authorgravatar for andrew@ziglang.orgAndrew Kelley <andrew@ziglang.org> 2025-09-19 09:27:25-07:00
committergravatar for noreply@github.comGitHub <noreply@github.com> 2025-09-19 09:27:25-07:00
log164c598cd85092323ef16640fd533d325fa74944
tree6528a4954dd35a49af4895b532f84321c6885388
parentbc921fec122f4fa9e9a3957657c6ee74d6870f5b
parent7c6ccca46d31a69cd6fddfdacd0aa7fba1d1e922
signaturebadge-check Signed by PGP key B5690EEEBB952194

Merge pull request #23416 from gooncreeper/improved-fuzzer

greatly improve capabilities of the fuzzer

9 files changed, 1317 insertions(+), 502 deletions(-)

lib/compiler/test_runner.zig+18-29
...@@ -4,6 +4,7 @@ const builtin = @import("builtin");...@@ -4,6 +4,7 @@ const builtin = @import("builtin");
4const std = @import("std");4const std = @import("std");
5const testing = std.testing;5const testing = std.testing;
6const assert = std.debug.assert;6const assert = std.debug.assert;
7const fuzz_abi = std.Build.abi.fuzz;
78
8pub const std_options: std.Options = .{9pub const std_options: std.Options = .{
9 .logFn = log,10 .logFn = log,
...@@ -57,7 +58,7 @@ pub fn main() void {...@@ -57,7 +58,7 @@ pub fn main() void {
57 fba.reset();58 fba.reset();
58 if (builtin.fuzz) {59 if (builtin.fuzz) {
59 const cache_dir = opt_cache_dir orelse @panic("missing --cache-dir=[path] argument");60 const cache_dir = opt_cache_dir orelse @panic("missing --cache-dir=[path] argument");
60 fuzzer_init(FuzzerSlice.fromSlice(cache_dir));61 fuzz_abi.fuzzer_init(.fromSlice(cache_dir));
61 }62 }
6263
63 if (listen) {64 if (listen) {
...@@ -78,7 +79,7 @@ fn mainServer() !void {...@@ -78,7 +79,7 @@ fn mainServer() !void {
78 });79 });
7980
80 if (builtin.fuzz) {81 if (builtin.fuzz) {
81 const coverage_id = fuzzer_coverage_id();82 const coverage_id = fuzz_abi.fuzzer_coverage_id();
82 try server.serveU64Message(.coverage_id, coverage_id);83 try server.serveU64Message(.coverage_id, coverage_id);
83 }84 }
8485
...@@ -152,14 +153,19 @@ fn mainServer() !void {...@@ -152,14 +153,19 @@ fn mainServer() !void {
152 });153 });
153 },154 },
154 .start_fuzzing => {155 .start_fuzzing => {
156 // This ensures that this code won't be analyzed and hence reference fuzzer symbols
157 // since they are not present.
155 if (!builtin.fuzz) unreachable;158 if (!builtin.fuzz) unreachable;
159
156 const index = try server.receiveBody_u32();160 const index = try server.receiveBody_u32();
157 const test_fn = builtin.test_functions[index];161 const test_fn = builtin.test_functions[index];
158 const entry_addr = @intFromPtr(test_fn.func);162 const entry_addr = @intFromPtr(test_fn.func);
163
159 try server.serveU64Message(.fuzz_start_addr, entry_addr);164 try server.serveU64Message(.fuzz_start_addr, entry_addr);
160 defer if (testing.allocator_instance.deinit() == .leak) std.process.exit(1);165 defer if (testing.allocator_instance.deinit() == .leak) std.process.exit(1);
161 is_fuzz_test = false;166 is_fuzz_test = false;
162 fuzzer_set_name(test_fn.name.ptr, test_fn.name.len);167 fuzz_test_index = index;
168
163 test_fn.func() catch |err| switch (err) {169 test_fn.func() catch |err| switch (err) {
164 error.SkipZigTest => return,170 error.SkipZigTest => return,
165 else => {171 else => {
...@@ -184,6 +190,8 @@ fn mainServer() !void {...@@ -184,6 +190,8 @@ fn mainServer() !void {
184190
185fn mainTerminal() void {191fn mainTerminal() void {
186 @disableInstrumentation();192 @disableInstrumentation();
193 if (builtin.fuzz) @panic("fuzz test requires server");
194
187 const test_fn_list = builtin.test_functions;195 const test_fn_list = builtin.test_functions;
188 var ok_count: usize = 0;196 var ok_count: usize = 0;
189 var skip_count: usize = 0;197 var skip_count: usize = 0;
...@@ -333,28 +341,8 @@ pub fn mainSimple() anyerror!void {...@@ -333,28 +341,8 @@ pub fn mainSimple() anyerror!void {
333 if (failed != 0) std.process.exit(1);341 if (failed != 0) std.process.exit(1);
334}342}
335343
336const FuzzerSlice = extern struct {
337 ptr: [*]const u8,
338 len: usize,
339
340 /// Inline to avoid fuzzer instrumentation.
341 inline fn toSlice(s: FuzzerSlice) []const u8 {
342 return s.ptr[0..s.len];
343 }
344
345 /// Inline to avoid fuzzer instrumentation.
346 inline fn fromSlice(s: []const u8) FuzzerSlice {
347 return .{ .ptr = s.ptr, .len = s.len };
348 }
349};
350
351var is_fuzz_test: bool = undefined;344var is_fuzz_test: bool = undefined;
352345var fuzz_test_index: u32 = undefined;
353extern fn fuzzer_set_name(name_ptr: [*]const u8, name_len: usize) void;
354extern fn fuzzer_init(cache_dir: FuzzerSlice) void;
355extern fn fuzzer_init_corpus_elem(input_ptr: [*]const u8, input_len: usize) void;
356extern fn fuzzer_start(testOne: *const fn ([*]const u8, usize) callconv(.c) void) void;
357extern fn fuzzer_coverage_id() u64;
358346
359pub fn fuzz(347pub fn fuzz(
360 context: anytype,348 context: anytype,
...@@ -385,12 +373,12 @@ pub fn fuzz(...@@ -385,12 +373,12 @@ pub fn fuzz(
385 const global = struct {373 const global = struct {
386 var ctx: @TypeOf(context) = undefined;374 var ctx: @TypeOf(context) = undefined;
387375
388 fn fuzzer_one(input_ptr: [*]const u8, input_len: usize) callconv(.c) void {376 fn test_one(input: fuzz_abi.Slice) callconv(.c) void {
389 @disableInstrumentation();377 @disableInstrumentation();
390 testing.allocator_instance = .{};378 testing.allocator_instance = .{};
391 defer if (testing.allocator_instance.deinit() == .leak) std.process.exit(1);379 defer if (testing.allocator_instance.deinit() == .leak) std.process.exit(1);
392 log_err_count = 0;380 log_err_count = 0;
393 testOne(ctx, input_ptr[0..input_len]) catch |err| switch (err) {381 testOne(ctx, input.toSlice()) catch |err| switch (err) {
394 error.SkipZigTest => return,382 error.SkipZigTest => return,
395 else => {383 else => {
396 std.debug.lockStdErr();384 std.debug.lockStdErr();
...@@ -411,10 +399,11 @@ pub fn fuzz(...@@ -411,10 +399,11 @@ pub fn fuzz(
411 testing.allocator_instance = .{};399 testing.allocator_instance = .{};
412 defer testing.allocator_instance = prev_allocator_state;400 defer testing.allocator_instance = prev_allocator_state;
413401
414 for (options.corpus) |elem| fuzzer_init_corpus_elem(elem.ptr, elem.len);
415
416 global.ctx = context;402 global.ctx = context;
417 fuzzer_start(&global.fuzzer_one);403 fuzz_abi.fuzzer_init_test(&global.test_one, .fromSlice(builtin.test_functions[fuzz_test_index].name));
404 for (options.corpus) |elem|
405 fuzz_abi.fuzzer_new_input(.fromSlice(elem));
406 fuzz_abi.fuzzer_main();
418 return;407 return;
419 }408 }
420409
lib/fuzzer.zig+1193-435
...@@ -1,532 +1,1275 @@...@@ -1,532 +1,1275 @@
1const builtin = @import("builtin");1const builtin = @import("builtin");
2const std = @import("std");2const std = @import("std");
3const Allocator = std.mem.Allocator;3const mem = std.mem;
4const math = std.math;
5const Allocator = mem.Allocator;
4const assert = std.debug.assert;6const assert = std.debug.assert;
5const fatal = std.process.fatal;7const panic = std.debug.panic;
6const SeenPcsHeader = std.Build.abi.fuzz.SeenPcsHeader;8const abi = std.Build.abi.fuzz;
9const native_endian = builtin.cpu.arch.endian();
710
8pub const std_options = std.Options{11pub const std_options = std.Options{
9 .logFn = logOverride,12 .logFn = logOverride,
10};13};
1114
12var log_file_buffer: [256]u8 = undefined;
13var log_file_writer: ?std.fs.File.Writer = null;
14
15fn logOverride(15fn logOverride(
16 comptime level: std.log.Level,16 comptime level: std.log.Level,
17 comptime scope: @Type(.enum_literal),17 comptime scope: @Type(.enum_literal),
18 comptime format: []const u8,18 comptime format: []const u8,
19 args: anytype,19 args: anytype,
20) void {20) void {
21 const fw = if (log_file_writer) |*f| f else f: {21 const f = log_f orelse
22 const f = fuzzer.cache_dir.createFile("tmp/libfuzzer.log", .{}) catch22 panic("attempt to use log before initialization, message:\n" ++ format, args);
23 @panic("failed to open fuzzer log file");23 f.lock(.exclusive) catch |e| panic("failed to lock logging file: {t}", .{e});
24 log_file_writer = f.writer(&log_file_buffer);24 defer f.unlock();
25 break :f &log_file_writer.?;25
26 };26 var buf: [256]u8 = undefined;
27 var fw = f.writer(&buf);
28 const end = f.getEndPos() catch |e| panic("failed to get fuzzer log file end: {t}", .{e});
29 fw.seekTo(end) catch |e| panic("failed to seek to fuzzer log file end: {t}", .{e});
30
27 const prefix1 = comptime level.asText();31 const prefix1 = comptime level.asText();
28 const prefix2 = if (scope == .default) ": " else "(" ++ @tagName(scope) ++ "): ";32 const prefix2 = if (scope == .default) ": " else "(" ++ @tagName(scope) ++ "): ";
29 fw.interface.print(prefix1 ++ prefix2 ++ format ++ "\n", args) catch33 fw.interface.print(
30 @panic("failed to write to fuzzer log");34 "[{s}] " ++ prefix1 ++ prefix2 ++ format ++ "\n",
31 fw.interface.flush() catch @panic("failed to flush fuzzer log");35 .{current_test_name orelse "setup"} ++ args,
36 ) catch panic("failed to write to fuzzer log: {t}", .{fw.err.?});
37 fw.interface.flush() catch panic("failed to write to fuzzer log: {t}", .{fw.err.?});
32}38}
3339
34/// Helps determine run uniqueness in the face of recursion.40var debug_allocator: std.heap.DebugAllocator(.{}) = .init;
35export threadlocal var __sancov_lowest_stack: usize = 0;41const gpa = switch (builtin.mode) {
42 .Debug => debug_allocator.allocator(),
43 .ReleaseFast, .ReleaseSmall, .ReleaseSafe => std.heap.smp_allocator,
44};
3645
37export fn __sanitizer_cov_trace_const_cmp1(arg1: u8, arg2: u8) void {46/// Part of `exec`, however seperate to allow it to be set before `exec` is.
38 handleCmp(@returnAddress(), arg1, arg2);47var log_f: ?std.fs.File = null;
39}48var exec: Executable = .preinit;
49var inst: Instrumentation = .preinit;
50var fuzzer: Fuzzer = undefined;
51var current_test_name: ?[]const u8 = null;
4052
41export fn __sanitizer_cov_trace_cmp1(arg1: u8, arg2: u8) void {53fn bitsetUsizes(elems: usize) usize {
42 handleCmp(@returnAddress(), arg1, arg2);54 return math.divCeil(usize, elems, @bitSizeOf(usize)) catch unreachable;
43}55}
4456
45export fn __sanitizer_cov_trace_const_cmp2(arg1: u16, arg2: u16) void {57const Executable = struct {
46 handleCmp(@returnAddress(), arg1, arg2);58 /// Tracks the hit count for each pc as updated by the process's instrumentation.
47}59 pc_counters: []u8,
4860
49export fn __sanitizer_cov_trace_cmp2(arg1: u16, arg2: u16) void {61 cache_f: std.fs.Dir,
50 handleCmp(@returnAddress(), arg1, arg2);62 /// Shared copy of all pcs that have been hit stored in a memory-mapped file that can viewed
51}63 /// while the fuzzer is running.
64 shared_seen_pcs: MemoryMappedList,
65 /// Hash of pcs used to uniquely identify the shared coverage file
66 pc_digest: u64,
67
68 /// A minimal state for this struct which instrumentation can function on.
69 /// Used before this structure is initialized to avoid illegal behavior
70 /// from instrumentation functions being called and using undefined values.
71 pub const preinit: Executable = .{
72 .pc_counters = undefined, // instrumentation works off the __sancov_cntrs section
73 .cache_f = undefined,
74 .shared_seen_pcs = undefined,
75 .pc_digest = undefined,
76 };
5277
53export fn __sanitizer_cov_trace_const_cmp4(arg1: u32, arg2: u32) void {78 fn getCoverageFile(cache_dir: std.fs.Dir, pcs: []const usize, pc_digest: u64) MemoryMappedList {
54 handleCmp(@returnAddress(), arg1, arg2);79 const pc_bitset_usizes = bitsetUsizes(pcs.len);
55}80 const coverage_file_name = std.fmt.hex(pc_digest);
81 comptime assert(abi.SeenPcsHeader.trailing[0] == .pc_bits_usize);
82 comptime assert(abi.SeenPcsHeader.trailing[1] == .pc_addr);
5683
57export fn __sanitizer_cov_trace_cmp4(arg1: u32, arg2: u32) void {84 var v = cache_dir.makeOpenPath("v", .{}) catch |e|
58 handleCmp(@returnAddress(), arg1, arg2);85 panic("failed to create directory 'v': {t}", .{e});
59}86 defer v.close();
87 const coverage_file, const populate = if (v.createFile(&coverage_file_name, .{
88 .read = true,
89 // If we create the file, we want to block other processes while we populate it
90 .lock = .exclusive,
91 .exclusive = true,
92 })) |f|
93 .{ f, true }
94 else |e| switch (e) {
95 error.PathAlreadyExists => .{ v.openFile(&coverage_file_name, .{
96 .mode = .read_write,
97 .lock = .shared,
98 }) catch |e2| panic(
99 "failed to open existing coverage file '{s}': {t}",
100 .{ &coverage_file_name, e2 },
101 ), false },
102 else => panic("failed to create coverage file '{s}': {t}", .{ &coverage_file_name, e }),
103 };
60104
61export fn __sanitizer_cov_trace_const_cmp8(arg1: u64, arg2: u64) void {105 const coverage_file_len = @sizeOf(abi.SeenPcsHeader) +
62 handleCmp(@returnAddress(), arg1, arg2);106 pc_bitset_usizes * @sizeOf(usize) +
63}107 pcs.len * @sizeOf(usize);
108 if (populate) {
109 defer coverage_file.lock(.shared) catch |e| panic(
110 "failed to demote lock for coverage file '{s}': {t}",
111 .{ &coverage_file_name, e },
112 );
113 var map = MemoryMappedList.create(coverage_file, 0, coverage_file_len) catch |e| panic(
114 "failed to init memory map for coverage file '{s}': {t}",
115 .{ &coverage_file_name, e },
116 );
117 map.appendSliceAssumeCapacity(mem.asBytes(&abi.SeenPcsHeader{
118 .n_runs = 0,
119 .unique_runs = 0,
120 .pcs_len = pcs.len,
121 }));
122 map.appendNTimesAssumeCapacity(0, pc_bitset_usizes * @sizeOf(usize));
123 map.appendSliceAssumeCapacity(mem.sliceAsBytes(pcs));
124 return map;
125 } else {
126 const size = coverage_file.getEndPos() catch |e| panic(
127 "failed to stat coverage file '{s}': {t}",
128 .{ &coverage_file_name, e },
129 );
130 if (size != coverage_file_len) panic(
131 "incompatible existing coverage file '{s}' (differing lengths: {} != {})",
132 .{ &coverage_file_name, size, coverage_file_len },
133 );
134
135 const map = MemoryMappedList.init(
136 coverage_file,
137 coverage_file_len,
138 coverage_file_len,
139 ) catch |e| panic(
140 "failed to init memory map for coverage file '{s}': {t}",
141 .{ &coverage_file_name, e },
142 );
143
144 const seen_pcs_header: *const abi.SeenPcsHeader = @ptrCast(@volatileCast(map.items));
145 if (seen_pcs_header.pcs_len != pcs.len) panic(
146 "incompatible existing coverage file '{s}' (differing pcs length: {} != {})",
147 .{ &coverage_file_name, seen_pcs_header.pcs_len, pcs.len },
148 );
149 if (mem.indexOfDiff(usize, seen_pcs_header.pcAddrs(), pcs)) |i| panic(
150 "incompatible existing coverage file '{s}' (differing pc at index {d}: {x} != {x})",
151 .{ &coverage_file_name, i, seen_pcs_header.pcAddrs()[i], pcs[i] },
152 );
153
154 return map;
155 }
156 }
64157
65export fn __sanitizer_cov_trace_cmp8(arg1: u64, arg2: u64) void {158 pub fn init(cache_dir_path: []const u8) Executable {
66 handleCmp(@returnAddress(), arg1, arg2);159 var self: Executable = undefined;
67}
68160
69export fn __sanitizer_cov_trace_switch(val: u64, cases_ptr: [*]u64) void {161 const cache_dir = std.fs.cwd().makeOpenPath(cache_dir_path, .{}) catch |e| panic(
70 const pc = @returnAddress();162 "failed to open directory '{s}': {t}",
71 const len = cases_ptr[0];163 .{ cache_dir_path, e },
72 const val_size_in_bits = cases_ptr[1];164 );
73 const cases = cases_ptr[2..][0..len];165 log_f = cache_dir.createFile("tmp/libfuzzer.log", .{ .truncate = false }) catch |e|
74 fuzzer.traceValue(pc ^ val);166 panic("failed to create file 'tmp/libfuzzer.log': {t}", .{e});
75 _ = val_size_in_bits;167 self.cache_f = cache_dir.makeOpenPath("f", .{}) catch |e|
76 _ = cases;168 panic("failed to open directory 'f': {t}", .{e});
77 //std.log.debug("0x{x}: switch on value {d} ({d} bits) with {d} cases", .{169
78 // pc, val, val_size_in_bits, cases.len,170 // Linkers are expected to automatically add symbols prefixed with these for the start and
79 //});171 // end of sections whose names are valid C identifiers.
80}172 const ofmt = builtin.object_format;
173 const section_start_prefix, const section_end_prefix = switch (ofmt) {
174 .elf => .{ "__start_", "__stop_" },
175 .macho => .{ "\x01section$start$__DATA$", "\x01section$end$__DATA$" },
176 else => @compileError("unsupported fuzzing object format '" ++ @tagName(ofmt) ++ "'"),
177 };
81178
82export fn __sanitizer_cov_trace_pc_indir(callee: usize) void {179 self.pc_counters = blk: {
83 // Not valuable because we already have pc tracing via 8bit counters.180 const pc_counters_start_name = section_start_prefix ++ "__sancov_cntrs";
84 _ = callee;181 const pc_counters_start = @extern([*]u8, .{
85 //const pc = @returnAddress();182 .name = pc_counters_start_name,
86 //fuzzer.traceValue(pc ^ callee);183 .linkage = .weak,
87 //std.log.debug("0x{x}: indirect call to 0x{x}", .{ pc, callee });184 }) orelse panic("missing {s} symbol", .{pc_counters_start_name});
88}
89export fn __sanitizer_cov_8bit_counters_init(start: usize, end: usize) void {
90 // clang will emit a call to this function when compiling with code coverage instrumentation.
91 // however fuzzer_init() does not need this information, since it directly reads from the symbol table.
92 _ = start;
93 _ = end;
94}
95export fn __sanitizer_cov_pcs_init(start: usize, end: usize) void {
96 // clang will emit a call to this function when compiling with code coverage instrumentation.
97 // however fuzzer_init() does not need this information, since it directly reads from the symbol table.
98 _ = start;
99 _ = end;
100}
101185
102fn handleCmp(pc: usize, arg1: u64, arg2: u64) void {186 const pc_counters_end_name = section_end_prefix ++ "__sancov_cntrs";
103 fuzzer.traceValue(pc ^ arg1 ^ arg2);187 const pc_counters_end = @extern([*]u8, .{
104 //std.log.debug("0x{x}: comparison of {d} and {d}", .{ pc, arg1, arg2 });188 .name = pc_counters_end_name,
105}189 .linkage = .weak,
190 }) orelse panic("missing {s} symbol", .{pc_counters_end_name});
106191
107const Fuzzer = struct {192 break :blk pc_counters_start[0 .. pc_counters_end - pc_counters_start];
108 rng: std.Random.DefaultPrng,193 };
109 pcs: []const usize,
110 pc_counters: []u8,
111 n_runs: usize,
112 traced_comparisons: std.AutoArrayHashMapUnmanaged(usize, void),
113 /// Tracks which PCs have been seen across all runs that do not crash the fuzzer process.
114 /// Stored in a memory-mapped file so that it can be shared with other
115 /// processes and viewed while the fuzzer is running.
116 seen_pcs: MemoryMappedList,
117 cache_dir: std.fs.Dir,
118 /// Identifies the file name that will be used to store coverage
119 /// information, available to other processes.
120 coverage_id: u64,
121 unit_test_name: []const u8,
122
123 /// The index corresponds to the file name within the f/ subdirectory.
124 /// The string is the input.
125 /// This data is read-only; it caches what is on the filesystem.
126 corpus: std.ArrayListUnmanaged(Input),
127 corpus_directory: std.Build.Cache.Directory,
128194
129 /// The next input that will be given to the testOne function. When the195 const pcs = blk: {
130 /// current process crashes, this memory-mapped file is used to recover the196 const pcs_start_name = section_start_prefix ++ "__sancov_pcs1";
131 /// input.197 const pcs_start = @extern([*]usize, .{
132 ///198 .name = pcs_start_name,
133 /// The file size corresponds to the capacity. The length is not stored199 .linkage = .weak,
134 /// and that is the next thing to work on!200 }) orelse panic("missing {s} symbol", .{pcs_start_name});
135 input: MemoryMappedList,
136201
137 const Input = struct {202 const pcs_end_name = section_end_prefix ++ "__sancov_pcs1";
138 bytes: []u8,203 const pcs_end = @extern([*]usize, .{
139 last_traced_comparison: usize,204 .name = pcs_end_name,
140 };205 .linkage = .weak,
206 }) orelse panic("missing {s} symbol", .{pcs_end_name});
207
208 break :blk pcs_start[0 .. pcs_end - pcs_start];
209 };
141210
142 const Slice = extern struct {211 if (self.pc_counters.len != pcs.len) panic(
143 ptr: [*]const u8,212 "pc counters length and pcs length do not match ({} != {})",
144 len: usize,213 .{ self.pc_counters.len, pcs.len },
214 );
145215
146 fn toZig(s: Slice) []const u8 {216 self.pc_digest = std.hash.Wyhash.hash(0, mem.sliceAsBytes(pcs));
147 return s.ptr[0..s.len];217 self.shared_seen_pcs = getCoverageFile(cache_dir, pcs, self.pc_digest);
148 }
149218
150 fn fromZig(s: []const u8) Slice {219 return self;
151 return .{220 }
152 .ptr = s.ptr,221
153 .len = s.len,222 pub fn pcBitsetIterator(self: Executable) PcBitsetIterator {
154 };223 return .{ .pc_counters = self.pc_counters };
224 }
225
226 /// Iterates over pc_counters returning a bitset for if each of them have been hit
227 pub const PcBitsetIterator = struct {
228 index: usize = 0,
229 pc_counters: []u8,
230
231 pub fn next(self: *PcBitsetIterator) usize {
232 const rest = self.pc_counters[self.index..];
233 if (rest.len >= @bitSizeOf(usize)) {
234 defer self.index += @bitSizeOf(usize);
235 const V = @Vector(@bitSizeOf(usize), u8);
236 return @as(usize, @bitCast(@as(V, @splat(0)) != rest[0..@bitSizeOf(usize)].*));
237 } else if (rest.len != 0) {
238 defer self.index += rest.len;
239 var res: usize = 0;
240 for (0.., rest) |bit_index, byte| {
241 res |= @shlExact(@as(usize, @intFromBool(byte != 0)), @intCast(bit_index));
242 }
243 return res;
244 } else unreachable;
155 }245 }
156 };246 };
247};
157248
158 fn init(f: *Fuzzer, cache_dir: std.fs.Dir, pc_counters: []u8, pcs: []const usize) !void {249/// Data gathered from instrumentation functions.
159 f.cache_dir = cache_dir;250/// Seperate from Executable since its state is resetable and changes.
160 f.pc_counters = pc_counters;251/// Seperate from Fuzzer since it may be needed before fuzzing starts.
161 f.pcs = pcs;252const Instrumentation = struct {
162253 /// Bitset of seen pcs across all runs excluding fresh pcs.
163 // Choose a file name for the coverage based on a hash of the PCs that will be stored within.254 /// This is seperate then shared_seen_pcs because multiple fuzzing processes are likely using
164 const pc_digest = std.hash.Wyhash.hash(0, std.mem.sliceAsBytes(pcs));255 /// it which causes contention and unrelated pcs to our campaign being set.
165 f.coverage_id = pc_digest;256 seen_pcs: []usize,
166 const hex_digest = std.fmt.hex(pc_digest);257
167 const coverage_file_path = "v/" ++ hex_digest;258 /// Stores a fresh input's new pcs
168259 fresh_pcs: []usize,
169 // Layout of this file:260
170 // - Header261 /// Pcs which __sanitizer_cov_trace_switch and __sanitizer_cov_trace_const_cmpx
171 // - list of PC addresses (usize elements)262 /// have been called from and have had their already been added to const_x_vals
172 // - list of hit flag, 1 bit per address (stored in u8 elements)263 const_pcs: std.AutoArrayHashMapUnmanaged(usize, void) = .empty,
173 const coverage_file = createFileBail(cache_dir, coverage_file_path, .{264 /// Values that have been constant operands in comparisons and switch cases.
174 .read = true,265 /// There may be duplicates in this array if they came from different addresses, which is
175 .truncate = false,266 /// fine as they are likely more important and hence more likely to be selected.
176 });267 const_vals2: std.ArrayListUnmanaged(u16) = .empty,
177 const n_bitset_elems = (pcs.len + @bitSizeOf(usize) - 1) / @bitSizeOf(usize);268 const_vals4: std.ArrayListUnmanaged(u32) = .empty,
178 comptime assert(SeenPcsHeader.trailing[0] == .pc_bits_usize);269 const_vals8: std.ArrayListUnmanaged(u64) = .empty,
179 comptime assert(SeenPcsHeader.trailing[1] == .pc_addr);270 const_vals16: std.ArrayListUnmanaged(u128) = .empty,
180 const bytes_len = @sizeOf(SeenPcsHeader) +271
181 n_bitset_elems * @sizeOf(usize) +272 /// A minimal state for this struct which instrumentation can function on.
182 pcs.len * @sizeOf(usize);273 /// Used before this structure is initialized to avoid illegal behavior
183 const existing_len = coverage_file.getEndPos() catch |err| {274 /// from instrumentation functions being called and using undefined values.
184 fatal("unable to check len of coverage file: {s}", .{@errorName(err)});275 pub const preinit: Instrumentation = .{
276 .seen_pcs = undefined, // currently only updated by `Fuzzer`
277 .fresh_pcs = undefined,
278 };
279
280 pub fn depreinit(self: *Instrumentation) void {
281 self.const_vals2.deinit(gpa);
282 self.const_vals4.deinit(gpa);
283 self.const_vals8.deinit(gpa);
284 self.const_vals16.deinit(gpa);
285 self.* = undefined;
286 }
287
288 pub fn init() Instrumentation {
289 const pc_bitset_usizes = bitsetUsizes(exec.pc_counters.len);
290 const alloc_usizes = pc_bitset_usizes * 2;
291 const buf = gpa.alloc(u8, alloc_usizes * @sizeOf(usize)) catch @panic("OOM");
292 var fba_ctx: std.heap.FixedBufferAllocator = .init(buf);
293 const fba = fba_ctx.allocator();
294
295 var self: Instrumentation = .{
296 .seen_pcs = fba.alloc(usize, pc_bitset_usizes) catch unreachable,
297 .fresh_pcs = fba.alloc(usize, pc_bitset_usizes) catch unreachable,
185 };298 };
186 if (existing_len == 0) {299 self.reset();
187 coverage_file.setEndPos(bytes_len) catch |err| {300 return self;
188 fatal("unable to set len of coverage file: {s}", .{@errorName(err)});301 }
189 };302
190 } else if (existing_len != bytes_len) {303 pub fn reset(self: *Instrumentation) void {
191 fatal("incompatible existing coverage file (differing lengths)", .{});304 @memset(self.seen_pcs, 0);
305 @memset(self.fresh_pcs, 0);
306 self.const_pcs.clearRetainingCapacity();
307 self.const_vals2.clearRetainingCapacity();
308 self.const_vals4.clearRetainingCapacity();
309 self.const_vals8.clearRetainingCapacity();
310 self.const_vals16.clearRetainingCapacity();
311 }
312
313 /// If false is returned, then the pc is marked as seen
314 pub fn constPcSeen(self: *Instrumentation, pc: usize) bool {
315 return (self.const_pcs.getOrPut(gpa, pc) catch @panic("OOM")).found_existing;
316 }
317
318 pub fn isFresh(self: *Instrumentation) bool {
319 var hit_pcs = exec.pcBitsetIterator();
320 for (self.seen_pcs) |seen_pcs| {
321 if (hit_pcs.next() & ~seen_pcs != 0) return true;
192 }322 }
193 f.seen_pcs = MemoryMappedList.init(coverage_file, existing_len, bytes_len) catch |err| {323
194 fatal("unable to init coverage memory map: {s}", .{@errorName(err)});324 return false;
195 };325 }
196 if (existing_len != 0) {326
197 const existing_pcs_bytes = f.seen_pcs.items[@sizeOf(SeenPcsHeader) + @sizeOf(usize) * n_bitset_elems ..][0 .. pcs.len * @sizeOf(usize)];327 /// Updates `fresh_pcs`
198 const existing_pcs = std.mem.bytesAsSlice(usize, existing_pcs_bytes);328 pub fn setFresh(self: *Instrumentation) void {
199 for (existing_pcs, pcs, 0..) |old, new, i| {329 var hit_pcs = exec.pcBitsetIterator();
200 if (old != new) {330 for (self.seen_pcs, self.fresh_pcs) |seen_pcs, *fresh_pcs| {
201 fatal("incompatible existing coverage file (differing PC at index {d}: {x} != {x})", .{331 fresh_pcs.* = hit_pcs.next() & ~seen_pcs;
202 i, old, new,
203 });
204 }
205 }
206 } else {
207 const header: SeenPcsHeader = .{
208 .n_runs = 0,
209 .unique_runs = 0,
210 .pcs_len = pcs.len,
211 };
212 f.seen_pcs.appendSliceAssumeCapacity(std.mem.asBytes(&header));
213 f.seen_pcs.appendNTimesAssumeCapacity(0, n_bitset_elems * @sizeOf(usize));
214 f.seen_pcs.appendSliceAssumeCapacity(std.mem.sliceAsBytes(pcs));
215 }332 }
216 }333 }
217334
218 fn initNextInput(f: *Fuzzer) void {335 /// Returns if `exec.pc_counters` is a superset of `fresh_pcs`.
219 while (true) {336 pub fn atleastFresh(self: *Instrumentation) bool {
220 const i = f.corpus.items.len;337 var hit_pcs = exec.pcBitsetIterator();
221 var buf: [30]u8 = undefined;338 for (self.fresh_pcs) |fresh_pcs| {
222 const input_sub_path = std.fmt.bufPrint(&buf, "{d}", .{i}) catch unreachable;339 if (fresh_pcs & hit_pcs.next() != fresh_pcs) return false;
223 const input = f.corpus_directory.handle.readFileAlloc(input_sub_path, gpa, .limited(1 << 31)) catch |err| switch (err) {
224 error.FileNotFound => {
225 // Make this one the next input.
226 const input_file = f.corpus_directory.handle.createFile(input_sub_path, .{
227 .exclusive = true,
228 .truncate = false,
229 .read = true,
230 }) catch |e| switch (e) {
231 error.PathAlreadyExists => continue,
232 else => fatal("unable to create '{f}{d}: {s}", .{ f.corpus_directory, i, @errorName(err) }),
233 };
234 errdefer input_file.close();
235 // Initialize the mmap for the current input.
236 f.input = MemoryMappedList.create(input_file, 0, std.heap.page_size_max) catch |e| {
237 fatal("unable to init memory map for input at '{f}{d}': {s}", .{
238 f.corpus_directory, i, @errorName(e),
239 });
240 };
241 break;
242 },
243 else => fatal("unable to read '{f}{d}': {s}", .{ f.corpus_directory, i, @errorName(err) }),
244 };
245 errdefer gpa.free(input);
246 f.corpus.append(gpa, .{
247 .bytes = input,
248 .last_traced_comparison = 0,
249 }) catch |err| oom(err);
250 }340 }
341 return true;
251 }342 }
252343
253 fn addCorpusElem(f: *Fuzzer, input: []const u8) !void {344 /// Updates based off `fresh_pcs`
254 try f.corpus.append(gpa, .{345 fn updateSeen(self: *Instrumentation) void {
255 .bytes = try gpa.dupe(u8, input),346 comptime assert(abi.SeenPcsHeader.trailing[0] == .pc_bits_usize);
256 .last_traced_comparison = 0,347 const shared_seen_pcs: [*]volatile usize = @ptrCast(
257 });348 exec.shared_seen_pcs.items[@sizeOf(abi.SeenPcsHeader)..].ptr,
349 );
350
351 for (self.seen_pcs, shared_seen_pcs, self.fresh_pcs) |*seen, *shared_seen, fresh| {
352 seen.* |= fresh;
353 if (fresh != 0)
354 _ = @atomicRmw(usize, shared_seen, .Or, fresh, .monotonic);
355 }
258 }356 }
357};
259358
260 fn start(f: *Fuzzer) !void {359const Fuzzer = struct {
261 const rng = fuzzer.rng.random();360 arena_ctx: std.heap.ArenaAllocator = .init(gpa),
361 rng: std.Random.DefaultPrng = .init(0),
362 test_one: abi.TestOne,
363 /// The next input that will be given to the testOne function. When the
364 /// current process crashes, this memory-mapped file is used to recover the
365 /// input.
366 input: MemoryMappedList,
262367
263 // Grab the corpus which is namespaced based on `unit_test_name`.368 /// Minimized past inputs leading to new pc hits.
264 {369 /// These are randomly mutated in round-robin fashion
265 if (f.unit_test_name.len == 0) fatal("test runner never set unit test name", .{});370 /// Element zero is always an empty input. It is gauraunteed no other elements are empty.
266 const sub_path = try std.fmt.allocPrint(gpa, "f/{s}", .{f.unit_test_name});371 corpus: std.ArrayListUnmanaged([]const u8),
267 f.corpus_directory = .{372 corpus_pos: usize,
268 .handle = f.cache_dir.makeOpenPath(sub_path, .{}) catch |err|373 /// List of past mutations that have led to new inputs. This way, the mutations that are the
269 fatal("unable to open corpus directory 'f/{s}': {t}", .{ sub_path, err }),374 /// most effective are the most likely to be selected again. Starts with one of each mutation.
270 .path = sub_path,375 mutations: std.ArrayListUnmanaged(Mutation) = .empty,
376
377 /// Filesystem directory containing found inputs for future runs
378 corpus_dir: std.fs.Dir,
379 corpus_dir_idx: usize = 0,
380
381 pub fn init(test_one: abi.TestOne, unit_test_name: []const u8) Fuzzer {
382 var self: Fuzzer = .{
383 .test_one = test_one,
384 .input = undefined,
385 .corpus = .empty,
386 .corpus_pos = 0,
387 .mutations = .empty,
388 .corpus_dir = undefined,
389 };
390 const arena = self.arena_ctx.allocator();
391
392 self.corpus_dir = exec.cache_f.makeOpenPath(unit_test_name, .{}) catch |e|
393 panic("failed to open directory '{s}': {t}", .{ unit_test_name, e });
394 self.input = in: {
395 const f = self.corpus_dir.createFile("in", .{
396 .read = true,
397 .truncate = false,
398 // In case any other fuzz tests are running under the same test name,
399 // the input file is exclusively locked to ensures only one proceeds.
400 .lock = .exclusive,
401 .lock_nonblocking = true,
402 }) catch |e| switch (e) {
403 error.WouldBlock => @panic("input file 'in' is in use by another fuzzing process"),
404 else => panic("failed to create input file 'in': {t}", .{e}),
271 };405 };
272 initNextInput(f);406 const size = f.getEndPos() catch |e| panic("failed to stat input file 'in': {t}", .{e});
273 }407 const map = (if (size < std.heap.page_size_max)
408 MemoryMappedList.create(f, 8, std.heap.page_size_max)
409 else
410 MemoryMappedList.init(f, size, size)) catch |e|
411 panic("failed to memory map input file 'in': {t}", .{e});
412
413 // Perform a dry-run of the stored input if there was one in case it might reproduce a
414 // crash.
415 const old_in_len = mem.littleToNative(usize, mem.bytesAsValue(usize, map.items[0..8]).*);
416 if (size >= 8 and old_in_len != 0 and map.items.len - 8 < old_in_len) {
417 test_one(.fromSlice(@volatileCast(map.items[8..][0..old_in_len])));
418 }
274419
275 assert(f.n_runs == 0);420 break :in map;
276421 };
277 // If the corpus is empty, synthesize one input.422 inst.reset();
278 if (f.corpus.items.len == 0) {423
279 const len = rng.uintLessThanBiased(usize, 200);424 self.mutations.appendSlice(gpa, std.meta.tags(Mutation)) catch @panic("OOM");
280 const slice = try gpa.alloc(u8, len);425 // Ensure there is never an empty corpus. Additionally, an empty input usually leads to
281 rng.bytes(slice);426 // new inputs.
282 f.input.appendSliceAssumeCapacity(slice);427 self.addInput(&.{});
283 try f.corpus.append(gpa, .{
284 .bytes = slice,
285 .last_traced_comparison = 0,
286 });
287 runOne(f, 0);
288 }
289428
290 while (true) {429 while (true) {
291 const chosen_index = rng.uintLessThanBiased(usize, f.corpus.items.len);430 var name_buf: [@sizeOf(usize) * 2]u8 = undefined;
292 const modification = rng.enumValue(Mutation);431 const bytes = self.corpus_dir.readFileAlloc(
293 f.mutateAndRunOne(chosen_index, modification);432 std.fmt.bufPrint(&name_buf, "{x}", .{self.corpus_dir_idx}) catch unreachable,
433 arena,
434 .unlimited,
435 ) catch |e| switch (e) {
436 error.FileNotFound => break,
437 else => panic("failed to read corpus file '{x}': {t}", .{ self.corpus_dir_idx, e }),
438 };
439 // No corpus file of length zero will ever be created
440 if (bytes.len == 0)
441 panic("corrupt corpus file '{x}' (len of zero)", .{self.corpus_dir_idx});
442 self.addInput(bytes);
443 self.corpus_dir_idx += 1;
294 }444 }
445
446 return self;
295 }447 }
296448
297 /// `x` represents a possible branch. It is the PC address of the possible449 pub fn deinit(self: *Fuzzer) void {
298 /// branch site, hashed together with the value(s) used that determine to450 self.input.deinit();
299 /// where it branches.451 self.corpus.deinit(gpa);
300 fn traceValue(f: *Fuzzer, x: usize) void {452 self.mutations.deinit(gpa);
301 errdefer |err| oom(err);453 self.corpus_dir.close();
302 try f.traced_comparisons.put(gpa, x, {});454 self.arena_ctx.deinit();
455 self.* = undefined;
303 }456 }
304457
305 const Mutation = enum {458 pub fn addInput(self: *Fuzzer, bytes: []const u8) void {
306 remove_byte,459 self.corpus.append(gpa, bytes) catch @panic("OOM");
307 modify_byte,460 self.input.clearRetainingCapacity();
308 add_byte,461 self.input.ensureTotalCapacity(8 + bytes.len) catch |e|
309 };462 panic("could not resize shared input file: {t}", .{e});
463 self.input.items.len = 8;
464 self.input.appendSliceAssumeCapacity(bytes);
465 self.run();
466 inst.setFresh();
467 inst.updateSeen();
468 }
310469
311 fn mutateAndRunOne(f: *Fuzzer, corpus_index: usize, mutation: Mutation) void {470 /// Assumes `fresh_pcs` correspond to the input
312 const rng = fuzzer.rng.random();471 fn minimizeInput(self: *Fuzzer) void {
313 f.input.clearRetainingCapacity();472 // The minimization technique is kept relatively simple, we sequentially try to remove each
314 const old_input = f.corpus.items[corpus_index].bytes;473 // byte and check that the new pcs and memory loads are still hit.
315 f.input.ensureTotalCapacity(old_input.len + 1) catch @panic("mmap file resize failed");474 var i = self.input.items.len;
316 switch (mutation) {475 while (i != 8) {
317 .remove_byte => {476 i -= 1;
318 const omitted_index = rng.uintLessThanBiased(usize, old_input.len);477 const old = self.input.orderedRemove(i);
319 f.input.appendSliceAssumeCapacity(old_input[0..omitted_index]);478
320 f.input.appendSliceAssumeCapacity(old_input[omitted_index + 1 ..]);479 @memset(exec.pc_counters, 0);
321 },480 self.run();
322 .modify_byte => {481
323 const modified_index = rng.uintLessThanBiased(usize, old_input.len);482 if (!inst.atleastFresh()) {
324 f.input.appendSliceAssumeCapacity(old_input);483 self.input.insertAssumeCapacity(i, old);
325 f.input.items[modified_index] = rng.int(u8);484 } else {
326 },485 // This removal may have led to new pcs or memory loads being hit, so we need to
327 .add_byte => {486 // update them to avoid duplicates.
328 const modified_index = rng.uintLessThanBiased(usize, old_input.len);487 inst.setFresh();
329 f.input.appendSliceAssumeCapacity(old_input[0..modified_index]);488 }
330 f.input.appendAssumeCapacity(rng.int(u8));
331 f.input.appendSliceAssumeCapacity(old_input[modified_index..]);
332 },
333 }489 }
334 runOne(f, corpus_index);
335 }490 }
336491
337 fn runOne(f: *Fuzzer, corpus_index: usize) void {492 fn run(self: *Fuzzer) void {
338 const header: *volatile SeenPcsHeader = @ptrCast(f.seen_pcs.items[0..@sizeOf(SeenPcsHeader)]);493 // `pc_counters` is not cleared since only new hits are relevant.
339494
340 f.traced_comparisons.clearRetainingCapacity();495 mem.bytesAsValue(usize, self.input.items[0..8]).* =
341 @memset(f.pc_counters, 0);496 mem.nativeToLittle(usize, self.input.items.len - 8);
342 __sancov_lowest_stack = std.math.maxInt(usize);497 self.test_one(.fromSlice(@volatileCast(self.input.items[8..])));
343498
344 fuzzer_one(@volatileCast(f.input.items.ptr), f.input.items.len);499 const header = mem.bytesAsValue(
345500 abi.SeenPcsHeader,
346 f.n_runs += 1;501 exec.shared_seen_pcs.items[0..@sizeOf(abi.SeenPcsHeader)],
502 );
347 _ = @atomicRmw(usize, &header.n_runs, .Add, 1, .monotonic);503 _ = @atomicRmw(usize, &header.n_runs, .Add, 1, .monotonic);
504 }
348505
349 // Track code coverage from all runs.506 pub fn cycle(self: *Fuzzer) void {
350 comptime assert(SeenPcsHeader.trailing[0] == .pc_bits_usize);507 const input = self.corpus.items[self.corpus_pos];
351 const header_end_ptr: [*]volatile usize = @ptrCast(f.seen_pcs.items[@sizeOf(SeenPcsHeader)..]);508 self.corpus_pos += 1;
352 const remainder = f.pcs.len % @bitSizeOf(usize);509 if (self.corpus_pos == self.corpus.items.len)
353 const aligned_len = f.pcs.len - remainder;510 self.corpus_pos = 0;
354 const seen_pcs = header_end_ptr[0..aligned_len];511
355 const pc_counters = std.mem.bytesAsSlice([@bitSizeOf(usize)]u8, f.pc_counters[0..aligned_len]);512 const rng = self.rng.random();
356 const V = @Vector(@bitSizeOf(usize), u8);513 while (true) {
357 const zero_v: V = @splat(0);514 const m = self.mutations.items[rng.uintLessThanBiased(usize, self.mutations.items.len)];
358 var fresh = false;515 if (!m.mutate(
359 var superset = true;516 rng,
360517 input,
361 for (header_end_ptr[0..pc_counters.len], pc_counters) |*elem, *array| {518 &self.input,
362 const v: V = array.*;519 self.corpus.items,
363 const mask: usize = @bitCast(v != zero_v);520 inst.const_vals2.items,
364 const prev = @atomicRmw(usize, elem, .Or, mask, .monotonic);521 inst.const_vals4.items,
365 fresh = fresh or (prev | mask) != prev;522 inst.const_vals8.items,
366 superset = superset and (prev | mask) != mask;523 inst.const_vals16.items,
367 }524 )) continue;
368 if (remainder > 0) {525
369 const i = pc_counters.len;526 self.run();
370 const elem = &seen_pcs[i];527 if (inst.isFresh()) {
371 var mask: usize = 0;528 @branchHint(.unlikely);
372 for (f.pc_counters[i * @bitSizeOf(usize) ..][0..remainder], 0..) |byte, bit_index| {529
373 mask |= @as(usize, @intFromBool(byte != 0)) << @intCast(bit_index);530 const header = mem.bytesAsValue(
531 abi.SeenPcsHeader,
532 exec.shared_seen_pcs.items[0..@sizeOf(abi.SeenPcsHeader)],
533 );
534 _ = @atomicRmw(usize, &header.unique_runs, .Add, 1, .monotonic);
535
536 inst.setFresh();
537 self.minimizeInput();
538 inst.updateSeen();
539
540 // An empty-input has always been tried, so if an empty input is fresh then the
541 // test has to be non-deterministic. This has to be checked as duplicate empty
542 // entries are not allowed.
543 if (self.input.items.len - 8 == 0) {
544 std.log.warn("non-deterministic test (empty input produces different hits)", .{});
545 _ = @atomicRmw(usize, &header.unique_runs, .Sub, 1, .monotonic);
546 return;
547 }
548
549 const arena = self.arena_ctx.allocator();
550 const bytes = arena.dupe(u8, @volatileCast(self.input.items[8..])) catch @panic("OOM");
551
552 self.corpus.append(gpa, bytes) catch @panic("OOM");
553 self.mutations.appendNTimes(gpa, m, 6) catch @panic("OOM");
554
555 // Write new corpus to cache
556 var name_buf: [@sizeOf(usize) * 2]u8 = undefined;
557 self.corpus_dir.writeFile(.{
558 .sub_path = std.fmt.bufPrint(
559 &name_buf,
560 "{x}",
561 .{self.corpus_dir_idx},
562 ) catch unreachable,
563 .data = bytes,
564 }) catch |e| panic(
565 "failed to write corpus file '{x}': {t}",
566 .{ self.corpus_dir_idx, e },
567 );
568 self.corpus_dir_idx += 1;
374 }569 }
375 const prev = @atomicRmw(usize, elem, .Or, mask, .monotonic);
376 fresh = fresh or (prev | mask) != prev;
377 superset = superset and (prev | mask) != mask;
378 }
379570
380 // First check if this is a better version of an already existing571 break;
381 // input, replacing that input.
382 if (superset or f.traced_comparisons.entries.len >= f.corpus.items[corpus_index].last_traced_comparison) {
383 const new_input = gpa.realloc(f.corpus.items[corpus_index].bytes, f.input.items.len) catch |err| oom(err);
384 f.corpus.items[corpus_index] = .{
385 .bytes = new_input,
386 .last_traced_comparison = f.traced_comparisons.count(),
387 };
388 @memcpy(new_input, @volatileCast(f.input.items));
389 _ = @atomicRmw(usize, &header.unique_runs, .Add, 1, .monotonic);
390 return;
391 }572 }
573 }
574};
392575
393 if (!fresh) return;576/// Instrumentation must not be triggered before this function is called
577export fn fuzzer_init(cache_dir_path: abi.Slice) void {
578 inst.depreinit();
579 exec = .init(cache_dir_path.toSlice());
580 inst = .init();
581}
394582
395 // Input is already committed to the file system, we just need to open a new file583/// Invalid until `fuzzer_init` is called.
396 // for the next input.584export fn fuzzer_coverage_id() u64 {
397 // Pre-add it to the corpus list so that it does not get redundantly picked up.585 return exec.pc_digest;
398 f.corpus.append(gpa, .{586}
399 .bytes = gpa.dupe(u8, @volatileCast(f.input.items)) catch |err| oom(err),
400 .last_traced_comparison = f.traced_comparisons.entries.len,
401 }) catch |err| oom(err);
402 f.input.deinit();
403 initNextInput(f);
404587
405 // TODO: also mark input as "hot" so it gets prioritized for checking mutations above others.588/// fuzzer_init must be called beforehand
589export fn fuzzer_init_test(test_one: abi.TestOne, unit_test_name: abi.Slice) void {
590 current_test_name = unit_test_name.toSlice();
591 fuzzer = .init(test_one, unit_test_name.toSlice());
592}
406593
407 _ = @atomicRmw(usize, &header.unique_runs, .Add, 1, .monotonic);594/// fuzzer_init_test must be called beforehand
408 }595/// The callee owns the memory of bytes and must not free it until the fuzzer is finished.
409};596export fn fuzzer_new_input(bytes: abi.Slice) void {
597 // An entry of length zero is always added and duplicates of it are not allowed.
598 if (bytes.len != 0)
599 fuzzer.addInput(bytes.toSlice());
600}
410601
411fn createFileBail(dir: std.fs.Dir, sub_path: []const u8, flags: std.fs.File.CreateFlags) std.fs.File {602/// fuzzer_init_test must be called first
412 return dir.createFile(sub_path, flags) catch |err| switch (err) {603export fn fuzzer_main() void {
413 error.FileNotFound => {604 while (true) {
414 const dir_name = std.fs.path.dirname(sub_path).?;605 fuzzer.cycle();
415 dir.makePath(dir_name) catch |e| {606 }
416 fatal("unable to make path '{s}': {s}", .{ dir_name, @errorName(e) });
417 };
418 return dir.createFile(sub_path, flags) catch |e| {
419 fatal("unable to create file '{s}': {s}", .{ sub_path, @errorName(e) });
420 };
421 },
422 else => fatal("unable to create file '{s}': {s}", .{ sub_path, @errorName(err) }),
423 };
424}607}
425608
426fn oom(err: anytype) noreturn {609/// Helps determine run uniqueness in the face of recursion.
427 switch (err) {610/// Currently not used by the fuzzer.
428 error.OutOfMemory => @panic("out of memory"),611export threadlocal var __sancov_lowest_stack: usize = 0;
612
613/// Inline since the return address of the callee is required
614inline fn genericConstCmp(T: anytype, val: T, comptime const_vals_field: []const u8) void {
615 if (!inst.constPcSeen(@returnAddress())) {
616 @branchHint(.unlikely);
617 @field(inst, const_vals_field).append(gpa, val) catch @panic("OOM");
429 }618 }
430}619}
431620
432var debug_allocator: std.heap.GeneralPurposeAllocator(.{}) = .init;621export fn __sanitizer_cov_trace_const_cmp1(const_arg: u8, arg: u8) void {
622 _ = const_arg;
623 _ = arg;
624}
433625
434const gpa = switch (builtin.mode) {626export fn __sanitizer_cov_trace_const_cmp2(const_arg: u16, arg: u16) void {
435 .Debug => debug_allocator.allocator(),627 _ = arg;
436 .ReleaseFast, .ReleaseSmall, .ReleaseSafe => std.heap.smp_allocator,628 genericConstCmp(u16, const_arg, "const_vals2");
437};629}
438630
439var fuzzer: Fuzzer = .{631export fn __sanitizer_cov_trace_const_cmp4(const_arg: u32, arg: u32) void {
440 .rng = std.Random.DefaultPrng.init(0),632 _ = arg;
441 .input = undefined,633 genericConstCmp(u32, const_arg, "const_vals4");
442 .pcs = undefined,634}
443 .pc_counters = undefined,
444 .n_runs = 0,
445 .cache_dir = undefined,
446 .seen_pcs = undefined,
447 .coverage_id = undefined,
448 .unit_test_name = &.{},
449 .corpus = .empty,
450 .corpus_directory = undefined,
451 .traced_comparisons = .empty,
452};
453635
454/// Invalid until `fuzzer_init` is called.636export fn __sanitizer_cov_trace_const_cmp8(const_arg: u64, arg: u64) void {
455export fn fuzzer_coverage_id() u64 {637 _ = arg;
456 return fuzzer.coverage_id;638 genericConstCmp(u64, const_arg, "const_vals8");
457}639}
458640
459var fuzzer_one: *const fn (input_ptr: [*]const u8, input_len: usize) callconv(.c) void = undefined;641export fn __sanitizer_cov_trace_switch(val: u64, cases: [*]const u64) void {
642 _ = val;
643 if (!inst.constPcSeen(@returnAddress())) {
644 @branchHint(.unlikely);
645 const case_bits = cases[1];
646 const cases_slice = cases[2..][0..cases[0]];
647 switch (case_bits) {
648 // 8-bit cases are ignored because they are likely to be randomly generated
649 0...8 => {},
650 9...16 => for (cases_slice) |c|
651 inst.const_vals2.append(gpa, @truncate(c)) catch @panic("OOM"),
652 17...32 => for (cases_slice) |c|
653 inst.const_vals4.append(gpa, @truncate(c)) catch @panic("OOM"),
654 33...64 => for (cases_slice) |c|
655 inst.const_vals8.append(gpa, @truncate(c)) catch @panic("OOM"),
656 else => {}, // Should be impossible
657 }
658 }
659}
460660
461export fn fuzzer_start(testOne: @TypeOf(fuzzer_one)) void {661export fn __sanitizer_cov_trace_cmp1(arg1: u8, arg2: u8) void {
462 fuzzer_one = testOne;662 _ = arg1;
463 fuzzer.start() catch |err| oom(err);663 _ = arg2;
464}664}
465665
466export fn fuzzer_set_name(name_ptr: [*]const u8, name_len: usize) void {666export fn __sanitizer_cov_trace_cmp2(arg1: u16, arg2: u16) void {
467 fuzzer.unit_test_name = name_ptr[0..name_len];667 _ = arg1;
668 _ = arg2;
468}669}
469670
470export fn fuzzer_init(cache_dir_struct: Fuzzer.Slice) void {671export fn __sanitizer_cov_trace_cmp4(arg1: u32, arg2: u32) void {
471 // Linkers are expected to automatically add `__start_<section>` and672 _ = arg1;
472 // `__stop_<section>` symbols when section names are valid C identifiers.673 _ = arg2;
473674}
474 const ofmt = builtin.object_format;675
475676export fn __sanitizer_cov_trace_cmp8(arg1: u64, arg2: u64) void {
476 const start_symbol_prefix: []const u8 = if (ofmt == .macho)677 _ = arg1;
477 "\x01section$start$__DATA$__"678 _ = arg2;
478 else679}
479 "__start___";680
480 const end_symbol_prefix: []const u8 = if (ofmt == .macho)681export fn __sanitizer_cov_trace_pc_indir(callee: usize) void {
481 "\x01section$end$__DATA$__"682 // Not valuable because we already have pc tracing via 8bit counters.
482 else683 _ = callee;
483 "__stop___";684}
484685export fn __sanitizer_cov_8bit_counters_init(start: usize, end: usize) void {
485 const pc_counters_start_name = start_symbol_prefix ++ "sancov_cntrs";686 // clang will emit a call to this function when compiling with code coverage instrumentation.
486 const pc_counters_start = @extern([*]u8, .{687 // however, fuzzer_init() does not need this information since it directly reads from the
487 .name = pc_counters_start_name,688 // symbol table.
488 .linkage = .weak,689 _ = start;
489 }) orelse fatal("missing {s} symbol", .{pc_counters_start_name});690 _ = end;
490691}
491 const pc_counters_end_name = end_symbol_prefix ++ "sancov_cntrs";692export fn __sanitizer_cov_pcs_init(start: usize, end: usize) void {
492 const pc_counters_end = @extern([*]u8, .{693 // clang will emit a call to this function when compiling with code coverage instrumentation.
493 .name = pc_counters_end_name,694 // however, fuzzer_init() does not need this information since it directly reads from the
494 .linkage = .weak,695 // symbol table.
495 }) orelse fatal("missing {s} symbol", .{pc_counters_end_name});696 _ = start;
496697 _ = end;
497 const pc_counters = pc_counters_start[0 .. pc_counters_end - pc_counters_start];698}
498
499 const pcs_start_name = start_symbol_prefix ++ "sancov_pcs1";
500 const pcs_start = @extern([*]usize, .{
501 .name = pcs_start_name,
502 .linkage = .weak,
503 }) orelse fatal("missing {s} symbol", .{pcs_start_name});
504
505 const pcs_end_name = end_symbol_prefix ++ "sancov_pcs1";
506 const pcs_end = @extern([*]usize, .{
507 .name = pcs_end_name,
508 .linkage = .weak,
509 }) orelse fatal("missing {s} symbol", .{pcs_end_name});
510
511 const pcs = pcs_start[0 .. pcs_end - pcs_start];
512
513 const cache_dir_path = cache_dir_struct.toZig();
514 const cache_dir = if (cache_dir_path.len == 0)
515 std.fs.cwd()
516 else
517 std.fs.cwd().makeOpenPath(cache_dir_path, .{ .iterate = true }) catch |err| {
518 fatal("unable to open fuzz directory '{s}': {s}", .{ cache_dir_path, @errorName(err) });
519 };
520699
521 fuzzer.init(cache_dir, pc_counters, pcs) catch |err|700/// Copy all of source into dest at position 0.
522 fatal("unable to init fuzzer: {s}", .{@errorName(err)});701/// If the slices overlap, dest.ptr must be <= src.ptr.
702fn volatileCopyForwards(comptime T: type, dest: []volatile T, source: []const volatile T) void {
703 for (dest, source) |*d, s| d.* = s;
523}704}
524705
525export fn fuzzer_init_corpus_elem(input_ptr: [*]const u8, input_len: usize) void {706/// Copy all of source into dest at position 0.
526 fuzzer.addCorpusElem(input_ptr[0..input_len]) catch |err|707/// If the slices overlap, dest.ptr must be >= src.ptr.
527 fatal("failed to add corpus element: {s}", .{@errorName(err)});708fn volatileCopyBackwards(comptime T: type, dest: []volatile T, source: []const volatile T) void {
709 var i = source.len;
710 while (i > 0) {
711 i -= 1;
712 dest[i] = source[i];
713 }
528}714}
529715
716const Mutation = enum {
717 /// Applies .insert_*_span, .push_*_span
718 /// For wtf-8, this limits code units, not code points
719 const max_insert_len = 12;
720 /// Applies to .insert_large_*_span and .push_large_*_span
721 /// 4096 is used as it is a common sector size
722 const max_large_insert_len = 4096;
723 /// Applies to .delete_span and .pop_span
724 const max_delete_len = 16;
725 /// Applies to .set_*span, .move_span, .set_existing_span
726 const max_set_len = 12;
727 const max_replicate_len = 64;
728 const AddValue = i6;
729 const SmallValue = i10;
730
731 delete_byte,
732 delete_span,
733 /// Removes the last byte from the input
734 pop_byte,
735 pop_span,
736 /// Inserts a group of bytes which is already in the input and removes the original copy.
737 move_span,
738 /// Replaces a group of bytes in the input with another group of bytes in the input
739 set_existing_span,
740 insert_existing_span,
741 push_existing_span,
742 set_rng_byte,
743 set_rng_span,
744 insert_rng_byte,
745 insert_rng_span,
746 /// Adds a byte to the end of the input
747 push_rng_byte,
748 push_rng_span,
749 set_zero_byte,
750 set_zero_span,
751 insert_zero_byte,
752 insert_zero_span,
753 push_zero_byte,
754 push_zero_span,
755 /// Inserts a lot of zeros to the end of the input
756 /// This is intended to work with fuzz tests that require data in (large) blocks
757 push_large_zero_span,
758 /// Inserts a group of ascii printable character
759 insert_print_span,
760 /// Inserts a group of character from a...z, A...Z, 0...9, _, and ' '
761 insert_common_span,
762 /// Inserts a group of ascii digits possibly preceded by a `-`
763 insert_integer,
764 /// Code units are evenly distributed between one to four
765 insert_wtf8_char,
766 insert_wtf8_span,
767 /// Inserts a group of bytes from another input
768 insert_splice_span,
769 // utf16 is not yet included since insertion of random bytes should adaquetly check
770 // BMP character, surrogate handling, and occasionally chacters outside of the BMP.
771 set_print_span,
772 set_common_span,
773 set_splice_span,
774 /// Similar to set_splice_span, but the bytes are copied to the same index instead of a random
775 replicate_splice_span,
776 push_print_span,
777 push_common_span,
778 push_integer,
779 push_wtf8_char,
780 push_wtf8_span,
781 push_splice_span,
782 /// Clears a random amount of high bits of a byte
783 truncate_8,
784 truncate_16le,
785 truncate_16be,
786 truncate_32le,
787 truncate_32be,
788 truncate_64le,
789 truncate_64be,
790 /// Flips a random bit
791 xor_1,
792 /// Swaps up to three bits of a byte biased to less bits
793 xor_few_8,
794 /// Swaps up to six bits of a 16-bit value biased to less bits
795 xor_few_16,
796 /// Swaps up to nine bits of a 32-bit value biased to less bits
797 xor_few_32,
798 /// Swaps up to twelve bits of 64-bit value biased to less bits
799 xor_few_64,
800 /// Adds to a byte a value of type AddValue
801 add_8,
802 add_16le,
803 add_16be,
804 add_32le,
805 add_32be,
806 add_64le,
807 add_64be,
808 /// Sets a 16-bit little-endian value to a value of type SmallValue
809 set_small_16le,
810 set_small_16be,
811 set_small_32le,
812 set_small_32be,
813 set_small_64le,
814 set_small_64be,
815 insert_small_16le,
816 insert_small_16be,
817 insert_small_32le,
818 insert_small_32be,
819 insert_small_64le,
820 insert_small_64be,
821 push_small_16le,
822 push_small_16be,
823 push_small_32le,
824 push_small_32be,
825 push_small_64le,
826 push_small_64be,
827 set_const_16,
828 set_const_32,
829 set_const_64,
830 set_const_128,
831 insert_const_16,
832 insert_const_32,
833 insert_const_64,
834 insert_const_128,
835 push_const_16,
836 push_const_32,
837 push_const_64,
838 push_const_128,
839 /// Sets a byte with up to three bits set biased to less bits
840 set_few_8,
841 /// Sets a 16-bit value with up to six bits set biased to less bits
842 set_few_16,
843 /// Sets a 32-bit value with up to nine bits set biased to less bits
844 set_few_32,
845 /// Sets a 64-bit value with up to twelve bits set biased to less bits
846 set_few_64,
847 insert_few_8,
848 insert_few_16,
849 insert_few_32,
850 insert_few_64,
851 push_few_8,
852 push_few_16,
853 push_few_32,
854 push_few_64,
855 /// Randomizes a random contigous group of bits in a byte
856 packed_set_rng_8,
857 packed_set_rng_16le,
858 packed_set_rng_16be,
859 packed_set_rng_32le,
860 packed_set_rng_32be,
861 packed_set_rng_64le,
862 packed_set_rng_64be,
863
864 fn fewValue(rng: std.Random, T: type, comptime bits: u16) T {
865 var result: T = 0;
866 var remaining_bits = rng.intRangeAtMostBiased(u16, 1, bits);
867 while (remaining_bits > 0) {
868 result |= @shlExact(@as(T, 1), rng.int(math.Log2Int(T)));
869 remaining_bits -= 1;
870 }
871 return result;
872 }
873
874 /// Returns if the mutation was applicable to the input
875 pub fn mutate(
876 mutation: Mutation,
877 rng: std.Random,
878 in: []const u8,
879 out: *MemoryMappedList,
880 corpus: []const []const u8,
881 const_vals2: []const u16,
882 const_vals4: []const u32,
883 const_vals8: []const u64,
884 const_vals16: []const u128,
885 ) bool {
886 out.clearRetainingCapacity();
887 const new_capacity = 8 + in.len + @max(
888 16, // builtin 128 value
889 Mutation.max_insert_len,
890 Mutation.max_large_insert_len,
891 );
892 out.ensureTotalCapacity(new_capacity) catch |e|
893 panic("could not resize shared input file: {t}", .{e});
894 out.items.len = 8; // Length field
895
896 const applied = switch (mutation) {
897 inline else => |m| m.comptimeMutate(
898 rng,
899 in,
900 out,
901 corpus,
902 const_vals2,
903 const_vals4,
904 const_vals8,
905 const_vals16,
906 ),
907 };
908 if (!applied)
909 assert(out.items.len == 8)
910 else
911 assert(out.items.len <= new_capacity);
912 return applied;
913 }
914
915 /// Assumes out has already been cleared
916 fn comptimeMutate(
917 comptime mutation: Mutation,
918 rng: std.Random,
919 in: []const u8,
920 out: *MemoryMappedList,
921 corpus: []const []const u8,
922 const_vals2: []const u16,
923 const_vals4: []const u32,
924 const_vals8: []const u64,
925 const_vals16: []const u128,
926 ) bool {
927 const Class = enum { new, remove, rmw, move_span, replicate_splice_span };
928 const class: Class, const class_ctx = switch (mutation) {
929 // zig fmt: off
930 .move_span => .{ .move_span, null },
931 .replicate_splice_span => .{ .replicate_splice_span, null },
932
933 .delete_byte => .{ .remove, .{ .delete, 1 } },
934 .delete_span => .{ .remove, .{ .delete, max_delete_len } },
935
936 .pop_byte => .{ .remove, .{ .pop, 1 } },
937 .pop_span => .{ .remove, .{ .pop, max_delete_len } },
938
939 .set_rng_byte => .{ .new, .{ .set , 1, .rng , .one } },
940 .set_zero_byte => .{ .new, .{ .set , 1, .zero , .one } },
941 .set_rng_span => .{ .new, .{ .set , 1, .rng , .many } },
942 .set_zero_span => .{ .new, .{ .set , 1, .zero , .many } },
943 .set_common_span => .{ .new, .{ .set , 1, .common , .many } },
944 .set_print_span => .{ .new, .{ .set , 1, .print , .many } },
945 .set_existing_span => .{ .new, .{ .set , 2, .existing, .many } },
946 .set_splice_span => .{ .new, .{ .set , 1, .splice , .many } },
947 .set_const_16 => .{ .new, .{ .set , 2, .@"const", const_vals2 } },
948 .set_const_32 => .{ .new, .{ .set , 4, .@"const", const_vals4 } },
949 .set_const_64 => .{ .new, .{ .set , 8, .@"const", const_vals8 } },
950 .set_const_128 => .{ .new, .{ .set , 16, .@"const", const_vals16 } },
951 .set_small_16le => .{ .new, .{ .set , 2, .small , .{ i16, .little } } },
952 .set_small_32le => .{ .new, .{ .set , 4, .small , .{ i32, .little } } },
953 .set_small_64le => .{ .new, .{ .set , 8, .small , .{ i64, .little } } },
954 .set_small_16be => .{ .new, .{ .set , 2, .small , .{ i16, .big } } },
955 .set_small_32be => .{ .new, .{ .set , 4, .small , .{ i32, .big } } },
956 .set_small_64be => .{ .new, .{ .set , 8, .small , .{ i64, .big } } },
957 .set_few_8 => .{ .new, .{ .set , 1, .few , .{ u8 , 3 } } },
958 .set_few_16 => .{ .new, .{ .set , 2, .few , .{ u16, 6 } } },
959 .set_few_32 => .{ .new, .{ .set , 4, .few , .{ u32, 9 } } },
960 .set_few_64 => .{ .new, .{ .set , 8, .few , .{ u64, 12 } } },
961
962 .insert_rng_byte => .{ .new, .{ .insert, 0, .rng , .one } },
963 .insert_zero_byte => .{ .new, .{ .insert, 0, .zero , .one } },
964 .insert_rng_span => .{ .new, .{ .insert, 0, .rng , .many } },
965 .insert_zero_span => .{ .new, .{ .insert, 0, .zero , .many } },
966 .insert_print_span => .{ .new, .{ .insert, 0, .print , .many } },
967 .insert_common_span => .{ .new, .{ .insert, 0, .common , .many } },
968 .insert_integer => .{ .new, .{ .insert, 0, .integer , .many } },
969 .insert_wtf8_char => .{ .new, .{ .insert, 0, .wtf8 , .one } },
970 .insert_wtf8_span => .{ .new, .{ .insert, 0, .wtf8 , .many } },
971 .insert_existing_span => .{ .new, .{ .insert, 1, .existing, .many } },
972 .insert_splice_span => .{ .new, .{ .insert, 0, .splice , .many } },
973 .insert_const_16 => .{ .new, .{ .insert, 0, .@"const", const_vals2 } },
974 .insert_const_32 => .{ .new, .{ .insert, 0, .@"const", const_vals4 } },
975 .insert_const_64 => .{ .new, .{ .insert, 0, .@"const", const_vals8 } },
976 .insert_const_128 => .{ .new, .{ .insert, 0, .@"const", const_vals16 } },
977 .insert_small_16le => .{ .new, .{ .insert, 0, .small , .{ i16, .little } } },
978 .insert_small_32le => .{ .new, .{ .insert, 0, .small , .{ i32, .little } } },
979 .insert_small_64le => .{ .new, .{ .insert, 0, .small , .{ i64, .little } } },
980 .insert_small_16be => .{ .new, .{ .insert, 0, .small , .{ i16, .big } } },
981 .insert_small_32be => .{ .new, .{ .insert, 0, .small , .{ i32, .big } } },
982 .insert_small_64be => .{ .new, .{ .insert, 0, .small , .{ i64, .big } } },
983 .insert_few_8 => .{ .new, .{ .insert, 0, .few , .{ u8 , 3 } } },
984 .insert_few_16 => .{ .new, .{ .insert, 0, .few , .{ u16, 6 } } },
985 .insert_few_32 => .{ .new, .{ .insert, 0, .few , .{ u32, 9 } } },
986 .insert_few_64 => .{ .new, .{ .insert, 0, .few , .{ u64, 12 } } },
987
988 .push_rng_byte => .{ .new, .{ .push , 0, .rng , .one } },
989 .push_zero_byte => .{ .new, .{ .push , 0, .zero , .one } },
990 .push_rng_span => .{ .new, .{ .push , 0, .rng , .many } },
991 .push_zero_span => .{ .new, .{ .push , 0, .zero , .many } },
992 .push_print_span => .{ .new, .{ .push , 0, .print , .many } },
993 .push_common_span => .{ .new, .{ .push , 0, .common , .many } },
994 .push_integer => .{ .new, .{ .push , 0, .integer , .many } },
995 .push_large_zero_span => .{ .new, .{ .push , 0, .zero , .large } },
996 .push_wtf8_char => .{ .new, .{ .push , 0, .wtf8 , .one } },
997 .push_wtf8_span => .{ .new, .{ .push , 0, .wtf8 , .many } },
998 .push_existing_span => .{ .new, .{ .push , 1, .existing, .many } },
999 .push_splice_span => .{ .new, .{ .push , 0, .splice , .many } },
1000 .push_const_16 => .{ .new, .{ .push , 0, .@"const", const_vals2 } },
1001 .push_const_32 => .{ .new, .{ .push , 0, .@"const", const_vals4 } },
1002 .push_const_64 => .{ .new, .{ .push , 0, .@"const", const_vals8 } },
1003 .push_const_128 => .{ .new, .{ .push , 0, .@"const", const_vals16 } },
1004 .push_small_16le => .{ .new, .{ .push , 0, .small , .{ i16, .little } } },
1005 .push_small_32le => .{ .new, .{ .push , 0, .small , .{ i32, .little } } },
1006 .push_small_64le => .{ .new, .{ .push , 0, .small , .{ i64, .little } } },
1007 .push_small_16be => .{ .new, .{ .push , 0, .small , .{ i16, .big } } },
1008 .push_small_32be => .{ .new, .{ .push , 0, .small , .{ i32, .big } } },
1009 .push_small_64be => .{ .new, .{ .push , 0, .small , .{ i64, .big } } },
1010 .push_few_8 => .{ .new, .{ .push , 0, .few , .{ u8 , 3 } } },
1011 .push_few_16 => .{ .new, .{ .push , 0, .few , .{ u16, 6 } } },
1012 .push_few_32 => .{ .new, .{ .push , 0, .few , .{ u32, 9 } } },
1013 .push_few_64 => .{ .new, .{ .push , 0, .few , .{ u64, 12 } } },
1014
1015 .xor_1 => .{ .rmw, .{ .xor , u8 , native_endian, 1 } },
1016 .xor_few_8 => .{ .rmw, .{ .xor , u8 , native_endian, 3 } },
1017 .xor_few_16 => .{ .rmw, .{ .xor , u16, native_endian, 6 } },
1018 .xor_few_32 => .{ .rmw, .{ .xor , u32, native_endian, 9 } },
1019 .xor_few_64 => .{ .rmw, .{ .xor , u64, native_endian, 12 } },
1020
1021 .truncate_8 => .{ .rmw, .{ .truncate , u8 , native_endian, {} } },
1022 .truncate_16le => .{ .rmw, .{ .truncate , u16, .little , {} } },
1023 .truncate_32le => .{ .rmw, .{ .truncate , u32, .little , {} } },
1024 .truncate_64le => .{ .rmw, .{ .truncate , u64, .little , {} } },
1025 .truncate_16be => .{ .rmw, .{ .truncate , u16, .big , {} } },
1026 .truncate_32be => .{ .rmw, .{ .truncate , u32, .big , {} } },
1027 .truncate_64be => .{ .rmw, .{ .truncate , u64, .big , {} } },
1028
1029 .add_8 => .{ .rmw, .{ .add , i8 , native_endian, {} } },
1030 .add_16le => .{ .rmw, .{ .add , i16, .little , {} } },
1031 .add_32le => .{ .rmw, .{ .add , i32, .little , {} } },
1032 .add_64le => .{ .rmw, .{ .add , i64, .little , {} } },
1033 .add_16be => .{ .rmw, .{ .add , i16, .big , {} } },
1034 .add_32be => .{ .rmw, .{ .add , i32, .big , {} } },
1035 .add_64be => .{ .rmw, .{ .add , i64, .big , {} } },
1036
1037 .packed_set_rng_8 => .{ .rmw, .{ .packed_rng, u8 , native_endian, {} } },
1038 .packed_set_rng_16le => .{ .rmw, .{ .packed_rng, u16, .little , {} } },
1039 .packed_set_rng_32le => .{ .rmw, .{ .packed_rng, u32, .little , {} } },
1040 .packed_set_rng_64le => .{ .rmw, .{ .packed_rng, u64, .little , {} } },
1041 .packed_set_rng_16be => .{ .rmw, .{ .packed_rng, u16, .big , {} } },
1042 .packed_set_rng_32be => .{ .rmw, .{ .packed_rng, u32, .big , {} } },
1043 .packed_set_rng_64be => .{ .rmw, .{ .packed_rng, u64, .big , {} } },
1044 // zig fmt: on
1045 };
1046
1047 switch (class) {
1048 .new => {
1049 const op: enum {
1050 set,
1051 insert,
1052 push,
1053
1054 pub fn maxLen(comptime op: @This(), in_len: usize) usize {
1055 return switch (op) {
1056 .set => @min(in_len, max_set_len),
1057 .insert, .push => max_insert_len,
1058 };
1059 }
1060 }, const min_in_len, const data: enum {
1061 rng,
1062 zero,
1063 common,
1064 print,
1065 integer,
1066 wtf8,
1067 existing,
1068 splice,
1069 @"const",
1070 small,
1071 few,
1072 }, const data_ctx = class_ctx;
1073 const Size = enum { one, many, large };
1074 if (in.len < min_in_len) return false;
1075 if (data == .@"const" and data_ctx.len == 0) return false;
1076
1077 const splice_i = if (data == .splice) blk: {
1078 // Element zero always holds an empty input, so we do not select it
1079 if (corpus.len == 1) return false;
1080 break :blk rng.intRangeLessThanBiased(usize, 1, corpus.len);
1081 } else undefined;
1082
1083 // Only needs to be followed for set
1084 const len = switch (data) {
1085 else => switch (@as(Size, data_ctx)) {
1086 .one => 1,
1087 .many => rng.intRangeAtMostBiased(usize, 1, op.maxLen(in.len)),
1088 .large => rng.intRangeAtMostBiased(usize, 1, max_large_insert_len),
1089 },
1090 .wtf8 => undefined, // varies by size of each code unit
1091 .splice => rng.intRangeAtMostBiased(usize, 1, @min(
1092 corpus[splice_i].len,
1093 op.maxLen(in.len),
1094 )),
1095 .existing => rng.intRangeAtMostBiased(usize, 1, @min(
1096 in.len,
1097 op.maxLen(in.len),
1098 )),
1099 .@"const" => @sizeOf(@typeInfo(@TypeOf(data_ctx)).pointer.child),
1100 .small, .few => @sizeOf(data_ctx[0]),
1101 };
1102
1103 const i = switch (op) {
1104 .set => rng.uintAtMostBiased(usize, in.len - len),
1105 .insert => rng.uintAtMostBiased(usize, in.len),
1106 .push => in.len,
1107 };
1108
1109 out.appendSliceAssumeCapacity(in[0..i]);
1110 switch (data) {
1111 .rng => {
1112 var bytes: [@max(max_insert_len, max_set_len)]u8 = undefined;
1113 rng.bytes(bytes[0..len]);
1114 out.appendSliceAssumeCapacity(bytes[0..len]);
1115 },
1116 .zero => out.appendNTimesAssumeCapacity(0, len),
1117 .common => for (out.addManyAsSliceAssumeCapacity(len)) |*c| {
1118 c.* = switch (rng.int(u6)) {
1119 0 => ' ',
1120 1...10 => |x| '0' + (@as(u8, x) - 1),
1121 11...36 => |x| 'A' + (@as(u8, x) - 11),
1122 37 => '_',
1123 38...63 => |x| 'a' + (@as(u8, x) - 38),
1124 };
1125 },
1126 .print => for (out.addManyAsSliceAssumeCapacity(len)) |*c| {
1127 c.* = rng.intRangeAtMostBiased(u8, 0x20, 0x7E);
1128 },
1129 .integer => {
1130 const negative = len != 0 and rng.boolean();
1131 if (negative) {
1132 out.appendAssumeCapacity('-');
1133 }
1134
1135 for (out.addManyAsSliceAssumeCapacity(len - @intFromBool(negative))) |*c| {
1136 c.* = rng.intRangeAtMostBiased(u8, '0', '9');
1137 }
1138 },
1139 .wtf8 => {
1140 comptime assert(op != .set);
1141 var codepoints: usize = if (data_ctx == .one)
1142 1
1143 else
1144 rng.intRangeAtMostBiased(usize, 1, Mutation.max_insert_len / 4);
1145
1146 while (true) {
1147 const units1 = rng.int(u2);
1148 const value = switch (units1) {
1149 0 => rng.int(u7),
1150 1 => rng.intRangeAtMostBiased(u11, 0x000080, 0x0007FF),
1151 2 => rng.intRangeAtMostBiased(u16, 0x000800, 0x00FFFF),
1152 3 => rng.intRangeAtMostBiased(u21, 0x010000, 0x10FFFF),
1153 };
1154 const units = @as(u3, units1) + 1;
1155
1156 var buf: [4]u8 = undefined;
1157 assert(std.unicode.wtf8Encode(value, &buf) catch unreachable == units);
1158 out.appendSliceAssumeCapacity(buf[0..units]);
1159
1160 codepoints -= 1;
1161 if (codepoints == 0) break;
1162 }
1163 },
1164 .existing => {
1165 const j = rng.uintAtMostBiased(usize, in.len - len);
1166 out.appendSliceAssumeCapacity(in[j..][0..len]);
1167 },
1168 .splice => {
1169 const j = rng.uintAtMostBiased(usize, corpus[splice_i].len - len);
1170 out.appendSliceAssumeCapacity(corpus[splice_i][j..][0..len]);
1171 },
1172 .@"const" => out.appendSliceAssumeCapacity(mem.asBytes(
1173 &data_ctx[rng.uintLessThanBiased(usize, data_ctx.len)],
1174 )),
1175 .small => out.appendSliceAssumeCapacity(mem.asBytes(
1176 &mem.nativeTo(data_ctx[0], rng.int(SmallValue), data_ctx[1]),
1177 )),
1178 .few => out.appendSliceAssumeCapacity(mem.asBytes(
1179 &fewValue(rng, data_ctx[0], data_ctx[1]),
1180 )),
1181 }
1182 switch (op) {
1183 .set => out.appendSliceAssumeCapacity(in[i + len ..]),
1184 .insert => out.appendSliceAssumeCapacity(in[i..]),
1185 .push => {},
1186 }
1187 },
1188 .remove => {
1189 if (in.len == 0) return false;
1190 const Op = enum { delete, pop };
1191 const op: Op, const max_len = class_ctx;
1192 // LessThan is used so we don't delete the entire span (which is unproductive since
1193 // an empty input has always been tried)
1194 const len = if (max_len == 1) 1 else rng.uintLessThanBiased(
1195 usize,
1196 @min(max_len + 1, in.len),
1197 );
1198 switch (op) {
1199 .delete => {
1200 const i = rng.uintAtMostBiased(usize, in.len - len);
1201 out.appendSliceAssumeCapacity(in[0..i]);
1202 out.appendSliceAssumeCapacity(in[i + len ..]);
1203 },
1204 .pop => out.appendSliceAssumeCapacity(in[0 .. in.len - len]),
1205 }
1206 },
1207 .rmw => {
1208 const Op = enum { xor, truncate, add, packed_rng };
1209 const op: Op, const T, const endian, const xor_bits = class_ctx;
1210 if (in.len < @sizeOf(T)) return false;
1211 const Log2T = math.Log2Int(T);
1212
1213 const idx = rng.uintAtMostBiased(usize, in.len - @sizeOf(T));
1214 const old = mem.readInt(T, in[idx..][0..@sizeOf(T)], endian);
1215 const new = switch (op) {
1216 .xor => old ^ fewValue(rng, T, xor_bits),
1217 .truncate => old & (@as(T, math.maxInt(T)) >> rng.int(Log2T)),
1218 .add => old +% addend: {
1219 const val = rng.int(Mutation.AddValue);
1220 break :addend if (val == 0) 1 else val;
1221 },
1222 .packed_rng => blk: {
1223 const bits = rng.int(math.Log2Int(T)) +| 1;
1224 break :blk old ^ (rng.int(T) >> bits << rng.uintAtMostBiased(Log2T, bits));
1225 },
1226 };
1227 out.appendSliceAssumeCapacity(in);
1228 mem.bytesAsValue(T, out.items[8..][idx..][0..@sizeOf(T)]).* =
1229 mem.nativeTo(T, new, endian);
1230 },
1231 .move_span => {
1232 if (in.len < 2) return false;
1233 // One less since moving whole output will never change anything
1234 const len = rng.intRangeAtMostBiased(usize, 1, @min(
1235 in.len - 1,
1236 Mutation.max_set_len,
1237 ));
1238
1239 const src = rng.uintAtMostBiased(usize, in.len - len);
1240 // This indexes into the final input
1241 const dst = blk: {
1242 const res = rng.uintAtMostBiased(usize, in.len - len - 1);
1243 break :blk res + @intFromBool(res >= src);
1244 };
1245
1246 if (src < dst) {
1247 out.appendSliceAssumeCapacity(in[0..src]);
1248 out.appendSliceAssumeCapacity(in[src + len .. dst + len]);
1249 out.appendSliceAssumeCapacity(in[src..][0..len]);
1250 out.appendSliceAssumeCapacity(in[dst + len ..]);
1251 } else {
1252 out.appendSliceAssumeCapacity(in[0..dst]);
1253 out.appendSliceAssumeCapacity(in[src..][0..len]);
1254 out.appendSliceAssumeCapacity(in[dst..src]);
1255 out.appendSliceAssumeCapacity(in[src + len ..]);
1256 }
1257 },
1258 .replicate_splice_span => {
1259 if (in.len == 0) return false;
1260 if (corpus.len == 1) return false;
1261 const from = corpus[rng.intRangeLessThanBiased(usize, 1, corpus.len)];
1262 const len = rng.uintLessThanBiased(usize, @min(in.len, from.len, max_replicate_len));
1263 const i = rng.uintAtMostBiased(usize, @min(in.len, from.len) - len);
1264 out.appendSliceAssumeCapacity(in[0..i]);
1265 out.appendSliceAssumeCapacity(from[i..][0..len]);
1266 out.appendSliceAssumeCapacity(in[i + len ..]);
1267 },
1268 }
1269 return true;
1270 }
1271};
1272
530/// Like `std.ArrayListUnmanaged(u8)` but backed by memory mapping.1273/// Like `std.ArrayListUnmanaged(u8)` but backed by memory mapping.
531pub const MemoryMappedList = struct {1274pub const MemoryMappedList = struct {
532 /// Contents of the list.1275 /// Contents of the list.
...@@ -654,8 +1397,23 @@ pub const MemoryMappedList = struct {...@@ -654,8 +1397,23 @@ pub const MemoryMappedList = struct {
654 fn growCapacity(current: usize, minimum: usize) usize {1397 fn growCapacity(current: usize, minimum: usize) usize {
655 var new = current;1398 var new = current;
656 while (true) {1399 while (true) {
657 new = std.mem.alignForward(usize, new + new / 2, std.heap.page_size_max);1400 new = mem.alignForward(usize, new + new / 2, std.heap.page_size_max);
658 if (new >= minimum) return new;1401 if (new >= minimum) return new;
659 }1402 }
660 }1403 }
1404
1405 pub fn insertAssumeCapacity(l: *MemoryMappedList, i: usize, item: u8) void {
1406 assert(l.items.len + 1 <= l.capacity);
1407 l.items.len += 1;
1408 volatileCopyBackwards(u8, l.items[i + 1 ..], l.items[i .. l.items.len - 1]);
1409 l.items[i] = item;
1410 }
1411
1412 pub fn orderedRemove(l: *MemoryMappedList, i: usize) u8 {
1413 assert(l.items.len + 1 <= l.capacity);
1414 const old = l.items[i];
1415 volatileCopyForwards(u8, l.items[i .. l.items.len - 1], l.items[i + 1 ..]);
1416 l.items.len -= 1;
1417 return old;
1418 }
661};1419};
lib/std/Build/Fuzz.zig+2-3
...@@ -252,9 +252,8 @@ pub fn sendUpdate(...@@ -252,9 +252,8 @@ pub fn sendUpdate(
252 const seen_pcs = cov_header.seenBits();252 const seen_pcs = cov_header.seenBits();
253 const n_runs = @atomicLoad(usize, &cov_header.n_runs, .monotonic);253 const n_runs = @atomicLoad(usize, &cov_header.n_runs, .monotonic);
254 const unique_runs = @atomicLoad(usize, &cov_header.unique_runs, .monotonic);254 const unique_runs = @atomicLoad(usize, &cov_header.unique_runs, .monotonic);
255 if (prev.unique_runs != unique_runs) {255 {
256 // There has been an update.256 if (unique_runs != 0 and prev.unique_runs == 0) {
257 if (prev.unique_runs == 0) {
258 // We need to send initial context.257 // We need to send initial context.
259 const header: abi.SourceIndexHeader = .{258 const header: abi.SourceIndexHeader = .{
260 .directories_len = @intCast(coverage_map.coverage.directories.entries.len),259 .directories_len = @intCast(coverage_map.coverage.directories.entries.len),
lib/std/Build/abi.zig+20
...@@ -138,6 +138,26 @@ pub const Rebuild = extern struct {...@@ -138,6 +138,26 @@ pub const Rebuild = extern struct {
138138
139/// ABI bits specifically relating to the fuzzer interface.139/// ABI bits specifically relating to the fuzzer interface.
140pub const fuzz = struct {140pub const fuzz = struct {
141 pub const TestOne = *const fn (Slice) callconv(.c) void;
142 pub extern fn fuzzer_init(cache_dir_path: Slice) void;
143 pub extern fn fuzzer_coverage_id() u64;
144 pub extern fn fuzzer_init_test(test_one: TestOne, unit_test_name: Slice) void;
145 pub extern fn fuzzer_new_input(bytes: Slice) void;
146 pub extern fn fuzzer_main() void;
147
148 pub const Slice = extern struct {
149 ptr: [*]const u8,
150 len: usize,
151
152 pub fn toSlice(s: Slice) []const u8 {
153 return s.ptr[0..s.len];
154 }
155
156 pub fn fromSlice(s: []const u8) Slice {
157 return .{ .ptr = s.ptr, .len = s.len };
158 }
159 };
160
141 /// libfuzzer uses this and its usize is the one that counts. To match the ABI,161 /// libfuzzer uses this and its usize is the one that counts. To match the ABI,
142 /// make the ints be the size of the target used with libfuzzer.162 /// make the ints be the size of the target used with libfuzzer.
143 ///163 ///
lib/std/json/scanner_test.zig+17
...@@ -517,3 +517,20 @@ test isNumberFormattedLikeAnInteger {...@@ -517,3 +517,20 @@ test isNumberFormattedLikeAnInteger {
517 try std.testing.expect(!isNumberFormattedLikeAnInteger("1e10"));517 try std.testing.expect(!isNumberFormattedLikeAnInteger("1e10"));
518 try std.testing.expect(!isNumberFormattedLikeAnInteger("1E10"));518 try std.testing.expect(!isNumberFormattedLikeAnInteger("1E10"));
519}519}
520
521test "fuzz" {
522 try std.testing.fuzz({}, fuzzTestOne, .{});
523}
524
525fn fuzzTestOne(_: void, input: []const u8) !void {
526 var buf: [16384]u8 = undefined;
527 var fba: std.heap.FixedBufferAllocator = .init(&buf);
528
529 var scanner = Scanner.initCompleteInput(fba.allocator(), input);
530 // Property: There are at most input.len tokens
531 var tokens: usize = 0;
532 while ((scanner.next() catch return) != .end_of_document) {
533 tokens += 1;
534 if (tokens > input.len) return error.Overflow;
535 }
536}
lib/std/zig/parser_test.zig+16
...@@ -6451,3 +6451,19 @@ fn testError(source: [:0]const u8, expected_errors: []const Error) !void {...@@ -6451,3 +6451,19 @@ fn testError(source: [:0]const u8, expected_errors: []const Error) !void {
6451 try std.testing.expectEqual(expected, tree.errors[i].tag);6451 try std.testing.expectEqual(expected, tree.errors[i].tag);
6452 }6452 }
6453}6453}
6454
6455test "fuzz ast parse" {
6456 try std.testing.fuzz({}, fuzzTestOneParse, .{});
6457}
6458
6459fn fuzzTestOneParse(_: void, input: []const u8) !void {
6460 // The first byte holds if zig / zon
6461 if (input.len == 0) return;
6462 const mode: std.zig.Ast.Mode = if (input[0] & 1 == 0) .zig else .zon;
6463 const bytes = input[1..];
6464
6465 var fba: std.heap.FixedBufferAllocator = .init(&fixed_buffer_mem);
6466 const allocator = fba.allocator();
6467 const source = allocator.dupeZ(u8, bytes) catch return;
6468 _ = std.zig.Ast.parse(allocator, source, mode) catch return;
6469}
lib/std/zig/tokenizer.zig+15-14
...@@ -1721,10 +1721,14 @@ fn testTokenize(source: [:0]const u8, expected_token_tags: []const Token.Tag) !v...@@ -1721,10 +1721,14 @@ fn testTokenize(source: [:0]const u8, expected_token_tags: []const Token.Tag) !v
1721 try std.testing.expectEqual(source.len, last_token.loc.end);1721 try std.testing.expectEqual(source.len, last_token.loc.end);
1722}1722}
17231723
1724fn testPropertiesUpheld(context: void, source: []const u8) anyerror!void {1724fn testPropertiesUpheld(_: void, source: []const u8) !void {
1725 _ = context;1725 var source0_buf: [512]u8 = undefined;
1726 const source0 = try std.testing.allocator.dupeZ(u8, source);1726 if (source.len + 1 > source0_buf.len)
1727 defer std.testing.allocator.free(source0);1727 return;
1728 @memcpy(source0_buf[0..source.len], source);
1729 source0_buf[source.len] = 0;
1730 const source0 = source0_buf[0..source.len :0];
1731
1728 var tokenizer = Tokenizer.init(source0);1732 var tokenizer = Tokenizer.init(source0);
1729 var tokenization_failed = false;1733 var tokenization_failed = false;
1730 while (true) {1734 while (true) {
...@@ -1750,18 +1754,15 @@ fn testPropertiesUpheld(context: void, source: []const u8) anyerror!void {...@@ -1750,18 +1754,15 @@ fn testPropertiesUpheld(context: void, source: []const u8) anyerror!void {
1750 }1754 }
1751 }1755 }
17521756
1753 if (source0.len > 0) for (source0, source0[1..][0..source0.len]) |cur, next| {1757 if (tokenization_failed) return;
1758 for (source0) |cur| {
1754 // Property: No null byte allowed except at end.1759 // Property: No null byte allowed except at end.
1755 if (cur == 0) {1760 if (cur == 0) {
1756 try std.testing.expect(tokenization_failed);1761 return error.TestUnexpectedResult;
1757 }
1758 // Property: No ASCII control characters other than \n and \t are allowed.
1759 if (std.ascii.isControl(cur) and cur != '\n' and cur != '\t') {
1760 try std.testing.expect(tokenization_failed);
1761 }1762 }
1762 // Property: All '\r' must be followed by '\n'.1763 // Property: No ASCII control characters other than \n, \t, and \r are allowed.
1763 if (cur == '\r' and next != '\n') {1764 if (std.ascii.isControl(cur) and cur != '\n' and cur != '\t' and cur != '\r') {
1764 try std.testing.expect(tokenization_failed);1765 return error.TestUnexpectedResult;
1765 }1766 }
1766 };1767 }
1767}1768}
test/standalone/libfuzzer/build.zig+1
...@@ -16,6 +16,7 @@ pub fn build(b: *std.Build) void {...@@ -16,6 +16,7 @@ pub fn build(b: *std.Build) void {
16 .optimize = optimize,16 .optimize = optimize,
17 .fuzz = true,17 .fuzz = true,
18 }),18 }),
19 .use_llvm = true, // #23423
19 });20 });
2021
21 b.installArtifact(exe);22 b.installArtifact(exe);
test/standalone/libfuzzer/main.zig+35-21
...@@ -1,29 +1,43 @@...@@ -1,29 +1,43 @@
1const std = @import("std");1const std = @import("std");
2const abi = std.Build.abi.fuzz;
3const native_endian = @import("builtin").cpu.arch.endian();
24
3const FuzzerSlice = extern struct {5fn testOne(in: abi.Slice) callconv(.c) void {
4 ptr: [*]const u8,6 std.debug.assertReadable(in.toSlice());
5 len: usize,7}
68
7 fn fromSlice(s: []const u8) FuzzerSlice {9pub fn main() !void {
8 return .{ .ptr = s.ptr, .len = s.len };10 var debug_gpa_ctx: std.heap.DebugAllocator(.{}) = .init;
9 }11 defer _ = debug_gpa_ctx.deinit();
10};12 const gpa = debug_gpa_ctx.allocator();
1113
12extern fn fuzzer_set_name(name_ptr: [*]const u8, name_len: usize) void;14 var args = try std.process.argsWithAllocator(gpa);
13extern fn fuzzer_init(cache_dir: FuzzerSlice) void;15 defer args.deinit();
14extern fn fuzzer_init_corpus_elem(input_ptr: [*]const u8, input_len: usize) void;16 _ = args.skip(); // executable name
15extern fn fuzzer_coverage_id() u64;
1617
17pub fn main() !void {18 const cache_dir_path = args.next() orelse @panic("expected cache directory path argument");
18 var gpa: std.heap.GeneralPurposeAllocator(.{}) = .init;19 var cache_dir = try std.fs.cwd().openDir(cache_dir_path, .{});
19 defer _ = gpa.deinit();20 defer cache_dir.close();
20 const args = try std.process.argsAlloc(gpa.allocator());21
21 defer std.process.argsFree(gpa.allocator(), args);22 abi.fuzzer_init(.fromSlice(cache_dir_path));
23 abi.fuzzer_init_test(testOne, .fromSlice("test"));
24 abi.fuzzer_new_input(.fromSlice(""));
25 abi.fuzzer_new_input(.fromSlice("hello"));
26
27 const pc_digest = abi.fuzzer_coverage_id();
28 const coverage_file_path = "v/" ++ std.fmt.hex(pc_digest);
29 const coverage_file = try cache_dir.openFile(coverage_file_path, .{});
30 defer coverage_file.close();
2231
23 const cache_dir = args[1];32 var read_buf: [@sizeOf(abi.SeenPcsHeader)]u8 = undefined;
33 var r = coverage_file.reader(&read_buf);
34 const pcs_header = r.interface.takeStruct(abi.SeenPcsHeader, native_endian) catch return r.err.?;
2435
25 fuzzer_init(FuzzerSlice.fromSlice(cache_dir));36 if (pcs_header.pcs_len == 0)
26 fuzzer_init_corpus_elem("hello".ptr, "hello".len);37 return error.ZeroPcs;
27 fuzzer_set_name("test".ptr, "test".len);38 const expected_len = @sizeOf(abi.SeenPcsHeader) +
28 _ = fuzzer_coverage_id();39 try std.math.divCeil(usize, pcs_header.pcs_len, @bitSizeOf(usize)) * @sizeOf(usize) +
40 pcs_header.pcs_len * @sizeOf(usize);
41 if (try coverage_file.getEndPos() != expected_len)
42 return error.WrongEnd;
29}43}