| ... | @@ -1,75 +1,965 @@ | ... | @@ -1,75 +1,965 @@ |
| 1 | const assert = @import("debug.zig").assert; | 1 | const std = @import("index.zig"); |
| 2 | const mem = @import("mem.zig"); | 2 | const assert = std.debug.assert; |
| 3 | const math = @import("math/index.zig"); | 3 | const mem = std.mem; |
| | 4 | const math = std.math; |
| 4 | | 5 | |
| 5 | pub const Cmp = math.Cmp; | 6 | /// Stable in-place sort. O(n) best case, O(pow(n, 2)) worst case. O(1) memory (no allocator required). |
| 6 | | 7 | pub fn insertionSort(comptime T: type, items: []T, lessThan: fn(lhs: &const T, rhs: &const T)->bool) { |
| 7 | /// Stable sort using O(1) space. Currently implemented as insertion sort. | 8 | {var i: usize = 1; while (i < items.len) : (i += 1) { |
| 8 | pub fn sort_stable(comptime T: type, array: []T, comptime cmp: fn(a: &const T, b: &const T)->Cmp) { | 9 | const x = items[i]; |
| 9 | {var i: usize = 1; while (i < array.len) : (i += 1) { | | |
| 10 | const x = array[i]; | | |
| 11 | var j: usize = i; | 10 | var j: usize = i; |
| 12 | while (j > 0 and cmp(array[j - 1], x) == Cmp.Greater) : (j -= 1) { | 11 | while (j > 0 and lessThan(x, items[j - 1])) : (j -= 1) { |
| 13 | array[j] = array[j - 1]; | 12 | items[j] = items[j - 1]; |
| 14 | } | 13 | } |
| 15 | array[j] = x; | 14 | items[j] = x; |
| 16 | }} | 15 | }} |
| 17 | } | 16 | } |
| 18 | | 17 | |
| 19 | /// Unstable sort using O(n) stack space. Currently implemented as quicksort. | 18 | const Range = struct { |
| 20 | pub fn sort(comptime T: type, array: []T, comptime cmp: fn(a: &const T, b: &const T)->Cmp) { | 19 | start: usize, |
| 21 | if (array.len > 0) { | 20 | end: usize, |
| 22 | quicksort(T, array, 0, array.len - 1, cmp); | 21 | |
| | 22 | fn init(start: usize, end: usize) -> Range { |
| | 23 | return Range { .start = start, .end = end }; |
| | 24 | } |
| | 25 | |
| | 26 | fn length(self: &const Range) -> usize { |
| | 27 | return self.end - self.start; |
| | 28 | } |
| | 29 | }; |
| | 30 | |
| | 31 | |
| | 32 | const Iterator = struct { |
| | 33 | size: usize, |
| | 34 | power_of_two: usize, |
| | 35 | numerator: usize, |
| | 36 | decimal: usize, |
| | 37 | denominator: usize, |
| | 38 | decimal_step: usize, |
| | 39 | numerator_step: usize, |
| | 40 | |
| | 41 | fn init(size2: usize, min_level: usize) -> Iterator { |
| | 42 | const power_of_two = math.floorPowerOfTwo(usize, size2); |
| | 43 | const denominator = power_of_two / min_level; |
| | 44 | return Iterator { |
| | 45 | .numerator = 0, |
| | 46 | .decimal = 0, |
| | 47 | .size = size2, |
| | 48 | .power_of_two = power_of_two, |
| | 49 | .denominator = denominator, |
| | 50 | .decimal_step = size2 / denominator, |
| | 51 | .numerator_step = size2 % denominator, |
| | 52 | }; |
| | 53 | } |
| | 54 | |
| | 55 | fn begin(self: &Iterator) { |
| | 56 | self.numerator = 0; |
| | 57 | self.decimal = 0; |
| | 58 | } |
| | 59 | |
| | 60 | fn nextRange(self: &Iterator) -> Range { |
| | 61 | const start = self.decimal; |
| | 62 | |
| | 63 | self.decimal += self.decimal_step; |
| | 64 | self.numerator += self.numerator_step; |
| | 65 | if (self.numerator >= self.denominator) { |
| | 66 | self.numerator -= self.denominator; |
| | 67 | self.decimal += 1; |
| | 68 | } |
| | 69 | |
| | 70 | return Range {.start = start, .end = self.decimal}; |
| | 71 | } |
| | 72 | |
| | 73 | fn finished(self: &Iterator) -> bool { |
| | 74 | return self.decimal >= self.size; |
| | 75 | } |
| | 76 | |
| | 77 | fn nextLevel(self: &Iterator) -> bool { |
| | 78 | self.decimal_step += self.decimal_step; |
| | 79 | self.numerator_step += self.numerator_step; |
| | 80 | if (self.numerator_step >= self.denominator) { |
| | 81 | self.numerator_step -= self.denominator; |
| | 82 | self.decimal_step += 1; |
| | 83 | } |
| | 84 | |
| | 85 | return (self.decimal_step < self.size); |
| | 86 | } |
| | 87 | |
| | 88 | fn length(self: &Iterator) -> usize { |
| | 89 | return self.decimal_step; |
| | 90 | } |
| | 91 | }; |
| | 92 | |
| | 93 | const Pull = struct { |
| | 94 | from: usize, |
| | 95 | to: usize, |
| | 96 | count: usize, |
| | 97 | range: Range, |
| | 98 | }; |
| | 99 | |
| | 100 | /// Stable in-place sort. O(n) best case, O(n*log(n)) worst case and average case. O(1) memory (no allocator required). |
| | 101 | /// Currently implemented as block sort. |
| | 102 | pub fn sort(comptime T: type, items: []T, lessThan: fn(lhs: &const T, rhs: &const T)->bool) { |
| | 103 | // Implementation ported from https://github.com/BonzaiThePenguin/WikiSort/blob/master/WikiSort.c |
| | 104 | var cache: [512]T = undefined; |
| | 105 | |
| | 106 | if (items.len < 4) { |
| | 107 | if (items.len == 3) { |
| | 108 | // hard coded insertion sort |
| | 109 | if (lessThan(items[1], items[0])) mem.swap(T, &items[0], &items[1]); |
| | 110 | if (lessThan(items[2], items[1])) { |
| | 111 | mem.swap(T, &items[1], &items[2]); |
| | 112 | if (lessThan(items[1], items[0])) mem.swap(T, &items[0], &items[1]); |
| | 113 | } |
| | 114 | } else if (items.len == 2) { |
| | 115 | if (lessThan(items[1], items[0])) mem.swap(T, &items[0], &items[1]); |
| | 116 | } |
| | 117 | return; |
| | 118 | } |
| | 119 | |
| | 120 | // sort groups of 4-8 items at a time using an unstable sorting network, |
| | 121 | // but keep track of the original item orders to force it to be stable |
| | 122 | // http://pages.ripco.net/~jgamble/nw.html |
| | 123 | var iterator = Iterator.init(items.len, 4); |
| | 124 | while (!iterator.finished()) { |
| | 125 | var order = []u8{0, 1, 2, 3, 4, 5, 6, 7}; |
| | 126 | const range = iterator.nextRange(); |
| | 127 | |
| | 128 | const sliced_items = items[range.start..]; |
| | 129 | switch (range.length()) { |
| | 130 | 8 => { |
| | 131 | swap(T, sliced_items, lessThan, &order, 0, 1); |
| | 132 | swap(T, sliced_items, lessThan, &order, 2, 3); |
| | 133 | swap(T, sliced_items, lessThan, &order, 4, 5); |
| | 134 | swap(T, sliced_items, lessThan, &order, 6, 7); |
| | 135 | swap(T, sliced_items, lessThan, &order, 0, 2); |
| | 136 | swap(T, sliced_items, lessThan, &order, 1, 3); |
| | 137 | swap(T, sliced_items, lessThan, &order, 4, 6); |
| | 138 | swap(T, sliced_items, lessThan, &order, 5, 7); |
| | 139 | swap(T, sliced_items, lessThan, &order, 1, 2); |
| | 140 | swap(T, sliced_items, lessThan, &order, 5, 6); |
| | 141 | swap(T, sliced_items, lessThan, &order, 0, 4); |
| | 142 | swap(T, sliced_items, lessThan, &order, 3, 7); |
| | 143 | swap(T, sliced_items, lessThan, &order, 1, 5); |
| | 144 | swap(T, sliced_items, lessThan, &order, 2, 6); |
| | 145 | swap(T, sliced_items, lessThan, &order, 1, 4); |
| | 146 | swap(T, sliced_items, lessThan, &order, 3, 6); |
| | 147 | swap(T, sliced_items, lessThan, &order, 2, 4); |
| | 148 | swap(T, sliced_items, lessThan, &order, 3, 5); |
| | 149 | swap(T, sliced_items, lessThan, &order, 3, 4); |
| | 150 | }, |
| | 151 | 7 => { |
| | 152 | swap(T, sliced_items, lessThan, &order, 1, 2); |
| | 153 | swap(T, sliced_items, lessThan, &order, 3, 4); |
| | 154 | swap(T, sliced_items, lessThan, &order, 5, 6); |
| | 155 | swap(T, sliced_items, lessThan, &order, 0, 2); |
| | 156 | swap(T, sliced_items, lessThan, &order, 3, 5); |
| | 157 | swap(T, sliced_items, lessThan, &order, 4, 6); |
| | 158 | swap(T, sliced_items, lessThan, &order, 0, 1); |
| | 159 | swap(T, sliced_items, lessThan, &order, 4, 5); |
| | 160 | swap(T, sliced_items, lessThan, &order, 2, 6); |
| | 161 | swap(T, sliced_items, lessThan, &order, 0, 4); |
| | 162 | swap(T, sliced_items, lessThan, &order, 1, 5); |
| | 163 | swap(T, sliced_items, lessThan, &order, 0, 3); |
| | 164 | swap(T, sliced_items, lessThan, &order, 2, 5); |
| | 165 | swap(T, sliced_items, lessThan, &order, 1, 3); |
| | 166 | swap(T, sliced_items, lessThan, &order, 2, 4); |
| | 167 | swap(T, sliced_items, lessThan, &order, 2, 3); |
| | 168 | }, |
| | 169 | 6 => { |
| | 170 | swap(T, sliced_items, lessThan, &order, 1, 2); |
| | 171 | swap(T, sliced_items, lessThan, &order, 4, 5); |
| | 172 | swap(T, sliced_items, lessThan, &order, 0, 2); |
| | 173 | swap(T, sliced_items, lessThan, &order, 3, 5); |
| | 174 | swap(T, sliced_items, lessThan, &order, 0, 1); |
| | 175 | swap(T, sliced_items, lessThan, &order, 3, 4); |
| | 176 | swap(T, sliced_items, lessThan, &order, 2, 5); |
| | 177 | swap(T, sliced_items, lessThan, &order, 0, 3); |
| | 178 | swap(T, sliced_items, lessThan, &order, 1, 4); |
| | 179 | swap(T, sliced_items, lessThan, &order, 2, 4); |
| | 180 | swap(T, sliced_items, lessThan, &order, 1, 3); |
| | 181 | swap(T, sliced_items, lessThan, &order, 2, 3); |
| | 182 | }, |
| | 183 | 5 => { |
| | 184 | swap(T, sliced_items, lessThan, &order, 0, 1); |
| | 185 | swap(T, sliced_items, lessThan, &order, 3, 4); |
| | 186 | swap(T, sliced_items, lessThan, &order, 2, 4); |
| | 187 | swap(T, sliced_items, lessThan, &order, 2, 3); |
| | 188 | swap(T, sliced_items, lessThan, &order, 1, 4); |
| | 189 | swap(T, sliced_items, lessThan, &order, 0, 3); |
| | 190 | swap(T, sliced_items, lessThan, &order, 0, 2); |
| | 191 | swap(T, sliced_items, lessThan, &order, 1, 3); |
| | 192 | swap(T, sliced_items, lessThan, &order, 1, 2); |
| | 193 | }, |
| | 194 | 4 => { |
| | 195 | swap(T, sliced_items, lessThan, &order, 0, 1); |
| | 196 | swap(T, sliced_items, lessThan, &order, 2, 3); |
| | 197 | swap(T, sliced_items, lessThan, &order, 0, 2); |
| | 198 | swap(T, sliced_items, lessThan, &order, 1, 3); |
| | 199 | swap(T, sliced_items, lessThan, &order, 1, 2); |
| | 200 | }, |
| | 201 | else => {}, |
| | 202 | } |
| | 203 | } |
| | 204 | if (items.len < 8) return; |
| | 205 | |
| | 206 | // then merge sort the higher levels, which can be 8-15, 16-31, 32-63, 64-127, etc. |
| | 207 | while (true) { |
| | 208 | // if every A and B block will fit into the cache, use a special branch specifically for merging with the cache |
| | 209 | // (we use < rather than <= since the block size might be one more than iterator.length()) |
| | 210 | if (iterator.length() < cache.len) { |
| | 211 | // if four subarrays fit into the cache, it's faster to merge both pairs of subarrays into the cache, |
| | 212 | // then merge the two merged subarrays from the cache back into the original array |
| | 213 | if ((iterator.length() + 1) * 4 <= cache.len and iterator.length() * 4 <= items.len) { |
| | 214 | iterator.begin(); |
| | 215 | while (!iterator.finished()) { |
| | 216 | // merge A1 and B1 into the cache |
| | 217 | var A1 = iterator.nextRange(); |
| | 218 | var B1 = iterator.nextRange(); |
| | 219 | var A2 = iterator.nextRange(); |
| | 220 | var B2 = iterator.nextRange(); |
| | 221 | |
| | 222 | if (lessThan(items[B1.end - 1], items[A1.start])) { |
| | 223 | // the two ranges are in reverse order, so copy them in reverse order into the cache |
| | 224 | mem.copy(T, cache[B1.length()..], items[A1.start..A1.end]); |
| | 225 | mem.copy(T, cache[0..], items[B1.start..B1.end]); |
| | 226 | } else if (lessThan(items[B1.start], items[A1.end - 1])) { |
| | 227 | // these two ranges weren't already in order, so merge them into the cache |
| | 228 | mergeInto(T, items, A1, B1, lessThan, cache[0..]); |
| | 229 | } else { |
| | 230 | // if A1, B1, A2, and B2 are all in order, skip doing anything else |
| | 231 | if (!lessThan(items[B2.start], items[A2.end - 1]) and !lessThan(items[A2.start], items[B1.end - 1])) continue; |
| | 232 | |
| | 233 | // copy A1 and B1 into the cache in the same order |
| | 234 | mem.copy(T, cache[0..], items[A1.start..A1.end]); |
| | 235 | mem.copy(T, cache[A1.length()..], items[B1.start..B1.end]); |
| | 236 | } |
| | 237 | A1 = Range.init(A1.start, B1.end); |
| | 238 | |
| | 239 | // merge A2 and B2 into the cache |
| | 240 | if (lessThan(items[B2.end - 1], items[A2.start])) { |
| | 241 | // the two ranges are in reverse order, so copy them in reverse order into the cache |
| | 242 | mem.copy(T, cache[A1.length() + B2.length()..], items[A2.start..A2.end]); |
| | 243 | mem.copy(T, cache[A1.length()..], items[B2.start..B2.end]); |
| | 244 | } else if (lessThan(items[B2.start], items[A2.end - 1])) { |
| | 245 | // these two ranges weren't already in order, so merge them into the cache |
| | 246 | mergeInto(T, items, A2, B2, lessThan, cache[A1.length()..]); |
| | 247 | } else { |
| | 248 | // copy A2 and B2 into the cache in the same order |
| | 249 | mem.copy(T, cache[A1.length()..], items[A2.start..A2.end]); |
| | 250 | mem.copy(T, cache[A1.length() + A2.length()..], items[B2.start..B2.end]); |
| | 251 | } |
| | 252 | A2 = Range.init(A2.start, B2.end); |
| | 253 | |
| | 254 | // merge A1 and A2 from the cache into the items |
| | 255 | const A3 = Range.init(0, A1.length()); |
| | 256 | const B3 = Range.init(A1.length(), A1.length() + A2.length()); |
| | 257 | |
| | 258 | if (lessThan(cache[B3.end - 1], cache[A3.start])) { |
| | 259 | // the two ranges are in reverse order, so copy them in reverse order into the items |
| | 260 | mem.copy(T, items[A1.start + A2.length()..], cache[A3.start..A3.end]); |
| | 261 | mem.copy(T, items[A1.start..], cache[B3.start..B3.end]); |
| | 262 | } else if (lessThan(cache[B3.start], cache[A3.end - 1])) { |
| | 263 | // these two ranges weren't already in order, so merge them back into the items |
| | 264 | mergeInto(T, cache[0..], A3, B3, lessThan, items[A1.start..]); |
| | 265 | } else { |
| | 266 | // copy A3 and B3 into the items in the same order |
| | 267 | mem.copy(T, items[A1.start..], cache[A3.start..A3.end]); |
| | 268 | mem.copy(T, items[A1.start + A1.length()..], cache[B3.start..B3.end]); |
| | 269 | } |
| | 270 | } |
| | 271 | |
| | 272 | // we merged two levels at the same time, so we're done with this level already |
| | 273 | // (iterator.nextLevel() is called again at the bottom of this outer merge loop) |
| | 274 | _ = iterator.nextLevel(); |
| | 275 | |
| | 276 | } else { |
| | 277 | iterator.begin(); |
| | 278 | while (!iterator.finished()) { |
| | 279 | var A = iterator.nextRange(); |
| | 280 | var B = iterator.nextRange(); |
| | 281 | |
| | 282 | if (lessThan(items[B.end - 1], items[A.start])) { |
| | 283 | // the two ranges are in reverse order, so a simple rotation should fix it |
| | 284 | mem.rotate(T, items[A.start..B.end], A.length()); |
| | 285 | } else if (lessThan(items[B.start], items[A.end - 1])) { |
| | 286 | // these two ranges weren't already in order, so we'll need to merge them! |
| | 287 | mem.copy(T, cache[0..], items[A.start..A.end]); |
| | 288 | mergeExternal(T, items, A, B, lessThan, cache[0..]); |
| | 289 | } |
| | 290 | } |
| | 291 | } |
| | 292 | } else { |
| | 293 | // this is where the in-place merge logic starts! |
| | 294 | // 1. pull out two internal buffers each containing √A unique values |
| | 295 | // 1a. adjust block_size and buffer_size if we couldn't find enough unique values |
| | 296 | // 2. loop over the A and B subarrays within this level of the merge sort |
| | 297 | // 3. break A and B into blocks of size 'block_size' |
| | 298 | // 4. "tag" each of the A blocks with values from the first internal buffer |
| | 299 | // 5. roll the A blocks through the B blocks and drop/rotate them where they belong |
| | 300 | // 6. merge each A block with any B values that follow, using the cache or the second internal buffer |
| | 301 | // 7. sort the second internal buffer if it exists |
| | 302 | // 8. redistribute the two internal buffers back into the items |
| | 303 | |
| | 304 | var block_size: usize = math.sqrt(iterator.length()); |
| | 305 | var buffer_size = iterator.length()/block_size + 1; |
| | 306 | |
| | 307 | // as an optimization, we really only need to pull out the internal buffers once for each level of merges |
| | 308 | // after that we can reuse the same buffers over and over, then redistribute it when we're finished with this level |
| | 309 | var A: Range = undefined; |
| | 310 | var B: Range = undefined; |
| | 311 | var index: usize = 0; |
| | 312 | var last: usize = 0; |
| | 313 | var count: usize = 0; |
| | 314 | var find: usize = 0; |
| | 315 | var start: usize = 0; |
| | 316 | var pull_index: usize = 0; |
| | 317 | var pull = []Pull{ |
| | 318 | Pull {.from = 0, .to = 0, .count = 0, .range = Range.init(0, 0),}, |
| | 319 | Pull {.from = 0, .to = 0, .count = 0, .range = Range.init(0, 0),}, |
| | 320 | }; |
| | 321 | |
| | 322 | var buffer1 = Range.init(0, 0); |
| | 323 | var buffer2 = Range.init(0, 0); |
| | 324 | |
| | 325 | // find two internal buffers of size 'buffer_size' each |
| | 326 | find = buffer_size + buffer_size; |
| | 327 | var find_separately = false; |
| | 328 | |
| | 329 | if (block_size <= cache.len) { |
| | 330 | // if every A block fits into the cache then we won't need the second internal buffer, |
| | 331 | // so we really only need to find 'buffer_size' unique values |
| | 332 | find = buffer_size; |
| | 333 | } else if (find > iterator.length()) { |
| | 334 | // we can't fit both buffers into the same A or B subarray, so find two buffers separately |
| | 335 | find = buffer_size; |
| | 336 | find_separately = true; |
| | 337 | } |
| | 338 | |
| | 339 | // we need to find either a single contiguous space containing 2√A unique values (which will be split up into two buffers of size √A each), |
| | 340 | // or we need to find one buffer of < 2√A unique values, and a second buffer of √A unique values, |
| | 341 | // OR if we couldn't find that many unique values, we need the largest possible buffer we can get |
| | 342 | |
| | 343 | // in the case where it couldn't find a single buffer of at least √A unique values, |
| | 344 | // all of the Merge steps must be replaced by a different merge algorithm (MergeInPlace) |
| | 345 | iterator.begin(); |
| | 346 | while (!iterator.finished()) { |
| | 347 | A = iterator.nextRange(); |
| | 348 | B = iterator.nextRange(); |
| | 349 | |
| | 350 | // just store information about where the values will be pulled from and to, |
| | 351 | // as well as how many values there are, to create the two internal buffers |
| | 352 | |
| | 353 | // check A for the number of unique values we need to fill an internal buffer |
| | 354 | // these values will be pulled out to the start of A |
| | 355 | last = A.start; |
| | 356 | count = 1; |
| | 357 | while (count < find) : ({last = index; count += 1}) { |
| | 358 | index = findLastForward(T, items, items[last], Range.init(last + 1, A.end), lessThan, find - count); |
| | 359 | if (index == A.end) break; |
| | 360 | } |
| | 361 | index = last; |
| | 362 | |
| | 363 | if (count >= buffer_size) { |
| | 364 | // keep track of the range within the items where we'll need to "pull out" these values to create the internal buffer |
| | 365 | pull[pull_index] = Pull { |
| | 366 | .range = Range.init(A.start, B.end), |
| | 367 | .count = count, |
| | 368 | .from = index, |
| | 369 | .to = A.start, |
| | 370 | }; |
| | 371 | pull_index = 1; |
| | 372 | |
| | 373 | if (count == buffer_size + buffer_size) { |
| | 374 | // we were able to find a single contiguous section containing 2√A unique values, |
| | 375 | // so this section can be used to contain both of the internal buffers we'll need |
| | 376 | buffer1 = Range.init(A.start, A.start + buffer_size); |
| | 377 | buffer2 = Range.init(A.start + buffer_size, A.start + count); |
| | 378 | break; |
| | 379 | } else if (find == buffer_size + buffer_size) { |
| | 380 | // we found a buffer that contains at least √A unique values, but did not contain the full 2√A unique values, |
| | 381 | // so we still need to find a second separate buffer of at least √A unique values |
| | 382 | buffer1 = Range.init(A.start, A.start + count); |
| | 383 | find = buffer_size; |
| | 384 | } else if (block_size <= cache.len) { |
| | 385 | // we found the first and only internal buffer that we need, so we're done! |
| | 386 | buffer1 = Range.init(A.start, A.start + count); |
| | 387 | break; |
| | 388 | } else if (find_separately) { |
| | 389 | // found one buffer, but now find the other one |
| | 390 | buffer1 = Range.init(A.start, A.start + count); |
| | 391 | find_separately = false; |
| | 392 | } else { |
| | 393 | // we found a second buffer in an 'A' subarray containing √A unique values, so we're done! |
| | 394 | buffer2 = Range.init(A.start, A.start + count); |
| | 395 | break; |
| | 396 | } |
| | 397 | } else if (pull_index == 0 and count > buffer1.length()) { |
| | 398 | // keep track of the largest buffer we were able to find |
| | 399 | buffer1 = Range.init(A.start, A.start + count); |
| | 400 | pull[pull_index] = Pull { |
| | 401 | .range = Range.init(A.start, B.end), |
| | 402 | .count = count, |
| | 403 | .from = index, |
| | 404 | .to = A.start, |
| | 405 | }; |
| | 406 | } |
| | 407 | |
| | 408 | // check B for the number of unique values we need to fill an internal buffer |
| | 409 | // these values will be pulled out to the end of B |
| | 410 | last = B.end - 1; |
| | 411 | count = 1; |
| | 412 | while (count < find) : ({last = index - 1; count += 1}) { |
| | 413 | index = findFirstBackward(T, items, items[last], Range.init(B.start, last), lessThan, find - count); |
| | 414 | if (index == B.start) break; |
| | 415 | } |
| | 416 | index = last; |
| | 417 | |
| | 418 | if (count >= buffer_size) { |
| | 419 | // keep track of the range within the items where we'll need to "pull out" these values to create the internal buffe |
| | 420 | pull[pull_index] = Pull { |
| | 421 | .range = Range.init(A.start, B.end), |
| | 422 | .count = count, |
| | 423 | .from = index, |
| | 424 | .to = B.end, |
| | 425 | }; |
| | 426 | pull_index = 1; |
| | 427 | |
| | 428 | if (count == buffer_size + buffer_size) { |
| | 429 | // we were able to find a single contiguous section containing 2√A unique values, |
| | 430 | // so this section can be used to contain both of the internal buffers we'll need |
| | 431 | buffer1 = Range.init(B.end - count, B.end - buffer_size); |
| | 432 | buffer2 = Range.init(B.end - buffer_size, B.end); |
| | 433 | break; |
| | 434 | } else if (find == buffer_size + buffer_size) { |
| | 435 | // we found a buffer that contains at least √A unique values, but did not contain the full 2√A unique values, |
| | 436 | // so we still need to find a second separate buffer of at least √A unique values |
| | 437 | buffer1 = Range.init(B.end - count, B.end); |
| | 438 | find = buffer_size; |
| | 439 | } else if (block_size <= cache.len) { |
| | 440 | // we found the first and only internal buffer that we need, so we're done! |
| | 441 | buffer1 = Range.init(B.end - count, B.end); |
| | 442 | break; |
| | 443 | } else if (find_separately) { |
| | 444 | // found one buffer, but now find the other one |
| | 445 | buffer1 = Range.init(B.end - count, B.end); |
| | 446 | find_separately = false; |
| | 447 | } else { |
| | 448 | // buffer2 will be pulled out from a 'B' subarray, so if the first buffer was pulled out from the corresponding 'A' subarray, |
| | 449 | // we need to adjust the end point for that A subarray so it knows to stop redistributing its values before reaching buffer2 |
| | 450 | if (pull[0].range.start == A.start) pull[0].range.end -= pull[1].count; |
| | 451 | |
| | 452 | // we found a second buffer in an 'B' subarray containing √A unique values, so we're done! |
| | 453 | buffer2 = Range.init(B.end - count, B.end); |
| | 454 | break; |
| | 455 | } |
| | 456 | } else if (pull_index == 0 and count > buffer1.length()) { |
| | 457 | // keep track of the largest buffer we were able to find |
| | 458 | buffer1 = Range.init(B.end - count, B.end); |
| | 459 | pull[pull_index] = Pull { |
| | 460 | .range = Range.init(A.start, B.end), |
| | 461 | .count = count, |
| | 462 | .from = index, |
| | 463 | .to = B.end, |
| | 464 | }; |
| | 465 | } |
| | 466 | } |
| | 467 | |
| | 468 | // pull out the two ranges so we can use them as internal buffers |
| | 469 | pull_index = 0; |
| | 470 | while (pull_index < 2) : (pull_index += 1) { |
| | 471 | const length = pull[pull_index].count; |
| | 472 | |
| | 473 | if (pull[pull_index].to < pull[pull_index].from) { |
| | 474 | // we're pulling the values out to the left, which means the start of an A subarray |
| | 475 | index = pull[pull_index].from; |
| | 476 | count = 1; |
| | 477 | while (count < length) : (count += 1) { |
| | 478 | index = findFirstBackward(T, items, items[index - 1], Range.init(pull[pull_index].to, pull[pull_index].from - (count - 1)), lessThan, length - count); |
| | 479 | const range = Range.init(index + 1, pull[pull_index].from + 1); |
| | 480 | mem.rotate(T, items[range.start..range.end], range.length() - count); |
| | 481 | pull[pull_index].from = index + count; |
| | 482 | } |
| | 483 | } else if (pull[pull_index].to > pull[pull_index].from) { |
| | 484 | // we're pulling values out to the right, which means the end of a B subarray |
| | 485 | index = pull[pull_index].from + 1; |
| | 486 | count = 1; |
| | 487 | while (count < length) : (count += 1) { |
| | 488 | index = findLastForward(T, items, items[index], Range.init(index, pull[pull_index].to), lessThan, length - count); |
| | 489 | const range = Range.init(pull[pull_index].from, index - 1); |
| | 490 | mem.rotate(T, items[range.start..range.end], count); |
| | 491 | pull[pull_index].from = index - 1 - count; |
| | 492 | } |
| | 493 | } |
| | 494 | } |
| | 495 | |
| | 496 | // adjust block_size and buffer_size based on the values we were able to pull out |
| | 497 | buffer_size = buffer1.length(); |
| | 498 | block_size = iterator.length()/buffer_size + 1; |
| | 499 | |
| | 500 | // the first buffer NEEDS to be large enough to tag each of the evenly sized A blocks, |
| | 501 | // so this was originally here to test the math for adjusting block_size above |
| | 502 | // assert((iterator.length() + 1)/block_size <= buffer_size); |
| | 503 | |
| | 504 | // now that the two internal buffers have been created, it's time to merge each A+B combination at this level of the merge sort! |
| | 505 | iterator.begin(); |
| | 506 | while (!iterator.finished()) { |
| | 507 | A = iterator.nextRange(); |
| | 508 | B = iterator.nextRange(); |
| | 509 | |
| | 510 | // remove any parts of A or B that are being used by the internal buffers |
| | 511 | start = A.start; |
| | 512 | if (start == pull[0].range.start) { |
| | 513 | if (pull[0].from > pull[0].to) { |
| | 514 | A.start += pull[0].count; |
| | 515 | |
| | 516 | // if the internal buffer takes up the entire A or B subarray, then there's nothing to merge |
| | 517 | // this only happens for very small subarrays, like √4 = 2, 2 * (2 internal buffers) = 4, |
| | 518 | // which also only happens when cache.len is small or 0 since it'd otherwise use MergeExternal |
| | 519 | if (A.length() == 0) continue; |
| | 520 | } else if (pull[0].from < pull[0].to) { |
| | 521 | B.end -= pull[0].count; |
| | 522 | if (B.length() == 0) continue; |
| | 523 | } |
| | 524 | } |
| | 525 | if (start == pull[1].range.start) { |
| | 526 | if (pull[1].from > pull[1].to) { |
| | 527 | A.start += pull[1].count; |
| | 528 | if (A.length() == 0) continue; |
| | 529 | } else if (pull[1].from < pull[1].to) { |
| | 530 | B.end -= pull[1].count; |
| | 531 | if (B.length() == 0) continue; |
| | 532 | } |
| | 533 | } |
| | 534 | |
| | 535 | if (lessThan(items[B.end - 1], items[A.start])) { |
| | 536 | // the two ranges are in reverse order, so a simple rotation should fix it |
| | 537 | mem.rotate(T, items[A.start..B.end], A.length()); |
| | 538 | } else if (lessThan(items[A.end], items[A.end - 1])) { |
| | 539 | // these two ranges weren't already in order, so we'll need to merge them! |
| | 540 | var findA: usize = undefined; |
| | 541 | |
| | 542 | // break the remainder of A into blocks. firstA is the uneven-sized first A block |
| | 543 | var blockA = Range.init(A.start, A.end); |
| | 544 | var firstA = Range.init(A.start, A.start + blockA.length() % block_size); |
| | 545 | |
| | 546 | // swap the first value of each A block with the value in buffer1 |
| | 547 | var indexA = buffer1.start; |
| | 548 | index = firstA.end; |
| | 549 | while (index < blockA.end) : ({indexA += 1; index += block_size}) { |
| | 550 | mem.swap(T, &items[indexA], &items[index]); |
| | 551 | } |
| | 552 | |
| | 553 | // start rolling the A blocks through the B blocks! |
| | 554 | // whenever we leave an A block behind, we'll need to merge the previous A block with any B blocks that follow it, so track that information as well |
| | 555 | var lastA = firstA; |
| | 556 | var lastB = Range.init(0, 0); |
| | 557 | var blockB = Range.init(B.start, B.start + math.min(block_size, B.length())); |
| | 558 | blockA.start += firstA.length(); |
| | 559 | indexA = buffer1.start; |
| | 560 | |
| | 561 | // if the first unevenly sized A block fits into the cache, copy it there for when we go to Merge it |
| | 562 | // otherwise, if the second buffer is available, block swap the contents into that |
| | 563 | if (lastA.length() <= cache.len) { |
| | 564 | mem.copy(T, cache[0..], items[lastA.start..lastA.end]); |
| | 565 | } else if (buffer2.length() > 0) { |
| | 566 | blockSwap(T, items, lastA.start, buffer2.start, lastA.length()); |
| | 567 | } |
| | 568 | |
| | 569 | if (blockA.length() > 0) { |
| | 570 | while (true) { |
| | 571 | // if there's a previous B block and the first value of the minimum A block is <= the last value of the previous B block, |
| | 572 | // then drop that minimum A block behind. or if there are no B blocks left then keep dropping the remaining A blocks. |
| | 573 | if ((lastB.length() > 0 and !lessThan(items[lastB.end - 1], items[indexA])) or blockB.length() == 0) { |
| | 574 | // figure out where to split the previous B block, and rotate it at the split |
| | 575 | const B_split = binaryFirst(T, items, items[indexA], lastB, lessThan); |
| | 576 | const B_remaining = lastB.end - B_split; |
| | 577 | |
| | 578 | // swap the minimum A block to the beginning of the rolling A blocks |
| | 579 | var minA = blockA.start; |
| | 580 | findA = minA + block_size; |
| | 581 | while (findA < blockA.end) : (findA += block_size) { |
| | 582 | if (lessThan(items[findA], items[minA])) { |
| | 583 | minA = findA; |
| | 584 | } |
| | 585 | } |
| | 586 | blockSwap(T, items, blockA.start, minA, block_size); |
| | 587 | |
| | 588 | // swap the first item of the previous A block back with its original value, which is stored in buffer1 |
| | 589 | mem.swap(T, &items[blockA.start], &items[indexA]); |
| | 590 | indexA += 1; |
| | 591 | |
| | 592 | // locally merge the previous A block with the B values that follow it |
| | 593 | // if lastA fits into the external cache we'll use that (with MergeExternal), |
| | 594 | // or if the second internal buffer exists we'll use that (with MergeInternal), |
| | 595 | // or failing that we'll use a strictly in-place merge algorithm (MergeInPlace) |
| | 596 | |
| | 597 | if (lastA.length() <= cache.len) { |
| | 598 | mergeExternal(T, items, lastA, Range.init(lastA.end, B_split), lessThan, cache[0..]); |
| | 599 | } else if (buffer2.length() > 0) { |
| | 600 | mergeInternal(T, items, lastA, Range.init(lastA.end, B_split), lessThan, buffer2); |
| | 601 | } else { |
| | 602 | mergeInPlace(T, items, lastA, Range.init(lastA.end, B_split), lessThan); |
| | 603 | } |
| | 604 | |
| | 605 | if (buffer2.length() > 0 or block_size <= cache.len) { |
| | 606 | // copy the previous A block into the cache or buffer2, since that's where we need it to be when we go to merge it anyway |
| | 607 | if (block_size <= cache.len) { |
| | 608 | mem.copy(T, cache[0..], items[blockA.start..blockA.start + block_size]); |
| | 609 | } else { |
| | 610 | blockSwap(T, items, blockA.start, buffer2.start, block_size); |
| | 611 | } |
| | 612 | |
| | 613 | // this is equivalent to rotating, but faster |
| | 614 | // the area normally taken up by the A block is either the contents of buffer2, or data we don't need anymore since we memcopied it |
| | 615 | // either way, we don't need to retain the order of those items, so instead of rotating we can just block swap B to where it belongs |
| | 616 | blockSwap(T, items, B_split, blockA.start + block_size - B_remaining, B_remaining); |
| | 617 | } else { |
| | 618 | // we are unable to use the 'buffer2' trick to speed up the rotation operation since buffer2 doesn't exist, so perform a normal rotation |
| | 619 | mem.rotate(T, items[B_split..blockA.start + block_size], blockA.start - B_split); |
| | 620 | } |
| | 621 | |
| | 622 | // update the range for the remaining A blocks, and the range remaining from the B block after it was split |
| | 623 | lastA = Range.init(blockA.start - B_remaining, blockA.start - B_remaining + block_size); |
| | 624 | lastB = Range.init(lastA.end, lastA.end + B_remaining); |
| | 625 | |
| | 626 | // if there are no more A blocks remaining, this step is finished! |
| | 627 | blockA.start += block_size; |
| | 628 | if (blockA.length() == 0) |
| | 629 | break; |
| | 630 | |
| | 631 | } else if (blockB.length() < block_size) { |
| | 632 | // move the last B block, which is unevenly sized, to before the remaining A blocks, by using a rotation |
| | 633 | // the cache is disabled here since it might contain the contents of the previous A block |
| | 634 | mem.rotate(T, items[blockA.start..blockB.end], blockB.start - blockA.start); |
| | 635 | |
| | 636 | lastB = Range.init(blockA.start, blockA.start + blockB.length()); |
| | 637 | blockA.start += blockB.length(); |
| | 638 | blockA.end += blockB.length(); |
| | 639 | blockB.end = blockB.start; |
| | 640 | } else { |
| | 641 | // roll the leftmost A block to the end by swapping it with the next B block |
| | 642 | blockSwap(T, items, blockA.start, blockB.start, block_size); |
| | 643 | lastB = Range.init(blockA.start, blockA.start + block_size); |
| | 644 | |
| | 645 | blockA.start += block_size; |
| | 646 | blockA.end += block_size; |
| | 647 | blockB.start += block_size; |
| | 648 | |
| | 649 | if (blockB.end > B.end - block_size) { |
| | 650 | blockB.end = B.end; |
| | 651 | } else { |
| | 652 | blockB.end += block_size; |
| | 653 | } |
| | 654 | } |
| | 655 | } |
| | 656 | } |
| | 657 | |
| | 658 | // merge the last A block with the remaining B values |
| | 659 | if (lastA.length() <= cache.len) { |
| | 660 | mergeExternal(T, items, lastA, Range.init(lastA.end, B.end), lessThan, cache[0..]); |
| | 661 | } else if (buffer2.length() > 0) { |
| | 662 | mergeInternal(T, items, lastA, Range.init(lastA.end, B.end), lessThan, buffer2); |
| | 663 | } else { |
| | 664 | mergeInPlace(T, items, lastA, Range.init(lastA.end, B.end), lessThan); |
| | 665 | } |
| | 666 | } |
| | 667 | } |
| | 668 | |
| | 669 | // when we're finished with this merge step we should have the one or two internal buffers left over, where the second buffer is all jumbled up |
| | 670 | // insertion sort the second buffer, then redistribute the buffers back into the items using the opposite process used for creating the buffer |
| | 671 | |
| | 672 | // while an unstable sort like quicksort could be applied here, in benchmarks it was consistently slightly slower than a simple insertion sort, |
| | 673 | // even for tens of millions of items. this may be because insertion sort is quite fast when the data is already somewhat sorted, like it is here |
| | 674 | insertionSort(T, items[buffer2.start..buffer2.end], lessThan); |
| | 675 | |
| | 676 | pull_index = 0; |
| | 677 | while (pull_index < 2) : (pull_index += 1) { |
| | 678 | var unique = pull[pull_index].count * 2; |
| | 679 | if (pull[pull_index].from > pull[pull_index].to) { |
| | 680 | // the values were pulled out to the left, so redistribute them back to the right |
| | 681 | var buffer = Range.init(pull[pull_index].range.start, pull[pull_index].range.start + pull[pull_index].count); |
| | 682 | while (buffer.length() > 0) { |
| | 683 | index = findFirstForward(T, items, items[buffer.start], Range.init(buffer.end, pull[pull_index].range.end), lessThan, unique); |
| | 684 | const amount = index - buffer.end; |
| | 685 | mem.rotate(T, items[buffer.start..index], buffer.length()); |
| | 686 | buffer.start += (amount + 1); |
| | 687 | buffer.end += amount; |
| | 688 | unique -= 2; |
| | 689 | } |
| | 690 | } else if (pull[pull_index].from < pull[pull_index].to) { |
| | 691 | // the values were pulled out to the right, so redistribute them back to the left |
| | 692 | var buffer = Range.init(pull[pull_index].range.end - pull[pull_index].count, pull[pull_index].range.end); |
| | 693 | while (buffer.length() > 0) { |
| | 694 | index = findLastBackward(T, items, items[buffer.end - 1], Range.init(pull[pull_index].range.start, buffer.start), lessThan, unique); |
| | 695 | const amount = buffer.start - index; |
| | 696 | mem.rotate(T, items[index..buffer.end], amount); |
| | 697 | buffer.start -= amount; |
| | 698 | buffer.end -= (amount + 1); |
| | 699 | unique -= 2; |
| | 700 | } |
| | 701 | } |
| | 702 | } |
| | 703 | } |
| | 704 | |
| | 705 | // double the size of each A and B subarray that will be merged in the next level |
| | 706 | if (!iterator.nextLevel()) break; |
| 23 | } | 707 | } |
| 24 | } | 708 | } |
| 25 | | 709 | |
| 26 | fn quicksort(comptime T: type, array: []T, left: usize, right: usize, comptime cmp: fn(a: &const T, b: &const T)->Cmp) { | 710 | // merge operation without a buffer |
| 27 | var i = left; | 711 | fn mergeInPlace(comptime T: type, items: []T, A_arg: &const Range, B_arg: &const Range, lessThan: fn(&const T,&const T)->bool) { |
| 28 | var j = right; | 712 | if (A_arg.length() == 0 or B_arg.length() == 0) return; |
| 29 | const p = (i + j) / 2; | 713 | |
| | 714 | // this just repeatedly binary searches into B and rotates A into position. |
| | 715 | // the paper suggests using the 'rotation-based Hwang and Lin algorithm' here, |
| | 716 | // but I decided to stick with this because it had better situational performance |
| | 717 | // |
| | 718 | // (Hwang and Lin is designed for merging subarrays of very different sizes, |
| | 719 | // but WikiSort almost always uses subarrays that are roughly the same size) |
| | 720 | // |
| | 721 | // normally this is incredibly suboptimal, but this function is only called |
| | 722 | // when none of the A or B blocks in any subarray contained 2√A unique values, |
| | 723 | // which places a hard limit on the number of times this will ACTUALLY need |
| | 724 | // to binary search and rotate. |
| | 725 | // |
| | 726 | // according to my analysis the worst case is √A rotations performed on √A items |
| | 727 | // once the constant factors are removed, which ends up being O(n) |
| | 728 | // |
| | 729 | // again, this is NOT a general-purpose solution – it only works well in this case! |
| | 730 | // kind of like how the O(n^2) insertion sort is used in some places |
| 30 | | 731 | |
| 31 | while (i <= j) { | 732 | var A = *A_arg; |
| 32 | while (cmp(array[i], array[p]) == Cmp.Less) { | 733 | var B = *B_arg; |
| 33 | i += 1; | 734 | |
| | 735 | while (true) { |
| | 736 | // find the first place in B where the first item in A needs to be inserted |
| | 737 | const mid = binaryFirst(T, items, items[A.start], B, lessThan); |
| | 738 | |
| | 739 | // rotate A into place |
| | 740 | const amount = mid - A.end; |
| | 741 | mem.rotate(T, items[A.start..mid], A.length()); |
| | 742 | if (B.end == mid) break; |
| | 743 | |
| | 744 | // calculate the new A and B ranges |
| | 745 | B.start = mid; |
| | 746 | A = Range.init(A.start + amount, B.start); |
| | 747 | A.start = binaryLast(T, items, items[A.start], A, lessThan); |
| | 748 | if (A.length() == 0) break; |
| | 749 | } |
| | 750 | } |
| | 751 | |
| | 752 | // merge operation using an internal buffer |
| | 753 | fn mergeInternal(comptime T: type, items: []T, A: &const Range, B: &const Range, lessThan: fn(&const T,&const T)->bool, buffer: &const Range) { |
| | 754 | // whenever we find a value to add to the final array, swap it with the value that's already in that spot |
| | 755 | // when this algorithm is finished, 'buffer' will contain its original contents, but in a different order |
| | 756 | var A_count: usize = 0; |
| | 757 | var B_count: usize = 0; |
| | 758 | var insert: usize = 0; |
| | 759 | |
| | 760 | if (B.length() > 0 and A.length() > 0) { |
| | 761 | while (true) { |
| | 762 | if (!lessThan(items[B.start + B_count], items[buffer.start + A_count])) { |
| | 763 | mem.swap(T, &items[A.start + insert], &items[buffer.start + A_count]); |
| | 764 | A_count += 1; |
| | 765 | insert += 1; |
| | 766 | if (A_count >= A.length()) break; |
| | 767 | } else { |
| | 768 | mem.swap(T, &items[A.start + insert], &items[B.start + B_count]); |
| | 769 | B_count += 1; |
| | 770 | insert += 1; |
| | 771 | if (B_count >= B.length()) break; |
| | 772 | } |
| 34 | } | 773 | } |
| 35 | while (cmp(array[j], array[p]) == Cmp.Greater) { | 774 | } |
| 36 | j -= 1; | 775 | |
| | 776 | // swap the remainder of A into the final array |
| | 777 | blockSwap(T, items, buffer.start + A_count, A.start + insert, A.length() - A_count); |
| | 778 | } |
| | 779 | |
| | 780 | fn blockSwap(comptime T: type, items: []T, start1: usize, start2: usize, block_size: usize) { |
| | 781 | var index: usize = 0; |
| | 782 | while (index < block_size) : (index += 1) { |
| | 783 | mem.swap(T, &items[start1 + index], &items[start2 + index]); |
| | 784 | } |
| | 785 | } |
| | 786 | |
| | 787 | // combine a linear search with a binary search to reduce the number of comparisons in situations |
| | 788 | // where have some idea as to how many unique values there are and where the next value might be |
| | 789 | fn findFirstForward(comptime T: type, items: []T, value: &const T, range: &const Range, lessThan: fn(&const T,&const T)->bool, unique: usize) -> usize { |
| | 790 | if (range.length() == 0) return range.start; |
| | 791 | const skip = math.max(range.length()/unique, usize(1)); |
| | 792 | |
| | 793 | var index = range.start + skip; |
| | 794 | while (lessThan(items[index - 1], value)) : (index += skip) { |
| | 795 | if (index >= range.end - skip) { |
| | 796 | return binaryFirst(T, items, value, Range.init(index, range.end), lessThan); |
| 37 | } | 797 | } |
| 38 | if (i <= j) { | 798 | } |
| 39 | const tmp = array[i]; | 799 | |
| 40 | array[i] = array[j]; | 800 | return binaryFirst(T, items, value, Range.init(index - skip, index), lessThan); |
| 41 | array[j] = tmp; | 801 | } |
| 42 | i += 1; | 802 | |
| 43 | if (j > 0) j -= 1; | 803 | fn findFirstBackward(comptime T: type, items: []T, value: &const T, range: &const Range, lessThan: fn(&const T,&const T)->bool, unique: usize) -> usize { |
| | 804 | if (range.length() == 0) return range.start; |
| | 805 | const skip = math.max(range.length()/unique, usize(1)); |
| | 806 | |
| | 807 | var index = range.end - skip; |
| | 808 | while (index > range.start and !lessThan(items[index - 1], value)) : (index -= skip) { |
| | 809 | if (index < range.start + skip) { |
| | 810 | return binaryFirst(T, items, value, Range.init(range.start, index), lessThan); |
| 44 | } | 811 | } |
| 45 | } | 812 | } |
| | 813 | |
| | 814 | return binaryFirst(T, items, value, Range.init(index, index + skip), lessThan); |
| | 815 | } |
| 46 | | 816 | |
| 47 | if (left < j) quicksort(T, array, left, j, cmp); | 817 | fn findLastForward(comptime T: type, items: []T, value: &const T, range: &const Range, lessThan: fn(&const T,&const T)->bool, unique: usize) -> usize { |
| 48 | if (i < right) quicksort(T, array, i, right, cmp); | 818 | if (range.length() == 0) return range.start; |
| | 819 | const skip = math.max(range.length()/unique, usize(1)); |
| | 820 | |
| | 821 | var index = range.start + skip; |
| | 822 | while (!lessThan(value, items[index - 1])) : (index += skip) { |
| | 823 | if (index >= range.end - skip) { |
| | 824 | return binaryLast(T, items, value, Range.init(index, range.end), lessThan); |
| | 825 | } |
| | 826 | } |
| | 827 | |
| | 828 | return binaryLast(T, items, value, Range.init(index - skip, index), lessThan); |
| 49 | } | 829 | } |
| 50 | | 830 | |
| 51 | pub fn i32asc(a: &const i32, b: &const i32) -> Cmp { | 831 | fn findLastBackward(comptime T: type, items: []T, value: &const T, range: &const Range, lessThan: fn(&const T,&const T)->bool, unique: usize) -> usize { |
| 52 | return if (*a > *b) Cmp.Greater else if (*a < *b) Cmp.Less else Cmp.Equal | 832 | if (range.length() == 0) return range.start; |
| | 833 | const skip = math.max(range.length()/unique, usize(1)); |
| | 834 | |
| | 835 | var index = range.end - skip; |
| | 836 | while (index > range.start and lessThan(value, items[index - 1])) : (index -= skip) { |
| | 837 | if (index < range.start + skip) { |
| | 838 | return binaryLast(T, items, value, Range.init(range.start, index), lessThan); |
| | 839 | } |
| | 840 | } |
| | 841 | |
| | 842 | return binaryLast(T, items, value, Range.init(index, index + skip), lessThan); |
| 53 | } | 843 | } |
| 54 | | 844 | |
| 55 | pub fn i32desc(a: &const i32, b: &const i32) -> Cmp { | 845 | fn binaryFirst(comptime T: type, items: []T, value: &const T, range: &const Range, lessThan: fn(&const T,&const T)->bool) -> usize { |
| 56 | reverse(i32asc(a, b)) | 846 | var start = range.start; |
| | 847 | var end = range.end - 1; |
| | 848 | if (range.start >= range.end) return range.end; |
| | 849 | while (start < end) { |
| | 850 | const mid = start + (end - start)/2; |
| | 851 | if (lessThan(items[mid], value)) { |
| | 852 | start = mid + 1; |
| | 853 | } else { |
| | 854 | end = mid; |
| | 855 | } |
| | 856 | } |
| | 857 | if (start == range.end - 1 and lessThan(items[start], value)) { |
| | 858 | start += 1; |
| | 859 | } |
| | 860 | return start; |
| 57 | } | 861 | } |
| 58 | | 862 | |
| 59 | pub fn u8asc(a: &const u8, b: &const u8) -> Cmp { | 863 | fn binaryLast(comptime T: type, items: []T, value: &const T, range: &const Range, lessThan: fn(&const T,&const T)->bool) -> usize { |
| 60 | if (*a > *b) Cmp.Greater else if (*a < *b) Cmp.Less else Cmp.Equal | 864 | var start = range.start; |
| | 865 | var end = range.end - 1; |
| | 866 | if (range.start >= range.end) return range.end; |
| | 867 | while (start < end) { |
| | 868 | const mid = start + (end - start)/2; |
| | 869 | if (!lessThan(value, items[mid])) { |
| | 870 | start = mid + 1; |
| | 871 | } else { |
| | 872 | end = mid; |
| | 873 | } |
| | 874 | } |
| | 875 | if (start == range.end - 1 and !lessThan(value, items[start])) { |
| | 876 | start += 1; |
| | 877 | } |
| | 878 | return start; |
| 61 | } | 879 | } |
| 62 | | 880 | |
| 63 | pub fn u8desc(a: &const u8, b: &const u8) -> Cmp { | 881 | fn mergeInto(comptime T: type, from: []T, A: &const Range, B: &const Range, lessThan: fn(&const T,&const T)->bool, into: []T) { |
| 64 | reverse(u8asc(a, b)) | 882 | var A_index: usize = A.start; |
| | 883 | var B_index: usize = B.start; |
| | 884 | const A_last = A.end; |
| | 885 | const B_last = B.end; |
| | 886 | var insert_index: usize = 0; |
| | 887 | |
| | 888 | while (true) { |
| | 889 | if (!lessThan(from[B_index], from[A_index])) { |
| | 890 | into[insert_index] = from[A_index]; |
| | 891 | A_index += 1; |
| | 892 | insert_index += 1; |
| | 893 | if (A_index == A_last) { |
| | 894 | // copy the remainder of B into the final array |
| | 895 | mem.copy(T, into[insert_index..], from[B_index..B_last]); |
| | 896 | break; |
| | 897 | } |
| | 898 | } else { |
| | 899 | into[insert_index] = from[B_index]; |
| | 900 | B_index += 1; |
| | 901 | insert_index += 1; |
| | 902 | if (B_index == B_last) { |
| | 903 | // copy the remainder of A into the final array |
| | 904 | mem.copy(T, into[insert_index..], from[A_index..A_last]); |
| | 905 | break; |
| | 906 | } |
| | 907 | } |
| | 908 | } |
| 65 | } | 909 | } |
| 66 | | 910 | |
| 67 | fn reverse(was: Cmp) -> Cmp { | 911 | fn mergeExternal(comptime T: type, items: []T, A: &const Range, B: &const Range, lessThan: fn(&const T,&const T)->bool, cache: []T) { |
| 68 | if (was == Cmp.Greater) Cmp.Less else if (was == Cmp.Less) Cmp.Greater else Cmp.Equal | 912 | // A fits into the cache, so use that instead of the internal buffer |
| | 913 | var A_index: usize = 0; |
| | 914 | var B_index: usize = B.start; |
| | 915 | var insert_index: usize = A.start; |
| | 916 | const A_last = A.length(); |
| | 917 | const B_last = B.end; |
| | 918 | |
| | 919 | if (B.length() > 0 and A.length() > 0) { |
| | 920 | while (true) { |
| | 921 | if (!lessThan(items[B_index], cache[A_index])) { |
| | 922 | items[insert_index] = cache[A_index]; |
| | 923 | A_index += 1; |
| | 924 | insert_index += 1; |
| | 925 | if (A_index == A_last) break; |
| | 926 | } else { |
| | 927 | items[insert_index] = items[B_index]; |
| | 928 | B_index += 1; |
| | 929 | insert_index += 1; |
| | 930 | if (B_index == B_last) break; |
| | 931 | } |
| | 932 | } |
| | 933 | } |
| | 934 | |
| | 935 | // copy the remainder of A into the final array |
| | 936 | mem.copy(T, items[insert_index..], cache[A_index..A_last]); |
| 69 | } | 937 | } |
| 70 | | 938 | |
| 71 | // --------------------------------------- | 939 | fn swap(comptime T: type, items: []T, lessThan: fn(lhs: &const T, rhs: &const T)->bool, order: &[8]u8, x: usize, y: usize) { |
| 72 | // tests | 940 | if (lessThan(items[y], items[x]) or |
| | 941 | ((*order)[x] > (*order)[y] and !lessThan(items[x], items[y]))) |
| | 942 | { |
| | 943 | mem.swap(T, &items[x], &items[y]); |
| | 944 | mem.swap(u8, &(*order)[x], &(*order)[y]); |
| | 945 | } |
| | 946 | } |
| | 947 | |
| | 948 | fn i32asc(lhs: &const i32, rhs: &const i32) -> bool { |
| | 949 | return *lhs < *rhs; |
| | 950 | } |
| | 951 | |
| | 952 | fn i32desc(lhs: &const i32, rhs: &const i32) -> bool { |
| | 953 | return *rhs < *lhs; |
| | 954 | } |
| | 955 | |
| | 956 | fn u8asc(lhs: &const u8, rhs: &const u8) -> bool { |
| | 957 | return *lhs < *rhs; |
| | 958 | } |
| | 959 | |
| | 960 | fn u8desc(lhs: &const u8, rhs: &const u8) -> bool { |
| | 961 | return *rhs < *lhs; |
| | 962 | } |
| 73 | | 963 | |
| 74 | test "stable sort" { | 964 | test "stable sort" { |
| 75 | testStableSort(); | 965 | testStableSort(); |
| ... | @@ -113,7 +1003,7 @@ fn testStableSort() { | ... | @@ -113,7 +1003,7 @@ fn testStableSort() { |
| 113 | }, | 1003 | }, |
| 114 | }; | 1004 | }; |
| 115 | for (cases) |*case| { | 1005 | for (cases) |*case| { |
| 116 | sort_stable(IdAndValue, (*case)[0..], cmpByValue); | 1006 | insertionSort(IdAndValue, (*case)[0..], cmpByValue); |
| 117 | for (*case) |item, i| { | 1007 | for (*case) |item, i| { |
| 118 | assert(item.id == expected[i].id); | 1008 | assert(item.id == expected[i].id); |
| 119 | assert(item.value == expected[i].value); | 1009 | assert(item.value == expected[i].value); |
| ... | @@ -121,14 +1011,14 @@ fn testStableSort() { | ... | @@ -121,14 +1011,14 @@ fn testStableSort() { |
| 121 | } | 1011 | } |
| 122 | } | 1012 | } |
| 123 | const IdAndValue = struct { | 1013 | const IdAndValue = struct { |
| 124 | id: i32, | 1014 | id: usize, |
| 125 | value: i32, | 1015 | value: i32, |
| 126 | }; | 1016 | }; |
| 127 | fn cmpByValue(a: &const IdAndValue, b: &const IdAndValue) -> Cmp { | 1017 | fn cmpByValue(a: &const IdAndValue, b: &const IdAndValue) -> bool { |
| 128 | return i32asc(a.value, b.value); | 1018 | return i32asc(a.value, b.value); |
| 129 | } | 1019 | } |
| 130 | | 1020 | |
| 131 | test "testSort" { | 1021 | test "std.sort" { |
| 132 | const u8cases = [][]const []const u8 { | 1022 | const u8cases = [][]const []const u8 { |
| 133 | [][]const u8{"", ""}, | 1023 | [][]const u8{"", ""}, |
| 134 | [][]const u8{"a", "a"}, | 1024 | [][]const u8{"a", "a"}, |
| ... | @@ -164,7 +1054,7 @@ test "testSort" { | ... | @@ -164,7 +1054,7 @@ test "testSort" { |
| 164 | } | 1054 | } |
| 165 | } | 1055 | } |
| 166 | | 1056 | |
| 167 | test "testSortDesc" { | 1057 | test "std.sort descending" { |
| 168 | const rev_cases = [][]const []const i32 { | 1058 | const rev_cases = [][]const []const i32 { |
| 169 | [][]const i32{[]i32{}, []i32{}}, | 1059 | [][]const i32{[]i32{}, []i32{}}, |
| 170 | [][]const i32{[]i32{1}, []i32{1}}, | 1060 | [][]const i32{[]i32{1}, []i32{1}}, |
| ... | @@ -182,3 +1072,42 @@ test "testSortDesc" { | ... | @@ -182,3 +1072,42 @@ test "testSortDesc" { |
| 182 | assert(mem.eql(i32, slice, case[1])); | 1072 | assert(mem.eql(i32, slice, case[1])); |
| 183 | } | 1073 | } |
| 184 | } | 1074 | } |
| | 1075 | |
| | 1076 | test "another sort case" { |
| | 1077 | var arr = []i32{ 5, 3, 1, 2, 4 }; |
| | 1078 | sort(i32, arr[0..], i32asc); |
| | 1079 | |
| | 1080 | assert(mem.eql(i32, arr, []i32{ 1, 2, 3, 4, 5 })) |
| | 1081 | } |
| | 1082 | |
| | 1083 | test "sort fuzz testing" { |
| | 1084 | var rng = std.rand.Rand.init(0x12345678); |
| | 1085 | const test_case_count = 10; |
| | 1086 | var i: usize = 0; |
| | 1087 | while (i < test_case_count) : (i += 1) { |
| | 1088 | fuzzTest(&rng); |
| | 1089 | } |
| | 1090 | } |
| | 1091 | |
| | 1092 | var fixed_buffer_mem: [100 * 1024]u8 = undefined; |
| | 1093 | |
| | 1094 | fn fuzzTest(rng: &std.rand.Rand) { |
| | 1095 | const array_size = rng.range(usize, 0, 1000); |
| | 1096 | var fixed_allocator = mem.FixedBufferAllocator.init(fixed_buffer_mem[0..]); |
| | 1097 | var array = %%fixed_allocator.allocator.alloc(IdAndValue, array_size); |
| | 1098 | // populate with random data |
| | 1099 | for (array) |*item, index| { |
| | 1100 | item.id = index; |
| | 1101 | item.value = rng.range(i32, 0, 100); |
| | 1102 | } |
| | 1103 | sort(IdAndValue, array, cmpByValue); |
| | 1104 | |
| | 1105 | var index: usize = 1; |
| | 1106 | while (index < array.len) : (index += 1) { |
| | 1107 | if (array[index].value == array[index - 1].value) { |
| | 1108 | assert(array[index].id > array[index - 1].id); |
| | 1109 | } else { |
| | 1110 | assert(array[index].value > array[index - 1].value); |
| | 1111 | } |
| | 1112 | } |
| | 1113 | } |