Coverage Report

Created: 2022-11-03 06:34

/src/bzip2/blocksort.c
Line
Count
Source (jump to first uncovered line)
1
2
/*-------------------------------------------------------------*/
3
/*--- Block sorting machinery                               ---*/
4
/*---                                           blocksort.c ---*/
5
/*-------------------------------------------------------------*/
6
7
/* ------------------------------------------------------------------
8
   This file is part of bzip2/libbzip2, a program and library for
9
   lossless, block-sorting data compression.
10
11
   bzip2/libbzip2 version 1.0.8 of 13 July 2019
12
   Copyright (C) 1996-2019 Julian Seward <jseward@acm.org>
13
14
   Please read the WARNING, DISCLAIMER and PATENTS sections in the 
15
   README file.
16
17
   This program is released under the terms of the license contained
18
   in the file LICENSE.
19
   ------------------------------------------------------------------ */
20
21
22
#include "bzlib_private.h"
23
24
/*---------------------------------------------*/
25
/*--- Fallback O(N log(N)^2) sorting        ---*/
26
/*--- algorithm, for repetitive blocks      ---*/
27
/*---------------------------------------------*/
28
29
/*---------------------------------------------*/
30
static 
31
__inline__
32
void fallbackSimpleSort ( UInt32* fmap, 
33
                          UInt32* eclass, 
34
                          Int32   lo, 
35
                          Int32   hi )
36
421M
{
37
421M
   Int32 i, j, tmp;
38
421M
   UInt32 ec_tmp;
39
40
421M
   if (lo == hi) return;
41
42
411M
   if (hi - lo > 3) {
43
253M
      for ( i = hi-4; i >= lo; i-- ) {
44
193M
         tmp = fmap[i];
45
193M
         ec_tmp = eclass[tmp];
46
281M
         for ( j = i+4; j <= hi && ec_tmp > eclass[fmap[j]]; j += 4 )
47
87.8M
            fmap[j-4] = fmap[j];
48
193M
         fmap[j-4] = tmp;
49
193M
      }
50
59.7M
   }
51
52
1.23G
   for ( i = hi-1; i >= lo; i-- ) {
53
825M
      tmp = fmap[i];
54
825M
      ec_tmp = eclass[tmp];
55
1.22G
      for ( j = i+1; j <= hi && ec_tmp > eclass[fmap[j]]; j++ )
56
394M
         fmap[j-1] = fmap[j];
57
825M
      fmap[j-1] = tmp;
58
825M
   }
59
411M
}
60
61
62
/*---------------------------------------------*/
63
#define fswap(zz1, zz2) \
64
7.97G
   { Int32 zztmp = zz1; zz1 = zz2; zz2 = zztmp; }
65
66
170M
#define fvswap(zzp1, zzp2, zzn)       \
67
170M
{                                     \
68
170M
   Int32 yyp1 = (zzp1);               \
69
170M
   Int32 yyp2 = (zzp2);               \
70
170M
   Int32 yyn  = (zzn);                \
71
1.32G
   while (yyn > 0) {                  \
72
1.15G
      fswap(fmap[yyp1], fmap[yyp2]);  \
73
1.15G
      yyp1++; yyp2++; yyn--;          \
74
1.15G
   }                                  \
75
170M
}
76
77
78
170M
#define fmin(a,b) ((a) < (b)) ? (a) : (b)
79
80
508M
#define fpush(lz,hz) { stackLo[sp] = lz; \
81
508M
                       stackHi[sp] = hz; \
82
508M
                       sp++; }
83
84
508M
#define fpop(lz,hz) { sp--;              \
85
508M
                      lz = stackLo[sp];  \
86
508M
                      hz = stackHi[sp]; }
87
88
508M
#define FALLBACK_QSORT_SMALL_THRESH 10
89
#define FALLBACK_QSORT_STACK_SIZE   100
90
91
92
static
93
void fallbackQSort3 ( UInt32* fmap, 
94
                      UInt32* eclass,
95
                      Int32   loSt, 
96
                      Int32   hiSt )
97
338M
{
98
338M
   Int32 unLo, unHi, ltLo, gtHi, n, m;
99
338M
   Int32 sp, lo, hi;
100
338M
   UInt32 med, r, r3;
101
338M
   Int32 stackLo[FALLBACK_QSORT_STACK_SIZE];
102
338M
   Int32 stackHi[FALLBACK_QSORT_STACK_SIZE];
103
104
338M
   r = 0;
105
106
338M
   sp = 0;
107
338M
   fpush ( loSt, hiSt );
108
109
847M
   while (sp > 0) {
110
111
508M
      AssertH ( sp < FALLBACK_QSORT_STACK_SIZE - 1, 1004 );
112
113
508M
      fpop ( lo, hi );
114
508M
      if (hi - lo < FALLBACK_QSORT_SMALL_THRESH) {
115
421M
         fallbackSimpleSort ( fmap, eclass, lo, hi );
116
421M
         continue;
117
421M
      }
118
119
      /* Random partitioning.  Median of 3 sometimes fails to
120
         avoid bad cases.  Median of 9 seems to help but 
121
         looks rather expensive.  This too seems to work but
122
         is cheaper.  Guidance for the magic constants 
123
         7621 and 32768 is taken from Sedgewick's algorithms
124
         book, chapter 35.
125
      */
126
87.8M
      r = ((r * 7621) + 1) % 32768;
127
87.8M
      r3 = r % 3;
128
87.8M
      if (r3 == 0) med = eclass[fmap[lo]]; else
129
72.0M
      if (r3 == 1) med = eclass[fmap[(lo+hi)>>1]]; else
130
30.0M
                   med = eclass[fmap[hi]];
131
132
87.8M
      unLo = ltLo = lo;
133
87.8M
      unHi = gtHi = hi;
134
135
891M
      while (1) {
136
5.55G
         while (1) {
137
5.55G
            if (unLo > unHi) break;
138
5.50G
            n = (Int32)eclass[fmap[unLo]] - (Int32)med;
139
5.50G
            if (n == 0) { 
140
2.17G
               fswap(fmap[unLo], fmap[ltLo]); 
141
2.17G
               ltLo++; unLo++; 
142
2.17G
               continue; 
143
3.33G
            };
144
3.33G
            if (n > 0) break;
145
2.48G
            unLo++;
146
2.48G
         }
147
7.18G
         while (1) {
148
7.18G
            if (unLo > unHi) break;
149
7.09G
            n = (Int32)eclass[fmap[unHi]] - (Int32)med;
150
7.09G
            if (n == 0) { 
151
3.83G
               fswap(fmap[unHi], fmap[gtHi]); 
152
3.83G
               gtHi--; unHi--; 
153
3.83G
               continue; 
154
3.83G
            };
155
3.25G
            if (n < 0) break;
156
2.45G
            unHi--;
157
2.45G
         }
158
891M
         if (unLo > unHi) break;
159
803M
         fswap(fmap[unLo], fmap[unHi]); unLo++; unHi--;
160
803M
      }
161
162
87.8M
      AssertD ( unHi == unLo-1, "fallbackQSort3(2)" );
163
164
87.8M
      if (gtHi < ltLo) continue;
165
166
85.0M
      n = fmin(ltLo-lo, unLo-ltLo); fvswap(lo, unLo-n, n);
167
85.0M
      m = fmin(hi-gtHi, gtHi-unHi); fvswap(unLo, hi-m+1, m);
168
169
85.0M
      n = lo + unLo - ltLo - 1;
170
85.0M
      m = hi - (gtHi - unHi) + 1;
171
172
85.0M
      if (n - lo > hi - m) {
173
38.9M
         fpush ( lo, n );
174
38.9M
         fpush ( m, hi );
175
46.0M
      } else {
176
46.0M
         fpush ( m, hi );
177
46.0M
         fpush ( lo, n );
178
46.0M
      }
179
85.0M
   }
180
338M
}
181
182
#undef fmin
183
#undef fpush
184
#undef fpop
185
#undef fswap
186
#undef fvswap
187
#undef FALLBACK_QSORT_SMALL_THRESH
188
#undef FALLBACK_QSORT_STACK_SIZE
189
190
191
/*---------------------------------------------*/
192
/* Pre:
193
      nblock > 0
194
      eclass exists for [0 .. nblock-1]
195
      ((UChar*)eclass) [0 .. nblock-1] holds block
196
      ptr exists for [0 .. nblock-1]
197
198
   Post:
199
      ((UChar*)eclass) [0 .. nblock-1] holds block
200
      All other areas of eclass destroyed
201
      fmap [0 .. nblock-1] holds sorted order
202
      bhtab [ 0 .. 2+(nblock/32) ] destroyed
203
*/
204
205
892M
#define       SET_BH(zz)  bhtab[(zz) >> 5] |= ((UInt32)1 << ((zz) & 31))
206
358k
#define     CLEAR_BH(zz)  bhtab[(zz) >> 5] &= ~((UInt32)1 << ((zz) & 31))
207
15.3G
#define     ISSET_BH(zz)  (bhtab[(zz) >> 5] & ((UInt32)1 << ((zz) & 31)))
208
295M
#define      WORD_BH(zz)  bhtab[(zz) >> 5]
209
1.73G
#define UNALIGNED_BH(zz)  ((zz) & 0x01f)
210
211
static
212
void fallbackSort ( UInt32* fmap, 
213
                    UInt32* eclass, 
214
                    UInt32* bhtab,
215
                    Int32   nblock,
216
                    Int32   verb )
217
11.2k
{
218
11.2k
   Int32 ftab[257];
219
11.2k
   Int32 ftabCopy[256];
220
11.2k
   Int32 H, i, j, k, l, r, cc, cc1;
221
11.2k
   Int32 nNotDone;
222
11.2k
   Int32 nBhtab;
223
11.2k
   UChar* eclass8 = (UChar*)eclass;
224
225
   /*--
226
      Initial 1-char radix sort to generate
227
      initial fmap and initial BH bits.
228
   --*/
229
11.2k
   if (verb >= 4)
230
0
      VPrintf0 ( "        bucket sorting ...\n" );
231
2.88M
   for (i = 0; i < 257;    i++) ftab[i] = 0;
232
551M
   for (i = 0; i < nblock; i++) ftab[eclass8[i]]++;
233
2.87M
   for (i = 0; i < 256;    i++) ftabCopy[i] = ftab[i];
234
2.87M
   for (i = 1; i < 257;    i++) ftab[i] += ftab[i-1];
235
236
551M
   for (i = 0; i < nblock; i++) {
237
551M
      j = eclass8[i];
238
551M
      k = ftab[j] - 1;
239
551M
      ftab[j] = k;
240
551M
      fmap[k] = i;
241
551M
   }
242
243
11.2k
   nBhtab = 2 + (nblock / 32);
244
17.2M
   for (i = 0; i < nBhtab; i++) bhtab[i] = 0;
245
2.87M
   for (i = 0; i < 256; i++) SET_BH(ftab[i]);
246
247
   /*--
248
      Inductively refine the buckets.  Kind-of an
249
      "exponential radix sort" (!), inspired by the
250
      Manber-Myers suffix array construction algorithm.
251
   --*/
252
253
   /*-- set sentinel bits for block-end detection --*/
254
369k
   for (i = 0; i < 32; i++) { 
255
358k
      SET_BH(nblock + 2*i);
256
358k
      CLEAR_BH(nblock + 2*i + 1);
257
358k
   }
258
259
   /*-- the log(N) loop --*/
260
11.2k
   H = 1;
261
98.4k
   while (1) {
262
263
98.4k
      if (verb >= 4) 
264
0
         VPrintf1 ( "        depth %6d has ", H );
265
266
98.4k
      j = 0;
267
9.43G
      for (i = 0; i < nblock; i++) {
268
9.43G
         if (ISSET_BH(i)) j = i;
269
9.43G
         k = fmap[i] - H; if (k < 0) k += nblock;
270
9.43G
         eclass[k] = j;
271
9.43G
      }
272
273
98.4k
      nNotDone = 0;
274
98.4k
      r = -1;
275
338M
      while (1) {
276
277
   /*-- find the next non-singleton bucket --*/
278
338M
         k = r + 1;
279
1.32G
         while (ISSET_BH(k) && UNALIGNED_BH(k)) k++;
280
338M
         if (ISSET_BH(k)) {
281
79.4M
            while (WORD_BH(k) == 0xffffffff) k += 32;
282
267M
            while (ISSET_BH(k)) k++;
283
38.1M
         }
284
338M
         l = k - 1;
285
338M
         if (l >= nblock) break;
286
1.02G
         while (!ISSET_BH(k) && UNALIGNED_BH(k)) k++;
287
338M
         if (!ISSET_BH(k)) {
288
216M
            while (WORD_BH(k) == 0x00000000) k += 32;
289
237M
            while (!ISSET_BH(k)) k++;
290
28.1M
         }
291
338M
         r = k - 1;
292
338M
         if (r >= nblock) break;
293
294
         /*-- now [l, r] bracket current bucket --*/
295
338M
         if (r > l) {
296
338M
            nNotDone += (r - l + 1);
297
338M
            fallbackQSort3 ( fmap, eclass, l, r );
298
299
            /*-- scan bucket and generate header bits-- */
300
338M
            cc = -1;
301
7.58G
            for (i = l; i <= r; i++) {
302
7.24G
               cc1 = eclass[fmap[i]];
303
7.24G
               if (cc != cc1) { SET_BH(i); cc = cc1; };
304
7.24G
            }
305
338M
         }
306
338M
      }
307
308
98.4k
      if (verb >= 4) 
309
0
         VPrintf1 ( "%6d unresolved strings\n", nNotDone );
310
311
98.4k
      H *= 2;
312
98.4k
      if (H > nblock || nNotDone == 0) break;
313
98.4k
   }
314
315
   /*-- 
316
      Reconstruct the original block in
317
      eclass8 [0 .. nblock-1], since the
318
      previous phase destroyed it.
319
   --*/
320
11.2k
   if (verb >= 4)
321
0
      VPrintf0 ( "        reconstructing block ...\n" );
322
11.2k
   j = 0;
323
551M
   for (i = 0; i < nblock; i++) {
324
554M
      while (ftabCopy[j] == 0) j++;
325
551M
      ftabCopy[j]--;
326
551M
      eclass8[fmap[i]] = (UChar)j;
327
551M
   }
328
11.2k
   AssertH ( j < 256, 1005 );
329
11.2k
}
330
331
#undef       SET_BH
332
#undef     CLEAR_BH
333
#undef     ISSET_BH
334
#undef      WORD_BH
335
#undef UNALIGNED_BH
336
337
338
/*---------------------------------------------*/
339
/*--- The main, O(N^2 log(N)) sorting       ---*/
340
/*--- algorithm.  Faster for "normal"       ---*/
341
/*--- non-repetitive blocks.                ---*/
342
/*---------------------------------------------*/
343
344
/*---------------------------------------------*/
345
static
346
__inline__
347
Bool mainGtU ( UInt32  i1, 
348
               UInt32  i2,
349
               UChar*  block, 
350
               UInt16* quadrant,
351
               UInt32  nblock,
352
               Int32*  budget )
353
727M
{
354
727M
   Int32  k;
355
727M
   UChar  c1, c2;
356
727M
   UInt16 s1, s2;
357
358
727M
   AssertD ( i1 != i2, "mainGtU" );
359
   /* 1 */
360
727M
   c1 = block[i1]; c2 = block[i2];
361
727M
   if (c1 != c2) return (c1 > c2);
362
455M
   i1++; i2++;
363
   /* 2 */
364
455M
   c1 = block[i1]; c2 = block[i2];
365
455M
   if (c1 != c2) return (c1 > c2);
366
431M
   i1++; i2++;
367
   /* 3 */
368
431M
   c1 = block[i1]; c2 = block[i2];
369
431M
   if (c1 != c2) return (c1 > c2);
370
419M
   i1++; i2++;
371
   /* 4 */
372
419M
   c1 = block[i1]; c2 = block[i2];
373
419M
   if (c1 != c2) return (c1 > c2);
374
407M
   i1++; i2++;
375
   /* 5 */
376
407M
   c1 = block[i1]; c2 = block[i2];
377
407M
   if (c1 != c2) return (c1 > c2);
378
395M
   i1++; i2++;
379
   /* 6 */
380
395M
   c1 = block[i1]; c2 = block[i2];
381
395M
   if (c1 != c2) return (c1 > c2);
382
385M
   i1++; i2++;
383
   /* 7 */
384
385M
   c1 = block[i1]; c2 = block[i2];
385
385M
   if (c1 != c2) return (c1 > c2);
386
377M
   i1++; i2++;
387
   /* 8 */
388
377M
   c1 = block[i1]; c2 = block[i2];
389
377M
   if (c1 != c2) return (c1 > c2);
390
371M
   i1++; i2++;
391
   /* 9 */
392
371M
   c1 = block[i1]; c2 = block[i2];
393
371M
   if (c1 != c2) return (c1 > c2);
394
367M
   i1++; i2++;
395
   /* 10 */
396
367M
   c1 = block[i1]; c2 = block[i2];
397
367M
   if (c1 != c2) return (c1 > c2);
398
364M
   i1++; i2++;
399
   /* 11 */
400
364M
   c1 = block[i1]; c2 = block[i2];
401
364M
   if (c1 != c2) return (c1 > c2);
402
361M
   i1++; i2++;
403
   /* 12 */
404
361M
   c1 = block[i1]; c2 = block[i2];
405
361M
   if (c1 != c2) return (c1 > c2);
406
358M
   i1++; i2++;
407
408
358M
   k = nblock + 8;
409
410
8.90G
   do {
411
      /* 1 */
412
8.90G
      c1 = block[i1]; c2 = block[i2];
413
8.90G
      if (c1 != c2) return (c1 > c2);
414
8.89G
      s1 = quadrant[i1]; s2 = quadrant[i2];
415
8.89G
      if (s1 != s2) return (s1 > s2);
416
8.83G
      i1++; i2++;
417
      /* 2 */
418
8.83G
      c1 = block[i1]; c2 = block[i2];
419
8.83G
      if (c1 != c2) return (c1 > c2);
420
8.81G
      s1 = quadrant[i1]; s2 = quadrant[i2];
421
8.81G
      if (s1 != s2) return (s1 > s2);
422
8.75G
      i1++; i2++;
423
      /* 3 */
424
8.75G
      c1 = block[i1]; c2 = block[i2];
425
8.75G
      if (c1 != c2) return (c1 > c2);
426
8.74G
      s1 = quadrant[i1]; s2 = quadrant[i2];
427
8.74G
      if (s1 != s2) return (s1 > s2);
428
8.70G
      i1++; i2++;
429
      /* 4 */
430
8.70G
      c1 = block[i1]; c2 = block[i2];
431
8.70G
      if (c1 != c2) return (c1 > c2);
432
8.69G
      s1 = quadrant[i1]; s2 = quadrant[i2];
433
8.69G
      if (s1 != s2) return (s1 > s2);
434
8.66G
      i1++; i2++;
435
      /* 5 */
436
8.66G
      c1 = block[i1]; c2 = block[i2];
437
8.66G
      if (c1 != c2) return (c1 > c2);
438
8.65G
      s1 = quadrant[i1]; s2 = quadrant[i2];
439
8.65G
      if (s1 != s2) return (s1 > s2);
440
8.63G
      i1++; i2++;
441
      /* 6 */
442
8.63G
      c1 = block[i1]; c2 = block[i2];
443
8.63G
      if (c1 != c2) return (c1 > c2);
444
8.62G
      s1 = quadrant[i1]; s2 = quadrant[i2];
445
8.62G
      if (s1 != s2) return (s1 > s2);
446
8.60G
      i1++; i2++;
447
      /* 7 */
448
8.60G
      c1 = block[i1]; c2 = block[i2];
449
8.60G
      if (c1 != c2) return (c1 > c2);
450
8.58G
      s1 = quadrant[i1]; s2 = quadrant[i2];
451
8.58G
      if (s1 != s2) return (s1 > s2);
452
8.57G
      i1++; i2++;
453
      /* 8 */
454
8.57G
      c1 = block[i1]; c2 = block[i2];
455
8.57G
      if (c1 != c2) return (c1 > c2);
456
8.56G
      s1 = quadrant[i1]; s2 = quadrant[i2];
457
8.56G
      if (s1 != s2) return (s1 > s2);
458
8.55G
      i1++; i2++;
459
460
8.55G
      if (i1 >= nblock) i1 -= nblock;
461
8.55G
      if (i2 >= nblock) i2 -= nblock;
462
463
8.55G
      k -= 8;
464
8.55G
      (*budget)--;
465
8.55G
   }
466
8.55G
      while (k >= 0);
467
468
1.96k
   return False;
469
358M
}
470
471
472
/*---------------------------------------------*/
473
/*--
474
   Knuth's increments seem to work better
475
   than Incerpi-Sedgewick here.  Possibly
476
   because the number of elems to sort is
477
   usually small, typically <= 20.
478
--*/
479
static
480
Int32 incs[14] = { 1, 4, 13, 40, 121, 364, 1093, 3280,
481
                   9841, 29524, 88573, 265720,
482
                   797161, 2391484 };
483
484
static
485
void mainSimpleSort ( UInt32* ptr,
486
                      UChar*  block,
487
                      UInt16* quadrant,
488
                      Int32   nblock,
489
                      Int32   lo, 
490
                      Int32   hi, 
491
                      Int32   d,
492
                      Int32*  budget )
493
27.5M
{
494
27.5M
   Int32 i, j, h, bigN, hp;
495
27.5M
   UInt32 v;
496
497
27.5M
   bigN = hi - lo + 1;
498
27.5M
   if (bigN < 2) return;
499
500
26.6M
   hp = 0;
501
66.4M
   while (incs[hp] < bigN) hp++;
502
26.6M
   hp--;
503
504
66.4M
   for (; hp >= 0; hp--) {
505
39.8M
      h = incs[hp];
506
507
39.8M
      i = lo + h;
508
157M
      while (True) {
509
510
         /*-- copy 1 --*/
511
157M
         if (i > hi) break;
512
147M
         v = ptr[i];
513
147M
         j = i;
514
251M
         while ( mainGtU ( 
515
251M
                    ptr[j-h]+d, v+d, block, quadrant, nblock, budget 
516
251M
                 ) ) {
517
133M
            ptr[j] = ptr[j-h];
518
133M
            j = j - h;
519
133M
            if (j <= (lo + h - 1)) break;
520
133M
         }
521
147M
         ptr[j] = v;
522
147M
         i++;
523
524
         /*-- copy 2 --*/
525
147M
         if (i > hi) break;
526
129M
         v = ptr[i];
527
129M
         j = i;
528
242M
         while ( mainGtU ( 
529
242M
                    ptr[j-h]+d, v+d, block, quadrant, nblock, budget 
530
242M
                 ) ) {
531
131M
            ptr[j] = ptr[j-h];
532
131M
            j = j - h;
533
131M
            if (j <= (lo + h - 1)) break;
534
131M
         }
535
129M
         ptr[j] = v;
536
129M
         i++;
537
538
         /*-- copy 3 --*/
539
129M
         if (i > hi) break;
540
117M
         v = ptr[i];
541
117M
         j = i;
542
233M
         while ( mainGtU ( 
543
233M
                    ptr[j-h]+d, v+d, block, quadrant, nblock, budget 
544
233M
                 ) ) {
545
130M
            ptr[j] = ptr[j-h];
546
130M
            j = j - h;
547
130M
            if (j <= (lo + h - 1)) break;
548
130M
         }
549
117M
         ptr[j] = v;
550
117M
         i++;
551
552
117M
         if (*budget < 0) return;
553
117M
      }
554
39.8M
   }
555
26.6M
}
556
557
558
/*---------------------------------------------*/
559
/*--
560
   The following is an implementation of
561
   an elegant 3-way quicksort for strings,
562
   described in a paper "Fast Algorithms for
563
   Sorting and Searching Strings", by Robert
564
   Sedgewick and Jon L. Bentley.
565
--*/
566
567
#define mswap(zz1, zz2) \
568
1.09G
   { Int32 zztmp = zz1; zz1 = zz2; zz2 = zztmp; }
569
570
3.29M
#define mvswap(zzp1, zzp2, zzn)       \
571
3.29M
{                                     \
572
3.29M
   Int32 yyp1 = (zzp1);               \
573
3.29M
   Int32 yyp2 = (zzp2);               \
574
3.29M
   Int32 yyn  = (zzn);                \
575
43.0M
   while (yyn > 0) {                  \
576
39.7M
      mswap(ptr[yyp1], ptr[yyp2]);    \
577
39.7M
      yyp1++; yyp2++; yyn--;          \
578
39.7M
   }                                  \
579
3.29M
}
580
581
static 
582
__inline__
583
UChar mmed3 ( UChar a, UChar b, UChar c )
584
2.75M
{
585
2.75M
   UChar t;
586
2.75M
   if (a > b) { t = a; a = b; b = t; };
587
2.75M
   if (b > c) { 
588
818k
      b = c;
589
818k
      if (a > b) b = a;
590
818k
   }
591
2.75M
   return b;
592
2.75M
}
593
594
3.29M
#define mmin(a,b) ((a) < (b)) ? (a) : (b)
595
596
30.2M
#define mpush(lz,hz,dz) { stackLo[sp] = lz; \
597
30.2M
                          stackHi[sp] = hz; \
598
30.2M
                          stackD [sp] = dz; \
599
30.2M
                          sp++; }
600
601
30.2M
#define mpop(lz,hz,dz) { sp--;             \
602
30.2M
                         lz = stackLo[sp]; \
603
30.2M
                         hz = stackHi[sp]; \
604
30.2M
                         dz = stackD [sp]; }
605
606
607
9.87M
#define mnextsize(az) (nextHi[az]-nextLo[az])
608
609
#define mnextswap(az,bz)                                        \
610
2.01M
   { Int32 tz;                                                  \
611
2.01M
     tz = nextLo[az]; nextLo[az] = nextLo[bz]; nextLo[bz] = tz; \
612
2.01M
     tz = nextHi[az]; nextHi[az] = nextHi[bz]; nextHi[bz] = tz; \
613
2.01M
     tz = nextD [az]; nextD [az] = nextD [bz]; nextD [bz] = tz; }
614
615
616
60.5M
#define MAIN_QSORT_SMALL_THRESH 20
617
2.89M
#define MAIN_QSORT_DEPTH_THRESH (BZ_N_RADIX + BZ_N_QSORT)
618
#define MAIN_QSORT_STACK_SIZE 100
619
620
static
621
void mainQSort3 ( UInt32* ptr,
622
                  UChar*  block,
623
                  UInt16* quadrant,
624
                  Int32   nblock,
625
                  Int32   loSt, 
626
                  Int32   hiSt, 
627
                  Int32   dSt,
628
                  Int32*  budget )
629
24.2M
{
630
24.2M
   Int32 unLo, unHi, ltLo, gtHi, n, m, med;
631
24.2M
   Int32 sp, lo, hi, d;
632
633
24.2M
   Int32 stackLo[MAIN_QSORT_STACK_SIZE];
634
24.2M
   Int32 stackHi[MAIN_QSORT_STACK_SIZE];
635
24.2M
   Int32 stackD [MAIN_QSORT_STACK_SIZE];
636
637
24.2M
   Int32 nextLo[3];
638
24.2M
   Int32 nextHi[3];
639
24.2M
   Int32 nextD [3];
640
641
24.2M
   sp = 0;
642
24.2M
   mpush ( loSt, hiSt, dSt );
643
644
54.5M
   while (sp > 0) {
645
646
30.2M
      AssertH ( sp < MAIN_QSORT_STACK_SIZE - 2, 1001 );
647
648
30.2M
      mpop ( lo, hi, d );
649
30.2M
      if (hi - lo < MAIN_QSORT_SMALL_THRESH || 
650
30.2M
          d > MAIN_QSORT_DEPTH_THRESH) {
651
27.5M
         mainSimpleSort ( ptr, block, quadrant, nblock, lo, hi, d, budget );
652
27.5M
         if (*budget < 0) return;
653
27.5M
         continue;
654
27.5M
      }
655
656
2.75M
      med = (Int32) 
657
2.75M
            mmed3 ( block[ptr[ lo         ]+d],
658
2.75M
                    block[ptr[ hi         ]+d],
659
2.75M
                    block[ptr[ (lo+hi)>>1 ]+d] );
660
661
2.75M
      unLo = ltLo = lo;
662
2.75M
      unHi = gtHi = hi;
663
664
15.4M
      while (True) {
665
710M
         while (True) {
666
710M
            if (unLo > unHi) break;
667
708M
            n = ((Int32)block[ptr[unLo]+d]) - med;
668
708M
            if (n == 0) { 
669
643M
               mswap(ptr[unLo], ptr[ltLo]); 
670
643M
               ltLo++; unLo++; continue; 
671
643M
            };
672
64.7M
            if (n >  0) break;
673
51.1M
            unLo++;
674
51.1M
         }
675
462M
         while (True) {
676
462M
            if (unLo > unHi) break;
677
459M
            n = ((Int32)block[ptr[unHi]+d]) - med;
678
459M
            if (n == 0) { 
679
399M
               mswap(ptr[unHi], ptr[gtHi]); 
680
399M
               gtHi--; unHi--; continue; 
681
399M
            };
682
59.9M
            if (n <  0) break;
683
47.1M
            unHi--;
684
47.1M
         }
685
15.4M
         if (unLo > unHi) break;
686
12.7M
         mswap(ptr[unLo], ptr[unHi]); unLo++; unHi--;
687
12.7M
      }
688
689
2.75M
      AssertD ( unHi == unLo-1, "mainQSort3(2)" );
690
691
2.75M
      if (gtHi < ltLo) {
692
1.10M
         mpush(lo, hi, d+1 );
693
1.10M
         continue;
694
1.10M
      }
695
696
1.64M
      n = mmin(ltLo-lo, unLo-ltLo); mvswap(lo, unLo-n, n);
697
1.64M
      m = mmin(hi-gtHi, gtHi-unHi); mvswap(unLo, hi-m+1, m);
698
699
1.64M
      n = lo + unLo - ltLo - 1;
700
1.64M
      m = hi - (gtHi - unHi) + 1;
701
702
1.64M
      nextLo[0] = lo;  nextHi[0] = n;   nextD[0] = d;
703
1.64M
      nextLo[1] = m;   nextHi[1] = hi;  nextD[1] = d;
704
1.64M
      nextLo[2] = n+1; nextHi[2] = m-1; nextD[2] = d+1;
705
706
1.64M
      if (mnextsize(0) < mnextsize(1)) mnextswap(0,1);
707
1.64M
      if (mnextsize(1) < mnextsize(2)) mnextswap(1,2);
708
1.64M
      if (mnextsize(0) < mnextsize(1)) mnextswap(0,1);
709
710
1.64M
      AssertD (mnextsize(0) >= mnextsize(1), "mainQSort3(8)" );
711
1.64M
      AssertD (mnextsize(1) >= mnextsize(2), "mainQSort3(9)" );
712
713
1.64M
      mpush (nextLo[0], nextHi[0], nextD[0]);
714
1.64M
      mpush (nextLo[1], nextHi[1], nextD[1]);
715
1.64M
      mpush (nextLo[2], nextHi[2], nextD[2]);
716
1.64M
   }
717
24.2M
}
718
719
#undef mswap
720
#undef mvswap
721
#undef mpush
722
#undef mpop
723
#undef mmin
724
#undef mnextsize
725
#undef mnextswap
726
#undef MAIN_QSORT_SMALL_THRESH
727
#undef MAIN_QSORT_DEPTH_THRESH
728
#undef MAIN_QSORT_STACK_SIZE
729
730
731
/*---------------------------------------------*/
732
/* Pre:
733
      nblock > N_OVERSHOOT
734
      block32 exists for [0 .. nblock-1 +N_OVERSHOOT]
735
      ((UChar*)block32) [0 .. nblock-1] holds block
736
      ptr exists for [0 .. nblock-1]
737
738
   Post:
739
      ((UChar*)block32) [0 .. nblock-1] holds block
740
      All other areas of block32 destroyed
741
      ftab [0 .. 65536 ] destroyed
742
      ptr [0 .. nblock-1] holds sorted order
743
      if (*budget < 0), sorting was abandoned
744
*/
745
746
21.1M
#define BIGFREQ(b) (ftab[((b)+1) << 8] - ftab[(b) << 8])
747
2.47G
#define SETMASK (1 << 21)
748
1.24G
#define CLEARMASK (~(SETMASK))
749
750
static
751
void mainSort ( UInt32* ptr, 
752
                UChar*  block,
753
                UInt16* quadrant, 
754
                UInt32* ftab,
755
                Int32   nblock,
756
                Int32   verb,
757
                Int32*  budget )
758
6.46k
{
759
6.46k
   Int32  i, j, k, ss, sb;
760
6.46k
   Int32  runningOrder[256];
761
6.46k
   Bool   bigDone[256];
762
6.46k
   Int32  copyStart[256];
763
6.46k
   Int32  copyEnd  [256];
764
6.46k
   UChar  c1;
765
6.46k
   Int32  numQSorted;
766
6.46k
   UInt16 s;
767
6.46k
   if (verb >= 4) VPrintf0 ( "        main sort initialise ...\n" );
768
769
   /*-- set up the 2-byte frequency table --*/
770
423M
   for (i = 65536; i >= 0; i--) ftab[i] = 0;
771
772
6.46k
   j = block[0] << 8;
773
6.46k
   i = nblock-1;
774
231M
   for (; i >= 3; i -= 4) {
775
231M
      quadrant[i] = 0;
776
231M
      j = (j >> 8) | ( ((UInt16)block[i]) << 8);
777
231M
      ftab[j]++;
778
231M
      quadrant[i-1] = 0;
779
231M
      j = (j >> 8) | ( ((UInt16)block[i-1]) << 8);
780
231M
      ftab[j]++;
781
231M
      quadrant[i-2] = 0;
782
231M
      j = (j >> 8) | ( ((UInt16)block[i-2]) << 8);
783
231M
      ftab[j]++;
784
231M
      quadrant[i-3] = 0;
785
231M
      j = (j >> 8) | ( ((UInt16)block[i-3]) << 8);
786
231M
      ftab[j]++;
787
231M
   }
788
14.5k
   for (; i >= 0; i--) {
789
8.05k
      quadrant[i] = 0;
790
8.05k
      j = (j >> 8) | ( ((UInt16)block[i]) << 8);
791
8.05k
      ftab[j]++;
792
8.05k
   }
793
794
   /*-- (emphasises close relationship of block & quadrant) --*/
795
226k
   for (i = 0; i < BZ_N_OVERSHOOT; i++) {
796
219k
      block   [nblock+i] = block[i];
797
219k
      quadrant[nblock+i] = 0;
798
219k
   }
799
800
6.46k
   if (verb >= 4) VPrintf0 ( "        bucket sorting ...\n" );
801
802
   /*-- Complete the initial radix sort --*/
803
423M
   for (i = 1; i <= 65536; i++) ftab[i] += ftab[i-1];
804
805
6.46k
   s = block[0] << 8;
806
6.46k
   i = nblock-1;
807
231M
   for (; i >= 3; i -= 4) {
808
231M
      s = (s >> 8) | (block[i] << 8);
809
231M
      j = ftab[s] -1;
810
231M
      ftab[s] = j;
811
231M
      ptr[j] = i;
812
231M
      s = (s >> 8) | (block[i-1] << 8);
813
231M
      j = ftab[s] -1;
814
231M
      ftab[s] = j;
815
231M
      ptr[j] = i-1;
816
231M
      s = (s >> 8) | (block[i-2] << 8);
817
231M
      j = ftab[s] -1;
818
231M
      ftab[s] = j;
819
231M
      ptr[j] = i-2;
820
231M
      s = (s >> 8) | (block[i-3] << 8);
821
231M
      j = ftab[s] -1;
822
231M
      ftab[s] = j;
823
231M
      ptr[j] = i-3;
824
231M
   }
825
14.5k
   for (; i >= 0; i--) {
826
8.05k
      s = (s >> 8) | (block[i] << 8);
827
8.05k
      j = ftab[s] -1;
828
8.05k
      ftab[s] = j;
829
8.05k
      ptr[j] = i;
830
8.05k
   }
831
832
   /*--
833
      Now ftab contains the first loc of every small bucket.
834
      Calculate the running order, from smallest to largest
835
      big bucket.
836
   --*/
837
1.66M
   for (i = 0; i <= 255; i++) {
838
1.65M
      bigDone     [i] = False;
839
1.65M
      runningOrder[i] = i;
840
1.65M
   }
841
842
6.46k
   {
843
6.46k
      Int32 vv;
844
6.46k
      Int32 h = 1;
845
32.3k
      do h = 3 * h + 1; while (h <= 256);
846
32.3k
      do {
847
32.3k
         h = h / 3;
848
7.15M
         for (i = h; i <= 255; i++) {
849
7.12M
            vv = runningOrder[i];
850
7.12M
            j = i;
851
10.5M
            while ( BIGFREQ(runningOrder[j-h]) > BIGFREQ(vv) ) {
852
3.83M
               runningOrder[j] = runningOrder[j-h];
853
3.83M
               j = j - h;
854
3.83M
               if (j <= (h - 1)) goto zero;
855
3.83M
            }
856
7.12M
            zero:
857
7.12M
            runningOrder[j] = vv;
858
7.12M
         }
859
32.3k
      } while (h != 1);
860
6.46k
   }
861
862
   /*--
863
      The main sorting loop.
864
   --*/
865
866
6.46k
   numQSorted = 0;
867
868
1.61M
   for (i = 0; i <= 255; i++) {
869
870
      /*--
871
         Process big buckets, starting with the least full.
872
         Basically this is a 3-step process in which we call
873
         mainQSort3 to sort the small buckets [ss, j], but
874
         also make a big effort to avoid the calls if we can.
875
      --*/
876
1.61M
      ss = runningOrder[i];
877
878
      /*--
879
         Step 1:
880
         Complete the big bucket [ss] by quicksorting
881
         any unsorted small buckets [ss, j], for j != ss.  
882
         Hopefully previous pointer-scanning phases have already
883
         completed many of the small buckets [ss, j], so
884
         we don't have to sort them at all.
885
      --*/
886
413M
      for (j = 0; j <= 255; j++) {
887
412M
         if (j != ss) {
888
410M
            sb = (ss << 8) + j;
889
410M
            if ( ! (ftab[sb] & SETMASK) ) {
890
206M
               Int32 lo = ftab[sb]   & CLEARMASK;
891
206M
               Int32 hi = (ftab[sb+1] & CLEARMASK) - 1;
892
206M
               if (hi > lo) {
893
24.2M
                  if (verb >= 4)
894
0
                     VPrintf4 ( "        qsort [0x%x, 0x%x]   "
895
24.2M
                                "done %d   this %d\n",
896
24.2M
                                ss, j, numQSorted, hi - lo + 1 );
897
24.2M
                  mainQSort3 ( 
898
24.2M
                     ptr, block, quadrant, nblock, 
899
24.2M
                     lo, hi, BZ_N_RADIX, budget 
900
24.2M
                  );   
901
24.2M
                  numQSorted += (hi - lo + 1);
902
24.2M
                  if (*budget < 0) return;
903
24.2M
               }
904
206M
            }
905
410M
            ftab[sb] |= SETMASK;
906
410M
         }
907
412M
      }
908
909
1.60M
      AssertH ( !bigDone[ss], 1006 );
910
911
      /*--
912
         Step 2:
913
         Now scan this big bucket [ss] so as to synthesise the
914
         sorted order for small buckets [t, ss] for all t,
915
         including, magically, the bucket [ss,ss] too.
916
         This will avoid doing Real Work in subsequent Step 1's.
917
      --*/
918
1.60M
      {
919
413M
         for (j = 0; j <= 255; j++) {
920
411M
            copyStart[j] =  ftab[(j << 8) + ss]     & CLEARMASK;
921
411M
            copyEnd  [j] = (ftab[(j << 8) + ss + 1] & CLEARMASK) - 1;
922
411M
         }
923
206M
         for (j = ftab[ss << 8] & CLEARMASK; j < copyStart[ss]; j++) {
924
205M
            k = ptr[j]-1; if (k < 0) k += nblock;
925
205M
            c1 = block[k];
926
205M
            if (!bigDone[c1])
927
116M
               ptr[ copyStart[c1]++ ] = k;
928
205M
         }
929
219M
         for (j = (ftab[(ss+1) << 8] & CLEARMASK) - 1; j > copyEnd[ss]; j--) {
930
217M
            k = ptr[j]-1; if (k < 0) k += nblock;
931
217M
            c1 = block[k];
932
217M
            if (!bigDone[c1]) 
933
118M
               ptr[ copyEnd[c1]-- ] = k;
934
217M
         }
935
1.60M
      }
936
937
1.60M
      AssertH ( (copyStart[ss]-1 == copyEnd[ss])
938
1.60M
                || 
939
                /* Extremely rare case missing in bzip2-1.0.0 and 1.0.1.
940
                   Necessity for this case is demonstrated by compressing 
941
                   a sequence of approximately 48.5 million of character 
942
                   251; 1.0.0/1.0.1 will then die here. */
943
1.60M
                (copyStart[ss] == 0 && copyEnd[ss] == nblock-1),
944
1.60M
                1007 )
945
946
413M
      for (j = 0; j <= 255; j++) ftab[(j << 8) + ss] |= SETMASK;
947
948
      /*--
949
         Step 3:
950
         The [ss] big bucket is now done.  Record this fact,
951
         and update the quadrant descriptors.  Remember to
952
         update quadrants in the overshoot area too, if
953
         necessary.  The "if (i < 255)" test merely skips
954
         this updating for the last bucket processed, since
955
         updating for the last bucket is pointless.
956
957
         The quadrant array provides a way to incrementally
958
         cache sort orderings, as they appear, so as to 
959
         make subsequent comparisons in fullGtU() complete
960
         faster.  For repetitive blocks this makes a big
961
         difference (but not big enough to be able to avoid
962
         the fallback sorting mechanism, exponential radix sort).
963
964
         The precise meaning is: at all times:
965
966
            for 0 <= i < nblock and 0 <= j <= nblock
967
968
            if block[i] != block[j], 
969
970
               then the relative values of quadrant[i] and 
971
                    quadrant[j] are meaningless.
972
973
               else {
974
                  if quadrant[i] < quadrant[j]
975
                     then the string starting at i lexicographically
976
                     precedes the string starting at j
977
978
                  else if quadrant[i] > quadrant[j]
979
                     then the string starting at j lexicographically
980
                     precedes the string starting at i
981
982
                  else
983
                     the relative ordering of the strings starting
984
                     at i and j has not yet been determined.
985
               }
986
      --*/
987
1.60M
      bigDone[ss] = True;
988
989
1.60M
      if (i < 255) {
990
1.60M
         Int32 bbStart  = ftab[ss << 8] & CLEARMASK;
991
1.60M
         Int32 bbSize   = (ftab[(ss+1) << 8] & CLEARMASK) - bbStart;
992
1.60M
         Int32 shifts   = 0;
993
994
1.60M
         while ((bbSize >> shifts) > 65534) shifts++;
995
996
374M
         for (j = bbSize-1; j >= 0; j--) {
997
373M
            Int32 a2update     = ptr[bbStart + j];
998
373M
            UInt16 qVal        = (UInt16)(j >> shifts);
999
373M
            quadrant[a2update] = qVal;
1000
373M
            if (a2update < BZ_N_OVERSHOOT)
1001
104k
               quadrant[a2update + nblock] = qVal;
1002
373M
         }
1003
1.60M
         AssertH ( ((bbSize-1) >> shifts) <= 65535, 1002 );
1004
1.60M
      }
1005
1006
1.60M
   }
1007
1008
2.70k
   if (verb >= 4)
1009
0
      VPrintf3 ( "        %d pointers, %d sorted, %d scanned\n",
1010
2.70k
                 nblock, numQSorted, nblock - numQSorted );
1011
2.70k
}
1012
1013
#undef BIGFREQ
1014
#undef SETMASK
1015
#undef CLEARMASK
1016
1017
1018
/*---------------------------------------------*/
1019
/* Pre:
1020
      nblock > 0
1021
      arr2 exists for [0 .. nblock-1 +N_OVERSHOOT]
1022
      ((UChar*)arr2)  [0 .. nblock-1] holds block
1023
      arr1 exists for [0 .. nblock-1]
1024
1025
   Post:
1026
      ((UChar*)arr2) [0 .. nblock-1] holds block
1027
      All other areas of block destroyed
1028
      ftab [ 0 .. 65536 ] destroyed
1029
      arr1 [0 .. nblock-1] holds sorted order
1030
*/
1031
void BZ2_blockSort ( EState* s )
1032
13.9k
{
1033
13.9k
   UInt32* ptr    = s->ptr; 
1034
13.9k
   UChar*  block  = s->block;
1035
13.9k
   UInt32* ftab   = s->ftab;
1036
13.9k
   Int32   nblock = s->nblock;
1037
13.9k
   Int32   verb   = s->verbosity;
1038
13.9k
   Int32   wfact  = s->workFactor;
1039
13.9k
   UInt16* quadrant;
1040
13.9k
   Int32   budget;
1041
13.9k
   Int32   budgetInit;
1042
13.9k
   Int32   i;
1043
1044
13.9k
   if (nblock < 10000) {
1045
7.43k
      fallbackSort ( s->arr1, s->arr2, ftab, nblock, verb );
1046
7.43k
   } else {
1047
      /* Calculate the location for quadrant, remembering to get
1048
         the alignment right.  Assumes that &(block[0]) is at least
1049
         2-byte aligned -- this should be ok since block is really
1050
         the first section of arr2.
1051
      */
1052
6.46k
      i = nblock+BZ_N_OVERSHOOT;
1053
6.46k
      if (i & 1) i++;
1054
6.46k
      quadrant = (UInt16*)(&(block[i]));
1055
1056
      /* (wfact-1) / 3 puts the default-factor-30
1057
         transition point at very roughly the same place as 
1058
         with v0.1 and v0.9.0.  
1059
         Not that it particularly matters any more, since the
1060
         resulting compressed stream is now the same regardless
1061
         of whether or not we use the main sort or fallback sort.
1062
      */
1063
6.46k
      if (wfact < 1  ) wfact = 1;
1064
6.46k
      if (wfact > 100) wfact = 100;
1065
6.46k
      budgetInit = nblock * ((wfact-1) / 3);
1066
6.46k
      budget = budgetInit;
1067
1068
6.46k
      mainSort ( ptr, block, quadrant, ftab, nblock, verb, &budget );
1069
6.46k
      if (verb >= 3) 
1070
0
         VPrintf3 ( "      %d work, %d block, ratio %5.2f\n",
1071
6.46k
                    budgetInit - budget,
1072
6.46k
                    nblock, 
1073
6.46k
                    (float)(budgetInit - budget) /
1074
6.46k
                    (float)(nblock==0 ? 1 : nblock) ); 
1075
6.46k
      if (budget < 0) {
1076
3.76k
         if (verb >= 2) 
1077
0
            VPrintf0 ( "    too repetitive; using fallback"
1078
3.76k
                       " sorting algorithm\n" );
1079
3.76k
         fallbackSort ( s->arr1, s->arr2, ftab, nblock, verb );
1080
3.76k
      }
1081
6.46k
   }
1082
1083
13.9k
   s->origPtr = -1;
1084
545M
   for (i = 0; i < s->nblock; i++)
1085
545M
      if (ptr[i] == 0)
1086
13.9k
         { s->origPtr = i; break; };
1087
1088
13.9k
   AssertH( s->origPtr != -1, 1003 );
1089
13.9k
}
1090
1091
1092
/*-------------------------------------------------------------*/
1093
/*--- end                                       blocksort.c ---*/
1094
/*-------------------------------------------------------------*/