Coverage Report

Created: 2026-09-06 07:31

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/bzip2/blocksort.c
Line
Count
Source
1
2
/*-------------------------------------------------------------*/
3
/*--- Block sorting machinery                               ---*/
4
/*---                                           blocksort.c ---*/
5
/*-------------------------------------------------------------*/
6
7
/* ------------------------------------------------------------------
8
   This file is part of bzip2/libbzip2, a program and library for
9
   lossless, block-sorting data compression.
10
11
   bzip2/libbzip2 version 1.0.6 of 6 September 2010
12
   Copyright (C) 1996-2010 Julian Seward <jseward@acm.org>
13
14
   Please read the WARNING, DISCLAIMER and PATENTS sections in the 
15
   README file.
16
17
   This program is released under the terms of the license contained
18
   in the file LICENSE.
19
   ------------------------------------------------------------------ */
20
21
22
#include "bzlib_private.h"
23
24
/*---------------------------------------------*/
25
/*--- Fallback O(N log(N)^2) sorting        ---*/
26
/*--- algorithm, for repetitive blocks      ---*/
27
/*---------------------------------------------*/
28
29
/*---------------------------------------------*/
30
static 
31
__inline__
32
void fallbackSimpleSort ( UInt32* fmap, 
33
                          UInt32* eclass, 
34
                          Int32   lo, 
35
                          Int32   hi )
36
105M
{
37
105M
   Int32 i, j, tmp;
38
105M
   UInt32 ec_tmp;
39
40
105M
   if (lo == hi) return;
41
42
102M
   if (hi - lo > 3) {
43
88.7M
      for ( i = hi-4; i >= lo; i-- ) {
44
66.9M
         tmp = fmap[i];
45
66.9M
         ec_tmp = eclass[tmp];
46
90.3M
         for ( j = i+4; j <= hi && ec_tmp > eclass[fmap[j]]; j += 4 )
47
23.4M
            fmap[j-4] = fmap[j];
48
66.9M
         fmap[j-4] = tmp;
49
66.9M
      }
50
21.7M
   }
51
52
339M
   for ( i = hi-1; i >= lo; i-- ) {
53
237M
      tmp = fmap[i];
54
237M
      ec_tmp = eclass[tmp];
55
351M
      for ( j = i+1; j <= hi && ec_tmp > eclass[fmap[j]]; j++ )
56
114M
         fmap[j-1] = fmap[j];
57
237M
      fmap[j-1] = tmp;
58
237M
   }
59
102M
}
60
61
62
/*---------------------------------------------*/
63
#define fswap(zz1, zz2) \
64
1.32G
   { Int32 zztmp = zz1; zz1 = zz2; zz2 = zztmp; }
65
66
46.3M
#define fvswap(zzp1, zzp2, zzn)       \
67
46.3M
{                                     \
68
46.3M
   Int32 yyp1 = (zzp1);               \
69
46.3M
   Int32 yyp2 = (zzp2);               \
70
46.3M
   Int32 yyn  = (zzn);                \
71
245M
   while (yyn > 0) {                  \
72
199M
      fswap(fmap[yyp1], fmap[yyp2]);  \
73
199M
      yyp1++; yyp2++; yyn--;          \
74
199M
   }                                  \
75
46.3M
}
76
77
78
46.3M
#define fmin(a,b) ((a) < (b)) ? (a) : (b)
79
80
130M
#define fpush(lz,hz) { stackLo[sp] = lz; \
81
130M
                       stackHi[sp] = hz; \
82
130M
                       sp++; }
83
84
130M
#define fpop(lz,hz) { sp--;              \
85
130M
                      lz = stackLo[sp];  \
86
130M
                      hz = stackHi[sp]; }
87
88
130M
#define FALLBACK_QSORT_SMALL_THRESH 10
89
#define FALLBACK_QSORT_STACK_SIZE   100
90
91
92
static
93
void fallbackQSort3 ( UInt32* fmap, 
94
                      UInt32* eclass,
95
                      Int32   loSt, 
96
                      Int32   hiSt )
97
84.6M
{
98
84.6M
   Int32 unLo, unHi, ltLo, gtHi, n, m;
99
84.6M
   Int32 sp, lo, hi;
100
84.6M
   UInt32 med, r, r3;
101
84.6M
   Int32 stackLo[FALLBACK_QSORT_STACK_SIZE];
102
84.6M
   Int32 stackHi[FALLBACK_QSORT_STACK_SIZE];
103
104
84.6M
   r = 0;
105
106
84.6M
   sp = 0;
107
84.6M
   fpush ( loSt, hiSt );
108
109
215M
   while (sp > 0) {
110
111
130M
      AssertH ( sp < FALLBACK_QSORT_STACK_SIZE - 1, 1004 );
112
113
130M
      fpop ( lo, hi );
114
130M
      if (hi - lo < FALLBACK_QSORT_SMALL_THRESH) {
115
105M
         fallbackSimpleSort ( fmap, eclass, lo, hi );
116
105M
         continue;
117
105M
      }
118
119
      /* Random partitioning.  Median of 3 sometimes fails to
120
         avoid bad cases.  Median of 9 seems to help but 
121
         looks rather expensive.  This too seems to work but
122
         is cheaper.  Guidance for the magic constants 
123
         7621 and 32768 is taken from Sedgewick's algorithms
124
         book, chapter 35.
125
      */
126
25.2M
      r = ((r * 7621) + 1) % 32768;
127
25.2M
      r3 = r % 3;
128
25.2M
      if (r3 == 0) med = eclass[fmap[lo]]; else
129
22.2M
      if (r3 == 1) med = eclass[fmap[(lo+hi)>>1]]; else
130
8.32M
                   med = eclass[fmap[hi]];
131
132
25.2M
      unLo = ltLo = lo;
133
25.2M
      unHi = gtHi = hi;
134
135
129M
      while (1) {
136
1.28G
         while (1) {
137
1.28G
            if (unLo > unHi) break;
138
1.26G
            n = (Int32)eclass[fmap[unLo]] - (Int32)med;
139
1.26G
            if (n == 0) { 
140
811M
               fswap(fmap[unLo], fmap[ltLo]); 
141
811M
               ltLo++; unLo++; 
142
811M
               continue; 
143
811M
            };
144
457M
            if (n > 0) break;
145
341M
            unLo++;
146
341M
         }
147
688M
         while (1) {
148
688M
            if (unLo > unHi) break;
149
663M
            n = (Int32)eclass[fmap[unHi]] - (Int32)med;
150
663M
            if (n == 0) { 
151
214M
               fswap(fmap[unHi], fmap[gtHi]); 
152
214M
               gtHi--; unHi--; 
153
214M
               continue; 
154
448M
            };
155
448M
            if (n < 0) break;
156
344M
            unHi--;
157
344M
         }
158
129M
         if (unLo > unHi) break;
159
104M
         fswap(fmap[unLo], fmap[unHi]); unLo++; unHi--;
160
104M
      }
161
162
25.2M
      AssertD ( unHi == unLo-1, "fallbackQSort3(2)" );
163
164
25.2M
      if (gtHi < ltLo) continue;
165
166
23.1M
      n = fmin(ltLo-lo, unLo-ltLo); fvswap(lo, unLo-n, n);
167
23.1M
      m = fmin(hi-gtHi, gtHi-unHi); fvswap(unLo, hi-m+1, m);
168
169
23.1M
      n = lo + unLo - ltLo - 1;
170
23.1M
      m = hi - (gtHi - unHi) + 1;
171
172
23.1M
      if (n - lo > hi - m) {
173
11.6M
         fpush ( lo, n );
174
11.6M
         fpush ( m, hi );
175
11.6M
      } else {
176
11.5M
         fpush ( m, hi );
177
11.5M
         fpush ( lo, n );
178
11.5M
      }
179
23.1M
   }
180
84.6M
}
181
182
#undef fmin
183
#undef fpush
184
#undef fpop
185
#undef fswap
186
#undef fvswap
187
#undef FALLBACK_QSORT_SMALL_THRESH
188
#undef FALLBACK_QSORT_STACK_SIZE
189
190
191
/*---------------------------------------------*/
192
/* Pre:
193
      nblock > 0
194
      eclass exists for [0 .. nblock-1]
195
      ((UChar*)eclass) [0 .. nblock-1] holds block
196
      ptr exists for [0 .. nblock-1]
197
198
   Post:
199
      ((UChar*)eclass) [0 .. nblock-1] holds block
200
      All other areas of eclass destroyed
201
      fmap [0 .. nblock-1] holds sorted order
202
      bhtab [ 0 .. 2+(nblock/32) ] destroyed
203
*/
204
205
266M
#define       SET_BH(zz)  bhtab[(zz) >> 5] |= ((UInt32)1 << ((zz) & 31))
206
3.16M
#define     CLEAR_BH(zz)  bhtab[(zz) >> 5] &= ~((UInt32)1 << ((zz) & 31))
207
3.45G
#define     ISSET_BH(zz)  (bhtab[(zz) >> 5] & ((UInt32)1 << ((zz) & 31)))
208
58.3M
#define      WORD_BH(zz)  bhtab[(zz) >> 5]
209
488M
#define UNALIGNED_BH(zz)  ((zz) & 0x01f)
210
211
static
212
void fallbackSort ( UInt32* fmap, 
213
                    UInt32* eclass, 
214
                    UInt32* bhtab,
215
                    Int32   nblock,
216
                    Int32   verb )
217
99.0k
{
218
99.0k
   Int32 ftab[257];
219
99.0k
   Int32 ftabCopy[256];
220
99.0k
   Int32 H, i, j, k, l, r, cc, cc1;
221
99.0k
   Int32 nNotDone;
222
99.0k
   Int32 nBhtab;
223
99.0k
   UChar* eclass8 = (UChar*)eclass;
224
225
   /*--
226
      Initial 1-char radix sort to generate
227
      initial fmap and initial BH bits.
228
   --*/
229
99.0k
   if (verb >= 4)
230
0
      VPrintf0 ( "        bucket sorting ...\n" );
231
25.5M
   for (i = 0; i < 257;    i++) ftab[i] = 0;
232
161M
   for (i = 0; i < nblock; i++) ftab[eclass8[i]]++;
233
25.4M
   for (i = 0; i < 256;    i++) ftabCopy[i] = ftab[i];
234
25.4M
   for (i = 1; i < 257;    i++) ftab[i] += ftab[i-1];
235
236
161M
   for (i = 0; i < nblock; i++) {
237
160M
      j = eclass8[i];
238
160M
      k = ftab[j] - 1;
239
160M
      ftab[j] = k;
240
160M
      fmap[k] = i;
241
160M
   }
242
243
99.0k
   nBhtab = 2 + (nblock / 32);
244
5.28M
   for (i = 0; i < nBhtab; i++) bhtab[i] = 0;
245
25.4M
   for (i = 0; i < 256; i++) SET_BH(ftab[i]);
246
247
   /*--
248
      Inductively refine the buckets.  Kind-of an
249
      "exponential radix sort" (!), inspired by the
250
      Manber-Myers suffix array construction algorithm.
251
   --*/
252
253
   /*-- set sentinel bits for block-end detection --*/
254
3.26M
   for (i = 0; i < 32; i++) { 
255
3.16M
      SET_BH(nblock + 2*i);
256
3.16M
      CLEAR_BH(nblock + 2*i + 1);
257
3.16M
   }
258
259
   /*-- the log(N) loop --*/
260
99.0k
   H = 1;
261
605k
   while (1) {
262
263
605k
      if (verb >= 4) 
264
0
         VPrintf1 ( "        depth %6d has ", H );
265
266
605k
      j = 0;
267
1.85G
      for (i = 0; i < nblock; i++) {
268
1.85G
         if (ISSET_BH(i)) j = i;
269
1.85G
         k = fmap[i] - H; if (k < 0) k += nblock;
270
1.85G
         eclass[k] = j;
271
1.85G
      }
272
273
605k
      nNotDone = 0;
274
605k
      r = -1;
275
85.2M
      while (1) {
276
277
   /*-- find the next non-singleton bucket --*/
278
85.2M
         k = r + 1;
279
310M
         while (ISSET_BH(k) && UNALIGNED_BH(k)) k++;
280
85.2M
         if (ISSET_BH(k)) {
281
18.5M
            while (WORD_BH(k) == 0xffffffff) k += 32;
282
68.9M
            while (ISSET_BH(k)) k++;
283
9.47M
         }
284
85.2M
         l = k - 1;
285
85.2M
         if (l >= nblock) break;
286
327M
         while (!ISSET_BH(k) && UNALIGNED_BH(k)) k++;
287
84.6M
         if (!ISSET_BH(k)) {
288
39.8M
            while (WORD_BH(k) == 0x00000000) k += 32;
289
91.9M
            while (!ISSET_BH(k)) k++;
290
10.0M
         }
291
84.6M
         r = k - 1;
292
84.6M
         if (r >= nblock) break;
293
294
         /*-- now [l, r] bracket current bucket --*/
295
84.6M
         if (r > l) {
296
84.6M
            nNotDone += (r - l + 1);
297
84.6M
            fallbackQSort3 ( fmap, eclass, l, r );
298
299
            /*-- scan bucket and generate header bits-- */
300
84.6M
            cc = -1;
301
1.44G
            for (i = l; i <= r; i++) {
302
1.36G
               cc1 = eclass[fmap[i]];
303
1.36G
               if (cc != cc1) { SET_BH(i); cc = cc1; };
304
1.36G
            }
305
84.6M
         }
306
84.6M
      }
307
308
605k
      if (verb >= 4) 
309
0
         VPrintf1 ( "%6d unresolved strings\n", nNotDone );
310
311
605k
      H *= 2;
312
605k
      if (H > nblock || nNotDone == 0) break;
313
605k
   }
314
315
   /*-- 
316
      Reconstruct the original block in
317
      eclass8 [0 .. nblock-1], since the
318
      previous phase destroyed it.
319
   --*/
320
99.0k
   if (verb >= 4)
321
0
      VPrintf0 ( "        reconstructing block ...\n" );
322
99.0k
   j = 0;
323
161M
   for (i = 0; i < nblock; i++) {
324
177M
      while (ftabCopy[j] == 0) j++;
325
160M
      ftabCopy[j]--;
326
160M
      eclass8[fmap[i]] = (UChar)j;
327
160M
   }
328
99.0k
   AssertH ( j < 256, 1005 );
329
99.0k
}
330
331
#undef       SET_BH
332
#undef     CLEAR_BH
333
#undef     ISSET_BH
334
#undef      WORD_BH
335
#undef UNALIGNED_BH
336
337
338
/*---------------------------------------------*/
339
/*--- The main, O(N^2 log(N)) sorting       ---*/
340
/*--- algorithm.  Faster for "normal"       ---*/
341
/*--- non-repetitive blocks.                ---*/
342
/*---------------------------------------------*/
343
344
/*---------------------------------------------*/
345
static
346
__inline__
347
Bool mainGtU ( UInt32  i1, 
348
               UInt32  i2,
349
               UChar*  block, 
350
               UInt16* quadrant,
351
               UInt32  nblock,
352
               Int32*  budget )
353
94.3M
{
354
94.3M
   Int32  k;
355
94.3M
   UChar  c1, c2;
356
94.3M
   UInt16 s1, s2;
357
358
94.3M
   AssertD ( i1 != i2, "mainGtU" );
359
   /* 1 */
360
94.3M
   c1 = block[i1]; c2 = block[i2];
361
94.3M
   if (c1 != c2) return (c1 > c2);
362
91.7M
   i1++; i2++;
363
   /* 2 */
364
91.7M
   c1 = block[i1]; c2 = block[i2];
365
91.7M
   if (c1 != c2) return (c1 > c2);
366
90.0M
   i1++; i2++;
367
   /* 3 */
368
90.0M
   c1 = block[i1]; c2 = block[i2];
369
90.0M
   if (c1 != c2) return (c1 > c2);
370
88.7M
   i1++; i2++;
371
   /* 4 */
372
88.7M
   c1 = block[i1]; c2 = block[i2];
373
88.7M
   if (c1 != c2) return (c1 > c2);
374
87.3M
   i1++; i2++;
375
   /* 5 */
376
87.3M
   c1 = block[i1]; c2 = block[i2];
377
87.3M
   if (c1 != c2) return (c1 > c2);
378
86.0M
   i1++; i2++;
379
   /* 6 */
380
86.0M
   c1 = block[i1]; c2 = block[i2];
381
86.0M
   if (c1 != c2) return (c1 > c2);
382
84.7M
   i1++; i2++;
383
   /* 7 */
384
84.7M
   c1 = block[i1]; c2 = block[i2];
385
84.7M
   if (c1 != c2) return (c1 > c2);
386
83.5M
   i1++; i2++;
387
   /* 8 */
388
83.5M
   c1 = block[i1]; c2 = block[i2];
389
83.5M
   if (c1 != c2) return (c1 > c2);
390
82.3M
   i1++; i2++;
391
   /* 9 */
392
82.3M
   c1 = block[i1]; c2 = block[i2];
393
82.3M
   if (c1 != c2) return (c1 > c2);
394
81.3M
   i1++; i2++;
395
   /* 10 */
396
81.3M
   c1 = block[i1]; c2 = block[i2];
397
81.3M
   if (c1 != c2) return (c1 > c2);
398
80.1M
   i1++; i2++;
399
   /* 11 */
400
80.1M
   c1 = block[i1]; c2 = block[i2];
401
80.1M
   if (c1 != c2) return (c1 > c2);
402
79.2M
   i1++; i2++;
403
   /* 12 */
404
79.2M
   c1 = block[i1]; c2 = block[i2];
405
79.2M
   if (c1 != c2) return (c1 > c2);
406
78.3M
   i1++; i2++;
407
408
78.3M
   k = nblock + 8;
409
410
805M
   do {
411
      /* 1 */
412
805M
      c1 = block[i1]; c2 = block[i2];
413
805M
      if (c1 != c2) return (c1 > c2);
414
802M
      s1 = quadrant[i1]; s2 = quadrant[i2];
415
802M
      if (s1 != s2) return (s1 > s2);
416
792M
      i1++; i2++;
417
      /* 2 */
418
792M
      c1 = block[i1]; c2 = block[i2];
419
792M
      if (c1 != c2) return (c1 > c2);
420
788M
      s1 = quadrant[i1]; s2 = quadrant[i2];
421
788M
      if (s1 != s2) return (s1 > s2);
422
777M
      i1++; i2++;
423
      /* 3 */
424
777M
      c1 = block[i1]; c2 = block[i2];
425
777M
      if (c1 != c2) return (c1 > c2);
426
773M
      s1 = quadrant[i1]; s2 = quadrant[i2];
427
773M
      if (s1 != s2) return (s1 > s2);
428
769M
      i1++; i2++;
429
      /* 4 */
430
769M
      c1 = block[i1]; c2 = block[i2];
431
769M
      if (c1 != c2) return (c1 > c2);
432
766M
      s1 = quadrant[i1]; s2 = quadrant[i2];
433
766M
      if (s1 != s2) return (s1 > s2);
434
761M
      i1++; i2++;
435
      /* 5 */
436
761M
      c1 = block[i1]; c2 = block[i2];
437
761M
      if (c1 != c2) return (c1 > c2);
438
758M
      s1 = quadrant[i1]; s2 = quadrant[i2];
439
758M
      if (s1 != s2) return (s1 > s2);
440
754M
      i1++; i2++;
441
      /* 6 */
442
754M
      c1 = block[i1]; c2 = block[i2];
443
754M
      if (c1 != c2) return (c1 > c2);
444
750M
      s1 = quadrant[i1]; s2 = quadrant[i2];
445
750M
      if (s1 != s2) return (s1 > s2);
446
743M
      i1++; i2++;
447
      /* 7 */
448
743M
      c1 = block[i1]; c2 = block[i2];
449
743M
      if (c1 != c2) return (c1 > c2);
450
740M
      s1 = quadrant[i1]; s2 = quadrant[i2];
451
740M
      if (s1 != s2) return (s1 > s2);
452
737M
      i1++; i2++;
453
      /* 8 */
454
737M
      c1 = block[i1]; c2 = block[i2];
455
737M
      if (c1 != c2) return (c1 > c2);
456
734M
      s1 = quadrant[i1]; s2 = quadrant[i2];
457
734M
      if (s1 != s2) return (s1 > s2);
458
727M
      i1++; i2++;
459
460
727M
      if (i1 >= nblock) i1 -= nblock;
461
727M
      if (i2 >= nblock) i2 -= nblock;
462
463
727M
      k -= 8;
464
727M
      (*budget)--;
465
727M
   }
466
727M
      while (k >= 0);
467
468
16.9k
   return False;
469
78.3M
}
470
471
472
/*---------------------------------------------*/
473
/*--
474
   Knuth's increments seem to work better
475
   than Incerpi-Sedgewick here.  Possibly
476
   because the number of elems to sort is
477
   usually small, typically <= 20.
478
--*/
479
static
480
Int32 incs[14] = { 1, 4, 13, 40, 121, 364, 1093, 3280,
481
                   9841, 29524, 88573, 265720,
482
                   797161, 2391484 };
483
484
static
485
void mainSimpleSort ( UInt32* ptr,
486
                      UChar*  block,
487
                      UInt16* quadrant,
488
                      Int32   nblock,
489
                      Int32   lo, 
490
                      Int32   hi, 
491
                      Int32   d,
492
                      Int32*  budget )
493
902k
{
494
902k
   Int32 i, j, h, bigN, hp;
495
902k
   UInt32 v;
496
497
902k
   bigN = hi - lo + 1;
498
902k
   if (bigN < 2) return;
499
500
700k
   hp = 0;
501
1.92M
   while (incs[hp] < bigN) hp++;
502
700k
   hp--;
503
504
1.89M
   for (; hp >= 0; hp--) {
505
1.20M
      h = incs[hp];
506
507
1.20M
      i = lo + h;
508
15.7M
      while (True) {
509
510
         /*-- copy 1 --*/
511
15.7M
         if (i > hi) break;
512
15.4M
         v = ptr[i];
513
15.4M
         j = i;
514
31.7M
         while ( mainGtU ( 
515
31.7M
                    ptr[j-h]+d, v+d, block, quadrant, nblock, budget 
516
31.7M
                 ) ) {
517
18.6M
            ptr[j] = ptr[j-h];
518
18.6M
            j = j - h;
519
18.6M
            if (j <= (lo + h - 1)) break;
520
18.6M
         }
521
15.4M
         ptr[j] = v;
522
15.4M
         i++;
523
524
         /*-- copy 2 --*/
525
15.4M
         if (i > hi) break;
526
14.8M
         v = ptr[i];
527
14.8M
         j = i;
528
31.4M
         while ( mainGtU ( 
529
31.4M
                    ptr[j-h]+d, v+d, block, quadrant, nblock, budget 
530
31.4M
                 ) ) {
531
18.6M
            ptr[j] = ptr[j-h];
532
18.6M
            j = j - h;
533
18.6M
            if (j <= (lo + h - 1)) break;
534
18.6M
         }
535
14.8M
         ptr[j] = v;
536
14.8M
         i++;
537
538
         /*-- copy 3 --*/
539
14.8M
         if (i > hi) break;
540
14.5M
         v = ptr[i];
541
14.5M
         j = i;
542
31.2M
         while ( mainGtU ( 
543
31.2M
                    ptr[j-h]+d, v+d, block, quadrant, nblock, budget 
544
31.2M
                 ) ) {
545
18.5M
            ptr[j] = ptr[j-h];
546
18.5M
            j = j - h;
547
18.5M
            if (j <= (lo + h - 1)) break;
548
18.5M
         }
549
14.5M
         ptr[j] = v;
550
14.5M
         i++;
551
552
14.5M
         if (*budget < 0) return;
553
14.5M
      }
554
1.20M
   }
555
700k
}
556
557
558
/*---------------------------------------------*/
559
/*--
560
   The following is an implementation of
561
   an elegant 3-way quicksort for strings,
562
   described in a paper "Fast Algorithms for
563
   Sorting and Searching Strings", by Robert
564
   Sedgewick and Jon L. Bentley.
565
--*/
566
567
#define mswap(zz1, zz2) \
568
208M
   { Int32 zztmp = zz1; zz1 = zz2; zz2 = zztmp; }
569
570
405k
#define mvswap(zzp1, zzp2, zzn)       \
571
405k
{                                     \
572
405k
   Int32 yyp1 = (zzp1);               \
573
405k
   Int32 yyp2 = (zzp2);               \
574
405k
   Int32 yyn  = (zzn);                \
575
8.28M
   while (yyn > 0) {                  \
576
7.87M
      mswap(ptr[yyp1], ptr[yyp2]);    \
577
7.87M
      yyp1++; yyp2++; yyn--;          \
578
7.87M
   }                                  \
579
405k
}
580
581
static 
582
__inline__
583
UChar mmed3 ( UChar a, UChar b, UChar c )
584
929k
{
585
929k
   UChar t;
586
929k
   if (a > b) { t = a; a = b; b = t; };
587
929k
   if (b > c) { 
588
43.8k
      b = c;
589
43.8k
      if (a > b) b = a;
590
43.8k
   }
591
929k
   return b;
592
929k
}
593
594
405k
#define mmin(a,b) ((a) < (b)) ? (a) : (b)
595
596
1.83M
#define mpush(lz,hz,dz) { stackLo[sp] = lz; \
597
1.83M
                          stackHi[sp] = hz; \
598
1.83M
                          stackD [sp] = dz; \
599
1.83M
                          sp++; }
600
601
1.83M
#define mpop(lz,hz,dz) { sp--;             \
602
1.83M
                         lz = stackLo[sp]; \
603
1.83M
                         hz = stackHi[sp]; \
604
1.83M
                         dz = stackD [sp]; }
605
606
607
1.21M
#define mnextsize(az) (nextHi[az]-nextLo[az])
608
609
#define mnextswap(az,bz)                                        \
610
430k
   { Int32 tz;                                                  \
611
430k
     tz = nextLo[az]; nextLo[az] = nextLo[bz]; nextLo[bz] = tz; \
612
430k
     tz = nextHi[az]; nextHi[az] = nextHi[bz]; nextHi[bz] = tz; \
613
430k
     tz = nextD [az]; nextD [az] = nextD [bz]; nextD [bz] = tz; }
614
615
616
3.66M
#define MAIN_QSORT_SMALL_THRESH 20
617
1.01M
#define MAIN_QSORT_DEPTH_THRESH (BZ_N_RADIX + BZ_N_QSORT)
618
#define MAIN_QSORT_STACK_SIZE 100
619
620
static
621
void mainQSort3 ( UInt32* ptr,
622
                  UChar*  block,
623
                  UInt16* quadrant,
624
                  Int32   nblock,
625
                  Int32   loSt, 
626
                  Int32   hiSt, 
627
                  Int32   dSt,
628
                  Int32*  budget )
629
498k
{
630
498k
   Int32 unLo, unHi, ltLo, gtHi, n, m, med;
631
498k
   Int32 sp, lo, hi, d;
632
633
498k
   Int32 stackLo[MAIN_QSORT_STACK_SIZE];
634
498k
   Int32 stackHi[MAIN_QSORT_STACK_SIZE];
635
498k
   Int32 stackD [MAIN_QSORT_STACK_SIZE];
636
637
498k
   Int32 nextLo[3];
638
498k
   Int32 nextHi[3];
639
498k
   Int32 nextD [3];
640
641
498k
   sp = 0;
642
498k
   mpush ( loSt, hiSt, dSt );
643
644
2.32M
   while (sp > 0) {
645
646
1.83M
      AssertH ( sp < MAIN_QSORT_STACK_SIZE - 2, 1001 );
647
648
1.83M
      mpop ( lo, hi, d );
649
1.83M
      if (hi - lo < MAIN_QSORT_SMALL_THRESH || 
650
1.01M
          d > MAIN_QSORT_DEPTH_THRESH) {
651
902k
         mainSimpleSort ( ptr, block, quadrant, nblock, lo, hi, d, budget );
652
902k
         if (*budget < 0) return;
653
897k
         continue;
654
902k
      }
655
656
929k
      med = (Int32) 
657
929k
            mmed3 ( block[ptr[ lo         ]+d],
658
929k
                    block[ptr[ hi         ]+d],
659
929k
                    block[ptr[ (lo+hi)>>1 ]+d] );
660
661
929k
      unLo = ltLo = lo;
662
929k
      unHi = gtHi = hi;
663
664
1.40M
      while (True) {
665
180M
         while (True) {
666
180M
            if (unLo > unHi) break;
667
179M
            n = ((Int32)block[ptr[unLo]+d]) - med;
668
179M
            if (n == 0) { 
669
174M
               mswap(ptr[unLo], ptr[ltLo]); 
670
174M
               ltLo++; unLo++; continue; 
671
174M
            };
672
5.24M
            if (n >  0) break;
673
4.68M
            unLo++;
674
4.68M
         }
675
32.2M
         while (True) {
676
32.2M
            if (unLo > unHi) break;
677
31.2M
            n = ((Int32)block[ptr[unHi]+d]) - med;
678
31.2M
            if (n == 0) { 
679
26.2M
               mswap(ptr[unHi], ptr[gtHi]); 
680
26.2M
               gtHi--; unHi--; continue; 
681
26.2M
            };
682
5.04M
            if (n <  0) break;
683
4.56M
            unHi--;
684
4.56M
         }
685
1.40M
         if (unLo > unHi) break;
686
475k
         mswap(ptr[unLo], ptr[unHi]); unLo++; unHi--;
687
475k
      }
688
689
929k
      AssertD ( unHi == unLo-1, "mainQSort3(2)" );
690
691
929k
      if (gtHi < ltLo) {
692
726k
         mpush(lo, hi, d+1 );
693
726k
         continue;
694
726k
      }
695
696
202k
      n = mmin(ltLo-lo, unLo-ltLo); mvswap(lo, unLo-n, n);
697
202k
      m = mmin(hi-gtHi, gtHi-unHi); mvswap(unLo, hi-m+1, m);
698
699
202k
      n = lo + unLo - ltLo - 1;
700
202k
      m = hi - (gtHi - unHi) + 1;
701
702
202k
      nextLo[0] = lo;  nextHi[0] = n;   nextD[0] = d;
703
202k
      nextLo[1] = m;   nextHi[1] = hi;  nextD[1] = d;
704
202k
      nextLo[2] = n+1; nextHi[2] = m-1; nextD[2] = d+1;
705
706
202k
      if (mnextsize(0) < mnextsize(1)) mnextswap(0,1);
707
202k
      if (mnextsize(1) < mnextsize(2)) mnextswap(1,2);
708
202k
      if (mnextsize(0) < mnextsize(1)) mnextswap(0,1);
709
710
202k
      AssertD (mnextsize(0) >= mnextsize(1), "mainQSort3(8)" );
711
202k
      AssertD (mnextsize(1) >= mnextsize(2), "mainQSort3(9)" );
712
713
202k
      mpush (nextLo[0], nextHi[0], nextD[0]);
714
202k
      mpush (nextLo[1], nextHi[1], nextD[1]);
715
202k
      mpush (nextLo[2], nextHi[2], nextD[2]);
716
202k
   }
717
498k
}
718
719
#undef mswap
720
#undef mvswap
721
#undef mpush
722
#undef mpop
723
#undef mmin
724
#undef mnextsize
725
#undef mnextswap
726
#undef MAIN_QSORT_SMALL_THRESH
727
#undef MAIN_QSORT_DEPTH_THRESH
728
#undef MAIN_QSORT_STACK_SIZE
729
730
731
/*---------------------------------------------*/
732
/* Pre:
733
      nblock > N_OVERSHOOT
734
      block32 exists for [0 .. nblock-1 +N_OVERSHOOT]
735
      ((UChar*)block32) [0 .. nblock-1] holds block
736
      ptr exists for [0 .. nblock-1]
737
738
   Post:
739
      ((UChar*)block32) [0 .. nblock-1] holds block
740
      All other areas of block32 destroyed
741
      ftab [0 .. 65536 ] destroyed
742
      ptr [0 .. nblock-1] holds sorted order
743
      if (*budget < 0), sorting was abandoned
744
*/
745
746
19.2M
#define BIGFREQ(b) (ftab[((b)+1) << 8] - ftab[(b) << 8])
747
2.92G
#define SETMASK (1 << 21)
748
1.46G
#define CLEARMASK (~(SETMASK))
749
750
static
751
void mainSort ( UInt32* ptr, 
752
                UChar*  block,
753
                UInt16* quadrant, 
754
                UInt32* ftab,
755
                Int32   nblock,
756
                Int32   verb,
757
                Int32*  budget )
758
7.52k
{
759
7.52k
   Int32  i, j, k, ss, sb;
760
7.52k
   Int32  runningOrder[256];
761
7.52k
   Bool   bigDone[256];
762
7.52k
   Int32  copyStart[256];
763
7.52k
   Int32  copyEnd  [256];
764
7.52k
   UChar  c1;
765
7.52k
   Int32  numQSorted;
766
7.52k
   UInt16 s;
767
7.52k
   if (verb >= 4) VPrintf0 ( "        main sort initialise ...\n" );
768
769
   /*-- set up the 2-byte frequency table --*/
770
492M
   for (i = 65536; i >= 0; i--) ftab[i] = 0;
771
772
7.52k
   j = block[0] << 8;
773
7.52k
   i = nblock-1;
774
23.4M
   for (; i >= 3; i -= 4) {
775
23.4M
      quadrant[i] = 0;
776
23.4M
      j = (j >> 8) | ( ((UInt16)block[i]) << 8);
777
23.4M
      ftab[j]++;
778
23.4M
      quadrant[i-1] = 0;
779
23.4M
      j = (j >> 8) | ( ((UInt16)block[i-1]) << 8);
780
23.4M
      ftab[j]++;
781
23.4M
      quadrant[i-2] = 0;
782
23.4M
      j = (j >> 8) | ( ((UInt16)block[i-2]) << 8);
783
23.4M
      ftab[j]++;
784
23.4M
      quadrant[i-3] = 0;
785
23.4M
      j = (j >> 8) | ( ((UInt16)block[i-3]) << 8);
786
23.4M
      ftab[j]++;
787
23.4M
   }
788
16.7k
   for (; i >= 0; i--) {
789
9.18k
      quadrant[i] = 0;
790
9.18k
      j = (j >> 8) | ( ((UInt16)block[i]) << 8);
791
9.18k
      ftab[j]++;
792
9.18k
   }
793
794
   /*-- (emphasises close relationship of block & quadrant) --*/
795
263k
   for (i = 0; i < BZ_N_OVERSHOOT; i++) {
796
255k
      block   [nblock+i] = block[i];
797
255k
      quadrant[nblock+i] = 0;
798
255k
   }
799
800
7.52k
   if (verb >= 4) VPrintf0 ( "        bucket sorting ...\n" );
801
802
   /*-- Complete the initial radix sort --*/
803
492M
   for (i = 1; i <= 65536; i++) ftab[i] += ftab[i-1];
804
805
7.52k
   s = block[0] << 8;
806
7.52k
   i = nblock-1;
807
23.4M
   for (; i >= 3; i -= 4) {
808
23.4M
      s = (s >> 8) | (block[i] << 8);
809
23.4M
      j = ftab[s] -1;
810
23.4M
      ftab[s] = j;
811
23.4M
      ptr[j] = i;
812
23.4M
      s = (s >> 8) | (block[i-1] << 8);
813
23.4M
      j = ftab[s] -1;
814
23.4M
      ftab[s] = j;
815
23.4M
      ptr[j] = i-1;
816
23.4M
      s = (s >> 8) | (block[i-2] << 8);
817
23.4M
      j = ftab[s] -1;
818
23.4M
      ftab[s] = j;
819
23.4M
      ptr[j] = i-2;
820
23.4M
      s = (s >> 8) | (block[i-3] << 8);
821
23.4M
      j = ftab[s] -1;
822
23.4M
      ftab[s] = j;
823
23.4M
      ptr[j] = i-3;
824
23.4M
   }
825
16.7k
   for (; i >= 0; i--) {
826
9.18k
      s = (s >> 8) | (block[i] << 8);
827
9.18k
      j = ftab[s] -1;
828
9.18k
      ftab[s] = j;
829
9.18k
      ptr[j] = i;
830
9.18k
   }
831
832
   /*--
833
      Now ftab contains the first loc of every small bucket.
834
      Calculate the running order, from smallest to largest
835
      big bucket.
836
   --*/
837
1.93M
   for (i = 0; i <= 255; i++) {
838
1.92M
      bigDone     [i] = False;
839
1.92M
      runningOrder[i] = i;
840
1.92M
   }
841
842
7.52k
   {
843
7.52k
      Int32 vv;
844
7.52k
      Int32 h = 1;
845
37.6k
      do h = 3 * h + 1; while (h <= 256);
846
37.6k
      do {
847
37.6k
         h = h / 3;
848
8.31M
         for (i = h; i <= 255; i++) {
849
8.28M
            vv = runningOrder[i];
850
8.28M
            j = i;
851
9.63M
            while ( BIGFREQ(runningOrder[j-h]) > BIGFREQ(vv) ) {
852
1.50M
               runningOrder[j] = runningOrder[j-h];
853
1.50M
               j = j - h;
854
1.50M
               if (j <= (h - 1)) goto zero;
855
1.50M
            }
856
8.28M
            zero:
857
8.28M
            runningOrder[j] = vv;
858
8.28M
         }
859
37.6k
      } while (h != 1);
860
7.52k
   }
861
862
   /*--
863
      The main sorting loop.
864
   --*/
865
866
7.52k
   numQSorted = 0;
867
868
1.90M
   for (i = 0; i <= 255; i++) {
869
870
      /*--
871
         Process big buckets, starting with the least full.
872
         Basically this is a 3-step process in which we call
873
         mainQSort3 to sort the small buckets [ss, j], but
874
         also make a big effort to avoid the calls if we can.
875
      --*/
876
1.90M
      ss = runningOrder[i];
877
878
      /*--
879
         Step 1:
880
         Complete the big bucket [ss] by quicksorting
881
         any unsorted small buckets [ss, j], for j != ss.  
882
         Hopefully previous pointer-scanning phases have already
883
         completed many of the small buckets [ss, j], so
884
         we don't have to sort them at all.
885
      --*/
886
487M
      for (j = 0; j <= 255; j++) {
887
485M
         if (j != ss) {
888
484M
            sb = (ss << 8) + j;
889
484M
            if ( ! (ftab[sb] & SETMASK) ) {
890
245M
               Int32 lo = ftab[sb]   & CLEARMASK;
891
245M
               Int32 hi = (ftab[sb+1] & CLEARMASK) - 1;
892
245M
               if (hi > lo) {
893
498k
                  if (verb >= 4)
894
0
                     VPrintf4 ( "        qsort [0x%x, 0x%x]   "
895
498k
                                "done %d   this %d\n",
896
498k
                                ss, j, numQSorted, hi - lo + 1 );
897
498k
                  mainQSort3 ( 
898
498k
                     ptr, block, quadrant, nblock, 
899
498k
                     lo, hi, BZ_N_RADIX, budget 
900
498k
                  );   
901
498k
                  numQSorted += (hi - lo + 1);
902
498k
                  if (*budget < 0) return;
903
498k
               }
904
245M
            }
905
484M
            ftab[sb] |= SETMASK;
906
484M
         }
907
485M
      }
908
909
1.89M
      AssertH ( !bigDone[ss], 1006 );
910
911
      /*--
912
         Step 2:
913
         Now scan this big bucket [ss] so as to synthesise the
914
         sorted order for small buckets [t, ss] for all t,
915
         including, magically, the bucket [ss,ss] too.
916
         This will avoid doing Real Work in subsequent Step 1's.
917
      --*/
918
1.89M
      {
919
487M
         for (j = 0; j <= 255; j++) {
920
485M
            copyStart[j] =  ftab[(j << 8) + ss]     & CLEARMASK;
921
485M
            copyEnd  [j] = (ftab[(j << 8) + ss + 1] & CLEARMASK) - 1;
922
485M
         }
923
13.7M
         for (j = ftab[ss << 8] & CLEARMASK; j < copyStart[ss]; j++) {
924
11.8M
            k = ptr[j]-1; if (k < 0) k += nblock;
925
11.8M
            c1 = block[k];
926
11.8M
            if (!bigDone[c1])
927
7.28M
               ptr[ copyStart[c1]++ ] = k;
928
11.8M
         }
929
16.5M
         for (j = (ftab[(ss+1) << 8] & CLEARMASK) - 1; j > copyEnd[ss]; j--) {
930
14.6M
            k = ptr[j]-1; if (k < 0) k += nblock;
931
14.6M
            c1 = block[k];
932
14.6M
            if (!bigDone[c1]) 
933
8.77M
               ptr[ copyEnd[c1]-- ] = k;
934
14.6M
         }
935
1.89M
      }
936
937
1.89M
      AssertH ( (copyStart[ss]-1 == copyEnd[ss])
938
1.89M
                || 
939
                /* Extremely rare case missing in bzip2-1.0.0 and 1.0.1.
940
                   Necessity for this case is demonstrated by compressing 
941
                   a sequence of approximately 48.5 million of character 
942
                   251; 1.0.0/1.0.1 will then die here. */
943
1.89M
                (copyStart[ss] == 0 && copyEnd[ss] == nblock-1),
944
1.89M
                1007 )
945
946
487M
      for (j = 0; j <= 255; j++) ftab[(j << 8) + ss] |= SETMASK;
947
948
      /*--
949
         Step 3:
950
         The [ss] big bucket is now done.  Record this fact,
951
         and update the quadrant descriptors.  Remember to
952
         update quadrants in the overshoot area too, if
953
         necessary.  The "if (i < 255)" test merely skips
954
         this updating for the last bucket processed, since
955
         updating for the last bucket is pointless.
956
957
         The quadrant array provides a way to incrementally
958
         cache sort orderings, as they appear, so as to 
959
         make subsequent comparisons in fullGtU() complete
960
         faster.  For repetitive blocks this makes a big
961
         difference (but not big enough to be able to avoid
962
         the fallback sorting mechanism, exponential radix sort).
963
964
         The precise meaning is: at all times:
965
966
            for 0 <= i < nblock and 0 <= j <= nblock
967
968
            if block[i] != block[j], 
969
970
               then the relative values of quadrant[i] and 
971
                    quadrant[j] are meaningless.
972
973
               else {
974
                  if quadrant[i] < quadrant[j]
975
                     then the string starting at i lexicographically
976
                     precedes the string starting at j
977
978
                  else if quadrant[i] > quadrant[j]
979
                     then the string starting at j lexicographically
980
                     precedes the string starting at i
981
982
                  else
983
                     the relative ordering of the strings starting
984
                     at i and j has not yet been determined.
985
               }
986
      --*/
987
1.89M
      bigDone[ss] = True;
988
989
1.89M
      if (i < 255) {
990
1.89M
         Int32 bbStart  = ftab[ss << 8] & CLEARMASK;
991
1.89M
         Int32 bbSize   = (ftab[(ss+1) << 8] & CLEARMASK) - bbStart;
992
1.89M
         Int32 shifts   = 0;
993
994
1.89M
         while ((bbSize >> shifts) > 65534) shifts++;
995
996
20.0M
         for (j = bbSize-1; j >= 0; j--) {
997
18.1M
            Int32 a2update     = ptr[bbStart + j];
998
18.1M
            UInt16 qVal        = (UInt16)(j >> shifts);
999
18.1M
            quadrant[a2update] = qVal;
1000
18.1M
            if (a2update < BZ_N_OVERSHOOT)
1001
54.4k
               quadrant[a2update + nblock] = qVal;
1002
18.1M
         }
1003
1.89M
         AssertH ( ((bbSize-1) >> shifts) <= 65535, 1002 );
1004
1.89M
      }
1005
1006
1.89M
   }
1007
1008
2.07k
   if (verb >= 4)
1009
0
      VPrintf3 ( "        %d pointers, %d sorted, %d scanned\n",
1010
2.07k
                 nblock, numQSorted, nblock - numQSorted );
1011
2.07k
}
1012
1013
#undef BIGFREQ
1014
#undef SETMASK
1015
#undef CLEARMASK
1016
1017
1018
/*---------------------------------------------*/
1019
/* Pre:
1020
      nblock > 0
1021
      arr2 exists for [0 .. nblock-1 +N_OVERSHOOT]
1022
      ((UChar*)arr2)  [0 .. nblock-1] holds block
1023
      arr1 exists for [0 .. nblock-1]
1024
1025
   Post:
1026
      ((UChar*)arr2) [0 .. nblock-1] holds block
1027
      All other areas of block destroyed
1028
      ftab [ 0 .. 65536 ] destroyed
1029
      arr1 [0 .. nblock-1] holds sorted order
1030
*/
1031
void BZ2_blockSort ( EState* s )
1032
101k
{
1033
101k
   UInt32* ptr    = s->ptr; 
1034
101k
   UChar*  block  = s->block;
1035
101k
   UInt32* ftab   = s->ftab;
1036
101k
   Int32   nblock = s->nblock;
1037
101k
   Int32   verb   = s->verbosity;
1038
101k
   Int32   wfact  = s->workFactor;
1039
101k
   UInt16* quadrant;
1040
101k
   Int32   budget;
1041
101k
   Int32   budgetInit;
1042
101k
   Int32   i;
1043
1044
101k
   if (nblock < 10000) {
1045
93.6k
      fallbackSort ( s->arr1, s->arr2, ftab, nblock, verb );
1046
93.6k
   } else {
1047
      /* Calculate the location for quadrant, remembering to get
1048
         the alignment right.  Assumes that &(block[0]) is at least
1049
         2-byte aligned -- this should be ok since block is really
1050
         the first section of arr2.
1051
      */
1052
7.52k
      i = nblock+BZ_N_OVERSHOOT;
1053
7.52k
      if (i & 1) i++;
1054
7.52k
      quadrant = (UInt16*)(&(block[i]));
1055
1056
      /* (wfact-1) / 3 puts the default-factor-30
1057
         transition point at very roughly the same place as 
1058
         with v0.1 and v0.9.0.  
1059
         Not that it particularly matters any more, since the
1060
         resulting compressed stream is now the same regardless
1061
         of whether or not we use the main sort or fallback sort.
1062
      */
1063
7.52k
      if (wfact < 1  ) wfact = 1;
1064
7.52k
      if (wfact > 100) wfact = 100;
1065
7.52k
      budgetInit = nblock * ((wfact-1) / 3);
1066
7.52k
      budget = budgetInit;
1067
1068
7.52k
      mainSort ( ptr, block, quadrant, ftab, nblock, verb, &budget );
1069
7.52k
      if (verb >= 3) 
1070
0
         VPrintf3 ( "      %d work, %d block, ratio %5.2f\n",
1071
7.52k
                    budgetInit - budget,
1072
7.52k
                    nblock, 
1073
7.52k
                    (float)(budgetInit - budget) /
1074
7.52k
                    (float)(nblock==0 ? 1 : nblock) ); 
1075
7.52k
      if (budget < 0) {
1076
5.45k
         if (verb >= 2) 
1077
0
            VPrintf0 ( "    too repetitive; using fallback"
1078
5.45k
                       " sorting algorithm\n" );
1079
5.45k
         fallbackSort ( s->arr1, s->arr2, ftab, nblock, verb );
1080
5.45k
      }
1081
7.52k
   }
1082
1083
101k
   s->origPtr = -1;
1084
86.7M
   for (i = 0; i < s->nblock; i++)
1085
86.7M
      if (ptr[i] == 0)
1086
101k
         { s->origPtr = i; break; };
1087
1088
101k
   AssertH( s->origPtr != -1, 1003 );
1089
101k
}
1090
1091
1092
/*-------------------------------------------------------------*/
1093
/*--- end                                       blocksort.c ---*/
1094
/*-------------------------------------------------------------*/