Coverage Report

Created: 2026-09-02 06:43

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/vvenc/source/Lib/Utilities/NoMallocThreadPool.cpp
Line
Count
Source
1
/* -----------------------------------------------------------------------------
2
The copyright in this software is being made available under the Clear BSD
3
License, included below. No patent rights, trademark rights and/or 
4
other Intellectual Property Rights other than the copyrights concerning 
5
the Software are granted under this license.
6
7
The Clear BSD License
8
9
Copyright (c) 2019-2026, Fraunhofer-Gesellschaft zur Förderung der angewandten Forschung e.V. & The VVenC Authors.
10
All rights reserved.
11
12
Redistribution and use in source and binary forms, with or without modification,
13
are permitted (subject to the limitations in the disclaimer below) provided that
14
the following conditions are met:
15
16
     * Redistributions of source code must retain the above copyright notice,
17
     this list of conditions and the following disclaimer.
18
19
     * Redistributions in binary form must reproduce the above copyright
20
     notice, this list of conditions and the following disclaimer in the
21
     documentation and/or other materials provided with the distribution.
22
23
     * Neither the name of the copyright holder nor the names of its
24
     contributors may be used to endorse or promote products derived from this
25
     software without specific prior written permission.
26
27
NO EXPRESS OR IMPLIED LICENSES TO ANY PARTY'S PATENT RIGHTS ARE GRANTED BY
28
THIS LICENSE. THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND
29
CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
30
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A
31
PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
32
CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
33
EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
34
PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR
35
BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER
36
IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
37
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
38
POSSIBILITY OF SUCH DAMAGE.
39
40
41
------------------------------------------------------------------------------------------- */
42
43
44
/** \file     NoMallocThreadPool.cpp
45
    \brief    thread pool
46
*/
47
48
#include "NoMallocThreadPool.h"
49
50
#ifdef HAVE_PTHREADS
51
#  include <pthread.h>
52
4.58k
#  define THREAD_MIN_STACK_SIZE 1024 * 1024
53
#endif
54
55
56
//! \ingroup Utilities
57
//! \{
58
59
namespace vvenc {
60
61
#if ENABLE_TIME_PROFILING_MT_MODE
62
thread_local std::unique_ptr<TProfiler> ptls;
63
#endif
64
65
NoMallocThreadPool::NoMallocThreadPool( int numThreads, const char * threadPoolName, const VVEncCfg* encCfg )
66
1.14k
  : m_poolName( threadPoolName )
67
1.14k
{
68
1.14k
  if( numThreads < 0 )
69
0
  {
70
0
    numThreads = std::thread::hardware_concurrency();
71
0
  }
72
73
5.73k
  for( int i = 0; i < numThreads; ++i )
74
4.58k
  {
75
4.58k
    m_threads.emplace_back( &NoMallocThreadPool::threadProc, this, i, *encCfg );
76
4.58k
  }
77
1.14k
}
78
79
NoMallocThreadPool::~NoMallocThreadPool()
80
1.14k
{
81
1.14k
  m_exitThreads = true;
82
83
1.14k
  waitForThreads();
84
1.14k
}
85
86
bool NoMallocThreadPool::processTasksOnMainThread()
87
0
{
88
0
  CHECK( m_threads.size() != 0, "should not be used with multiple threads" );
89
90
0
  bool         progress      = false;
91
0
  TaskIterator firstFailedIt = m_tasks.end();
92
0
  for( auto taskIt = findNextTask( 0, m_tasks.begin() ); taskIt.isValid(); taskIt = findNextTask( 0, taskIt ) )
93
0
  {
94
0
    const bool success = processTask( 0, *taskIt );
95
0
    progress |= success;
96
97
0
    if( taskIt == firstFailedIt )
98
0
    {
99
0
      if( success )
100
0
      {
101
        // first failed was successful -> reset
102
0
        firstFailedIt = m_tasks.end();
103
0
      }
104
0
      else if( progress )
105
0
      {
106
        // reset progress, try another round
107
0
        progress = false;
108
0
      }
109
0
      else
110
0
      {
111
        // no progress -> exit
112
0
        break;
113
0
      }
114
0
    }
115
0
    else if( !success && !firstFailedIt.isValid() )
116
0
    {
117
0
      firstFailedIt = taskIt;
118
0
    }
119
0
  }
120
121
  // return true if all done (-> false if some tasks blocked due to barriers)
122
0
  return std::all_of( m_tasks.begin(), m_tasks.end(), []( Slot& t ) { return t.state == FREE; } );
123
0
}
124
125
void NoMallocThreadPool::shutdown( bool block )
126
1.14k
{
127
1.14k
  m_exitThreads = true;
128
1.14k
  if( block )
129
1.14k
  {
130
1.14k
    waitForThreads();
131
1.14k
  }
132
1.14k
}
133
134
void NoMallocThreadPool::waitForThreads()
135
2.29k
{
136
2.29k
  for( auto& t: m_threads )
137
9.17k
  {
138
9.17k
    if( t.joinable() )
139
4.58k
      t.join();
140
9.17k
  }
141
2.29k
}
142
143
void NoMallocThreadPool::threadProc( int threadId, const VVEncCfg& encCfg )
144
4.58k
{
145
4.58k
#if __linux
146
4.58k
  if( !m_poolName.empty() )
147
4.58k
  {
148
4.58k
    std::string threadName( m_poolName + std::to_string( threadId ) );
149
4.58k
    pthread_setname_np( pthread_self(), threadName.c_str() );
150
4.58k
  }
151
4.58k
#endif
152
#if ENABLE_TIME_PROFILING_MT_MODE
153
  ptls.reset( timeProfilerCreate( encCfg ) );
154
  {
155
    std::unique_lock< std::mutex > lock( m_nextFillSlotMutex );
156
    TProfiler *tp = ptls.get();
157
    profilers.push_back( tp );
158
  }
159
#endif
160
161
4.58k
  auto nextTaskIt = m_tasks.begin();
162
29.8k
  while( !m_exitThreads )
163
29.8k
  {
164
29.8k
    auto taskIt = findNextTask( threadId, nextTaskIt );
165
29.8k
    if( !taskIt.isValid() )
166
16.8k
    {
167
16.8k
      std::unique_lock<std::mutex> l( m_idleMutex, std::defer_lock );
168
169
16.8k
      ITT_TASKSTART( itt_domain_thrd, itt_handle_TPspinWait );
170
16.8k
      m_waitingThreads.fetch_add( 1, std::memory_order_relaxed );
171
16.8k
      const auto startWait = std::chrono::steady_clock::now();
172
84.2M
      while( !m_exitThreads )
173
84.2M
      {
174
84.2M
        taskIt = findNextTask( threadId, nextTaskIt );
175
84.2M
        if( taskIt.isValid() || m_exitThreads )
176
15.4k
        {
177
15.4k
          break;
178
15.4k
        }
179
180
84.2M
        if( !l.owns_lock()
181
686k
            && m_waitingThreads.load( std::memory_order_relaxed ) > 1
182
606k
            && ( BUSY_WAIT_TIME.count() == 0 || std::chrono::steady_clock::now() - startWait > BUSY_WAIT_TIME )
183
9.99k
            && !m_exitThreads )
184
9.98k
        {
185
9.98k
          ITT_TASKSTART(itt_domain_thrd, itt_handle_TPblocked);
186
9.98k
          l.lock();
187
9.98k
          ITT_TASKEND(itt_domain_thrd, itt_handle_TPblocked);
188
9.98k
        }
189
84.1M
        else
190
84.1M
        {
191
84.1M
          std::this_thread::yield();
192
84.1M
        }
193
84.2M
      }
194
16.8k
      m_waitingThreads.fetch_sub( 1, std::memory_order_relaxed );
195
16.8k
      ITT_TASKEND( itt_domain_thrd, itt_handle_TPspinWait );
196
16.8k
    }
197
29.8k
    if( m_exitThreads )
198
4.57k
    {
199
4.57k
      return;
200
4.57k
    }
201
202
25.2k
    processTask( threadId, *taskIt );
203
204
25.2k
    nextTaskIt = taskIt;
205
25.2k
    nextTaskIt.incWrap();
206
25.2k
  }
207
4.58k
}
208
209
NoMallocThreadPool::TaskIterator NoMallocThreadPool::findNextTask( int threadId, TaskIterator startSearch )
210
84.2M
{
211
84.2M
  if( !startSearch.isValid() )
212
0
  {
213
0
    startSearch = m_tasks.begin();
214
0
  }
215
84.2M
  bool first = true;
216
10.8G
  for( auto it = startSearch; it != startSearch || first; it.incWrap() )
217
10.7G
  {
218
#if ENABLE_VALGRIND_CODE
219
    MutexLock lock( m_extraMutex );
220
#endif
221
222
10.7G
    first = false;
223
224
10.7G
    Slot& t = *it;
225
10.7G
    auto expected = WAITING;
226
10.7G
    if( t.state.load( std::memory_order_relaxed ) == WAITING && t.state.compare_exchange_strong( expected, RUNNING ) )
227
132M
    {
228
132M
      if( !t.barriers.empty() )
229
39.1M
      {
230
39.1M
        if( std::any_of( t.barriers.cbegin(), t.barriers.cend(), []( const Barrier* b ) { return b && b->isBlocked(); } ) )
231
39.1M
        {
232
          // reschedule
233
39.1M
          t.state.store( WAITING );
234
39.1M
          continue;
235
39.1M
        }
236
1.14k
        t.barriers.clear();   // clear barriers, so we don't need to check them on the next try (we assume they won't get locked again)
237
1.14k
      }
238
92.9M
      if( t.readyCheck && t.readyCheck( threadId, t.param ) == false )
239
92.9M
      {
240
        // reschedule
241
92.9M
        t.state.store( WAITING );
242
92.9M
        continue;
243
92.9M
      }
244
245
34.6k
      return it;
246
92.9M
    }
247
10.7G
  }
248
84.2M
  return {};
249
84.2M
}
250
251
bool NoMallocThreadPool::processTask( int threadId, NoMallocThreadPool::Slot& task )
252
25.2k
{
253
25.2k
  const bool success = task.func( threadId, task.param );
254
#if ENABLE_VALGRIND_CODE
255
  MutexLock lock( m_extraMutex );
256
#endif
257
25.2k
  if( !success )
258
20.4k
  {
259
20.4k
    task.state = WAITING;
260
20.4k
    return false;
261
20.4k
  }
262
263
4.72k
  if( task.done != nullptr )
264
0
  {
265
0
    task.done->unlock();
266
0
  }
267
4.72k
  if( task.counter != nullptr )
268
3.59k
  {
269
3.59k
    --(*task.counter);
270
3.59k
  }
271
272
4.72k
  task.state = FREE;
273
274
4.72k
  return true;
275
25.2k
}
276
277
#ifdef HAVE_PTHREADS
278
279
template<class TFunc, class... TArgs>
280
NoMallocThreadPool::PThread::PThread( TFunc&& func, TArgs&&... args )
281
4.58k
{
282
4.58k
  using WrappedCall     = std::function<void()>;
283
4.58k
  std::unique_ptr<WrappedCall> call = std::make_unique<WrappedCall>( std::bind( func, args... ) );
284
285
4.58k
  using PThreadsStartFn = void* (*) ( void* );
286
4.58k
  PThreadsStartFn threadFn = []( void* p ) -> void*
287
4.58k
  {
288
4.58k
    std::unique_ptr<WrappedCall> call( static_cast<WrappedCall*>( p ) );
289
290
4.58k
    ( *call )();
291
292
4.58k
    return nullptr;
293
4.58k
  };
294
295
4.58k
  pthread_attr_t attr;
296
4.58k
  int ret = pthread_attr_init( &attr );
297
4.58k
  CHECK( ret != 0, "pthread_attr_init() failed" );
298
299
4.58k
  try
300
4.58k
  {
301
4.58k
    size_t currStackSize = 0;
302
4.58k
    ret = pthread_attr_getstacksize( &attr, &currStackSize );
303
4.58k
    CHECK( ret != 0, "pthread_attr_getstacksize() failed" );
304
305
4.58k
    if( currStackSize < THREAD_MIN_STACK_SIZE )
306
0
    {
307
0
      ret = pthread_attr_setstacksize( &attr, THREAD_MIN_STACK_SIZE );
308
0
      CHECK( ret != 0, "pthread_attr_setstacksize() failed" );
309
310
0
#  if defined( _DEBUG ) && !defined( __MINGW32__ ) && !defined( __MINGW64__ )
311
0
      ret = pthread_attr_setguardsize( &attr, 1024 * 1024 );   // set stack guard size to 1MB to more reliably deteck stack overflows
312
0
      CHECK( ret != 0, "pthread_attr_setguardsize() failed" );
313
0
#  endif
314
0
    }
315
4.58k
    m_joinable = 0 == pthread_create( &m_id, &attr, threadFn, call.get() );
316
4.58k
    CHECK( !m_joinable, "pthread_create() faild" );
317
318
4.58k
    call.release();   // will now be freed by the thread
319
320
4.58k
    pthread_attr_destroy( &attr );
321
4.58k
  }
322
4.58k
  catch( ... )
323
4.58k
  {
324
0
    pthread_attr_destroy( &attr );
325
0
    throw;
326
0
  }
327
4.58k
}
328
329
NoMallocThreadPool::PThread& NoMallocThreadPool::PThread::operator=( PThread&& other )
330
3.44k
{
331
3.44k
  m_id             = other.m_id;
332
3.44k
  m_joinable       = other.m_joinable;
333
3.44k
  other.m_id       = 0;
334
3.44k
  other.m_joinable = false;
335
3.44k
  return *this;
336
3.44k
}
337
338
void NoMallocThreadPool::PThread::join()
339
4.58k
{
340
4.58k
  if( m_joinable )
341
4.58k
  {
342
4.58k
    m_joinable = false;
343
4.58k
    pthread_join( m_id, nullptr );
344
4.58k
  }
345
4.58k
}
346
347
#endif   // HAVE_PTHREADS
348
349
} // namespace vvenc
350
351
//! \}
352