Compare commits

...
Author SHA1 Message Date
ExPikaPaka dbe342c978 Fix data races in the parallel slicing paths
Multi-material segmentation derived the output slot from range.begin() divided
by the granularity, which assumes tbb::blocked_range splits on grainsize
multiples. It bisects at midpoints, so two sub-ranges of one group get the same
slot and append into the same ExPolygons at once. With ten layers and a
granularity of two, the sub-ranges [2,3) and [3,5) collide. The granularity is
two or more whenever the shell layers are three or more, which is the default.
The loop now iterates groups, so the slot follows the element index and the
partitioner cannot cut a group in half.

CutSurface wrote std::vector<bool> elements from a parallel_for. The bits share
words across chunk boundaries, so flags were lost.

name_tbb_thread_pool_threads_set_locale() held every chunk of a parallel_for on
a condition variable until max_concurrency() of them were running. TBB does not
promise that, and there is no timeout, so the slicing thread could wait forever.
A task_scheduler_observer sets each worker up as it joins the arena, which needs
no simultaneity and covers workers the chunked version never reached.

TriangleSetSampling dereferenced upper_bound() without checking for end(), which
is reachable because the sum is a float and the keys are doubles.

TreeModelVolumes re-read getMaxCalculatedLayer() after releasing the lock, so the
layer difference could go negative and become a huge size_t. The two sibling
functions already carry the guard this adds.
2026-10-01 09:15:16 +02:00
5 changed files with 147 additions and 106 deletions
+9 -2
View File
@@ -809,13 +809,20 @@ void priv::set_skip_for_out_of_aoi(std::vector<bool> &skip_indicies,
}); // END parallel for
// inspect all triangles, when it is out of bounding box
// NOTE: std::vector<bool> is bit packed, thus setting its items from multiple threads is a
// read-modify-write race on the shared words and silently loses flags. Collect the flags into
// a byte per triangle, where the chunks do not share memory, and merge them afterwards.
std::vector<unsigned char> skip_triangle(its.indices.size(), 0);
tbb::parallel_for(tbb::blocked_range<size_t>(0, its.indices.size()),
[&its, &is_on_sides, &skip_indicies](const tbb::blocked_range<size_t> &range) {
[&its, &is_on_sides, &skip_triangle](const tbb::blocked_range<size_t> &range) {
for (size_t i = range.begin(); i < range.end(); ++i) {
if (is_all_on_one_side(its.indices[i], is_on_sides))
skip_indicies[i] = true;
skip_triangle[i] = 1;
}
}); // END parallel for
for (size_t i = 0; i < skip_triangle.size(); ++i)
if (skip_triangle[i])
skip_indicies[i] = true;
}
indexed_triangle_set Slic3r::its_mask(const indexed_triangle_set &its,
+15 -4
View File
@@ -1378,12 +1378,22 @@ static inline std::vector<std::vector<ExPolygons>> segmentation_top_and_bottom_l
return out;
};
tbb::parallel_for(tbb::blocked_range<size_t>(0, num_layers, granularity), [&granularity, &num_layers, &num_facets_states, &layer_color_stat, &top_raw, &triangles_by_color_top,
// The layers are processed in groups of "granularity" layers. A layer projects its shells up to "granularity"
// layers away, thus a group may write into the slots of its neighbor groups. The even and the odd groups
// therefore write into two disjoint halves of the output vectors (the 2nd half is offset by num_layers) and
// both halves are merged below. The group index has to be derived from the layer index and not from the extent
// of the TBB sub-range: tbb::blocked_range bisects at midpoints, thus a sub-range neither starts at a multiple
// of the grain size nor covers a whole group, and two sub-ranges of one group would append into a single
// ExPolygons concurrently. Iterating over the groups keeps every group on a single thread, in ascending order.
const size_t num_groups = (num_layers + size_t(granularity) - 1) / size_t(granularity);
tbb::parallel_for(tbb::blocked_range<size_t>(0, num_groups, 1), [&granularity, &num_layers, &num_facets_states, &layer_color_stat, &top_raw, &triangles_by_color_top,
&throw_on_cancel_callback, &input_expolygons, &bottom_raw, &triangles_by_color_bottom,
&shell_triangles_by_color_top, &shell_triangles_by_color_bottom](const tbb::blocked_range<size_t> &range) {
size_t group_idx = range.begin() / granularity;
size_t layer_idx_offset = (group_idx & 1) * num_layers;
for (size_t layer_idx = range.begin(); layer_idx < range.end(); ++ layer_idx) {
for (size_t group_idx = range.begin(); group_idx < range.end(); ++ group_idx) {
const size_t layer_idx_offset = (group_idx & 1) * num_layers;
const size_t layer_idx_begin = group_idx * size_t(granularity);
const size_t layer_idx_end = std::min(num_layers, layer_idx_begin + size_t(granularity));
for (size_t layer_idx = layer_idx_begin; layer_idx < layer_idx_end; ++ layer_idx) {
for (size_t color_idx = 0; color_idx < num_facets_states; ++color_idx) {
throw_on_cancel_callback();
LayerColorStat stat = layer_color_stat(layer_idx, color_idx);
@@ -1429,6 +1439,7 @@ static inline std::vector<std::vector<ExPolygons>> segmentation_top_and_bottom_l
}
}
}
}
});
std::vector<std::vector<ExPolygons>> triangles_by_color_merged(num_facets_states);
+15 -1
View File
@@ -424,8 +424,16 @@ void TreeModelVolumes::calculateCollision(const coord_t radius, const LayerIndex
[this](size_t i, size_t j) { return m_layer_outlines[i].second.size() < m_layer_outlines[j].second.size(); });
// Layer range for which the collisions will be calculated.
// Another thread may have advanced getMaxCalculatedLayer() past max_layer_idx after this calculation
// was requested. Bail out in that case, otherwise the layer range would be negative and allocating
// it would throw std::length_error out of a parallel task.
const LayerIndex start_layer = 1 + m_collision_cache.getMaxCalculatedLayer(radius);
if (start_layer > max_layer_idx) {
BOOST_LOG_TRIVIAL(debug) << "Requested calculation for value already calculated ?";
return;
}
LayerPolygonCache data;
data.allocate(m_collision_cache.getMaxCalculatedLayer(radius) + 1, max_layer_idx + 1);
data.allocate(start_layer, max_layer_idx + 1);
const bool calculate_placable = m_support_rests_on_model && radius == 0;
LayerPolygonCache data_placeable;
@@ -804,6 +812,12 @@ void TreeModelVolumes::calculateWallRestrictions(const std::vector<RadiusLayerPa
const coord_t radius = keys[key_idx].first;
const LayerIndex max_required_layer = keys[key_idx].second;
const coord_t min_layer_bottom = std::max(1, m_wall_restrictions_cache.getMaxCalculatedLayer(radius));
if (min_layer_bottom > max_required_layer) {
// Another thread has calculated this range in the meantime. Continuing would make
// buffer_size negative and allocating it would throw std::length_error.
BOOST_LOG_TRIVIAL(debug) << "Requested calculation for value already calculated ?";
continue;
}
const size_t buffer_size = max_required_layer + 1 - min_layer_bottom;
std::vector<Polygons> data(buffer_size, Polygons{});
std::vector<Polygons> data_min;
+48 -46
View File
@@ -12,6 +12,7 @@
#include <thread>
#include <tbb/parallel_for.h>
#include <tbb/task_arena.h>
#include <tbb/task_scheduler_observer.h>
#include "Thread.hpp"
#include "Utils.hpp"
@@ -212,54 +213,14 @@ bool is_main_thread_active()
return get_main_thread_id() == boost::this_thread::get_id();
}
// Spawn (n - 1) worker threads on Intel TBB thread pool and name them by an index and a system thread ID.
// Also it sets locale of the worker threads to "C" for the G-code generator to produce "." as a decimal separator.
void name_tbb_thread_pool_threads_set_locale()
// Name the current TBB worker thread and set its locale to "C", so that the G-code generator
// produces "." as a decimal separator. Called once per worker thread, before it runs its first task.
static void setup_tbb_worker_thread()
{
static bool initialized = false;
if (initialized)
return;
initialized = true;
// see GH issue #5661 PrusaSlicer hangs on Linux when run with non standard task affinity
// TBB will respect the task affinity mask on Linux and spawn less threads than std::thread::hardware_concurrency().
// const size_t nthreads_hw = std::thread::hardware_concurrency();
const size_t nthreads_hw = tbb::this_task_arena::max_concurrency();
size_t nthreads = nthreads_hw;
#ifdef SLIC3R_PROFILE
// Shiny profiler is not thread safe, thus disable parallelization.
disable_multi_threading();
nthreads = 1;
#endif
size_t nthreads_running(0);
std::condition_variable cv;
std::mutex cv_m;
auto master_thread_id = std::this_thread::get_id();
tbb::parallel_for(
tbb::blocked_range<size_t>(0, nthreads, 1),
[&nthreads_running, nthreads, &master_thread_id, &cv, &cv_m](const tbb::blocked_range<size_t> &range) {
assert(range.begin() + 1 == range.end());
if (std::unique_lock<std::mutex> lk(cv_m); ++nthreads_running == nthreads) {
lk.unlock();
// All threads are spinning.
// Wake them up.
cv.notify_all();
} else {
// Wait for the last thread to wake the others.
cv.wait(lk, [&nthreads_running, nthreads]{return nthreads_running == nthreads;});
}
auto thread_id = std::this_thread::get_id();
if (thread_id == master_thread_id) {
// The calling thread runs the 0'th task.
assert(range.begin() == 0);
} else {
assert(range.begin() > 0);
static std::atomic<size_t> s_worker_idx{ 0 };
std::ostringstream name;
name << "slic3r_tbb_" << range.begin();
name << "slic3r_tbb_" << (1 + s_worker_idx.fetch_add(1, std::memory_order_relaxed));
set_current_thread_name(name.str().c_str());
// Set locales of the worker thread to "C".
#ifdef _WIN32
_configthreadlocale(_ENABLE_PER_THREAD_LOCALE);
std::setlocale(LC_ALL, "C");
@@ -275,7 +236,48 @@ void name_tbb_thread_pool_threads_set_locale()
, "C", nullptr));
#endif
}
});
// Sets up the TBB worker threads of the arena of the thread, which activated the observation.
// A worker sets itself up on entry to the arena, before it executes its first task, thus unlike a barrier
// inside a parallel_for, this does not depend on TBB running any number of tasks simultaneously.
class TBBWorkerThreadSetupObserver : public tbb::task_scheduler_observer
{
public:
TBBWorkerThreadSetupObserver() { this->observe(true); }
void on_scheduler_entry(bool is_worker) override
{
// Leave the external threads (the calling / UI thread) alone, their name and locale must not be modified here.
if (! is_worker)
return;
// A worker thread enters an arena many times, while its name and locale have to be set just once.
static thread_local bool initialized = false;
if (initialized)
return;
initialized = true;
setup_tbb_worker_thread();
}
};
// Name the threads of the Intel TBB thread pool by an index and set their locale to "C"
// for the G-code generator to produce "." as a decimal separator.
// Formerly all the worker threads were caught inside a single parallel_for, which was held on a condition
// variable barrier until max_concurrency() of its chunks were running. TBB guarantees no such simultaneity,
// thus the barrier was able to block the slicing threads indefinitely. The TBB scheduler observer below
// sets each worker up on its own, thus no two chunks have to run at the same time.
void name_tbb_thread_pool_threads_set_locale()
{
#ifdef SLIC3R_PROFILE
// Shiny profiler is not thread safe, thus disable parallelization.
disable_multi_threading();
#endif
// An observer is local to the arena of the thread which activates it, thus one observer is registered
// per calling thread. Being function local and thread local, it is also initialized exactly once per
// thread without a race. It is intentionally never destroyed, as it has to stay alive as long as the
// TBB scheduler may notify it, which includes the shutdown of the process.
static thread_local tbb::task_scheduler_observer *observer = new TBBWorkerThreadSetupObserver();
(void)observer;
}
}
+8 -1
View File
@@ -28,6 +28,10 @@ TriangleSetSamples sample_its_uniform_parallel(size_t samples_count, const index
area_sum_to_triangle_idx[area_sum] = t_idx;
}
if (area_sum_to_triangle_idx.empty())
// No triangle to sample from.
return {};
std::mt19937_64 mersenne_engine { 27644437 };
// random numbers on interval [0, 1)
std::uniform_real_distribution<double> fdistribution;
@@ -50,7 +54,10 @@ TriangleSetSamples sample_its_uniform_parallel(size_t samples_count, const index
tbb::blocked_range<size_t> r) {
for (size_t s_idx = r.begin(); s_idx < r.end(); ++s_idx) {
double t_sample = random_samples[s_idx].x() * area_sum;
size_t t_idx = area_sum_to_triangle_idx.upper_bound(t_sample)->second;
// The keys of area_sum_to_triangle_idx are accumulated areas in double precision, while area_sum
// is a float, thus t_sample may reach or exceed the largest key and upper_bound() may return end().
auto t_it = area_sum_to_triangle_idx.upper_bound(t_sample);
size_t t_idx = (t_it == area_sum_to_triangle_idx.end() ? std::prev(t_it) : t_it)->second;
double sq_u = std::sqrt(random_samples[s_idx].y());
double v = random_samples[s_idx].z();