diff --git a/src/libslic3r/CutSurface.cpp b/src/libslic3r/CutSurface.cpp index 606a5b8aaa..5e5698c1b2 100644 --- a/src/libslic3r/CutSurface.cpp +++ b/src/libslic3r/CutSurface.cpp @@ -809,13 +809,20 @@ void priv::set_skip_for_out_of_aoi(std::vector &skip_indicies, }); // END parallel for // inspect all triangles, when it is out of bounding box + // NOTE: std::vector is bit packed, thus setting its items from multiple threads is a + // read-modify-write race on the shared words and silently loses flags. Collect the flags into + // a byte per triangle, where the chunks do not share memory, and merge them afterwards. + std::vector skip_triangle(its.indices.size(), 0); tbb::parallel_for(tbb::blocked_range(0, its.indices.size()), - [&its, &is_on_sides, &skip_indicies](const tbb::blocked_range &range) { + [&its, &is_on_sides, &skip_triangle](const tbb::blocked_range &range) { for (size_t i = range.begin(); i < range.end(); ++i) { if (is_all_on_one_side(its.indices[i], is_on_sides)) - skip_indicies[i] = true; + skip_triangle[i] = 1; } }); // END parallel for + for (size_t i = 0; i < skip_triangle.size(); ++i) + if (skip_triangle[i]) + skip_indicies[i] = true; } indexed_triangle_set Slic3r::its_mask(const indexed_triangle_set &its, diff --git a/src/libslic3r/MultiMaterialSegmentation.cpp b/src/libslic3r/MultiMaterialSegmentation.cpp index 6f80f7b759..6307ac2080 100644 --- a/src/libslic3r/MultiMaterialSegmentation.cpp +++ b/src/libslic3r/MultiMaterialSegmentation.cpp @@ -1378,55 +1378,66 @@ static inline std::vector> segmentation_top_and_bottom_l return out; }; - tbb::parallel_for(tbb::blocked_range(0, num_layers, granularity), [&granularity, &num_layers, &num_facets_states, &layer_color_stat, &top_raw, &triangles_by_color_top, - &throw_on_cancel_callback, &input_expolygons, &bottom_raw, &triangles_by_color_bottom, - &shell_triangles_by_color_top, &shell_triangles_by_color_bottom](const tbb::blocked_range &range) { - size_t group_idx = range.begin() / granularity; - size_t layer_idx_offset = (group_idx & 1) * num_layers; - for (size_t layer_idx = range.begin(); layer_idx < range.end(); ++ layer_idx) { - for (size_t color_idx = 0; color_idx < num_facets_states; ++color_idx) { - throw_on_cancel_callback(); - LayerColorStat stat = layer_color_stat(layer_idx, color_idx); - if (std::vector &top = top_raw[color_idx]; ! top.empty() && ! top[layer_idx].empty()) - if (ExPolygons top_ex = union_ex(top[layer_idx]); ! top_ex.empty()) { - // Clean up thin projections. They are not printable anyways. - top_ex = opening_ex(top_ex, stat.small_region_threshold); - if (! top_ex.empty()) { - append(triangles_by_color_top[color_idx][layer_idx + layer_idx_offset], top_ex); - float offset = 0.f; - ExPolygons layer_slices_trimmed = input_expolygons[layer_idx]; - for (int last_idx = int(layer_idx) - 1; last_idx > std::max(int(layer_idx - stat.top_shell_layers), int(0)); --last_idx) { - //BBS: offset width should be 2*spacing to avoid too narrow area which has overlap of wall line - //offset -= stat.extrusion_width ; - offset -= (stat.extrusion_spacing + stat.extrusion_width); - layer_slices_trimmed = intersection_ex(layer_slices_trimmed, input_expolygons[last_idx]); - ExPolygons last = opening_ex(intersection_ex(top_ex, offset_ex(layer_slices_trimmed, offset)), stat.small_region_threshold); - if (last.empty()) - break; - append(shell_triangles_by_color_top[color_idx][last_idx + layer_idx_offset], std::move(last)); + // The layers are processed in groups of "granularity" layers. A layer projects its shells up to "granularity" + // layers away, thus a group may write into the slots of its neighbor groups. The even and the odd groups + // therefore write into two disjoint halves of the output vectors (the 2nd half is offset by num_layers) and + // both halves are merged below. The group index has to be derived from the layer index and not from the extent + // of the TBB sub-range: tbb::blocked_range bisects at midpoints, thus a sub-range neither starts at a multiple + // of the grain size nor covers a whole group, and two sub-ranges of one group would append into a single + // ExPolygons concurrently. Iterating over the groups keeps every group on a single thread, in ascending order. + const size_t num_groups = (num_layers + size_t(granularity) - 1) / size_t(granularity); + tbb::parallel_for(tbb::blocked_range(0, num_groups, 1), [&granularity, &num_layers, &num_facets_states, &layer_color_stat, &top_raw, &triangles_by_color_top, + &throw_on_cancel_callback, &input_expolygons, &bottom_raw, &triangles_by_color_bottom, + &shell_triangles_by_color_top, &shell_triangles_by_color_bottom](const tbb::blocked_range &range) { + for (size_t group_idx = range.begin(); group_idx < range.end(); ++ group_idx) { + const size_t layer_idx_offset = (group_idx & 1) * num_layers; + const size_t layer_idx_begin = group_idx * size_t(granularity); + const size_t layer_idx_end = std::min(num_layers, layer_idx_begin + size_t(granularity)); + for (size_t layer_idx = layer_idx_begin; layer_idx < layer_idx_end; ++ layer_idx) { + for (size_t color_idx = 0; color_idx < num_facets_states; ++color_idx) { + throw_on_cancel_callback(); + LayerColorStat stat = layer_color_stat(layer_idx, color_idx); + if (std::vector &top = top_raw[color_idx]; ! top.empty() && ! top[layer_idx].empty()) + if (ExPolygons top_ex = union_ex(top[layer_idx]); ! top_ex.empty()) { + // Clean up thin projections. They are not printable anyways. + top_ex = opening_ex(top_ex, stat.small_region_threshold); + if (! top_ex.empty()) { + append(triangles_by_color_top[color_idx][layer_idx + layer_idx_offset], top_ex); + float offset = 0.f; + ExPolygons layer_slices_trimmed = input_expolygons[layer_idx]; + for (int last_idx = int(layer_idx) - 1; last_idx > std::max(int(layer_idx - stat.top_shell_layers), int(0)); --last_idx) { + //BBS: offset width should be 2*spacing to avoid too narrow area which has overlap of wall line + //offset -= stat.extrusion_width ; + offset -= (stat.extrusion_spacing + stat.extrusion_width); + layer_slices_trimmed = intersection_ex(layer_slices_trimmed, input_expolygons[last_idx]); + ExPolygons last = opening_ex(intersection_ex(top_ex, offset_ex(layer_slices_trimmed, offset)), stat.small_region_threshold); + if (last.empty()) + break; + append(shell_triangles_by_color_top[color_idx][last_idx + layer_idx_offset], std::move(last)); + } } } - } - if (std::vector &bottom = bottom_raw[color_idx]; ! bottom.empty() && ! bottom[layer_idx].empty()) - if (ExPolygons bottom_ex = union_ex(bottom[layer_idx]); ! bottom_ex.empty()) { - // Clean up thin projections. They are not printable anyways. - bottom_ex = opening_ex(bottom_ex, stat.small_region_threshold); - if (! bottom_ex.empty()) { - append(triangles_by_color_bottom[color_idx][layer_idx + layer_idx_offset], bottom_ex); - float offset = 0.f; - ExPolygons layer_slices_trimmed = input_expolygons[layer_idx]; - for (size_t last_idx = layer_idx + 1; last_idx < std::min(layer_idx + stat.bottom_shell_layers, num_layers); ++last_idx) { - //BBS: offset width should be 2*spacing to avoid too narrow area which has overlap of wall line - //offset -= stat.extrusion_width; - offset -= (stat.extrusion_spacing + stat.extrusion_width); - layer_slices_trimmed = intersection_ex(layer_slices_trimmed, input_expolygons[last_idx]); - ExPolygons last = opening_ex(intersection_ex(bottom_ex, offset_ex(layer_slices_trimmed, offset)), stat.small_region_threshold); - if (last.empty()) - break; - append(shell_triangles_by_color_bottom[color_idx][last_idx + layer_idx_offset], std::move(last)); + if (std::vector &bottom = bottom_raw[color_idx]; ! bottom.empty() && ! bottom[layer_idx].empty()) + if (ExPolygons bottom_ex = union_ex(bottom[layer_idx]); ! bottom_ex.empty()) { + // Clean up thin projections. They are not printable anyways. + bottom_ex = opening_ex(bottom_ex, stat.small_region_threshold); + if (! bottom_ex.empty()) { + append(triangles_by_color_bottom[color_idx][layer_idx + layer_idx_offset], bottom_ex); + float offset = 0.f; + ExPolygons layer_slices_trimmed = input_expolygons[layer_idx]; + for (size_t last_idx = layer_idx + 1; last_idx < std::min(layer_idx + stat.bottom_shell_layers, num_layers); ++last_idx) { + //BBS: offset width should be 2*spacing to avoid too narrow area which has overlap of wall line + //offset -= stat.extrusion_width; + offset -= (stat.extrusion_spacing + stat.extrusion_width); + layer_slices_trimmed = intersection_ex(layer_slices_trimmed, input_expolygons[last_idx]); + ExPolygons last = opening_ex(intersection_ex(bottom_ex, offset_ex(layer_slices_trimmed, offset)), stat.small_region_threshold); + if (last.empty()) + break; + append(shell_triangles_by_color_bottom[color_idx][last_idx + layer_idx_offset], std::move(last)); + } } } - } + } } } }); diff --git a/src/libslic3r/Support/TreeModelVolumes.cpp b/src/libslic3r/Support/TreeModelVolumes.cpp index 70c33aab0d..2990d698c2 100644 --- a/src/libslic3r/Support/TreeModelVolumes.cpp +++ b/src/libslic3r/Support/TreeModelVolumes.cpp @@ -424,8 +424,16 @@ void TreeModelVolumes::calculateCollision(const coord_t radius, const LayerIndex [this](size_t i, size_t j) { return m_layer_outlines[i].second.size() < m_layer_outlines[j].second.size(); }); // Layer range for which the collisions will be calculated. + // Another thread may have advanced getMaxCalculatedLayer() past max_layer_idx after this calculation + // was requested. Bail out in that case, otherwise the layer range would be negative and allocating + // it would throw std::length_error out of a parallel task. + const LayerIndex start_layer = 1 + m_collision_cache.getMaxCalculatedLayer(radius); + if (start_layer > max_layer_idx) { + BOOST_LOG_TRIVIAL(debug) << "Requested calculation for value already calculated ?"; + return; + } LayerPolygonCache data; - data.allocate(m_collision_cache.getMaxCalculatedLayer(radius) + 1, max_layer_idx + 1); + data.allocate(start_layer, max_layer_idx + 1); const bool calculate_placable = m_support_rests_on_model && radius == 0; LayerPolygonCache data_placeable; @@ -804,6 +812,12 @@ void TreeModelVolumes::calculateWallRestrictions(const std::vector max_required_layer) { + // Another thread has calculated this range in the meantime. Continuing would make + // buffer_size negative and allocating it would throw std::length_error. + BOOST_LOG_TRIVIAL(debug) << "Requested calculation for value already calculated ?"; + continue; + } const size_t buffer_size = max_required_layer + 1 - min_layer_bottom; std::vector data(buffer_size, Polygons{}); std::vector data_min; diff --git a/src/libslic3r/Thread.cpp b/src/libslic3r/Thread.cpp index edd7c2a3d0..24b3286175 100644 --- a/src/libslic3r/Thread.cpp +++ b/src/libslic3r/Thread.cpp @@ -12,6 +12,7 @@ #include #include #include +#include #include "Thread.hpp" #include "Utils.hpp" @@ -212,70 +213,71 @@ bool is_main_thread_active() return get_main_thread_id() == boost::this_thread::get_id(); } -// Spawn (n - 1) worker threads on Intel TBB thread pool and name them by an index and a system thread ID. -// Also it sets locale of the worker threads to "C" for the G-code generator to produce "." as a decimal separator. +// Name the current TBB worker thread and set its locale to "C", so that the G-code generator +// produces "." as a decimal separator. Called once per worker thread, before it runs its first task. +static void setup_tbb_worker_thread() +{ + static std::atomic s_worker_idx{ 0 }; + std::ostringstream name; + name << "slic3r_tbb_" << (1 + s_worker_idx.fetch_add(1, std::memory_order_relaxed)); + set_current_thread_name(name.str().c_str()); +#ifdef _WIN32 + _configthreadlocale(_ENABLE_PER_THREAD_LOCALE); + std::setlocale(LC_ALL, "C"); +#else + // We are leaking some memory here, because the newlocale() produced memory will never be released. + // This is not a problem though, as there will be a maximum one worker thread created per physical thread. + uselocale(newlocale( +#ifdef __APPLE__ + LC_ALL_MASK +#else // some Unix / Linux / BSD + LC_ALL +#endif + , "C", nullptr)); +#endif +} + +// Sets up the TBB worker threads of the arena of the thread, which activated the observation. +// A worker sets itself up on entry to the arena, before it executes its first task, thus unlike a barrier +// inside a parallel_for, this does not depend on TBB running any number of tasks simultaneously. +class TBBWorkerThreadSetupObserver : public tbb::task_scheduler_observer +{ +public: + TBBWorkerThreadSetupObserver() { this->observe(true); } + + void on_scheduler_entry(bool is_worker) override + { + // Leave the external threads (the calling / UI thread) alone, their name and locale must not be modified here. + if (! is_worker) + return; + // A worker thread enters an arena many times, while its name and locale have to be set just once. + static thread_local bool initialized = false; + if (initialized) + return; + initialized = true; + setup_tbb_worker_thread(); + } +}; + +// Name the threads of the Intel TBB thread pool by an index and set their locale to "C" +// for the G-code generator to produce "." as a decimal separator. +// Formerly all the worker threads were caught inside a single parallel_for, which was held on a condition +// variable barrier until max_concurrency() of its chunks were running. TBB guarantees no such simultaneity, +// thus the barrier was able to block the slicing threads indefinitely. The TBB scheduler observer below +// sets each worker up on its own, thus no two chunks have to run at the same time. void name_tbb_thread_pool_threads_set_locale() { - static bool initialized = false; - if (initialized) - return; - initialized = true; - - // see GH issue #5661 PrusaSlicer hangs on Linux when run with non standard task affinity - // TBB will respect the task affinity mask on Linux and spawn less threads than std::thread::hardware_concurrency(). -// const size_t nthreads_hw = std::thread::hardware_concurrency(); - const size_t nthreads_hw = tbb::this_task_arena::max_concurrency(); - size_t nthreads = nthreads_hw; - #ifdef SLIC3R_PROFILE // Shiny profiler is not thread safe, thus disable parallelization. disable_multi_threading(); - nthreads = 1; #endif - size_t nthreads_running(0); - std::condition_variable cv; - std::mutex cv_m; - auto master_thread_id = std::this_thread::get_id(); - tbb::parallel_for( - tbb::blocked_range(0, nthreads, 1), - [&nthreads_running, nthreads, &master_thread_id, &cv, &cv_m](const tbb::blocked_range &range) { - assert(range.begin() + 1 == range.end()); - if (std::unique_lock lk(cv_m); ++nthreads_running == nthreads) { - lk.unlock(); - // All threads are spinning. - // Wake them up. - cv.notify_all(); - } else { - // Wait for the last thread to wake the others. - cv.wait(lk, [&nthreads_running, nthreads]{return nthreads_running == nthreads;}); - } - auto thread_id = std::this_thread::get_id(); - if (thread_id == master_thread_id) { - // The calling thread runs the 0'th task. - assert(range.begin() == 0); - } else { - assert(range.begin() > 0); - std::ostringstream name; - name << "slic3r_tbb_" << range.begin(); - set_current_thread_name(name.str().c_str()); - // Set locales of the worker thread to "C". -#ifdef _WIN32 - _configthreadlocale(_ENABLE_PER_THREAD_LOCALE); - std::setlocale(LC_ALL, "C"); -#else - // We are leaking some memory here, because the newlocale() produced memory will never be released. - // This is not a problem though, as there will be a maximum one worker thread created per physical thread. - uselocale(newlocale( -#ifdef __APPLE__ - LC_ALL_MASK -#else // some Unix / Linux / BSD - LC_ALL -#endif - , "C", nullptr)); -#endif - } - }); + // An observer is local to the arena of the thread which activates it, thus one observer is registered + // per calling thread. Being function local and thread local, it is also initialized exactly once per + // thread without a race. It is intentionally never destroyed, as it has to stay alive as long as the + // TBB scheduler may notify it, which includes the shutdown of the process. + static thread_local tbb::task_scheduler_observer *observer = new TBBWorkerThreadSetupObserver(); + (void)observer; } } diff --git a/src/libslic3r/TriangleSetSampling.cpp b/src/libslic3r/TriangleSetSampling.cpp index bb03ff6d75..fb7cce9121 100644 --- a/src/libslic3r/TriangleSetSampling.cpp +++ b/src/libslic3r/TriangleSetSampling.cpp @@ -28,6 +28,10 @@ TriangleSetSamples sample_its_uniform_parallel(size_t samples_count, const index area_sum_to_triangle_idx[area_sum] = t_idx; } + if (area_sum_to_triangle_idx.empty()) + // No triangle to sample from. + return {}; + std::mt19937_64 mersenne_engine { 27644437 }; // random numbers on interval [0, 1) std::uniform_real_distribution fdistribution; @@ -50,7 +54,10 @@ TriangleSetSamples sample_its_uniform_parallel(size_t samples_count, const index tbb::blocked_range r) { for (size_t s_idx = r.begin(); s_idx < r.end(); ++s_idx) { double t_sample = random_samples[s_idx].x() * area_sum; - size_t t_idx = area_sum_to_triangle_idx.upper_bound(t_sample)->second; + // The keys of area_sum_to_triangle_idx are accumulated areas in double precision, while area_sum + // is a float, thus t_sample may reach or exceed the largest key and upper_bound() may return end(). + auto t_it = area_sum_to_triangle_idx.upper_bound(t_sample); + size_t t_idx = (t_it == area_sum_to_triangle_idx.end() ? std::prev(t_it) : t_it)->second; double sq_u = std::sqrt(random_samples[s_idx].y()); double v = random_samples[s_idx].z();