Compare commits

..
Author SHA1 Message Date
ExPikaPaka 391c53a8da Take the print config by reference while simplifying extrusions
simplify_path, simplify_multi_path and simplify_loop each copied the whole
PrintConfig, 10520 bytes with around 200 heap owning members, once per extrusion
entity, from inside the parallel_for over all layers. LayerRegion.cpp already
binds a reference a few hundred lines above. Three more sites in Layer.cpp do the
same thing.

The tree support changes in the same area: draw_circles() dropped layers from an
owning vector without deleting them, so every re-slice leaked the layers above
the topmost overhang. std::stable_partition rather than std::remove_if, because
remove_if leaves moved-from duplicates of kept pointers in the tail and deleting
those would be a double free. The collision and avoidance accessors returned
ExPolygons by value out of a cache that hands them out by reference, several
times per node per layer. The preview cache is released when the support step
finishes instead of at the start of the next slice; nothing reads it in between.

set_shared_object() now releases the layers the object still owns before it
starts aliasing another object's. Both clear functions are no-ops once the alias
is set, so this cannot free the sharee's layers.

PrintObjectSlice.cpp: eps was a function local static inside a parallel_for, so
slice_closing_radius was latched once per process. This one changes output for
projects that override it per object, or on a re-slice after editing it. The old
behaviour was undefined: whichever object won the race set it for the session.
2026-10-01 09:15:15 +02:00
12 changed files with 143 additions and 197 deletions
+2 -9
View File
@@ -809,20 +809,13 @@ void priv::set_skip_for_out_of_aoi(std::vector<bool> &skip_indicies,
}); // END parallel for
// inspect all triangles, when it is out of bounding box
// NOTE: std::vector<bool> is bit packed, thus setting its items from multiple threads is a
// read-modify-write race on the shared words and silently loses flags. Collect the flags into
// a byte per triangle, where the chunks do not share memory, and merge them afterwards.
std::vector<unsigned char> skip_triangle(its.indices.size(), 0);
tbb::parallel_for(tbb::blocked_range<size_t>(0, its.indices.size()),
[&its, &is_on_sides, &skip_triangle](const tbb::blocked_range<size_t> &range) {
[&its, &is_on_sides, &skip_indicies](const tbb::blocked_range<size_t> &range) {
for (size_t i = range.begin(); i < range.end(); ++i) {
if (is_all_on_one_side(its.indices[i], is_on_sides))
skip_triangle[i] = 1;
skip_indicies[i] = true;
}
}); // END parallel for
for (size_t i = 0; i < skip_triangle.size(); ++i)
if (skip_triangle[i])
skip_indicies[i] = true;
}
indexed_triangle_set Slic3r::its_mask(const indexed_triangle_set &its,
+3 -3
View File
@@ -360,7 +360,7 @@ void Layer::simplify_support_entity_collection(ExtrusionEntityCollection* entity
//BBS: method to simplify support path
void Layer::simplify_support_path(ExtrusionPath * path)
{
const auto print_config = this->object()->print()->config();
const PrintConfig &print_config = this->object()->print()->config();
const bool spiral_mode = print_config.spiral_mode;
const bool enable_arc_fitting = print_config.enable_arc_fitting;
const auto scaled_resolution = scaled<double>(print_config.resolution.value);
@@ -375,7 +375,7 @@ void Layer::simplify_support_path(ExtrusionPath * path)
//BBS: method to simplify support path
void Layer::simplify_support_multi_path(ExtrusionMultiPath* multipath)
{
const auto print_config = this->object()->print()->config();
const PrintConfig &print_config = this->object()->print()->config();
const bool spiral_mode = print_config.spiral_mode;
const bool enable_arc_fitting = print_config.enable_arc_fitting;
const auto scaled_resolution = scaled<double>(print_config.resolution.value);
@@ -392,7 +392,7 @@ void Layer::simplify_support_multi_path(ExtrusionMultiPath* multipath)
//BBS: method to simplify support path
void Layer::simplify_support_loop(ExtrusionLoop* loop)
{
const auto print_config = this->object()->print()->config();
const PrintConfig &print_config = this->object()->print()->config();
const bool spiral_mode = print_config.spiral_mode;
const bool enable_arc_fitting = print_config.enable_arc_fitting;
const auto scaled_resolution = scaled<double>(print_config.resolution.value);
+3 -3
View File
@@ -1074,7 +1074,7 @@ void LayerRegion::simplify_entity_collection(ExtrusionEntityCollection* entity_c
void LayerRegion::simplify_path(ExtrusionPath* path)
{
const auto print_config = this->layer()->object()->print()->config();
const PrintConfig &print_config = this->layer()->object()->print()->config();
const bool spiral_mode = print_config.spiral_mode;
const bool enable_arc_fitting = print_config.enable_arc_fitting;
const auto scaled_resolution = scaled<double>(print_config.resolution.value);
@@ -1092,7 +1092,7 @@ void LayerRegion::simplify_path(ExtrusionPath* path)
void LayerRegion::simplify_multi_path(ExtrusionMultiPath* multipath)
{
const auto print_config = this->layer()->object()->print()->config();
const PrintConfig &print_config = this->layer()->object()->print()->config();
const bool spiral_mode = print_config.spiral_mode;
const bool enable_arc_fitting = print_config.enable_arc_fitting;
const auto scaled_resolution = scaled<double>(print_config.resolution.value);
@@ -1112,7 +1112,7 @@ void LayerRegion::simplify_multi_path(ExtrusionMultiPath* multipath)
void LayerRegion::simplify_loop(ExtrusionLoop* loop)
{
const auto print_config = this->layer()->object()->print()->config();
const PrintConfig &print_config = this->layer()->object()->print()->config();
const bool spiral_mode = print_config.spiral_mode;
const bool enable_arc_fitting = print_config.enable_arc_fitting;
const auto scaled_resolution = scaled<double>(print_config.resolution.value);
+45 -56
View File
@@ -1378,66 +1378,55 @@ static inline std::vector<std::vector<ExPolygons>> segmentation_top_and_bottom_l
return out;
};
// The layers are processed in groups of "granularity" layers. A layer projects its shells up to "granularity"
// layers away, thus a group may write into the slots of its neighbor groups. The even and the odd groups
// therefore write into two disjoint halves of the output vectors (the 2nd half is offset by num_layers) and
// both halves are merged below. The group index has to be derived from the layer index and not from the extent
// of the TBB sub-range: tbb::blocked_range bisects at midpoints, thus a sub-range neither starts at a multiple
// of the grain size nor covers a whole group, and two sub-ranges of one group would append into a single
// ExPolygons concurrently. Iterating over the groups keeps every group on a single thread, in ascending order.
const size_t num_groups = (num_layers + size_t(granularity) - 1) / size_t(granularity);
tbb::parallel_for(tbb::blocked_range<size_t>(0, num_groups, 1), [&granularity, &num_layers, &num_facets_states, &layer_color_stat, &top_raw, &triangles_by_color_top,
&throw_on_cancel_callback, &input_expolygons, &bottom_raw, &triangles_by_color_bottom,
&shell_triangles_by_color_top, &shell_triangles_by_color_bottom](const tbb::blocked_range<size_t> &range) {
for (size_t group_idx = range.begin(); group_idx < range.end(); ++ group_idx) {
const size_t layer_idx_offset = (group_idx & 1) * num_layers;
const size_t layer_idx_begin = group_idx * size_t(granularity);
const size_t layer_idx_end = std::min(num_layers, layer_idx_begin + size_t(granularity));
for (size_t layer_idx = layer_idx_begin; layer_idx < layer_idx_end; ++ layer_idx) {
for (size_t color_idx = 0; color_idx < num_facets_states; ++color_idx) {
throw_on_cancel_callback();
LayerColorStat stat = layer_color_stat(layer_idx, color_idx);
if (std::vector<Polygons> &top = top_raw[color_idx]; ! top.empty() && ! top[layer_idx].empty())
if (ExPolygons top_ex = union_ex(top[layer_idx]); ! top_ex.empty()) {
// Clean up thin projections. They are not printable anyways.
top_ex = opening_ex(top_ex, stat.small_region_threshold);
if (! top_ex.empty()) {
append(triangles_by_color_top[color_idx][layer_idx + layer_idx_offset], top_ex);
float offset = 0.f;
ExPolygons layer_slices_trimmed = input_expolygons[layer_idx];
for (int last_idx = int(layer_idx) - 1; last_idx > std::max(int(layer_idx - stat.top_shell_layers), int(0)); --last_idx) {
//BBS: offset width should be 2*spacing to avoid too narrow area which has overlap of wall line
//offset -= stat.extrusion_width ;
offset -= (stat.extrusion_spacing + stat.extrusion_width);
layer_slices_trimmed = intersection_ex(layer_slices_trimmed, input_expolygons[last_idx]);
ExPolygons last = opening_ex(intersection_ex(top_ex, offset_ex(layer_slices_trimmed, offset)), stat.small_region_threshold);
if (last.empty())
break;
append(shell_triangles_by_color_top[color_idx][last_idx + layer_idx_offset], std::move(last));
}
tbb::parallel_for(tbb::blocked_range<size_t>(0, num_layers, granularity), [&granularity, &num_layers, &num_facets_states, &layer_color_stat, &top_raw, &triangles_by_color_top,
&throw_on_cancel_callback, &input_expolygons, &bottom_raw, &triangles_by_color_bottom,
&shell_triangles_by_color_top, &shell_triangles_by_color_bottom](const tbb::blocked_range<size_t> &range) {
size_t group_idx = range.begin() / granularity;
size_t layer_idx_offset = (group_idx & 1) * num_layers;
for (size_t layer_idx = range.begin(); layer_idx < range.end(); ++ layer_idx) {
for (size_t color_idx = 0; color_idx < num_facets_states; ++color_idx) {
throw_on_cancel_callback();
LayerColorStat stat = layer_color_stat(layer_idx, color_idx);
if (std::vector<Polygons> &top = top_raw[color_idx]; ! top.empty() && ! top[layer_idx].empty())
if (ExPolygons top_ex = union_ex(top[layer_idx]); ! top_ex.empty()) {
// Clean up thin projections. They are not printable anyways.
top_ex = opening_ex(top_ex, stat.small_region_threshold);
if (! top_ex.empty()) {
append(triangles_by_color_top[color_idx][layer_idx + layer_idx_offset], top_ex);
float offset = 0.f;
ExPolygons layer_slices_trimmed = input_expolygons[layer_idx];
for (int last_idx = int(layer_idx) - 1; last_idx > std::max(int(layer_idx - stat.top_shell_layers), int(0)); --last_idx) {
//BBS: offset width should be 2*spacing to avoid too narrow area which has overlap of wall line
//offset -= stat.extrusion_width ;
offset -= (stat.extrusion_spacing + stat.extrusion_width);
layer_slices_trimmed = intersection_ex(layer_slices_trimmed, input_expolygons[last_idx]);
ExPolygons last = opening_ex(intersection_ex(top_ex, offset_ex(layer_slices_trimmed, offset)), stat.small_region_threshold);
if (last.empty())
break;
append(shell_triangles_by_color_top[color_idx][last_idx + layer_idx_offset], std::move(last));
}
}
if (std::vector<Polygons> &bottom = bottom_raw[color_idx]; ! bottom.empty() && ! bottom[layer_idx].empty())
if (ExPolygons bottom_ex = union_ex(bottom[layer_idx]); ! bottom_ex.empty()) {
// Clean up thin projections. They are not printable anyways.
bottom_ex = opening_ex(bottom_ex, stat.small_region_threshold);
if (! bottom_ex.empty()) {
append(triangles_by_color_bottom[color_idx][layer_idx + layer_idx_offset], bottom_ex);
float offset = 0.f;
ExPolygons layer_slices_trimmed = input_expolygons[layer_idx];
for (size_t last_idx = layer_idx + 1; last_idx < std::min(layer_idx + stat.bottom_shell_layers, num_layers); ++last_idx) {
//BBS: offset width should be 2*spacing to avoid too narrow area which has overlap of wall line
//offset -= stat.extrusion_width;
offset -= (stat.extrusion_spacing + stat.extrusion_width);
layer_slices_trimmed = intersection_ex(layer_slices_trimmed, input_expolygons[last_idx]);
ExPolygons last = opening_ex(intersection_ex(bottom_ex, offset_ex(layer_slices_trimmed, offset)), stat.small_region_threshold);
if (last.empty())
break;
append(shell_triangles_by_color_bottom[color_idx][last_idx + layer_idx_offset], std::move(last));
}
}
if (std::vector<Polygons> &bottom = bottom_raw[color_idx]; ! bottom.empty() && ! bottom[layer_idx].empty())
if (ExPolygons bottom_ex = union_ex(bottom[layer_idx]); ! bottom_ex.empty()) {
// Clean up thin projections. They are not printable anyways.
bottom_ex = opening_ex(bottom_ex, stat.small_region_threshold);
if (! bottom_ex.empty()) {
append(triangles_by_color_bottom[color_idx][layer_idx + layer_idx_offset], bottom_ex);
float offset = 0.f;
ExPolygons layer_slices_trimmed = input_expolygons[layer_idx];
for (size_t last_idx = layer_idx + 1; last_idx < std::min(layer_idx + stat.bottom_shell_layers, num_layers); ++last_idx) {
//BBS: offset width should be 2*spacing to avoid too narrow area which has overlap of wall line
//offset -= stat.extrusion_width;
offset -= (stat.extrusion_spacing + stat.extrusion_width);
layer_slices_trimmed = intersection_ex(layer_slices_trimmed, input_expolygons[last_idx]);
ExPolygons last = opening_ex(intersection_ex(bottom_ex, offset_ex(layer_slices_trimmed, offset)), stat.small_region_threshold);
if (last.empty())
break;
append(shell_triangles_by_color_bottom[color_idx][last_idx + layer_idx_offset], std::move(last));
}
}
}
}
}
}
});
+13 -11
View File
@@ -2604,6 +2604,11 @@ void Print::auto_assign_extruders(ModelObject* model_object) const
void PrintObject::set_shared_object(PrintObject *object)
{
// Orca: from now on m_layers / m_support_layers only alias the shared object's layers, so release the
// ones this object still owns (it may have sliced itself before it became shareable again).
// Both are no-ops once m_shared_object is set, so this cannot free layers owned by another object.
clear_support_layers();
clear_layers();
m_shared_object = object;
BOOST_LOG_TRIVIAL(info) << __FUNCTION__ << boost::format(": this=%1%, found shared object from %2%")%this%m_shared_object;
}
@@ -4329,9 +4334,9 @@ bool Print::is_dynamic_group_reorder() const
return true;
}
int Print::get_filament_config_indx(int filament_id, int layer_id, bool use_cache)
int Print::get_filament_config_indx(int filament_id, int layer_id)
{
return get_config_index(filament_id, layer_id, m_config.filament_extruder_variant.values, m_filament_self_index, use_cache ? &m_filament_index_map : nullptr);
return get_config_index(filament_id, layer_id, m_config.filament_extruder_variant.values, m_filament_self_index, m_filament_index_map);
}
void Print::update_filament_self_index_cache()
@@ -4374,7 +4379,7 @@ int Print::get_nozzle_config_index(int filament_id, int layer_id)
return get_config_index(filament_id, layer_id, m_default_region_config.print_extruder_variant.values, m_default_region_config.print_extruder_id.values, m_nozzle_index_map);
}
int Print::get_config_index(int filament_id, int layer_id, const std::vector<std::string> &variant_list, const std::vector<int>& self_index_list, FilamentIndexMap *index_map)
int Print::get_config_index(int filament_id, int layer_id, const std::vector<std::string> &variant_list, const std::vector<int>& self_index_list, FilamentIndexMap &index_map)
{
auto group_result = get_layered_nozzle_group_result();
// Orca: defensive — when no grouping producer has published a result yet, fall back to the
@@ -4385,8 +4390,7 @@ int Print::get_config_index(int filament_id, int layer_id, const std::vector<std
if (!nozzle_info.has_value()) {
// Orca: this fallback runs per-filament/per-layer in the g-code hot path — log once per filament
// (reset each slice) instead of flooding thousands of identical lines that bury the real error.
// Without the cache, the log set is left alone too; the cached caller reports the same filament.
if (index_map && m_missing_nozzle_group_logged.insert(filament_id).second)
if (m_missing_nozzle_group_logged.insert(filament_id).second)
BOOST_LOG_TRIVIAL(error) << __FUNCTION__
<< boost::format(", Line %1%: could not found group_nozzle_info corresponding to filament_id %2%, layer_id %3% (further occurrences for this filament suppressed)") % __LINE__ % filament_id %
layer_id;
@@ -4395,17 +4399,15 @@ int Print::get_config_index(int filament_id, int layer_id, const std::vector<std
ExtruderType extruder_type = ExtruderType(m_config.extruder_type.get_at(nozzle_info->extruder_id));
NozzleVolumeType nozzle_volume_type = nozzle_info->volume_type;
if (!index_map)
return get_config_index_base(nozzle_volume_type, extruder_type, filament_id + 1, variant_list, self_index_list);
FilamentIndexKey key{filament_id, extruder_type, nozzle_volume_type};
auto iter = index_map->find(key);
if (iter == index_map->end()) {
auto iter = index_map.find(key);
if (iter == index_map.end()) {
int index = get_config_index_base(nozzle_volume_type, extruder_type, filament_id + 1, variant_list, self_index_list);
(*index_map)[key] = index;
index_map[key] = index;
return index;
} else {
return iter->second;
return index_map[key];
}
}
+3
View File
@@ -987,6 +987,9 @@ void PrintObject::generate_support_material()
this->_generate_support_material();
m_print->throw_if_canceled();
}
// Orca: the tree support collision/avoidance caches and support nodes are only used while this step runs
// (detect_overhangs() rebuilds them from scratch), so don't keep them resident until the next slice.
this->clear_tree_support_preview_cache();
this->set_done(posSupportMaterial);
}
}
+1 -1
View File
@@ -1346,7 +1346,7 @@ void PrintObject::slice_volumes()
if (min_growth < 0.f || elfoot > 0.f) {
// Apply the negative XY compensation. (the ones that is <0)
ExPolygons trimming;
static const float eps = float(scale_(m_config.slice_closing_radius.value) * 1.5);
const float eps = float(scale_(m_config.slice_closing_radius.value) * 1.5);
if (elfoot > 0.f) {
ExPolygons expolygons_to_compensate = offset_ex(layer->merged(eps), -eps);
lslices_elfoot_uncompensated[layer_id] = expolygons_to_compensate;
+1 -15
View File
@@ -424,16 +424,8 @@ void TreeModelVolumes::calculateCollision(const coord_t radius, const LayerIndex
[this](size_t i, size_t j) { return m_layer_outlines[i].second.size() < m_layer_outlines[j].second.size(); });
// Layer range for which the collisions will be calculated.
// Another thread may have advanced getMaxCalculatedLayer() past max_layer_idx after this calculation
// was requested. Bail out in that case, otherwise the layer range would be negative and allocating
// it would throw std::length_error out of a parallel task.
const LayerIndex start_layer = 1 + m_collision_cache.getMaxCalculatedLayer(radius);
if (start_layer > max_layer_idx) {
BOOST_LOG_TRIVIAL(debug) << "Requested calculation for value already calculated ?";
return;
}
LayerPolygonCache data;
data.allocate(start_layer, max_layer_idx + 1);
data.allocate(m_collision_cache.getMaxCalculatedLayer(radius) + 1, max_layer_idx + 1);
const bool calculate_placable = m_support_rests_on_model && radius == 0;
LayerPolygonCache data_placeable;
@@ -812,12 +804,6 @@ void TreeModelVolumes::calculateWallRestrictions(const std::vector<RadiusLayerPa
const coord_t radius = keys[key_idx].first;
const LayerIndex max_required_layer = keys[key_idx].second;
const coord_t min_layer_bottom = std::max(1, m_wall_restrictions_cache.getMaxCalculatedLayer(radius));
if (min_layer_bottom > max_required_layer) {
// Another thread has calculated this range in the meantime. Continuing would make
// buffer_size negative and allocating it would throw std::length_error.
BOOST_LOG_TRIVIAL(debug) << "Requested calculation for value already calculated ?";
continue;
}
const size_t buffer_size = max_required_layer + 1 - min_layer_bottom;
std::vector<Polygons> data(buffer_size, Polygons{});
std::vector<Polygons> data_min;
+12 -30
View File
@@ -1820,37 +1820,15 @@ coordf_t TreeSupport::get_radius(const SupportNode* node)
return node->radius;
}
ExPolygons TreeSupport::get_avoidance(coordf_t radius, size_t obj_layer_nr)
// Orca: these are hit up to several times per node per layer in drop_nodes(), so hand out a
// reference into the TreeSupportData cache instead of copying the ExPolygons out of it.
const ExPolygons& TreeSupport::get_avoidance(coordf_t radius, size_t obj_layer_nr)
{
#if USE_SUPPORT_3D
if (m_model_volumes) {
bool on_build_plate = m_object_config->support_on_build_plate_only.value;
const Polygons& avoid_polys = m_model_volumes->getAvoidance(radius, obj_layer_nr, TreeSupport3D::TreeModelVolumes::AvoidanceType::FastSafe, on_build_plate, true);
ExPolygons expolys;
for (auto& poly : avoid_polys)
expolys.emplace_back(std::move(poly));
return expolys;
}
return ExPolygons();
#else
return m_ts_data->get_avoidance(radius, obj_layer_nr);
#endif
}
ExPolygons TreeSupport::get_collision(coordf_t radius, size_t layer_nr)
const ExPolygons& TreeSupport::get_collision(coordf_t radius, size_t layer_nr)
{
#if USE_SUPPORT_3D
if (m_model_volumes) {
bool on_build_plate = m_object_config->support_on_build_plate_only.value;
const Polygons& collision_polys = m_model_volumes->getCollision(radius, layer_nr, true);
ExPolygons expolys;
for (auto& poly : collision_polys)
expolys.emplace_back(std::move(poly));
return expolys;
}
#else
return m_ts_data->get_collision(radius, layer_nr);
#endif
return ExPolygons();
}
Polygons TreeSupport::get_collision_polys(coordf_t radius, size_t layer_nr)
{
@@ -2639,7 +2617,11 @@ void TreeSupport::draw_circles()
#endif // SUPPORT_TREE_DEBUG_TO_SVG
SupportLayerPtrs& ts_layers = m_object->support_layers();
auto iter = std::remove_if(ts_layers.begin(), ts_layers.end(), [](SupportLayer* ts_layer) { return ts_layer->height < EPSILON; });
// Orca: the vector owns its layers, so the dropped ones have to be deleted, not just unlinked.
// std::stable_partition (unlike std::remove_if) leaves exactly the dropped layers in the tail.
auto iter = std::stable_partition(ts_layers.begin(), ts_layers.end(), [](SupportLayer* ts_layer) { return ts_layer->height >= EPSILON; });
for (auto it = iter; it != ts_layers.end(); ++it)
delete *it;
ts_layers.erase(iter, ts_layers.end());
for (int layer_nr = 0; layer_nr < ts_layers.size(); layer_nr++) {
ts_layers[layer_nr]->upper_layer = layer_nr != ts_layers.size() - 1 ? ts_layers[layer_nr + 1] : nullptr;
@@ -2882,7 +2864,7 @@ void TreeSupport::drop_nodes()
//Insert a completely new node and let both original nodes fade.
Point next_position = (node.position + neighbours[0]) / 2; //Average position of the two nodes.
coordf_t next_radius = calc_radius(node.dist_mm_to_top+height_next);
auto avoid_layer = get_avoidance(next_radius, obj_layer_nr_next);
const ExPolygons& avoid_layer = get_avoidance(next_radius, obj_layer_nr_next);
if (group_index == 0)
{
//Avoid collisions.
@@ -3069,7 +3051,7 @@ void TreeSupport::drop_nodes()
}
#endif
coordf_t next_radius = calc_radius(node.dist_mm_to_top + height_next);
auto avoidance_next = get_avoidance(next_radius, obj_layer_nr_next);
const ExPolygons& avoidance_next = get_avoidance(next_radius, obj_layer_nr_next);
Point to_outside = projection_onto(avoidance_next, node.position);
Point direction_to_outer = to_outside - node.position;
@@ -3123,7 +3105,7 @@ void TreeSupport::drop_nodes()
if (is_outside) { next_layer_vertex = candidate_vertex; }
}
}
auto next_collision = get_collision(0, obj_layer_nr_next);
const ExPolygons& next_collision = get_collision(0, obj_layer_nr_next);
const bool to_buildplate = !is_inside_ex(m_ts_data->m_layer_outlines[obj_layer_nr_next], next_layer_vertex);
// don't increase radius if next node will collide partially with the object (STUDIO-7883)
to_outside = projection_onto(next_collision, next_layer_vertex);
+2 -2
View File
@@ -511,9 +511,9 @@ private:
coordf_t calc_branch_radius(coordf_t base_radius, coordf_t mm_to_top, double diameter_angle_scale_factor, bool use_min_distance=true);
coordf_t calc_radius(coordf_t mm_to_top);
coordf_t get_radius(const SupportNode* node);
ExPolygons get_avoidance(coordf_t radius, size_t obj_layer_nr);
const ExPolygons& get_avoidance(coordf_t radius, size_t obj_layer_nr);
// layer's expolygon expanded by radius+m_xy_distance
ExPolygons get_collision(coordf_t radius, size_t layer_nr);
const ExPolygons& get_collision(coordf_t radius, size_t layer_nr);
// get Polygons instead of ExPolygons
Polygons get_collision_polys(coordf_t radius, size_t layer_nr);
+57 -59
View File
@@ -12,7 +12,6 @@
#include <thread>
#include <tbb/parallel_for.h>
#include <tbb/task_arena.h>
#include <tbb/task_scheduler_observer.h>
#include "Thread.hpp"
#include "Utils.hpp"
@@ -213,71 +212,70 @@ bool is_main_thread_active()
return get_main_thread_id() == boost::this_thread::get_id();
}
// Name the current TBB worker thread and set its locale to "C", so that the G-code generator
// produces "." as a decimal separator. Called once per worker thread, before it runs its first task.
static void setup_tbb_worker_thread()
{
static std::atomic<size_t> s_worker_idx{ 0 };
std::ostringstream name;
name << "slic3r_tbb_" << (1 + s_worker_idx.fetch_add(1, std::memory_order_relaxed));
set_current_thread_name(name.str().c_str());
#ifdef _WIN32
_configthreadlocale(_ENABLE_PER_THREAD_LOCALE);
std::setlocale(LC_ALL, "C");
#else
// We are leaking some memory here, because the newlocale() produced memory will never be released.
// This is not a problem though, as there will be a maximum one worker thread created per physical thread.
uselocale(newlocale(
#ifdef __APPLE__
LC_ALL_MASK
#else // some Unix / Linux / BSD
LC_ALL
#endif
, "C", nullptr));
#endif
}
// Sets up the TBB worker threads of the arena of the thread, which activated the observation.
// A worker sets itself up on entry to the arena, before it executes its first task, thus unlike a barrier
// inside a parallel_for, this does not depend on TBB running any number of tasks simultaneously.
class TBBWorkerThreadSetupObserver : public tbb::task_scheduler_observer
{
public:
TBBWorkerThreadSetupObserver() { this->observe(true); }
void on_scheduler_entry(bool is_worker) override
{
// Leave the external threads (the calling / UI thread) alone, their name and locale must not be modified here.
if (! is_worker)
return;
// A worker thread enters an arena many times, while its name and locale have to be set just once.
static thread_local bool initialized = false;
if (initialized)
return;
initialized = true;
setup_tbb_worker_thread();
}
};
// Name the threads of the Intel TBB thread pool by an index and set their locale to "C"
// for the G-code generator to produce "." as a decimal separator.
// Formerly all the worker threads were caught inside a single parallel_for, which was held on a condition
// variable barrier until max_concurrency() of its chunks were running. TBB guarantees no such simultaneity,
// thus the barrier was able to block the slicing threads indefinitely. The TBB scheduler observer below
// sets each worker up on its own, thus no two chunks have to run at the same time.
// Spawn (n - 1) worker threads on Intel TBB thread pool and name them by an index and a system thread ID.
// Also it sets locale of the worker threads to "C" for the G-code generator to produce "." as a decimal separator.
void name_tbb_thread_pool_threads_set_locale()
{
static bool initialized = false;
if (initialized)
return;
initialized = true;
// see GH issue #5661 PrusaSlicer hangs on Linux when run with non standard task affinity
// TBB will respect the task affinity mask on Linux and spawn less threads than std::thread::hardware_concurrency().
// const size_t nthreads_hw = std::thread::hardware_concurrency();
const size_t nthreads_hw = tbb::this_task_arena::max_concurrency();
size_t nthreads = nthreads_hw;
#ifdef SLIC3R_PROFILE
// Shiny profiler is not thread safe, thus disable parallelization.
disable_multi_threading();
nthreads = 1;
#endif
// An observer is local to the arena of the thread which activates it, thus one observer is registered
// per calling thread. Being function local and thread local, it is also initialized exactly once per
// thread without a race. It is intentionally never destroyed, as it has to stay alive as long as the
// TBB scheduler may notify it, which includes the shutdown of the process.
static thread_local tbb::task_scheduler_observer *observer = new TBBWorkerThreadSetupObserver();
(void)observer;
size_t nthreads_running(0);
std::condition_variable cv;
std::mutex cv_m;
auto master_thread_id = std::this_thread::get_id();
tbb::parallel_for(
tbb::blocked_range<size_t>(0, nthreads, 1),
[&nthreads_running, nthreads, &master_thread_id, &cv, &cv_m](const tbb::blocked_range<size_t> &range) {
assert(range.begin() + 1 == range.end());
if (std::unique_lock<std::mutex> lk(cv_m); ++nthreads_running == nthreads) {
lk.unlock();
// All threads are spinning.
// Wake them up.
cv.notify_all();
} else {
// Wait for the last thread to wake the others.
cv.wait(lk, [&nthreads_running, nthreads]{return nthreads_running == nthreads;});
}
auto thread_id = std::this_thread::get_id();
if (thread_id == master_thread_id) {
// The calling thread runs the 0'th task.
assert(range.begin() == 0);
} else {
assert(range.begin() > 0);
std::ostringstream name;
name << "slic3r_tbb_" << range.begin();
set_current_thread_name(name.str().c_str());
// Set locales of the worker thread to "C".
#ifdef _WIN32
_configthreadlocale(_ENABLE_PER_THREAD_LOCALE);
std::setlocale(LC_ALL, "C");
#else
// We are leaking some memory here, because the newlocale() produced memory will never be released.
// This is not a problem though, as there will be a maximum one worker thread created per physical thread.
uselocale(newlocale(
#ifdef __APPLE__
LC_ALL_MASK
#else // some Unix / Linux / BSD
LC_ALL
#endif
, "C", nullptr));
#endif
}
});
}
}
+1 -8
View File
@@ -28,10 +28,6 @@ TriangleSetSamples sample_its_uniform_parallel(size_t samples_count, const index
area_sum_to_triangle_idx[area_sum] = t_idx;
}
if (area_sum_to_triangle_idx.empty())
// No triangle to sample from.
return {};
std::mt19937_64 mersenne_engine { 27644437 };
// random numbers on interval [0, 1)
std::uniform_real_distribution<double> fdistribution;
@@ -54,10 +50,7 @@ TriangleSetSamples sample_its_uniform_parallel(size_t samples_count, const index
tbb::blocked_range<size_t> r) {
for (size_t s_idx = r.begin(); s_idx < r.end(); ++s_idx) {
double t_sample = random_samples[s_idx].x() * area_sum;
// The keys of area_sum_to_triangle_idx are accumulated areas in double precision, while area_sum
// is a float, thus t_sample may reach or exceed the largest key and upper_bound() may return end().
auto t_it = area_sum_to_triangle_idx.upper_bound(t_sample);
size_t t_idx = (t_it == area_sum_to_triangle_idx.end() ? std::prev(t_it) : t_it)->second;
size_t t_idx = area_sum_to_triangle_idx.upper_bound(t_sample)->second;
double sq_u = std::sqrt(random_samples[s_idx].y());
double v = random_samples[s_idx].z();