Compare commits

..
Author SHA1 Message Date
ExPikaPaka bea412fddf Move the G-code processing result instead of copying it
GCodeProcessorResult declares a copy assignment, which suppresses the implicit
move assignment, so `*result = std::move(processor.extract_result())` binds to
the copy and duplicates the whole moves array: 96 bytes per move, measured at
147.7 MiB in one allocation for a 597k facet model at 0.08 mm, with the source
staying alive until the export returns.

Add the move assignment. It assigns exactly the same 39 members in the same
order as the copy, including the ones the copy deliberately leaves alone, so a
target that carries its own filament maps and nozzle type keeps them. A move
constructor is intentionally not added: the type holds a std::mutex, so it is
neither copy nor move constructible today and a partial one would leave ten
members uninitialised.

reset() now releases the two print sized vectors rather than clearing them. Its
callers are the paths that discard a result, so the memory went back only at the
next slice.

ViewerImpl::reset() does the same for the two vectors sized by the print, which
is what keeps a discarded preview resident, and counts m_vertices_colors in the
reported CPU memory, where it was missing.
2026-10-01 09:15:15 +02:00
8 changed files with 168 additions and 152 deletions
+2 -9
View File
@@ -809,20 +809,13 @@ void priv::set_skip_for_out_of_aoi(std::vector<bool> &skip_indicies,
}); // END parallel for
// inspect all triangles, when it is out of bounding box
// NOTE: std::vector<bool> is bit packed, thus setting its items from multiple threads is a
// read-modify-write race on the shared words and silently loses flags. Collect the flags into
// a byte per triangle, where the chunks do not share memory, and merge them afterwards.
std::vector<unsigned char> skip_triangle(its.indices.size(), 0);
tbb::parallel_for(tbb::blocked_range<size_t>(0, its.indices.size()),
[&its, &is_on_sides, &skip_triangle](const tbb::blocked_range<size_t> &range) {
[&its, &is_on_sides, &skip_indicies](const tbb::blocked_range<size_t> &range) {
for (size_t i = range.begin(); i < range.end(); ++i) {
if (is_all_on_one_side(its.indices[i], is_on_sides))
skip_triangle[i] = 1;
skip_indicies[i] = true;
}
}); // END parallel for
for (size_t i = 0; i < skip_triangle.size(); ++i)
if (skip_triangle[i])
skip_indicies[i] = true;
}
indexed_triangle_set Slic3r::its_mask(const indexed_triangle_set &its,
+6 -2
View File
@@ -2582,8 +2582,12 @@ void GCodeProcessorResult::reset() {
//BBS: add mutex for protection of gcode result
lock();
moves.clear();
lines_ends.clear();
// release rather than clear: these two are sized by the print - one entry per move and one
// per g-code line - and a reset is where the memory is expected to go back to the allocator
// (see BackgroundSlicingProcess::apply()). The capacity would not be reused anyway: the
// result is refilled by move-assigning the processor's own result.
moves = std::vector<MoveVertex>();
lines_ends = std::vector<size_t>();
printable_area = Pointfs();
//BBS: add bed exclude area
bed_exclude_area = Pointfs();
+50
View File
@@ -369,6 +369,56 @@ class Print;
initial_layer_time = other.initial_layer_time;
#if ENABLE_GCODE_VIEWER_STATISTICS
time = other.time;
#endif
return *this;
}
// Orca: the user-declared copy assignment above suppresses the implicit move assignment, so
// `*result = std::move(processor.extract_result())` used to deep copy 'moves' (one MoveVertex
// per move, gigabytes on a large print) while the source stayed alive. This moves exactly the
// same members as the copy above, with the same omissions, so the members the copy leaves
// untouched on the target are left untouched here too.
GCodeProcessorResult& operator=(GCodeProcessorResult &&other)
{
filename = std::move(other.filename);
id = other.id;
moves = std::move(other.moves);
lines_ends = std::move(other.lines_ends);
printable_area = std::move(other.printable_area);
bed_exclude_area = std::move(other.bed_exclude_area);
wrapping_exclude_area = std::move(other.wrapping_exclude_area);
toolpath_outside = other.toolpath_outside;
label_object_enabled = other.label_object_enabled;
long_retraction_when_cut = other.long_retraction_when_cut;
timelapse_warning_code = other.timelapse_warning_code;
printable_height = other.printable_height;
settings_ids = std::move(other.settings_ids);
filaments_count = other.filaments_count;
extruder_colors = std::move(other.extruder_colors);
filament_diameters = std::move(other.filament_diameters);
filament_densities = std::move(other.filament_densities);
filament_costs = std::move(other.filament_costs);
print_statistics = std::move(other.print_statistics);
custom_gcode_per_print_z = std::move(other.custom_gcode_per_print_z);
spiral_vase_mode = other.spiral_vase_mode;
warnings = std::move(other.warnings);
bed_type = other.bed_type;
gcode_check_result = std::move(other.gcode_check_result);
limit_filament_maps = std::move(other.limit_filament_maps);
filament_printable_reuslt = std::move(other.filament_printable_reuslt);
nozzle_group_result = std::move(other.nozzle_group_result);
extruder_types = std::move(other.extruder_types);
printer_extruder_variant = std::move(other.printer_extruder_variant);
printer_extruder_id = std::move(other.printer_extruder_id);
layer_filaments = std::move(other.layer_filaments);
filament_change_sequence = std::move(other.filament_change_sequence);
used_mixed_filaments = std::move(other.used_mixed_filaments);
nozzle_change_sequence = std::move(other.nozzle_change_sequence);
optimal_assignment = std::move(other.optimal_assignment);
filament_change_count_map = std::move(other.filament_change_count_map);
skippable_part_time = std::move(other.skippable_part_time);
initial_layer_time = other.initial_layer_time;
#if ENABLE_GCODE_VIEWER_STATISTICS
time = other.time;
#endif
return *this;
}
+4 -15
View File
@@ -1378,22 +1378,12 @@ static inline std::vector<std::vector<ExPolygons>> segmentation_top_and_bottom_l
return out;
};
// The layers are processed in groups of "granularity" layers. A layer projects its shells up to "granularity"
// layers away, thus a group may write into the slots of its neighbor groups. The even and the odd groups
// therefore write into two disjoint halves of the output vectors (the 2nd half is offset by num_layers) and
// both halves are merged below. The group index has to be derived from the layer index and not from the extent
// of the TBB sub-range: tbb::blocked_range bisects at midpoints, thus a sub-range neither starts at a multiple
// of the grain size nor covers a whole group, and two sub-ranges of one group would append into a single
// ExPolygons concurrently. Iterating over the groups keeps every group on a single thread, in ascending order.
const size_t num_groups = (num_layers + size_t(granularity) - 1) / size_t(granularity);
tbb::parallel_for(tbb::blocked_range<size_t>(0, num_groups, 1), [&granularity, &num_layers, &num_facets_states, &layer_color_stat, &top_raw, &triangles_by_color_top,
tbb::parallel_for(tbb::blocked_range<size_t>(0, num_layers, granularity), [&granularity, &num_layers, &num_facets_states, &layer_color_stat, &top_raw, &triangles_by_color_top,
&throw_on_cancel_callback, &input_expolygons, &bottom_raw, &triangles_by_color_bottom,
&shell_triangles_by_color_top, &shell_triangles_by_color_bottom](const tbb::blocked_range<size_t> &range) {
for (size_t group_idx = range.begin(); group_idx < range.end(); ++ group_idx) {
const size_t layer_idx_offset = (group_idx & 1) * num_layers;
const size_t layer_idx_begin = group_idx * size_t(granularity);
const size_t layer_idx_end = std::min(num_layers, layer_idx_begin + size_t(granularity));
for (size_t layer_idx = layer_idx_begin; layer_idx < layer_idx_end; ++ layer_idx) {
size_t group_idx = range.begin() / granularity;
size_t layer_idx_offset = (group_idx & 1) * num_layers;
for (size_t layer_idx = range.begin(); layer_idx < range.end(); ++ layer_idx) {
for (size_t color_idx = 0; color_idx < num_facets_states; ++color_idx) {
throw_on_cancel_callback();
LayerColorStat stat = layer_color_stat(layer_idx, color_idx);
@@ -1439,7 +1429,6 @@ static inline std::vector<std::vector<ExPolygons>> segmentation_top_and_bottom_l
}
}
}
}
});
std::vector<std::vector<ExPolygons>> triangles_by_color_merged(num_facets_states);
+1 -15
View File
@@ -424,16 +424,8 @@ void TreeModelVolumes::calculateCollision(const coord_t radius, const LayerIndex
[this](size_t i, size_t j) { return m_layer_outlines[i].second.size() < m_layer_outlines[j].second.size(); });
// Layer range for which the collisions will be calculated.
// Another thread may have advanced getMaxCalculatedLayer() past max_layer_idx after this calculation
// was requested. Bail out in that case, otherwise the layer range would be negative and allocating
// it would throw std::length_error out of a parallel task.
const LayerIndex start_layer = 1 + m_collision_cache.getMaxCalculatedLayer(radius);
if (start_layer > max_layer_idx) {
BOOST_LOG_TRIVIAL(debug) << "Requested calculation for value already calculated ?";
return;
}
LayerPolygonCache data;
data.allocate(start_layer, max_layer_idx + 1);
data.allocate(m_collision_cache.getMaxCalculatedLayer(radius) + 1, max_layer_idx + 1);
const bool calculate_placable = m_support_rests_on_model && radius == 0;
LayerPolygonCache data_placeable;
@@ -812,12 +804,6 @@ void TreeModelVolumes::calculateWallRestrictions(const std::vector<RadiusLayerPa
const coord_t radius = keys[key_idx].first;
const LayerIndex max_required_layer = keys[key_idx].second;
const coord_t min_layer_bottom = std::max(1, m_wall_restrictions_cache.getMaxCalculatedLayer(radius));
if (min_layer_bottom > max_required_layer) {
// Another thread has calculated this range in the meantime. Continuing would make
// buffer_size negative and allocating it would throw std::length_error.
BOOST_LOG_TRIVIAL(debug) << "Requested calculation for value already calculated ?";
continue;
}
const size_t buffer_size = max_required_layer + 1 - min_layer_bottom;
std::vector<Polygons> data(buffer_size, Polygons{});
std::vector<Polygons> data_min;
+46 -48
View File
@@ -12,7 +12,6 @@
#include <thread>
#include <tbb/parallel_for.h>
#include <tbb/task_arena.h>
#include <tbb/task_scheduler_observer.h>
#include "Thread.hpp"
#include "Utils.hpp"
@@ -213,14 +212,54 @@ bool is_main_thread_active()
return get_main_thread_id() == boost::this_thread::get_id();
}
// Name the current TBB worker thread and set its locale to "C", so that the G-code generator
// produces "." as a decimal separator. Called once per worker thread, before it runs its first task.
static void setup_tbb_worker_thread()
// Spawn (n - 1) worker threads on Intel TBB thread pool and name them by an index and a system thread ID.
// Also it sets locale of the worker threads to "C" for the G-code generator to produce "." as a decimal separator.
void name_tbb_thread_pool_threads_set_locale()
{
static std::atomic<size_t> s_worker_idx{ 0 };
static bool initialized = false;
if (initialized)
return;
initialized = true;
// see GH issue #5661 PrusaSlicer hangs on Linux when run with non standard task affinity
// TBB will respect the task affinity mask on Linux and spawn less threads than std::thread::hardware_concurrency().
// const size_t nthreads_hw = std::thread::hardware_concurrency();
const size_t nthreads_hw = tbb::this_task_arena::max_concurrency();
size_t nthreads = nthreads_hw;
#ifdef SLIC3R_PROFILE
// Shiny profiler is not thread safe, thus disable parallelization.
disable_multi_threading();
nthreads = 1;
#endif
size_t nthreads_running(0);
std::condition_variable cv;
std::mutex cv_m;
auto master_thread_id = std::this_thread::get_id();
tbb::parallel_for(
tbb::blocked_range<size_t>(0, nthreads, 1),
[&nthreads_running, nthreads, &master_thread_id, &cv, &cv_m](const tbb::blocked_range<size_t> &range) {
assert(range.begin() + 1 == range.end());
if (std::unique_lock<std::mutex> lk(cv_m); ++nthreads_running == nthreads) {
lk.unlock();
// All threads are spinning.
// Wake them up.
cv.notify_all();
} else {
// Wait for the last thread to wake the others.
cv.wait(lk, [&nthreads_running, nthreads]{return nthreads_running == nthreads;});
}
auto thread_id = std::this_thread::get_id();
if (thread_id == master_thread_id) {
// The calling thread runs the 0'th task.
assert(range.begin() == 0);
} else {
assert(range.begin() > 0);
std::ostringstream name;
name << "slic3r_tbb_" << (1 + s_worker_idx.fetch_add(1, std::memory_order_relaxed));
name << "slic3r_tbb_" << range.begin();
set_current_thread_name(name.str().c_str());
// Set locales of the worker thread to "C".
#ifdef _WIN32
_configthreadlocale(_ENABLE_PER_THREAD_LOCALE);
std::setlocale(LC_ALL, "C");
@@ -235,49 +274,8 @@ static void setup_tbb_worker_thread()
#endif
, "C", nullptr));
#endif
}
// Sets up the TBB worker threads of the arena of the thread, which activated the observation.
// A worker sets itself up on entry to the arena, before it executes its first task, thus unlike a barrier
// inside a parallel_for, this does not depend on TBB running any number of tasks simultaneously.
class TBBWorkerThreadSetupObserver : public tbb::task_scheduler_observer
{
public:
TBBWorkerThreadSetupObserver() { this->observe(true); }
void on_scheduler_entry(bool is_worker) override
{
// Leave the external threads (the calling / UI thread) alone, their name and locale must not be modified here.
if (! is_worker)
return;
// A worker thread enters an arena many times, while its name and locale have to be set just once.
static thread_local bool initialized = false;
if (initialized)
return;
initialized = true;
setup_tbb_worker_thread();
}
};
// Name the threads of the Intel TBB thread pool by an index and set their locale to "C"
// for the G-code generator to produce "." as a decimal separator.
// Formerly all the worker threads were caught inside a single parallel_for, which was held on a condition
// variable barrier until max_concurrency() of its chunks were running. TBB guarantees no such simultaneity,
// thus the barrier was able to block the slicing threads indefinitely. The TBB scheduler observer below
// sets each worker up on its own, thus no two chunks have to run at the same time.
void name_tbb_thread_pool_threads_set_locale()
{
#ifdef SLIC3R_PROFILE
// Shiny profiler is not thread safe, thus disable parallelization.
disable_multi_threading();
#endif
// An observer is local to the arena of the thread which activates it, thus one observer is registered
// per calling thread. Being function local and thread local, it is also initialized exactly once per
// thread without a race. It is intentionally never destroyed, as it has to stay alive as long as the
// TBB scheduler may notify it, which includes the shutdown of the process.
static thread_local tbb::task_scheduler_observer *observer = new TBBWorkerThreadSetupObserver();
(void)observer;
});
}
}
+1 -8
View File
@@ -28,10 +28,6 @@ TriangleSetSamples sample_its_uniform_parallel(size_t samples_count, const index
area_sum_to_triangle_idx[area_sum] = t_idx;
}
if (area_sum_to_triangle_idx.empty())
// No triangle to sample from.
return {};
std::mt19937_64 mersenne_engine { 27644437 };
// random numbers on interval [0, 1)
std::uniform_real_distribution<double> fdistribution;
@@ -54,10 +50,7 @@ TriangleSetSamples sample_its_uniform_parallel(size_t samples_count, const index
tbb::blocked_range<size_t> r) {
for (size_t s_idx = r.begin(); s_idx < r.end(); ++s_idx) {
double t_sample = random_samples[s_idx].x() * area_sum;
// The keys of area_sum_to_triangle_idx are accumulated areas in double precision, while area_sum
// is a float, thus t_sample may reach or exceed the largest key and upper_bound() may return end().
auto t_it = area_sum_to_triangle_idx.upper_bound(t_sample);
size_t t_idx = (t_it == area_sum_to_triangle_idx.end() ? std::prev(t_it) : t_it)->second;
size_t t_idx = area_sum_to_triangle_idx.upper_bound(t_sample)->second;
double sq_u = std::sqrt(random_samples[s_idx].y());
double v = random_samples[s_idx].z();
+6 -3
View File
@@ -883,15 +883,17 @@ void ViewerImpl::reset()
m_used_extruders.clear();
m_total_time = { 0.0f, 0.0f };
m_travels_time = { 0.0f, 0.0f };
m_vertices.clear();
m_vertices_colors.clear();
// swap rather than clear: these are sized by the print, and a reset means the memory
// should go back, not sit reserved until the next load
std::vector<PathVertex>().swap(m_vertices);
std::vector<float>().swap(m_vertices_colors);
for (std::vector<float>& times : m_layer_start_times)
std::vector<float>().swap(times);
std::vector<uint32_t>().swap(m_layer_first_vertex);
std::vector<float>().swap(m_colors_scratch);
m_valid_lines_bitset.clear();
// BitSet::clear() only zeroes the bits, it keeps the blocks allocated; load() builds a new
// bitset anyway and it is never read while m_vertices is empty
m_valid_lines_bitset = BitSet<>();
#if VGCODE_ENABLE_COG_AND_TOOL_MARKERS
m_cog_marker.reset();
#endif // VGCODE_ENABLE_COG_AND_TOOL_MARKERS
@@ -1812,6 +1814,7 @@ size_t ViewerImpl::get_used_cpu_memory() const
ret += sizeof(m_extrusion_roles_colors);
ret += sizeof(m_options_colors);
ret += STDVEC_MEMSIZE(m_vertices, PathVertex);
ret += STDVEC_MEMSIZE(m_vertices_colors, float);
for (const std::vector<float>& times : m_layer_start_times)
ret += STDVEC_MEMSIZE(times, float);
ret += STDVEC_MEMSIZE(m_layer_first_vertex, uint32_t);