Img2Num C++ (Internal Developer Docs) dev
API Documentation
Loading...
Searching...
No Matches
kmeans_gpu.cpp
1
2#include "internal/kmeans_gpu.h"
3
4#include "img2num.h"
5#include "internal/cielab.h"
6#include "internal/gpu.h"
7#include "internal/Image.h"
8#include "internal/LABAPixel.h"
9#include "internal/log.h"
10#include "internal/PixelConverters.h"
11#include "internal/RGBAPixel.h"
12
13#include <algorithm>
14#include <cmath>
15#include <cstddef>
16#include <cstdint>
17#include <cstdlib>
18#include <cstring>
19#include <ctime>
20#include <functional>
21#include <limits>
22#include <numeric>
23#include <random>
24#include <type_traits> // Required for std::is_same_v
25#include <vector>
26
27static constexpr uint8_t COLOR_SPACE_OPTION_CIELAB {0};
28static constexpr uint8_t COLOR_SPACE_OPTION_RGB {1};
29
30#ifdef _MSC_VER
31#pragma pack(push, 1)
32#endif
33struct Params {
34 uint32_t numPoints;
35 uint32_t numCentroids;
36 uint32_t pad[2];
37}
38#ifndef _MSC_VER
39__attribute__((packed))
40#endif
41;
42#ifdef _MSC_VER
43#pragma pack(pop)
44#endif
45
46#ifdef _MSC_VER
47#pragma pack(push, 1)
48#endif
50 int32_t sumR;
51 int32_t sumG;
52 int32_t sumB;
53 uint32_t count;
54}
55#ifndef _MSC_VER
56__attribute__((packed))
57#endif
58;
59#ifdef _MSC_VER
60#pragma pack(pop)
61#endif
62
63#ifdef _MSC_VER
64#pragma pack(push, 1)
65#endif
67 float r, g, b, a;
68 uint32_t width;
69 uint32_t pad[3]; // Padding to align to 16 bytes
70}
71#ifndef _MSC_VER
72__attribute__((packed))
73#endif
74;
75#ifdef _MSC_VER
76#pragma pack(pop)
77#endif
78
79// The K-Means++ Initialization Function
80template <typename PixelT>
81void kMeansPlusPlusInitGpu(
82 const ImageLib::Image<PixelT>& pixels, ImageLib::Image<PixelT>& out_centroids, int k,
83 const uint8_t color_space
84) {
85 if (k <= 0)
86 return;
87
88 size_t width = pixels.getWidth();
89 size_t height = pixels.getHeight();
90 size_t num_pixels = width * height;
91
92 std::vector<PixelT> centroids;
93
94 // --- WEBGPU SETUP START ---
95 // (Assuming 'device' and 'queue' are globally available or passed in)
96 // 1. Upload Image Texture
97 wgpu::TextureDescriptor texDesc = {};
98 texDesc.size = {static_cast<uint32_t>(width), static_cast<uint32_t>(height), 1};
99 texDesc.format = wgpu::TextureFormat::RGBA32Float;
100 texDesc.usage = wgpu::TextureUsage::TextureBinding | wgpu::TextureUsage::CopyDst;
101 texDesc.label = "inputTextureInit";
102 wgpu::Texture inputTexture = GPU::getClassInstance().get_device().CreateTexture(&texDesc);
103
104 // Upload pixel data (Normalization to 0.0-1.0 assumed)
105 std::vector<float> gpu_pixels;
106 gpu_pixels.reserve(num_pixels * 4);
107
108 for (int i = 0; i < num_pixels; i++) {
109 PixelT p = pixels[i];
110 if constexpr (std::is_same_v<PixelT, ImageLib::LABAPixel<float>>) {
111 gpu_pixels.push_back(p.l / 255.0f);
112 gpu_pixels.push_back(p.a / 255.0f);
113 gpu_pixels.push_back(p.b / 255.0f);
114 gpu_pixels.push_back(p.alpha / 255.0f);
115 } else {
116 gpu_pixels.push_back(p.red / 255.0f);
117 gpu_pixels.push_back(p.green / 255.0f);
118 gpu_pixels.push_back(p.blue / 255.0f);
119 gpu_pixels.push_back(p.alpha / 255.0f);
120 }
121 }
122
123 wgpu::TexelCopyTextureInfo texDst = {};
124 texDst.texture = inputTexture;
125 wgpu::TexelCopyBufferLayout texLayout = {};
126 texLayout.bytesPerRow = width * 16;
127 texLayout.rowsPerImage = height;
128 GPU::getClassInstance().get_queue().WriteTexture(
129 &texDst, gpu_pixels.data(), gpu_pixels.size() * 4, &texLayout, &texDesc.size
130 );
131
132 // 2. Create MinDist Buffer (Storage)
133 // Initialize with FLT_MAX so the first centroid overwrites everything
134 std::vector<float> initial_dists(num_pixels, std::numeric_limits<float>::max());
135
136 wgpu::BufferDescriptor distDesc = {};
137 distDesc.size = num_pixels * sizeof(float);
138 distDesc.usage =
139 wgpu::BufferUsage::Storage | wgpu::BufferUsage::CopySrc | wgpu::BufferUsage::CopyDst;
140 wgpu::Buffer minDistBuffer = GPU::getClassInstance().get_device().CreateBuffer(&distDesc);
141 GPU::getClassInstance().get_queue().WriteBuffer(
142 minDistBuffer, 0, initial_dists.data(), distDesc.size
143 );
144
145 // 3. Create Uniform Buffer (For passing new centroid color)
146
147 wgpu::BufferDescriptor uniDesc = {};
148 uniDesc.size = sizeof(CentroidParams);
149 uniDesc.usage = wgpu::BufferUsage::Uniform | wgpu::BufferUsage::CopyDst;
150 wgpu::Buffer paramBuffer = GPU::getClassInstance().get_device().CreateBuffer(&uniDesc);
151
152 // 4. Create Readback Buffer
153 wgpu::BufferDescriptor readDesc = {};
154 readDesc.size = num_pixels * sizeof(float);
155 readDesc.usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst;
156 wgpu::Buffer readBuffer = GPU::getClassInstance().get_device().CreateBuffer(&readDesc);
157
158 // 5. Compile Shader & Pipeline
159 wgpu::ComputePipeline pipeline =
160 GPU::getClassInstance().createPipeline("dist_shader", "updateDistShader");
161
162 // 6. Bind Group
163 wgpu::BindGroupEntry entries[3];
164 entries[0].binding = 0;
165 entries[0].textureView = inputTexture.CreateView();
166 entries[1].binding = 1;
167 entries[1].buffer = minDistBuffer;
168 entries[1].size = distDesc.size;
169 entries[2].binding = 2;
170 entries[2].buffer = paramBuffer;
171 entries[2].size = uniDesc.size;
172
173 wgpu::BindGroupDescriptor bgDesc = {};
174 bgDesc.layout = pipeline.GetBindGroupLayout(0);
175 bgDesc.entryCount = 3;
176 bgDesc.entries = entries;
177 wgpu::BindGroup bindGroup = GPU::getClassInstance().get_device().CreateBindGroup(&bgDesc);
178 // --- WEBGPU SETUP END ---
179
180 // RNG Setup
181 std::random_device rd;
182 std::mt19937 gen(rd());
183
184 // --- Step 1: Choose the first centroid randomly ---
185 std::uniform_int_distribution<> dis(0, num_pixels - 1);
186 int first_index = dis(gen);
187 centroids.push_back(pixels[first_index]);
188
189 // --- Step 2 & 3: Repeat until we have k centroids ---
190 // static volatile bool done = false;
191 bool* done = new bool(false);
192
193 for (int i = 1; i < k; ++i) {
194 *done = false;
195 // A. Upload Current Centroid to GPU
196 PixelT c = centroids.back();
197 CentroidParams params;
198 if constexpr (std::is_same_v<PixelT, ImageLib::LABAPixel<float>>) {
199 params = CentroidParams {
200 c.l / 255.0f, c.a / 255.0f, c.b / 255.0f, 1.0f, static_cast<uint32_t>(width)
201 };
202 } else {
203 params = CentroidParams {
204 c.red / 255.0f, c.green / 255.0f, c.blue / 255.0f, 1.0f,
205 static_cast<uint32_t>(width)
206 };
207 }
208
209 GPU::getClassInstance().get_queue().WriteBuffer(
210 paramBuffer, 0, &params, sizeof(CentroidParams)
211 );
212
213 // B. Dispatch Shader (Updates min_dist buffer on GPU)
214 wgpu::CommandEncoder encoder = GPU::getClassInstance().get_device().CreateCommandEncoder();
215 wgpu::ComputePassEncoder pass = encoder.BeginComputePass();
216 pass.SetPipeline(pipeline);
217 pass.SetBindGroup(0, bindGroup);
218 pass.DispatchWorkgroups((width + 15) / 16, (height + 15) / 16, 1);
219 pass.End();
220
221 // C. Copy Result to ReadBuffer
222 encoder.CopyBufferToBuffer(minDistBuffer, 0, readBuffer, 0, readDesc.size);
223 wgpu::CommandBuffer commands = encoder.Finish();
224 GPU::getClassInstance().get_queue().Submit(1, &commands);
225
226 // D. Map and Read
227
228 readBuffer.MapAsync(
229 wgpu::MapMode::Read, 0, readDesc.size, wgpu::CallbackMode::AllowProcessEvents,
230 [](wgpu::MapAsyncStatus status, wgpu::StringView msg, void* userdata) {
231 bool* flag = static_cast<bool*>(userdata);
232 bool success = false;
233 if (status == wgpu::MapAsyncStatus::Success) {
234 success = true;
235 } else {
236 // Handle error
237 success = false;
238 }
239 *flag = true;
240 },
241 (void*)done
242 );
243
244 // E. Wait for GPU
245 while (!*done) {
246 GPU::getClassInstance().get_instance().ProcessEvents();
247#if defined(__EMSCRIPTEN__)
248 emscripten_sleep(10);
249#endif
250 }
251
252 const float* dists = (const float*)readBuffer.GetConstMappedRange();
253 // --- CPU SIDE: Selection Logic ---
254 double sum_dist_sq = 0.0;
255
256 // 1. Sum (We have to iterate anyway for roulette, so sum here)
257 // Note: dists[] contains the SQUARED distance because shader calculated distSq
258 for (size_t j = 0; j < num_pixels; ++j) {
259 sum_dist_sq += dists[j];
260 }
261
262 // 2. Select
263 std::uniform_real_distribution<> dist_selector(0.0, sum_dist_sq);
264 double random_value = dist_selector(gen);
265 double current_sum = 0.0;
266 int selected_index = -1;
267
268 for (size_t j = 0; j < num_pixels; ++j) {
269 current_sum += dists[j];
270 if (current_sum >= random_value) {
271 selected_index = j;
272 break;
273 }
274 }
275
276 if (selected_index == -1)
277 selected_index = num_pixels - 1;
278
279 // Add new centroid
280 centroids.push_back(pixels[selected_index]);
281 readBuffer.Unmap();
282#if defined(__EMSCRIPTEN__)
283 emscripten_sleep(10);
284#endif
285 }
286
287 std::copy(centroids.begin(), centroids.end(), out_centroids.begin());
288
289 // explicit clean up
290 if (inputTexture)
291 inputTexture.Destroy();
292 readBuffer.Destroy();
293 minDistBuffer.Destroy();
294 paramBuffer.Destroy();
295 delete done;
296
297#if defined(__EMSCRIPTEN__)
298 emscripten_sleep(50);
299#endif
300}
301
302void setup(
306 ImageLib::Image<ImageLib::LABAPixel<float>>& centroids_lab, const int32_t width,
307 const int32_t height, const int32_t k, wgpu::Texture& inputTexture, wgpu::Texture& labelTexture,
308 wgpu::Texture& centroidTexture, wgpu::TextureDescriptor& labelDesc,
309 wgpu::TextureDescriptor& centroidDesc, wgpu::ComputePipeline& pipeline1,
310 wgpu::ComputePipeline& pipeline2, wgpu::BindGroup& bindGroup1, wgpu::BindGroup& bindGroup2,
311 const uint8_t color_space
312) {
313 int bytesPerPixel {16};
314 const int32_t num_pixels {pixels.getSize()};
315
316 wgpu::TextureDescriptor texDesc = {};
317 texDesc.size = {static_cast<uint32_t>(width), static_cast<uint32_t>(height), 1};
318 texDesc.format = wgpu::TextureFormat::RGBA32Float;
319 texDesc.usage = wgpu::TextureUsage::TextureBinding | wgpu::TextureUsage::CopyDst;
320 texDesc.label = "inputTexture";
321 inputTexture = GPU::getClassInstance().get_device().CreateTexture(&texDesc);
322
323 wgpu::TexelCopyTextureInfo dst = {};
324 dst.texture = inputTexture;
325 wgpu::TexelCopyBufferLayout layout = {};
326 layout.offset = 0;
327 layout.bytesPerRow = width * bytesPerPixel; // Tightly packed for upload
328 layout.rowsPerImage = height;
329
330 std::vector<float> pixels_;
331 for (int i = 0; i < num_pixels; i++) {
332 switch (color_space) {
333 case COLOR_SPACE_OPTION_RGB: {
334 auto p = pixels[i];
335 pixels_.push_back(p.red / 255.0f);
336 pixels_.push_back(p.green / 255.0f);
337 pixels_.push_back(p.blue / 255.0f);
338 pixels_.push_back(p.alpha / 255.0f);
339 break;
340 }
341 case COLOR_SPACE_OPTION_CIELAB: {
342 auto p = lab[i];
343 pixels_.push_back(p.l / 255.0f);
344 pixels_.push_back(p.a / 255.0f);
345 pixels_.push_back(p.b / 255.0f);
346 pixels_.push_back(p.alpha / 255.0f);
347 break;
348 }
349 }
350 }
351
352 GPU::getClassInstance().get_queue().WriteTexture(
353 &dst, pixels_.data(), pixels_.size() * sizeof(float), &layout, &texDesc.size
354 );
355
356 // centroids
357 centroidDesc.size = {static_cast<uint32_t>(k), 1, 1};
358 centroidDesc.format = wgpu::TextureFormat::RGBA32Float;
359 centroidDesc.usage = wgpu::TextureUsage::TextureBinding | wgpu::TextureUsage::StorageBinding |
360 wgpu::TextureUsage::CopyDst | wgpu::TextureUsage::CopySrc;
361 centroidDesc.label = "centroidTexture";
362 centroidTexture = GPU::getClassInstance().get_device().CreateTexture(&centroidDesc);
363
364 wgpu::TexelCopyTextureInfo cdst = {};
365 cdst.texture = centroidTexture;
366 wgpu::TexelCopyBufferLayout clayout = {};
367 clayout.offset = 0;
368 clayout.bytesPerRow = k * bytesPerPixel; // Tightly packed for upload
369 clayout.rowsPerImage = 1;
370
371 std::vector<float> centroids_; // rgba
372 switch (color_space) {
373 case COLOR_SPACE_OPTION_RGB: {
374 for (int i = 0; i < k; i++) {
375 auto p = centroids[i];
376 centroids_.push_back(p.red / 255.0f);
377 centroids_.push_back(p.green / 255.0f);
378 centroids_.push_back(p.blue / 255.0f);
379 centroids_.push_back(p.alpha / 255.0f);
380 }
381 break;
382 }
383 case COLOR_SPACE_OPTION_CIELAB: {
384 for (int i = 0; i < k; i++) {
385 auto p = centroids_lab[i];
386 centroids_.push_back(p.l / 255.0f);
387 centroids_.push_back(p.a / 255.0f);
388 centroids_.push_back(p.b / 255.0f);
389 centroids_.push_back(p.alpha / 255.0f);
390 }
391 break;
392 }
393 }
394
395 GPU::getClassInstance().get_queue().WriteTexture(
396 &cdst, centroids_.data(), centroids_.size() * sizeof(float), &clayout, &centroidDesc.size
397 );
398
399 // labels
400 labelDesc.size = {static_cast<uint32_t>(width), static_cast<uint32_t>(height), 1};
401 labelDesc.format = wgpu::TextureFormat::RGBA32Uint;
402 labelDesc.usage = wgpu::TextureUsage::TextureBinding | wgpu::TextureUsage::StorageBinding |
403 wgpu::TextureUsage::CopyDst | wgpu::TextureUsage::CopySrc;
404 labelDesc.label = "labelTexture";
405 labelTexture = GPU::getClassInstance().get_device().CreateTexture(&labelDesc);
406
407 // params
408 Params params = {static_cast<uint32_t>(num_pixels), static_cast<uint32_t>(k)};
409 wgpu::BufferDescriptor bufDesc = {};
410 bufDesc.size = sizeof(Params);
411 bufDesc.usage = wgpu::BufferUsage::Uniform | wgpu::BufferUsage::CopyDst;
412 wgpu::Buffer paramBuffer = GPU::getClassInstance().get_device().CreateBuffer(&bufDesc);
413 GPU::getClassInstance().get_queue().WriteBuffer(paramBuffer, 0, &params, sizeof(Params));
414
415 // centroid accumulator
416 std::vector<ClusterAccumulator> reset_centroids(k, {0, 0, 0, 0});
417 wgpu::BufferDescriptor accDesc = {};
418 accDesc.size = sizeof(ClusterAccumulator) * k;
419 accDesc.usage = wgpu::BufferUsage::Storage | wgpu::BufferUsage::CopyDst;
420 wgpu::Buffer accBuffer = GPU::getClassInstance().get_device().CreateBuffer(&accDesc);
421 GPU::getClassInstance().get_queue().WriteBuffer(
422 accBuffer, 0, reset_centroids.data(), accDesc.size
423 );
424
425 // shaders
426 pipeline1 =
427 GPU::getClassInstance().createPipeline("assign_update_shader", "assignUpdateShader");
428 pipeline2 = GPU::getClassInstance().createPipeline("resolve_shader", "resolveShader");
429
430 // binding groups
431 wgpu::BindGroupDescriptor bindGroupDesc1 = {};
432 bindGroupDesc1.layout = pipeline1.GetBindGroupLayout(0);
433 wgpu::BindGroupEntry entries1[5]; // 4
434 // Entry 0: Input Texture View
435 entries1[0].binding = 0;
436 entries1[0].textureView = inputTexture.CreateView();
437 // Entry 1: Centroid Texture View
438 entries1[1].binding = 1;
439 entries1[1].textureView = centroidTexture.CreateView();
440 // Entry 2: Label Texture View
441 entries1[2].binding = 2;
442 entries1[2].textureView = labelTexture.CreateView();
443 // Entry 2: Uniform Buffer
444 entries1[3].binding = 3;
445 entries1[3].buffer = paramBuffer;
446 entries1[3].size = sizeof(Params);
447
448 entries1[4].binding = 4;
449 entries1[4].buffer = accBuffer;
450 entries1[4].size = sizeof(ClusterAccumulator) * k;
451
452 bindGroupDesc1.entryCount = 5; // 4;
453 bindGroupDesc1.entries = entries1;
454 bindGroup1 = GPU::getClassInstance().get_device().CreateBindGroup(&bindGroupDesc1);
455
456 wgpu::BindGroupDescriptor bindGroupDesc2 = {};
457 bindGroupDesc2.layout = pipeline2.GetBindGroupLayout(0);
458 wgpu::BindGroupEntry entries2[2];
459 entries2[0].binding = 0;
460 entries2[0].buffer = accBuffer;
461 entries2[0].size = accDesc.size;
462 entries2[1].binding = 1;
463 entries2[1].textureView = centroidTexture.CreateView();
464 bindGroupDesc2.entryCount = 2;
465 bindGroupDesc2.entries = entries2;
466 bindGroup2 = GPU::getClassInstance().get_device().CreateBindGroup(&bindGroupDesc2);
467}
468
469void kmeans_gpu(
470 const uint8_t* data, uint8_t* out_data, int32_t* out_labels, const int32_t width,
471 const int32_t height, const int32_t k, const int32_t max_iter, const uint8_t color_space
472) {
474 pixels.loadFromBuffer(data, width, height, ImageLib::RGBA_CONVERTER<float>);
475 const int32_t num_pixels {pixels.getSize()};
476
477 // width = k, height = 1
478 // k centroids, initialized to rgba(0,0,0,255)
479 // Init of each pixel is from default in Image constructor
481 ImageLib::Image<ImageLib::LABAPixel<float>> centroids_lab {k, 1};
482 std::vector<int32_t> labels(num_pixels, -1);
483
484 ImageLib::Image<ImageLib::LABAPixel<float>> lab(pixels.getWidth(), pixels.getHeight());
485
486 if (color_space == COLOR_SPACE_OPTION_CIELAB) {
487 for (int i {0}; i < pixels.getSize(); ++i) {
488 rgb_to_lab<float, float>(pixels[i], lab[i]);
489 }
490 }
491
492 IMG2NUM_LOG_INFO("starting");
493 // Step 2: Initialize centroids
494
495 switch (color_space) {
496 case COLOR_SPACE_OPTION_RGB: {
497 kMeansPlusPlusInitGpu<ImageLib::RGBAPixel<float>>(pixels, centroids, k, color_space);
498 break;
499 }
500 case COLOR_SPACE_OPTION_CIELAB: {
501 kMeansPlusPlusInitGpu<ImageLib::LABAPixel<float>>(lab, centroids_lab, k, color_space);
502 break;
503 }
504 }
505
506 IMG2NUM_LOG_INFO("kmeans++ init done");
507 // Step 3: Run k-means iterations
508
509 int bytesPerPixel {16}; // float pixels
510
511 // shaders - 2 pipelines:
512 // 1. assign and update clusters
513 // 2. resolve cluster centroids
514 wgpu::ComputePipeline pipeline1;
515 wgpu::ComputePipeline pipeline2;
516 wgpu::BindGroup bindGroup1;
517 wgpu::BindGroup bindGroup2;
518 wgpu::Texture inputTexture;
519 wgpu::Texture labelTexture;
520 wgpu::Texture centroidTexture;
521 wgpu::TextureDescriptor labelDesc = {};
522 wgpu::TextureDescriptor centroidDesc = {};
523
524 // setup all textures and buffers needed for the kmeans loop on gpu
525 setup(
526 pixels, lab, centroids, centroids_lab, width, height, k, inputTexture, labelTexture,
527 centroidTexture, labelDesc, centroidDesc, pipeline1, pipeline2, bindGroup1, bindGroup2,
528 color_space
529 );
530
531 uint32_t wgX = (width + 15) / 16;
532 uint32_t wgY = (height + 15) / 16;
533
534 // Label Readback RGBA32Uint is 16 bytes/ pixel
535 uint32_t bytesPerRowLabels =
536 GPU::getAlignedBytesPerRow(width, static_cast<uint32_t>(bytesPerPixel));
537 wgpu::BufferDescriptor readLabelsDesc = {};
538 readLabelsDesc.size = bytesPerRowLabels * height;
539 readLabelsDesc.usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst;
540 wgpu::Buffer readLabelsBuffer =
541 GPU::getClassInstance().get_device().CreateBuffer(&readLabelsDesc);
542
543 // Centroid Readback
544 uint32_t bytesPerRowCentroids =
545 GPU::getAlignedBytesPerRow(width, static_cast<uint32_t>(bytesPerPixel));
546 wgpu::BufferDescriptor readCentroidsDesc = {};
547 readCentroidsDesc.size = bytesPerRowCentroids; // Height is 1
548 readCentroidsDesc.usage = wgpu::BufferUsage::MapRead | wgpu::BufferUsage::CopyDst;
549 wgpu::Buffer readCentroidsBuffer =
550 GPU::getClassInstance().get_device().CreateBuffer(&readCentroidsDesc);
551
552 // This is the actual KMeans loop
553 IMG2NUM_LOG_INFO("start iterations");
554 wgpu::CommandEncoder encoder = GPU::getClassInstance().get_device().CreateCommandEncoder();
555 for (int32_t iter {0}; iter < max_iter; ++iter) {
556 wgpu::ComputePassEncoder pass1 = encoder.BeginComputePass();
557 pass1.SetPipeline(pipeline1);
558 pass1.SetBindGroup(0, bindGroup1);
559 pass1.DispatchWorkgroups(wgX, wgY);
560 pass1.End();
561
562 wgpu::ComputePassEncoder pass2 = encoder.BeginComputePass();
563 pass2.SetPipeline(pipeline2);
564 pass2.SetBindGroup(0, bindGroup2);
565 pass2.DispatchWorkgroups((k + 255) / 256, 1);
566 pass2.End();
567 }
568
569 // 3. Readback (After Loop Finishes)
570
571 // Copy Labels
572 wgpu::TexelCopyTextureInfo srcLabels = {};
573 srcLabels.texture = labelTexture;
574 wgpu::TexelCopyBufferInfo dstLabels = {};
575 dstLabels.buffer = readLabelsBuffer;
576 dstLabels.layout.bytesPerRow = bytesPerRowLabels;
577 dstLabels.layout.rowsPerImage = height;
578 encoder.CopyTextureToBuffer(&srcLabels, &dstLabels, &labelDesc.size);
579
580 // Copy Centroids
581 wgpu::TexelCopyTextureInfo srcCentroids = {};
582 srcCentroids.texture = centroidTexture;
583 wgpu::TexelCopyBufferInfo dstCentroids = {};
584 dstCentroids.buffer = readCentroidsBuffer;
585 dstCentroids.layout.bytesPerRow = bytesPerRowCentroids;
586 dstCentroids.layout.rowsPerImage = 1;
587 encoder.CopyTextureToBuffer(&srcCentroids, &dstCentroids, &centroidDesc.size);
588
589 wgpu::CommandBuffer commands = encoder.Finish();
590 GPU::getClassInstance().get_queue().Submit(1, &commands);
591 IMG2NUM_LOG_INFO("done iterations");
592
593 // 4. Map Async & Wait
594 bool* done1 = new bool(false);
595 bool* done2 = new bool(false);
596
597 // Map Labels
598 readLabelsBuffer.MapAsync(
599 wgpu::MapMode::Read, 0, readLabelsDesc.size, wgpu::CallbackMode::AllowProcessEvents,
600 [](wgpu::MapAsyncStatus status, wgpu::StringView msg, void* userdata) {
601 bool* flag = static_cast<bool*>(userdata);
602 bool success = false;
603 if (status == wgpu::MapAsyncStatus::Success) {
604 success = true;
605 }
606 *flag = true;
607 },
608 (void*)done1
609 );
610
611 IMG2NUM_LOG_INFO("read out");
612
613 while (!*done1) {
614 GPU::getClassInstance().get_instance().ProcessEvents();
615#if defined(__EMSCRIPTEN__)
616 emscripten_sleep(10);
617#endif
618 }
619
620 IMG2NUM_LOG_INFO("mapping labels");
621 const uint8_t* mappedData = (const uint8_t*)readLabelsBuffer.GetConstMappedRange();
622 // ... Copy data to your C++ vector ...
623 // Copy row by row to remove padding and put data into 'result'
624 for (size_t y = 0; y < height; ++y) {
625 const uint8_t* rowPtr = mappedData + (y * bytesPerRowLabels);
626 for (size_t x = 0; x < width; ++x) {
627 const uint8_t* pixelPtr = rowPtr + (x * bytesPerPixel);
628 uint32_t r = 0;
629 std::memcpy(&r, pixelPtr, sizeof(uint32_t));
630
631 size_t dstIndex = y * width + x;
632 labels[dstIndex] = static_cast<int32_t>(r);
633 }
634 }
635
636 readLabelsBuffer.Unmap();
637
638 // Map Centroids
639 readCentroidsBuffer.MapAsync(
640 wgpu::MapMode::Read, 0, readCentroidsDesc.size, wgpu::CallbackMode::AllowProcessEvents,
641 [](wgpu::MapAsyncStatus status, wgpu::StringView msg, void* userdata) {
642 bool* flag = static_cast<bool*>(userdata);
643 bool success = false;
644 if (status == wgpu::MapAsyncStatus::Success) {
645 success = true;
646 }
647 *flag = true; // Signal completion
648 },
649 (void*)done2
650 );
651
652 while (!*done2) {
653 GPU::getClassInstance().get_instance().ProcessEvents();
654#if defined(__EMSCRIPTEN__)
655 emscripten_sleep(10);
656#endif
657 }
658
659 IMG2NUM_LOG_INFO("mapping centroids");
660 const float* mappedDataFloat = (const float*)readCentroidsBuffer.GetConstMappedRange();
661 // ... Copy data to your C++ vector ...
662
663 for (int i = 0; i < k; i++) {
664 // if CIELAB color space these represent l, a, b, alpha
665 const float* centroidPtr = mappedDataFloat + (i * 4);
666
667 float r = *(centroidPtr);
668 float g = *(centroidPtr + 1);
669 float b = *(centroidPtr + 2);
670 float a = *(centroidPtr + 3);
671 switch (color_space) {
672 case COLOR_SPACE_OPTION_RGB: {
673 centroids[i] = ImageLib::RGBAPixel<float>(r * 255.f, g * 255.f, b * 255.f, a * 255.f);
674 break;
675 }
676 case COLOR_SPACE_OPTION_CIELAB: {
677 centroids_lab[i] =
678 ImageLib::LABAPixel<float>(r * 255.f, g * 255.f, b * 255.f, a * 255.f);
679 break;
680 }
681 }
682 }
683
684 readCentroidsBuffer.Unmap();
685
686 // Write the final centroid values to each pixel in the cluster
687 if (color_space == COLOR_SPACE_OPTION_CIELAB) {
688 for (int32_t i {0}; i < k; ++i) {
689 lab_to_rgb<float, float>(centroids_lab[i], centroids[i]);
690 }
691 }
692
693 for (int32_t i = 0; i < num_pixels; ++i) {
694 const int32_t cluster = labels[i];
695 out_data[i * 4 + 0] = static_cast<uint8_t>(centroids[cluster].red);
696 out_data[i * 4 + 1] = static_cast<uint8_t>(centroids[cluster].green);
697 out_data[i * 4 + 2] = static_cast<uint8_t>(centroids[cluster].blue);
698 out_data[i * 4 + 3] = 255;
699 }
700
701 // Write labels to out_labels
702 IMG2NUM_LOG_INFO("copying labels out");
703 std::memcpy(out_labels, labels.data(), labels.size() * sizeof(int32_t));
704
705 if (inputTexture)
706 inputTexture.Destroy();
707 if (labelTexture)
708 labelTexture.Destroy();
709 if (centroidTexture)
710 centroidTexture.Destroy();
711 readLabelsBuffer.Destroy();
712 readCentroidsBuffer.Destroy();
713 delete done1;
714 delete done2;
715
716 labels.clear();
717 labels.shrink_to_fit();
718#if defined(__EMSCRIPTEN__)
719 emscripten_sleep(50);
720#endif
721}
Core image processing functions for img2num project.