Ansel 0.0
A darktable fork - bloat + design vision
Loading...
Searching...
No Matches
test_drawlayer_batch_raster.c
Go to the documentation of this file.
1/*
2 This file is part of the Ansel project.
3 Copyright (C) 2026 Aurélien PIERRE.
4
5 Ansel is free software: you can redistribute it and/or modify
6 it under the terms of the GNU General Public License as published by
7 the Free Software Foundation, either version 3 of the License, or
8 (at your option) any later version.
9
10 Ansel is distributed in the hope that it will be useful,
11 but WITHOUT ANY WARRANTY; without even the implied warranty of
12 MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
13 GNU General Public License for more details.
14
15 You should have received a copy of the GNU General Public License
16 along with Ansel. If not, see <http://www.gnu.org/licenses/>.
17*/
18
34#include "iop/drawlayer/brush.h"
35#include "iop/drawlayer/cache.h"
36#include "iop/drawlayer/paint.h"
37
38#include <setjmp.h>
39#include <stdarg.h>
40#include <stddef.h>
41#include <stdint.h>
42#include <stdlib.h>
43#include <string.h>
44#include <math.h>
45#include <time.h>
46#ifdef _OPENMP
47#include <omp.h>
48#endif
49#include <cmocka.h>
50
51#define W 96
52#define H 96
53
59
60static void _plane_init(plane_t *p)
61{
62 p->patch = (dt_drawlayer_cache_patch_t){ .x = 0, .y = 0, .width = W, .height = H,
63 .pixels = calloc((size_t)W * H * 4, sizeof(float)),
64 .external_alloc = TRUE };
65 p->mask = (dt_drawlayer_cache_patch_t){ .x = 0, .y = 0, .width = W, .height = H,
66 .pixels = calloc((size_t)W * H, sizeof(float)),
67 .external_alloc = TRUE };
68 assert_non_null(p->patch.pixels);
69 assert_non_null(p->mask.pixels);
70}
71
72static void _plane_free(plane_t *p)
73{
74 free(p->patch.pixels);
75 free(p->mask.pixels);
76}
77
79static void _build_dabs_tex(dt_drawlayer_brush_dab_t *dabs, const int count, const int mode,
80 const float opacity, const float radius, const float step,
81 const float sprinkles)
82{
83 for(int i = 0; i < count; i++)
84 {
86 .x = 24.0f + step * (float)i,
87 .y = 48.0f,
88 .radius = radius,
89 .dir_x = 1.0f,
90 .dir_y = 0.0f,
91 .sample_spacing = step,
92 .sample_opacity_scale = 1.0f,
93 .opacity = opacity,
94 .flow = 1.0f, /* UI 100% -> internal 0, the regime the closed form covers */
95 .sprinkles = sprinkles,
96 .sprinkle_size = 3.0f,
97 .sprinkle_coarseness = 0.5f,
98 .hardness = 0.5f,
99 .color = { 0.8f, 0.4f, 0.2f, 1.0f },
101 .mode = mode,
102 .stroke_batch = 7u,
103 .stroke_pos = (uint8_t)(i == 0 ? DT_DRAWLAYER_PAINT_STROKE_FIRST
105 };
106 }
107}
108
109static void _build_dabs(dt_drawlayer_brush_dab_t *dabs, const int count, const int mode,
110 const float opacity, const float radius, const float step)
111{
112 _build_dabs_tex(dabs, count, mode, opacity, radius, step, 0.0f);
113}
114
116static void _run_serial(plane_t *p, const dt_drawlayer_brush_dab_t *dabs, const int count)
117{
119 assert_non_null(runtime);
120 for(int i = 0; i < count; i++)
121 dt_drawlayer_brush_rasterize(NULL, &p->patch, 1.0f, &dabs[i], 1.0f, &p->mask, runtime);
123}
124
125static void _run_batch(plane_t *p, const dt_drawlayer_brush_dab_t *dabs, const int count)
126{
127 dt_drawlayer_brush_batch_t batch = { 0 };
128 assert_true(dt_drawlayer_brush_batch_is_uniform(dabs, (guint)count, &batch));
129
130 float *transmittance = calloc((size_t)W * H, sizeof(float));
131 float *noise = calloc((size_t)W * H, sizeof(float));
134 assert_true(dt_drawlayer_brush_rasterize_batch(&batch, &p->patch, 1.0f, &p->mask, transmittance, noise, NULL));
135 free(transmittance);
136 free(noise);
137}
138
139static double _max_abs_diff(const float *a, const float *b, const size_t n)
140{
141 double worst = 0.0;
142 for(size_t i = 0; i < n; i++)
143 {
144 const double d = fabs((double)a[i] - (double)b[i]);
145 if(d > worst) worst = d;
146 }
147 return worst;
148}
149
150static void _compare_tex(const int mode, const float opacity, const float radius, const float step,
151 const int count, const float sprinkles)
152{
153 dt_drawlayer_brush_dab_t *dabs = calloc((size_t)count, sizeof(*dabs));
154 assert_non_null(dabs);
155 _build_dabs_tex(dabs, count, mode, opacity, radius, step, sprinkles);
156
157 plane_t serial, batch;
158 _plane_init(&serial);
159 _plane_init(&batch);
160
161 /* ERASE needs something to erase: seed both planes identically with opaque white. */
163 {
164 for(size_t i = 0; i < (size_t)W * H * 4; i++) serial.patch.pixels[i] = 1.0f;
165 memcpy(batch.patch.pixels, serial.patch.pixels, (size_t)W * H * 4 * sizeof(float));
166 }
167
168 _run_serial(&serial, dabs, count);
169 _run_batch(&batch, dabs, count);
170
171 const double pixel_diff = _max_abs_diff(serial.patch.pixels, batch.patch.pixels, (size_t)W * H * 4);
172 const double mask_diff = _max_abs_diff(serial.mask.pixels, batch.mask.pixels, (size_t)W * H);
173
174 /* Sanity: the dabs must actually have painted, or the comparison proves nothing. */
175 double painted = 0.0;
176 for(size_t i = 0; i < (size_t)W * H; i++) painted += batch.mask.pixels[i];
177 assert_true(painted > 1.0);
178
179 print_message("mode=%d opacity=%.2f r=%.1f step=%.2f n=%d sprinkles=%.2f -> "
180 "max |dpixel|=%.3e max |dmask|=%.3e\n",
181 mode, opacity, radius, step, count, sprinkles, pixel_diff, mask_diff);
182
183 /* Measured on x86-64 GCC with -ffast-math: 1.192e-07, i.e. ONE ulp at 1.0 in float32.
184 * The bound is two orders of magnitude looser to absorb platform variation, and still
185 * three orders tighter than any real algebraic mistake -- a dropped cap or a wrong order
186 * of operations moves this to O(0.1). */
187 assert_true(pixel_diff < 1e-5);
188 assert_true(mask_diff < 1e-5);
189
190 _plane_free(&serial);
191 _plane_free(&batch);
192 free(dabs);
193}
194
195static void _compare(const int mode, const float opacity, const float radius, const float step,
196 const int count)
197{
198 _compare_tex(mode, opacity, radius, step, count, 0.0f);
199}
200
210{
211 (void)state;
212 _compare_tex(DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.9f, 12.0f, 1.0f, 24, 0.6f);
213}
214
217{
218 (void)state;
219 _compare(DT_DRAWLAYER_BRUSH_MODE_PAINT, 1.0f, 12.0f, 1.0f, 32);
220}
221
224{
225 (void)state;
226 _compare(DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.25f, 12.0f, 1.0f, 40);
227}
228
231{
232 (void)state;
233 _compare(DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.6f, 8.0f, 11.0f, 6);
234}
235
237{
238 (void)state;
239 _compare(DT_DRAWLAYER_BRUSH_MODE_ERASE, 0.75f, 12.0f, 1.5f, 24);
240}
241
244{
245 (void)state;
247 dt_drawlayer_brush_batch_t batch = { 0 };
248
249 _build_dabs(dabs, 4, DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.8f, 10.0f, 2.0f);
251
252 /* A pressure-mapped opacity breaks the shared cap the single clamp depends on. */
253 _build_dabs(dabs, 4, DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.8f, 10.0f, 2.0f);
254 dabs[2].opacity = 0.4f;
256
257 /* Flow below 100% keeps `accum_alpha`, which does not collapse into a product. */
258 _build_dabs(dabs, 4, DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.8f, 10.0f, 2.0f);
259 dabs[1].flow = 0.5f;
261
262 /* A mid-batch colour change would make one composite wrong. */
263 _build_dabs(dabs, 4, DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.8f, 10.0f, 2.0f);
264 dabs[3].color[1] = 0.9f;
266
267 /* Mixed modes cannot share one composite. */
268 _build_dabs(dabs, 4, DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.8f, 10.0f, 2.0f);
271
272 /* The sprinkle field is shared, so its parameters are part of the contract. */
273 _build_dabs(dabs, 4, DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.8f, 10.0f, 2.0f);
274 dabs[2].sprinkles = 0.5f;
276
277 _build_dabs_tex(dabs, 4, DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.8f, 10.0f, 2.0f, 0.5f);
278 dabs[1].sprinkle_size = 9.0f;
280
281 /* SMUDGE and BLUR read the destination per dab and never qualify. */
282 _build_dabs(dabs, 4, DT_DRAWLAYER_BRUSH_MODE_SMUDGE, 0.8f, 10.0f, 2.0f);
284 _build_dabs(dabs, 4, DT_DRAWLAYER_BRUSH_MODE_BLUR, 0.8f, 10.0f, 2.0f);
286}
287
298{
299 (void)state;
300 const int count = 32;
301 dt_drawlayer_brush_dab_t *dabs = calloc((size_t)count, sizeof(*dabs));
302 assert_non_null(dabs);
303 _build_dabs(dabs, count, DT_DRAWLAYER_BRUSH_MODE_PAINT, 0.7f, 14.0f, 1.0f);
304
308
309#ifdef _OPENMP
310 const int saved = omp_get_max_threads();
312#endif
313 _run_batch(&one, dabs, count);
314#ifdef _OPENMP
315 omp_set_num_threads(saved > 1 ? saved : 8);
316#endif
317 _run_batch(&many, dabs, count);
318#ifdef _OPENMP
319 omp_set_num_threads(saved);
320#endif
321
322 assert_memory_equal(one.patch.pixels, many.patch.pixels, (size_t)W * H * 4 * sizeof(float));
323 assert_memory_equal(one.mask.pixels, many.mask.pixels, (size_t)W * H * sizeof(float));
324
327 free(dabs);
328}
329
338/* Written as a loop rather than memset() on purpose: SonarCloud reads any memset() that zeroes
339 * a buffer as the "clearing sensitive data" pattern and rates it a security finding, which for a
340 * benchmark scratch buffer it is not. The loop says the same thing and says it only once, so the
341 * quality gate is not spending anybody's attention on it. Both timed paths below clear the mask
342 * the same way, so whatever this costs cancels in the comparison. */
343static void _clear_mask(float *const mask, const size_t px)
344{
345 for(size_t i = 0; i < px; i++) mask[i] = 0.0f;
346}
347
348static void _report_speedup(const float sprinkles)
349{
350 const int count = 32;
351 const float radius = 64.0f;
352 const int plane = 512;
353
354 dt_drawlayer_brush_dab_t *dabs = calloc((size_t)count, sizeof(*dabs));
355 assert_non_null(dabs);
356 for(int i = 0; i < count; i++)
357 {
358 dabs[i] = (dt_drawlayer_brush_dab_t){
359 .x = 160.0f + (float)i, .y = 256.0f, .radius = radius, .dir_x = 1.0f, .dir_y = 0.0f,
360 .sample_spacing = 1.0f, .sample_opacity_scale = 1.0f, .opacity = 1.0f, .flow = 1.0f,
361 .sprinkles = sprinkles, .sprinkle_size = 3.0f, .sprinkle_coarseness = 0.5f, .hardness = 0.5f,
362 .color = { 0.8f, 0.4f, 0.2f, 1.0f }, .shape = DT_DRAWLAYER_BRUSH_SHAPE_LINEAR,
363 .mode = DT_DRAWLAYER_BRUSH_MODE_PAINT, .stroke_batch = 3u,
364 .stroke_pos = (uint8_t)(i == 0 ? DT_DRAWLAYER_PAINT_STROKE_FIRST
366 };
367 }
368
369 const size_t px = (size_t)plane * plane;
370 float *rgba = calloc(px * 4, sizeof(float));
371 float *mask = calloc(px, sizeof(float));
372 float *transmittance = calloc(px, sizeof(float));
373 float *noise = calloc(px, sizeof(float));
375 assert_non_null(mask);
378
379 dt_drawlayer_cache_patch_t patch = { .width = plane, .height = plane, .pixels = rgba,
380 .external_alloc = TRUE };
381 dt_drawlayer_cache_patch_t mpatch = { .width = plane, .height = plane, .pixels = mask,
382 .external_alloc = TRUE };
383
384 const int rounds = 40;
385 struct timespec t0, t1;
386
388 assert_non_null(runtime);
390 for(int r = 0; r < rounds; r++)
391 {
392 _clear_mask(mask, px);
393 for(int i = 0; i < count; i++)
394 dt_drawlayer_brush_rasterize(NULL, &patch, 1.0f, &dabs[i], 1.0f, &mpatch, runtime);
395 }
397 const double serial_ms = ((double)(t1.tv_sec - t0.tv_sec) * 1e3
398 + (double)(t1.tv_nsec - t0.tv_nsec) / 1e6) / (double)rounds;
400
401 dt_drawlayer_brush_batch_t batch = { 0 };
402 assert_true(dt_drawlayer_brush_batch_is_uniform(dabs, (guint)count, &batch));
404 for(int r = 0; r < rounds; r++)
405 {
406 _clear_mask(mask, px);
408 }
410 const double batch_ms = ((double)(t1.tv_sec - t0.tv_sec) * 1e3
411 + (double)(t1.tv_nsec - t0.tv_nsec) / 1e6) / (double)rounds;
412
413 print_message("one heartbeat batch, r=%.0f spacing=1 dabs=%d sprinkles=%.2f: "
414 "per-dab %.3f ms, batch %.3f ms (%.1fx)\n",
415 radius, count, sprinkles, serial_ms, batch_ms, serial_ms / fmax(batch_ms, 1e-9));
416
417 free(rgba);
418 free(mask);
419 free(transmittance);
420 free(noise);
421 free(dabs);
422}
423
425{
426 (void)state;
427 _report_speedup(0.0f);
428 _report_speedup(0.6f);
429}
430
#define TRUE
Definition ashift_lsd.c:162
Dab-level brush rasterization API for drawlayer.
@ DT_DRAWLAYER_BRUSH_MODE_ERASE
Definition brush.h:51
@ DT_DRAWLAYER_BRUSH_MODE_PAINT
Definition brush.h:50
@ DT_DRAWLAYER_BRUSH_MODE_BLUR
Definition brush.h:52
@ DT_DRAWLAYER_BRUSH_MODE_SMUDGE
Definition brush.h:53
@ DT_DRAWLAYER_BRUSH_SHAPE_LINEAR
Definition brush.h:41
typedef void((*dt_cache_allocate_t)(void *userdata, dt_cache_entry_t *entry))
GdkRGBA color[]
Definition geotagging.c:541
gboolean dt_drawlayer_brush_batch_is_uniform(const dt_drawlayer_brush_dab_t *dabs, const guint count, dt_drawlayer_brush_batch_t *out)
Decide whether a run of dabs can take the batch path, and describe it if so.
gboolean dt_drawlayer_brush_rasterize(const dt_drawlayer_cache_patch_t *sample_patch, dt_drawlayer_cache_patch_t *patch, const float scale, const dt_drawlayer_brush_dab_t *dab, const float sample_opacity_scale, dt_drawlayer_cache_patch_t *stroke_mask, dt_drawlayer_paint_stroke_t *runtime_private)
Public dab rasterization entry point.
gboolean dt_drawlayer_brush_rasterize_batch(const dt_drawlayer_brush_batch_t *batch, dt_drawlayer_cache_patch_t *patch, const float scale, dt_drawlayer_cache_patch_t *stroke_mask, float *const transmittance, float *const noise_scratch, dt_drawlayer_damaged_rect_t *batch_damage)
Rasterize a whole batch in two passes: accumulate transmittance, then composite once.
Patch/cache helpers for drawlayer process and preview buffers.
void dt_drawlayer_paint_runtime_private_destroy(dt_drawlayer_paint_stroke_t **state)
Destroy stroke runtime payload and null pointer.
dt_drawlayer_paint_stroke_t * dt_drawlayer_paint_runtime_private_create(void)
Allocate stroke runtime payload object used by paint+brush internals.
Stroke-level path sampling and runtime-state API for drawlayer.
@ DT_DRAWLAYER_PAINT_STROKE_MIDDLE
@ DT_DRAWLAYER_PAINT_STROKE_FIRST
#define omp_get_max_threads()
Definition openmp.h:90
const float uint32_t state[4]
const float r
const float noise
A run of dabs that share everything the batch composite needs to be a single step.
Definition brush.h:120
Fully resolved input dab descriptor.
Definition brush.h:66
Generic float RGBA patch stored either in malloc memory or pixel cache.
Mutable stroke runtime state owned by worker/backend code.
dt_drawlayer_cache_patch_t mask
dt_drawlayer_cache_patch_t patch
typedef double((*spd)(unsigned long int wavelength, double TempK))
static void _plane_free(plane_t *p)
static void test_batch_matches_serial_paint_capped(void **state)
The cap binds mid-batch: opacity well below 1 with many overlapping dabs.
static void _run_serial(plane_t *p, const dt_drawlayer_brush_dab_t *dabs, const int count)
Run the reference path: one dt_drawlayer_brush_rasterize per dab, in order.
static void _compare_tex(const int mode, const float opacity, const float radius, const float step, const int count, const float sprinkles)
static void test_batch_is_thread_count_independent(void **state)
The batch result must not depend on the thread count.
static void _build_dabs(dt_drawlayer_brush_dab_t *dabs, const int count, const int mode, const float opacity, const float radius, const float step)
static void test_report_batch_speedup(void **state)
static void test_batch_matches_serial_erase(void **state)
static void _clear_mask(float *const mask, const size_t px)
Time both paths on the shipped regime, so the speedup is measured and not derived.
static void _build_dabs_tex(dt_drawlayer_brush_dab_t *dabs, const int count, const int mode, const float opacity, const float radius, const float step, const float sprinkles)
A short stroke of overlapping dabs, uniform in everything the gate tests.
static void _compare(const int mode, const float opacity, const float radius, const float step, const int count)
static double _max_abs_diff(const float *a, const float *b, const size_t n)
int main(void)
static void test_batch_matches_serial_sprinkles(void **state)
Texture on: the batch shares ONE sprinkle field across every dab.
static void _plane_init(plane_t *p)
static void test_batch_matches_serial_paint_dense(void **state)
Heavy overlap, the shipped regime: spacing far below the diameter.
static void _run_batch(plane_t *p, const dt_drawlayer_brush_dab_t *dabs, const int count)
static void test_batch_matches_serial_paint_sparse(void **state)
Sparse dabs, barely overlapping.
static void _report_speedup(const float sprinkles)
static void test_gate_refuses_non_uniform(void **state)
The gate must refuse everything the closed form is not derived for.