Ansel 0.0
A darktable fork - bloat + design vision
Loading...
Searching...
No Matches
rawdenoiseai.c
Go to the documentation of this file.
1/*
2 This file is part of Ansel,
3 Copyright (C) 2026 Aurélien PIERRE.
4
5 Ansel is free software: you can redistribute it and/or modify
6 it under the terms of the GNU General Public License as published by
7 the Free Software Foundation, either version 3 of the License, or
8 (at your option) any later version.
9
10 Ansel is distributed in the hope that it will be useful,
11 but WITHOUT ANY WARRANTY; without even the implied warranty of
12 MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
13 GNU General Public License for more details.
14
15 You should have received a copy of the GNU General Public License
16 along with Ansel. If not, see <http://www.gnu.org/licenses/>.
17*/
18
19/*
20 * Neural raw (CFA-domain) denoiser.
21 *
22 * Runs a sigma-map-conditioned U-Net (common/nn_model.{h,c}) on the mosaiced
23 * buffer, before demosaicing, where sensor noise is still per-sensel
24 * independent. The network was trained on synthetic Poisson-Gaussian noise
25 * drawn from the community noise-profile database, conditioned on the
26 * per-pixel noise standard deviation — so one set of weights serves every
27 * profiled camera, Bayer and X-Trans alike, and a newly profiled camera is
28 * supported without retraining. Training pipeline:
29 * https://github.com/aurelienpierreeng/ansel-denoise
30 *
31 * The sigma map is computed from the per-channel variance line
32 * Var(x) = a*x + b of the matched noise profile at the image ISO, applied in
33 * the post-rawprepare normalized domain — the exact domain the profiles were
34 * fitted in and the training used. Do NOT copy denoiseprofile's white-balance
35 * adjustments here: those compensate its post-demosaic, post-WB position.
36 *
37 * User parameters:
38 * - "strength": opacity of the correction, an alpha blend of the inferred
39 * noise residual: out = in + strength * (denoised - in).
40 * - model version / size (large|half|quarter network width) / variant (single-scale
41 * or multiscale) select the weights file
42 * denoise-<size>-<single|multi>-<version>.anselnn.
43 * - "global correction" and the per-channel R/G/B corrections scale the
44 * sigma map (see the calibration note above the params struct).
45 *
46 * Weights are loaded once per session from the user config dir (override for
47 * testing) or <datadir>/. Without a weights file the module stays disabled.
48 * The training counterpart of every inference step lives in the
49 * ansel-denoise repository: _k_assemble() <-> dataset.py,
50 * _k_bin_planes() <-> cfa.bin_mosaic_torch()/bin_sigma_torch(),
51 * the coarse->fine guide flow <-> train.ms_forward(), and
52 * _apply_low_band_anchor() <-> cfa.fuse_low_bands(). Keep them in sync.
53 */
54
55#ifdef HAVE_CONFIG_H
56#include "config.h"
57#endif
58#include "widgets/bauhaus.h"
59#include "common/paths.h" // DT_PATH_MAX
60#include "system/macros.h"
61#include "system/openmp.h"
63#include "system/mem_alloc.h"
64#include "common/logging.h"
66#include <json-glib/json-glib.h>
69#include "common/imagebuf.h"
70#include "common/nn_model.h"
72#include "common/opencl.h"
73#include "develop/imageop.h"
74#include "develop/imageop_gui.h"
76#include "develop/tiling.h"
77#include "iop/iop_api.h"
78
79#include <gtk/gtk.h>
80#include <stdlib.h>
82#include "widgets/label.h"
83
84#define DT_RAWDENOISEAI_MODEL_LEN 128
85
87
88/* Model version: pins which trained network a history entry uses, so results
89 * stay reproducible across app updates. A new release that retrains the net
90 * appends a value here (and ships the matching weights file) instead of
91 * replacing v1 — old edits keep rendering with the model they were made on. */
96
97/* Model size: same architecture family, different width of the FINE net
98 * (32 / 16 / 8; the coarse chroma net of a multiscale model stays at 32 for
99 * every size — it runs on the superpixel-binned image and costs a few
100 * percent). large is the reference quality (practical with OpenCL), half is
101 * ~4x cheaper for the CPU path, quarter ~4x cheaper again for very weak
102 * hardware and near-realtime editing. The outputs are NOT interchangeable,
103 * hence a user parameter rather than a silent runtime choice. */
105{
106 DT_RAWDENOISEAI_LARGE = 0, // $DESCRIPTION: "large"
107 DT_RAWDENOISEAI_HALF = 1, // $DESCRIPTION: "half"
108 DT_RAWDENOISEAI_QUARTER = 2, // $DESCRIPTION: "quarter"
110
111/* Model variant: single-scale (the fine mosaic net alone — fast, no
112 * low-frequency chroma handling) vs multiscale (coarse chroma net guiding
113 * the fine net, plus the hybrid low-band fusion — high quality). */
115{
116 DT_RAWDENOISEAI_SINGLE = 0, // $DESCRIPTION: "single-scale"
117 DT_RAWDENOISEAI_MULTI = 1, // $DESCRIPTION: "multiscale"
119
120#define DT_RAWDENOISEAI_NUM_VERSIONS 1
121#define DT_RAWDENOISEAI_NUM_SIZES 3
122#define DT_RAWDENOISEAI_NUM_SCALES 2
123
124// filename components per enum value; the weights file is
125// denoise-<size>-<single|multi>-<version>.anselnn
126static const char *const _version_tag[DT_RAWDENOISEAI_NUM_VERSIONS] = { "v1" };
127static const char *const _size_tag[DT_RAWDENOISEAI_NUM_SIZES] = { "large", "half", "quarter" };
128static const char *const _scale_tag[DT_RAWDENOISEAI_NUM_SCALES] = { "single", "multi" };
129
130/* The shipped noise profiles understate the true mosaic-domain sigma by an
131 * exact factor 2 before any demosaic effect: tools/noise/noiseprofile.c
132 * estimates sigma as MAD/0.6745 of the HH band of a decimated lifting Haar
133 * whose normalization is HH = (x00 - x01 - x10 + x11)/4, so std(HH) = sigma/2
134 * for iid noise (the orthonormal Haar assumed by the MAD rule divides by 2).
135 * The gnuplot fit in ansel-gen-noiseprofile squares that std into (a, b)
136 * without correction, so every profile carries 1/4 of the physical variance.
137 * Historical consumers (denoiseprofile) were tuned end to end around these
138 * units; this module is the first to treat (a, b) as absolute physical
139 * variance, so the correction lives here — the shared profile database must
140 * stay consistent with a decade of fits and cannot change.
141 *
142 * The remaining, channel-dependent part of the deviation (the profiles are
143 * fitted on demosaiced pixels, and interpolation averages away high-frequency
144 * noise — most on the dense green lattice) is exposed as the per-channel
145 * corrections below, calibrated by measuring flat-region noise on raw mosaics
146 * against the profile prediction: 253 profiled cameras (one raw.pixls.us
147 * sample each) plus 64 images across ISO 64-12800 on three local bodies.
148 * Cross-camera medians (estimator-bias corrected): R 1.41, G 1.97, B 1.48
149 * after the factor 2 above; the deviation is ISO-stable.
150 *
151 * The whole correction is carried by the three GUI sliders and nothing
152 * else: what the user sees is exactly what multiplies the profile sigma
153 * (times the global correction). No hidden constants, no model-side
154 * multiplication — a model's cfg may document the sigma convention it was
155 * trained under, but the module never applies it behind the user's back
156 * (a hidden factor stacked with visible sliders is how the calibration got
157 * silently applied twice — the yellow-cast field bug). The slider defaults
158 * are the calibration itself: 2 x the sweep medians above. */
159
161{
162 float strength; // $MIN: 0.0 $MAX: 1.0 $DEFAULT: 0.85 $DESCRIPTION: "strength"
163 dt_iop_rawdenoiseai_version_t version; // $DEFAULT: DT_RAWDENOISEAI_V1 $DESCRIPTION: "model version"
164 dt_iop_rawdenoiseai_size_t size; // $DEFAULT: DT_RAWDENOISEAI_QUARTER $DESCRIPTION: "model size"
165 float noise_level; // $MIN: 0.0 $MAX: 2.0 $DEFAULT: 1.0 $DESCRIPTION: "global correction"
166 float sigma_red; // $MIN: 0.5 $MAX: 8.0 $DEFAULT: 2.82 $DESCRIPTION: "red correction"
167 float sigma_green; // $MIN: 0.5 $MAX: 8.0 $DEFAULT: 3.94 $DESCRIPTION: "green correction"
168 float sigma_blue; // $MIN: 0.5 $MAX: 8.0 $DEFAULT: 2.96 $DESCRIPTION: "blue correction"
169 dt_iop_rawdenoiseai_scale_t scale_variant; // $DEFAULT: DT_RAWDENOISEAI_MULTI $DESCRIPTION: "model variant"
170 /* Empty: use the shipped model selected by (version, size, scale) above.
171 * Otherwise the basename of a .anselnn in the user config dir, which
172 * overrides all three. Stored by NAME, never by list position: the set of
173 * files on disk changes between sessions, and an index would silently
174 * re-point every history entry that used it (same reason colorin stores
175 * its ICC filename). */
178
193
195{
196 float strength; // opacity of the correction: out = in + strength * (denoised - in)
197 float noise_level; // scales the sigma map fed to the network (1.0 = trust the profile)
198 float sigma_scale[3]; // per-channel demosaic-bias correction, applied on top of noise_level
199 float a[3], b[3]; // noise variance line per RGB channel, normalized domain
200 dt_nn_model_t *model; // resolved from (version, variant); NULL disables the piece
202
204{
205 // lazily loaded, cached for the session; guarded by lock. tried[][] records
206 // a load attempt so a missing file is probed only once.
209 // user models from the config dir, keyed by basename; a NULL value records
210 // a failed load so a broken file is probed once, like tried[][] above
212 dt_pthread_mutex_t lock;
213#ifdef HAVE_OPENCL
214 dt_nn_cl_t *nn_cl; // U-Net kernel handles, from rawdenoiseai.cl
215 // device-resident glue kernels: the whole tile runs dev_in -> dev_out with
216 // no mid-tile host round-trip (command-queue syncs dominate GPU cost)
219#endif
221
222/* Thread-safe lazy loader: returns the model for (version, size, scale),
223 * loading it on first request from
224 * <configdir>/denoise-<size>-<single|multi>-<version>.anselnn (user
225 * override) or <datadir>/ (shipped). NULL if the file is absent or invalid.
226 * Runs on the pipeline thread via commit_params, hence the mutex. */
229{
230 if((int)ver < 0 || (int)ver >= DT_RAWDENOISEAI_NUM_VERSIONS || (int)sz < 0
231 || (int)sz >= DT_RAWDENOISEAI_NUM_SIZES || (int)sc < 0 || (int)sc >= DT_RAWDENOISEAI_NUM_SCALES)
232 return NULL;
233
235 if(!gd->tried[ver][sz][sc])
236 {
237 gd->tried[ver][sz][sc] = TRUE;
238 char name[64];
239 snprintf(name, sizeof(name), "denoise-%s-%s-%s.anselnn", _size_tag[sz], _scale_tag[sc], _version_tag[ver]);
240
241 char dir[DT_PATH_MAX] = { 0 };
242 char path[DT_PATH_MAX] = { 0 };
243 char err[256] = "";
244 dt_loc_get_user_config_dir(dir, sizeof(dir));
245 dt_concat_path_file(path, dir, name);
246 gd->models[ver][sz][sc] = dt_nn_model_load(path, err, sizeof(err));
247 if(!gd->models[ver][sz][sc])
248 {
249 dt_loc_get_datadir(dir, sizeof(dir));
250 dt_concat_path_file(path, dir, name);
251 gd->models[ver][sz][sc] = dt_nn_model_load(path, err, sizeof(err));
252 }
253 if(gd->models[ver][sz][sc])
254 dt_print(DT_DEBUG_ALWAYS, "[rawdenoiseai] loaded %s\n", path);
255 else
256 dt_print(DT_DEBUG_ALWAYS, "[rawdenoiseai] %s unavailable (%s)\n", name, err);
257 }
258 dt_nn_model_t *m = gd->models[ver][sz][sc];
260 return m;
261}
262
263/* A user model: any .anselnn dropped in the config dir under a name that is
264 * not one of the shipped ones. Cached by basename for the session, with a
265 * NULL entry recording a failed load so a broken file is probed once. Same
266 * mutex as the shipped matrix — this also runs on the pipeline thread. */
268{
269 if(!base || !*base || strchr(base, '/') || strchr(base, '\\')) return NULL;
270
272 if(!gd->custom) gd->custom = g_hash_table_new_full(g_str_hash, g_str_equal, g_free, NULL);
274 if(!g_hash_table_lookup_extended(gd->custom, base, NULL, (gpointer *)&m))
275 {
276 char dir[DT_PATH_MAX] = { 0 };
277 char path[DT_PATH_MAX] = { 0 };
278 char err[256] = "";
279 dt_loc_get_user_config_dir(dir, sizeof(dir));
280 dt_concat_path_file(path, dir, base);
281 m = dt_nn_model_load(path, err, sizeof(err));
282 g_hash_table_insert(gd->custom, g_strdup(base), m);
283 if(m)
284 dt_print(DT_DEBUG_ALWAYS, "[rawdenoiseai] loaded user model %s\n", path);
285 else
286 dt_print(DT_DEBUG_ALWAYS, "[rawdenoiseai] user model %s unusable (%s)\n", base, err);
287 }
289 return m;
290}
291
292/* Basenames of every .anselnn in the config dir, sorted, shipped names
293 * excluded (those are overrides of the matrix, already reachable through the
294 * size/variant combos). Caller frees with g_list_free_full(l, g_free). */
296{
297 char dir[DT_PATH_MAX] = { 0 };
298 dt_loc_get_user_config_dir(dir, sizeof(dir));
299 GDir *d = g_dir_open(dir, 0, NULL);
300 if(!d) return NULL;
301 GList *out = NULL;
302 const gchar *fn;
303 while((fn = g_dir_read_name(d)))
304 {
305 if(!g_str_has_suffix(fn, ".anselnn")) continue;
306 gboolean shipped = FALSE;
307 for(int v = 0; v < DT_RAWDENOISEAI_NUM_VERSIONS && !shipped; v++)
308 for(int z = 0; z < DT_RAWDENOISEAI_NUM_SIZES && !shipped; z++)
309 for(int c = 0; c < DT_RAWDENOISEAI_NUM_SCALES && !shipped; c++)
310 {
311 char name[64];
312 snprintf(name, sizeof(name), "denoise-%s-%s-%s.anselnn", _size_tag[z], _scale_tag[c], _version_tag[v]);
314 }
316 }
317 g_dir_close(d);
319}
320
321const char *name()
322{
323 return _("raw denoise (AI)");
324}
325
326const char **description(struct dt_iop_module_t *self)
327{
328 return dt_iop_set_description(self,
329 _("denoise the raw picture with a neural network conditioned "
330 "on the camera noise profile"),
331 _("corrective"), _("linear, raw, scene-referred"), _("linear, raw"),
332 _("linear, raw, scene-referred"));
333}
334
339
341{
342 return IOP_GROUP_REPAIR;
343}
344
349
357
358/* Route the executor's scratch through the pixelpipe cache arena, per the
359 * project rule that pixel buffers never come from bare malloc. nn_model
360 * cannot include darktable.h (it is deliberately pipeline-free), so the
361 * arena is injected here, once, before any pipeline runs. */
362/* Per-tile scratch REGION: the whole working set of one process() call —
363 * input planes, output plane and every executor tensor — is reserved from the
364 * pixelpipe arena as ONE allocation, up front, sized by the executor's exact
365 * ledger plus an internal-fragmentation margin. The executor's alloc/free
366 * churn then happens inside the region through the sub-allocator below and is
367 * invisible to the arena: no interleaving with cache entries, no fragmenting
368 * of the arena's free runs, and the tiling engine's largest-free-run cap
369 * guarantees the single reservation fits by construction. Memory is planned,
370 * reserved, then executed within — never discovered at runtime; if the
371 * reservation itself fails, the tile fails BEFORE any compute.
372 *
373 * The sub-allocator is a trivial first-fit block list: at most a dozen live
374 * tensors exist at once, all sized in whole planes. The region pointer is
375 * thread-local because the darkroom's full and preview pipes may run
376 * process() concurrently. */
377#define NN_REGION_MAX_BLOCKS 32
378// the two-ended layout (churn bottom-up, skips top-down) achieves the
379// ledger's live peak exactly; the slack only covers per-block 64-byte
380// alignment crumbs
381#define NN_REGION_SLACK 1.02f
382
383typedef struct nn_region_t
384{
385 char *base;
386 size_t size;
387 struct
388 {
389 size_t off, len;
393
395
396static void *_region_alloc(nn_region_t *r, size_t bytes, int long_lived)
397{
398 bytes = (bytes + 63) & ~(size_t)63; // keep 64-byte alignment inside the region
399 if(r->n_blocks >= NN_REGION_MAX_BLOCKS) return NULL;
400 size_t off = (size_t)-1;
401 int at = 0;
402 if(!long_lived)
403 {
404 // churn packs bottom-up, first fit; blocks are kept sorted by offset
405 size_t gap = 0;
406 for(int i = 0; i <= r->n_blocks; i++)
407 {
408 const size_t end = (i < r->n_blocks) ? r->blocks[i].off : r->size;
409 if(end - gap >= bytes)
410 {
411 off = gap;
412 at = i;
413 break;
414 }
415 if(i == r->n_blocks) break;
416 gap = r->blocks[i].off + r->blocks[i].len;
417 }
418 }
419 else
420 {
421 // long-lived blocks (skip connections) pack top-down, last fit, so they
422 // never split the churn area mid-region — this is what lets the region be
423 // sized at exactly planes + the ledger's live peak, no slack
424 size_t gap_end = r->size;
425 for(int i = r->n_blocks; i >= 0; i--)
426 {
427 const size_t gap_start = (i > 0) ? r->blocks[i - 1].off + r->blocks[i - 1].len : 0;
428 if(gap_end - gap_start >= bytes)
429 {
430 off = gap_end - bytes;
431 at = i;
432 break;
433 }
434 if(i > 0) gap_end = r->blocks[i - 1].off;
435 }
436 }
437 if(off == (size_t)-1)
438 {
439 size_t live = 0, largest_gap = 0, gap = 0;
440 for(int i = 0; i <= r->n_blocks; i++)
441 {
442 const size_t end = (i < r->n_blocks) ? r->blocks[i].off : r->size;
443 if(end - gap > largest_gap) largest_gap = end - gap;
444 if(i == r->n_blocks) break;
445 live += r->blocks[i].len;
446 gap = r->blocks[i].off + r->blocks[i].len;
447 }
449 "[rawdenoiseai] region alloc failed: %" G_GSIZE_FORMAT " bytes (%s), region %" G_GSIZE_FORMAT
450 ", live %" G_GSIZE_FORMAT " in %d blocks, largest gap %" G_GSIZE_FORMAT "\n",
451 bytes, long_lived ? "long-lived" : "churn", r->size, live, r->n_blocks, largest_gap);
452 return NULL;
453 }
454 for(int i = r->n_blocks; i > at; i--) r->blocks[i] = r->blocks[i - 1];
455 r->blocks[at].off = off;
456 r->blocks[at].len = bytes;
457 r->n_blocks++;
458 return r->base + off;
459}
460
461static void _region_free(nn_region_t *r, void *p)
462{
463 const size_t off = (size_t)((char *)p - r->base);
464 for(int i = 0; i < r->n_blocks; i++)
465 if(r->blocks[i].off == off)
466 {
467 for(int j = i; j < r->n_blocks - 1; j++) r->blocks[j] = r->blocks[j + 1];
468 r->n_blocks--;
469 return;
470 }
471}
472
478
479static void _nn_arena_free(void *p)
480{
482 if(r && (char *)p >= r->base && (char *)p < r->base + r->size)
483 {
484 _region_free(r, p);
485 return;
486 }
488}
489
491{
495#ifdef HAVE_OPENCL
496 gd->nn_cl = dt_nn_cl_create(39); // rawdenoiseai.cl, from programs.conf
497 gd->k_assemble = dt_opencl_create_kernel(39, "nn_assemble");
498 gd->k_bin_planes = dt_opencl_create_kernel(39, "nn_bin_planes");
499 gd->k_residual = dt_opencl_create_kernel(39, "nn_residual");
500 gd->k_upsample_n = dt_opencl_create_kernel(39, "nn_upsample_n");
501 gd->k_bin16_mdv = dt_opencl_create_kernel(39, "nn_bin16_mdv");
502 gd->k_avg2x2 = dt_opencl_create_kernel(39, "nn_avg2x2");
503 gd->k_floor_fuse = dt_opencl_create_kernel(39, "nn_floor_fuse");
504 gd->k_fuse_step = dt_opencl_create_kernel(39, "nn_fuse_step");
505 gd->k_bilerp_add = dt_opencl_create_kernel(39, "nn_bilerp_add");
506 gd->k_blend_crop = dt_opencl_create_kernel(39, "nn_blend_crop");
507#endif
508 module->data = gd;
509 // models are loaded lazily on first use per (version, variant)
510}
511
513{
514 // the hooks point into this module's code: leaving them set after dlclose
515 // would leave the core library with dangling function pointers
518 if(gd)
519 {
520 for(int v = 0; v < DT_RAWDENOISEAI_NUM_VERSIONS; v++)
521 for(int sz = 0; sz < DT_RAWDENOISEAI_NUM_SIZES; sz++)
522 for(int sc = 0; sc < DT_RAWDENOISEAI_NUM_SCALES; sc++) dt_nn_model_free(gd->models[v][sz][sc]);
523#ifdef HAVE_OPENCL
524 dt_nn_cl_destroy(gd->nn_cl);
525 dt_opencl_free_kernel(gd->k_assemble);
526 dt_opencl_free_kernel(gd->k_bin_planes);
527 dt_opencl_free_kernel(gd->k_residual);
528 dt_opencl_free_kernel(gd->k_upsample_n);
529 dt_opencl_free_kernel(gd->k_bin16_mdv);
530 dt_opencl_free_kernel(gd->k_avg2x2);
531 dt_opencl_free_kernel(gd->k_floor_fuse);
532 dt_opencl_free_kernel(gd->k_fuse_step);
533 dt_opencl_free_kernel(gd->k_bilerp_add);
534 dt_opencl_free_kernel(gd->k_blend_crop);
535#endif
536 if(gd->custom)
537 {
539 gpointer k, v;
540 g_hash_table_iter_init(&it, gd->custom);
541 while(g_hash_table_iter_next(&it, &k, &v))
543 g_hash_table_destroy(gd->custom);
544 gd->custom = NULL;
545 }
547 }
548 free(module->data);
549 module->data = NULL;
550}
551
552/* The size a new history entry gets, from what the machine can actually run:
553 * half where OpenCL will carry it, quarter on CPU alone — the only size that
554 * stays interactive there. The variant is multiscale either way; the coarse
555 * chroma pass is worth most exactly where capacity is scarce (it buys quarter
556 * ~2.6 dB of low-frequency chroma error and large almost nothing), so the
557 * hardware picks the width, not the variant. */
562
563/* Supported when the input is mosaiced and the model a new entry would DEFAULT
564 * to can be loaded (i.e. weights are installed). Probing that exact model and
565 * not a fixed one is what keeps the enable button honest on an installation
566 * carrying only part of the matrix. A specific (version, size, variant) the
567 * user selects that turns out to be missing is handled per-piece by
568 * copy-through. */
575
577{
578 /* Pick the default size from what the machine can actually run, because a
579 * user who enables the module without knowing what it is must get a result
580 * in seconds — not a frozen application. The variant is multiscale in both
581 * cases: it is what keeps the smaller networks free of low-frequency chroma
582 * blotches, and that matters most exactly where capacity is scarce (see
583 * doc/rawdenoiseai.md — the coarse pass buys quarter ~2.6 dB of chroma
584 * error and large almost nothing).
585 *
586 * OpenCL present: half, ~4x the quarter cost but clearly better.
587 * CPU only: quarter, the only size that stays interactive on a CPU.
588 *
589 * dt_opencl_is_enabled() reflects both the build and the user's preference,
590 * and is a static stub returning 0 without HAVE_OPENCL. Users who want a
591 * different trade-off still pick any size by hand; this only seeds a NEW
592 * history entry, so existing edits keep whatever they were created with. */
594 d->size = _default_size();
595 d->scale_variant = DT_RAWDENOISEAI_MULTI;
596
597 module->hide_enable_button = !_rawdenoiseai_supported(module);
598 module->default_enabled = 0;
599}
600
601gboolean force_enable(struct dt_iop_module_t *self, const gboolean current_state)
602{
603 // history sanitization: an entry pasted onto a non-mosaic image, or loaded
604 // without a model available, is forced off at history-read time
606}
607
608/* Fill d->a/d->b from the best noise profile for this image: exact ISO match,
609 * interpolation between the bracketing profiled ISOs, clamping outside the
610 * profiled range, generic Poissonian when the camera has no profiles. Mirrors
611 * the training-time ProfileSampler semantics (clamped interpolation). */
613{
616 const float iso = self->dev->image_storage.exif_iso;
617
618 if(profiles)
619 {
620 dt_noiseprofile_t *first = (dt_noiseprofile_t *)profiles->data;
621 dt_noiseprofile_t *last = NULL;
622 interpolated = *first; // clamp below the profiled range
623 for(GList *iter = profiles; iter; iter = g_list_next(iter))
624 {
625 dt_noiseprofile_t *current = (dt_noiseprofile_t *)iter->data;
626 if(current->iso == iso)
627 {
628 interpolated = *current;
629 break;
630 }
631 if(last && last->iso < iso && current->iso > iso)
632 {
633 interpolated.iso = iso;
634 dt_noiseprofile_interpolate(last, current, &interpolated);
635 break;
636 }
637 interpolated = *current; // clamp above the profiled range
638 last = current;
639 }
640 }
641 for(int k = 0; k < 3; k++)
642 {
643 d->a[k] = interpolated.a[k];
644 d->b[k] = interpolated.b[k];
645 }
647}
648
651{
654
655 d->strength = p->strength;
656 d->noise_level = p->noise_level;
657 d->sigma_scale[0] = p->sigma_red;
658 d->sigma_scale[1] = p->sigma_green;
659 d->sigma_scale[2] = p->sigma_blue;
660 // neural inference is among the most expensive nodes of the pipe: always
661 // materialize this piece's output in the RAM cache (same policy as
662 // diffuse/atrous), so interactive edits downstream never re-run it
663 piece->cache_output_on_ram = TRUE;
664 _fetch_noise_profile(self, d);
665
667 /* A named user model wins over the shipped matrix. If it has gone missing
668 * since the edit was made, the piece is disabled rather than silently
669 * rendered through a DIFFERENT network: the two are not interchangeable,
670 * and quietly substituting one would change the picture without telling
671 * anyone. The GUI keeps showing the name so the situation is legible. */
672 const gboolean want_custom = gd && p->custom_model[0];
673 d->model = want_custom ? _get_custom_model(gd, p->custom_model)
674 : gd ? _get_model(gd, p->version, p->size, p->scale_variant)
675 : NULL;
676
678 "[rawdenoiseai] commit: camera '%s' iso %.0f model %s-%s-%s%s -> "
679 "a=(%.3g %.3g %.3g) b=(%.3g %.3g %.3g) strength %.2f noise level %.2f "
680 "channel correction (%.2f %.2f %.2f)\n",
683 _scale_tag[CLAMP(p->scale_variant, 0, DT_RAWDENOISEAI_NUM_SCALES - 1)],
685 d->model ? (want_custom ? " (user model)" : "") : " (missing)",
686 d->a[0], d->a[1], d->a[2], d->b[0], d->b[1], d->b[2], p->strength, p->noise_level, p->sigma_red,
687 p->sigma_green, p->sigma_blue);
688
689 // plane-layout contract with the training repo: a model that does not
690 // match what _assemble_planes builds must be treated as missing, not fed
691 if(d->model)
692 {
693 const int fine_in = dt_nn_model_in_channels(d->model);
694 const int c_out = dt_nn_model_coarse_out_channels(d->model);
695 const gboolean ok = c_out > 0 ? (dt_nn_model_coarse_in_channels(d->model) == 6 && c_out == 3 && fine_in == 8)
696 : (fine_in == 5);
697 if(!ok)
698 {
700 "[rawdenoiseai] model plane layout unsupported "
701 "(fine_in %d coarse_out %d) — module disabled\n",
702 fine_in, c_out);
703 d->model = NULL;
704 }
705 }
706 if(!d->model || !(p->strength > 0.0f)) piece->enabled = 0;
707}
708
709static unsigned _align_lcm(const unsigned a, const unsigned b)
710{
711 if(a == 0 || b == 0) return a > b ? a : b;
712 unsigned x = a, y = b;
713 while(y)
714 {
715 const unsigned t = x % y;
716 x = y;
717 y = t;
718 }
719 return a / x * b;
720}
721
724{
726 // per input pixel (4 bytes): 5 input planes + 1 output plane + network
727 // scratch, all float32. factor is relative to the unit buffer size.
728 float extra = 6.0f;
729 float extra_cl = 6.0f;
730 unsigned overlap = 0;
731 if(d && d->model)
732 {
733 // input planes + output plane (in_ch is 8 for a multi-scale model), the
734 // coarse buffers at 1/bin^2, and the executor scratch (which already
735 // includes the coarse net's share for unet-ms)
736 const gboolean xtrans_cfa = piece->dsc_in.filters == 9u;
737 const int bin = dt_nn_model_bin(d->model, xtrans_cfa);
738 const float planes = (float)(dt_nn_model_in_channels(d->model) + 1)
739 + (bin > 1 ? 12.0f / (float)(bin * bin) : 0.0f);
740 /* Host: process() reserves the tile's whole working set as ONE arena
741 * region — the (in_ch + 1) planes plus the executor's ledger peak times
742 * the sub-allocator's fragmentation margin. The factor declares exactly
743 * that reservation, so the tiling plan, the reservation and the execution
744 * are the same number (the coarse-stage module buffers, 12/bin^2, are the
745 * only separate short-lived arena entries). Expressed as a factor of the
746 * input image size — floats per input pixel — never absolute bytes. */
747 extra = planes + NN_REGION_SLACK * dt_nn_unet_scratch_per_px(d->model);
748 /* Device: no region — buffers are device-side. The CL executor's sequence
749 * differs (it materializes the decoder concat), and the CL path adds
750 * three device planes of its own (dev_den, plus the buffer-format copies
751 * dev_in_buf/dev_out_buf) and the fusion grids (~0.13 plane); count all
752 * of it explicitly rather than letting it ride inside factor_cl's
753 * padding headroom. */
754 extra_cl = planes + dt_nn_unet_scratch_per_px_cl(d->model) + 3.2f;
755
756 /* Overlap is set from the MEASURED seam profile, not from the impulse
757 * response. An earlier revision sized it at 48 px because the trained
758 * model's response to a delta decays to 1.4e-6 of peak by radius 32; that
759 * reasoning is wrong for this purpose. A tile boundary does not remove one
760 * tap, it removes an entire half-plane of context, and the summed
761 * contribution of that half-plane is orders of magnitude above any single
762 * tap. Measured on DSCF9668.RAF (X-Trans, 3 tiles across), CPU untiled vs
763 * GPU tiled, as a fraction of full scale:
764 *
765 * overlap 48 -> seam peaks at 0.17-0.23 %, i.e. 238-317x the interior
766 * difference, and does not return to it until ~50 px in
767 * overlap 192 -> seam peaks at 0.006-0.016 %, 5.5-14.5x interior
768 *
769 * The 48 px case is visible; it is the tile grid users report. The decay
770 * measured with 48 px of overlap reaches the interior baseline exactly
771 * 48 px inside the owned region, so a boundary pixel needs 48 + 48 = 96 px
772 * of context: that is the fine net's effective receptive field, and the
773 * floor for the single-scale model.
774 *
775 * The multiscale model adds the coarse stage, whose receptive field is its
776 * own (in binned pixels) times `bin`, so it needs substantially more. 192
777 * is the value actually validated above. Note X-Trans multiscale reached it
778 * only by accident: the engine rounds the requested overlap UP to a
779 * multiple of lcm(xalign, yalign), which is 192 there, so a request of 96
780 * silently became 192. Bayer multiscale aligns to 64 and would have stayed
781 * at 128. Ask for what the measurement supports instead of relying on an
782 * alignment that varies with the sensor. */
783 overlap = (bin > 1) ? 192 : 96;
784 }
785 tiling->factor = 2.0f + extra;
786 /* The GPU budget must be more conservative than the CPU one, for two
787 * reasons the tiling engine cannot see. First, every buffer here is
788 * allocated at the alignment-padded tile size, up to (align-1) larger per
789 * axis than the tile the engine budgeted — only for the ragged last tile of
790 * a row or column now that xalign below equals the alignment, so an interior
791 * tile is padded by nothing at all and this term is pure headroom. Second,
792 * the engine hands out up to 100 % of the device's reported memory, and a
793 * budget that spends all of VRAM never allocates in practice (display,
794 * driver and allocator overhead) — CPU RAM overcommits, VRAM does not,
795 * and the U-Net scratch dominates this factor so the whole budget is
796 * exact where other modules' estimates carry implicit slack. 1.4 covers
797 * the worst realistic padding inflation (~1.15) times an allocation
798 * headroom (~1.2); the cost of overshooting is a smaller tile, the cost
799 * of undershooting is CL_MEM_OBJECT_ALLOCATION_FAILURE and a silent
800 * fallback of the whole tile to CPU. */
801 tiling->factor_cl = 2.0f + extra_cl * 1.4f;
802 tiling->maxbuf = 1.0f;
803 // device-resident weight blob (up to ~38 MB) plus fixed slack
804 tiling->overhead = 64u * 1024u * 1024u;
805 tiling->overlap = overlap;
806 /* Tiles must preserve the CFA phase — and, for a multi-scale model, every
807 * other LATTICE this module lays over its input: the superpixel binning of
808 * the coarse stage (period `bin`) and the low-band fusion pyramid (period
809 * DT_NN_FUSION_COARSEST). All of them are anchored to the tile's own origin,
810 * because that is the only origin process() is handed. The tiling engine
811 * places tile origins on multiples of lcm(xalign, yalign), so anything short
812 * of the full lattice period lets two different tile grids bin the same
813 * sensels into differently-phased superpixels — which changes the coarse
814 * guide, and with it the result, EVERYWHERE inside the tile rather than only
815 * near its seams. No amount of overlap fixes that; only aligning the origins
816 * does. It is why the same edit rendered differently on CPU and GPU: the two
817 * budget their memory differently, so they tile differently, so they binned
818 * on different lattices.
819 *
820 * dt_nn_model_alignment() is the one function that owns those periods (the
821 * fine net's stride pyramid, the coarse net's binned one, and the fusion
822 * bands) — ask it rather than re-deriving any of them here, or the next
823 * lattice added to the model silently reopens this bug. Aligning to the FULL
824 * value also means an interior tile needs no reflect padding at all, so no
825 * mirrored data is folded into the fusion's per-tile statistics. */
826 const gboolean xtrans = piece->dsc_in.filters == 9u;
827 unsigned align = xtrans ? 6u : 2u;
828 if(d && d->model) align = _align_lcm(align, (unsigned)dt_nn_model_alignment(d->model));
829 tiling->xalign = align;
830 tiling->yalign = align;
831}
832
833// mirror-reflect coordinate into [0, n): same row/column parity is preserved
834// for even n-1 steps; the CFA color planes are computed from the SOURCE
835// sensel, so the network always sees consistent (value, color) pairs even
836// where reflection breaks the periodic layout (X-Trans borders).
837static inline __attribute__((always_inline)) int _reflect(int v, int n)
838{
839 if(n == 1) return 0;
840 while(v < 0 || v >= n)
841 {
842 if(v < 0) v = -v;
843 if(v >= n) v = 2 * n - 2 - v;
844 }
845 return v;
846}
847
848/* Total sigma scale per CFA color: global correction x the per-channel
849 * slider, nothing else — the GUI values ARE the conditioning (see the
850 * calibration comment at the top of this file). */
851static void _sigma_scale(const dt_iop_rawdenoiseai_data_t *d, float scale[3])
852{
853 for(int c = 0; c < 3; c++) scale[c] = d->noise_level * d->sigma_scale[c];
854}
855
856/* Build the coarse stage's 6 input planes [R, G, B, sigmaR, sigmaG, sigmaB]
857 * from the assembled fine planes 0-3 (count-weighted superpixel means; the
858 * binning itself is dt_nn_bin_planes, the bit-exact contract with the
859 * training repo). Coarse sigma is the analytic sigma of the mean of n
860 * sensels: scale[c] * sqrt((a*x + b) / n). Shared verbatim by the CPU and
861 * OpenCL paths. */
863static void _k_bin_planes(const float *const nn_in, float *const coarse_in, float *const cnt,
864 const int pw, const int ph, const int bin,
865 const dt_iop_rawdenoiseai_data_t *const d)
866{
867 const int cw = pw / bin, chh = ph / bin;
868 const size_t cplane = (size_t)cw * chh;
870 float scale[3];
871 _sigma_scale(d, scale);
873 for(int c = 0; c < 3; c++)
874 {
875 const float *const mean = coarse_in + (size_t)c * cplane;
876 const float *const n_c = cnt + (size_t)c * cplane;
877 float *const sigma = coarse_in + (size_t)(3 + c) * cplane;
878 for(size_t i = 0; i < cplane; i++)
879 {
880 const float n = n_c[i] > 1.0f ? n_c[i] : 1.0f;
881 const float var = (d->a[c] * MAX(mean[i], 0.0f) + d->b[c]) / n;
882 sigma[i] = scale[c] * sqrtf(MAX(var, 1e-12f));
883 }
884 }
885}
886
887/* Build the network's 5 input planes [mosaic, R, G, B one-hot, sigma] from the
888 * mosaic, reflect-padded from (width, height) to (pw, ph). Shared verbatim by
889 * the CPU and OpenCL paths so both feed the network identical data. */
891static void _k_assemble(const float *const in, float *const nn_in, const int width, const int height,
892 const int pw, const int ph, const dt_iop_rawdenoiseai_data_t *const d,
893 const uint32_t filters, const uint8_t (*const xtrans)[6],
894 const dt_iop_roi_t *const roi)
895{
896 const size_t plane = (size_t)pw * ph;
897 float scale[3];
898 _sigma_scale(d, scale);
900 for(int y = 0; y < ph; y++)
901 {
902 const int sy = _reflect(y, height);
903 const float *const irow = in + (size_t)sy * width;
904 float *const mosaic = nn_in + (size_t)y * pw;
905 float *const onehot_r = nn_in + plane + (size_t)y * pw;
906 float *const onehot_g = nn_in + 2 * plane + (size_t)y * pw;
907 float *const onehot_b = nn_in + 3 * plane + (size_t)y * pw;
908 float *const sigma = nn_in + 4 * plane + (size_t)y * pw;
909 for(int x = 0; x < pw; x++)
910 {
911 const int sx = _reflect(x, width);
912 const float v = irow[sx];
913 int c = (filters == 9u) ? FCxtrans(sy, sx, roi, xtrans) : (int)FC(sy, sx, filters);
914 if(c < 0 || c > 2) c = 1; // both greens share the G statistics; clamp junk CFA data
915 mosaic[x] = v;
916 onehot_r[x] = (c == 0) ? 1.0f : 0.0f;
917 onehot_g[x] = (c == 1) ? 1.0f : 0.0f;
918 onehot_b[x] = (c == 2) ? 1.0f : 0.0f;
919 const float var = d->a[c] * MAX(v, 0.0f) + d->b[c];
920 sigma[x] = scale[c] * sqrtf(MAX(var, 1e-12f));
921 }
922 }
923}
924
925
926/* ---- CPU kernels ------------------------------------------------------
927 * One function per OpenCL kernel in rawdenoiseai.cl, same name, same
928 * arguments in the same order, so process() and process_cl() read as the same
929 * procedure and a change to one has an obvious counterpart in the other.
930 * Every pixel loop in this module lives in one of these. */
931
932/* Per-channel Bayer densities: the fraction of a block's sensels carrying each
933 * colour. Shared with the OpenCL twins, which take them as dens0/dens1/dens2.
934 * Bayer values are used for both CFA families, matching the torch reference. */
935static const float DT_NN_FUSION_DENS[3] = { 0.25f, 0.5f, 0.25f };
936
937/* chi^2-quantile guard: the local mean of a squared noise term over a 3x3 cell
938 * neighbourhood has ~9 effective samples, so pure-noise cells fluctuate up to
939 * ~2x their expectation. Must equal the literal in nn_floor_fuse/nn_fuse_step. */
940#define DT_NN_FUSION_T_CHI2 2.5
941
942/* mirrors nn_residual: the network predicts what to REMOVE, so the denoised
943 * signal is input minus head. Applied by the caller on BOTH devices — the
944 * executor writes the raw head output either way. */
946static void _k_residual(const float *const in, const float *const head, float *const out, const size_t n)
947{
949 for(size_t i = 0; i < n; i++) out[i] = in[i] - head[i];
950}
951
952/* mirrors nn_blend_crop: strength is the opacity of the correction, so the
953 * output lerps between the original mosaic and the denoised one while dropping
954 * the alignment padding. */
956static void _k_blend_crop(const float *const in, const float *const den, float *const out, const int width,
957 const int height, const int pw, const float strength)
958{
960 for(int y = 0; y < height; y++)
961 {
962 const float *const src_in = in + (size_t)y * width;
963 const float *const src_nn = den + (size_t)y * pw;
964 float *const dst = out + (size_t)y * width;
965 for(int x = 0; x < width; x++) dst[x] = src_in[x] + strength * (src_nn[x] - src_in[x]);
966 }
967}
968
969/* mirrors nn_bin16_mdv: count-weighted per-channel mean of the mosaic, the
970 * denoised plane and sigma^2 over each 16x16 block. */
972static void _k_bin16_mdv(const float *const nn_in, const float *const denoised, float *const M,
973 float *const D, float *const V, const int pw, const int ph)
974{
975 const size_t plane = (size_t)pw * ph;
976 const float *const sig = nn_in + 4 * plane;
977 const int cw = pw / DT_NN_FUSION_FINEST, chh = ph / DT_NN_FUSION_FINEST;
978 const size_t p0 = (size_t)cw * chh;
979 for(int c = 0; c < 3; c++)
980 {
981 const float *const oh = nn_in + (size_t)(1 + c) * plane;
983 for(int cy = 0; cy < chh; cy++)
984 for(int cx = 0; cx < cw; cx++)
985 {
986 float sm = 0.f, sd = 0.f, sv = 0.f, cnt = 0.f;
987 for(int y = cy * DT_NN_FUSION_FINEST; y < (cy + 1) * DT_NN_FUSION_FINEST; y++)
988 for(int x = cx * DT_NN_FUSION_FINEST; x < (cx + 1) * DT_NN_FUSION_FINEST; x++)
989 {
990 const size_t i = (size_t)y * pw + x;
991 sm += nn_in[i] * oh[i];
992 sd += denoised[i] * oh[i];
993 sv += sig[i] * sig[i] * oh[i];
994 cnt += oh[i];
995 }
996 const float n = cnt > 1.f ? cnt : 1.f;
997 const size_t o = (size_t)c * p0 + (size_t)cy * cw + cx;
998 M[o] = sm / n;
999 D[o] = sd / n;
1000 V[o] = sv / n;
1001 }
1002 }
1003}
1004
1005/* mirrors nn_avg2x2: 2x2 average pooling of a 3-plane grid. Per-channel counts
1006 * are uniform at these scales, so a plain mean of four children equals
1007 * re-binning from full resolution. */
1009static void _k_avg2x2(const float *const in, float *const out, const int sw, const int sh, const size_t p0)
1010{
1011 const int w2 = sw / 2, h2 = sh / 2;
1012 for(int c = 0; c < 3; c++)
1013 {
1014 const float *const src = in + (size_t)c * p0;
1015 float *const dst = out + (size_t)c * p0;
1016 for(int y = 0; y < h2; y++)
1017 for(int x = 0; x < w2; x++)
1018 dst[(size_t)y * w2 + x]
1019 = 0.25f
1020 * (src[(size_t)(2 * y) * sw + 2 * x] + src[(size_t)(2 * y) * sw + 2 * x + 1]
1021 + src[(size_t)(2 * y + 1) * sw + 2 * x] + src[(size_t)(2 * y + 1) * sw + 2 * x + 1]);
1022 }
1023}
1024
1025/* Hybrid low-band fusion (mirrors cfa.fuse_low_bands in the training repo):
1026 * per-band self-calibrated Wiener weights at scales 16/32, pure measurement
1027 * at the coarsest band. All band corrections are upsampled BILINEARLY
1028 * (align_corners=false, matching torch F.interpolate): nearest upsampling
1029 * turned the per-block measurement noise (sigma/sqrt(n), non-negligible on
1030 * very noisy images) into visible checkers of colored squares. The finest
1031 * fusion band is 16 px for the same reason. The coarsest band is the
1032 * n-averaged measurement outright, so the hallucination-free guarantee
1033 * holds. The pyramid is always 16/32/64 — the same bands the training
1034 * reference fuses — because dt_nn_model_alignment() guarantees a padded tile
1035 * that divides by the coarsest one. Bayer channel densities are used for both
1036 * CFA families, matching the torch reference. */
1037static inline __attribute__((always_inline)) float _bilerp_tap(const float *const p, const int w,
1038 const int h, const float fx,
1039 const float fy)
1040{
1041 const float cx = fx < 0.f ? 0.f : (fx > w - 1.f ? w - 1.f : fx);
1042 const float cy = fy < 0.f ? 0.f : (fy > h - 1.f ? h - 1.f : fy);
1043 const int x0 = (int)cx, y0 = (int)cy;
1044 const int x1 = x0 + 1 < w ? x0 + 1 : x0;
1045 const int y1 = y0 + 1 < h ? y0 + 1 : y0;
1046 const float ax = cx - x0, ay = cy - y0;
1047 const float top = p[(size_t)y0 * w + x0] * (1.f - ax) + p[(size_t)y0 * w + x1] * ax;
1048 const float bot = p[(size_t)y1 * w + x0] * (1.f - ax) + p[(size_t)y1 * w + x1] * ax;
1049 return top * (1.f - ay) + bot * ay;
1050}
1051
1052// dst (fw x fh) = bilinear upsample of src (sw x sh) by integer factor f
1054static void _upsample_bilinear(const float *const src, const int sw, const int sh, const int f, float *const dst)
1055{
1056 const int fw = sw * f, fh = sh * f;
1057 for(int y = 0; y < fh; y++)
1058 {
1059 const float sy = (y + 0.5f) / f - 0.5f;
1060 for(int x = 0; x < fw; x++) dst[(size_t)y * fw + x] = _bilerp_tap(src, sw, sh, (x + 0.5f) / f - 0.5f, sy);
1061 }
1062}
1063
1064/* Number of pyramid levels between the finest and the coarsest fusion band.
1065 * Constant by construction — both are fixed by the training reference. */
1066static inline int _fusion_levels(void)
1067{
1068 int n = 1;
1069 for(int s = DT_NN_FUSION_FINEST; s < DT_NN_FUSION_COARSEST; s *= 2) n++;
1070 return n;
1071}
1072
1073/* Returns 0 on success, non-zero if the fusion could not run — the caller must
1074 * treat that as a failed tile, exactly as process_cl() does when a device
1075 * buffer for the same pyramid cannot be allocated. Rendering the tile with the
1076 * fine network's raw low band instead would be a silent, tile-shaped quality
1077 * regression, and would differ from whatever the other device did. */
1079/* mirrors nn_floor_fuse: structure-gated blend — the measurement owns every
1080 * cell whose own mean-removed local energy is noise-sized (the dilution
1081 * guarantee), the model owns structured cells, because a box average across an
1082 * edge mixes both sides into a saturated outline. The gate reads the
1083 * MEASUREMENT, not the model discrepancy: D-M cannot tell a real edge from the
1084 * model drifting on flat content. REQUIRES a model trained with the
1085 * DC-ownership loss; one from the older fused loss drifts in deep shadows. */
1087static void _k_floor_fuse(const float *const M, const float *const D, const float *const V,
1088 float *const fused, const int sw, const int sh, const size_t p0, const int S)
1089{
1090 for(int c = 0; c < 3; c++)
1091 {
1092 const double vscale = 1.0 / (DT_NN_FUSION_DENS[c] * S * S);
1093 const float *const Mp = M + (size_t)c * p0;
1094 const float *const Dp = D + (size_t)c * p0;
1095 const float *const Vp = V + (size_t)c * p0;
1096 float *const fs = fused + (size_t)c * p0;
1097 for(int y = 0; y < sh; y++)
1098 for(int x = 0; x < sw; x++)
1099 {
1100 // structure = blur3((M - blur3(M))^2): insensitive to smooth gradients,
1101 // unlike a plain window variance
1102 double structure = 0.0;
1103 for(int ny = -1; ny <= 1; ny++)
1104 for(int nx = -1; nx <= 1; nx++)
1105 {
1106 const int cy2 = CLAMP(y + ny, 0, sh - 1), cx2 = CLAMP(x + nx, 0, sw - 1);
1107 double mean = 0.0;
1108 for(int dy = -1; dy <= 1; dy++)
1109 for(int dx = -1; dx <= 1; dx++)
1110 {
1111 const int yy = CLAMP(cy2 + dy, 0, sh - 1), xx = CLAMP(cx2 + dx, 0, sw - 1);
1112 mean += Mp[(size_t)yy * sw + xx];
1113 }
1114 const double mloc = Mp[(size_t)cy2 * sw + cx2] - mean / 9.0;
1115 structure += mloc * mloc;
1116 }
1117 const size_t i = (size_t)y * sw + x;
1118 // this cell's own mean sigma^2, not the tile's (cfa.fuse_low_bands)
1119 const double vn = (double)Vp[i] * vscale;
1121 if(structure < 0.0) structure = 0.0;
1122 const float w = (float)(structure / (structure + vn + 1e-20));
1123 fs[i] = w * Dp[i] + (1.f - w) * Mp[i];
1124 }
1125 }
1126}
1127
1128/* mirrors nn_fuse_step: upsample the running fusion and add the finer band,
1129 * weighted per cell by a Wiener gain on the band discrepancy. `ups` is scratch
1130 * for two upsampled planes. */
1132static void _k_fuse_step(const float *const fused_c, const float *const Mf, const float *const Df,
1133 const float *const Vf, const float *const Mc, const float *const Dc,
1134 float *const fused_f, float *const ups, const int sw, const int sh,
1135 const size_t p0, const int sc)
1136{
1137 const int fw = sw * 2, fh = sh * 2;
1138 for(int c = 0; c < 3; c++)
1139 {
1140 const float *const Dfp = Df + (size_t)c * p0;
1141 const float *const Mfp = Mf + (size_t)c * p0;
1142 const float *const Vfp = Vf + (size_t)c * p0;
1143 float *const upD = ups, *const upM = ups + p0;
1144 _upsample_bilinear(Dc + (size_t)c * p0, sw, sh, 2, upD);
1145 _upsample_bilinear(Mc + (size_t)c * p0, sw, sh, 2, upM);
1146 /* Var(mean_s - up(mean_2s)) = Var(mean_s) * 3/4 once the covariance with the
1147 * 2x2 parent is folded in, which is what this reciprocal difference is; so
1148 * the scale-s cell mean is the right sigma^2 for the whole term. */
1149 const double vscale
1150 = 1.0 / (DT_NN_FUSION_DENS[c] * sc * sc) - 1.0 / (DT_NN_FUSION_DENS[c] * 4.0 * sc * sc);
1151 _upsample_bilinear(fused_c + (size_t)c * p0, sw, sh, 2, fused_f + (size_t)c * p0);
1152 float *const fs = fused_f + (size_t)c * p0;
1153 for(int y = 0; y < fh; y++)
1154 for(int x = 0; x < fw; x++)
1155 {
1156 // per-cell Wiener weight from the 3x3-smoothed band discrepancy
1157 double acc = 0.0;
1158 int n = 0;
1159 for(int dy = -1; dy <= 1; dy++)
1160 for(int dx = -1; dx <= 1; dx++)
1161 {
1162 const int yy = CLAMP(y + dy, 0, fh - 1), xx = CLAMP(x + dx, 0, fw - 1);
1163 const size_t j = (size_t)yy * fw + xx;
1164 const double d = (double)(Dfp[j] - upD[j]) - (double)(Mfp[j] - upM[j]);
1165 acc += d * d;
1166 n++;
1167 }
1168 const size_t i = (size_t)y * fw + x;
1169 const double vn = (double)Vfp[i] * vscale;
1170 double vm = acc / n - DT_NN_FUSION_T_CHI2 * vn;
1171 if(vm < 0.0) vm = 0.0;
1172 const float w = (float)(vn / (vn + vm + 1e-20));
1173 fs[i] += w * (Dfp[i] - upD[i]) + (1.f - w) * (Mfp[i] - upM[i]);
1174 }
1175 }
1176}
1177
1178/* mirrors nn_bilerp_add: scatter (fused - D16) bilinearly upsampled from the
1179 * level-0 grid onto the denoised plane, on whichever colour plane owns each
1180 * sensel. `scratch` is reused as the correction plane. */
1182static void _k_bilerp_add(const float *const fused, const float *const D16, float *const scratch,
1183 const float *const nn_in, float *const denoised, const int pw, const int ph,
1184 const size_t p0, const int cw0, const int ch0)
1185{
1186 const size_t plane = (size_t)pw * ph;
1187 const float inv = 1.f / (float)DT_NN_FUSION_FINEST;
1188 for(int c = 0; c < 3; c++)
1189 {
1190 float *const cs = scratch + (size_t)c * p0;
1191 const float *const fa = fused + (size_t)c * p0;
1192 const float *const d0 = D16 + (size_t)c * p0;
1193 for(size_t i = 0; i < p0; i++) cs[i] = fa[i] - d0[i];
1194 }
1196 for(int y = 0; y < ph; y++)
1197 {
1198 const float sy = (y + 0.5f) * inv - 0.5f;
1199 for(int x = 0; x < pw; x++)
1200 {
1201 const size_t i = (size_t)y * pw + x;
1202 for(int c = 0; c < 3; c++)
1203 if(nn_in[(size_t)(1 + c) * plane + i] > 0.0f)
1204 {
1205 denoised[i] += _bilerp_tap(scratch + (size_t)c * p0, cw0, ch0, (x + 0.5f) * inv - 0.5f, sy);
1206 break;
1207 }
1208 }
1209 }
1210}
1211
1212static int _apply_low_band_anchor(const float *const nn_in, float *const denoised, const int pw, const int ph,
1213 const int scale)
1214{
1215 if(scale <= 0) return 0;
1216 // The pyramid is 16/32/64, fixed by the training reference. The padded tile
1217 // divides by 64 because dt_nn_model_alignment() folds DT_NN_FUSION_COARSEST
1218 // in for any model that declares an anchor; deriving the number of levels
1219 // from the tile size instead is what used to make the render depend on how
1220 // the pipe happened to tile (i.e. on the machine, and on CPU vs GPU).
1221 const int S = DT_NN_FUSION_COARSEST;
1222 if(pw % S || ph % S) return 0;
1223 const int cw0 = pw / 16, ch0 = ph / 16; // level-0 grid (scale 16)
1224 const size_t p0 = (size_t)cw0 * ch0;
1225 // slots: M/D/V per level (16, 32, 64) + fused ping-pong + upsample scratch
1226 float *const buf = dt_pixelpipe_cache_alloc_align_float_cache(p0 * 3 * (3 * 3 + 3), 0);
1227 if(IS_NULL_PTR(buf)) return 1;
1228 float *lv[9]; // lv[3 * level + {0: mosaic, 1: denoised, 2: sigma^2}]
1229 for(int k = 0; k < 9; k++) lv[k] = buf + (size_t)k * p0 * 3;
1230 float *fusedA = buf + p0 * 3 * 9, *fusedB = fusedA + p0 * 3, *ups = fusedB + p0 * 3;
1231
1232 _k_bin16_mdv(nn_in, denoised, lv[0], lv[1], lv[2], pw, ph);
1233 const int nlev = _fusion_levels();
1234 int w_l = cw0, h_l = ch0;
1235 for(int k = 1; k < nlev; k++)
1236 {
1237 for(int md = 0; md < 3; md++) _k_avg2x2(lv[3 * (k - 1) + md], lv[3 * k + md], w_l, h_l, p0);
1238 w_l /= 2;
1239 h_l /= 2;
1240 }
1241
1242 // Coarse-to-fine fusion with LOCAL weights (mirrors cfa.fuse_low_bands).
1243 // Two distinct gates (see the training repo for the full rationale):
1244 // - floor band: anchor to the measurement wherever the MEASUREMENT itself
1245 // is smooth at this scale (local variance vs noise); the model-vs-
1246 // measurement discrepancy cannot distinguish a real edge from the model
1247 // drifting on flat content;
1248 // - soft bands: per-cell Wiener on the band discrepancy with a chi^2
1249 // guard (T = 2.5, ~9 effective samples per 3x3 cell neighbourhood).
1250 int sw = w_l, sh = h_l;
1251 // FLOOR: structure-gated blend (mirrors cfa.fuse_low_bands) — the
1252 // measurement owns every cell where its own mean-removed local energy is
1253 // noise-sized (the dilution guarantee), the model owns structured cells
1254 // (a box average across an edge mixes both sides -> saturated outline).
1255 // REQUIRES a model trained with the DC-ownership loss: models from the
1256 // older fused loss drift in deep shadows and must not be used with this
1257 // code.
1258 _k_floor_fuse(lv[3 * (nlev - 1)], lv[3 * (nlev - 1) + 1], lv[3 * (nlev - 1) + 2], fusedA, sw, sh, p0,
1259 DT_NN_FUSION_FINEST << (nlev - 1));
1260 for(int k = nlev - 2; k >= 0; k--)
1261 {
1262 _k_fuse_step(fusedA, lv[3 * k], lv[3 * k + 1], lv[3 * k + 2], lv[3 * (k + 1)], lv[3 * (k + 1) + 1], fusedB,
1263 ups, sw, sh, p0, DT_NN_FUSION_FINEST << k);
1264 float *tmp = fusedA;
1265 fusedA = fusedB;
1266 fusedB = tmp;
1267 sw *= 2;
1268 sh *= 2;
1269 }
1270
1273 return 0;
1274}
1275
1278 const void *const ivoid, void *const ovoid)
1279{
1280 const dt_iop_roi_t *const roi = &piece->roi_in;
1282 dt_nn_model_t *const model = d->model;
1283
1284 const int width = roi->width, height = roi->height;
1285 const float *const in = (const float *)ivoid;
1286 float *const out = (float *)ovoid;
1287
1288 if(!model || !(d->strength > 0.0f))
1289 {
1291 return 0;
1292 }
1293
1294 /* The Bayer CFA lookup below is tile-local (FC() has no ROI awareness), so it
1295 * needs the word already rotated to this ROI's phase — see the CFA-phase rule
1296 * in CLAUDE.md. The X-Trans branch stays self-correcting: it takes the raw
1297 * table plus `roi` and adds the offset itself, and dt_dev_get_roi_filters()
1298 * no-ops on X-Trans, so `filters == 9u` still identifies it. */
1299 const uint32_t filters = dt_dev_get_roi_filters(piece, roi);
1300 const uint8_t(*const xtrans)[6] = (const uint8_t(*const)[6])piece->dsc_in.xtrans;
1301
1302 // reflect-pad to the network alignment
1303 const int align = dt_nn_model_alignment(model);
1304 const int pw = (width + align - 1) / align * align;
1305 const int ph = (height + align - 1) / align * align;
1306 const size_t plane = (size_t)pw * ph;
1307
1308 /* The tile's whole working set — input planes, output plane, executor
1309 * scratch — is ONE arena reservation, made before any compute, sized by the
1310 * executor's exact ledger plus the sub-allocator's fragmentation margin.
1311 * This is the same quantity tiling_callback declares, so the plan, the
1312 * reservation and the execution are one number. */
1313 const int in_ch = dt_nn_model_in_channels(model);
1314 nn_region_t region = { 0 };
1315 region.size = ((plane * (in_ch + 1) * sizeof(float) + 63) & ~(size_t)63)
1316 + (size_t)(NN_REGION_SLACK * (float)dt_nn_unet_scratch_bytes(model, pw, ph));
1317 region.size = (region.size + 63) & ~(size_t)63;
1319 if(IS_NULL_PTR(region.base))
1320 {
1321 dt_print(DT_DEBUG_ALWAYS, "[rawdenoiseai] region alloc failed for %dx%d tile (%.1f MB)\n", pw, ph,
1322 region.size / 1048576.0);
1324 return 1;
1325 }
1326 _nn_region = &region;
1327 float *const nn_in = _region_alloc(&region, plane * in_ch * sizeof(float), 0);
1328 float *const nn_out = _region_alloc(&region, plane * sizeof(float), 0);
1329 // by construction these cannot fail: the region was sized for them
1330
1331 _k_assemble(in, nn_in, width, height, pw, ph, d, filters, xtrans, roi);
1332
1333 int rc = 0;
1334 const int bin = dt_nn_model_bin(model, filters == 9u);
1335 if(bin > 1)
1336 {
1337 // coarse (low-frequency chroma) pass: denoise the superpixel-binned RGB
1338 // and inject the nearest-upsampled result as guide planes 5-7 of the
1339 // fine network's input (mirrors ms_forward() in ansel-denoise train.py)
1340 const int cw = pw / bin, chh = ph / bin;
1341 const size_t cplane = (size_t)cw * chh;
1346 rc = 1;
1347 else
1348 {
1351 // the coarse head predicts the correction to the binned RGB planes
1353 if(!rc) dt_nn_upsample_nearest(coarse_out, 3, cw, chh, bin, nn_in + plane * 5);
1354 }
1358 }
1359
1360 // 3. fine pass writes the RAW noise prediction; the residual is a kernel of
1361 // ours, sequenced exactly as process_cl() sequences nn_residual
1362 if(!rc) rc = dt_nn_unet_apply_stage(model, 0, nn_in, nn_out, pw, ph, 0);
1363 if(!rc) _k_residual(nn_in, nn_out, nn_out, plane);
1365 if(rc)
1366 {
1367 dt_print(DT_DEBUG_ALWAYS, "[rawdenoiseai] inference failed (%d) on %dx%d tile, scratch %.1f MB\n", rc, pw, ph,
1368 dt_nn_unet_scratch_bytes(model, pw, ph) / 1048576.0);
1370 }
1371 else
1372 {
1373 _k_blend_crop(in, nn_out, out, width, height, pw, d->strength);
1374 }
1375
1376 _nn_region = NULL;
1378 return rc;
1379}
1380
1381#ifdef HAVE_OPENCL
1382/* GPU path. Step for step the same procedure as process(), each _k_* function
1383 * there having the kernel of the same name here; the whole tile runs
1384 * dev_in -> dev_out with no mid-tile host round-trip, since command-queue syncs
1385 * dominate GPU cost. Returns FALSE on any failure so the pipeline falls back
1386 * to CPU. */
1388 cl_mem dev_in, cl_mem dev_out)
1389{
1390 const dt_iop_roi_t *const roi = &piece->roi_in;
1393 dt_nn_model_t *const model = d->model;
1394 const int devid = pipe->devid;
1395
1396 if(!model || !gd || !gd->nn_cl || !(d->strength > 0.0f)) return FALSE;
1397
1398 const int width = roi->width, height = roi->height;
1399 // pre-shifted for the kernel's tile-local Bayer branch, exactly as in process()
1400 const uint32_t filters = dt_dev_get_roi_filters(piece, roi);
1401 const int is_xtrans = filters == 9u;
1402
1403 const int align = dt_nn_model_alignment(model);
1404 const int pw = (width + align - 1) / align * align;
1405 const int ph = (height + align - 1) / align * align;
1406 const size_t plane = (size_t)pw * ph;
1407 const int in_ch = dt_nn_model_in_channels(model);
1408 const int bin = dt_nn_model_bin(model, is_xtrans);
1409 const int cw = pw / bin, chh = ph / bin;
1410 const size_t cplane = (size_t)cw * chh;
1411 const int anchor = dt_nn_model_anchor(model);
1412 const int gw = pw / 16, gh = ph / 16;
1413 const size_t p16 = (size_t)gw * gh;
1414
1415 float scale[3];
1416 _sigma_scale(d, scale);
1417
1418 /* The whole tile runs on the device: assembly, binning, both network
1419 * stages, residuals, low-band fusion and the final blend are enqueued as
1420 * one chain with no mid-tile readback — command-queue syncs dominate GPU
1421 * cost, not arithmetic. The CPU path (process()) is the bit-parity
1422 * reference for every step. */
1423 gboolean success = FALSE;
1424 int err = CL_SUCCESS;
1425 cl_mem dev_planes = dt_opencl_alloc_device_buffer(devid, plane * in_ch * sizeof(float));
1426 cl_mem dev_noise = dt_opencl_alloc_device_buffer(devid, plane * sizeof(float));
1427 cl_mem dev_den = dt_opencl_alloc_device_buffer(devid, plane * sizeof(float));
1428 cl_mem dev_xtrans = dt_opencl_alloc_device_buffer(devid, 36);
1429 /* The pipeline hands modules their I/O as image2d objects (CL_R float here),
1430 * but every kernel in this chain addresses plain float buffers — passing an
1431 * image where a buffer is declared makes clSetKernelArg fail with
1432 * CL_INVALID_MEM_OBJECT and the enqueue with CL_INVALID_KERNEL_ARGS. Convert
1433 * at the endpoints only: one device-side image->buffer copy of the input, one
1434 * buffer->image copy of the blended result. The interior chain is untouched
1435 * and stays device-resident. */
1436 cl_mem dev_in_buf = dt_opencl_alloc_device_buffer(devid, (size_t)width * height * sizeof(float));
1437 cl_mem dev_out_buf = dt_opencl_alloc_device_buffer(devid, (size_t)width * height * sizeof(float));
1438 cl_mem dev_cin = NULL, dev_chead = NULL, dev_cden = NULL;
1439 // M/D/V per level (16, 32, 64) then the fused ping-pong pair
1440 cl_mem grids[11] = { NULL };
1441 if(!dev_planes || !dev_noise || !dev_den || !dev_xtrans || !dev_in_buf || !dev_out_buf) goto cleanup;
1442 {
1443 size_t origin[3] = { 0, 0, 0 };
1444 size_t region[3] = { (size_t)width, (size_t)height, 1 };
1445 err = dt_opencl_enqueue_copy_image_to_buffer(devid, dev_in, dev_in_buf, origin, region, 0);
1446 if(err != CL_SUCCESS) goto cleanup;
1447 }
1448 if(bin > 1)
1449 {
1450 dev_cin = dt_opencl_alloc_device_buffer(devid, cplane * 6 * sizeof(float));
1451 dev_chead = dt_opencl_alloc_device_buffer(devid, cplane * 3 * sizeof(float));
1452 dev_cden = dt_opencl_alloc_device_buffer(devid, cplane * 3 * sizeof(float));
1453 if(!dev_cin || !dev_chead || !dev_cden) goto cleanup;
1454 }
1455 // same gate as the CPU path's _apply_low_band_anchor(), so both devices fuse
1456 // or skip together — the alignment guarantees the modulo holds
1457 const int do_anchor = anchor > 0 && pw % DT_NN_FUSION_COARSEST == 0 && ph % DT_NN_FUSION_COARSEST == 0;
1458 if(do_anchor)
1459 {
1460 for(int k = 0; k < 11; k++)
1461 {
1462 grids[k] = dt_opencl_alloc_device_buffer(devid, p16 * 3 * sizeof(float));
1463 if(!grids[k]) goto cleanup;
1464 }
1465 }
1466
1467 {
1468 unsigned char xtrans_host[36]; // staging copy: the CL API takes a non-const pointer
1469 memcpy(xtrans_host, piece->dsc_in.xtrans, sizeof(xtrans_host));
1471 goto cleanup;
1472 }
1473
1474 // 1. assemble the base planes (reflect pad + one-hot + sigma)
1475 {
1476 const int K = gd->k_assemble;
1477 const unsigned int f = filters;
1478 dt_opencl_set_kernel_arg(devid, K, 0, sizeof(cl_mem), &dev_in_buf);
1479 dt_opencl_set_kernel_arg(devid, K, 1, sizeof(cl_mem), &dev_planes);
1480 dt_opencl_set_kernel_arg(devid, K, 2, sizeof(int), &width);
1481 dt_opencl_set_kernel_arg(devid, K, 3, sizeof(int), &height);
1482 dt_opencl_set_kernel_arg(devid, K, 4, sizeof(int), &pw);
1483 dt_opencl_set_kernel_arg(devid, K, 5, sizeof(int), &ph);
1484 dt_opencl_set_kernel_arg(devid, K, 6, sizeof(unsigned int), &f);
1485 dt_opencl_set_kernel_arg(devid, K, 7, sizeof(cl_mem), &dev_xtrans);
1486 dt_opencl_set_kernel_arg(devid, K, 8, sizeof(int), &is_xtrans);
1487 dt_opencl_set_kernel_arg(devid, K, 9, sizeof(int), &roi->x);
1488 dt_opencl_set_kernel_arg(devid, K, 10, sizeof(int), &roi->y);
1489 for(int c = 0; c < 3; c++) dt_opencl_set_kernel_arg(devid, K, 11 + c, sizeof(float), &d->a[c]);
1490 for(int c = 0; c < 3; c++) dt_opencl_set_kernel_arg(devid, K, 14 + c, sizeof(float), &d->b[c]);
1491 for(int c = 0; c < 3; c++) dt_opencl_set_kernel_arg(devid, K, 17 + c, sizeof(float), &scale[c]);
1492 size_t sizes[3] = { ROUNDUPDWD(pw, devid), ROUNDUPDHT(ph, devid), 1 };
1494 if(err != CL_SUCCESS) goto cleanup;
1495 }
1496
1497 // 2. coarse chroma pass (multiscale models)
1498 if(bin > 1)
1499 {
1500 const int K = gd->k_bin_planes;
1501 dt_opencl_set_kernel_arg(devid, K, 0, sizeof(cl_mem), &dev_planes);
1502 dt_opencl_set_kernel_arg(devid, K, 1, sizeof(cl_mem), &dev_cin);
1503 dt_opencl_set_kernel_arg(devid, K, 2, sizeof(int), &pw);
1504 dt_opencl_set_kernel_arg(devid, K, 3, sizeof(int), &ph);
1505 dt_opencl_set_kernel_arg(devid, K, 4, sizeof(int), &bin);
1506 for(int c = 0; c < 3; c++) dt_opencl_set_kernel_arg(devid, K, 5 + c, sizeof(float), &d->a[c]);
1507 for(int c = 0; c < 3; c++) dt_opencl_set_kernel_arg(devid, K, 8 + c, sizeof(float), &d->b[c]);
1508 for(int c = 0; c < 3; c++) dt_opencl_set_kernel_arg(devid, K, 11 + c, sizeof(float), &scale[c]);
1509 size_t sizes[3] = { ROUNDUPDWD(cw, devid), ROUNDUPDHT(chh * 3, devid), 1 };
1511 if(err != CL_SUCCESS) goto cleanup;
1512
1513 err = dt_nn_unet_apply_stage_cl(model, 1, gd->nn_cl, devid, dev_cin, dev_chead, cw, chh);
1514 if(err != CL_SUCCESS) goto cleanup;
1515 const int KR = gd->k_residual;
1516 const int n3 = (int)(cplane * 3);
1517 dt_opencl_set_kernel_arg(devid, KR, 0, sizeof(cl_mem), &dev_cin);
1518 dt_opencl_set_kernel_arg(devid, KR, 1, sizeof(cl_mem), &dev_chead);
1519 dt_opencl_set_kernel_arg(devid, KR, 2, sizeof(cl_mem), &dev_cden);
1520 dt_opencl_set_kernel_arg(devid, KR, 3, sizeof(int), &n3);
1521 size_t sz[3] = { ROUNDUPDWD(n3, devid), 1, 1 };
1523 if(err != CL_SUCCESS) goto cleanup;
1524 cl_mem guide_src = dev_cden;
1525 const int KU = gd->k_upsample_n;
1526 const int three = 3;
1527 const cl_long dst_off = (cl_long)plane * 5;
1528 dt_opencl_set_kernel_arg(devid, KU, 0, sizeof(cl_mem), &guide_src);
1529 dt_opencl_set_kernel_arg(devid, KU, 1, sizeof(cl_mem), &dev_planes);
1530 dt_opencl_set_kernel_arg(devid, KU, 2, sizeof(int), &cw);
1531 dt_opencl_set_kernel_arg(devid, KU, 3, sizeof(int), &chh);
1532 dt_opencl_set_kernel_arg(devid, KU, 4, sizeof(int), &bin);
1533 dt_opencl_set_kernel_arg(devid, KU, 5, sizeof(int), &three);
1534 dt_opencl_set_kernel_arg(devid, KU, 6, sizeof(cl_long), &dst_off);
1535 size_t sizes_u[3] = { ROUNDUPDWD(pw, devid), ROUNDUPDHT(ph * 3, devid), 1 };
1537 if(err != CL_SUCCESS) goto cleanup;
1538 }
1539
1540 // 3. fine pass (raw noise) + residual -> denoised plane
1541 err = dt_nn_unet_apply_stage_cl(model, 0, gd->nn_cl, devid, dev_planes, dev_noise, pw, ph);
1542 if(err != CL_SUCCESS)
1543 {
1544 dt_print(DT_DEBUG_OPENCL, "[rawdenoiseai] GPU inference failed on %dx%d tile, falling back to CPU\n", pw, ph);
1545 goto cleanup;
1546 }
1547 {
1548 const int KR = gd->k_residual;
1549 const int n1 = (int)plane;
1550 dt_opencl_set_kernel_arg(devid, KR, 0, sizeof(cl_mem), &dev_planes);
1551 dt_opencl_set_kernel_arg(devid, KR, 1, sizeof(cl_mem), &dev_noise);
1552 dt_opencl_set_kernel_arg(devid, KR, 2, sizeof(cl_mem), &dev_den);
1553 dt_opencl_set_kernel_arg(devid, KR, 3, sizeof(int), &n1);
1554 size_t sz[3] = { ROUNDUPDWD(n1, devid), 1, 1 };
1556 if(err != CL_SUCCESS) goto cleanup;
1557 }
1558
1559 // 4. hybrid low-band fusion, entirely on device
1560 if(do_anchor)
1561 {
1562 {
1563 const int K = gd->k_bin16_mdv;
1564 dt_opencl_set_kernel_arg(devid, K, 0, sizeof(cl_mem), &dev_planes);
1565 dt_opencl_set_kernel_arg(devid, K, 1, sizeof(cl_mem), &dev_den);
1566 dt_opencl_set_kernel_arg(devid, K, 2, sizeof(cl_mem), &grids[0]);
1567 dt_opencl_set_kernel_arg(devid, K, 3, sizeof(cl_mem), &grids[1]);
1568 dt_opencl_set_kernel_arg(devid, K, 4, sizeof(cl_mem), &grids[2]);
1569 dt_opencl_set_kernel_arg(devid, K, 5, sizeof(int), &pw);
1570 dt_opencl_set_kernel_arg(devid, K, 6, sizeof(int), &ph);
1571 size_t sizes[3] = { ROUNDUPDWD(gw, devid), ROUNDUPDHT(gh * 3, devid), 1 };
1573 if(err != CL_SUCCESS) goto cleanup;
1574 }
1575 const int nlev = _fusion_levels();
1576 int lw = gw, lh = gh;
1577 for(int k = 1; k < nlev; k++)
1578 {
1579 for(int md = 0; md < 3; md++)
1580 {
1581 const int K = gd->k_avg2x2;
1582 dt_opencl_set_kernel_arg(devid, K, 0, sizeof(cl_mem), &grids[3 * (k - 1) + md]);
1583 dt_opencl_set_kernel_arg(devid, K, 1, sizeof(cl_mem), &grids[3 * k + md]);
1584 dt_opencl_set_kernel_arg(devid, K, 2, sizeof(int), &lw);
1585 dt_opencl_set_kernel_arg(devid, K, 3, sizeof(int), &lh);
1586 size_t sizes[3] = { ROUNDUPDWD(lw / 2, devid), ROUNDUPDHT((lh / 2) * 3, devid), 1 };
1588 if(err != CL_SUCCESS) goto cleanup;
1589 }
1590 lw /= 2;
1591 lh /= 2;
1592 }
1593 // floor band: structure-gated blend (see the CPU comment)
1594 int fA = 9, fB = 10; // fused ping-pong, past the 3 x 3 level grids
1595 {
1596 const int K = gd->k_floor_fuse;
1597 const int S = 16 << (nlev - 1);
1598 dt_opencl_set_kernel_arg(devid, K, 0, sizeof(cl_mem), &grids[3 * (nlev - 1)]);
1599 dt_opencl_set_kernel_arg(devid, K, 1, sizeof(cl_mem), &grids[3 * (nlev - 1) + 1]);
1600 dt_opencl_set_kernel_arg(devid, K, 2, sizeof(cl_mem), &grids[fA]);
1601 dt_opencl_set_kernel_arg(devid, K, 3, sizeof(cl_mem), &grids[3 * (nlev - 1) + 2]);
1602 dt_opencl_set_kernel_arg(devid, K, 4, sizeof(int), &lw);
1603 dt_opencl_set_kernel_arg(devid, K, 5, sizeof(int), &lh);
1604 dt_opencl_set_kernel_arg(devid, K, 6, sizeof(int), &S);
1605 for(int c = 0; c < 3; c++) dt_opencl_set_kernel_arg(devid, K, 7 + c, sizeof(float), &DT_NN_FUSION_DENS[c]);
1606 size_t sizes[3] = { ROUNDUPDWD(lw, devid), ROUNDUPDHT(lh * 3, devid), 1 };
1608 if(err != CL_SUCCESS) goto cleanup;
1609 }
1610 for(int k = nlev - 2; k >= 0; k--)
1611 {
1612 const int fw = lw * 2, fh = lh * 2, sc = 16 << k;
1613 {
1614 const int K = gd->k_fuse_step;
1615 dt_opencl_set_kernel_arg(devid, K, 0, sizeof(cl_mem), &grids[fA]);
1616 dt_opencl_set_kernel_arg(devid, K, 1, sizeof(cl_mem), &grids[3 * k]);
1617 dt_opencl_set_kernel_arg(devid, K, 2, sizeof(cl_mem), &grids[3 * k + 1]);
1618 dt_opencl_set_kernel_arg(devid, K, 3, sizeof(cl_mem), &grids[3 * (k + 1)]);
1619 dt_opencl_set_kernel_arg(devid, K, 4, sizeof(cl_mem), &grids[3 * (k + 1) + 1]);
1620 dt_opencl_set_kernel_arg(devid, K, 5, sizeof(cl_mem), &grids[fB]);
1621 dt_opencl_set_kernel_arg(devid, K, 6, sizeof(cl_mem), &grids[3 * k + 2]);
1622 dt_opencl_set_kernel_arg(devid, K, 7, sizeof(int), &fw);
1623 dt_opencl_set_kernel_arg(devid, K, 8, sizeof(int), &fh);
1624 dt_opencl_set_kernel_arg(devid, K, 9, sizeof(int), &sc);
1625 for(int c = 0; c < 3; c++) dt_opencl_set_kernel_arg(devid, K, 10 + c, sizeof(float), &DT_NN_FUSION_DENS[c]);
1626 size_t sizes[3] = { ROUNDUPDWD(fw, devid), ROUNDUPDHT(fh * 3, devid), 1 };
1628 if(err != CL_SUCCESS) goto cleanup;
1629 }
1630 const int t = fA;
1631 fA = fB;
1632 fB = t;
1633 lw = fw;
1634 lh = fh;
1635 }
1636 {
1637 const int K = gd->k_bilerp_add;
1638 dt_opencl_set_kernel_arg(devid, K, 0, sizeof(cl_mem), &grids[fA]);
1639 dt_opencl_set_kernel_arg(devid, K, 1, sizeof(cl_mem), &grids[1]);
1640 dt_opencl_set_kernel_arg(devid, K, 2, sizeof(cl_mem), &dev_planes);
1641 dt_opencl_set_kernel_arg(devid, K, 3, sizeof(cl_mem), &dev_den);
1642 dt_opencl_set_kernel_arg(devid, K, 4, sizeof(int), &pw);
1643 dt_opencl_set_kernel_arg(devid, K, 5, sizeof(int), &ph);
1644 size_t sizes[3] = { ROUNDUPDWD(pw, devid), ROUNDUPDHT(ph, devid), 1 };
1646 if(err != CL_SUCCESS) goto cleanup;
1647 }
1648 }
1649
1650 // 5. strength blend + crop straight into the pipeline output
1651 {
1652 const int K = gd->k_blend_crop;
1653 dt_opencl_set_kernel_arg(devid, K, 0, sizeof(cl_mem), &dev_in_buf);
1654 dt_opencl_set_kernel_arg(devid, K, 1, sizeof(cl_mem), &dev_den);
1655 dt_opencl_set_kernel_arg(devid, K, 2, sizeof(cl_mem), &dev_out_buf);
1656 dt_opencl_set_kernel_arg(devid, K, 3, sizeof(int), &width);
1657 dt_opencl_set_kernel_arg(devid, K, 4, sizeof(int), &height);
1658 dt_opencl_set_kernel_arg(devid, K, 5, sizeof(int), &pw);
1659 dt_opencl_set_kernel_arg(devid, K, 6, sizeof(float), &d->strength);
1660 size_t sizes[3] = { ROUNDUPDWD(width, devid), ROUNDUPDHT(height, devid), 1 };
1662 if(err != CL_SUCCESS) goto cleanup;
1663 }
1664 {
1665 // hand the result back in the pipeline's format (see the entry copy above)
1666 size_t origin[3] = { 0, 0, 0 };
1667 size_t region[3] = { (size_t)width, (size_t)height, 1 };
1668 err = dt_opencl_enqueue_copy_buffer_to_image(devid, dev_out_buf, dev_out, 0, origin, region);
1669 if(err != CL_SUCCESS) goto cleanup;
1670 }
1671 success = TRUE;
1672
1673cleanup:
1683 for(int k = 0; k < 11; k++)
1685 return success;
1686}
1687#endif // HAVE_OPENCL
1688
1693
1699
1701{
1702 dt_free_align(piece->data);
1703 piece->data = NULL;
1704}
1705
1706/* The combo lists "(shipped model)" then whatever .anselnn files the config
1707 * dir holds. The VALUE written to params is the basename, never the index —
1708 * see the params comment. gui_update re-derives the index from the name and
1709 * appends an entry for a name that is no longer on disk, so an edit made with
1710 * a since-removed model still shows what it wants instead of silently
1711 * displaying the wrong one. */
1713{
1714 if(dt_gui_widgets_suppressed()) return;
1717 const int idx = dt_bauhaus_combobox_get(w);
1718 const char *txt = idx > 0 ? dt_bauhaus_combobox_get_text(w) : NULL;
1719 if(txt)
1720 g_strlcpy(p->custom_model, txt, sizeof(p->custom_model));
1721 else
1722 p->custom_model[0] = '\0';
1723 dt_dev_add_history_item(self->dev, self, TRUE, TRUE);
1724}
1725
1727{
1730 dt_bauhaus_combobox_clear(g->custom_model);
1731 dt_bauhaus_combobox_add(g->custom_model, _("(shipped model)"));
1732 GList *files = _list_custom_models();
1733 int sel = 0, i = 0;
1734 for(GList *l = files; l; l = g_list_next(l))
1735 {
1736 dt_bauhaus_combobox_add(g->custom_model, (const char *)l->data);
1737 i++;
1738 if(!g_strcmp0((const char *)l->data, p->custom_model)) sel = i;
1739 }
1740 // the edit names a file that is not there any more: keep it visible
1741 if(p->custom_model[0] && !sel)
1742 {
1743 dt_bauhaus_combobox_add(g->custom_model, p->custom_model);
1744 sel = i + 1;
1745 }
1746 g_list_free_full(files, g_free);
1747 dt_bauhaus_combobox_set(g->custom_model, sel);
1748 // the shipped-matrix combos are meaningless while a user model is selected
1749 const gboolean shipped = !p->custom_model[0];
1752 gtk_widget_set_sensitive(g->scale_variant, shipped);
1753}
1754
1756{
1760 gtk_stack_set_visible_child_name(GTK_STACK(self->gui->widget), self->hide_enable_button ? "unsupported" : "raw");
1761
1762 // rescan on every panel update: the user may have dropped a file in since
1764
1765 // two-line status: the selected model (warn if its weights are missing) and
1766 // the noise profile the sigma map will use
1767 const gboolean have_model = gd && (p->custom_model[0] ? _get_custom_model(gd, p->custom_model)
1768 : _get_model(gd, p->version, p->size, p->scale_variant));
1770 gchar *prof = profiles
1771 ? g_strdup_printf(_("noise profile: %s at ISO %d"), self->dev->image_storage.camera_makermodel,
1772 (int)self->dev->image_storage.exif_iso)
1773 : g_strdup(_("no noise profile for this camera — using the generic profile"));
1774 gchar *label = have_model
1775 ? g_strdup(prof)
1776 : g_strdup_printf(_("selected model (%s %s, %s) is not installed — module inactive\n%s"),
1778 _scale_tag[CLAMP(p->scale_variant, 0, DT_RAWDENOISEAI_NUM_SCALES - 1)],
1780 gtk_label_set_text(GTK_LABEL(g->profile_label), label);
1782 g_free(label);
1783 g_free(prof);
1785}
1786
1788{
1790
1792
1793 g->strength = dt_bauhaus_slider_from_params(self, "strength");
1794 dt_bauhaus_slider_set_digits(g->strength, 3);
1795 dt_bauhaus_slider_set_format(g->strength, "%");
1796 gtk_widget_set_tooltip_text(g->strength, _("opacity of the noise removal: blends between the original\n"
1797 "image (0%) and the fully denoised result (100%).\n"
1798 "lower it to keep some residual grain"));
1799
1800 g->version = dt_bauhaus_combobox_from_params(self, "version");
1801 gtk_widget_set_tooltip_text(g->version, _("neural model version. Older edits keep their original\n"
1802 "version so their result never changes across updates."));
1803
1804 g->size = dt_bauhaus_combobox_from_params(self, "size");
1805 gtk_widget_set_tooltip_text(g->size, _("network width. large: reference quality, practical on GPU\n"
1806 "(OpenCL) and the default there. half: ~4x faster.\n"
1807 "quarter: ~4x faster again, the default without OpenCL\n"
1808 "and the choice for weak hardware or near-realtime editing."));
1809
1810 g->scale_variant = dt_bauhaus_combobox_from_params(self, "scale_variant");
1811 gtk_widget_set_tooltip_text(g->scale_variant, _("single-scale: the fine full-resolution pass only — fast,\n"
1812 "no low-frequency chroma handling. multiscale: adds the\n"
1813 "coarse chroma pass and the low-band fusion — high quality,\n"
1814 "recommended for high ISO."));
1815
1817 dt_bauhaus_widget_set_label(g->custom_model, N_("custom model"));
1818 gtk_box_pack_start(GTK_BOX(box_raw), g->custom_model, TRUE, TRUE, 0);
1819 gtk_widget_set_tooltip_text(g->custom_model,
1820 _("use a neural model of your own instead of the shipped ones.\n"
1821 "drop a .anselnn file into your Ansel config directory and it\n"
1822 "appears here; the edit records the file NAME, so it keeps\n"
1823 "pointing at the same model as the folder changes."));
1824 g_signal_connect(G_OBJECT(g->custom_model), "value-changed", G_CALLBACK(_custom_model_callback), self);
1825
1826 gtk_box_pack_start(GTK_BOX(box_raw), dt_ui_section_label_new(_("noise profile correction")), FALSE, FALSE, 0);
1827
1828 g->profile_label = dt_ui_label_new("");
1829 gtk_label_set_line_wrap(GTK_LABEL(g->profile_label), TRUE);
1830 gtk_box_pack_start(GTK_BOX(box_raw), g->profile_label, FALSE, FALSE, 0);
1831
1832 g->noise_level = dt_bauhaus_slider_from_params(self, "noise_level");
1833 dt_bauhaus_slider_set_digits(g->noise_level, 3);
1834 dt_bauhaus_slider_set_format(g->noise_level, "%");
1835 gtk_widget_set_tooltip_text(g->noise_level, _("global scale on the assumed noise amplitude, relative to the\n"
1836 "calibrated noise for this camera at this ISO (100% trusts the\n"
1837 "calibration). raise it if noise remains, lower it if fine\n"
1838 "detail is eaten"));
1839
1840 dt_gui_new_collapsible_section(&g->cs, "plugins/darkroom/rawdenoiseai/expand_channel",
1841 _("per-channel corrections"), GTK_BOX(box_raw), GTK_PACK_START);
1842 self->gui->widget = GTK_WIDGET(g->cs.container); // sliders below pack into the section
1843
1844 const char *sigma_tooltip = _("per-channel correction of the camera noise profile, applied on top\n"
1845 "of the global correction. Profiles are measured after demosaicing,\n"
1846 "which averages away part of the noise — most on the dense green\n"
1847 "lattice — while this module sees the raw sensor noise at full\n"
1848 "strength. The defaults are calibrated against raw-mosaic\n"
1849 "measurements over 253 cameras.");
1850 g->sigma_red = dt_bauhaus_slider_from_params(self, "sigma_red");
1851 dt_bauhaus_slider_set_digits(g->sigma_red, 3);
1852 dt_bauhaus_slider_set_format(g->sigma_red, "%");
1854
1855 g->sigma_green = dt_bauhaus_slider_from_params(self, "sigma_green");
1856 dt_bauhaus_slider_set_digits(g->sigma_green, 3);
1857 dt_bauhaus_slider_set_format(g->sigma_green, "%");
1859
1860 g->sigma_blue = dt_bauhaus_slider_from_params(self, "sigma_blue");
1861 dt_bauhaus_slider_set_digits(g->sigma_blue, 3);
1862 dt_bauhaus_slider_set_format(g->sigma_blue, "%");
1864
1865 self->gui->widget = box_raw; // done packing into the collapsible section
1866
1867 GtkWidget *label_unsupported = dt_ui_label_new(_("AI raw denoising needs a mosaiced raw image\n"
1868 "and an installed rawdenoiseai model file."));
1869
1870 self->gui->widget = gtk_stack_new();
1872 gtk_stack_add_named(GTK_STACK(self->gui->widget), label_unsupported, "unsupported");
1874}
1875
1877{
1879}
1880// clang-format off
1881// modelines: These editor modelines have been set for all relevant files by tools/update_modelines.py
1882// vim: shiftwidth=2 expandtab tabstop=2 cindent
1883// kate: tab-indents: off; indent-width 2; replace-tabs on; indent-mode cstyle; remove-trailing-spaces modified;
1884// clang-format on
static double * inv
#define TRUE
Definition ashift_lsd.c:162
#define FALSE
Definition ashift_lsd.c:158
void cleanup(dt_imageio_module_format_t *self)
Definition avif.c:170
#define m
Definition basecurve.c:283
void dt_bauhaus_slider_set_digits(GtkWidget *widget, int val)
Definition bauhaus.c:3343
void dt_bauhaus_combobox_clear(GtkWidget *widget)
Definition bauhaus.c:1997
int dt_bauhaus_combobox_get(GtkWidget *widget)
Definition bauhaus.c:2136
const char * dt_bauhaus_combobox_get_text(GtkWidget *widget)
Definition bauhaus.c:1970
void dt_bauhaus_combobox_set(GtkWidget *widget, const int pos)
Definition bauhaus.c:2090
void dt_bauhaus_widget_set_label(GtkWidget *widget, const char *label)
Definition bauhaus.c:1504
GtkWidget * dt_bauhaus_combobox_new(dt_bauhaus_t *bh, dt_gui_module_t *self)
Definition bauhaus.c:1698
void dt_bauhaus_slider_set_format(GtkWidget *widget, const char *format)
Definition bauhaus.c:3407
void dt_bauhaus_combobox_add(GtkWidget *widget, const char *text)
Definition bauhaus.c:1824
void dt_gui_new_collapsible_section(dt_gui_collapsible_section_t *cs, const char *confname, const char *label, GtkBox *parent, GtkPackType pack)
Create a collapsible section and pack it into the parent box.
void dt_gui_update_collapsible_section(dt_gui_collapsible_section_t *cs)
@ IOP_CS_RAW
static const float x
const float f
const int t
const float v
struct _GtkWidget GtkWidget
GtkWidget, opaque, spelled exactly as GTK spells it.
Definition colorspaces.h:98
const dt_colormatrix_t dt_aligned_pixel_t out
const float top
static const dt_colormatrix_t M
static float strength(float value, float strength)
Definition colorzones.c:431
#define S(V, params)
gboolean dt_image_needs_demosaic(const dt_image_t *img)
struct dt_bauhaus_t * dt_bauhaus_get_global(void)
Definition darktable.c:646
static int FCxtrans(const int row, const int col, global const unsigned char(*const xtrans)[6])
static int FC(const int row, const int col, const unsigned int filters)
#define dt_dev_add_history_item(dev, module, enable, redraw)
void dt_iop_params_t
Definition dev_history.h:43
void default_input_format(dt_iop_module_t *self, dt_dev_pixelpipe_t *pipe, dt_dev_pixelpipe_iop_t *piece, dt_iop_buffer_dsc_t *dsc)
static int dt_pthread_mutex_unlock(dt_pthread_mutex_t *mutex) RELEASE(mutex) NO_THREAD_SAFETY_ANALYSIS
Definition dtpthread.h:127
static int dt_pthread_mutex_init(dt_pthread_mutex_t *mutex, const pthread_mutexattr_t *mutexattr)
Initialise a mutex. With mutexattr NULL – which is how 54 of the 56 call sites in this tree spell it ...
Definition dtpthread.h:104
static int dt_pthread_mutex_destroy(dt_pthread_mutex_t *mutex)
Definition dtpthread.h:132
static int dt_pthread_mutex_lock(dt_pthread_mutex_t *mutex) ACQUIRE(mutex) NO_THREAD_SAFETY_ANALYSIS
Definition dtpthread.h:117
void dt_loc_get_datadir(char *datadir, size_t bufsize)
void dt_loc_get_user_config_dir(char *configdir, size_t bufsize)
#define DT_GUI_MODULE(x)
static void dt_iop_image_copy_by_size(float *const __restrict__ out, const float *const __restrict__ in, const size_t width, const size_t height, const size_t ch)
Definition imagebuf.h:91
uint32_t dt_dev_get_roi_filters(const dt_dev_pixelpipe_iop_t *const piece, const dt_iop_roi_t *const roi_in)
Definition imageop.c:139
void dt_iop_default_init(dt_iop_module_t *module)
Definition imageop.c:308
const char ** dt_iop_set_description(dt_iop_module_t *module, const char *main_text, const char *purpose, const char *input, const char *process, const char *output)
Definition imageop.c:1909
void dt_iop_request_focus(dt_iop_module_t *module)
Move darkroom focus to module, or clear it with NULL.
@ IOP_FLAGS_SUPPORTS_BLENDING
Definition imageop.h:186
@ IOP_FLAGS_ALLOW_TILING
Definition imageop.h:188
@ IOP_GROUP_REPAIR
Definition imageop.h:159
GtkWidget * dt_bauhaus_slider_from_params(dt_iop_module_t *self, const char *param)
GtkWidget * dt_bauhaus_combobox_from_params(dt_iop_module_t *self, const char *param)
#define IOP_GUI_FREE
Definition imageop_gui.h:96
static dt_iop_gui_data_t * dt_iop_gui_data(const struct dt_iop_module_t *m)
The module's GUI data blob, NULL-safe for headless callers: IOP process() implementations read it for...
Definition imageop_gui.h:81
#define IOP_GUI_ALLOC(module)
Definition imageop_gui.h:93
void *const ovoid
const char * model
GtkWidget * dt_ui_section_label_new(const gchar *str)
Definition label.c:114
GtkWidget * dt_ui_label_new(const gchar *str)
Definition label.c:125
static const dt_dev_pixelpipe_t * pipe
Definition layers.h:49
static const dt_dev_pixelpipe_t const dt_dev_pixelpipe_iop_t * piece
Definition layers.h:50
static const dt_iop_drawlayer_params_t * params
Definition layers.h:80
#define w2
Definition lmmse.c:60
@ DT_DEBUG_OPENCL
Definition logging.h:57
@ DT_DEBUG_PARAMS
Definition logging.h:71
@ DT_DEBUG_ALWAYS
Definition logging.h:49
void dt_print(dt_debug_thread_t thread, const char *msg,...) __attribute__((format(printf
Print to stdout when thread is enabled, prefixed with seconds since startup.
float *const restrict const size_t k
#define IS_NULL_PTR(p)
C is way too permissive with !=, == and if(var) checks, which can mean too many things depending on w...
Definition macros.h:96
#define dt_free_align(ptr)
Release memory from dt_alloc_align() and set ptr to NULL.
Definition mem_alloc.h:214
static void * dt_calloc_align(size_t size)
dt_alloc_align() followed by a zero fill.
Definition mem_alloc.h:225
uint32_t width
Definition mipmap_cache.c:0
uint32_t height
Definition mipmap_cache.c:1
#define DT_MODULE_INTROSPECTION(MODVER, PARAMSTYPE)
DT_MODULE() for a module whose params struct is introspected.
static float gh(const float f)
void dt_nn_cl_destroy(dt_nn_cl_t *cl)
Definition nn_model.c:1098
int dt_nn_unet_apply_stage_cl(const dt_nn_model_t *m, int stage, dt_nn_cl_t *cl, int devid, cl_mem dev_in, cl_mem dev_out, int width, int height)
Definition nn_model.c:1338
int dt_nn_model_anchor(const dt_nn_model_t *m)
Definition nn_model.c:442
void dt_nn_model_free(dt_nn_model_t *m)
Definition nn_model.c:405
int dt_nn_model_in_channels(const dt_nn_model_t *m)
Definition nn_model.c:416
int dt_nn_model_coarse_in_channels(const dt_nn_model_t *m)
Definition nn_model.c:432
dt_nn_cl_t * dt_nn_cl_create(int program)
Definition nn_model.c:1088
size_t dt_nn_unet_scratch_bytes(const dt_nn_model_t *m, int width, int height)
Definition nn_model.c:736
int dt_nn_unet_apply_stage(const dt_nn_model_t *m, int stage, const float *in, float *out, int width, int height, int apply_residual)
Definition nn_model.c:1009
float dt_nn_unet_scratch_per_px(const dt_nn_model_t *m)
Definition nn_model.c:707
dt_nn_model_t * dt_nn_model_load(const char *path, char *err, size_t err_len)
Definition nn_model.c:237
__DT_CLONE_TARGETS__ void dt_nn_upsample_nearest(const float *in, int ch, int w, int h, int factor, float *out)
Definition nn_model.c:1058
int dt_nn_model_coarse_out_channels(const dt_nn_model_t *m)
Definition nn_model.c:437
float dt_nn_unet_scratch_per_px_cl(const dt_nn_model_t *m)
Definition nn_model.c:712
int dt_nn_model_alignment(const dt_nn_model_t *m)
Definition nn_model.c:460
__DT_CLONE_TARGETS__ void dt_nn_bin_planes(const float *planes, int pw, int ph, int bin, float *out_rgb, float *out_cnt)
Definition nn_model.c:1022
int dt_nn_model_bin(const dt_nn_model_t *m, const int is_xtrans)
Definition nn_model.c:426
void dt_nn_set_allocator(dt_nn_alloc_f alloc_fn, dt_nn_free_f free_fn)
Definition nn_model.c:77
#define DT_NN_FUSION_COARSEST
Definition nn_model.h:98
#define DT_NN_FUSION_FINEST
Definition nn_model.h:97
const dt_noiseprofile_t dt_noiseprofile_generic
void dt_noiseprofile_interpolate(const dt_noiseprofile_t *const p1, const dt_noiseprofile_t *const p2, dt_noiseprofile_t *out)
void dt_noiseprofile_free(gpointer data)
GList * dt_noiseprofile_get_matching(const dt_image_t *cimg)
int dt_opencl_enqueue_kernel_2d(const int dev, const int kernel, const size_t *sizes)
Definition opencl.c:2595
void * dt_opencl_alloc_device_buffer(const int devid, const size_t size)
Definition opencl.c:3011
int dt_opencl_enqueue_copy_buffer_to_image(const int devid, cl_mem src_buffer, cl_mem dst_image, size_t offset, size_t *origin, size_t *region)
Definition opencl.c:2743
int dt_opencl_create_kernel(const int prog, const char *name)
Definition opencl.c:2489
int dt_opencl_write_buffer_to_device(const int devid, void *host, void *device, const size_t offset, const size_t size, const int blocking)
Definition opencl.c:2779
int dt_opencl_is_enabled(void)
Definition opencl.c:3303
void dt_opencl_free_kernel(const int kernel)
Definition opencl.c:2532
int dt_opencl_set_kernel_arg(const int dev, const int kernel, const int num, const size_t size, const void *arg)
Definition opencl.c:2586
int dt_opencl_enqueue_copy_image_to_buffer(const int devid, cl_mem src_image, cl_mem dst_buffer, size_t *origin, size_t *region, size_t offset)
Definition opencl.c:2731
void dt_opencl_release_mem_object(cl_mem mem)
Definition opencl.c:2846
#define ROUNDUPDHT(a, b)
Definition opencl.h:86
#define ROUNDUPDWD(a, b)
Definition opencl.h:85
#define __OMP_PARALLEL_FOR__(...)
Definition openmp.h:95
#define DT_PATH_MAX
Buffer size for a filesystem path anywhere in Ansel.
Definition paths.h:57
void dt_concat_path_file(char destination[4096], const char path[4096], const char *const file)
Append a constant filename to a variable, stack-based, fixed-sized, directory, and add a / in-between...
Definition darktable.c:2674
void dt_iop_buffer_dsc_update_bpp(dt_iop_buffer_dsc_t *dsc)
#define dt_pixelpipe_cache_alloc_align_float_cache(pixels, id)
#define dt_pixelpipe_cache_alloc_align_cache(size, id)
#define dt_pixelpipe_cache_free_align(mem)
static dt_nn_model_t * _get_custom_model(dt_iop_rawdenoiseai_global_data_t *gd, const char *base)
void init(dt_iop_module_t *module)
const char ** description(struct dt_iop_module_t *self)
int default_group()
static const char *const _scale_tag[2]
__DT_CLONE_TARGETS__ int process(struct dt_iop_module_t *self, const dt_dev_pixelpipe_t *pipe, const dt_dev_pixelpipe_iop_t *piece, const void *const ivoid, void *const ovoid)
static void _region_free(nn_region_t *r, void *p)
#define DT_NN_FUSION_T_CHI2
static __DT_CLONE_TARGETS__ void _k_bin16_mdv(const float *const nn_in, const float *const denoised, float *const M, float *const D, float *const V, const int pw, const int ph)
void reload_defaults(dt_iop_module_t *module)
static int _apply_low_band_anchor(const float *const nn_in, float *const denoised, const int pw, const int ph, const int scale)
static void _sigma_scale(const dt_iop_rawdenoiseai_data_t *d, float scale[3])
void commit_params(struct dt_iop_module_t *self, dt_iop_params_t *params, dt_dev_pixelpipe_t *pipe, dt_dev_pixelpipe_iop_t *piece)
#define NN_REGION_MAX_BLOCKS
dt_iop_rawdenoiseai_scale_t
@ DT_RAWDENOISEAI_MULTI
@ DT_RAWDENOISEAI_SINGLE
void gui_update(dt_iop_module_t *self)
Refresh GUI controls from current params and configuration.
#define NN_REGION_SLACK
static __DT_CLONE_TARGETS__ void _upsample_bilinear(const float *const src, const int sw, const int sh, const int f, float *const dst)
static void _nn_arena_free(void *p)
static __DT_CLONE_TARGETS__ void _k_residual(const float *const in, const float *const head, float *const out, const size_t n)
void init_pipe(struct dt_iop_module_t *self, dt_dev_pixelpipe_t *pipe, dt_dev_pixelpipe_iop_t *piece)
static gboolean _rawdenoiseai_supported(dt_iop_module_t *module)
static GList * _list_custom_models(void)
const char * name()
void gui_init(dt_iop_module_t *self)
static void _custom_model_populate(dt_iop_module_t *self)
static __DT_CLONE_TARGETS__ void _k_bin_planes(const float *const nn_in, float *const coarse_in, float *const cnt, const int pw, const int ph, const int bin, const dt_iop_rawdenoiseai_data_t *const d)
#define DT_RAWDENOISEAI_MODEL_LEN
static __DT_CLONE_TARGETS__ void _k_assemble(const float *const in, float *const nn_in, const int width, const int height, const int pw, const int ph, const dt_iop_rawdenoiseai_data_t *const d, const uint32_t filters, const uint8_t(*const xtrans)[6], const dt_iop_roi_t *const roi)
static __thread nn_region_t * _nn_region
void tiling_callback(struct dt_iop_module_t *self, const struct dt_dev_pixelpipe_t *pipe, const struct dt_dev_pixelpipe_iop_t *piece, struct dt_develop_tiling_t *tiling)
static __DT_CLONE_TARGETS__ void _k_bilerp_add(const float *const fused, const float *const D16, float *const scratch, const float *const nn_in, float *const denoised, const int pw, const int ph, const size_t p0, const int cw0, const int ch0)
void gui_cleanup(dt_iop_module_t *self)
static void _fetch_noise_profile(dt_iop_module_t *self, dt_iop_rawdenoiseai_data_t *d)
void cleanup_global(dt_iop_module_so_t *module)
__DT_CLONE_TARGETS__ static __DT_CLONE_TARGETS__ void _k_floor_fuse(const float *const M, const float *const D, const float *const V, float *const fused, const int sw, const int sh, const size_t p0, const int S)
static void _custom_model_callback(GtkWidget *w, dt_iop_module_t *self)
int default_colorspace(dt_iop_module_t *self, dt_dev_pixelpipe_t *pipe, const dt_dev_pixelpipe_iop_t *piece)
void input_format(dt_iop_module_t *self, dt_dev_pixelpipe_t *pipe, dt_dev_pixelpipe_iop_t *piece, dt_iop_buffer_dsc_t *dsc)
static const char *const _version_tag[1]
int flags()
dt_iop_rawdenoiseai_size_t
@ DT_RAWDENOISEAI_LARGE
@ DT_RAWDENOISEAI_QUARTER
@ DT_RAWDENOISEAI_HALF
static const char *const _size_tag[3]
static void * _nn_arena_alloc(size_t bytes, int long_lived)
gboolean force_enable(struct dt_iop_module_t *self, const gboolean current_state)
void cleanup_pipe(struct dt_iop_module_t *self, dt_dev_pixelpipe_t *pipe, dt_dev_pixelpipe_iop_t *piece)
static unsigned _align_lcm(const unsigned a, const unsigned b)
static void * _region_alloc(nn_region_t *r, size_t bytes, int long_lived)
#define DT_RAWDENOISEAI_NUM_SCALES
void init_global(dt_iop_module_so_t *module)
static __DT_CLONE_TARGETS__ void _k_avg2x2(const float *const in, float *const out, const int sw, const int sh, const size_t p0)
static __DT_CLONE_TARGETS__ void _k_blend_crop(const float *const in, const float *const den, float *const out, const int width, const int height, const int pw, const float strength)
dt_iop_rawdenoiseai_version_t
@ DT_RAWDENOISEAI_V1
int process_cl(struct dt_iop_module_t *self, const dt_dev_pixelpipe_t *pipe, const dt_dev_pixelpipe_iop_t *piece, cl_mem dev_in, cl_mem dev_out)
#define DT_RAWDENOISEAI_NUM_SIZES
static dt_nn_model_t * _get_model(dt_iop_rawdenoiseai_global_data_t *gd, dt_iop_rawdenoiseai_version_t ver, dt_iop_rawdenoiseai_size_t sz, dt_iop_rawdenoiseai_scale_t sc)
static int _fusion_levels(void)
static const float DT_NN_FUSION_DENS[3]
static __DT_CLONE_TARGETS__ void _k_fuse_step(const float *const fused_c, const float *const Mf, const float *const Df, const float *const Vf, const float *const Mc, const float *const Dc, float *const fused_f, float *const ups, const int sw, const int sh, const size_t p0, const int sc)
#define DT_RAWDENOISEAI_NUM_VERSIONS
static dt_iop_rawdenoiseai_size_t _default_size(void)
#define lw
Definition retouch.c:1121
float dt_aligned_pixel_simd_t __attribute__((vector_size(16), aligned(16)))
Apply one channel's tone curve to each of the three colour channels, or pass the channel through unto...
Definition simd.h:55
const float sigma
const float r
dt_image_t image_storage
Definition develop.h:225
char camera_makermodel[128]
Definition image.h:382
float exif_iso
Definition image.h:370
unsigned int channels
Definition format.h:83
GtkWidget * widget
Definition imageop_gui.h:47
int32_t hide_enable_button
Definition imageop.h:270
struct dt_iop_module_gui_t * gui
Definition imageop.h:346
struct dt_develop_t * dev
Definition imageop.h:311
dt_iop_global_data_t * global_data
Definition imageop.h:337
dt_iop_params_t * params
Definition imageop.h:333
dt_nn_model_t * models[1][3][2]
dt_gui_collapsible_section_t cs
dt_iop_rawdenoiseai_version_t version
dt_iop_rawdenoiseai_size_t size
dt_iop_rawdenoiseai_scale_t scale_variant
Region of interest passed through the pixelpipe.
Definition format.h:49
int width
Definition format.h:50
int height
Definition format.h:50
dt_aligned_pixel_t a
dt_aligned_pixel_t b
struct nn_region_t::@47 blocks[32]
#define __DT_CLONE_TARGETS__
typedef double((*spd)(unsigned long int wavelength, double TempK))
#define MAX(a, b)
Definition thinplate.c:29
gboolean dt_gui_widgets_suppressed(void)
#define DT_GUI_BOX_SPACING