terra_texture_rs/soft_light.rs
1//! Photoshop-style soft light blend, over 2-D (H, W) and 3-D (H, W, C)
2//! arrays.
3//!
4//! Matches `blend.py`'s `soft_light()` exactly (verified in
5//! `tests/test_blend.py`'s Rust-parity tests, when the extension is
6//! built). Per element, with base `a` and blend `b`:
7//!
8//! ```text
9//! b <= 0.5: 2ab + a²(1 - 2b)
10//! b > 0.5: 2a(1 - b) + sqrt(clamp(a, 0, 1)) · (2b - 1)
11//! ```
12//!
13//! and the result is clamped to \[0, 1\].
14//!
15//! # Which function to call
16//!
17//! Call [`soft_light_core`] (2-D) or [`soft_light_rgb_core`] (3-D). They
18//! pick the serial or parallel path using
19//! [`PARALLEL_THRESHOLD`].
20//!
21//! The `*_serial`/`*_parallel` variants are public only so
22//! `benches/blend_bench.rs` can time both paths at every array size to
23//! tune that threshold. Benches are separate crates and can only call
24//! `pub` items.
25
26use ndarray::{Array2, Array3, ArrayView2, ArrayView3, Zip};
27
28use crate::common::PARALLEL_THRESHOLD;
29
30/// Soft light for one element: base `a`, blend `b`, both `f32` nominally
31/// in \[0, 1\]. Returns an `f32` clamped to \[0, 1\]. See the module docs
32/// for the formula.
33#[inline]
34fn soft_light_pixel(a: f32, b: f32) -> f32 {
35 let v = if b <= 0.5 {
36 2.0 * a * b + a * a * (1.0 - 2.0 * b)
37 } else {
38 // matches Python's np.sqrt(np.clip(a, 0, 1)) -- clamped to BOTH
39 // 0 and 1 before the sqrt, not just floored at 0.
40 2.0 * a * (1.0 - b) + a.clamp(0.0, 1.0).sqrt() * (2.0 * b - 1.0)
41 };
42 v.clamp(0.0, 1.0)
43}
44
45// ---------------------------------------------------------------------------
46// 2-D (H, W)
47// ---------------------------------------------------------------------------
48
49/// Soft light blend of two 2-D arrays, single-threaded.
50///
51/// Same arguments, output and panics as [`soft_light_core`], but always
52/// runs serially regardless of size. Public for benchmarking only (see
53/// module docs).
54pub fn soft_light_serial(a: ArrayView2<f32>, b: ArrayView2<f32>, out: &mut Array2<f32>) {
55 Zip::from(out)
56 .and(&a)
57 .and(&b)
58 .for_each(|o, &a, &b| *o = soft_light_pixel(a, b));
59}
60
61/// Soft light blend of two 2-D arrays, parallelised with rayon.
62///
63/// Same arguments, output and panics as [`soft_light_core`], but always
64/// runs in parallel regardless of size. Public for benchmarking only
65/// (see module docs).
66pub fn soft_light_parallel(a: ArrayView2<f32>, b: ArrayView2<f32>, out: &mut Array2<f32>) {
67 Zip::from(out)
68 .and(&a)
69 .and(&b)
70 .par_for_each(|o, &a, &b| *o = soft_light_pixel(a, b));
71}
72
73/// Soft light blend of two 2-D arrays. The entry point for 2-D inputs.
74///
75/// # Arguments
76///
77/// * `a` - `ArrayView2<f32>`, shape (H, W): the base layer, values
78/// nominally in \[0, 1\]. Any memory layout.
79/// * `b` - `ArrayView2<f32>`, shape (H, W): the blend layer, values
80/// nominally in \[0, 1\]. Any memory layout.
81/// * `out` - `&mut Array2<f32>`, shape (H, W): overwritten with the
82/// result, every element in \[0, 1\]. Its previous contents are ignored.
83///
84/// Runs serially below [`PARALLEL_THRESHOLD`]
85/// elements and in parallel at or above it.
86///
87/// # Panics
88///
89/// If `a`, `b` and `out` do not all have the same shape.
90///
91/// # Example
92///
93/// ```
94/// use ndarray::{array, Array2};
95/// use terra_texture_rs::soft_light_core;
96///
97/// let base = array![[0.2_f32, 0.8], [0.5, 1.0]];
98/// let blend = array![[0.5_f32, 0.5], [0.0, 1.0]];
99/// let mut out = Array2::<f32>::zeros(base.raw_dim());
100/// soft_light_core(base.view(), blend.view(), &mut out);
101///
102/// // blend = 0.5 leaves the base unchanged
103/// assert!((out[[0, 0]] - 0.2).abs() < 1e-6);
104/// assert!(out.iter().all(|&v| (0.0..=1.0).contains(&v)));
105/// ```
106pub fn soft_light_core(a: ArrayView2<f32>, b: ArrayView2<f32>, out: &mut Array2<f32>) {
107 if a.len() >= PARALLEL_THRESHOLD {
108 soft_light_parallel(a, b, out);
109 } else {
110 soft_light_serial(a, b, out);
111 }
112}
113
114// ---------------------------------------------------------------------------
115// 3-D (H, W, C)
116// ---------------------------------------------------------------------------
117//
118// Same per-element math as the 2-D path (soft_light_pixel has no
119// cross-channel interaction). This exists as a separate kernel only to
120// avoid the per-channel np.ascontiguousarray() copy the Python-side loop
121// needed: each a[..., c] slice of a contiguous (H, W, C) array is itself
122// non-contiguous. Working on the whole buffer directly means one input
123// read and one output write per element, with no per-channel arrays.
124
125/// Soft light blend of two 3-D arrays, single-threaded.
126///
127/// Same arguments, output and panics as [`soft_light_rgb_core`], but
128/// always runs serially. Public for benchmarking only (see module docs).
129pub fn soft_light_rgb_serial(a: ArrayView3<f32>, b: ArrayView3<f32>, out: &mut Array3<f32>) {
130 Zip::from(out)
131 .and(&a)
132 .and(&b)
133 .for_each(|o, &a, &b| *o = soft_light_pixel(a, b));
134}
135
136/// Soft light blend of two 3-D arrays, parallelised with rayon.
137///
138/// Same arguments, output and panics as [`soft_light_rgb_core`], but
139/// always runs in parallel. Public for benchmarking only (see module
140/// docs).
141pub fn soft_light_rgb_parallel(a: ArrayView3<f32>, b: ArrayView3<f32>, out: &mut Array3<f32>) {
142 Zip::from(out)
143 .and(&a)
144 .and(&b)
145 .par_for_each(|o, &a, &b| *o = soft_light_pixel(a, b));
146}
147
148/// Soft light blend of two 3-D (H, W, C) arrays, e.g. RGB or RGBA
149/// imagery, in a single pass. The entry point for multi-channel inputs.
150///
151/// Every channel is blended independently with the same formula as
152/// [`soft_light_core`]; any channel count C works.
153///
154/// # Arguments
155///
156/// * `a` - `ArrayView3<f32>`, shape (H, W, C): the base layer, values
157/// nominally in \[0, 1\]. Any memory layout.
158/// * `b` - `ArrayView3<f32>`, shape (H, W, C): the blend layer, values
159/// nominally in \[0, 1\]. Any memory layout.
160/// * `out` - `&mut Array3<f32>`, shape (H, W, C): overwritten with the
161/// result, every element in \[0, 1\].
162///
163/// Runs serially below [`PARALLEL_THRESHOLD`]
164/// elements (H × W × C) and in parallel at or above it.
165///
166/// # Panics
167///
168/// If `a`, `b` and `out` do not all have the same shape.
169pub fn soft_light_rgb_core(a: ArrayView3<f32>, b: ArrayView3<f32>, out: &mut Array3<f32>) {
170 if a.len() >= PARALLEL_THRESHOLD {
171 soft_light_rgb_parallel(a, b, out);
172 } else {
173 soft_light_rgb_serial(a, b, out);
174 }
175}