fix: eliminate diagonal dither pattern from display driver

Root cause: Panel_ED2208's epd_quality mode applies _dither_row_rgb_pair
(diagonal bias pattern with dither=140) during _exec_transfer(). Since our
ImagePipeline already does proper Floyd-Steinberg dithering, the driver's
additional dithering was creating visible diagonal artifacts.

Fix: setEpdMode(epd_fastest) selects _dither_row_none (clean nearest-color
lookup, no spatial bias). On Panel_ED2208 this only affects the dither
function — NOT refresh quality or waveform.

Also: default dither_noise setting to OFF since error scatter was targeting
the wrong layer (it made gradient areas fuzzier without fixing the real
problem in the display driver).

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-08-03 19:52:22 -04:00
parent 0399573edf
commit 1944ce4f93
20 changed files with 1100 additions and 138 deletions

View File

@@ -1,4 +1,5 @@
#include "image_pipeline.h"
#include "blue_noise.h"
#include <M5GFX.h>
#include <esp_heap_caps.h>
#include <cstring>
@@ -15,6 +16,301 @@ static const uint8_t PALETTE_CALIBRATED[6][3] = {
{0xC1, 0xBB, 0x1E} // Yellow -> appears as olive/mustard
};
// ---------------------------------------------------------------------------
// Color science helpers (sRGB <-> LAB, luma, saturation)
// ---------------------------------------------------------------------------
static inline uint8_t clampByte(float v) {
if (v <= 0.0f) return 0;
if (v >= 255.0f) return 255;
return (uint8_t)(v + 0.5f);
}
static inline float clampF(float v, float lo, float hi) {
return (v < lo) ? lo : (v > hi) ? hi : v;
}
static inline float luma709(float r, float g, float b) {
return 0.2126f * r + 0.7152f * g + 0.0722f * b;
}
// Pre-computed sRGB-to-linear LUT (avoids per-pixel powf)
static const float* getSrgbToLinear() {
static float lut[256];
static bool initialized = false;
if (!initialized) {
for (int i = 0; i < 256; i++) {
float n = i / 255.0f;
lut[i] = (n > 0.04045f)
? powf((n + 0.055f) / 1.055f, 2.4f)
: n / 12.92f;
}
initialized = true;
}
return lut;
}
static inline float labForwardPivot(float v) {
return (v > 0.008856f) ? cbrtf(v) : 7.787f * v + 16.0f / 116.0f;
}
// Lightness-only conversion (for DRC histogram pass)
static float rgbToLabLightness(uint8_t r, uint8_t g, uint8_t b) {
const float* lut = getSrgbToLinear();
float y = lut[r] * 0.2126729f + lut[g] * 0.7151522f + lut[b] * 0.0721750f;
return 116.0f * labForwardPivot(y) - 16.0f;
}
// Full sRGB -> CIE LAB (D65 illuminant)
static void srgbToLab(uint8_t r, uint8_t g, uint8_t b,
float* L, float* a, float* bOut) {
const float* lut = getSrgbToLinear();
float rn = lut[r], gn = lut[g], bn = lut[b];
float x = rn * 0.4124564f + gn * 0.3575761f + bn * 0.1804375f;
float y = rn * 0.2126729f + gn * 0.7151522f + bn * 0.0721750f;
float z = rn * 0.0193339f + gn * 0.1191920f + bn * 0.9503041f;
float fx = labForwardPivot(x / 0.95047f);
float fy = labForwardPivot(y);
float fz = labForwardPivot(z / 1.08883f);
*L = 116.0f * fy - 16.0f;
*a = 500.0f * (fx - fy);
*bOut = 200.0f * (fy - fz);
}
// CIE LAB -> sRGB (D65 illuminant)
static void labToSrgb(float L, float a, float b,
uint8_t* rOut, uint8_t* gOut, uint8_t* bOut) {
float fy = (L + 16.0f) / 116.0f;
float fx = a / 500.0f + fy;
float fz = fy - b / 200.0f;
float x = (fx > 0.206897f) ? fx * fx * fx : (fx - 16.0f / 116.0f) / 7.787f;
float y = (fy > 0.206897f) ? fy * fy * fy : (fy - 16.0f / 116.0f) / 7.787f;
float z = (fz > 0.206897f) ? fz * fz * fz : (fz - 16.0f / 116.0f) / 7.787f;
x *= 0.95047f;
z *= 1.08883f;
float rl = x * 3.2404542f + y * -1.5371385f + z * -0.4985314f;
float gl = x * -0.9692660f + y * 1.8760108f + z * 0.0415560f;
float bl = x * 0.0556434f + y * -0.2040259f + z * 1.0572252f;
auto linearToSrgb = [](float v) -> float {
if (v <= 0.0f) return 0.0f;
return (v > 0.0031308f)
? 1.055f * powf(v, 1.0f / 2.4f) - 0.055f
: 12.92f * v;
};
*rOut = clampByte(linearToSrgb(rl) * 255.0f);
*gOut = clampByte(linearToSrgb(gl) * 255.0f);
*bOut = clampByte(linearToSrgb(bl) * 255.0f);
}
// HSV-style saturation (max-min)/max in 0..1 range
static inline float pixelSaturation(float r, float g, float b) {
float mx = fmaxf(r, fmaxf(g, b));
float mn = fminf(r, fminf(g, b));
return (mx > 0.0f) ? (mx - mn) / mx : 0.0f;
}
// ---------------------------------------------------------------------------
// Tone mapping (replaces enhanceContrast)
//
// Uses epdoptimize's "dynamic" preset:
// - Asymmetric power-curve S-curve built into a 256-entry LUT
// - HSL-space saturation adjustment (preserves hue fidelity)
// - Lookup-table contrast and exposure
// ---------------------------------------------------------------------------
static constexpr float SHADOW_TONE_RESPONSE = 1.5f;
static void toneMap(uint8_t* rgb, size_t pixelCount) {
static constexpr float EXPOSURE = 0.0f; // stops (2^0 = 1.0x)
static constexpr float SATURATION_ADJ = 0.3f; // -> 1.3x multiplier
static constexpr float CONTRAST_ADJ = 0.0f; // -> 1.0x multiplier
static constexpr float STRENGTH = 0.9f;
static constexpr float SHADOW_BOOST = 0.0f;
static constexpr float HIGHLIGHT_COMP = -1.5f;
static constexpr float MIDPOINT = 0.5f;
float exposureMul = powf(2.0f, EXPOSURE);
float satMul = fmaxf(0.0f, SATURATION_ADJ + 1.0f);
float contrastMul = (CONTRAST_ADJ < 0.0f)
? fmaxf(0.5f, 1.0f + CONTRAST_ADJ * 0.5f)
: CONTRAST_ADJ + 1.0f;
// Build S-curve LUT (asymmetric power curve per epdoptimize)
float mid = clampF(MIDPOINT, 0.01f, 0.99f);
float shadowExp = clampF(1.0f - STRENGTH * SHADOW_BOOST * SHADOW_TONE_RESPONSE, 0.15f, 3.0f);
float highlightExp = clampF(1.0f - STRENGTH * HIGHLIGHT_COMP, 0.15f, 3.0f);
uint8_t exposureLut[256];
uint8_t toneLut[256];
for (int v = 0; v < 256; v++) {
exposureLut[v] = clampByte((float)v * exposureMul);
float tv = clampF(((float)v - 128.0f) * contrastMul + 128.0f, 0.0f, 255.0f);
if (STRENGTH != 0.0f) {
float n = tv / 255.0f;
float curved;
if (n <= mid) {
curved = powf(n / mid, shadowExp) * mid;
} else {
curved = mid + powf((n - mid) / (1.0f - mid), highlightExp) * (1.0f - mid);
}
tv = clampF(curved * 255.0f, 0.0f, 255.0f);
}
toneLut[v] = (uint8_t)tv;
}
bool needsSaturation = (satMul != 1.0f);
for (size_t i = 0; i < pixelCount; i++) {
size_t idx = i * 3;
if (!needsSaturation) {
rgb[idx] = toneLut[exposureLut[rgb[idx]]];
rgb[idx + 1] = toneLut[exposureLut[rgb[idx + 1]]];
rgb[idx + 2] = toneLut[exposureLut[rgb[idx + 2]]];
continue;
}
// HSL-space saturation (preserves hue, matches epdoptimize)
float r0 = exposureLut[rgb[idx]] / 255.0f;
float g0 = exposureLut[rgb[idx + 1]] / 255.0f;
float b0 = exposureLut[rgb[idx + 2]] / 255.0f;
float maxC = fmaxf(r0, fmaxf(g0, b0));
float minC = fminf(r0, fminf(g0, b0));
float light = (maxC + minC) * 0.5f;
float r = r0, g = g0, b = b0;
if (maxC != minC) {
float delta = maxC - minC;
float sat = (light > 0.5f)
? delta / (2.0f - maxC - minC)
: delta / fmaxf(maxC + minC, 1e-6f);
float hue;
if (maxC == r0) {
hue = (g0 - b0) / delta;
if (g0 < b0) hue += 6.0f;
hue /= 6.0f;
} else if (maxC == g0) {
hue = ((b0 - r0) / delta + 2.0f) / 6.0f;
} else {
hue = ((r0 - g0) / delta + 4.0f) / 6.0f;
}
float newSat = clampF(sat * satMul, 0.0f, 1.0f);
float c = (1.0f - fabsf(2.0f * light - 1.0f)) * newSat;
float x = c * (1.0f - fabsf(fmodf(hue * 6.0f, 2.0f) - 1.0f));
float m = light - c * 0.5f;
int sector = (int)(hue * 6.0f);
if (sector >= 6) sector = 5;
switch (sector) {
case 0: r = c + m; g = x + m; b = m; break;
case 1: r = x + m; g = c + m; b = m; break;
case 2: r = m; g = c + m; b = x + m; break;
case 3: r = m; g = x + m; b = c + m; break;
case 4: r = x + m; g = m; b = c + m; break;
case 5: r = c + m; g = m; b = x + m; break;
}
}
rgb[idx] = toneLut[clampByte(r * 255.0f)];
rgb[idx + 1] = toneLut[clampByte(g * 255.0f)];
rgb[idx + 2] = toneLut[clampByte(b * 255.0f)];
}
Serial.printf("[pipeline] toneMap: exposure=%.1f sat=%.1fx strength=%.1f\n",
EXPOSURE, satMul, STRENGTH);
}
// ---------------------------------------------------------------------------
// Dynamic range compression (fast luma path with chroma protection)
//
// Uses epdoptimize's "balanced" preset approach:
// - Histogram percentile scan for source range detection
// - Remap into palette luminance range
// - smoothstep chroma protection prevents saturated color washout
// ---------------------------------------------------------------------------
static void compressDynamicRange(uint8_t* rgb, size_t pixelCount) {
static constexpr float STRENGTH = 1.0f;
static constexpr float LOW_PERCENTILE = 0.01f;
static constexpr float HIGH_PERCENTILE = 0.99f;
float blackY = luma709(PALETTE_CALIBRATED[0][0],
PALETTE_CALIBRATED[0][1],
PALETTE_CALIBRATED[0][2]);
float whiteY = luma709(PALETTE_CALIBRATED[1][0],
PALETTE_CALIBRATED[1][1],
PALETTE_CALIBRATED[1][2]);
float targetRange = whiteY - blackY;
if (targetRange <= 0.0f) return;
// Build luma histogram for percentile detection
uint32_t histogram[256] = {0};
for (size_t i = 0; i < pixelCount; i++) {
size_t idx = i * 3;
histogram[clampByte(luma709(rgb[idx], rgb[idx + 1], rgb[idx + 2]))]++;
}
// Find percentile endpoints
auto findPercentile = [&](float p) -> float {
uint32_t target = (uint32_t)((pixelCount - 1) * p);
uint32_t seen = 0;
for (int i = 0; i < 256; i++) {
seen += histogram[i];
if (seen > target) return (float)i;
}
return 255.0f;
};
float sourceBlackY = findPercentile(LOW_PERCENTILE);
float sourceWhiteY = findPercentile(HIGH_PERCENTILE);
float sourceRange = sourceWhiteY - sourceBlackY;
if (sourceRange <= 0.0001f) return;
for (size_t i = 0; i < pixelCount; i++) {
size_t idx = i * 3;
float r = rgb[idx], g = rgb[idx + 1], b = rgb[idx + 2];
float y = luma709(r, g, b);
float normalizedY = clampF((y - sourceBlackY) / sourceRange, 0.0f, 1.0f);
float targetY = blackY + normalizedY * targetRange;
// Chroma protection: smoothstep(0.18, 0.68, saturation) * 0.85
float sat = pixelSaturation(r, g, b);
float chromaProtection = 0.0f;
if (sat > 0.18f) {
float t = clampF((sat - 0.18f) / (0.68f - 0.18f), 0.0f, 1.0f);
chromaProtection = t * t * (3.0f - 2.0f * t) * 0.85f;
}
float effectiveStrength = STRENGTH * (1.0f - chromaProtection);
float nextY = y + (targetY - y) * effectiveStrength;
float ratio = (y > 0.0f) ? nextY / y : 0.0f;
float maxChannel = fmaxf(r, fmaxf(g, b));
if (maxChannel > 0.0f) ratio = fminf(ratio, 255.0f / maxChannel);
rgb[idx] = clampByte(r * ratio);
rgb[idx + 1] = clampByte(g * ratio);
rgb[idx + 2] = clampByte(b * ratio);
}
Serial.printf("[pipeline] DRC: src=[%.0f..%.0f] -> dst=[%.0f..%.0f]\n",
sourceBlackY, sourceWhiteY, blackY, whiteY);
}
// JPEGDEC draw callback: receives decoded MCU blocks and writes RGB888 to buffer
static int jpegDrawCallback(JPEGDRAW* pDraw) {
auto* ctx = static_cast<DecodeContext*>(pDraw->pUser);
@@ -43,33 +339,6 @@ static int jpegDrawCallback(JPEGDRAW* pDraw) {
return 1;
}
// Contrast/saturation enhancement applied before dithering
static void enhanceContrast(uint8_t* rgb, size_t pixelCount) {
static constexpr float CONTRAST = 1.25f;
static constexpr float SATURATION = 1.15f;
static constexpr float MID = 128.0f;
for (size_t i = 0; i < pixelCount; i++) {
size_t idx = i * 3;
float r = rgb[idx];
float g = rgb[idx + 1];
float b = rgb[idx + 2];
r = (r - MID) * CONTRAST + MID;
g = (g - MID) * CONTRAST + MID;
b = (b - MID) * CONTRAST + MID;
float lum = 0.299f * r + 0.587f * g + 0.114f * b;
r = lum + (r - lum) * SATURATION;
g = lum + (g - lum) * SATURATION;
b = lum + (b - lum) * SATURATION;
rgb[idx] = (uint8_t)fminf(fmaxf(r, 0.0f), 255.0f);
rgb[idx + 1] = (uint8_t)fminf(fmaxf(g, 0.0f), 255.0f);
rgb[idx + 2] = (uint8_t)fminf(fmaxf(b, 0.0f), 255.0f);
}
}
// Average a single edge of the fitted image (4 rows or columns deep)
static constexpr int EDGE_DEPTH = 4;
@@ -670,10 +939,19 @@ ProcessedImage ImagePipeline::process(uint8_t* jpegData, size_t jpegSize) {
Serial.printf("[pipeline] Got %dx%d fitted RGB\n", outW, outH);
// Enhance contrast/saturation for e-ink readability
enhanceContrast(rgb, (size_t)outW * outH);
size_t totalPixels = (size_t)outW * outH;
switch (_mode) {
case PipelineMode::DYNAMIC:
toneMap(rgb, totalPixels);
break;
case PipelineMode::BALANCED:
compressDynamicRange(rgb, totalPixels);
break;
case PipelineMode::NONE:
break;
// No default — compiler warns on unhandled PipelineMode via -Wswitch
}
// Row-by-row Floyd-Steinberg dithering to 6-color palette
uint8_t* dithered = ditherRowByRow(rgb, outW, outH);
free(rgb);
@@ -685,7 +963,11 @@ ProcessedImage ImagePipeline::process(uint8_t* jpegData, size_t jpegSize) {
result.width = outW;
result.height = outH;
result.valid = true;
Serial.printf("[pipeline] Processing complete (%dx%d)\n", outW, outH);
const char* modeName = (_mode == PipelineMode::DYNAMIC) ? "dynamic"
: (_mode == PipelineMode::BALANCED) ? "balanced"
: "none";
Serial.printf("[pipeline] Processing complete (%dx%d, mode=%s)\n",
outW, outH, modeName);
return result;
}
@@ -727,7 +1009,18 @@ ProcessedImage ImagePipeline::processPortraitPair(uint8_t* jpeg1Data, size_t jpe
free(rgb2);
}
enhanceContrast(combined, (size_t)DISPLAY_WIDTH * DISPLAY_HEIGHT);
size_t combinedPixels = (size_t)DISPLAY_WIDTH * DISPLAY_HEIGHT;
switch (_mode) {
case PipelineMode::DYNAMIC:
toneMap(combined, combinedPixels);
break;
case PipelineMode::BALANCED:
compressDynamicRange(combined, combinedPixels);
break;
case PipelineMode::NONE:
break;
// No default — compiler warns on unhandled PipelineMode via -Wswitch
}
uint8_t* dithered = ditherRowByRow(combined, DISPLAY_WIDTH, DISPLAY_HEIGHT);
free(combined);
@@ -792,8 +1085,14 @@ uint8_t* ImagePipeline::ditherRowByRow(uint8_t* rgb, uint16_t width, uint16_t he
memset(errNext, 0, rowBytes);
}
for (uint16_t x = 0; x < width; x++) {
size_t errIdx = x * 3;
// Serpentine: alternate scan direction each row
bool forward = (y % 2 == 0);
int xStart = forward ? 0 : (int)width - 1;
int xEnd = forward ? (int)width : -1;
int xStep = forward ? 1 : -1;
for (int x = xStart; x != xEnd; x += xStep) {
size_t errIdx = (size_t)x * 3;
int r = constrain(errCurrent[errIdx], 0, 255);
int g = constrain(errCurrent[errIdx + 1], 0, 255);
@@ -806,29 +1105,77 @@ uint8_t* ImagePipeline::ditherRowByRow(uint8_t* rgb, uint16_t width, uint16_t he
int errG = g - PALETTE_CALIBRATED[nearest][1];
int errB = b - PALETTE_CALIBRATED[nearest][2];
if (x + 1 < width) {
size_t ni = (x + 1) * 3;
// Floyd-Steinberg: mirror dx offsets on reverse rows
int xRight = forward ? x + 1 : x - 1;
int xLeft = forward ? x - 1 : x + 1;
// 7/16 to next pixel in scan direction
if (xRight >= 0 && xRight < (int)width) {
size_t ni = (size_t)xRight * 3;
errCurrent[ni] += errR * 7 / 16;
errCurrent[ni + 1] += errG * 7 / 16;
errCurrent[ni + 2] += errB * 7 / 16;
}
if (y + 1 < height && x > 0) {
size_t ni = (x - 1) * 3;
errNext[ni] += errR * 3 / 16;
errNext[ni + 1] += errG * 3 / 16;
errNext[ni + 2] += errB * 3 / 16;
}
if (y + 1 < height) {
size_t ni = x * 3;
errNext[ni] += errR * 5 / 16;
errNext[ni + 1] += errG * 5 / 16;
errNext[ni + 2] += errB * 5 / 16;
}
if (y + 1 < height && x + 1 < width) {
size_t ni = (x + 1) * 3;
errNext[ni] += errR * 1 / 16;
errNext[ni + 1] += errG * 1 / 16;
errNext[ni + 2] += errB * 1 / 16;
if (_blueNoise) {
// Randomized error scatter: the diagonal pattern is caused
// by the 5/16 "below" weight always landing error on the
// same column, creating vertical correlation that manifests
// as diagonal lines. We break this by randomly offsetting
// the entire below-row error target by -2..+2 pixels.
// Energy is perfectly conserved (same 9/16 total, same
// 3/5/1 ratio — just shifted horizontally).
uint32_t h = (uint32_t)x * 2654435761u
^ (uint32_t)y * 2246822519u;
int offset = (int)(h % 5u) - 2; // -2, -1, 0, +1, or +2
int xBL = xLeft + offset;
int xB = x + offset;
int xBR = xRight + offset;
// 3/16 to below-left (shifted)
if (xBL >= 0 && xBL < (int)width) {
size_t ni = (size_t)xBL * 3;
errNext[ni] += errR * 3 / 16;
errNext[ni + 1] += errG * 3 / 16;
errNext[ni + 2] += errB * 3 / 16;
}
// 5/16 to below (shifted)
if (xB >= 0 && xB < (int)width) {
size_t ni = (size_t)xB * 3;
errNext[ni] += errR * 5 / 16;
errNext[ni + 1] += errG * 5 / 16;
errNext[ni + 2] += errB * 5 / 16;
}
// 1/16 to below-right (shifted)
if (xBR >= 0 && xBR < (int)width) {
size_t ni = (size_t)xBR * 3;
errNext[ni] += errR * 1 / 16;
errNext[ni + 1] += errG * 1 / 16;
errNext[ni + 2] += errB * 1 / 16;
}
} else {
// Standard fixed Floyd-Steinberg weights
if (xLeft >= 0 && xLeft < (int)width) {
size_t ni = (size_t)xLeft * 3;
errNext[ni] += errR * 3 / 16;
errNext[ni + 1] += errG * 3 / 16;
errNext[ni + 2] += errB * 3 / 16;
}
{
size_t ni = (size_t)x * 3;
errNext[ni] += errR * 5 / 16;
errNext[ni + 1] += errG * 5 / 16;
errNext[ni + 2] += errB * 5 / 16;
}
if (xRight >= 0 && xRight < (int)width) {
size_t ni = (size_t)xRight * 3;
errNext[ni] += errR * 1 / 16;
errNext[ni + 1] += errG * 1 / 16;
errNext[ni + 2] += errB * 1 / 16;
}
}
}
}
@@ -844,12 +1191,13 @@ uint8_t* ImagePipeline::ditherRowByRow(uint8_t* rgb, uint16_t width, uint16_t he
uint8_t ImagePipeline::findNearest(int r, int g, int b) {
uint8_t best = 0;
int bestDist = INT32_MAX;
int32_t bestDist = INT32_MAX;
for (int i = 0; i < DISPLAY_COLORS; i++) {
int dr = r - PALETTE_CALIBRATED[i][0];
int dg = g - PALETTE_CALIBRATED[i][1];
int db = b - PALETTE_CALIBRATED[i][2];
int dist = dr * dr + dg * dg + db * db;
// Rec. 709 luminance-weighted distance (integer weights x10000)
int32_t dist = 2126 * dr * dr + 7152 * dg * dg + 722 * db * db;
if (dist < bestDist) {
bestDist = dist;
best = static_cast<uint8_t>(i);