[v5] libcamera: software_isp: Add support for raw monochrome formats
diff mbox series

Message ID 10d396ff-2270-46ef-a641-967df76917be@laposte.net
State New
Headers show
Series
  • [v5] libcamera: software_isp: Add support for raw monochrome formats
Related show

Commit Message

Alain Cousinie Oct. 11, 2026, 2:31 p.m. UTC
Hi Barnabás, Hi Milan,

Thank you very much for your feedback, which allows me to improve and also makes you co-authors of this work.
I have integrated all your suggestions into a major unified v5 refactoring.

Here are the three major improvements included in this v5:

1. Extended Format Support: Added comprehensive pipelines for the native 16-bit 
monochrome format (R16), covering CPU, EGL, and stats loops.

2. CPU-Driven GPU Optimizations: Streamlined the GPU rendering pipelines by 
pushing format parameters directly from the CPU.
This eliminates runtime branchings and duplicate calculations inside the shaders.

3. Fixed Bit-Alignment Math: Corrected the internal bit-shift logic between left
and right alignments in swstats_cpu, ensuring that the cumulative sums for all
formats (R8 to R16, including CSI2 packeds) are properly scaled to the 8-bit range
expected by the AE algorithm.

4. Branchless CPU Routines: Completely refactored the CPU debayering implementation
by splitting it into dedicated functions.

Note: GPU processing handles full bit-depth (including LSBs) and the R16 format 
with a view to supporting future R16 HDR outputs. (Please note that I am unable 
to hardware-test the packed format path.)

Looking forward to your feedback on this streamlined v5!

Best regards,
Alain

Signed-off-by: Alain Cousinié <alain.cousinie@laposte.net>

---

 .../internal/software_isp/swstats_cpu.h       |  13 ++
 src/libcamera/shaders/meson.build             |   3 +
 src/libcamera/shaders/mono_1x_packed.frag     |  72 +++++++
 src/libcamera/shaders/mono_unpacked.frag      |  47 +++++
 src/libcamera/shaders/mono_unpacked.vert      |  18 ++
 src/libcamera/software_isp/debayer_cpu.cpp    | 134 ++++++++++++++
 src/libcamera/software_isp/debayer_cpu.h      |  18 ++
 src/libcamera/software_isp/debayer_egl.cpp    |  89 +++++++++
 src/libcamera/software_isp/swstats_cpu.cpp    | 175 ++++++++++++++++++
 9 files changed, 569 insertions(+)
 create mode 100644 src/libcamera/shaders/mono_1x_packed.frag
 create mode 100644 src/libcamera/shaders/mono_unpacked.frag
 create mode 100644 src/libcamera/shaders/mono_unpacked.vert

Patch
diff mbox series

diff --git a/include/libcamera/internal/software_isp/swstats_cpu.h b/include/libcamera/internal/software_isp/swstats_cpu.h
index 551870921..f702cf126 100644
--- a/include/libcamera/internal/software_isp/swstats_cpu.h
+++ b/include/libcamera/internal/software_isp/swstats_cpu.h
@@ -101,8 +101,21 @@  private:
 	/* Bayer 12 bpp packed */
 	void statsBGGR12PLine0(const uint8_t *src[], SwIspStats &stats);
 	void statsGBRG12PLine0(const uint8_t *src[], SwIspStats &stats);
+	/* Mono 8 bpp unpacked */
+	void statsMono8Line0(const uint8_t *src[], SwIspStats &stats);
+	/* Mono 10 bpp unpacked */
+	void statsMono10Line0(const uint8_t *src[], SwIspStats &stats);
+	/* Mono 10 bpp packed */
+	void statsMono10PLine0(const uint8_t *src[], SwIspStats &stats);
+	/* Mono 12 bpp unpacked */
+	void statsMono12Line0(const uint8_t *src[], SwIspStats &stats);
+	/* Mono 12 bpp packed */
+	void statsMono12PLine0(const uint8_t *src[], SwIspStats &stats);
+	/* Mono 16 bpp unpacked */
+	void statsMono16Line0(const uint8_t *src[], SwIspStats &stats);
 
 	void processBayerFrame2(MappedFrameBuffer &in);
+	void processMonoFrame2(MappedFrameBuffer &in);
 
 	processFrameFn processFrame_;
 
diff --git a/src/libcamera/shaders/meson.build b/src/libcamera/shaders/meson.build
index c409ff9b0..ed19768e0 100644
--- a/src/libcamera/shaders/meson.build
+++ b/src/libcamera/shaders/meson.build
@@ -7,6 +7,9 @@  shader_files = files([
     'bayer_unpacked.frag',
     'bayer_unpacked.vert',
     'identity.vert',
+    'mono_1x_packed.frag',
+    'mono_unpacked.frag',
+    'mono_unpacked.vert',
 ])
 
 # Generate header from shaders
diff --git a/src/libcamera/shaders/mono_1x_packed.frag b/src/libcamera/shaders/mono_1x_packed.frag
new file mode 100644
index 000000000..76e676b50
--- /dev/null
+++ b/src/libcamera/shaders/mono_1x_packed.frag
@@ -0,0 +1,72 @@ 
+/* SPDX-License-Identifier: BSD-2-Clause */
+/*
+ * Copyright (C) 2026, Alain Cousinié
+ * Unified Fragment shader for MIPI CSI-2 packed monochrome formats (R10P / R12P)
+ * Full bit-depth extraction (with LSB) optimized for HDR & Computer Vision
+ */
+
+#ifdef GL_ES
+#ifdef GL_FRAGMENT_PRECISION_HIGH
+precision highp float;
+#else
+precision mediump float;
+#endif
+#endif
+
+varying vec2            textureOut;
+
+uniform vec2            tex_size;
+uniform vec2            tex_step;
+
+uniform sampler2D       tex_y;
+uniform float           gamma;
+uniform float           contrastExp;
+uniform vec3            blacklevel;
+
+void main(void)
+{
+	vec2 pixelCoords = floor(textureOut * tex_size);
+	float px = pixelCoords.x;
+	float py = pixelCoords.y;
+
+	float blockStart  = floor(px / BLOCK_PIXELS) * BLOCK_BYTES;
+	float idx         = px - (blockStart * (BLOCK_PIXELS / BLOCK_BYTES));
+
+	float xMSB        = (blockStart + idx + 0.5) * tex_step.x;
+	float xLSB        = (blockStart + BLOCK_PIXELS + 0.5) * tex_step.x;
+	float normY       = (py + 0.5) * tex_step.y;
+
+	float byteHigh = floor(texture2D(tex_y, vec2(xMSB, normY)).r * 255.0 + 0.5);
+	float byteLow  = floor(texture2D(tex_y, vec2(xLSB, normY)).r * 255.0 + 0.5);
+
+	float lsbMask = 0.0;
+	float msbMultiplier = 0.0;
+	float shiftedLsb = 0.0;
+
+	vec4 pixelPos = step(vec4(-0.5, 0.5, 1.5, 2.5), vec4(idx)) * step(vec4(idx), vec4(0.5, 1.5, 2.5, 3.5));
+
+#if defined(RAW10P)
+	lsbMask = 4.0;
+	msbMultiplier = 4.0;
+	float divisor = dot(pixelPos, vec4(1.0, 4.0, 16.0, 64.0));
+	shiftedLsb = floor(byteLow / divisor);
+#elif defined(RAW12P)
+	lsbMask = 16.0;
+	msbMultiplier = 16.0;
+	float divisor = dot(pixelPos, vec4(1.0, 16.0, 0.0, 0.0));
+	shiftedLsb = floor(byteLow / divisor);
+#endif
+
+	float lsbBits = shiftedLsb - (floor(shiftedLsb / lsbMask) * lsbMask);
+	float rawValue = (byteHigh * msbMultiplier) + lsbBits;
+
+	float normalizedIntensity = rawValue * INV_MAX_VAL;
+	normalizedIntensity = clamp(normalizedIntensity - blacklevel.g, 0.0, 1.0);
+
+	float centerDist = 1.0 - 2.0 * abs(normalizedIntensity - 0.5);
+	float singlePow  = 0.5 * pow(centerDist, contrastExp);
+	normalizedIntensity = mix(singlePow, 1.0 - singlePow, step(0.5, normalizedIntensity));
+
+	vec3 rgb = pow(vec3(normalizedIntensity), vec3(gamma));
+	gl_FragColor = vec4(rgb, 1.0);
+}
diff --git a/src/libcamera/shaders/mono_unpacked.frag b/src/libcamera/shaders/mono_unpacked.frag
new file mode 100644
index 000000000..a320b2936
--- /dev/null
+++ b/src/libcamera/shaders/mono_unpacked.frag
@@ -0,0 +1,47 @@ 
+/* SPDX-License-Identifier: BSD-2-Clause */
+/*
+ * Copyright (C) 2026, Alain Cousinié
+ * Unified Fragment shader for unpacked monochrome formats (R8, R10, R12, R16)
+ * Optimized for libcamera EGL pipeline (GLSL ES 1.00)
+ */
+
+#ifdef GL_ES
+#ifdef GL_FRAGMENT_PRECISION_HIGH
+precision highp float;
+#else
+precision mediump float;
+#endif
+#endif
+
+varying vec4            center;
+
+uniform sampler2D       tex_y;
+uniform float           gamma;
+uniform float           contrastExp;
+uniform vec3            blacklevel;
+
+void main(void)
+{
+	float rawValue = 0.0;
+
+#ifdef RAW8
+	rawValue = floor(texture2D(tex_y, center.xy).r * 255.0 + 0.5);
+#else
+	vec4 texel = texture2D(tex_y, center.xy);
+
+	float byteLow  = floor(texel.r * 255.0 + 0.5);
+	float byteHigh = floor(texel.g * 255.0 + 0.5);
+
+	rawValue = (byteHigh * 256.0) + byteLow;
+#endif
+
+	float normalizedIntensity = rawValue * INV_MAX_VAL;
+	normalizedIntensity = clamp(normalizedIntensity - blacklevel.g, 0.0, 1.0);
+
+	float centerDist = 1.0 - 2.0 * abs(normalizedIntensity - 0.5);
+	float singlePow  = 0.5 * pow(centerDist, contrastExp);
+	normalizedIntensity = mix(singlePow, 1.0 - singlePow, step(0.5, normalizedIntensity));
+
+	vec3 rgb = pow(vec3(normalizedIntensity), vec3(gamma));
+	gl_FragColor = vec4(rgb, 1.0);
+}
diff --git a/src/libcamera/shaders/mono_unpacked.vert b/src/libcamera/shaders/mono_unpacked.vert
new file mode 100644
index 000000000..690cd4fb2
--- /dev/null
+++ b/src/libcamera/shaders/mono_unpacked.vert
@@ -0,0 +1,18 @@ 
+/* SPDX-License-Identifier: BSD-2-Clause */
+/*
+ * Copyright (C) 2026, Alain Cousinié
+ * Lightweight vertex shader for unpacked monochrome formats
+ */
+
+attribute vec4 vertexIn;
+attribute vec2 textureIn;
+
+uniform mat4 proj_matrix;
+uniform float stride_factor;
+
+varying vec4            center; /* Restored to original vec4 */
+
+void main(void) {
+    center.xy = vec2(textureIn.x * stride_factor, textureIn.y);
+    gl_Position = proj_matrix * vertexIn;
+}
diff --git a/src/libcamera/software_isp/debayer_cpu.cpp b/src/libcamera/software_isp/debayer_cpu.cpp
index ce8b3c647..e8d9873f5 100644
--- a/src/libcamera/software_isp/debayer_cpu.cpp
+++ b/src/libcamera/software_isp/debayer_cpu.cpp
@@ -423,6 +423,89 @@  void DebayerCpu::debayer12P_RGRG_BGR888(uint8_t *dst, const uint8_t *src[])
 		x++;
 	}
 }
+template<bool addAlphaByte, bool ccmEnabled>
+void DebayerCpu::debayer8_MONO_BGR888(uint8_t *dst, const uint8_t *src[])
+{
+	const uint8_t *cur = src[1];
+
+	for (int x = 0; x < (int)window_.width;) {
+		uint8_t pixel_val = *cur++;
+		STORE_PIXEL(pixel_val, pixel_val, pixel_val)
+	}
+}
+
+template<bool addAlphaByte, bool ccmEnabled>
+void DebayerCpu::debayer10_MONO_BGR888(uint8_t *dst, const uint8_t *src[])
+{
+	const uint16_t *cur = (const uint16_t *)src[1];
+
+	for (int x = 0; x < (int)window_.width;) {
+		uint16_t native_val = *cur++;
+		uint8_t pixel_val = static_cast<uint8_t>(native_val >> 2);
+
+		STORE_PIXEL(pixel_val, pixel_val, pixel_val)
+	}
+}
+
+template<bool addAlphaByte, bool ccmEnabled>
+void DebayerCpu::debayer12_MONO_BGR888(uint8_t *dst, const uint8_t *src[])
+{
+	const uint16_t *cur = (const uint16_t *)src[1];
+
+	for (int x = 0; x < (int)window_.width;) {
+		uint16_t native_val = *cur++;
+		uint8_t pixel_val = static_cast<uint8_t>(native_val >> 4);
+
+		STORE_PIXEL(pixel_val, pixel_val, pixel_val)
+	}
+}
+
+template<bool addAlphaByte, bool ccmEnabled>
+void DebayerCpu::debayer10P_MONO_BGR888(uint8_t *dst, const uint8_t *src[])
+{
+	const uint8_t *cur = src[1];
+
+	for (int x = 0; x < (int)window_.width;) {
+		uint8_t gray0 = *cur++;
+		uint8_t gray1 = *cur++;
+		uint8_t gray2 = *cur++;
+		uint8_t gray3 = *cur++;
+		cur++;
+
+		STORE_PIXEL(gray0, gray0, gray0)
+		STORE_PIXEL(gray1, gray1, gray1)
+		STORE_PIXEL(gray2, gray2, gray2)
+		STORE_PIXEL(gray3, gray3, gray3)
+	}
+}
+
+template<bool addAlphaByte, bool ccmEnabled>
+void DebayerCpu::debayer12P_MONO_BGR888(uint8_t *dst, const uint8_t *src[])
+{
+	const uint8_t *cur = src[1];
+
+	for (int x = 0; x < (int)window_.width;) {
+		uint8_t gray0 = *cur++;
+		uint8_t gray1 = *cur++;
+		cur++;
+
+		STORE_PIXEL(gray0, gray0, gray0)
+		STORE_PIXEL(gray1, gray1, gray1)
+	}
+}
+
+template<bool addAlphaByte, bool ccmEnabled>
+void DebayerCpu::debayer16_MONO_BGR888(uint8_t *dst, const uint8_t *src[])
+{
+	const uint16_t *cur = (const uint16_t *)src[1];
+
+	for (int x = 0; x < (int)window_.width;) {
+		uint16_t native_val = *cur++;
+		uint8_t pixel_val = static_cast<uint8_t>(native_val >> 8);
+
+		STORE_PIXEL(pixel_val, pixel_val, pixel_val)
+	}
+}
 
 /*
  * Setup the Debayer object according to the passed in parameters.
@@ -439,6 +522,20 @@  int DebayerCpu::getInputConfig(PixelFormat inputFormat, DebayerInputConfig &conf
 						   formats::BGR888,
 						   formats::XBGR8888,
 						   formats::ABGR8888 };
+	if (bayerFormat.order == BayerFormat::Order::MONO) {
+		if (bayerFormat.packing == BayerFormat::Packing::None) {
+			config.bpp = (bayerFormat.bitDepth + 7) & ~7;
+			config.patternSize.width = 2;
+			config.patternSize.height = 2;
+		} else if (bayerFormat.packing == BayerFormat::Packing::CSI2) {
+			config.bpp = bayerFormat.bitDepth;
+			config.patternSize.width = (bayerFormat.bitDepth == 10) ? 4 : 2;
+			config.patternSize.height = 2;
+		}
+
+		config.outputFormats = outputFormats;
+		return 0;
+	}
 
 	if ((bayerFormat.bitDepth == 8 || bayerFormat.bitDepth == 10 || bayerFormat.bitDepth == 12) &&
 	    bayerFormat.packing == BayerFormat::Packing::None &&
@@ -525,6 +622,43 @@  int DebayerCpu::setDebayerFunctions(PixelFormat inputFormat,
 		return -EINVAL;
 	};
 
+	if (bayerFormat.order == BayerFormat::Order::MONO) {
+		/* Disable CCM for monochrome to prevent image coloration */
+		ccmEnabled = false;
+
+		if (outputFormat == formats::XRGB8888 || outputFormat == formats::ARGB8888 ||
+		    outputFormat == formats::XBGR8888 || outputFormat == formats::ABGR8888) {
+			addAlphaByte = true;
+		}
+		if (bayerFormat.packing == BayerFormat::Packing::None) {
+			switch (bayerFormat.bitDepth) {
+			case 8:
+				SET_DEBAYER_METHODS(debayer8_MONO_BGR888, debayer8_MONO_BGR888)
+				return 0;
+			case 10:
+				SET_DEBAYER_METHODS(debayer10_MONO_BGR888, debayer10_MONO_BGR888)
+				return 0;
+			case 12:
+				SET_DEBAYER_METHODS(debayer12_MONO_BGR888, debayer12_MONO_BGR888)
+				return 0;
+			case 16:
+				SET_DEBAYER_METHODS(debayer16_MONO_BGR888, debayer16_MONO_BGR888)
+				return 0;
+			}
+		}
+		if (bayerFormat.packing == BayerFormat::Packing::CSI2) {
+			switch (bayerFormat.bitDepth) {
+			case 10:
+				SET_DEBAYER_METHODS(debayer10P_MONO_BGR888, debayer10P_MONO_BGR888)
+				return 0;
+			case 12:
+				SET_DEBAYER_METHODS(debayer12P_MONO_BGR888, debayer12P_MONO_BGR888)
+				return 0;
+			}
+		}
+		return invalidFmt();
+	}
+
 	switch (outputFormat) {
 	case formats::XRGB8888:
 	case formats::ARGB8888:
diff --git a/src/libcamera/software_isp/debayer_cpu.h b/src/libcamera/software_isp/debayer_cpu.h
index 2c88c9e1a..7ffbc0e53 100644
--- a/src/libcamera/software_isp/debayer_cpu.h
+++ b/src/libcamera/software_isp/debayer_cpu.h
@@ -119,6 +119,24 @@  private:
 	void debayer12P_GBGB_BGR888(uint8_t *dst, const uint8_t *src[]);
 	template<bool addAlphaByte, bool ccmEnabled>
 	void debayer12P_RGRG_BGR888(uint8_t *dst, const uint8_t *src[]);
+	/* 8-bit raw monochrome format */
+	template<bool addAlphaByte, bool ccmEnabled>
+	void debayer8_MONO_BGR888(uint8_t *dst, const uint8_t *src[]);
+	/* unpacked 10-bit raw monochrome format */
+	template<bool addAlphaByte, bool ccmEnabled>
+	void debayer10_MONO_BGR888(uint8_t *dst, const uint8_t *src[]);
+	/* unpacked 12-bit raw monochrome format */
+	template<bool addAlphaByte, bool ccmEnabled>
+	void debayer12_MONO_BGR888(uint8_t *dst, const uint8_t *src[]);
+	/* CSI-2 packed 10-bit raw monochrome format */
+	template<bool addAlphaByte, bool ccmEnabled>
+	void debayer10P_MONO_BGR888(uint8_t *dst, const uint8_t *src[]);
+	/* CSI-2 packed 12-bit raw monochrome format */
+	template<bool addAlphaByte, bool ccmEnabled>
+	void debayer12P_MONO_BGR888(uint8_t *dst, const uint8_t *src[]);
+	/* Mono 16 bpp unpacked */
+	template<bool addAlphaByte, bool ccmEnabled>
+	void debayer16_MONO_BGR888(uint8_t *dst, const uint8_t *src[]);
 
 	static int getInputConfig(PixelFormat inputFormat, DebayerInputConfig &config);
 	int setupStandardBayerOrder(BayerFormat::Order order);
diff --git a/src/libcamera/software_isp/debayer_egl.cpp b/src/libcamera/software_isp/debayer_egl.cpp
index 300822f9d..90eefc92e 100644
--- a/src/libcamera/software_isp/debayer_egl.cpp
+++ b/src/libcamera/software_isp/debayer_egl.cpp
@@ -61,6 +61,17 @@  int DebayerEGL::getInputConfig(PixelFormat inputFormat, DebayerInputConfig &conf
 						   formats::XBGR8888,
 						   formats::ABGR8888 };
 
+	/* Handle Monochrome case using bayerFormat */
+	if (bayerFormat.order == BayerFormat::Order::MONO) {
+		bool isPacked = (bayerFormat.packing == BayerFormat::Packing::CSI2);
+
+		config.bpp = isPacked ? bayerFormat.bitDepth : ((bayerFormat.bitDepth + 7) & ~7);
+		config.patternSize.width = isPacked ? 4 : 2;
+		config.patternSize.height = 2;
+		config.outputFormats = outputFormats;
+		return 0;
+	}
+
 	if ((bayerFormat.bitDepth == 8 || bayerFormat.bitDepth == 10) &&
 	    bayerFormat.packing == BayerFormat::Packing::None &&
 	    isStandardBayerOrder(bayerFormat.order)) {
@@ -166,6 +177,14 @@  int DebayerEGL::initBayerShaders(PixelFormat inputFormat, PixelFormat outputForm
 	shaderStridePixels_ = inputConfig_.stride;
 
 	switch (inputFormat) {
+	/* Monochrome cases: No color: phase, proper initialization to 0 for Rx and Rx_CSI2P */
+	case libcamera::formats::R8:
+	case libcamera::formats::R10:
+	case libcamera::formats::R12:
+	case libcamera::formats::R16:
+		firstRed_x_ = 0.0;
+		firstRed_y_ = 0.0;
+		break;
 	case libcamera::formats::SBGGR8:
 	case libcamera::formats::SBGGR10_CSI2P:
 	case libcamera::formats::SBGGR12_CSI2P:
@@ -197,6 +216,76 @@  int DebayerEGL::initBayerShaders(PixelFormat inputFormat, PixelFormat outputForm
 
 	/* Shader selection */
 	switch (inputFormat) {
+	case libcamera::formats::R8:
+		egl_.pushEnv(shaderEnv, "#define INV_MAX_VAL (1.0 / 255.0)");
+		egl_.pushEnv(shaderEnv, "#define RAW8");
+		fragmentShaderData = mono_unpacked_frag;
+		vertexShaderData = mono_unpacked_vert;
+		glFormat_ = GL_LUMINANCE;
+		bytesPerPixel_ = 1;
+		break;
+	case libcamera::formats::R10: {
+		BayerFormat bayerFormat = BayerFormat::fromPixelFormat(inputFormat);
+
+		egl_.pushEnv(shaderEnv, "#define INV_MAX_VAL (1.0 / 1023.0)");
+		if (bayerFormat.packing == BayerFormat::Packing::None) {
+			egl_.pushEnv(shaderEnv, "#define RAW10");
+			fragmentShaderData = mono_unpacked_frag;
+			vertexShaderData = mono_unpacked_vert;
+			glFormat_ = GL_RG;
+			bytesPerPixel_ = 2;
+		} else {
+			unsigned int stridePixels = width_ * 5 / 4;
+			egl_.pushEnv(shaderEnv, "#define BIT_DEPTH 10");
+			egl_.pushEnv(shaderEnv, "#define BLOCK_PIXELS 4.0");
+			egl_.pushEnv(shaderEnv, "#define BLOCK_BYTES 5.0");
+			egl_.pushEnv(shaderEnv, (std::string("#define TEX_WIDTH ") + std::to_string(stridePixels) + ".0").c_str());
+			egl_.pushEnv(shaderEnv, (std::string("#define TEX_HEIGHT ") + std::to_string(height_) + ".0").c_str());
+			egl_.pushEnv(shaderEnv, (std::string("#define INV_TEX_WIDTH (1.0 / ") + std::to_string(stridePixels) + ".0)").c_str());
+			egl_.pushEnv(shaderEnv, (std::string("#define INV_TEX_HEIGHT (1.0 / ") + std::to_string(height_) + ".0)").c_str());
+			fragmentShaderData = mono_1x_packed_frag;
+			vertexShaderData = identity_vert;
+			glFormat_ = GL_LUMINANCE;
+			bytesPerPixel_ = 1;
+			shaderStridePixels_ = stridePixels;
+		}
+		break;
+	}
+	case libcamera::formats::R12: {
+		BayerFormat bayerFormat = BayerFormat::fromPixelFormat(inputFormat);
+
+		egl_.pushEnv(shaderEnv, "#define INV_MAX_VAL (1.0 / 4095.0)");
+		if (bayerFormat.packing == BayerFormat::Packing::None) {
+			egl_.pushEnv(shaderEnv, "#define RAW12");
+			fragmentShaderData = mono_unpacked_frag;
+			vertexShaderData = mono_unpacked_vert;
+			glFormat_ = GL_RG;
+			bytesPerPixel_ = 2;
+		} else {
+			unsigned int stridePixels = width_ * 3 / 2;
+			egl_.pushEnv(shaderEnv, "#define BIT_DEPTH 12");
+			egl_.pushEnv(shaderEnv, "#define BLOCK_PIXELS 2.0");
+			egl_.pushEnv(shaderEnv, "#define BLOCK_BYTES 3.0");
+			egl_.pushEnv(shaderEnv, (std::string("#define TEX_WIDTH ") + std::to_string(stridePixels) + ".0").c_str());
+			egl_.pushEnv(shaderEnv, (std::string("#define TEX_HEIGHT ") + std::to_string(height_) + ".0").c_str());
+			egl_.pushEnv(shaderEnv, (std::string("#define INV_TEX_WIDTH (1.0 / ") + std::to_string(stridePixels) + ".0)").c_str());
+			egl_.pushEnv(shaderEnv, (std::string("#define INV_TEX_HEIGHT (1.0 / ") + std::to_string(height_) + ".0)").c_str());
+			fragmentShaderData = mono_1x_packed_frag;
+			vertexShaderData = identity_vert;
+			glFormat_ = GL_LUMINANCE;
+			bytesPerPixel_ = 1;
+			shaderStridePixels_ = stridePixels;
+		}
+		break;
+	}
+	case libcamera::formats::R16:
+		egl_.pushEnv(shaderEnv, "#define INV_MAX_VAL (1.0 / 65535.0)");
+		egl_.pushEnv(shaderEnv, "#define RAW16");
+		fragmentShaderData = mono_unpacked_frag;
+		vertexShaderData = mono_unpacked_vert;
+		glFormat_ = GL_RG;
+		bytesPerPixel_ = 2;
+		break;
 	case libcamera::formats::SBGGR8:
 	case libcamera::formats::SGBRG8:
 	case libcamera::formats::SGRBG8:
diff --git a/src/libcamera/software_isp/swstats_cpu.cpp b/src/libcamera/software_isp/swstats_cpu.cpp
index 7fb77ce7d..28c639ca1 100644
--- a/src/libcamera/software_isp/swstats_cpu.cpp
+++ b/src/libcamera/software_isp/swstats_cpu.cpp
@@ -375,6 +375,117 @@  void SwStatsCpu::statsGBRG12PLine0(const uint8_t *src[], SwIspStats &stats)
 	SWSTATS_FINISH_LINE_STATS()
 }
 
+/**
+ * Horizontal sub-sampling optimization: evaluates 1 pixel out of 4 (x += 4)
+ * to reduce CPU overhead. Each sample gets a weight of +4 to preserve global
+ * frame statistical density required by the Auto Exposure (AE) algorithm.
+ */
+void SwStatsCpu::statsMono8Line0(const uint8_t *src[], SwIspStats &stats)
+{
+	const uint8_t *cur = src[1] + window_.x;
+	uint64_t sum = 0;
+
+	for (unsigned int x = 0; x < window_.width; x += 4) {
+		sum += *cur;
+		stats.yHistogram[*cur >> 2] += 4;
+		cur += 4;
+	}
+	/* x << 2 (or x * 4) process 4 pixels per iteration */
+	stats.sum_.r() += (sum << 2);
+}
+
+void SwStatsCpu::statsMono10Line0(const uint8_t *src[], SwIspStats &stats)
+{
+	/**
+	 * Little-Endian optimization: Unpacked 10-bit format stores data
+	 * right-aligned in a 2-byte container. Reading via a 16-bit pointer
+	 * allows clean data access before applying the bit-shift.
+	 */
+	const uint16_t *cur = (const uint16_t *)(src[1] + window_.x * 2);
+	uint64_t sum = 0;
+
+	for (unsigned int x = 0; x < window_.width; x += 4) {
+		uint16_t val = *cur;
+		sum += val;
+		stats.yHistogram[val >> 4] += 4;
+		cur += 4;
+	}
+	/* x 10 >> 2 to 8bits and x << 2 (or x * 4) process 4 pixels per iteration */
+	stats.sum_.r() += sum;
+}
+
+void SwStatsCpu::statsMono12Line0(const uint8_t *src[], SwIspStats &stats)
+{
+	/**
+	 * 12-bit unpacked processing. Data is read via a native 16-bit pointer
+	 * and right-shifted to fit the histogram range.
+	 */
+	const uint16_t *cur = (const uint16_t *)(src[1] + window_.x * 2);
+	uint64_t sum = 0;
+
+	for (unsigned int x = 0; x < window_.width; x += 4) {
+		uint16_t val = *cur;
+		sum += val;
+		stats.yHistogram[val >> 6] += 4;
+		cur += 4;
+	}
+	/* 12 >> 4 to 8bits and x << 2 (or x * 4) process 4 pixels per iteration */
+	stats.sum_.r() += (sum >> 2);
+}
+
+void SwStatsCpu::statsMono10PLine0(const uint8_t *src[], SwIspStats &stats)
+{
+	/**
+	 * MIPI CSI-2 10-bit packed: 4 pixels in 5 bytes. Stepping 5 bytes
+	 * evaluates the head MSB pixel of each block as a native 8-bit value.
+	 */
+	const uint8_t *cur = src[1] + window_.x * 5 / 4;
+	uint64_t sum = 0;
+
+	for (unsigned int x = 0; x < window_.width; x += 4) {
+		uint8_t val = *cur;
+		sum += val;
+		stats.yHistogram[val >> 2] += 4;
+		cur += 5;
+	}
+	/* first byte x << 2 (or x * 4) process 4 pixels per iteration */
+	stats.sum_.r() += (sum << 2);
+}
+
+void SwStatsCpu::statsMono12PLine0(const uint8_t *src[], SwIspStats &stats)
+{
+	/**
+	 * MIPI CSI-2 12-bit packed: 2 pixels in 3 bytes. Stepping 6 bytes
+	 * advances by two full blocks for optimal data alignment.
+	 */
+	const uint8_t *cur = src[1] + window_.x * 3 / 2;
+	uint64_t sum = 0;
+
+	for (unsigned int x = 0; x < window_.width; x += 4) {
+		uint8_t val = *cur;
+		sum += val;
+		stats.yHistogram[val >> 2] += 4;
+		cur += 6;
+	}
+	/* first byte x << 2 (or x * 4) process 4 pixels per iteration */
+	stats.sum_.r() += (sum << 2);
+}
+
+void SwStatsCpu::statsMono16Line0(const uint8_t *src[], SwIspStats &stats)
+{
+	const uint16_t *cur = (const uint16_t *)(src[1] + window_.x * 2);
+	uint64_t sum = 0;
+
+	for (unsigned int x = 0; x < window_.width; x += 4) {
+		uint16_t val = *cur;
+		sum += val;
+		stats.yHistogram[val >> 10] += 4;
+		cur += 4;
+	}
+	/* 16 >> 8 to 8bits and x << 2 (or x * 4) process 4 pixels per iteration */
+	stats.sum_.r() += (sum >> 6);
+}
+
 /**
  * \brief Reset state to start statistics gathering for a new frame
  * \param[in] frame The frame number
@@ -494,6 +605,48 @@  int SwStatsCpu::configure(const StreamConfiguration &inputCfg, unsigned int stat
 
 	uint8_t bitDepth = bayerFormat.bitDepth;
 
+	if (bayerFormat.order == BayerFormat::Order::MONO) {
+		ySkipMask_ = 0x00;
+		xShift_ = 0;
+		sumShift_ = 0;
+		processFrame_ = &SwStatsCpu::processMonoFrame2;
+
+		if (bayerFormat.packing == BayerFormat::Packing::None) {
+			patternSize_.width = 2;
+			patternSize_.height = 2;
+
+			switch (bitDepth) {
+			case 8:
+				stats0_ = &SwStatsCpu::statsMono8Line0;
+				return 0;
+			case 10:
+				stats0_ = &SwStatsCpu::statsMono10Line0;
+				return 0;
+			case 12:
+				stats0_ = &SwStatsCpu::statsMono12Line0;
+				return 0;
+			case 16:
+				stats0_ = &SwStatsCpu::statsMono16Line0;
+				return 0;
+			}
+		}
+		if (bayerFormat.packing == BayerFormat::Packing::CSI2) {
+			patternSize_.height = 2;
+
+			switch (bitDepth) {
+			case 10:
+				patternSize_.width = 4;
+				stats0_ = &SwStatsCpu::statsMono10PLine0;
+				return 0;
+			case 12:
+				patternSize_.width = 2;
+				stats0_ = &SwStatsCpu::statsMono12PLine0;
+				return 0;
+			}
+		}
+		return -EINVAL;
+	}
+
 	if ((bitDepth == 10 || bitDepth == 12) &&
 	    bayerFormat.packing == BayerFormat::Packing::CSI2) {
 		if (bitDepth == 10)
@@ -594,6 +747,28 @@  void SwStatsCpu::processBayerFrame2(MappedFrameBuffer &in)
 	}
 }
 
+void SwStatsCpu::processMonoFrame2(MappedFrameBuffer &in)
+{
+	const uint8_t *src = in.planes()[0].data();
+	const uint8_t *linePointers[3] = { nullptr, nullptr, nullptr };
+
+	src += window_.y * stride_;
+
+	for (unsigned int y = 0; y < window_.height; y++) {
+		if (y & ySkipMask_) {
+			src += stride_;
+			continue;
+		}
+
+		linePointers[1] = src;
+		(this->*stats0_)(linePointers, stats_[0]);
+		src += stride_;
+	}
+
+	stats_[0].sum_.g() = stats_[0].sum_.r();
+	stats_[0].sum_.b() = stats_[0].sum_.r();
+}
+
 /**
  * \brief Calculate statistics for a frame in one go
  * \param[in] frame The frame number