forgectrl: VPU JPEG encode for the camera stream - 7.9 fps

Demosaic the superpixels straight to planar YUV420 and encode on the CODA960 (mainline coda V4L2 mem2mem, node found by personality); libjpeg stays as the automatic fallback and the snapshot path. All camera paths now demosaic from a cached bounce copy of the frame: the V4L2 MMAP capture buffers are uncached, and reading them in-place costs ~340 ms/frame vs 43 ms memcpy + 75 ms cached convert. Per-frame stats logged every 100 frames; /cam/status reports the encoder.
This commit is contained in:
ScottW514
2026-08-03 13:31:48 -04:00
parent 44137ae650
commit d75ff717ac
10 changed files with 425 additions and 23 deletions
+23 -11
View File
@@ -156,7 +156,8 @@ The cameras share the hardware video-mux; the NEWEST request wins it
The per-camera lamp (`pic/lid_led` / `head/white_led`) is raised to The per-camera lamp (`pic/lid_led` / `head/white_led`) is raised to
`FORGECTRL_LAMP` (default 132) while capturing and restored on idle. `FORGECTRL_LAMP` (default 132) while capturing and restored on idle.
Bench (2026-08-03, on the board): stream 3.2 fps sustained at 1296×972; Bench (2026-08-03, on the board): stream **7.9 fps** sustained at
1296×972 (VPU encode; 3.2 fps on the software fallback);
full-res snapshot 2.4 s warm / 2.7 s cold (cold includes the pipeline full-res snapshot 2.4 s warm / 2.7 s cold (cold includes the pipeline
bring-up); two parallel same-camera clients share the frame rate; idle bring-up); two parallel same-camera clients share the frame rate; idle
teardown observed. Borrow verified: head snapshot 200 during a lid teardown observed. Borrow verified: head snapshot 200 during a lid
@@ -174,16 +175,27 @@ core). Run by hand: `/usr/bin/forgectrl >> /data/forgectrl.log 2>&1 &`
`/?action=stream` alias) while jogging the machine from the same `/?action=stream` alias) while jogging the machine from the same
LightBurn session. LightBurn session.
Frame-rate ceiling and the offload path: 3.2 fps is CPU-bound in the **VPU JPEG offload: DONE 2026-08-03, bench-verified — 7.9 fps** (2.5×
JPEG encode (single A9, libjpeg-turbo NEON). The hardware answer is the the software rate). The stream path demosaics the 2×2 superpixels
**CODA960 VPU JPEG encoder — already probed with firmware on the image, straight to planar YUV420 (JFIF full-range 601) and the **CODA960 VPU
registered at /dev/video0** (V4L2 mem2mem, YUV input): demosaic the 2×2 JPEG encoder** (mainline coda, V4L2 mem2mem; found by personality, not
superpixels straight to YUV420 on the CPU (cheap) and let the VPU node number) does the encode: per-frame **copy 43 ms + convert 75 ms +
encode → est. 8–15 fps, likely CSI/memory-bound. The IPU cannot help encode 7 ms**. Two hard-won facts:
with demosaic (its IC is CSC/scale only, YUV/RGB in — that is the - **V4L2 MMAP capture buffers are uncached** — demosaicing in-place out
`imx-csc-scaler` at /dev/video8, useful only for a future full-res of one costs ~340 ms/frame at this resolution; one bulk memcpy into a
stream). Contained follow-up in forgectrl cam.c; keep the libjpeg path cached bounce buffer first (43 ms) makes the same demosaic run in
as fallback. Not yet done: lens calibration / bed alignment (the 75 ms. All camera paths (stream, snapshot, borrow) read from the
bounce copy.
- The VPU encoder accepts 1296×972 exactly (no MCU-alignment padding
needed) with quality via V4L2_CID_JPEG_COMPRESSION_QUALITY.
libjpeg remains the automatic fallback (`FORGECTRL_NO_VPU=1` forces
it) and the snapshot path; `/cam/status` reports `"encoder"`.
Remaining headroom: the scalar convert dominates — NEON would push
toward the sensor/CSI limit. If motion contention ever shows clamps
during streaming, a stream-fps cap knob is the easy relief valve.
The IPU cannot help with demosaic (its IC is CSC/scale only — the
`imx-csc-scaler` at /dev/video8 matters only for a future full-res
stream). Not yet done: lens calibration / bed alignment (the
fisheye needs LightBurn's camera calibration pass), and the deferred fisheye needs LightBurn's camera calibration pass), and the deferred
5.6 emulator homing-image smoke (the cloud emulator can now be pointed 5.6 emulator homing-image smoke (the cloud emulator can now be pointed
at live snapshots). at live snapshots).
@@ -11,6 +11,8 @@ SRC_URI = "\
file://cam.h \ file://cam.h \
file://debayer.c \ file://debayer.c \
file://debayer.h \ file://debayer.h \
file://vpu_jpeg.c \
file://vpu_jpeg.h \
file://forgectrl.init \ file://forgectrl.init \
" "
@@ -3,7 +3,7 @@ project(forgectrl C)
set(CMAKE_C_STANDARD 11) set(CMAKE_C_STANDARD 11)
add_executable(forgectrl main.c cam.c debayer.c) add_executable(forgectrl main.c cam.c debayer.c vpu_jpeg.c)
target_compile_options(forgectrl PRIVATE -Wall -Wextra -O2) target_compile_options(forgectrl PRIVATE -Wall -Wextra -O2)
target_link_libraries(forgectrl ulfius jpeg pthread m) target_link_libraries(forgectrl ulfius jpeg pthread m)
@@ -19,6 +19,7 @@
#define _GNU_SOURCE #define _GNU_SOURCE
#include "cam.h" #include "cam.h"
#include "debayer.h" #include "debayer.h"
#include "vpu_jpeg.h"
#include <errno.h> #include <errno.h>
#include <fcntl.h> #include <fcntl.h>
@@ -123,6 +124,7 @@ static struct {
/* config */ /* config */
int stream_quality; int stream_quality;
int lamp_level; int lamp_level;
int vpu_active; /* last stream frame went through the VPU */
} eng = { } eng = {
.ctl = PTHREAD_MUTEX_INITIALIZER, .ctl = PTHREAD_MUTEX_INITIALIZER,
.lock = PTHREAD_MUTEX_INITIALIZER, .lock = PTHREAD_MUTEX_INITIALIZER,
@@ -139,6 +141,11 @@ const char *cam_name(cam_id_t cam)
return camdefs[cam].name; return camdefs[cam].name;
} }
/* CODA960 hardware JPEG encoder for the stream path; libjpeg remains the
* fallback (and the snapshot path). Worker-thread use only. */
static vpu_jpeg_t *vpu;
static int vpu_disabled;
/* ------------------------------------------------------------------ util */ /* ------------------------------------------------------------------ util */
static void now_ts(struct timespec *ts) static void now_ts(struct timespec *ts)
@@ -539,7 +546,8 @@ static void fail_snap(void)
/* Capture one frame from the currently-started pipeline and feed it to /* Capture one frame from the currently-started pipeline and feed it to
* deliver_snap. Used by the borrow path. */ * deliver_snap. Used by the borrow path. */
static int grab_one_snap(uint8_t *rgb_half, uint8_t **prgb_full) static int grab_one_snap(uint8_t *raw_cached, uint8_t *rgb_half,
uint8_t **prgb_full)
{ {
for (int tries = 0; tries < MAX_DQ_TIMEOUTS; tries++) { for (int tries = 0; tries < MAX_DQ_TIMEOUTS; tries++) {
fd_set fds; fd_set fds;
@@ -561,7 +569,9 @@ static int grab_one_snap(uint8_t *rgb_half, uint8_t **prgb_full)
continue; continue;
return -1; return -1;
} }
deliver_snap(eng.bufs[buf.index].start, rgb_half, prgb_full); memcpy(raw_cached, eng.bufs[buf.index].start,
(size_t)CAM_W * CAM_H);
deliver_snap(raw_cached, rgb_half, prgb_full);
xioctl(eng.fd, VIDIOC_QBUF, &buf); xioctl(eng.fd, VIDIOC_QBUF, &buf);
return 0; return 0;
} }
@@ -573,12 +583,20 @@ static void *worker(void *arg)
(void)arg; (void)arg;
uint8_t *rgb_half = malloc((size_t)HALF_W * HALF_H * 3); uint8_t *rgb_half = malloc((size_t)HALF_W * HALF_H * 3);
uint8_t *rgb_full = NULL; /* allocated on first full-res snapshot */ uint8_t *rgb_full = NULL; /* allocated on first full-res snapshot */
/* The V4L2 MMAP capture buffers are DMA-coherent = UNCACHED: byte
* reads from them cost a bus transaction each and demosaicing
* straight out of one measures ~340 ms/frame. One bulk memcpy into
* this cached bounce buffer first makes the demosaic run at cached
* speed. */
uint8_t *raw_cached = malloc((size_t)CAM_W * CAM_H);
double stat_copy_ms = 0, stat_conv_ms = 0, stat_enc_ms = 0;
unsigned stat_n = 0;
int dq_timeouts = 0; int dq_timeouts = 0;
struct timespec fps_t0; struct timespec fps_t0;
now_ts(&fps_t0); now_ts(&fps_t0);
uint64_t fps_frames = 0; uint64_t fps_frames = 0;
if (!rgb_half) { if (!rgb_half || !raw_cached) {
fprintf(stderr, "cam: worker OOM\n"); fprintf(stderr, "cam: worker OOM\n");
goto out; goto out;
} }
@@ -607,7 +625,7 @@ static void *worker(void *arg)
char berr[128]; char berr[128];
release_capture(); release_capture();
if (start_capture(borrow_cam, berr, sizeof(berr)) == 0) { if (start_capture(borrow_cam, berr, sizeof(berr)) == 0) {
if (grab_one_snap(rgb_half, &rgb_full)) if (grab_one_snap(raw_cached, rgb_half, &rgb_full))
fail_snap(); fail_snap();
release_capture(); release_capture();
} else { } else {
@@ -653,23 +671,77 @@ static void *worker(void *arg)
break; break;
} }
dq_timeouts = 0; dq_timeouts = 0;
const uint8_t *raw = eng.bufs[buf.index].start; struct timespec c0, c1;
now_ts(&c0);
memcpy(raw_cached, eng.bufs[buf.index].start,
(size_t)CAM_W * CAM_H);
now_ts(&c1);
const uint8_t *raw = raw_cached;
/* Snapshot request rides on the same raw frame */ /* Snapshot request rides on the same raw frame */
if (snap) if (snap)
deliver_snap(raw, rgb_half, &rgb_full); deliver_snap(raw, rgb_half, &rgb_full);
/* Stream frame */ /* Stream frame: demosaic + encode, VPU first, libjpeg fallback */
if (clients > 0) { if (clients > 0) {
uint8_t *jpg = NULL; uint8_t *jpg = NULL;
size_t len = 0; size_t len = 0;
debayer_bggr_half(raw, rgb_half, CAM_W, CAM_H, HFLIP); int via_vpu = 0;
if (jpeg_encode_rgb(rgb_half, HALF_W, HALF_H, eng.stream_quality, struct timespec e0, e1, e2;
1, &jpg, &len) == 0) { now_ts(&e0);
if (!vpu_disabled && !vpu) {
vpu = vpu_jpeg_open(HALF_W, HALF_H, eng.stream_quality);
if (!vpu) {
vpu_disabled = 1;
fprintf(stderr, "cam: no VPU JPEG encoder, "
"using software encode\n");
}
}
if (vpu) {
uint8_t *yp, *up, *vp;
int ys, uvs;
vpu_jpeg_planes(vpu, &yp, &up, &vp, &ys, &uvs);
debayer_bggr_half_yuv420(raw, CAM_W, CAM_H, HFLIP,
yp, ys, up, vp, uvs);
now_ts(&e1);
if (vpu_jpeg_encode(vpu, &jpg, &len) == 0) {
via_vpu = 1;
} else {
fprintf(stderr, "cam: VPU encode failed, "
"falling back to software\n");
vpu_jpeg_close(vpu);
vpu = NULL;
vpu_disabled = 1;
}
}
if (!via_vpu) {
debayer_bggr_half(raw, rgb_half, CAM_W, CAM_H, HFLIP);
now_ts(&e1);
if (jpeg_encode_rgb(rgb_half, HALF_W, HALF_H,
eng.stream_quality, 1, &jpg, &len))
jpg = NULL;
}
now_ts(&e2);
if (jpg) {
stat_copy_ms += ts_diff(&c1, &c0) * 1e3;
stat_conv_ms += ts_diff(&e1, &e0) * 1e3;
stat_enc_ms += ts_diff(&e2, &e1) * 1e3;
if (++stat_n >= 100) {
fprintf(stderr, "cam: stream stats: copy %.0f ms, "
"convert %.0f ms, encode %.0f ms avg (%s)\n",
stat_copy_ms / stat_n, stat_conv_ms / stat_n,
stat_enc_ms / stat_n,
via_vpu ? "vpu" : "software");
stat_copy_ms = stat_conv_ms = stat_enc_ms = 0;
stat_n = 0;
}
pthread_mutex_lock(&eng.lock); pthread_mutex_lock(&eng.lock);
free(eng.stream_jpg); free(eng.stream_jpg);
eng.stream_jpg = jpg; eng.stream_jpg = jpg;
eng.stream_len = len; eng.stream_len = len;
eng.vpu_active = via_vpu;
eng.seq++; eng.seq++;
fps_frames++; fps_frames++;
struct timespec t; struct timespec t;
@@ -695,6 +767,7 @@ out:
release_capture(); release_capture();
free(rgb_half); free(rgb_half);
free(rgb_full); free(rgb_full);
free(raw_cached);
pthread_mutex_lock(&eng.lock); pthread_mutex_lock(&eng.lock);
eng.running = 0; eng.running = 0;
/* fail any waiter: stream clients see running==0, a pending snapshot /* fail any waiter: stream clients see running==0, a pending snapshot
@@ -807,6 +880,8 @@ void cam_engine_init(void)
if (l >= 0 && l <= 1023) if (l >= 0 && l <= 1023)
eng.lamp_level = l; eng.lamp_level = l;
} }
if (getenv("FORGECTRL_NO_VPU"))
vpu_disabled = 1;
} }
void cam_engine_shutdown(void) void cam_engine_shutdown(void)
@@ -822,6 +897,10 @@ void cam_engine_shutdown(void)
eng.tid_valid = 0; eng.tid_valid = 0;
pthread_mutex_unlock(&eng.lock); pthread_mutex_unlock(&eng.lock);
} }
if (vpu) {
vpu_jpeg_close(vpu);
vpu = NULL;
}
pthread_mutex_unlock(&eng.ctl); pthread_mutex_unlock(&eng.ctl);
} }
@@ -970,5 +1049,6 @@ void cam_get_status(struct cam_status *st)
st->clients = eng.clients; st->clients = eng.clients;
st->seq = eng.seq; st->seq = eng.seq;
st->fps = eng.fps; st->fps = eng.fps;
st->vpu = eng.vpu_active;
pthread_mutex_unlock(&eng.lock); pthread_mutex_unlock(&eng.lock);
} }
@@ -57,6 +57,7 @@ struct cam_status {
int clients; int clients;
uint64_t seq; uint64_t seq;
double fps; double fps;
int vpu; /* stream frames are VPU-encoded */
}; };
void cam_get_status(struct cam_status *st); void cam_get_status(struct cam_status *st);
@@ -7,6 +7,8 @@
* even rows: B G B G ... * even rows: B G B G ...
* odd rows: G R G R ... * odd rows: G R G R ...
*/ */
#include <stddef.h>
#include "debayer.h" #include "debayer.h"
static inline int clampi(int v, int lo, int hi) static inline int clampi(int v, int lo, int hi)
@@ -62,6 +64,51 @@ void debayer_bggr_bilinear(const uint8_t *raw, uint8_t *rgb,
} }
} }
void debayer_bggr_half_yuv420(const uint8_t *raw, int w, int h, int hflip,
uint8_t *yp, int y_stride,
uint8_t *up, uint8_t *vp, int uv_stride)
{
const int ow = w / 2; /* luma dimensions */
const int oh = h / 2;
const int uvw = ow / 2;
/* JFIF full-range ITU-R 601, x256 fixed point:
* Y = 0.299 R + 0.587 G + 0.114 B -> 77 150 29
* Cb = -0.169 R - 0.331 G + 0.500 B + 128 -> -43 -85 128
* Cr = 0.500 R - 0.419 G - 0.081 B + 128 -> 128 -107 -21 */
for (int y2 = 0; y2 < oh / 2; y2++) {
uint8_t *yrow0 = yp + (size_t)(2 * y2) * y_stride;
uint8_t *yrow1 = yrow0 + y_stride;
uint8_t *urow = up + (size_t)y2 * uv_stride;
uint8_t *vrow = vp + (size_t)y2 * uv_stride;
for (int x2 = 0; x2 < uvw; x2++) {
int rs = 0, gs = 0, bs = 0;
for (int sy = 0; sy < 2; sy++) {
const int row = 2 * y2 + sy;
const uint8_t *quad_row = raw + (size_t)(2 * row) * w;
uint8_t *yrow = sy ? yrow1 : yrow0;
for (int sx = 0; sx < 2; sx++) {
const int col = 2 * x2 + sx;
const uint8_t *q = quad_row + 2 * col;
const int b = q[0];
const int g = (q[1] + q[w] + 1) >> 1;
const int r = q[w + 1];
rs += r;
gs += g;
bs += b;
yrow[hflip ? ow - 1 - col : col] =
(uint8_t)((77 * r + 150 * g + 29 * b + 128) >> 8);
}
}
const int cx = hflip ? uvw - 1 - x2 : x2;
int cb = ((-43 * rs - 85 * gs + 128 * bs + 512) >> 10) + 128;
int cr = ((128 * rs - 107 * gs - 21 * bs + 512) >> 10) + 128;
urow[cx] = (uint8_t)(cb < 0 ? 0 : (cb > 255 ? 255 : cb));
vrow[cx] = (uint8_t)(cr < 0 ? 0 : (cr > 255 ? 255 : cr));
}
}
}
void debayer_bggr_half(const uint8_t *raw, uint8_t *rgb, void debayer_bggr_half(const uint8_t *raw, uint8_t *rgb,
int w, int h, int hflip) int w, int h, int hflip)
{ {
@@ -21,4 +21,12 @@ void debayer_bggr_bilinear(const uint8_t *raw, uint8_t *rgb,
void debayer_bggr_half(const uint8_t *raw, uint8_t *rgb, void debayer_bggr_half(const uint8_t *raw, uint8_t *rgb,
int w, int h, int hflip); int w, int h, int hflip);
/* Half-resolution demosaic straight to planar YUV420 (JFIF full-range,
* ITU-R 601) for the VPU JPEG encoder: luma per 2x2 BGGR quad at
* (w/2)x(h/2), chroma averaged per 2x2 luma block at (w/4)x(h/4).
* w/2 and h/2 must be even. Strides are in bytes. */
void debayer_bggr_half_yuv420(const uint8_t *raw, int w, int h, int hflip,
uint8_t *yp, int y_stride,
uint8_t *up, uint8_t *vp, int uv_stride);
#endif #endif
@@ -210,11 +210,12 @@ static int cb_status(const struct _u_request *req, struct _u_response *res,
char body[256]; char body[256];
snprintf(body, sizeof(body), snprintf(body, sizeof(body),
"{\"running\":%s,\"cam\":\"%s\",\"clients\":%d," "{\"running\":%s,\"cam\":\"%s\",\"clients\":%d,"
"\"frames\":%llu,\"fps\":%.1f," "\"frames\":%llu,\"fps\":%.1f,\"encoder\":\"%s\","
"\"stream\":{\"width\":1296,\"height\":972}," "\"stream\":{\"width\":1296,\"height\":972},"
"\"snapshot\":{\"width\":2592,\"height\":1944}}", "\"snapshot\":{\"width\":2592,\"height\":1944}}",
st.running ? "true" : "false", cam_name(st.cam), st.clients, st.running ? "true" : "false", cam_name(st.cam), st.clients,
(unsigned long long)st.seq, st.fps); (unsigned long long)st.seq, st.fps,
st.vpu ? "vpu" : "software");
ulfius_set_string_body_response(res, 200, body); ulfius_set_string_body_response(res, 200, body);
ulfius_add_header_to_response(res, "Content-Type", "application/json"); ulfius_add_header_to_response(res, "Content-Type", "application/json");
return U_CALLBACK_CONTINUE; return U_CALLBACK_CONTINUE;
@@ -0,0 +1,218 @@
/*
* vpu_jpeg.c - hardware JPEG encoding on the i.MX6 CODA960 VPU
* Copyright (c) 2026 Scott Wiederhold <s.e.wiederhold@gmail.com>
* SPDX-License-Identifier: MIT
*
* V4L2 mem2mem, single-planar API against the mainline coda driver: one
* MMAP buffer on each queue, synchronous QBUF/DQBUF per frame. The node
* is found by personality (driver "coda", JPEG on the capture side,
* YUV420 accepted on the output side), never by number - coda registers
* four nodes and the numbering depends on probe order.
*/
#include "vpu_jpeg.h"
#include <errno.h>
#include <fcntl.h>
#include <linux/videodev2.h>
#include <poll.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/ioctl.h>
#include <sys/mman.h>
#include <unistd.h>
#define ENCODE_TIMEOUT_MS 1000
struct vpu_jpeg {
int fd;
int w, h;
int bpl; /* OUTPUT luma stride from S_FMT */
uint8_t *out; /* mapped OUTPUT (YUV420) buffer */
size_t out_size;
uint8_t *cap; /* mapped CAPTURE (JPEG) buffer */
size_t cap_size;
};
static int xioctl(int fd, unsigned long req, void *arg)
{
int r;
do {
r = ioctl(fd, req, arg);
} while (r == -1 && errno == EINTR);
return r;
}
/* Is this node the coda JPEG encoder? (JPEG capture, YUV420 output) */
static int is_jpeg_encoder(int fd)
{
struct v4l2_capability cap = {0};
if (xioctl(fd, VIDIOC_QUERYCAP, &cap) < 0 ||
strcmp((const char *)cap.driver, "coda") != 0 ||
!(cap.device_caps & V4L2_CAP_VIDEO_M2M))
return 0;
struct v4l2_fmtdesc fd0 = { .type = V4L2_BUF_TYPE_VIDEO_CAPTURE };
if (xioctl(fd, VIDIOC_ENUM_FMT, &fd0) < 0 ||
fd0.pixelformat != V4L2_PIX_FMT_JPEG)
return 0;
for (unsigned i = 0; ; i++) {
struct v4l2_fmtdesc fo = { .type = V4L2_BUF_TYPE_VIDEO_OUTPUT,
.index = i };
if (xioctl(fd, VIDIOC_ENUM_FMT, &fo) < 0)
return 0;
if (fo.pixelformat == V4L2_PIX_FMT_YUV420)
return 1;
}
}
static int find_encoder(void)
{
for (int i = 0; i < 32; i++) {
char path[32];
snprintf(path, sizeof(path), "/dev/video%d", i);
int fd = open(path, O_RDWR | O_NONBLOCK, 0);
if (fd < 0)
continue;
if (is_jpeg_encoder(fd))
return fd;
close(fd);
}
return -1;
}
static int map_one(int fd, enum v4l2_buf_type type, uint8_t **mem,
size_t *size)
{
struct v4l2_requestbuffers req = { .count = 1, .type = type,
.memory = V4L2_MEMORY_MMAP };
if (xioctl(fd, VIDIOC_REQBUFS, &req) < 0 || req.count < 1)
return -1;
struct v4l2_buffer buf = { .type = type, .memory = V4L2_MEMORY_MMAP,
.index = 0 };
if (xioctl(fd, VIDIOC_QUERYBUF, &buf) < 0)
return -1;
*mem = mmap(NULL, buf.length, PROT_READ | PROT_WRITE, MAP_SHARED,
fd, buf.m.offset);
if (*mem == MAP_FAILED) {
*mem = NULL;
return -1;
}
*size = buf.length;
return 0;
}
vpu_jpeg_t *vpu_jpeg_open(int w, int h, int quality)
{
vpu_jpeg_t *v = calloc(1, sizeof(*v));
if (!v)
return NULL;
v->fd = find_encoder();
if (v->fd < 0)
goto fail;
struct v4l2_format fo = { .type = V4L2_BUF_TYPE_VIDEO_OUTPUT };
fo.fmt.pix.width = (unsigned)w;
fo.fmt.pix.height = (unsigned)h;
fo.fmt.pix.pixelformat = V4L2_PIX_FMT_YUV420;
fo.fmt.pix.field = V4L2_FIELD_NONE;
if (xioctl(v->fd, VIDIOC_S_FMT, &fo) < 0 ||
fo.fmt.pix.width != (unsigned)w ||
fo.fmt.pix.height != (unsigned)h) {
fprintf(stderr, "vpu: S_FMT output rejected %dx%d\n", w, h);
goto fail;
}
v->w = w;
v->h = h;
v->bpl = (int)fo.fmt.pix.bytesperline;
struct v4l2_format fc = { .type = V4L2_BUF_TYPE_VIDEO_CAPTURE };
fc.fmt.pix.width = (unsigned)w;
fc.fmt.pix.height = (unsigned)h;
fc.fmt.pix.pixelformat = V4L2_PIX_FMT_JPEG;
if (xioctl(v->fd, VIDIOC_S_FMT, &fc) < 0)
goto fail;
struct v4l2_control q = { .id = V4L2_CID_JPEG_COMPRESSION_QUALITY,
.value = quality };
xioctl(v->fd, VIDIOC_S_CTRL, &q); /* best effort */
if (map_one(v->fd, V4L2_BUF_TYPE_VIDEO_OUTPUT, &v->out, &v->out_size) ||
map_one(v->fd, V4L2_BUF_TYPE_VIDEO_CAPTURE, &v->cap, &v->cap_size))
goto fail;
enum v4l2_buf_type t = V4L2_BUF_TYPE_VIDEO_OUTPUT;
if (xioctl(v->fd, VIDIOC_STREAMON, &t) < 0)
goto fail;
t = V4L2_BUF_TYPE_VIDEO_CAPTURE;
if (xioctl(v->fd, VIDIOC_STREAMON, &t) < 0)
goto fail;
return v;
fail:
vpu_jpeg_close(v);
return NULL;
}
void vpu_jpeg_planes(vpu_jpeg_t *v, uint8_t **y, uint8_t **u, uint8_t **vv,
int *y_stride, int *uv_stride)
{
*y = v->out;
*u = v->out + (size_t)v->bpl * v->h;
*vv = *u + (size_t)(v->bpl / 2) * (v->h / 2);
*y_stride = v->bpl;
*uv_stride = v->bpl / 2;
}
int vpu_jpeg_encode(vpu_jpeg_t *v, uint8_t **jpeg, size_t *len)
{
struct v4l2_buffer cb = { .type = V4L2_BUF_TYPE_VIDEO_CAPTURE,
.memory = V4L2_MEMORY_MMAP, .index = 0 };
struct v4l2_buffer ob = { .type = V4L2_BUF_TYPE_VIDEO_OUTPUT,
.memory = V4L2_MEMORY_MMAP, .index = 0 };
ob.bytesused = (unsigned)v->out_size;
if (xioctl(v->fd, VIDIOC_QBUF, &cb) < 0 ||
xioctl(v->fd, VIDIOC_QBUF, &ob) < 0)
return -1;
struct pollfd pfd = { .fd = v->fd, .events = POLLIN };
int pr;
do {
pr = poll(&pfd, 1, ENCODE_TIMEOUT_MS);
} while (pr == -1 && errno == EINTR);
if (pr <= 0)
return -1;
if (xioctl(v->fd, VIDIOC_DQBUF, &cb) < 0)
return -1;
xioctl(v->fd, VIDIOC_DQBUF, &ob);
uint8_t *out = malloc(cb.bytesused);
if (!out)
return -1;
memcpy(out, v->cap, cb.bytesused);
*jpeg = out;
*len = cb.bytesused;
return 0;
}
void vpu_jpeg_close(vpu_jpeg_t *v)
{
if (!v)
return;
if (v->fd >= 0) {
enum v4l2_buf_type t = V4L2_BUF_TYPE_VIDEO_OUTPUT;
xioctl(v->fd, VIDIOC_STREAMOFF, &t);
t = V4L2_BUF_TYPE_VIDEO_CAPTURE;
xioctl(v->fd, VIDIOC_STREAMOFF, &t);
}
if (v->out)
munmap(v->out, v->out_size);
if (v->cap)
munmap(v->cap, v->cap_size);
if (v->fd >= 0)
close(v->fd);
free(v);
}
@@ -0,0 +1,33 @@
/*
* vpu_jpeg.h - hardware JPEG encoding on the i.MX6 CODA960 VPU
* Copyright (c) 2026 Scott Wiederhold <s.e.wiederhold@gmail.com>
* SPDX-License-Identifier: MIT
*
* Thin wrapper around the mainline coda V4L2 mem2mem JPEG encoder: the
* caller writes planar YUV420 directly into the encoder's OUTPUT buffer
* (vpu_jpeg_planes) and gets back a malloc'd JFIF JPEG.
*/
#ifndef FORGECTRL_VPU_JPEG_H
#define FORGECTRL_VPU_JPEG_H
#include <stddef.h>
#include <stdint.h>
typedef struct vpu_jpeg vpu_jpeg_t;
/* Locate the CODA JPEG encoder video node, configure it for w x h YUV420
* -> JPEG at the given quality (5..100), and map one buffer per queue.
* Returns NULL if no encoder exists or setup fails. */
vpu_jpeg_t *vpu_jpeg_open(int w, int h, int quality);
/* Planes of the mapped OUTPUT buffer for direct fill. */
void vpu_jpeg_planes(vpu_jpeg_t *v, uint8_t **y, uint8_t **u, uint8_t **vv,
int *y_stride, int *uv_stride);
/* Encode the currently-filled OUTPUT buffer. On success *jpeg is malloc'd
* (caller frees) and 0 is returned. */
int vpu_jpeg_encode(vpu_jpeg_t *v, uint8_t **jpeg, size_t *len);
void vpu_jpeg_close(vpu_jpeg_t *v);
#endif