diff --git a/docs/BRINGUP.md b/docs/BRINGUP.md index 7fdf8e4..53a86e0 100644 --- a/docs/BRINGUP.md +++ b/docs/BRINGUP.md @@ -156,7 +156,8 @@ The cameras share the hardware video-mux; the NEWEST request wins it The per-camera lamp (`pic/lid_led` / `head/white_led`) is raised to `FORGECTRL_LAMP` (default 132) while capturing and restored on idle. -Bench (2026-08-03, on the board): stream 3.2 fps sustained at 1296×972; +Bench (2026-08-03, on the board): stream **7.9 fps** sustained at +1296×972 (VPU encode; 3.2 fps on the software fallback); full-res snapshot 2.4 s warm / 2.7 s cold (cold includes the pipeline bring-up); two parallel same-camera clients share the frame rate; idle teardown observed. Borrow verified: head snapshot 200 during a lid @@ -174,16 +175,27 @@ core). Run by hand: `/usr/bin/forgectrl >> /data/forgectrl.log 2>&1 &` `/?action=stream` alias) while jogging the machine from the same LightBurn session. -Frame-rate ceiling and the offload path: 3.2 fps is CPU-bound in the -JPEG encode (single A9, libjpeg-turbo NEON). The hardware answer is the -**CODA960 VPU JPEG encoder — already probed with firmware on the image, -registered at /dev/video0** (V4L2 mem2mem, YUV input): demosaic the 2×2 -superpixels straight to YUV420 on the CPU (cheap) and let the VPU -encode → est. 8–15 fps, likely CSI/memory-bound. The IPU cannot help -with demosaic (its IC is CSC/scale only, YUV/RGB in — that is the -`imx-csc-scaler` at /dev/video8, useful only for a future full-res -stream). Contained follow-up in forgectrl cam.c; keep the libjpeg path -as fallback. Not yet done: lens calibration / bed alignment (the +**VPU JPEG offload: DONE 2026-08-03, bench-verified — 7.9 fps** (2.5× +the software rate). The stream path demosaics the 2×2 superpixels +straight to planar YUV420 (JFIF full-range 601) and the **CODA960 VPU +JPEG encoder** (mainline coda, V4L2 mem2mem; found by personality, not +node number) does the encode: per-frame **copy 43 ms + convert 75 ms + +encode 7 ms**. Two hard-won facts: +- **V4L2 MMAP capture buffers are uncached** — demosaicing in-place out + of one costs ~340 ms/frame at this resolution; one bulk memcpy into a + cached bounce buffer first (43 ms) makes the same demosaic run in + 75 ms. All camera paths (stream, snapshot, borrow) read from the + bounce copy. +- The VPU encoder accepts 1296×972 exactly (no MCU-alignment padding + needed) with quality via V4L2_CID_JPEG_COMPRESSION_QUALITY. +libjpeg remains the automatic fallback (`FORGECTRL_NO_VPU=1` forces +it) and the snapshot path; `/cam/status` reports `"encoder"`. +Remaining headroom: the scalar convert dominates — NEON would push +toward the sensor/CSI limit. If motion contention ever shows clamps +during streaming, a stream-fps cap knob is the easy relief valve. +The IPU cannot help with demosaic (its IC is CSC/scale only — the +`imx-csc-scaler` at /dev/video8 matters only for a future full-res +stream). Not yet done: lens calibration / bed alignment (the fisheye needs LightBurn's camera calibration pass), and the deferred 5.6 emulator homing-image smoke (the cloud emulator can now be pointed at live snapshots). diff --git a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl.bb b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl.bb index 28bd850..6289774 100644 --- a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl.bb +++ b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl.bb @@ -11,6 +11,8 @@ SRC_URI = "\ file://cam.h \ file://debayer.c \ file://debayer.h \ + file://vpu_jpeg.c \ + file://vpu_jpeg.h \ file://forgectrl.init \ " diff --git a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/CMakeLists.txt b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/CMakeLists.txt index 2f8fd75..4251931 100644 --- a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/CMakeLists.txt +++ b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/CMakeLists.txt @@ -3,7 +3,7 @@ project(forgectrl C) set(CMAKE_C_STANDARD 11) -add_executable(forgectrl main.c cam.c debayer.c) +add_executable(forgectrl main.c cam.c debayer.c vpu_jpeg.c) target_compile_options(forgectrl PRIVATE -Wall -Wextra -O2) target_link_libraries(forgectrl ulfius jpeg pthread m) diff --git a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/cam.c b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/cam.c index d522da9..491b7ab 100644 --- a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/cam.c +++ b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/cam.c @@ -19,6 +19,7 @@ #define _GNU_SOURCE #include "cam.h" #include "debayer.h" +#include "vpu_jpeg.h" #include #include @@ -123,6 +124,7 @@ static struct { /* config */ int stream_quality; int lamp_level; + int vpu_active; /* last stream frame went through the VPU */ } eng = { .ctl = PTHREAD_MUTEX_INITIALIZER, .lock = PTHREAD_MUTEX_INITIALIZER, @@ -139,6 +141,11 @@ const char *cam_name(cam_id_t cam) return camdefs[cam].name; } +/* CODA960 hardware JPEG encoder for the stream path; libjpeg remains the + * fallback (and the snapshot path). Worker-thread use only. */ +static vpu_jpeg_t *vpu; +static int vpu_disabled; + /* ------------------------------------------------------------------ util */ static void now_ts(struct timespec *ts) @@ -539,7 +546,8 @@ static void fail_snap(void) /* Capture one frame from the currently-started pipeline and feed it to * deliver_snap. Used by the borrow path. */ -static int grab_one_snap(uint8_t *rgb_half, uint8_t **prgb_full) +static int grab_one_snap(uint8_t *raw_cached, uint8_t *rgb_half, + uint8_t **prgb_full) { for (int tries = 0; tries < MAX_DQ_TIMEOUTS; tries++) { fd_set fds; @@ -561,7 +569,9 @@ static int grab_one_snap(uint8_t *rgb_half, uint8_t **prgb_full) continue; return -1; } - deliver_snap(eng.bufs[buf.index].start, rgb_half, prgb_full); + memcpy(raw_cached, eng.bufs[buf.index].start, + (size_t)CAM_W * CAM_H); + deliver_snap(raw_cached, rgb_half, prgb_full); xioctl(eng.fd, VIDIOC_QBUF, &buf); return 0; } @@ -573,12 +583,20 @@ static void *worker(void *arg) (void)arg; uint8_t *rgb_half = malloc((size_t)HALF_W * HALF_H * 3); uint8_t *rgb_full = NULL; /* allocated on first full-res snapshot */ + /* The V4L2 MMAP capture buffers are DMA-coherent = UNCACHED: byte + * reads from them cost a bus transaction each and demosaicing + * straight out of one measures ~340 ms/frame. One bulk memcpy into + * this cached bounce buffer first makes the demosaic run at cached + * speed. */ + uint8_t *raw_cached = malloc((size_t)CAM_W * CAM_H); + double stat_copy_ms = 0, stat_conv_ms = 0, stat_enc_ms = 0; + unsigned stat_n = 0; int dq_timeouts = 0; struct timespec fps_t0; now_ts(&fps_t0); uint64_t fps_frames = 0; - if (!rgb_half) { + if (!rgb_half || !raw_cached) { fprintf(stderr, "cam: worker OOM\n"); goto out; } @@ -607,7 +625,7 @@ static void *worker(void *arg) char berr[128]; release_capture(); if (start_capture(borrow_cam, berr, sizeof(berr)) == 0) { - if (grab_one_snap(rgb_half, &rgb_full)) + if (grab_one_snap(raw_cached, rgb_half, &rgb_full)) fail_snap(); release_capture(); } else { @@ -653,23 +671,77 @@ static void *worker(void *arg) break; } dq_timeouts = 0; - const uint8_t *raw = eng.bufs[buf.index].start; + struct timespec c0, c1; + now_ts(&c0); + memcpy(raw_cached, eng.bufs[buf.index].start, + (size_t)CAM_W * CAM_H); + now_ts(&c1); + const uint8_t *raw = raw_cached; /* Snapshot request rides on the same raw frame */ if (snap) deliver_snap(raw, rgb_half, &rgb_full); - /* Stream frame */ + /* Stream frame: demosaic + encode, VPU first, libjpeg fallback */ if (clients > 0) { uint8_t *jpg = NULL; size_t len = 0; - debayer_bggr_half(raw, rgb_half, CAM_W, CAM_H, HFLIP); - if (jpeg_encode_rgb(rgb_half, HALF_W, HALF_H, eng.stream_quality, - 1, &jpg, &len) == 0) { + int via_vpu = 0; + struct timespec e0, e1, e2; + now_ts(&e0); + + if (!vpu_disabled && !vpu) { + vpu = vpu_jpeg_open(HALF_W, HALF_H, eng.stream_quality); + if (!vpu) { + vpu_disabled = 1; + fprintf(stderr, "cam: no VPU JPEG encoder, " + "using software encode\n"); + } + } + if (vpu) { + uint8_t *yp, *up, *vp; + int ys, uvs; + vpu_jpeg_planes(vpu, &yp, &up, &vp, &ys, &uvs); + debayer_bggr_half_yuv420(raw, CAM_W, CAM_H, HFLIP, + yp, ys, up, vp, uvs); + now_ts(&e1); + if (vpu_jpeg_encode(vpu, &jpg, &len) == 0) { + via_vpu = 1; + } else { + fprintf(stderr, "cam: VPU encode failed, " + "falling back to software\n"); + vpu_jpeg_close(vpu); + vpu = NULL; + vpu_disabled = 1; + } + } + if (!via_vpu) { + debayer_bggr_half(raw, rgb_half, CAM_W, CAM_H, HFLIP); + now_ts(&e1); + if (jpeg_encode_rgb(rgb_half, HALF_W, HALF_H, + eng.stream_quality, 1, &jpg, &len)) + jpg = NULL; + } + now_ts(&e2); + + if (jpg) { + stat_copy_ms += ts_diff(&c1, &c0) * 1e3; + stat_conv_ms += ts_diff(&e1, &e0) * 1e3; + stat_enc_ms += ts_diff(&e2, &e1) * 1e3; + if (++stat_n >= 100) { + fprintf(stderr, "cam: stream stats: copy %.0f ms, " + "convert %.0f ms, encode %.0f ms avg (%s)\n", + stat_copy_ms / stat_n, stat_conv_ms / stat_n, + stat_enc_ms / stat_n, + via_vpu ? "vpu" : "software"); + stat_copy_ms = stat_conv_ms = stat_enc_ms = 0; + stat_n = 0; + } pthread_mutex_lock(&eng.lock); free(eng.stream_jpg); eng.stream_jpg = jpg; eng.stream_len = len; + eng.vpu_active = via_vpu; eng.seq++; fps_frames++; struct timespec t; @@ -695,6 +767,7 @@ out: release_capture(); free(rgb_half); free(rgb_full); + free(raw_cached); pthread_mutex_lock(&eng.lock); eng.running = 0; /* fail any waiter: stream clients see running==0, a pending snapshot @@ -807,6 +880,8 @@ void cam_engine_init(void) if (l >= 0 && l <= 1023) eng.lamp_level = l; } + if (getenv("FORGECTRL_NO_VPU")) + vpu_disabled = 1; } void cam_engine_shutdown(void) @@ -822,6 +897,10 @@ void cam_engine_shutdown(void) eng.tid_valid = 0; pthread_mutex_unlock(&eng.lock); } + if (vpu) { + vpu_jpeg_close(vpu); + vpu = NULL; + } pthread_mutex_unlock(&eng.ctl); } @@ -970,5 +1049,6 @@ void cam_get_status(struct cam_status *st) st->clients = eng.clients; st->seq = eng.seq; st->fps = eng.fps; + st->vpu = eng.vpu_active; pthread_mutex_unlock(&eng.lock); } diff --git a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/cam.h b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/cam.h index 6458c96..be7ad25 100644 --- a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/cam.h +++ b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/cam.h @@ -57,6 +57,7 @@ struct cam_status { int clients; uint64_t seq; double fps; + int vpu; /* stream frames are VPU-encoded */ }; void cam_get_status(struct cam_status *st); diff --git a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/debayer.c b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/debayer.c index e54e464..095ace6 100644 --- a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/debayer.c +++ b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/debayer.c @@ -7,6 +7,8 @@ * even rows: B G B G ... * odd rows: G R G R ... */ +#include + #include "debayer.h" static inline int clampi(int v, int lo, int hi) @@ -62,6 +64,51 @@ void debayer_bggr_bilinear(const uint8_t *raw, uint8_t *rgb, } } +void debayer_bggr_half_yuv420(const uint8_t *raw, int w, int h, int hflip, + uint8_t *yp, int y_stride, + uint8_t *up, uint8_t *vp, int uv_stride) +{ + const int ow = w / 2; /* luma dimensions */ + const int oh = h / 2; + const int uvw = ow / 2; + + /* JFIF full-range ITU-R 601, x256 fixed point: + * Y = 0.299 R + 0.587 G + 0.114 B -> 77 150 29 + * Cb = -0.169 R - 0.331 G + 0.500 B + 128 -> -43 -85 128 + * Cr = 0.500 R - 0.419 G - 0.081 B + 128 -> 128 -107 -21 */ + for (int y2 = 0; y2 < oh / 2; y2++) { + uint8_t *yrow0 = yp + (size_t)(2 * y2) * y_stride; + uint8_t *yrow1 = yrow0 + y_stride; + uint8_t *urow = up + (size_t)y2 * uv_stride; + uint8_t *vrow = vp + (size_t)y2 * uv_stride; + for (int x2 = 0; x2 < uvw; x2++) { + int rs = 0, gs = 0, bs = 0; + for (int sy = 0; sy < 2; sy++) { + const int row = 2 * y2 + sy; + const uint8_t *quad_row = raw + (size_t)(2 * row) * w; + uint8_t *yrow = sy ? yrow1 : yrow0; + for (int sx = 0; sx < 2; sx++) { + const int col = 2 * x2 + sx; + const uint8_t *q = quad_row + 2 * col; + const int b = q[0]; + const int g = (q[1] + q[w] + 1) >> 1; + const int r = q[w + 1]; + rs += r; + gs += g; + bs += b; + yrow[hflip ? ow - 1 - col : col] = + (uint8_t)((77 * r + 150 * g + 29 * b + 128) >> 8); + } + } + const int cx = hflip ? uvw - 1 - x2 : x2; + int cb = ((-43 * rs - 85 * gs + 128 * bs + 512) >> 10) + 128; + int cr = ((128 * rs - 107 * gs - 21 * bs + 512) >> 10) + 128; + urow[cx] = (uint8_t)(cb < 0 ? 0 : (cb > 255 ? 255 : cb)); + vrow[cx] = (uint8_t)(cr < 0 ? 0 : (cr > 255 ? 255 : cr)); + } + } +} + void debayer_bggr_half(const uint8_t *raw, uint8_t *rgb, int w, int h, int hflip) { diff --git a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/debayer.h b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/debayer.h index b2e13b2..a6f841c 100644 --- a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/debayer.h +++ b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/debayer.h @@ -21,4 +21,12 @@ void debayer_bggr_bilinear(const uint8_t *raw, uint8_t *rgb, void debayer_bggr_half(const uint8_t *raw, uint8_t *rgb, int w, int h, int hflip); +/* Half-resolution demosaic straight to planar YUV420 (JFIF full-range, + * ITU-R 601) for the VPU JPEG encoder: luma per 2x2 BGGR quad at + * (w/2)x(h/2), chroma averaged per 2x2 luma block at (w/4)x(h/4). + * w/2 and h/2 must be even. Strides are in bytes. */ +void debayer_bggr_half_yuv420(const uint8_t *raw, int w, int h, int hflip, + uint8_t *yp, int y_stride, + uint8_t *up, uint8_t *vp, int uv_stride); + #endif diff --git a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/main.c b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/main.c index 19823c0..e155a02 100644 --- a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/main.c +++ b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/main.c @@ -210,11 +210,12 @@ static int cb_status(const struct _u_request *req, struct _u_response *res, char body[256]; snprintf(body, sizeof(body), "{\"running\":%s,\"cam\":\"%s\",\"clients\":%d," - "\"frames\":%llu,\"fps\":%.1f," + "\"frames\":%llu,\"fps\":%.1f,\"encoder\":\"%s\"," "\"stream\":{\"width\":1296,\"height\":972}," "\"snapshot\":{\"width\":2592,\"height\":1944}}", st.running ? "true" : "false", cam_name(st.cam), st.clients, - (unsigned long long)st.seq, st.fps); + (unsigned long long)st.seq, st.fps, + st.vpu ? "vpu" : "software"); ulfius_set_string_body_response(res, 200, body); ulfius_add_header_to_response(res, "Content-Type", "application/json"); return U_CALLBACK_CONTINUE; diff --git a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/vpu_jpeg.c b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/vpu_jpeg.c new file mode 100644 index 0000000..f3bf318 --- /dev/null +++ b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/vpu_jpeg.c @@ -0,0 +1,218 @@ +/* + * vpu_jpeg.c - hardware JPEG encoding on the i.MX6 CODA960 VPU + * Copyright (c) 2026 Scott Wiederhold + * SPDX-License-Identifier: MIT + * + * V4L2 mem2mem, single-planar API against the mainline coda driver: one + * MMAP buffer on each queue, synchronous QBUF/DQBUF per frame. The node + * is found by personality (driver "coda", JPEG on the capture side, + * YUV420 accepted on the output side), never by number - coda registers + * four nodes and the numbering depends on probe order. + */ +#include "vpu_jpeg.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#define ENCODE_TIMEOUT_MS 1000 + +struct vpu_jpeg { + int fd; + int w, h; + int bpl; /* OUTPUT luma stride from S_FMT */ + uint8_t *out; /* mapped OUTPUT (YUV420) buffer */ + size_t out_size; + uint8_t *cap; /* mapped CAPTURE (JPEG) buffer */ + size_t cap_size; +}; + +static int xioctl(int fd, unsigned long req, void *arg) +{ + int r; + do { + r = ioctl(fd, req, arg); + } while (r == -1 && errno == EINTR); + return r; +} + +/* Is this node the coda JPEG encoder? (JPEG capture, YUV420 output) */ +static int is_jpeg_encoder(int fd) +{ + struct v4l2_capability cap = {0}; + if (xioctl(fd, VIDIOC_QUERYCAP, &cap) < 0 || + strcmp((const char *)cap.driver, "coda") != 0 || + !(cap.device_caps & V4L2_CAP_VIDEO_M2M)) + return 0; + + struct v4l2_fmtdesc fd0 = { .type = V4L2_BUF_TYPE_VIDEO_CAPTURE }; + if (xioctl(fd, VIDIOC_ENUM_FMT, &fd0) < 0 || + fd0.pixelformat != V4L2_PIX_FMT_JPEG) + return 0; + + for (unsigned i = 0; ; i++) { + struct v4l2_fmtdesc fo = { .type = V4L2_BUF_TYPE_VIDEO_OUTPUT, + .index = i }; + if (xioctl(fd, VIDIOC_ENUM_FMT, &fo) < 0) + return 0; + if (fo.pixelformat == V4L2_PIX_FMT_YUV420) + return 1; + } +} + +static int find_encoder(void) +{ + for (int i = 0; i < 32; i++) { + char path[32]; + snprintf(path, sizeof(path), "/dev/video%d", i); + int fd = open(path, O_RDWR | O_NONBLOCK, 0); + if (fd < 0) + continue; + if (is_jpeg_encoder(fd)) + return fd; + close(fd); + } + return -1; +} + +static int map_one(int fd, enum v4l2_buf_type type, uint8_t **mem, + size_t *size) +{ + struct v4l2_requestbuffers req = { .count = 1, .type = type, + .memory = V4L2_MEMORY_MMAP }; + if (xioctl(fd, VIDIOC_REQBUFS, &req) < 0 || req.count < 1) + return -1; + struct v4l2_buffer buf = { .type = type, .memory = V4L2_MEMORY_MMAP, + .index = 0 }; + if (xioctl(fd, VIDIOC_QUERYBUF, &buf) < 0) + return -1; + *mem = mmap(NULL, buf.length, PROT_READ | PROT_WRITE, MAP_SHARED, + fd, buf.m.offset); + if (*mem == MAP_FAILED) { + *mem = NULL; + return -1; + } + *size = buf.length; + return 0; +} + +vpu_jpeg_t *vpu_jpeg_open(int w, int h, int quality) +{ + vpu_jpeg_t *v = calloc(1, sizeof(*v)); + if (!v) + return NULL; + v->fd = find_encoder(); + if (v->fd < 0) + goto fail; + + struct v4l2_format fo = { .type = V4L2_BUF_TYPE_VIDEO_OUTPUT }; + fo.fmt.pix.width = (unsigned)w; + fo.fmt.pix.height = (unsigned)h; + fo.fmt.pix.pixelformat = V4L2_PIX_FMT_YUV420; + fo.fmt.pix.field = V4L2_FIELD_NONE; + if (xioctl(v->fd, VIDIOC_S_FMT, &fo) < 0 || + fo.fmt.pix.width != (unsigned)w || + fo.fmt.pix.height != (unsigned)h) { + fprintf(stderr, "vpu: S_FMT output rejected %dx%d\n", w, h); + goto fail; + } + v->w = w; + v->h = h; + v->bpl = (int)fo.fmt.pix.bytesperline; + + struct v4l2_format fc = { .type = V4L2_BUF_TYPE_VIDEO_CAPTURE }; + fc.fmt.pix.width = (unsigned)w; + fc.fmt.pix.height = (unsigned)h; + fc.fmt.pix.pixelformat = V4L2_PIX_FMT_JPEG; + if (xioctl(v->fd, VIDIOC_S_FMT, &fc) < 0) + goto fail; + + struct v4l2_control q = { .id = V4L2_CID_JPEG_COMPRESSION_QUALITY, + .value = quality }; + xioctl(v->fd, VIDIOC_S_CTRL, &q); /* best effort */ + + if (map_one(v->fd, V4L2_BUF_TYPE_VIDEO_OUTPUT, &v->out, &v->out_size) || + map_one(v->fd, V4L2_BUF_TYPE_VIDEO_CAPTURE, &v->cap, &v->cap_size)) + goto fail; + + enum v4l2_buf_type t = V4L2_BUF_TYPE_VIDEO_OUTPUT; + if (xioctl(v->fd, VIDIOC_STREAMON, &t) < 0) + goto fail; + t = V4L2_BUF_TYPE_VIDEO_CAPTURE; + if (xioctl(v->fd, VIDIOC_STREAMON, &t) < 0) + goto fail; + return v; + +fail: + vpu_jpeg_close(v); + return NULL; +} + +void vpu_jpeg_planes(vpu_jpeg_t *v, uint8_t **y, uint8_t **u, uint8_t **vv, + int *y_stride, int *uv_stride) +{ + *y = v->out; + *u = v->out + (size_t)v->bpl * v->h; + *vv = *u + (size_t)(v->bpl / 2) * (v->h / 2); + *y_stride = v->bpl; + *uv_stride = v->bpl / 2; +} + +int vpu_jpeg_encode(vpu_jpeg_t *v, uint8_t **jpeg, size_t *len) +{ + struct v4l2_buffer cb = { .type = V4L2_BUF_TYPE_VIDEO_CAPTURE, + .memory = V4L2_MEMORY_MMAP, .index = 0 }; + struct v4l2_buffer ob = { .type = V4L2_BUF_TYPE_VIDEO_OUTPUT, + .memory = V4L2_MEMORY_MMAP, .index = 0 }; + ob.bytesused = (unsigned)v->out_size; + + if (xioctl(v->fd, VIDIOC_QBUF, &cb) < 0 || + xioctl(v->fd, VIDIOC_QBUF, &ob) < 0) + return -1; + + struct pollfd pfd = { .fd = v->fd, .events = POLLIN }; + int pr; + do { + pr = poll(&pfd, 1, ENCODE_TIMEOUT_MS); + } while (pr == -1 && errno == EINTR); + if (pr <= 0) + return -1; + + if (xioctl(v->fd, VIDIOC_DQBUF, &cb) < 0) + return -1; + xioctl(v->fd, VIDIOC_DQBUF, &ob); + + uint8_t *out = malloc(cb.bytesused); + if (!out) + return -1; + memcpy(out, v->cap, cb.bytesused); + *jpeg = out; + *len = cb.bytesused; + return 0; +} + +void vpu_jpeg_close(vpu_jpeg_t *v) +{ + if (!v) + return; + if (v->fd >= 0) { + enum v4l2_buf_type t = V4L2_BUF_TYPE_VIDEO_OUTPUT; + xioctl(v->fd, VIDIOC_STREAMOFF, &t); + t = V4L2_BUF_TYPE_VIDEO_CAPTURE; + xioctl(v->fd, VIDIOC_STREAMOFF, &t); + } + if (v->out) + munmap(v->out, v->out_size); + if (v->cap) + munmap(v->cap, v->cap_size); + if (v->fd >= 0) + close(v->fd); + free(v); +} diff --git a/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/vpu_jpeg.h b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/vpu_jpeg.h new file mode 100644 index 0000000..673f19a --- /dev/null +++ b/meta-forgefirm/recipes-forgefirm/forgectrl/forgectrl/vpu_jpeg.h @@ -0,0 +1,33 @@ +/* + * vpu_jpeg.h - hardware JPEG encoding on the i.MX6 CODA960 VPU + * Copyright (c) 2026 Scott Wiederhold + * SPDX-License-Identifier: MIT + * + * Thin wrapper around the mainline coda V4L2 mem2mem JPEG encoder: the + * caller writes planar YUV420 directly into the encoder's OUTPUT buffer + * (vpu_jpeg_planes) and gets back a malloc'd JFIF JPEG. + */ +#ifndef FORGECTRL_VPU_JPEG_H +#define FORGECTRL_VPU_JPEG_H + +#include +#include + +typedef struct vpu_jpeg vpu_jpeg_t; + +/* Locate the CODA JPEG encoder video node, configure it for w x h YUV420 + * -> JPEG at the given quality (5..100), and map one buffer per queue. + * Returns NULL if no encoder exists or setup fails. */ +vpu_jpeg_t *vpu_jpeg_open(int w, int h, int quality); + +/* Planes of the mapped OUTPUT buffer for direct fill. */ +void vpu_jpeg_planes(vpu_jpeg_t *v, uint8_t **y, uint8_t **u, uint8_t **vv, + int *y_stride, int *uv_stride); + +/* Encode the currently-filled OUTPUT buffer. On success *jpeg is malloc'd + * (caller frees) and 0 is returned. */ +int vpu_jpeg_encode(vpu_jpeg_t *v, uint8_t **jpeg, size_t *len); + +void vpu_jpeg_close(vpu_jpeg_t *v); + +#endif