mirror of
https://github.com/openglow-org/forgefirm.git
synced 2026-09-28 01:01:12 -07:00
forgectrl: VPU JPEG encode for the camera stream - 7.9 fps
Demosaic the superpixels straight to planar YUV420 and encode on the CODA960 (mainline coda V4L2 mem2mem, node found by personality); libjpeg stays as the automatic fallback and the snapshot path. All camera paths now demosaic from a cached bounce copy of the frame: the V4L2 MMAP capture buffers are uncached, and reading them in-place costs ~340 ms/frame vs 43 ms memcpy + 75 ms cached convert. Per-frame stats logged every 100 frames; /cam/status reports the encoder.
This commit is contained in:
+23
-11
@@ -156,7 +156,8 @@ The cameras share the hardware video-mux; the NEWEST request wins it
|
|||||||
The per-camera lamp (`pic/lid_led` / `head/white_led`) is raised to
|
The per-camera lamp (`pic/lid_led` / `head/white_led`) is raised to
|
||||||
`FORGECTRL_LAMP` (default 132) while capturing and restored on idle.
|
`FORGECTRL_LAMP` (default 132) while capturing and restored on idle.
|
||||||
|
|
||||||
Bench (2026-08-03, on the board): stream 3.2 fps sustained at 1296×972;
|
Bench (2026-08-03, on the board): stream **7.9 fps** sustained at
|
||||||
|
1296×972 (VPU encode; 3.2 fps on the software fallback);
|
||||||
full-res snapshot 2.4 s warm / 2.7 s cold (cold includes the pipeline
|
full-res snapshot 2.4 s warm / 2.7 s cold (cold includes the pipeline
|
||||||
bring-up); two parallel same-camera clients share the frame rate; idle
|
bring-up); two parallel same-camera clients share the frame rate; idle
|
||||||
teardown observed. Borrow verified: head snapshot 200 during a lid
|
teardown observed. Borrow verified: head snapshot 200 during a lid
|
||||||
@@ -174,16 +175,27 @@ core). Run by hand: `/usr/bin/forgectrl >> /data/forgectrl.log 2>&1 &`
|
|||||||
`/?action=stream` alias) while jogging the machine from the same
|
`/?action=stream` alias) while jogging the machine from the same
|
||||||
LightBurn session.
|
LightBurn session.
|
||||||
|
|
||||||
Frame-rate ceiling and the offload path: 3.2 fps is CPU-bound in the
|
**VPU JPEG offload: DONE 2026-08-03, bench-verified — 7.9 fps** (2.5×
|
||||||
JPEG encode (single A9, libjpeg-turbo NEON). The hardware answer is the
|
the software rate). The stream path demosaics the 2×2 superpixels
|
||||||
**CODA960 VPU JPEG encoder — already probed with firmware on the image,
|
straight to planar YUV420 (JFIF full-range 601) and the **CODA960 VPU
|
||||||
registered at /dev/video0** (V4L2 mem2mem, YUV input): demosaic the 2×2
|
JPEG encoder** (mainline coda, V4L2 mem2mem; found by personality, not
|
||||||
superpixels straight to YUV420 on the CPU (cheap) and let the VPU
|
node number) does the encode: per-frame **copy 43 ms + convert 75 ms +
|
||||||
encode → est. 8–15 fps, likely CSI/memory-bound. The IPU cannot help
|
encode 7 ms**. Two hard-won facts:
|
||||||
with demosaic (its IC is CSC/scale only, YUV/RGB in — that is the
|
- **V4L2 MMAP capture buffers are uncached** — demosaicing in-place out
|
||||||
`imx-csc-scaler` at /dev/video8, useful only for a future full-res
|
of one costs ~340 ms/frame at this resolution; one bulk memcpy into a
|
||||||
stream). Contained follow-up in forgectrl cam.c; keep the libjpeg path
|
cached bounce buffer first (43 ms) makes the same demosaic run in
|
||||||
as fallback. Not yet done: lens calibration / bed alignment (the
|
75 ms. All camera paths (stream, snapshot, borrow) read from the
|
||||||
|
bounce copy.
|
||||||
|
- The VPU encoder accepts 1296×972 exactly (no MCU-alignment padding
|
||||||
|
needed) with quality via V4L2_CID_JPEG_COMPRESSION_QUALITY.
|
||||||
|
libjpeg remains the automatic fallback (`FORGECTRL_NO_VPU=1` forces
|
||||||
|
it) and the snapshot path; `/cam/status` reports `"encoder"`.
|
||||||
|
Remaining headroom: the scalar convert dominates — NEON would push
|
||||||
|
toward the sensor/CSI limit. If motion contention ever shows clamps
|
||||||
|
during streaming, a stream-fps cap knob is the easy relief valve.
|
||||||
|
The IPU cannot help with demosaic (its IC is CSC/scale only — the
|
||||||
|
`imx-csc-scaler` at /dev/video8 matters only for a future full-res
|
||||||
|
stream). Not yet done: lens calibration / bed alignment (the
|
||||||
fisheye needs LightBurn's camera calibration pass), and the deferred
|
fisheye needs LightBurn's camera calibration pass), and the deferred
|
||||||
5.6 emulator homing-image smoke (the cloud emulator can now be pointed
|
5.6 emulator homing-image smoke (the cloud emulator can now be pointed
|
||||||
at live snapshots).
|
at live snapshots).
|
||||||
|
|||||||
@@ -11,6 +11,8 @@ SRC_URI = "\
|
|||||||
file://cam.h \
|
file://cam.h \
|
||||||
file://debayer.c \
|
file://debayer.c \
|
||||||
file://debayer.h \
|
file://debayer.h \
|
||||||
|
file://vpu_jpeg.c \
|
||||||
|
file://vpu_jpeg.h \
|
||||||
file://forgectrl.init \
|
file://forgectrl.init \
|
||||||
"
|
"
|
||||||
|
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ project(forgectrl C)
|
|||||||
|
|
||||||
set(CMAKE_C_STANDARD 11)
|
set(CMAKE_C_STANDARD 11)
|
||||||
|
|
||||||
add_executable(forgectrl main.c cam.c debayer.c)
|
add_executable(forgectrl main.c cam.c debayer.c vpu_jpeg.c)
|
||||||
target_compile_options(forgectrl PRIVATE -Wall -Wextra -O2)
|
target_compile_options(forgectrl PRIVATE -Wall -Wextra -O2)
|
||||||
target_link_libraries(forgectrl ulfius jpeg pthread m)
|
target_link_libraries(forgectrl ulfius jpeg pthread m)
|
||||||
|
|
||||||
|
|||||||
@@ -19,6 +19,7 @@
|
|||||||
#define _GNU_SOURCE
|
#define _GNU_SOURCE
|
||||||
#include "cam.h"
|
#include "cam.h"
|
||||||
#include "debayer.h"
|
#include "debayer.h"
|
||||||
|
#include "vpu_jpeg.h"
|
||||||
|
|
||||||
#include <errno.h>
|
#include <errno.h>
|
||||||
#include <fcntl.h>
|
#include <fcntl.h>
|
||||||
@@ -123,6 +124,7 @@ static struct {
|
|||||||
/* config */
|
/* config */
|
||||||
int stream_quality;
|
int stream_quality;
|
||||||
int lamp_level;
|
int lamp_level;
|
||||||
|
int vpu_active; /* last stream frame went through the VPU */
|
||||||
} eng = {
|
} eng = {
|
||||||
.ctl = PTHREAD_MUTEX_INITIALIZER,
|
.ctl = PTHREAD_MUTEX_INITIALIZER,
|
||||||
.lock = PTHREAD_MUTEX_INITIALIZER,
|
.lock = PTHREAD_MUTEX_INITIALIZER,
|
||||||
@@ -139,6 +141,11 @@ const char *cam_name(cam_id_t cam)
|
|||||||
return camdefs[cam].name;
|
return camdefs[cam].name;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* CODA960 hardware JPEG encoder for the stream path; libjpeg remains the
|
||||||
|
* fallback (and the snapshot path). Worker-thread use only. */
|
||||||
|
static vpu_jpeg_t *vpu;
|
||||||
|
static int vpu_disabled;
|
||||||
|
|
||||||
/* ------------------------------------------------------------------ util */
|
/* ------------------------------------------------------------------ util */
|
||||||
|
|
||||||
static void now_ts(struct timespec *ts)
|
static void now_ts(struct timespec *ts)
|
||||||
@@ -539,7 +546,8 @@ static void fail_snap(void)
|
|||||||
|
|
||||||
/* Capture one frame from the currently-started pipeline and feed it to
|
/* Capture one frame from the currently-started pipeline and feed it to
|
||||||
* deliver_snap. Used by the borrow path. */
|
* deliver_snap. Used by the borrow path. */
|
||||||
static int grab_one_snap(uint8_t *rgb_half, uint8_t **prgb_full)
|
static int grab_one_snap(uint8_t *raw_cached, uint8_t *rgb_half,
|
||||||
|
uint8_t **prgb_full)
|
||||||
{
|
{
|
||||||
for (int tries = 0; tries < MAX_DQ_TIMEOUTS; tries++) {
|
for (int tries = 0; tries < MAX_DQ_TIMEOUTS; tries++) {
|
||||||
fd_set fds;
|
fd_set fds;
|
||||||
@@ -561,7 +569,9 @@ static int grab_one_snap(uint8_t *rgb_half, uint8_t **prgb_full)
|
|||||||
continue;
|
continue;
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
deliver_snap(eng.bufs[buf.index].start, rgb_half, prgb_full);
|
memcpy(raw_cached, eng.bufs[buf.index].start,
|
||||||
|
(size_t)CAM_W * CAM_H);
|
||||||
|
deliver_snap(raw_cached, rgb_half, prgb_full);
|
||||||
xioctl(eng.fd, VIDIOC_QBUF, &buf);
|
xioctl(eng.fd, VIDIOC_QBUF, &buf);
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
@@ -573,12 +583,20 @@ static void *worker(void *arg)
|
|||||||
(void)arg;
|
(void)arg;
|
||||||
uint8_t *rgb_half = malloc((size_t)HALF_W * HALF_H * 3);
|
uint8_t *rgb_half = malloc((size_t)HALF_W * HALF_H * 3);
|
||||||
uint8_t *rgb_full = NULL; /* allocated on first full-res snapshot */
|
uint8_t *rgb_full = NULL; /* allocated on first full-res snapshot */
|
||||||
|
/* The V4L2 MMAP capture buffers are DMA-coherent = UNCACHED: byte
|
||||||
|
* reads from them cost a bus transaction each and demosaicing
|
||||||
|
* straight out of one measures ~340 ms/frame. One bulk memcpy into
|
||||||
|
* this cached bounce buffer first makes the demosaic run at cached
|
||||||
|
* speed. */
|
||||||
|
uint8_t *raw_cached = malloc((size_t)CAM_W * CAM_H);
|
||||||
|
double stat_copy_ms = 0, stat_conv_ms = 0, stat_enc_ms = 0;
|
||||||
|
unsigned stat_n = 0;
|
||||||
int dq_timeouts = 0;
|
int dq_timeouts = 0;
|
||||||
struct timespec fps_t0;
|
struct timespec fps_t0;
|
||||||
now_ts(&fps_t0);
|
now_ts(&fps_t0);
|
||||||
uint64_t fps_frames = 0;
|
uint64_t fps_frames = 0;
|
||||||
|
|
||||||
if (!rgb_half) {
|
if (!rgb_half || !raw_cached) {
|
||||||
fprintf(stderr, "cam: worker OOM\n");
|
fprintf(stderr, "cam: worker OOM\n");
|
||||||
goto out;
|
goto out;
|
||||||
}
|
}
|
||||||
@@ -607,7 +625,7 @@ static void *worker(void *arg)
|
|||||||
char berr[128];
|
char berr[128];
|
||||||
release_capture();
|
release_capture();
|
||||||
if (start_capture(borrow_cam, berr, sizeof(berr)) == 0) {
|
if (start_capture(borrow_cam, berr, sizeof(berr)) == 0) {
|
||||||
if (grab_one_snap(rgb_half, &rgb_full))
|
if (grab_one_snap(raw_cached, rgb_half, &rgb_full))
|
||||||
fail_snap();
|
fail_snap();
|
||||||
release_capture();
|
release_capture();
|
||||||
} else {
|
} else {
|
||||||
@@ -653,23 +671,77 @@ static void *worker(void *arg)
|
|||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
dq_timeouts = 0;
|
dq_timeouts = 0;
|
||||||
const uint8_t *raw = eng.bufs[buf.index].start;
|
struct timespec c0, c1;
|
||||||
|
now_ts(&c0);
|
||||||
|
memcpy(raw_cached, eng.bufs[buf.index].start,
|
||||||
|
(size_t)CAM_W * CAM_H);
|
||||||
|
now_ts(&c1);
|
||||||
|
const uint8_t *raw = raw_cached;
|
||||||
|
|
||||||
/* Snapshot request rides on the same raw frame */
|
/* Snapshot request rides on the same raw frame */
|
||||||
if (snap)
|
if (snap)
|
||||||
deliver_snap(raw, rgb_half, &rgb_full);
|
deliver_snap(raw, rgb_half, &rgb_full);
|
||||||
|
|
||||||
/* Stream frame */
|
/* Stream frame: demosaic + encode, VPU first, libjpeg fallback */
|
||||||
if (clients > 0) {
|
if (clients > 0) {
|
||||||
uint8_t *jpg = NULL;
|
uint8_t *jpg = NULL;
|
||||||
size_t len = 0;
|
size_t len = 0;
|
||||||
|
int via_vpu = 0;
|
||||||
|
struct timespec e0, e1, e2;
|
||||||
|
now_ts(&e0);
|
||||||
|
|
||||||
|
if (!vpu_disabled && !vpu) {
|
||||||
|
vpu = vpu_jpeg_open(HALF_W, HALF_H, eng.stream_quality);
|
||||||
|
if (!vpu) {
|
||||||
|
vpu_disabled = 1;
|
||||||
|
fprintf(stderr, "cam: no VPU JPEG encoder, "
|
||||||
|
"using software encode\n");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (vpu) {
|
||||||
|
uint8_t *yp, *up, *vp;
|
||||||
|
int ys, uvs;
|
||||||
|
vpu_jpeg_planes(vpu, &yp, &up, &vp, &ys, &uvs);
|
||||||
|
debayer_bggr_half_yuv420(raw, CAM_W, CAM_H, HFLIP,
|
||||||
|
yp, ys, up, vp, uvs);
|
||||||
|
now_ts(&e1);
|
||||||
|
if (vpu_jpeg_encode(vpu, &jpg, &len) == 0) {
|
||||||
|
via_vpu = 1;
|
||||||
|
} else {
|
||||||
|
fprintf(stderr, "cam: VPU encode failed, "
|
||||||
|
"falling back to software\n");
|
||||||
|
vpu_jpeg_close(vpu);
|
||||||
|
vpu = NULL;
|
||||||
|
vpu_disabled = 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!via_vpu) {
|
||||||
debayer_bggr_half(raw, rgb_half, CAM_W, CAM_H, HFLIP);
|
debayer_bggr_half(raw, rgb_half, CAM_W, CAM_H, HFLIP);
|
||||||
if (jpeg_encode_rgb(rgb_half, HALF_W, HALF_H, eng.stream_quality,
|
now_ts(&e1);
|
||||||
1, &jpg, &len) == 0) {
|
if (jpeg_encode_rgb(rgb_half, HALF_W, HALF_H,
|
||||||
|
eng.stream_quality, 1, &jpg, &len))
|
||||||
|
jpg = NULL;
|
||||||
|
}
|
||||||
|
now_ts(&e2);
|
||||||
|
|
||||||
|
if (jpg) {
|
||||||
|
stat_copy_ms += ts_diff(&c1, &c0) * 1e3;
|
||||||
|
stat_conv_ms += ts_diff(&e1, &e0) * 1e3;
|
||||||
|
stat_enc_ms += ts_diff(&e2, &e1) * 1e3;
|
||||||
|
if (++stat_n >= 100) {
|
||||||
|
fprintf(stderr, "cam: stream stats: copy %.0f ms, "
|
||||||
|
"convert %.0f ms, encode %.0f ms avg (%s)\n",
|
||||||
|
stat_copy_ms / stat_n, stat_conv_ms / stat_n,
|
||||||
|
stat_enc_ms / stat_n,
|
||||||
|
via_vpu ? "vpu" : "software");
|
||||||
|
stat_copy_ms = stat_conv_ms = stat_enc_ms = 0;
|
||||||
|
stat_n = 0;
|
||||||
|
}
|
||||||
pthread_mutex_lock(&eng.lock);
|
pthread_mutex_lock(&eng.lock);
|
||||||
free(eng.stream_jpg);
|
free(eng.stream_jpg);
|
||||||
eng.stream_jpg = jpg;
|
eng.stream_jpg = jpg;
|
||||||
eng.stream_len = len;
|
eng.stream_len = len;
|
||||||
|
eng.vpu_active = via_vpu;
|
||||||
eng.seq++;
|
eng.seq++;
|
||||||
fps_frames++;
|
fps_frames++;
|
||||||
struct timespec t;
|
struct timespec t;
|
||||||
@@ -695,6 +767,7 @@ out:
|
|||||||
release_capture();
|
release_capture();
|
||||||
free(rgb_half);
|
free(rgb_half);
|
||||||
free(rgb_full);
|
free(rgb_full);
|
||||||
|
free(raw_cached);
|
||||||
pthread_mutex_lock(&eng.lock);
|
pthread_mutex_lock(&eng.lock);
|
||||||
eng.running = 0;
|
eng.running = 0;
|
||||||
/* fail any waiter: stream clients see running==0, a pending snapshot
|
/* fail any waiter: stream clients see running==0, a pending snapshot
|
||||||
@@ -807,6 +880,8 @@ void cam_engine_init(void)
|
|||||||
if (l >= 0 && l <= 1023)
|
if (l >= 0 && l <= 1023)
|
||||||
eng.lamp_level = l;
|
eng.lamp_level = l;
|
||||||
}
|
}
|
||||||
|
if (getenv("FORGECTRL_NO_VPU"))
|
||||||
|
vpu_disabled = 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
void cam_engine_shutdown(void)
|
void cam_engine_shutdown(void)
|
||||||
@@ -822,6 +897,10 @@ void cam_engine_shutdown(void)
|
|||||||
eng.tid_valid = 0;
|
eng.tid_valid = 0;
|
||||||
pthread_mutex_unlock(&eng.lock);
|
pthread_mutex_unlock(&eng.lock);
|
||||||
}
|
}
|
||||||
|
if (vpu) {
|
||||||
|
vpu_jpeg_close(vpu);
|
||||||
|
vpu = NULL;
|
||||||
|
}
|
||||||
pthread_mutex_unlock(&eng.ctl);
|
pthread_mutex_unlock(&eng.ctl);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -970,5 +1049,6 @@ void cam_get_status(struct cam_status *st)
|
|||||||
st->clients = eng.clients;
|
st->clients = eng.clients;
|
||||||
st->seq = eng.seq;
|
st->seq = eng.seq;
|
||||||
st->fps = eng.fps;
|
st->fps = eng.fps;
|
||||||
|
st->vpu = eng.vpu_active;
|
||||||
pthread_mutex_unlock(&eng.lock);
|
pthread_mutex_unlock(&eng.lock);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -57,6 +57,7 @@ struct cam_status {
|
|||||||
int clients;
|
int clients;
|
||||||
uint64_t seq;
|
uint64_t seq;
|
||||||
double fps;
|
double fps;
|
||||||
|
int vpu; /* stream frames are VPU-encoded */
|
||||||
};
|
};
|
||||||
void cam_get_status(struct cam_status *st);
|
void cam_get_status(struct cam_status *st);
|
||||||
|
|
||||||
|
|||||||
@@ -7,6 +7,8 @@
|
|||||||
* even rows: B G B G ...
|
* even rows: B G B G ...
|
||||||
* odd rows: G R G R ...
|
* odd rows: G R G R ...
|
||||||
*/
|
*/
|
||||||
|
#include <stddef.h>
|
||||||
|
|
||||||
#include "debayer.h"
|
#include "debayer.h"
|
||||||
|
|
||||||
static inline int clampi(int v, int lo, int hi)
|
static inline int clampi(int v, int lo, int hi)
|
||||||
@@ -62,6 +64,51 @@ void debayer_bggr_bilinear(const uint8_t *raw, uint8_t *rgb,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void debayer_bggr_half_yuv420(const uint8_t *raw, int w, int h, int hflip,
|
||||||
|
uint8_t *yp, int y_stride,
|
||||||
|
uint8_t *up, uint8_t *vp, int uv_stride)
|
||||||
|
{
|
||||||
|
const int ow = w / 2; /* luma dimensions */
|
||||||
|
const int oh = h / 2;
|
||||||
|
const int uvw = ow / 2;
|
||||||
|
|
||||||
|
/* JFIF full-range ITU-R 601, x256 fixed point:
|
||||||
|
* Y = 0.299 R + 0.587 G + 0.114 B -> 77 150 29
|
||||||
|
* Cb = -0.169 R - 0.331 G + 0.500 B + 128 -> -43 -85 128
|
||||||
|
* Cr = 0.500 R - 0.419 G - 0.081 B + 128 -> 128 -107 -21 */
|
||||||
|
for (int y2 = 0; y2 < oh / 2; y2++) {
|
||||||
|
uint8_t *yrow0 = yp + (size_t)(2 * y2) * y_stride;
|
||||||
|
uint8_t *yrow1 = yrow0 + y_stride;
|
||||||
|
uint8_t *urow = up + (size_t)y2 * uv_stride;
|
||||||
|
uint8_t *vrow = vp + (size_t)y2 * uv_stride;
|
||||||
|
for (int x2 = 0; x2 < uvw; x2++) {
|
||||||
|
int rs = 0, gs = 0, bs = 0;
|
||||||
|
for (int sy = 0; sy < 2; sy++) {
|
||||||
|
const int row = 2 * y2 + sy;
|
||||||
|
const uint8_t *quad_row = raw + (size_t)(2 * row) * w;
|
||||||
|
uint8_t *yrow = sy ? yrow1 : yrow0;
|
||||||
|
for (int sx = 0; sx < 2; sx++) {
|
||||||
|
const int col = 2 * x2 + sx;
|
||||||
|
const uint8_t *q = quad_row + 2 * col;
|
||||||
|
const int b = q[0];
|
||||||
|
const int g = (q[1] + q[w] + 1) >> 1;
|
||||||
|
const int r = q[w + 1];
|
||||||
|
rs += r;
|
||||||
|
gs += g;
|
||||||
|
bs += b;
|
||||||
|
yrow[hflip ? ow - 1 - col : col] =
|
||||||
|
(uint8_t)((77 * r + 150 * g + 29 * b + 128) >> 8);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const int cx = hflip ? uvw - 1 - x2 : x2;
|
||||||
|
int cb = ((-43 * rs - 85 * gs + 128 * bs + 512) >> 10) + 128;
|
||||||
|
int cr = ((128 * rs - 107 * gs - 21 * bs + 512) >> 10) + 128;
|
||||||
|
urow[cx] = (uint8_t)(cb < 0 ? 0 : (cb > 255 ? 255 : cb));
|
||||||
|
vrow[cx] = (uint8_t)(cr < 0 ? 0 : (cr > 255 ? 255 : cr));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
void debayer_bggr_half(const uint8_t *raw, uint8_t *rgb,
|
void debayer_bggr_half(const uint8_t *raw, uint8_t *rgb,
|
||||||
int w, int h, int hflip)
|
int w, int h, int hflip)
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -21,4 +21,12 @@ void debayer_bggr_bilinear(const uint8_t *raw, uint8_t *rgb,
|
|||||||
void debayer_bggr_half(const uint8_t *raw, uint8_t *rgb,
|
void debayer_bggr_half(const uint8_t *raw, uint8_t *rgb,
|
||||||
int w, int h, int hflip);
|
int w, int h, int hflip);
|
||||||
|
|
||||||
|
/* Half-resolution demosaic straight to planar YUV420 (JFIF full-range,
|
||||||
|
* ITU-R 601) for the VPU JPEG encoder: luma per 2x2 BGGR quad at
|
||||||
|
* (w/2)x(h/2), chroma averaged per 2x2 luma block at (w/4)x(h/4).
|
||||||
|
* w/2 and h/2 must be even. Strides are in bytes. */
|
||||||
|
void debayer_bggr_half_yuv420(const uint8_t *raw, int w, int h, int hflip,
|
||||||
|
uint8_t *yp, int y_stride,
|
||||||
|
uint8_t *up, uint8_t *vp, int uv_stride);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -210,11 +210,12 @@ static int cb_status(const struct _u_request *req, struct _u_response *res,
|
|||||||
char body[256];
|
char body[256];
|
||||||
snprintf(body, sizeof(body),
|
snprintf(body, sizeof(body),
|
||||||
"{\"running\":%s,\"cam\":\"%s\",\"clients\":%d,"
|
"{\"running\":%s,\"cam\":\"%s\",\"clients\":%d,"
|
||||||
"\"frames\":%llu,\"fps\":%.1f,"
|
"\"frames\":%llu,\"fps\":%.1f,\"encoder\":\"%s\","
|
||||||
"\"stream\":{\"width\":1296,\"height\":972},"
|
"\"stream\":{\"width\":1296,\"height\":972},"
|
||||||
"\"snapshot\":{\"width\":2592,\"height\":1944}}",
|
"\"snapshot\":{\"width\":2592,\"height\":1944}}",
|
||||||
st.running ? "true" : "false", cam_name(st.cam), st.clients,
|
st.running ? "true" : "false", cam_name(st.cam), st.clients,
|
||||||
(unsigned long long)st.seq, st.fps);
|
(unsigned long long)st.seq, st.fps,
|
||||||
|
st.vpu ? "vpu" : "software");
|
||||||
ulfius_set_string_body_response(res, 200, body);
|
ulfius_set_string_body_response(res, 200, body);
|
||||||
ulfius_add_header_to_response(res, "Content-Type", "application/json");
|
ulfius_add_header_to_response(res, "Content-Type", "application/json");
|
||||||
return U_CALLBACK_CONTINUE;
|
return U_CALLBACK_CONTINUE;
|
||||||
|
|||||||
@@ -0,0 +1,218 @@
|
|||||||
|
/*
|
||||||
|
* vpu_jpeg.c - hardware JPEG encoding on the i.MX6 CODA960 VPU
|
||||||
|
* Copyright (c) 2026 Scott Wiederhold <s.e.wiederhold@gmail.com>
|
||||||
|
* SPDX-License-Identifier: MIT
|
||||||
|
*
|
||||||
|
* V4L2 mem2mem, single-planar API against the mainline coda driver: one
|
||||||
|
* MMAP buffer on each queue, synchronous QBUF/DQBUF per frame. The node
|
||||||
|
* is found by personality (driver "coda", JPEG on the capture side,
|
||||||
|
* YUV420 accepted on the output side), never by number - coda registers
|
||||||
|
* four nodes and the numbering depends on probe order.
|
||||||
|
*/
|
||||||
|
#include "vpu_jpeg.h"
|
||||||
|
|
||||||
|
#include <errno.h>
|
||||||
|
#include <fcntl.h>
|
||||||
|
#include <linux/videodev2.h>
|
||||||
|
#include <poll.h>
|
||||||
|
#include <stdio.h>
|
||||||
|
#include <stdlib.h>
|
||||||
|
#include <string.h>
|
||||||
|
#include <sys/ioctl.h>
|
||||||
|
#include <sys/mman.h>
|
||||||
|
#include <unistd.h>
|
||||||
|
|
||||||
|
#define ENCODE_TIMEOUT_MS 1000
|
||||||
|
|
||||||
|
struct vpu_jpeg {
|
||||||
|
int fd;
|
||||||
|
int w, h;
|
||||||
|
int bpl; /* OUTPUT luma stride from S_FMT */
|
||||||
|
uint8_t *out; /* mapped OUTPUT (YUV420) buffer */
|
||||||
|
size_t out_size;
|
||||||
|
uint8_t *cap; /* mapped CAPTURE (JPEG) buffer */
|
||||||
|
size_t cap_size;
|
||||||
|
};
|
||||||
|
|
||||||
|
static int xioctl(int fd, unsigned long req, void *arg)
|
||||||
|
{
|
||||||
|
int r;
|
||||||
|
do {
|
||||||
|
r = ioctl(fd, req, arg);
|
||||||
|
} while (r == -1 && errno == EINTR);
|
||||||
|
return r;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* Is this node the coda JPEG encoder? (JPEG capture, YUV420 output) */
|
||||||
|
static int is_jpeg_encoder(int fd)
|
||||||
|
{
|
||||||
|
struct v4l2_capability cap = {0};
|
||||||
|
if (xioctl(fd, VIDIOC_QUERYCAP, &cap) < 0 ||
|
||||||
|
strcmp((const char *)cap.driver, "coda") != 0 ||
|
||||||
|
!(cap.device_caps & V4L2_CAP_VIDEO_M2M))
|
||||||
|
return 0;
|
||||||
|
|
||||||
|
struct v4l2_fmtdesc fd0 = { .type = V4L2_BUF_TYPE_VIDEO_CAPTURE };
|
||||||
|
if (xioctl(fd, VIDIOC_ENUM_FMT, &fd0) < 0 ||
|
||||||
|
fd0.pixelformat != V4L2_PIX_FMT_JPEG)
|
||||||
|
return 0;
|
||||||
|
|
||||||
|
for (unsigned i = 0; ; i++) {
|
||||||
|
struct v4l2_fmtdesc fo = { .type = V4L2_BUF_TYPE_VIDEO_OUTPUT,
|
||||||
|
.index = i };
|
||||||
|
if (xioctl(fd, VIDIOC_ENUM_FMT, &fo) < 0)
|
||||||
|
return 0;
|
||||||
|
if (fo.pixelformat == V4L2_PIX_FMT_YUV420)
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static int find_encoder(void)
|
||||||
|
{
|
||||||
|
for (int i = 0; i < 32; i++) {
|
||||||
|
char path[32];
|
||||||
|
snprintf(path, sizeof(path), "/dev/video%d", i);
|
||||||
|
int fd = open(path, O_RDWR | O_NONBLOCK, 0);
|
||||||
|
if (fd < 0)
|
||||||
|
continue;
|
||||||
|
if (is_jpeg_encoder(fd))
|
||||||
|
return fd;
|
||||||
|
close(fd);
|
||||||
|
}
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
|
||||||
|
static int map_one(int fd, enum v4l2_buf_type type, uint8_t **mem,
|
||||||
|
size_t *size)
|
||||||
|
{
|
||||||
|
struct v4l2_requestbuffers req = { .count = 1, .type = type,
|
||||||
|
.memory = V4L2_MEMORY_MMAP };
|
||||||
|
if (xioctl(fd, VIDIOC_REQBUFS, &req) < 0 || req.count < 1)
|
||||||
|
return -1;
|
||||||
|
struct v4l2_buffer buf = { .type = type, .memory = V4L2_MEMORY_MMAP,
|
||||||
|
.index = 0 };
|
||||||
|
if (xioctl(fd, VIDIOC_QUERYBUF, &buf) < 0)
|
||||||
|
return -1;
|
||||||
|
*mem = mmap(NULL, buf.length, PROT_READ | PROT_WRITE, MAP_SHARED,
|
||||||
|
fd, buf.m.offset);
|
||||||
|
if (*mem == MAP_FAILED) {
|
||||||
|
*mem = NULL;
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
|
*size = buf.length;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
vpu_jpeg_t *vpu_jpeg_open(int w, int h, int quality)
|
||||||
|
{
|
||||||
|
vpu_jpeg_t *v = calloc(1, sizeof(*v));
|
||||||
|
if (!v)
|
||||||
|
return NULL;
|
||||||
|
v->fd = find_encoder();
|
||||||
|
if (v->fd < 0)
|
||||||
|
goto fail;
|
||||||
|
|
||||||
|
struct v4l2_format fo = { .type = V4L2_BUF_TYPE_VIDEO_OUTPUT };
|
||||||
|
fo.fmt.pix.width = (unsigned)w;
|
||||||
|
fo.fmt.pix.height = (unsigned)h;
|
||||||
|
fo.fmt.pix.pixelformat = V4L2_PIX_FMT_YUV420;
|
||||||
|
fo.fmt.pix.field = V4L2_FIELD_NONE;
|
||||||
|
if (xioctl(v->fd, VIDIOC_S_FMT, &fo) < 0 ||
|
||||||
|
fo.fmt.pix.width != (unsigned)w ||
|
||||||
|
fo.fmt.pix.height != (unsigned)h) {
|
||||||
|
fprintf(stderr, "vpu: S_FMT output rejected %dx%d\n", w, h);
|
||||||
|
goto fail;
|
||||||
|
}
|
||||||
|
v->w = w;
|
||||||
|
v->h = h;
|
||||||
|
v->bpl = (int)fo.fmt.pix.bytesperline;
|
||||||
|
|
||||||
|
struct v4l2_format fc = { .type = V4L2_BUF_TYPE_VIDEO_CAPTURE };
|
||||||
|
fc.fmt.pix.width = (unsigned)w;
|
||||||
|
fc.fmt.pix.height = (unsigned)h;
|
||||||
|
fc.fmt.pix.pixelformat = V4L2_PIX_FMT_JPEG;
|
||||||
|
if (xioctl(v->fd, VIDIOC_S_FMT, &fc) < 0)
|
||||||
|
goto fail;
|
||||||
|
|
||||||
|
struct v4l2_control q = { .id = V4L2_CID_JPEG_COMPRESSION_QUALITY,
|
||||||
|
.value = quality };
|
||||||
|
xioctl(v->fd, VIDIOC_S_CTRL, &q); /* best effort */
|
||||||
|
|
||||||
|
if (map_one(v->fd, V4L2_BUF_TYPE_VIDEO_OUTPUT, &v->out, &v->out_size) ||
|
||||||
|
map_one(v->fd, V4L2_BUF_TYPE_VIDEO_CAPTURE, &v->cap, &v->cap_size))
|
||||||
|
goto fail;
|
||||||
|
|
||||||
|
enum v4l2_buf_type t = V4L2_BUF_TYPE_VIDEO_OUTPUT;
|
||||||
|
if (xioctl(v->fd, VIDIOC_STREAMON, &t) < 0)
|
||||||
|
goto fail;
|
||||||
|
t = V4L2_BUF_TYPE_VIDEO_CAPTURE;
|
||||||
|
if (xioctl(v->fd, VIDIOC_STREAMON, &t) < 0)
|
||||||
|
goto fail;
|
||||||
|
return v;
|
||||||
|
|
||||||
|
fail:
|
||||||
|
vpu_jpeg_close(v);
|
||||||
|
return NULL;
|
||||||
|
}
|
||||||
|
|
||||||
|
void vpu_jpeg_planes(vpu_jpeg_t *v, uint8_t **y, uint8_t **u, uint8_t **vv,
|
||||||
|
int *y_stride, int *uv_stride)
|
||||||
|
{
|
||||||
|
*y = v->out;
|
||||||
|
*u = v->out + (size_t)v->bpl * v->h;
|
||||||
|
*vv = *u + (size_t)(v->bpl / 2) * (v->h / 2);
|
||||||
|
*y_stride = v->bpl;
|
||||||
|
*uv_stride = v->bpl / 2;
|
||||||
|
}
|
||||||
|
|
||||||
|
int vpu_jpeg_encode(vpu_jpeg_t *v, uint8_t **jpeg, size_t *len)
|
||||||
|
{
|
||||||
|
struct v4l2_buffer cb = { .type = V4L2_BUF_TYPE_VIDEO_CAPTURE,
|
||||||
|
.memory = V4L2_MEMORY_MMAP, .index = 0 };
|
||||||
|
struct v4l2_buffer ob = { .type = V4L2_BUF_TYPE_VIDEO_OUTPUT,
|
||||||
|
.memory = V4L2_MEMORY_MMAP, .index = 0 };
|
||||||
|
ob.bytesused = (unsigned)v->out_size;
|
||||||
|
|
||||||
|
if (xioctl(v->fd, VIDIOC_QBUF, &cb) < 0 ||
|
||||||
|
xioctl(v->fd, VIDIOC_QBUF, &ob) < 0)
|
||||||
|
return -1;
|
||||||
|
|
||||||
|
struct pollfd pfd = { .fd = v->fd, .events = POLLIN };
|
||||||
|
int pr;
|
||||||
|
do {
|
||||||
|
pr = poll(&pfd, 1, ENCODE_TIMEOUT_MS);
|
||||||
|
} while (pr == -1 && errno == EINTR);
|
||||||
|
if (pr <= 0)
|
||||||
|
return -1;
|
||||||
|
|
||||||
|
if (xioctl(v->fd, VIDIOC_DQBUF, &cb) < 0)
|
||||||
|
return -1;
|
||||||
|
xioctl(v->fd, VIDIOC_DQBUF, &ob);
|
||||||
|
|
||||||
|
uint8_t *out = malloc(cb.bytesused);
|
||||||
|
if (!out)
|
||||||
|
return -1;
|
||||||
|
memcpy(out, v->cap, cb.bytesused);
|
||||||
|
*jpeg = out;
|
||||||
|
*len = cb.bytesused;
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
void vpu_jpeg_close(vpu_jpeg_t *v)
|
||||||
|
{
|
||||||
|
if (!v)
|
||||||
|
return;
|
||||||
|
if (v->fd >= 0) {
|
||||||
|
enum v4l2_buf_type t = V4L2_BUF_TYPE_VIDEO_OUTPUT;
|
||||||
|
xioctl(v->fd, VIDIOC_STREAMOFF, &t);
|
||||||
|
t = V4L2_BUF_TYPE_VIDEO_CAPTURE;
|
||||||
|
xioctl(v->fd, VIDIOC_STREAMOFF, &t);
|
||||||
|
}
|
||||||
|
if (v->out)
|
||||||
|
munmap(v->out, v->out_size);
|
||||||
|
if (v->cap)
|
||||||
|
munmap(v->cap, v->cap_size);
|
||||||
|
if (v->fd >= 0)
|
||||||
|
close(v->fd);
|
||||||
|
free(v);
|
||||||
|
}
|
||||||
@@ -0,0 +1,33 @@
|
|||||||
|
/*
|
||||||
|
* vpu_jpeg.h - hardware JPEG encoding on the i.MX6 CODA960 VPU
|
||||||
|
* Copyright (c) 2026 Scott Wiederhold <s.e.wiederhold@gmail.com>
|
||||||
|
* SPDX-License-Identifier: MIT
|
||||||
|
*
|
||||||
|
* Thin wrapper around the mainline coda V4L2 mem2mem JPEG encoder: the
|
||||||
|
* caller writes planar YUV420 directly into the encoder's OUTPUT buffer
|
||||||
|
* (vpu_jpeg_planes) and gets back a malloc'd JFIF JPEG.
|
||||||
|
*/
|
||||||
|
#ifndef FORGECTRL_VPU_JPEG_H
|
||||||
|
#define FORGECTRL_VPU_JPEG_H
|
||||||
|
|
||||||
|
#include <stddef.h>
|
||||||
|
#include <stdint.h>
|
||||||
|
|
||||||
|
typedef struct vpu_jpeg vpu_jpeg_t;
|
||||||
|
|
||||||
|
/* Locate the CODA JPEG encoder video node, configure it for w x h YUV420
|
||||||
|
* -> JPEG at the given quality (5..100), and map one buffer per queue.
|
||||||
|
* Returns NULL if no encoder exists or setup fails. */
|
||||||
|
vpu_jpeg_t *vpu_jpeg_open(int w, int h, int quality);
|
||||||
|
|
||||||
|
/* Planes of the mapped OUTPUT buffer for direct fill. */
|
||||||
|
void vpu_jpeg_planes(vpu_jpeg_t *v, uint8_t **y, uint8_t **u, uint8_t **vv,
|
||||||
|
int *y_stride, int *uv_stride);
|
||||||
|
|
||||||
|
/* Encode the currently-filled OUTPUT buffer. On success *jpeg is malloc'd
|
||||||
|
* (caller frees) and 0 is returned. */
|
||||||
|
int vpu_jpeg_encode(vpu_jpeg_t *v, uint8_t **jpeg, size_t *len);
|
||||||
|
|
||||||
|
void vpu_jpeg_close(vpu_jpeg_t *v);
|
||||||
|
|
||||||
|
#endif
|
||||||
Reference in New Issue
Block a user