JPEG软件编码,直接用就行

#ifndef JPEG_ENCODE_H
#define JPEG_ENCODE_H

#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>

#ifdef __cplusplus
extern "C" {
#endif

/* 软件基线 JPEG 编码器:ARGB8888 输入 → JFIF baseline、4:2:0、无 APP1。
 *
 * 纯软件实现,不碰任何外设,因此没有硬件相关条件编译。
 *
 * 色度采样 4:2:0:MCU = 16×16 像素 = 4 个亮度 8×8 块 + Cb/Cr 各 1 块,
 * 色度按 2×2 取平均后再编码。
 *
 *   argb        ARGB8888 像素首地址(A 被忽略)
 *   width/height 图像尺寸,不必是 16 的整数倍(边缘用最后一行/列复制补齐)
 *   stride      行距,单位像素,必须 >= width
 *   quality     1..100,超出会被夹到范围内
 *   output      输出缓冲
 *   capacity    输出缓冲容量,最小值 1024
 *   outputSize  收到实际写入字节数;返回 false 时保持 0
 *
 * 返回 false 表示参数非法或输出放不下(output 中已写入的部分无意义,
 * 调用方应整体丢弃)。 */
bool JpegEncode_Argb8888(const uint32_t *argb, uint16_t width, uint16_t height,
                         uint16_t stride, uint8_t quality, uint8_t *output,
                         size_t capacity, size_t *outputSize);

#ifdef __cplusplus
}
#endif

#endif /* JPEG_ENCODE_H */

#include "jpeg_encode.h"

#include <string.h>

typedef struct {
    uint8_t* data;
    size_t capacity;
    size_t size;
    uint32_t bits;
    uint8_t bitCount;
    bool ok;
} JpegWriter_t;

typedef struct { uint16_t code; uint8_t bits; } HuffCode_t;

/* Keep generated lookup tables out of PhotoTask's 4KB stack. */
static HuffCode_t s_dcYCode[256];
static HuffCode_t s_dcCCode[256];
static HuffCode_t s_acYCode[256];
static HuffCode_t s_acCCode[256];

static const uint8_t s_zigzag[64] = {
     0, 1, 8,16, 9, 2, 3,10,17,24,32,25,18,11, 4, 5,
    12,19,26,33,40,48,41,34,27,20,13, 6, 7,14,21,28,
    35,42,49,56,57,50,43,36,29,22,15,23,30,37,44,51,
    58,59,52,45,38,31,39,46,53,60,61,54,47,55,62,63
};

static const uint8_t s_qYBase[64] = {
    16,11,10,16,24,40,51,61, 12,12,14,19,26,58,60,55,
    14,13,16,24,40,57,69,56, 14,17,22,29,51,87,80,62,
    18,22,37,56,68,109,103,77, 24,35,55,64,81,104,113,92,
    49,64,78,87,103,121,120,101, 72,92,95,98,112,100,103,99
};
static const uint8_t s_qCBase[64] = {
    17,18,24,47,99,99,99,99, 18,21,26,66,99,99,99,99,
    24,26,56,99,99,99,99,99, 47,66,99,99,99,99,99,99,
    99,99,99,99,99,99,99,99, 99,99,99,99,99,99,99,99,
    99,99,99,99,99,99,99,99, 99,99,99,99,99,99,99,99
};

static const uint8_t s_bitsDcY[16] =
    {0,1,5,1,1,1,1,1,1,0,0,0,0,0,0,0};
static const uint8_t s_valDcY[12] =
    {0,1,2,3,4,5,6,7,8,9,10,11};
static const uint8_t s_bitsDcC[16] =
    {0,3,1,1,1,1,1,1,1,1,1,0,0,0,0,0};
static const uint8_t s_valDcC[12] =
    {0,1,2,3,4,5,6,7,8,9,10,11};
static const uint8_t s_bitsAcY[16] =
    {0,2,1,3,3,2,4,3,5,5,4,4,0,0,1,0x7d};
static const uint8_t s_valAcY[162] = {
    0x01,0x02,0x03,0x00,0x04,0x11,0x05,0x12,0x21,0x31,0x41,0x06,
    0x13,0x51,0x61,0x07,0x22,0x71,0x14,0x32,0x81,0x91,0xa1,0x08,
    0x23,0x42,0xb1,0xc1,0x15,0x52,0xd1,0xf0,0x24,0x33,0x62,0x72,
    0x82,0x09,0x0a,0x16,0x17,0x18,0x19,0x1a,0x25,0x26,0x27,0x28,
    0x29,0x2a,0x34,0x35,0x36,0x37,0x38,0x39,0x3a,0x43,0x44,0x45,
    0x46,0x47,0x48,0x49,0x4a,0x53,0x54,0x55,0x56,0x57,0x58,0x59,
    0x5a,0x63,0x64,0x65,0x66,0x67,0x68,0x69,0x6a,0x73,0x74,0x75,
    0x76,0x77,0x78,0x79,0x7a,0x83,0x84,0x85,0x86,0x87,0x88,0x89,
    0x8a,0x92,0x93,0x94,0x95,0x96,0x97,0x98,0x99,0x9a,0xa2,0xa3,
    0xa4,0xa5,0xa6,0xa7,0xa8,0xa9,0xaa,0xb2,0xb3,0xb4,0xb5,0xb6,
    0xb7,0xb8,0xb9,0xba,0xc2,0xc3,0xc4,0xc5,0xc6,0xc7,0xc8,0xc9,
    0xca,0xd2,0xd3,0xd4,0xd5,0xd6,0xd7,0xd8,0xd9,0xda,0xe1,0xe2,
    0xe3,0xe4,0xe5,0xe6,0xe7,0xe8,0xe9,0xea,0xf1,0xf2,0xf3,0xf4,
    0xf5,0xf6,0xf7,0xf8,0xf9,0xfa
};
static const uint8_t s_bitsAcC[16] =
    {0,2,1,2,4,4,3,4,7,5,4,4,0,1,2,0x77};
static const uint8_t s_valAcC[162] = {
    0x00,0x01,0x02,0x03,0x11,0x04,0x05,0x21,0x31,0x06,0x12,0x41,
    0x51,0x07,0x61,0x71,0x13,0x22,0x32,0x81,0x08,0x14,0x42,0x91,
    0xa1,0xb1,0xc1,0x09,0x23,0x33,0x52,0xf0,0x15,0x62,0x72,0xd1,
    0x0a,0x16,0x24,0x34,0xe1,0x25,0xf1,0x17,0x18,0x19,0x1a,0x26,
    0x27,0x28,0x29,0x2a,0x35,0x36,0x37,0x38,0x39,0x3a,0x43,0x44,
    0x45,0x46,0x47,0x48,0x49,0x4a,0x53,0x54,0x55,0x56,0x57,0x58,
    0x59,0x5a,0x63,0x64,0x65,0x66,0x67,0x68,0x69,0x6a,0x73,0x74,
    0x75,0x76,0x77,0x78,0x79,0x7a,0x82,0x83,0x84,0x85,0x86,0x87,
    0x88,0x89,0x8a,0x92,0x93,0x94,0x95,0x96,0x97,0x98,0x99,0x9a,
    0xa2,0xa3,0xa4,0xa5,0xa6,0xa7,0xa8,0xa9,0xaa,0xb2,0xb3,0xb4,
    0xb5,0xb6,0xb7,0xb8,0xb9,0xba,0xc2,0xc3,0xc4,0xc5,0xc6,0xc7,
    0xc8,0xc9,0xca,0xd2,0xd3,0xd4,0xd5,0xd6,0xd7,0xd8,0xd9,0xda,
    0xe2,0xe3,0xe4,0xe5,0xe6,0xe7,0xe8,0xe9,0xea,0xf2,0xf3,0xf4,
    0xf5,0xf6,0xf7,0xf8,0xf9,0xfa
};

/* alpha(u)*cos((2*x+1)*u*pi/16), scaled by 16384. */
static const int16_t s_dct[8][8] = {
    {11585,11585,11585,11585,11585,11585,11585,11585},
    {16069,13623, 9102, 3196,-3196,-9102,-13623,-16069},
    {15137, 6270,-6270,-15137,-15137,-6270, 6270,15137},
    {13623,-3196,-16069,-9102, 9102,16069, 3196,-13623},
    {11585,-11585,-11585,11585,11585,-11585,-11585,11585},
    { 9102,-16069,3196,13623,-13623,-3196,16069,-9102},
    { 6270,-15137,15137,-6270,-6270,15137,-15137,6270},
    { 3196,-9102,13623,-16069,16069,-13623,9102,-3196}
};

static void putByte(JpegWriter_t* w, uint8_t value)
{
    if (!w->ok || w->size >= w->capacity) { w->ok = false; return; }
    w->data[w->size++] = value;
}
static void putU16(JpegWriter_t* w, uint16_t value)
{
    putByte(w, (uint8_t)(value >> 8)); putByte(w, (uint8_t)value);
}
static void putMarker(JpegWriter_t* w, uint8_t marker)
{
    putByte(w, 0xffU); putByte(w, marker);
}
static void putBits(JpegWriter_t* w, uint16_t value, uint8_t count)
{
    if (count == 0U || !w->ok) return;
    w->bits = (w->bits << count) | (value & ((1UL << count) - 1UL));
    w->bitCount = (uint8_t)(w->bitCount + count);
    while (w->bitCount >= 8U) {
        uint8_t b = (uint8_t)(w->bits >> (w->bitCount - 8U));
        w->bitCount = (uint8_t)(w->bitCount - 8U);
        putByte(w, b);
        if (b == 0xffU) putByte(w, 0U);
    }
}
static void flushBits(JpegWriter_t* w)
{
    if (w->bitCount != 0U) {
        uint8_t pad = (uint8_t)(8U - w->bitCount);
        putBits(w, (uint16_t)((1U << pad) - 1U), pad);
    }
}
static void buildHuff(const uint8_t bits[16], const uint8_t* values,
                      HuffCode_t table[256])
{
    uint16_t code = 0U;
    size_t k = 0U;
    memset(table, 0, 256U * sizeof(table[0]));
    for (uint8_t len = 1U; len <= 16U; ++len) {
        for (uint8_t n = 0U; n < bits[len - 1U]; ++n) {
            table[values[k]].code = code++;
            table[values[k]].bits = len;
            ++k;
        }
        code <<= 1;
    }
}
static uint8_t magnitudeBits(int value)
{
    unsigned int v = (unsigned int)(value < 0 ? -value : value);
    uint8_t n = 0U;
    while (v != 0U) { ++n; v >>= 1; }
    return n;
}
static uint16_t magnitudeValue(int value, uint8_t bits)
{
    return value < 0 ? (uint16_t)(value + ((1 << bits) - 1)) : (uint16_t)value;
}
static void fdctQuant(const int16_t input[64], const uint8_t q[64], int16_t out[64])
{
    int32_t temp[64];
    for (uint8_t y = 0U; y < 8U; ++y) {
        for (uint8_t u = 0U; u < 8U; ++u) {
            int32_t sum = 0;
            for (uint8_t x = 0U; x < 8U; ++x)
                sum += (int32_t)input[y * 8U + x] * s_dct[u][x];
            temp[y * 8U + u] = sum;
        }
    }
    for (uint8_t v = 0U; v < 8U; ++v) {
        for (uint8_t u = 0U; u < 8U; ++u) {
            int64_t sum = 0;
            int32_t coeff;
            for (uint8_t y = 0U; y < 8U; ++y)
                sum += (int64_t)temp[y * 8U + u] * s_dct[v][y];
            if (sum >= 0) coeff = (int32_t)((sum + (1LL << 29)) >> 30);
            else coeff = -(int32_t)(((-sum) + (1LL << 29)) >> 30);
            if (coeff >= 0) coeff = (coeff + q[v * 8U + u] / 2) / q[v * 8U + u];
            else coeff = -((-coeff + q[v * 8U + u] / 2) / q[v * 8U + u]);
            out[v * 8U + u] = (int16_t)coeff;
        }
    }
}
static void encodeBlock(JpegWriter_t* w, const int16_t input[64],
                        const uint8_t q[64], const HuffCode_t dc[256],
                        const HuffCode_t ac[256], int16_t* previousDc)
{
    int16_t c[64];
    int diff;
    uint8_t n;
    uint8_t zeroRun = 0U;
    fdctQuant(input, q, c);
    if (c[0] > 2047) c[0] = 2047;
    if (c[0] < -2047) c[0] = -2047;
    diff = c[0] - *previousDc;
    *previousDc = c[0];
    n = magnitudeBits(diff);
    putBits(w, dc[n].code, dc[n].bits);
    if (n != 0U) putBits(w, magnitudeValue(diff, n), n);

    for (uint8_t k = 1U; k < 64U; ++k) {
        int value = c[s_zigzag[k]];
        if (value > 1023) value = 1023;
        if (value < -1023) value = -1023;
        if (value == 0) { ++zeroRun; continue; }
        while (zeroRun >= 16U) {
            putBits(w, ac[0xf0].code, ac[0xf0].bits);
            zeroRun = (uint8_t)(zeroRun - 16U);
        }
        n = magnitudeBits(value);
        {
            uint8_t symbol = (uint8_t)((zeroRun << 4) | n);
            putBits(w, ac[symbol].code, ac[symbol].bits);
        }
        putBits(w, magnitudeValue(value, n), n);
        zeroRun = 0U;
    }
    if (zeroRun != 0U) putBits(w, ac[0].code, ac[0].bits);
}
static void makeQuant(uint8_t quality, const uint8_t base[64], uint8_t out[64])
{
    int scale;
    if (quality < 1U) quality = 1U;
    if (quality > 100U) quality = 100U;
    scale = quality < 50U ? 5000 / quality : 200 - quality * 2;
    for (uint8_t i = 0U; i < 64U; ++i) {
        int v = (base[i] * scale + 50) / 100;
        if (v < 1) v = 1;
        if (v > 255) v = 255;
        out[i] = (uint8_t)v;
    }
}
static void writeDhtTable(JpegWriter_t* w, uint8_t id,
                          const uint8_t bits[16], const uint8_t* values)
{
    size_t count = 0U;
    putByte(w, id);
    for (uint8_t i = 0U; i < 16U; ++i) { putByte(w, bits[i]); count += bits[i]; }
    for (size_t i = 0U; i < count; ++i) putByte(w, values[i]);
}

/* 4:2:0 的 MCU 是 16×16 像素:4 个亮度块 + 每路色度 1 块,共 6 块。
 * 源图宽高不是 16 的整数倍时,越界的行/列用最后一行/列复制补齐——这是 JPEG
 * 对非整 MCU 边界的标准处理。参数用 int 收,避免 by+oy+iy 在 uint16_t 里回绕。 */
#define JPEG_MCU_SIZE 16U

static uint16_t clamp_index(int value, int limit)
{
    return (uint16_t)((value < limit) ? value : (limit - 1));
}

bool JpegEncode_Argb8888(const uint32_t* argb, uint16_t width,
                         uint16_t height, uint16_t stride, uint8_t quality,
                         uint8_t* output, size_t capacity, size_t* outputSize)
{
    JpegWriter_t w = { output, capacity, 0U, 0U, 0U, true };
    uint8_t qY[64], qC[64];
    int16_t previousDc[3] = {0,0,0};
    int16_t yBlock[64];
    int16_t cbBlock[64];
    int16_t crBlock[64];

    if (outputSize != NULL) *outputSize = 0U;
    if (argb == NULL || output == NULL || outputSize == NULL ||
        width == 0U || height == 0U || stride < width || capacity < 1024U) return false;

    makeQuant(quality, s_qYBase, qY); makeQuant(quality, s_qCBase, qC);
    buildHuff(s_bitsDcY, s_valDcY, s_dcYCode); buildHuff(s_bitsDcC, s_valDcC, s_dcCCode);
    buildHuff(s_bitsAcY, s_valAcY, s_acYCode); buildHuff(s_bitsAcC, s_valAcC, s_acCCode);

    putMarker(&w, 0xd8);                         /* SOI */
    putMarker(&w, 0xe0); putU16(&w, 16U);        /* APP0 JFIF */
    putByte(&w,'J'); putByte(&w,'F'); putByte(&w,'I'); putByte(&w,'F'); putByte(&w,0);
    putByte(&w,1); putByte(&w,1); putByte(&w,0); putU16(&w,1); putU16(&w,1); putByte(&w,0); putByte(&w,0);
    putMarker(&w, 0xdb); putU16(&w, 132U);       /* DQT before SOF0 */
    putByte(&w, 0U); for (uint8_t i=0U;i<64U;++i) putByte(&w,qY[s_zigzag[i]]);
    putByte(&w, 1U); for (uint8_t i=0U;i<64U;++i) putByte(&w,qC[s_zigzag[i]]);
    /* baseline SOF0, 4:2:0:亮度 Y 采样因子 2×2,两路色度 1×1,
     * 于是 MCU = 16×16 像素 = 4 个 Y 块 + 1 个 Cb + 1 个 Cr。 */
    putMarker(&w, 0xc0); putU16(&w, 17U);
    putByte(&w,8); putU16(&w,height); putU16(&w,width); putByte(&w,3);
    putByte(&w,1); putByte(&w,0x22); putByte(&w,0);
    putByte(&w,2); putByte(&w,0x11); putByte(&w,1);
    putByte(&w,3); putByte(&w,0x11); putByte(&w,1);
    putMarker(&w, 0xc4); putU16(&w, 0x01a2U);    /* standard Huffman tables */
    writeDhtTable(&w,0x00,s_bitsDcY,s_valDcY); writeDhtTable(&w,0x10,s_bitsAcY,s_valAcY);
    writeDhtTable(&w,0x01,s_bitsDcC,s_valDcC); writeDhtTable(&w,0x11,s_bitsAcC,s_valAcC);
    putMarker(&w, 0xda); putU16(&w, 12U);        /* SOS */
    putByte(&w,3); putByte(&w,1); putByte(&w,0x00); putByte(&w,2); putByte(&w,0x11);
    putByte(&w,3); putByte(&w,0x11); putByte(&w,0); putByte(&w,63); putByte(&w,0);

    for (uint16_t by = 0U; by < height; by = (uint16_t)(by + JPEG_MCU_SIZE)) {
        for (uint16_t bx = 0U; bx < width; bx = (uint16_t)(bx + JPEG_MCU_SIZE)) {
            /* ① 亮度:一个 MCU 里 4 个 8×8 块,按 raster 顺序
             *    (0,0) (8,0) (0,8) (8,8) —— 这就是熵编码的顺序,
             *    顺序写反图像的块会错位,而且不容易一眼看出来。 */
            for (uint8_t sub = 0U; sub < 4U; ++sub) {
                int oy = (sub & 2U) ? 8 : 0;
                int ox = (sub & 1U) ? 8 : 0;

                for (uint8_t iy = 0U; iy < 8U; ++iy) {
                    uint16_t sy = clamp_index((int)by + oy + (int)iy, (int)height);
                    for (uint8_t ix = 0U; ix < 8U; ++ix) {
                        uint16_t sx = clamp_index((int)bx + ox + (int)ix, (int)width);
                        uint32_t p = argb[(uint32_t)sy * stride + sx];
                        int r = (int)((p >> 16) & 0xffU);
                        int g = (int)((p >> 8) & 0xffU);
                        int b = (int)(p & 0xffU);
                        yBlock[iy * 8U + ix] =
                            (int16_t)(((77*r + 150*g + 29*b + 128) >> 8) - 128);
                    }
                }
                encodeBlock(&w, yBlock, qY, s_dcYCode, s_acYCode, &previousDc[0]);
            }

            /* ② 色度:把 16×16 里每个 2×2 取平均,得到 8×8 的 Cb / Cr 各一块。
             * 必须在 Cb/Cr 域上平均(YCbCr 是仿射变换,先平均 RGB 会多引入一次
             * 舍入),所以逐像素算完再累加。 */
            for (uint8_t cy = 0U; cy < 8U; ++cy) {
                for (uint8_t cx = 0U; cx < 8U; ++cx) {
                    int cbSum = 0;
                    int crSum = 0;

                    for (uint8_t qy = 0U; qy < 2U; ++qy) {
                        uint16_t sy = clamp_index((int)by + (int)(cy * 2U) + (int)qy,
                                                  (int)height);
                        for (uint8_t qx = 0U; qx < 2U; ++qx) {
                            uint16_t sx = clamp_index((int)bx + (int)(cx * 2U) + (int)qx,
                                                      (int)width);
                            uint32_t p = argb[(uint32_t)sy * stride + sx];
                            int r = (int)((p >> 16) & 0xffU);
                            int g = (int)((p >> 8) & 0xffU);
                            int b = (int)(p & 0xffU);
                            cbSum += (-43*r - 85*g + 128*b + 32768 + 128) >> 8;
                            crSum += (128*r - 107*g - 21*b + 32768 + 128) >> 8;
                        }
                    }
                    {
                        uint8_t n = (uint8_t)(cy * 8U + cx);
                        cbBlock[n] = (int16_t)(((cbSum + 2) >> 2) - 128);
                        crBlock[n] = (int16_t)(((crSum + 2) >> 2) - 128);
                    }
                }
            }
            encodeBlock(&w, cbBlock, qC, s_dcCCode, s_acCCode, &previousDc[1]);
            encodeBlock(&w, crBlock, qC, s_dcCCode, s_acCCode, &previousDc[2]);
            if (!w.ok) return false;
        }
    }
    flushBits(&w); putMarker(&w, 0xd9);           /* EOI */
    if (!w.ok) return false;
    *outputSize = w.size;
    return true;
}