codec: vendor VaptVupt 2.65.0 + fix two ratio-crippling wrapper defaults

Vendor codec 2.65.0 (from vaptvupt-codec tag v2.65.0), re-applying the two
in-tree audit patches on top: the ANS decode safe-zone 2*SAFEZONE_MAX_RUN
reserve (heap-overflow guard, not yet upstream) and the AVX2 offset-read
bound in vv_decoder.c.

Two settings in the integration layer were leaving most of the codec's ratio
on the table:

  * vaptvupt_api.c forced opts.format_v2=1 for balanced+extreme. Since codec
    v2.61.0 that routes text through the binary/greedy path and HALVES the
    extreme-mode text ratio (codec-level 7.6x -> 3.7x). The codec auto-enables
    format_v2 for binary-detected input on its own, so stop forcing it: text
    keeps the optimal parser, binary still gets v2.

  * auto_block_size() capped the extreme block at 512 KiB. The block IS the
    codec's LZ window, so the 'large-window extreme' parser could never match
    past 512 KiB. Scale block size with level (128 KiB fast -> 8 MiB extreme).
    Because block size also sets --dedup granularity (a large block rarely
    finds a byte-exact duplicate), --dedup now overrides to a small 256 KiB
    block so block-level dedup still works.

Measured, level 9 extreme: text 3.77x->5.98x (+58%), logs 7.21x->9.07x (+26%),
json 8.25x->9.38x, source 4.93x->5.63x. Wire format unchanged (v1.6); 5.0.0
and 5.1.0 archives interoperate both directions (verified, all modes). Bump
ZUPT_CODEC_RELEASE to 2.65.0 and ZUPT_VERSION_STRING to 5.1.0.
This commit is contained in:
Cristian Cezar Moisés 2026-07-11 22:53:44 -03:00
commit a2350dd0f6
8 changed files with 953 additions and 290 deletions

View file

@ -1,9 +1,9 @@
/* /*
* VaptVupt Zupt Integration API * VaptVupt VaptVupt Integration API
* SPDX-License-Identifier: GPL-3.0-or-later * SPDX-License-Identifier: GPL-3.0-or-later
* Copyright 2026 Cristian. * Copyright 2026 Cristian.
* *
* ZUPT-COMPAT: This is the API that Zupt calls. It wraps the internal * EMBED-COMPAT: This is the API that a host application calls. It wraps the internal
* VaptVupt API with sensible defaults for backup workloads: * VaptVupt API with sensible defaults for backup workloads:
* - Checksum always enabled (data integrity is critical for backups) * - Checksum always enabled (data integrity is critical for backups)
* - Adaptive window selection (auto-detect optimal wlog per file) * - Adaptive window selection (auto-detect optimal wlog per file)

View file

@ -50,10 +50,10 @@
#define ZUPT_PRODUCT_EXTENSION ".zupt" /* on-disk archive extension (kept stable) */ #define ZUPT_PRODUCT_EXTENSION ".zupt" /* on-disk archive extension (kept stable) */
#define ZUPT_PRODUCT_TAGLINE "Post-quantum backup compression" #define ZUPT_PRODUCT_TAGLINE "Post-quantum backup compression"
#define ZUPT_VERSION_STRING "5.0.0" #define ZUPT_VERSION_STRING "5.1.0"
/* Vendored codec release (upstream tag) — single source for display strings. /* Vendored codec release (upstream tag) — single source for display strings.
* The codec's own VV_VERSION_* is its internal API version, not the release. */ * The codec's own VV_VERSION_* is its internal API version, not the release. */
#define ZUPT_CODEC_RELEASE "2.60.4" #define ZUPT_CODEC_RELEASE "2.65.0"
#define ZUPT_FORMAT_MAJOR 1 #define ZUPT_FORMAT_MAJOR 1
#define ZUPT_FORMAT_MINOR 6 #define ZUPT_FORMAT_MINOR 6

View file

@ -4,15 +4,22 @@
* Copyright (c) 2025-2026 Cristian Cezar Moisés * Copyright (c) 2025-2026 Cristian Cezar Moisés
* *
* ZUPT-COMPAT: thin wrapper over vv_compress/vv_decompress with * ZUPT-COMPAT: thin wrapper over vv_compress/vv_decompress with
* backup-optimized defaults for VaptVupt 2.60.4. * backup-optimized defaults for VaptVupt 2.65.0.
* *
* Defaults applied here (per ZUPT_INTEGRATION.md, Sprint 122): * Defaults applied here (per ZUPT_INTEGRATION.md, Sprint 122):
* - opts.checksum = 0 (Zupt's HMAC-SHA256 / AES-GCM-SIV outer * - opts.checksum = 0 (Zupt's HMAC-SHA256 / AES-GCM-SIV outer
* already authenticates the compressed * already authenticates the compressed
* bytes; XXH64 footer is redundant work * bytes; XXH64 footer is redundant work
* and saves ~10% encode time) * and saves ~10% encode time)
* - opts.format_v2 = 1 (4-7% better binary ratio; v2.33.0+ * - opts.format_v2 = 0 (AUTO). Since codec v2.61.0 the encoder
* decoders read v2 frames transparently) * auto-enables min_match=3 ('T' blocks) for
* binary-detected input and keeps 'S' blocks
* for text. FORCING format_v2=1 routes text
* through the binary/greedy path and HALVES the
* extreme-mode ratio (text 7.6x -> 3.7x,
* measured on codec 2.65.0); auto keeps the
* optimal parser on text and still wins on
* binary. Never force it here.
* - VV_DECOMPRESS_SKIP_CHECKSUM on decode (matched pair to * - VV_DECOMPRESS_SKIP_CHECKSUM on decode (matched pair to
* checksum=0 on encode; saves ~30% on real * checksum=0 on encode; saves ~30% on real
* fixtures, 2-5x on AEAD-wrapped data) * fixtures, 2-5x on AEAD-wrapped data)
@ -43,14 +50,15 @@ int64_t vvz_compress(const uint8_t *src, size_t src_len,
opts.format_v2 = 0; opts.format_v2 = 0;
} else if (level <= 7) { } else if (level <= 7) {
opts.mode = VV_MODE_BALANCED; opts.mode = VV_MODE_BALANCED;
opts.format_v2 = 1; /* 4-7% better binary ratio (v2.33.0+ decoders) */ opts.format_v2 = 0; /* AUTO: v2 for binary, optimal 'S' for text.
* Forcing v2 halves text ratio see header. */
opts.filter_auto = 1; /* BCJ on recognised ELF/PE/Mach-O input opts.filter_auto = 1; /* BCJ on recognised ELF/PE/Mach-O input
* (codec 2.55.0): no-op on everything else. * (codec 2.55.0): no-op on everything else.
* Blocks where a filter fired need a * Blocks where a filter fired need a
* v2.54.0+ decoder (tool >= 3.9.0). */ * v2.54.0+ decoder (tool >= 3.9.0). */
} else { } else {
opts.mode = VV_MODE_EXTREME; opts.mode = VV_MODE_EXTREME;
opts.format_v2 = 1; /* 4-7% better binary ratio (v2.33.0+ decoders) */ opts.format_v2 = 0; /* AUTO (see BALANCED / header note) */
opts.filter_auto = 1; /* see BALANCED note above */ opts.filter_auto = 1; /* see BALANCED note above */
} }

View file

@ -181,6 +181,21 @@ static void build_dec(const uint16_t norm[NSYM], const uint8_t sp[ANS_L],
vva_dec_entry_t dec[ANS_L]) { vva_dec_entry_t dec[ANS_L]) {
uint16_t occ[NSYM]; uint16_t occ[NSYM];
memset(occ, 0, sizeof(occ)); memset(occ, 0, sizeof(occ));
/* SPRINT 125: per-symbol nb_max/low_count were recomputed (including
* an ilog2 while-loop) for every one of the 4096 slots; hoist them
* to one 256-entry precompute pass identical values, ~16× fewer
* ilog2 evaluations per table build (3-4 builds per block on both
* encode and decode sides). */
int8_t nbmax_tab[NSYM];
int16_t lowcnt_tab[NSYM];
for (int s = 0; s < NSYM; s++) {
uint16_t f = norm[s];
if (f == 0 || f == (uint16_t)ANS_L) { nbmax_tab[s] = 0; lowcnt_tab[s] = 0; continue; }
int flg = ilog2(f);
int nb = ANS_LOG - flg;
nbmax_tab[s] = (int8_t)nb;
lowcnt_tab[s] = (int16_t)((1 << (flg + 1)) - (int)f);
}
for (int x = 0; x < ANS_L; x++) { for (int x = 0; x < ANS_L; x++) {
uint8_t s = sp[x]; uint8_t s = sp[x];
uint16_t f = norm[s]; uint16_t f = norm[s];
@ -189,9 +204,8 @@ static void build_dec(const uint16_t norm[NSYM], const uint8_t sp[ANS_L],
dec[x].symbol = s; dec[x].nbits = 0; dec[x].baseline = 0; dec[x].symbol = s; dec[x].nbits = 0; dec[x].baseline = 0;
continue; continue;
} }
int flg = ilog2(f); int nb_max = nbmax_tab[s];
int nb_max = ANS_LOG - flg; int low_count = lowcnt_tab[s];
int low_count = (1 << (flg + 1)) - (int)f;
/* On a VALID normalized table, f ∈ [1, ANS_L) here (f==0 and /* On a VALID normalized table, f ∈ [1, ANS_L) here (f==0 and
* f==ANS_L are handled above), so flg ANS_LOG-1 and nb_max 1, * f==ANS_L are handled above), so flg ANS_LOG-1 and nb_max 1,
* and the shifts below are well-defined. A CORRUPT stream can * and the shifts below are well-defined. A CORRUPT stream can
@ -265,15 +279,34 @@ static inline int enc_sym(const enc_ctx_t *c, uint32_t state, uint8_t sym,
int base = c->cum[sym], cnt = c->cum[sym + 1] - base; int base = c->cum[sym], cnt = c->cum[sym + 1] - base;
if (!cnt) return -1; if (!cnt) return -1;
if (cnt == ANS_L) { *bv = 0; *bn = 0; return 0; } if (cnt == ANS_L) { *bv = 0; *bn = 0; return 0; }
for (int i = base; i < base + cnt; i++) { /* SPRINT 124: O(1) slot lookup replacing a linear scan that
uint32_t bl = c->o[i].bl; * averaged f/2 iterations (up to ~2048 for a dominant symbol
int nb = c->o[i].nb; * 10-15% of encode wall).
if (state >= bl && state < bl + (1u << nb)) { *
*bv = state - bl; *bn = nb; * The occurrence windows for a symbol with normalized freq f
return (int)c->o[i].slot; * tile [0, ANS_L) exactly (see build_dec): occurrences
} * k < low_count have nb_max = ANS_LOG - ilog2(f) bits and
* baseline k << nb_max; the rest have nb_max-1 bits. Baselines
* ascend with k and build_enc keeps c->o[] baseline-sorted, so
* c->o[base + k] IS occurrence k the window containing `state`
* is directly computable. Produces bit-identical output to the
* scan (same slot, same bits). */
int flg = ilog2((uint32_t)cnt);
int nb_max = ANS_LOG - flg;
uint32_t low_count = (1u << (flg + 1)) - (uint32_t)cnt;
uint32_t threshold = low_count << nb_max;
uint32_t k, nb;
if (state < threshold) {
nb = (uint32_t)nb_max;
k = state >> nb_max;
} else {
nb = (uint32_t)(nb_max - 1);
k = low_count + ((state - threshold) >> nb);
} }
return -1; const enc_occ_t *e = &c->o[base + k];
*bv = state - e->bl;
*bn = (int)nb;
return (int)e->slot;
} }
/* ═══════════════════════════════════════════════════════════════ /* ═══════════════════════════════════════════════════════════════
@ -1477,6 +1510,11 @@ static size_t parse_sequences(const uint8_t *tokens, size_t tok_len,
ll -= LL_MAX; ll -= LL_MAX;
} }
/* SPRINT 125: re-check after the split loop — the while() guard
* at the top of the outer loop does not cover seqs consumed by
* splits within this iteration. */
if (nseq >= seq_cap) return 0;
seqs[nseq].litlen = (uint32_t)ll; seqs[nseq].litlen = (uint32_t)ll;
seqs[nseq].lit_offset = (uint32_t)nlits; seqs[nseq].lit_offset = (uint32_t)nlits;
nlits += ll; nlits += ll;
@ -1511,10 +1549,89 @@ static size_t parse_sequences(const uint8_t *tokens, size_t tok_len,
nseq++; nseq++;
} }
/* SPRINT 125 (defense in depth): if the loop stopped because
* seq_cap was reached with tokens still unparsed, the parse is
* TRUNCATED encoding it would silently drop sequences and emit a
* corrupt block. Unreachable with a correctly-sized seq_cap (see
* the caller's bound derivation), but fail closed regardless. */
if (tp < tp_end) return 0;
*total_lits = nlits; *total_lits = nlits;
return nseq; return nseq;
} }
/* ═══════════════════════════════════════════════════════════════
* LITERAL-CODER SIZE ESTIMATION (SPRINT 124)
*
* The literal-format race used to FULLY encode every candidate
* (ANS4 + ANS1 + Huffman + Huffman4) and keep one measured at
* 6-21% of encode wall, nearly all discarded. One histogram plus
* analytic size estimates picks the winner first; only the winner
* is actually encoded.
* */
/* Unlimited-depth Huffman code lengths, for size estimation only.
* (The real coder limits depth to 15; the difference is a handful of
* bits on pathological distributions irrelevant for choosing.) */
static void est_huff_lengths(const uint32_t freq[NSYM], uint8_t len[NSYM]) {
int leaf_sym[NSYM];
int n = 0;
for (int s = 0; s < NSYM; s++) {
len[s] = 0;
if (freq[s]) leaf_sym[n++] = s;
}
if (n == 0) return;
if (n == 1) { len[leaf_sym[0]] = 1; return; }
/* Leaves sorted ascending by freq (insertion sort, n ≤ 256). */
for (int i = 1; i < n; i++) {
int t = leaf_sym[i];
int j = i - 1;
while (j >= 0 && freq[leaf_sym[j]] > freq[t]) {
leaf_sym[j + 1] = leaf_sym[j];
j--;
}
leaf_sym[j + 1] = t;
}
/* Two-queue Huffman: leaves (sorted) + internal nodes (created in
* nondecreasing weight order). Nodes 0..n-1 are leaves; n.. are
* internal. 2n-1 511 nodes total. */
uint64_t w[2 * NSYM];
int16_t parent[2 * NSYM];
for (int i = 0; i < n; i++) { w[i] = freq[leaf_sym[i]]; parent[i] = -1; }
int q1 = 0; /* next unconsumed leaf */
int q2 = n; /* next unconsumed internal node */
int nn = n; /* next node id to create */
for (int made = 0; made < n - 1; made++) {
int a, b;
/* pick two smallest among q1-front and q2-front */
a = (q2 >= nn || (q1 < n && w[q1] <= w[q2])) ? q1++ : q2++;
b = (q2 >= nn || (q1 < n && w[q1] <= w[q2])) ? q1++ : q2++;
w[nn] = w[a] + w[b];
parent[nn] = -1;
parent[a] = (int16_t)nn;
parent[b] = (int16_t)nn;
nn++;
}
/* Depth of each node = depth(parent) + 1; parents always have
* higher ids, so one reverse pass suffices. */
uint8_t depth[2 * NSYM];
depth[nn - 1] = 0;
for (int i = nn - 2; i >= 0; i--)
depth[i] = (uint8_t)(depth[parent[i]] + 1);
for (int i = 0; i < n; i++)
len[leaf_sym[i]] = depth[i] ? depth[i] : 1;
}
/* log2(v) in 1/256 units via ilog2 + linear mantissa interpolation
* (max error ~0.09 bits fine for candidate selection). */
static inline uint32_t log2_fp8(uint32_t v) {
int t = ilog2(v);
uint32_t mant = ((v << 8) >> t); /* in [256, 512) */
return (uint32_t)t * 256u + (mant - 256u);
}
/* ═══════════════════════════════════════════════════════════════ /* ═══════════════════════════════════════════════════════════════
* ENCODE SEQUENCES * ENCODE SEQUENCES
* *
@ -1529,12 +1646,26 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
const uint32_t *ml_base_tab, const uint32_t *ml_base_tab,
int disable_huf4) { int disable_huf4) {
if (!tok_len) { *dst_len = 0; return VVA_OK; } if (!tok_len) { *dst_len = 0; return VVA_OK; }
/* SPRINT 126: API-misuse guard. Every internal caller passes one
* block's tokens (<= ~1.13 MB), but this entry point is public;
* bound tok_len so the arena size arithmetic below cannot wrap on
* absurd direct-API inputs. 1 GiB is orders of magnitude above any
* legal block token stream. */
if (tok_len > ((size_t)1 << 30)) return VVA_ERR_PARAM;
/* Parse into sequences. /* Parse into sequences.
* PERF: one combined alloc for seqs + lit_buf. The sizeof(seq_t) * PERF: one combined alloc for seqs + lit_buf. The sizeof(seq_t)
* is 4 bytes so natural alignment for both is satisfied. Saves * is 4 bytes so natural alignment for both is satisfied. Saves
* 1 malloc/free pair per call. */ * 1 malloc/free pair per call. */
size_t max_seqs = tok_len; /* Upper bound */ /* SPRINT 125: tight sequence-count bound. Every sequence with a
* match consumes >= 3 token bytes (1 token byte + 2-3 offset bytes);
* zero-match sequences arise only from the final literal-only token
* (<= 1) and from LL_MAX splits (<= total_lits/65535 <=
* tok_len/65535). The old bound (max_seqs = tok_len) allocated
* 16 bytes of seq_t per TOKEN BYTE ~17 MB of scratch per 1 MB
* block; this bound cuts that ~3x. parse_sequences fails closed if
* the bound were ever wrong (truncation guard). */
size_t max_seqs = tok_len / 3 + tok_len / 65535 + 8;
size_t seqs_sz = max_seqs * sizeof(seq_t); size_t seqs_sz = max_seqs * sizeof(seq_t);
size_t total_scratch = seqs_sz + tok_len; size_t total_scratch = seqs_sz + tok_len;
uint8_t *base_scratch = (uint8_t *)malloc(total_scratch); uint8_t *base_scratch = (uint8_t *)malloc(total_scratch);
@ -1550,10 +1681,35 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
size_t nseq = parse_sequences(tokens, tok_len, lit_buf, tok_len, seqs, max_seqs, &total_lits, off_bytes, min_match); size_t nseq = parse_sequences(tokens, tok_len, lit_buf, tok_len, seqs, max_seqs, &total_lits, off_bytes, min_match);
if (nseq == 0) { free(base_scratch); return VVA_ERR_CORRUPT; } if (nseq == 0) { free(base_scratch); return VVA_ERR_CORRUPT; }
/* ─── Encode literals with 4-way ANS ─── */ /* ─── SPRINT 126: one block-scratch arena ───
*
* After parse_sequences, nseq and total_lits pin every remaining
* scratch size, so the 6 per-block mallocs that used to follow
* (lit_enc, seq_scratch memoization arrays, LL build tables, ML/OF
* build tables, the bitpair staging array, and the sequence
* bitstream) collapse into ONE allocation with computed offsets
* one malloc/free pair per block instead of six, and one cleanup
* pointer on every error path. Layout keeps 4/8-byte-aligned
* sections first; sizes are the exact bounds the individual
* allocations used. ML/OF tables are reserved unconditionally
* (40 KB) even when match_count == 0 a bound, not a leak. */
size_t lit_cap = vva_bound(total_lits); size_t lit_cap = vva_bound(total_lits);
uint8_t *lit_enc = (uint8_t *)malloc(lit_cap); size_t a_codes_sz = (nseq * sizeof(uint8_t) + 3) & ~(size_t)3;
if (!lit_enc) { free(base_scratch); return VVA_ERR_NOMEM; } size_t a_stream_sz = a_codes_sz + nseq * sizeof(uint32_t) + nseq * sizeof(int);
size_t tab_one_sz = ANS_L + ANS_L * sizeof(vva_dec_entry_t);
#define VVA_A8(x) (((x) + 7) & ~(size_t)7)
size_t off_pairs = 0;
size_t off_scratch = off_pairs + VVA_A8(nseq * 6 * sizeof(bitpair_t));
size_t off_lltab = off_scratch + VVA_A8(3 * a_stream_sz);
size_t off_mloftab = off_lltab + VVA_A8(tab_one_sz);
size_t off_lit = off_mloftab + VVA_A8(2 * tab_one_sz);
size_t off_bs = off_lit + VVA_A8(lit_cap);
size_t arena_sz = off_bs + VVA_A8(nseq * 6 * 4 + 16);
uint8_t *arena = (uint8_t *)malloc(arena_sz);
if (!arena) { free(base_scratch); return VVA_ERR_NOMEM; }
/* ─── Encode literals with 4-way ANS ─── */
uint8_t *lit_enc = arena + off_lit;
size_t lit_enc_len = 0; size_t lit_enc_len = 0;
uint8_t lit_fmt = 0; /* 0=raw, 1=ANS4, 2=ANS1, 3=Huffman, 4=Huffman4 (Sprint 104) */ uint8_t lit_fmt = 0; /* 0=raw, 1=ANS4, 2=ANS1, 3=Huffman, 4=Huffman4 (Sprint 104) */
@ -1582,6 +1738,99 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
* unchanged otherwise existing decoders reject lit_fmt={3,4} * unchanged otherwise existing decoders reject lit_fmt={3,4}
* with VVA_ERR_CORRUPT, so this is a decoder-incompatible * with VVA_ERR_CORRUPT, so this is a decoder-incompatible
* format change (requires v2.46.0+ for fmt=3, v2.47+ for fmt=4). */ * format change (requires v2.46.0+ for fmt=3, v2.47+ for fmt=4). */
if (total_lits >= 4096) {
/* ─── SPRINT 124: estimate-based single-encode selection.
*
* One histogram, then analytic sizes: ANS4 cost is the
* table-quantized Σ f·(ANS_LOG log2(norm_f)) plus its
* header; Huffman cost is exact given code lengths (built
* without a bitstream pass). Only the winner is encoded,
* directly into lit_enc. ANS1 is dropped here: it can
* undercut ANS4 by at most ~26 header bytes, which is
* noise at 4096 literals. The old full race burned
* 6-21% of total encode wall on discarded encodes. */
uint32_t hist[NSYM];
memset(hist, 0, sizeof(hist));
for (size_t i = 0; i < total_lits; i++) hist[lit_buf[i]]++;
int active = 0, max_sym = 0;
for (int s = 0; s < NSYM; s++)
if (hist[s]) { active++; max_sym = s; }
uint16_t norm_est[NSYM];
memset(norm_est, 0, sizeof(norm_est));
normalize_freq(hist, norm_est);
uint64_t bits256 = 0;
for (int s = 0; s < NSYM; s++) {
if (!hist[s]) continue;
uint32_t nf = norm_est[s] ? norm_est[s] : 1;
bits256 += (uint64_t)hist[s] *
((uint32_t)ANS_LOG * 256u - log2_fp8(nf));
}
size_t tbl_hdr = (active <= 64) ? (size_t)(2 + 3 * active)
: (size_t)(2 + 2 * (max_sym + 1));
size_t ans4_est = (size_t)(bits256 / 2048u) + tbl_hdr + 26;
uint8_t hlen[NSYM];
est_huff_lengths(hist, hlen);
uint64_t hbits = 0;
for (int s = 0; s < NSYM; s++)
hbits += (uint64_t)hist[s] * hlen[s];
size_t huf_est = (size_t)(hbits / 8u) + 130;
size_t huf4_est = huf_est + 12;
/* Two-finalist race with estimate-gated skips.
*
* The estimates are systematically OPTIMISTIC (linear log2
* interpolation undershoots; tANS state costs and lane
* overheads are approximated low), so `est >= raw` proves
* the real encode cannot beat raw literals a safe skip
* that turns incompressible-literal blocks (sensor data)
* into an immediate raw store with zero encode passes.
* When a candidate is plausible it is actually encoded:
* measured sizes decide, exactly like the old 4-way race,
* but with at most 2 encodes (ANS1 dropped bounded
* ~26 B win; huf-vs-huf4 resolved by their fixed ~12 B
* structural delta instead of dual encodes). */
uint8_t hb_fmt = disable_huf4 ? 3 : 4;
size_t hb_est = disable_huf4 ? huf_est : huf4_est;
if (!disable_huf4 && huf_est + 32 < huf4_est) {
hb_fmt = 3; hb_est = huf_est;
}
lit_fmt = 0;
lit_enc_len = 0;
if (ans4_est < total_lits) {
size_t out_len = 0;
if (vva_encode4(lit_buf, total_lits, lit_enc, lit_cap, &out_len) == VVA_OK &&
out_len < total_lits) {
lit_enc_len = out_len;
lit_fmt = 1;
}
}
if (hb_est < total_lits &&
(lit_fmt == 0 || hb_est < lit_enc_len + lit_enc_len / 8)) {
uint8_t *alt_buf = (uint8_t *)malloc(lit_cap);
if (alt_buf) {
size_t alt_len = 0;
int aok = (hb_fmt == 4)
? (vvh_encode4(lit_buf, total_lits, alt_buf, lit_cap, &alt_len) == VVH_OK)
: (vvh_encode(lit_buf, total_lits, alt_buf, lit_cap, &alt_len) == VVH_OK);
if (aok && alt_len < total_lits &&
(lit_fmt == 0 || alt_len < lit_enc_len)) {
memcpy(lit_enc, alt_buf, alt_len);
lit_enc_len = alt_len;
lit_fmt = hb_fmt;
}
free(alt_buf);
}
}
if (lit_fmt == 0) {
/* Raw literals (lit_cap = vva_bound(total_lits) ≥ total_lits). */
memcpy(lit_enc, lit_buf, total_lits);
lit_enc_len = total_lits;
}
} else {
size_t ans4_len = 0, ans1_len = 0, huf_len = 0, huf4_len = 0; size_t ans4_len = 0, ans1_len = 0, huf_len = 0, huf4_len = 0;
uint8_t *ans4_buf = (uint8_t *)malloc(lit_cap); uint8_t *ans4_buf = (uint8_t *)malloc(lit_cap);
uint8_t *ans1_buf = (uint8_t *)malloc(lit_cap); uint8_t *ans1_buf = (uint8_t *)malloc(lit_cap);
@ -1654,6 +1903,7 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
lit_fmt = 0; lit_fmt = 0;
} }
free(ans4_buf); free(ans1_buf); free(huf_buf); free(huf4_buf); free(ans4_buf); free(ans1_buf); free(huf_buf); free(huf4_buf);
}
} }
/* ─── Count ML, OF, and LL code frequencies ─── */ /* ─── Count ML, OF, and LL code frequencies ─── */
@ -1687,16 +1937,11 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
* Net cost: 1 extra malloc region (~14 × nseq bytes), 0 extra * Net cost: 1 extra malloc region (~14 × nseq bytes), 0 extra
* malloc calls. Net saving: the backward pass becomes lookups * malloc calls. Net saving: the backward pass becomes lookups
* instead of re-computation. */ * instead of re-computation. */
size_t codes_sz = (nseq * sizeof(uint8_t) + 3) & ~(size_t)3; size_t codes_sz = a_codes_sz;
size_t extra_sz = nseq * sizeof(uint32_t); size_t extra_sz = nseq * sizeof(uint32_t);
size_t nbits_sz = nseq * sizeof(int); /* 3 streams × (codes + extra + nbits) — carved from the arena. */
/* 3 streams × (codes + extra + nbits) */ uint8_t *seq_scratch = arena + off_scratch;
uint8_t *seq_scratch = (uint8_t *)malloc(3 * (codes_sz + extra_sz + nbits_sz)); size_t stream_sz = a_stream_sz;
if (!seq_scratch) {
free(base_scratch); free(lit_enc);
return VVA_ERR_NOMEM;
}
size_t stream_sz = codes_sz + extra_sz + nbits_sz;
uint8_t *seq_of_code = seq_scratch; uint8_t *seq_of_code = seq_scratch;
uint32_t *seq_of_extra = (uint32_t *)(seq_scratch + codes_sz); uint32_t *seq_of_extra = (uint32_t *)(seq_scratch + codes_sz);
int *seq_of_nbits = (int *)(seq_scratch + codes_sz + extra_sz); int *seq_of_nbits = (int *)(seq_scratch + codes_sz + extra_sz);
@ -1799,23 +2044,20 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
ll_hdr_sz = write_hdr_v2(norm_ll, ll_hdr_buf, 600); ll_hdr_sz = write_hdr_v2(norm_ll, ll_hdr_buf, 600);
if (!ll_hdr_sz) goto seq_fail; if (!ll_hdr_sz) goto seq_fail;
/* PERF: one combined alloc for sp_ll + dec_ll. sp_ll lives in /* sp_ll lives in the first ANS_L bytes of the arena's LL-table
* the first ANS_L bytes, dec_ll follows with alignment (16-byte * section, dec_ll follows (ANS_L=4096 keeps dec_ll aligned). */
* aligned vs 8-byte reads is satisfied since ANS_L=4096 is
* already 4KB-aligned). Saves 1 malloc/free pair. */
size_t sp_sz = ANS_L; size_t sp_sz = ANS_L;
size_t dec_sz = ANS_L * sizeof(vva_dec_entry_t); uint8_t *ll_tables = arena + off_lltab;
uint8_t *ll_tables = (uint8_t *)malloc(sp_sz + dec_sz);
if (!ll_tables) goto seq_fail;
uint8_t *sp_ll = ll_tables; uint8_t *sp_ll = ll_tables;
vva_dec_entry_t *dec_ll = (vva_dec_entry_t *)(ll_tables + sp_sz); vva_dec_entry_t *dec_ll = (vva_dec_entry_t *)(ll_tables + sp_sz);
spread_symbols(norm_ll, sp_ll); spread_symbols(norm_ll, sp_ll);
build_dec(norm_ll, sp_ll, dec_ll); build_dec(norm_ll, sp_ll, dec_ll);
enc_ll_ctx = build_enc(norm_ll, sp_ll, dec_ll); enc_ll_ctx = build_enc(norm_ll, sp_ll, dec_ll);
free(ll_tables);
if (!enc_ll_ctx) goto seq_fail; if (!enc_ll_ctx) goto seq_fail;
} }
enc_ctx_t *enc_ml_ctx = NULL;
enc_ctx_t *enc_of_ctx = NULL;
if (match_count > 0) { if (match_count > 0) {
/* Treat ML codes as a small-alphabet problem */ /* Treat ML codes as a small-alphabet problem */
uint32_t raw_ml[NSYM], raw_of[NSYM]; uint32_t raw_ml[NSYM], raw_of[NSYM];
@ -1834,14 +2076,10 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
/* ─── Build encode tables ─── /* ─── Build encode tables (in the arena's ML/OF section) ─── */
* PERF: one combined alloc for sp_ml + dec_ml + sp_of + dec_of
* (4 fixed-size ANS_L-based buffers). Saves 3 malloc/free pairs. */
size_t sp_sz = ANS_L; size_t sp_sz = ANS_L;
size_t dec_sz = ANS_L * sizeof(vva_dec_entry_t); size_t dec_sz = ANS_L * sizeof(vva_dec_entry_t);
size_t combo_sz = (sp_sz + dec_sz) * 2; uint8_t *ml_of_tables = arena + off_mloftab;
uint8_t *ml_of_tables = (uint8_t *)malloc(combo_sz);
if (!ml_of_tables) goto seq_fail;
uint8_t *sp_ml = ml_of_tables; uint8_t *sp_ml = ml_of_tables;
vva_dec_entry_t *dec_ml = (vva_dec_entry_t *)(ml_of_tables + sp_sz); vva_dec_entry_t *dec_ml = (vva_dec_entry_t *)(ml_of_tables + sp_sz);
uint8_t *sp_of = ml_of_tables + sp_sz + dec_sz; uint8_t *sp_of = ml_of_tables + sp_sz + dec_sz;
@ -1849,23 +2087,36 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
spread_symbols(norm_ml, sp_ml); spread_symbols(norm_ml, sp_ml);
build_dec(norm_ml, sp_ml, dec_ml); build_dec(norm_ml, sp_ml, dec_ml);
enc_ctx_t *enc_ml_ctx = build_enc(norm_ml, sp_ml, dec_ml); enc_ml_ctx = build_enc(norm_ml, sp_ml, dec_ml);
spread_symbols(norm_of, sp_of); spread_symbols(norm_of, sp_of);
build_dec(norm_of, sp_of, dec_of); build_dec(norm_of, sp_of, dec_of);
enc_ctx_t *enc_of_ctx = build_enc(norm_of, sp_of, dec_of); enc_of_ctx = build_enc(norm_of, sp_of, dec_of);
free(ml_of_tables);
if (!enc_ml_ctx || !enc_of_ctx) { if (!enc_ml_ctx || !enc_of_ctx) {
free_enc(enc_ml_ctx); free_enc(enc_of_ctx); free_enc(enc_ml_ctx); free_enc(enc_of_ctx);
enc_ml_ctx = enc_of_ctx = NULL;
goto seq_fail; goto seq_fail;
} }
}
/* ─── Encode ML/OF codes + extra bits in reverse ─── */ /* ─── Encode ML/OF/LL codes + extra bits in reverse ───
/* Collect bitpairs for ANS-coded symbols + raw extra bits */ *
size_t pair_cap = nseq * 6; /* 3 ANS + 3 extra max per seq */ * SPRINT 124 (latent-corruption fix): this section including the
bitpair_t *pairs = (bitpair_t *)malloc(pair_cap * sizeof(bitpair_t)); * LL encoding used to live INSIDE the match_count > 0 branch. A
if (!pairs) { free_enc(enc_ml_ctx); free_enc(enc_of_ctx); goto seq_fail; } * block whose token stream contains no matches at all (pure
* literal run) then wrote the LL table header but NO sequence
* bitstream, while the decoder unconditionally decodes an LL code
* per sequence it read garbage from an empty stream and failed
* (or worse, produced short output). The case was unreachable
* while emit_block sent every csz >= braw token stream straight
* to RAW storage; the relaxed raw_gate made it reachable. The LL
* bitstream must be written whenever nseq > 0, with ML/OF work
* still gated per-sequence on matchlen > 0. */
{
/* Collect bitpairs for ANS-coded symbols + raw extra bits
* (arena section; capacity nseq * 6 = 3 ANS + 3 extra per seq). */
bitpair_t *pairs = (bitpair_t *)(arena + off_pairs);
state_ml = 0; state_of = 0; state_ll = 0; state_ml = 0; state_of = 0; state_ll = 0;
size_t npairs = 0; size_t npairs = 0;
@ -1904,7 +2155,7 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
uint32_t bv; int bn; uint32_t bv; int bn;
int slot = enc_sym(enc_ml_ctx, state_ml, mc, &bv, &bn); int slot = enc_sym(enc_ml_ctx, state_ml, mc, &bv, &bn);
if (slot < 0) { if (slot < 0) {
free(pairs); free_enc(enc_ml_ctx); free_enc(enc_of_ctx); free_enc(enc_ml_ctx); free_enc(enc_of_ctx);
goto seq_fail; goto seq_fail;
} }
pairs[npairs].val = (uint32_t)bv; pairs[npairs].val = (uint32_t)bv;
@ -1925,7 +2176,7 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
uint32_t bv; int bn; uint32_t bv; int bn;
int slot = enc_sym(enc_of_ctx, state_of, oc, &bv, &bn); int slot = enc_sym(enc_of_ctx, state_of, oc, &bv, &bn);
if (slot < 0) { if (slot < 0) {
free(pairs); free_enc(enc_ml_ctx); free_enc(enc_of_ctx); free_enc(enc_ml_ctx); free_enc(enc_of_ctx);
goto seq_fail; goto seq_fail;
} }
pairs[npairs].val = (uint32_t)bv; pairs[npairs].val = (uint32_t)bv;
@ -1950,7 +2201,7 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
uint32_t bv; int bn; uint32_t bv; int bn;
int slot = enc_sym(enc_ll_ctx, state_ll, lc, &bv, &bn); int slot = enc_sym(enc_ll_ctx, state_ll, lc, &bv, &bn);
if (slot < 0) { if (slot < 0) {
free(pairs); free_enc(enc_ml_ctx); free_enc(enc_of_ctx); free_enc(enc_ml_ctx); free_enc(enc_of_ctx);
goto seq_fail; goto seq_fail;
} }
pairs[npairs].val = (uint32_t)bv; pairs[npairs].val = (uint32_t)bv;
@ -1967,15 +2218,13 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
* Each pair is up to 32 bits (ANS slot = 14 bits + extra up to 18). * Each pair is up to 32 bits (ANS slot = 14 bits + extra up to 18).
* Allocate 4 bytes per pair + 16-byte safety margin. */ * Allocate 4 bytes per pair + 16-byte safety margin. */
size_t bs_cap = npairs * 4 + 16; size_t bs_cap = npairs * 4 + 16;
seq_bs = (uint8_t *)malloc(bs_cap); seq_bs = arena + off_bs; /* arena section, sized nseq*6*4 + 16 >= bs_cap */
if (!seq_bs) { free(pairs); goto seq_fail; }
ans_bw_t w; ans_bw_t w;
ans_bw_init(&w, seq_bs, bs_cap); ans_bw_init(&w, seq_bs, bs_cap);
for (size_t i = npairs; i > 0; i--) for (size_t i = npairs; i > 0; i--)
ans_bw_add(&w, pairs[i - 1].val, pairs[i - 1].nb); ans_bw_add(&w, pairs[i - 1].val, pairs[i - 1].nb);
seq_bs_len = ans_bw_flush(&w); seq_bs_len = ans_bw_flush(&w);
free(pairs);
} }
/* Litlens are now ANS-coded in the sequence bitstream — no varints needed */ /* Litlens are now ANS-coded in the sequence bitstream — no varints needed */
@ -2033,17 +2282,15 @@ static vva_error_t vva_encode_sequences_impl(const uint8_t *tokens, size_t tok_l
*dst_len = (size_t)(op - dst); *dst_len = (size_t)(op - dst);
} }
free(base_scratch); free(lit_enc); free(base_scratch);
free(seq_scratch); free(arena);
free_enc(enc_ll_ctx); free_enc(enc_ll_ctx);
free(seq_bs);
return VVA_OK; return VVA_OK;
seq_fail: seq_fail:
free(base_scratch); free(lit_enc); free(base_scratch);
free(seq_scratch); free(arena);
free_enc(enc_ll_ctx); free_enc(enc_ll_ctx);
free(seq_bs);
return VVA_ERR_OVERFLOW; return VVA_ERR_OVERFLOW;
} }
@ -2125,7 +2372,14 @@ vva_error_t vva_decode_sequences_impl(const uint8_t *src, size_t src_len,
if (VV_UNLIKELY(total_lits > dst_cap)) return VVA_ERR_CORRUPT; if (VV_UNLIKELY(total_lits > dst_cap)) return VVA_ERR_CORRUPT;
/* Decode literals based on format byte */ /* Decode literals based on format byte */
uint8_t *lit_buf = (uint8_t *)malloc(total_lits + 16); /* SPRINT 126: one allocation for the literal buffer AND the decode
* tables (previously 2 mallocs; the table section was itself fused
* from 4 in Sprint 125). The table space (52 KB) is reserved
* unconditionally up front so the whole block scratch is a single
* malloc/free its exact use is decided at table-build below. */
size_t lit_sec = (total_lits + 16 + 7) & ~(size_t)7;
size_t tab_sec = ANS_L + 3 * (ANS_L * sizeof(vva_dec_entry_t));
uint8_t *lit_buf = (uint8_t *)malloc(lit_sec + tab_sec);
if (!lit_buf) return VVA_ERR_NOMEM; if (!lit_buf) return VVA_ERR_NOMEM;
if (total_lits > 0 && lit_enc_len > 0) { if (total_lits > 0 && lit_enc_len > 0) {
@ -2208,6 +2462,46 @@ vva_error_t vva_decode_sequences_impl(const uint8_t *src, size_t src_len,
if (ll_hdr_sz > 0) read_hdr_v2(p, ll_hdr_sz, norm_ll); if (ll_hdr_sz > 0) read_hdr_v2(p, ll_hdr_sz, norm_ll);
p += ll_hdr_sz; p += ll_hdr_sz;
/* SPRINT 125: hoisted table validation. Two invariants are enforced
* once per block so the old per-sequence `code >= VVA_*_CODES`
* branch (one per iteration, on the critical path between the table
* load and the bit read) becomes tautological and is removed from
* the hot loop below:
*
* (1) No out-of-range symbol has nonzero frequency bounds every
* spread-table entry's symbol.
* (2) Frequencies sum to exactly ANS_L guarantees spread_symbols
* fills ALL 4096 slots. Without this, a corrupt underfull
* header leaves stale scratch bytes in unfilled slots, whose
* "symbols" bypass check (1) entirely (caught by UBSan as an
* OOB index into ll_extra[36] during validation of this very
* change). normalize_freq guarantees sum == ANS_L on every
* valid stream, so this rejects only corrupt input.
*
* This is STRICTER than the old per-sequence check: malformed
* tables are rejected up front instead of only when a decode path
* lands on a bad entry. Tables that the decode loop never consults
* (ML/OF when match_count == 0; all of them when the loop body
* cannot run) are exempt from (2) for wire compatibility. */
{
uint32_t sum_ml = 0, sum_of = 0, sum_ll = 0;
for (int s = 0; s < NSYM; s++) {
sum_ml += norm_ml[s]; sum_of += norm_of[s]; sum_ll += norm_ll[s];
if (s >= VVA_OF_CODES && VV_UNLIKELY(norm_of[s])) {
free(lit_buf); return VVA_ERR_CORRUPT;
}
if (s >= VVA_ML_CODES && VV_UNLIKELY(norm_ml[s] | norm_ll[s])) {
free(lit_buf); return VVA_ERR_CORRUPT;
}
}
if (VV_UNLIKELY(sum_ll != ANS_L && (total_lits > 0 || match_count > 0))) {
free(lit_buf); return VVA_ERR_CORRUPT;
}
if (VV_UNLIKELY(match_count > 0 && (sum_ml != ANS_L || sum_of != ANS_L))) {
free(lit_buf); return VVA_ERR_CORRUPT;
}
}
/* Read initial states */ /* Read initial states */
if (p + 6 > end) { free(lit_buf); return VVA_ERR_CORRUPT; } if (p + 6 > end) { free(lit_buf); return VVA_ERR_CORRUPT; }
uint32_t state_ml = (uint32_t)p[0] | ((uint32_t)p[1] << 8); p += 2; uint32_t state_ml = (uint32_t)p[0] | ((uint32_t)p[1] << 8); p += 2;
@ -2228,34 +2522,32 @@ vva_error_t vva_decode_sequences_impl(const uint8_t *src, size_t src_len,
* NULL-deref's dec_of and dec_ml. Found by libFuzzer + ASan. * NULL-deref's dec_of and dec_ml. Found by libFuzzer + ASan.
* Fix: always allocate all 3 tables. The decode-loop dereferences * Fix: always allocate all 3 tables. The decode-loop dereferences
* are safe because state masks bound the index to ANS_L. */ * are safe because state masks bound the index to ANS_L. */
/* SPRINT 125: one allocation for the spread scratch + decode tables
* (previously 4 separate mallocs measurable on small blocks).
* When match_count == 0, the ML/OF tables are never consulted for
* real decode work (the loop `continue`s before the OF/ML reads),
* but the ILP eager-loads at the loop top still index them alias
* them to the LL table: valid, initialized memory, zero build and
* zero memset cost (replaces two 16 KB sentinel memsets). */
vva_dec_entry_t *dec_ml = NULL, *dec_of = NULL, *dec_ll = NULL; vva_dec_entry_t *dec_ml = NULL, *dec_of = NULL, *dec_ll = NULL;
{ {
uint8_t *sp_tmp = (uint8_t *)malloc(ANS_L); size_t dec_sz = ANS_L * sizeof(vva_dec_entry_t);
dec_ml = (vva_dec_entry_t *)malloc(ANS_L * sizeof(vva_dec_entry_t)); uint8_t *seq_tables = lit_buf + lit_sec; /* reserved above */
dec_of = (vva_dec_entry_t *)malloc(ANS_L * sizeof(vva_dec_entry_t)); uint8_t *sp_tmp = seq_tables;
dec_ll = (vva_dec_entry_t *)malloc(ANS_L * sizeof(vva_dec_entry_t)); dec_ll = (vva_dec_entry_t *)(seq_tables + ANS_L);
if (!sp_tmp || !dec_ll || !dec_ml || !dec_of) { spread_symbols(norm_ll, sp_tmp);
free(sp_tmp); free(dec_ml); free(dec_of); free(dec_ll); free(lit_buf); build_dec(norm_ll, sp_tmp, dec_ll);
return VVA_ERR_NOMEM;
}
if (match_count > 0) { if (match_count > 0) {
dec_ml = (vva_dec_entry_t *)(seq_tables + ANS_L + dec_sz);
dec_of = (vva_dec_entry_t *)(seq_tables + ANS_L + 2 * dec_sz);
spread_symbols(norm_ml, sp_tmp); spread_symbols(norm_ml, sp_tmp);
build_dec(norm_ml, sp_tmp, dec_ml); build_dec(norm_ml, sp_tmp, dec_ml);
spread_symbols(norm_of, sp_tmp); spread_symbols(norm_of, sp_tmp);
build_dec(norm_of, sp_tmp, dec_of); build_dec(norm_of, sp_tmp, dec_of);
} else { } else {
/* Initialize ml/of tables to safe sentinel values so any dec_ml = dec_ll;
* unintended read (e.g., the ILP eager-load in the decode dec_of = dec_ll;
* loop when match_count == 0) returns predictable data
* rather than dereferencing uninitialized memory. The
* loop guard prevents these values from being used in
* actual sequence reconstruction. */
memset(dec_ml, 0, ANS_L * sizeof(vva_dec_entry_t));
memset(dec_of, 0, ANS_L * sizeof(vva_dec_entry_t));
} }
spread_symbols(norm_ll, sp_tmp);
build_dec(norm_ll, sp_tmp, dec_ll);
free(sp_tmp);
} }
/* Initialize bitstream reader for sequence data */ /* Initialize bitstream reader for sequence data */
@ -2307,7 +2599,8 @@ vva_error_t vva_decode_sequences_impl(const uint8_t *src, size_t src_len,
* guarantees litlen+matchlen fit without per-iter overflow checking. * guarantees litlen+matchlen fit without per-iter overflow checking.
* (Reserving only ONE run let a crafted final sequence write up to * (Reserving only ONE run let a crafted final sequence write up to
* 65535 bytes past op_end a heap overflow; the +64 caller slack was * 65535 bytes past op_end a heap overflow; the +64 caller slack was
* far too small to absorb it.) * far too small to absorb it. ZUPT AUDIT FIX, carried across codec
* re-vendors until upstreamed.)
* *
* SPRINT 46: raised from 1<<20 to 1<<24. The 3-byte offset wire * SPRINT 46: raised from 1<<20 to 1<<24. The 3-byte offset wire
* encoding (off_bytes==3 for wlog>16) represents offsets up to * encoding (off_bytes==3 for wlog>16) represents offsets up to
@ -2361,7 +2654,7 @@ vva_error_t vva_decode_sequences_impl(const uint8_t *src, size_t src_len,
size_t iter_count = 0; size_t iter_count = 0;
while (lit_pos < total_lits || matches_decoded < match_count) { while (lit_pos < total_lits || matches_decoded < match_count) {
if (VV_UNLIKELY(++iter_count > max_iters)) { if (VV_UNLIKELY(++iter_count > max_iters)) {
free(dec_ml); free(dec_of); free(dec_ll); free(lit_buf); free(lit_buf);
return VVA_ERR_CORRUPT; return VVA_ERR_CORRUPT;
} }
/* PERF: issue all 3 ANS table lookups early so CPU can overlap /* PERF: issue all 3 ANS table lookups early so CPU can overlap
@ -2391,33 +2684,13 @@ vva_error_t vva_decode_sequences_impl(const uint8_t *src, size_t src_len,
vva_dec_entry_t eof = dec_of[state_of & (ANS_L - 1)]; vva_dec_entry_t eof = dec_of[state_of & (ANS_L - 1)];
vva_dec_entry_t eml = dec_ml[state_ml & (ANS_L - 1)]; vva_dec_entry_t eml = dec_ml[state_ml & (ANS_L - 1)];
/* SPRINT 27 (v2.50.1): combine the 3 per-iteration OOB code /* SPRINT 125: the per-iteration OOB code check (Sprint 27's
* validators into 1 branch. Previously each of ll_code, of_code, * combined branch) is gone table symbols are validated once
* ml_code had a separate `if (VV_UNLIKELY(code >= MAX)) return` * at header-parse time above, so every entry in dec_ll/dec_of/
* three predicted-not-taken branches per iteration. ORing * dec_ml carries an in-range symbol by construction. Same
* the three bool comparisons into a single mask lets the compiler * security property (out-of-range codes on corrupt input are
* use one branch and parallel SIMD-style comparisons. * rejected, now earlier and unconditionally), one branch less
* * on the critical path between the table load and the bit read. */
* Found via profile-driven analysis on v2.50.0 (Sprint 27). The
* three branches were each individually cheap when not taken,
* but they sit on the critical path between the table-read
* latency (L1/L2 miss on the random-walk index) and the
* subsequent bit-read, where they delay state-update of the
* NEXT iteration. Folding to one branch removes 2 branch slots
* and lets the comparator ALU run in parallel with the load
* latency for ell/eof/eml.
*
* Note: VVA_LL_CODES == VVA_ML_CODES == 36, VVA_OF_CODES == 27.
* Use the strictest bound (27) as a quick-fail mask; codes 27-35
* are still legal for LL/ML and fall through to the per-code
* tail check below. This catches the most common adversarial
* encoding (high-symbol garbage) at zero cost on the common path. */
if (VV_UNLIKELY(((unsigned)ell.symbol >= VVA_LL_CODES) |
((unsigned)eof.symbol >= VVA_OF_CODES) |
((unsigned)eml.symbol >= VVA_ML_CODES))) {
free(dec_ml); free(dec_of); free(dec_ll); free(lit_buf);
return VVA_ERR_CORRUPT;
}
/* ── Decode LL: state, extra, final litlen ── */ /* ── Decode LL: state, extra, final litlen ── */
uint32_t ll_bits = ans_br_read(&r, ell.nbits); uint32_t ll_bits = ans_br_read(&r, ell.nbits);
@ -2428,11 +2701,11 @@ vva_error_t vva_decode_sequences_impl(const uint8_t *src, size_t src_len,
size_t litlen = ll_decode(ll_code, ll_extra_val); size_t litlen = ll_decode(ll_code, ll_extra_val);
if (VV_UNLIKELY(lit_pos + litlen > total_lits)) { if (VV_UNLIKELY(lit_pos + litlen > total_lits)) {
free(dec_ml); free(dec_of); free(dec_ll); free(lit_buf); free(lit_buf);
return VVA_ERR_CORRUPT; return VVA_ERR_CORRUPT;
} }
if (VV_UNLIKELY(!in_safe_zone && op + litlen > op_end)) { if (VV_UNLIKELY(!in_safe_zone && op + litlen > op_end)) {
free(dec_ml); free(dec_of); free(dec_ll); free(lit_buf); free(lit_buf);
return VVA_ERR_OVERFLOW; return VVA_ERR_OVERFLOW;
} }
if (litlen > 0) { if (litlen > 0) {
@ -2510,15 +2783,15 @@ vva_error_t vva_decode_sequences_impl(const uint8_t *src, size_t src_len,
* op_safe_end = op_end - SAFEZONE_MAX_MATCH, and matchlen is * op_safe_end = op_end - SAFEZONE_MAX_MATCH, and matchlen is
* always SAFEZONE_MAX_MATCH by wire format. */ * always SAFEZONE_MAX_MATCH by wire format. */
if (VV_UNLIKELY(offset == 0 || offset > SAFEZONE_MAX_OFFSET)) { if (VV_UNLIKELY(offset == 0 || offset > SAFEZONE_MAX_OFFSET)) {
free(dec_ml); free(dec_of); free(dec_ll); free(lit_buf); free(lit_buf);
return VVA_ERR_CORRUPT; return VVA_ERR_CORRUPT;
} }
if (VV_UNLIKELY(!in_safe_zone && offset > (uint32_t)(op - dst_base))) { if (VV_UNLIKELY(!in_safe_zone && offset > (uint32_t)(op - dst_base))) {
free(dec_ml); free(dec_of); free(dec_ll); free(lit_buf); free(lit_buf);
return VVA_ERR_CORRUPT; return VVA_ERR_CORRUPT;
} }
if (VV_UNLIKELY(!in_safe_zone && op + matchlen > op_end)) { if (VV_UNLIKELY(!in_safe_zone && op + matchlen > op_end)) {
free(dec_ml); free(dec_of); free(dec_ll); free(lit_buf); free(lit_buf);
return VVA_ERR_OVERFLOW; return VVA_ERR_OVERFLOW;
} }
@ -2605,7 +2878,7 @@ vva_error_t vva_decode_sequences_impl(const uint8_t *src, size_t src_len,
} }
*dst_len = (size_t)(op - dst); *dst_len = (size_t)(op - dst);
free(dec_ml); free(dec_of); free(dec_ll); free(lit_buf); free(lit_buf);
return VVA_OK; return VVA_OK;
} }

View file

@ -195,19 +195,23 @@ decode_block_tokens_impl(
VV_PREFETCH(op + ll - off_raw); VV_PREFETCH(op + ll - off_raw);
} }
if (ll > 0) /* SPRINT 125: wildcopy for the dominant ll <= 14 case. The loop
* guards reserve 48 bytes of readable input (ip < ip_safe; ip has
* advanced by only the 1 token byte since, as ll <= 14 implies no
* extension bytes) and 72 bytes of writable output (op < op_safe;
* op unchanged since entry), so one unconditional 16-byte copy is
* in-bounds and replaces memcpy's branchy variable-size dispatch.
* The extra bytes past ll are overwritten by the next copy. */
if (VV_LIKELY(ll <= 14)) {
memcpy(op, ip, 16);
} else {
memcpy(op, ip, ll); memcpy(op, ip, ll);
}
ip += ll; ip += ll;
op += ll; op += ll;
if (VV_UNLIKELY(ip >= ip_end)) break; if (VV_UNLIKELY(ip >= ip_end)) break;
/* Bound the 2-/3-byte offset read against the compressed-block end.
* Without this a crafted block whose last literal advances ip to
* ip_end-1 (or ip_end-2 for 3-byte offsets) makes vv_read16 / the
* 3-byte load read past the heap buffer. The general/tail decode path
* already carries this guard; the AVX2 fast paths were missing it. */
if (VV_UNLIKELY(ip + off_bytes > ip_end)) return VV_ERR_CORRUPT;
uint32_t offset; uint32_t offset;
if (off_bytes == 2) { if (off_bytes == 2) {
offset = vv_read16(ip); offset = vv_read16(ip);
@ -284,19 +288,23 @@ decode_block_tokens_impl(
VV_PREFETCH(op + ll - off_raw); VV_PREFETCH(op + ll - off_raw);
} }
if (ll > 0) /* SPRINT 125: wildcopy for the dominant ll <= 14 case. The loop
* guards reserve 48 bytes of readable input (ip < ip_safe; ip has
* advanced by only the 1 token byte since, as ll <= 14 implies no
* extension bytes) and 72 bytes of writable output (op < op_safe;
* op unchanged since entry), so one unconditional 16-byte copy is
* in-bounds and replaces memcpy's branchy variable-size dispatch.
* The extra bytes past ll are overwritten by the next copy. */
if (VV_LIKELY(ll <= 14)) {
memcpy(op, ip, 16);
} else {
memcpy(op, ip, ll); memcpy(op, ip, ll);
}
ip += ll; ip += ll;
op += ll; op += ll;
if (VV_UNLIKELY(ip >= ip_end)) break; if (VV_UNLIKELY(ip >= ip_end)) break;
/* Bound the 2-/3-byte offset read against the compressed-block end.
* Without this a crafted block whose last literal advances ip to
* ip_end-1 (or ip_end-2 for 3-byte offsets) makes vv_read16 / the
* 3-byte load read past the heap buffer. The general/tail decode path
* already carries this guard; the AVX2 fast paths were missing it. */
if (VV_UNLIKELY(ip + off_bytes > ip_end)) return VV_ERR_CORRUPT;
uint32_t offset; uint32_t offset;
if (off_bytes == 2) { if (off_bytes == 2) {
offset = vv_read16(ip); offset = vv_read16(ip);

View file

@ -159,10 +159,28 @@ static inline int32_t extend_match(const uint8_t *a, const uint8_t *b,
len += 32; len += 32;
} }
#endif #endif
/* SPRINT 124: 8-byte xor/ctz stride for the post-8 region. This TU
* is deliberately built without -mavx2 (baseline portability), so
* before this loop existed every match longer than 8 bytes extended
* one byte per iteration measured at 7-8% of encode wall on
* long-match corpora. Same technique as the fast path above. */
while (len + 8 <= max_len) {
uint64_t va, vb;
memcpy(&va, a + len, 8);
memcpy(&vb, b + len, 8);
uint64_t x = va ^ vb;
if (x) return len + (__builtin_ctzll(x) >> 3);
len += 8;
}
while (len < max_len && a[len] == b[len]) len++; while (len < max_len && a[len] == b[len]) len++;
return len; return len;
} }
/* Branch-free floor(log2(v)); v=0 maps to 0. */
static inline int enc_ilog2(uint32_t v) {
return 31 - __builtin_clz(v | 1);
}
/* ═══════════════════════════════════════════════════════════════ /* ═══════════════════════════════════════════════════════════════
* MATCHER: hash chain with 5-byte hash + rep-match * MATCHER: hash chain with 5-byte hash + rep-match
* */ * */
@ -512,8 +530,11 @@ chain_match_ex(const matcher_t *m, const uint8_t *data,
/* Pipeline priming: look 4 chain entries ahead. If chain is /* Pipeline priming: look 4 chain entries ahead. If chain is
* short, the prefetches become no-ops (chain entries below limit * short, the prefetches become no-ops (chain entries below limit
* just return -1 or an expired position). */ * just return -1 or an expired position).
if (ref >= limit && ref < pos) { * SPRINT 124: only prime for deep walks. At depth 4 (fast mode,
* window trial) the priming loads cost more than the misses they
* hide measured 5-8% of fast-mode encode wall. */
if (depth >= 8 && ref >= limit && ref < pos) {
__builtin_prefetch(data + ref, 0, 0); __builtin_prefetch(data + ref, 0, 0);
int32_t r1 = chain_arr[ref & chain_mask]; int32_t r1 = chain_arr[ref & chain_mask];
if (r1 >= limit && r1 < pos) { if (r1 >= limit && r1 < pos) {
@ -562,6 +583,22 @@ chain_match_ex(const matcher_t *m, const uint8_t *data,
if (max > (int32_t)m->max_match) max = (int32_t)m->max_match; if (max > (int32_t)m->max_match) max = (int32_t)m->max_match;
int32_t len = 4 + extend_match(data + pos + 4, data + ref + 4, max - 4); int32_t len = 4 + extend_match(data + pos + 4, data + ref + 4, max - 4);
if (len > best_len) { if (len > best_len) {
/* SPRINT 124: offset-cost-aware acceptance. The walk
* goes newestoldest, so a later candidate always has
* a larger offset. SEQ codes offsets as log2 buckets +
* extra bits, so the farther match costs ~dbits more;
* each extra matched byte saves ~6 bits of literals.
* Without this check a barely-longer match at 512 KB
* displaces a same-ish match at 200 B, and the diverse
* offsets also break rep-offset streaks downstream.
* Only affects greedy/lazy paths the optimal parser
* collects candidates via opt_collect and prices
* offsets itself. */
if (best_len >= 4) {
int dbits = enc_ilog2((uint32_t)(pos - ref))
- enc_ilog2((uint32_t)*best_off);
if ((len - best_len) * 6 < dbits) { ref = next_ref; continue; }
}
best_len = len; best_len = len;
*best_off = pos - ref; *best_off = pos - ref;
if (len >= 256) return best_len; if (len >= 256) return best_len;
@ -604,6 +641,12 @@ chain_match_ex(const matcher_t *m, const uint8_t *data,
if (max > (int32_t)m->max_match) max = (int32_t)m->max_match; if (max > (int32_t)m->max_match) max = (int32_t)m->max_match;
int32_t len = 4 + extend_match(data + pos + 4, data + ref4 + 4, max - 4); int32_t len = 4 + extend_match(data + pos + 4, data + ref4 + 4, max - 4);
if (len > best_len) { if (len > best_len) {
/* Same offset-cost-aware acceptance as the hash5 walk. */
if (best_len >= 4) {
int dbits = enc_ilog2((uint32_t)(pos - ref4))
- enc_ilog2((uint32_t)*best_off);
if ((len - best_len) * 6 < dbits) { ref4 = next_ref4; continue; }
}
best_len = len; best_len = len;
*best_off = pos - ref4; *best_off = pos - ref4;
if (len >= 256) return best_len; if (len >= 256) return best_len;
@ -857,20 +900,137 @@ typedef struct { uint32_t off; int32_t len; } opt_cand_t;
* lever has been measured (matters more for nci-class fixtures). */ * lever has been measured (matters more for nci-class fixtures). */
static inline int32_t opt_lit_price(void) { return 8; } static inline int32_t opt_lit_price(void) { return 8; }
/* match bit price: cost_const(14) + log2(off) + ml_extra; rep ~2 bits */ /* SPRINT 129: per-byte literal prices from the block's byte histogram.
static inline int32_t opt_match_price(const matcher_t *m, uint32_t off, int32_t len) { * The flat-8 model (Sprint 44) was chosen as the best single constant,
int is_rep = (off == m->rep[0] || off == m->rep[1] || off == m->rep[2]); * but the real literal coder delivers ~4-6 bits/byte on text and 7-8
* on dense binary the flat constant over-prices text literals, so
* the parser substitutes marginal matches where literals are cheaper
* in reality. This is the "two-pass repricing" refinement that Sprint
* 44's note deferred, using the raw block histogram as the literal-
* distribution estimate (the true literal stream excludes match-
* covered bytes, but the distributions track closely in practice).
* price[b] = round(log2(N / hist[b])) clamped to [VV_OPT_LIT_MIN, 14];
* unseen bytes cannot appear as literals and get the ceiling. The
* clamp floor guards degenerate blocks (a byte at ~100% frequency
* would price to 0 and make literal runs look free). Constants swept
* on the 11-file corpus see CHANGELOG v2.64.0. */
#ifndef VV_OPT_LIT_MIN
#define VV_OPT_LIT_MIN 2
#endif
#ifndef VV_OPT_LIT_BLEND
#define VV_OPT_LIT_BLEND 6
#endif
/* fwd decl: the greedy parser (defined below) doubles as the residual-
* literal estimator for the optimal parser's pricing prepass. */
static size_t compress_block(const uint8_t *src, size_t start_pos, size_t block_len,
uint8_t *dst, size_t dst_cap,
matcher_t *m, vv_mode_t mode, int min_match);
/* SPRINT 130: histogram the literal bytes of an LZ token stream (the
* residual literals a parse actually leaves), walking the same wire
* layout extract_literals does but only counting. Returns total
* literal count, or 0 on a malformed stream (caller falls back to the
* raw-block histogram). */
static size_t tok_lit_hist(const uint8_t *tokens, size_t tok_len,
int off_bytes, uint32_t hist[256]) {
const uint8_t *tp = tokens, *tp_end = tokens + tok_len;
size_t total = 0;
while (tp < tp_end) {
uint8_t token = *tp++;
size_t ll = token >> 4;
size_t mc = token & 0x0F;
if (ll == 15) {
do {
if (tp >= tp_end) return 0;
uint8_t b = *tp++;
ll += b;
if (b < 255) break;
} while (tp < tp_end);
}
if ((size_t)(tp_end - tp) < ll) return 0;
for (size_t i = 0; i < ll; i++) hist[tp[i]]++;
total += ll;
tp += ll;
if (tp >= tp_end) break;
if ((size_t)(tp_end - tp) < (size_t)off_bytes) return 0;
tp += off_bytes;
if (mc == 15) {
do {
if (tp >= tp_end) return 0;
uint8_t b = *tp++;
if (b < 255) break;
} while (tp < tp_end);
}
}
return total;
}
static void opt_build_lit_prices_from_hist(const uint32_t hist[256], size_t n,
int32_t lit_bits[256]) {
for (int s = 0; s < 256; s++) {
if (!hist[s] || !n) { lit_bits[s] = 14; continue; }
/* ratio8 = (n / hist[s]) in 24.8 fixed point; log2(ratio8) =
* log2(n/hist) + 8. Round via the mantissa bit below the MSB. */
uint32_t ratio8 = (uint32_t)(((uint64_t)n << 8) / hist[s]);
int t = enc_ilog2(ratio8);
int bits = t - 8;
if (t >= 1 && ((ratio8 >> (t - 1)) & 1)) bits++; /* round half up */
if (bits < VV_OPT_LIT_MIN) bits = VV_OPT_LIT_MIN;
if (bits > 14) bits = 14;
/* Blend toward the flat-8 prior: a histogram estimate is still
* an approximation of the coder's delivered cost, and pricing
* from it unblended over-buys literals (measured; see the
* v2.64.0 sweep). blend/8 parts per-byte estimate, rest flat. */
lit_bits[s] = (VV_OPT_LIT_BLEND * bits + (8 - VV_OPT_LIT_BLEND) * 8) / 8;
}
}
/* match bit price: cost_const(14) + log2(off) + ml_extra; rep ~2 bits.
*
* SPRINT 128: priced against a caller-supplied rep set instead of
* m->rep. The matcher's rep state is a greedy-parser search heuristic
* that nothing updates during an optimal parse (it stayed {0,0,0} for
* every all-extreme frame, so rep pricing here was dead code), and the
* wire's rep state is PER-BLOCK and PATH-DEPENDENT: the SEQ encoder
* and decoder both start each block at {0,0,0} and evolve it per
* emitted sequence. The DP now threads that exact state through
* per-position rep histories (see compress_block_optimal). */
/* A rep match saves the offset EXTRA bits, not the per-sequence
* overhead: it still spends full LL/OF/ML code symbols (~10 bits).
* The explicit-match constant 14 approximates that overhead plus
* slack, so the rep price must stay close beneath it pricing reps
* near-free makes the DP shred long matches into chains of short rep
* matches, each paying the un-modeled sequence overhead (measured:
* -15% ratio on logs at rep=2). Constant swept on the 11-file corpus. */
#ifndef VV_OPT_REP_BITS
#define VV_OPT_REP_BITS 10
#endif
static inline int32_t opt_match_price(const uint32_t reps[3], uint32_t off, int32_t len) {
int is_rep = (off == reps[0] || off == reps[1] || off == reps[2]);
int32_t log2_off = 0; uint32_t o = off; int32_t log2_off = 0; uint32_t o = off;
while (o > 1) { o >>= 1; log2_off++; } while (o > 1) { o >>= 1; log2_off++; }
int32_t off_bits = is_rep ? 2 : (14 + log2_off); int32_t off_bits = is_rep ? VV_OPT_REP_BITS : (14 + log2_off);
int32_t ml_extra = 0, v = len - VV_MIN_MATCH; int32_t ml_extra = 0, v = len - VV_MIN_MATCH;
if (v >= 15) ml_extra = 8 * (v / 255 + 1); if (v >= 15) ml_extra = 8 * (v / 255 + 1);
return off_bits + ml_extra; return off_bits + ml_extra;
} }
/* Collect match candidates at pos (longest per distinct offset). */ /* Wire rep-history update rule — must mirror vva_encode_sequences'
* enc_rep update (and the decoder's dec_rep) exactly: push only when
* the offset differs from rep[0]. */
static inline void opt_rep_push(uint32_t dst[3], const uint32_t src3[3], uint32_t off) {
if (off == src3[0]) {
dst[0] = src3[0]; dst[1] = src3[1]; dst[2] = src3[2];
} else {
dst[0] = off; dst[1] = src3[0]; dst[2] = src3[1];
}
}
/* Collect match candidates at pos (longest per distinct offset).
* SPRINT 128: rep candidates come from the DP path's rep history. */
static int opt_collect(const matcher_t *m, const uint8_t *data, static int opt_collect(const matcher_t *m, const uint8_t *data,
int32_t pos, int32_t end, opt_cand_t *cands) { int32_t pos, int32_t end, opt_cand_t *cands,
const uint32_t reps[3]) {
int n = 0; int n = 0;
int32_t max_dist = (int32_t)((1u << m->wlog) - 1); int32_t max_dist = (int32_t)((1u << m->wlog) - 1);
int32_t limit = pos - max_dist; if (limit < 0) limit = 0; int32_t limit = pos - max_dist; if (limit < 0) limit = 0;
@ -880,7 +1040,7 @@ static int opt_collect(const matcher_t *m, const uint8_t *data,
uint32_t pos4; memcpy(&pos4, data + pos, 4); uint32_t pos4; memcpy(&pos4, data + pos, 4);
for (int r = 0; r < 3; r++) { for (int r = 0; r < 3; r++) {
uint32_t roff = m->rep[r]; uint32_t roff = reps[r];
if (roff == 0 || (int32_t)roff > pos) continue; if (roff == 0 || (int32_t)roff > pos) continue;
const uint8_t *a = data + pos, *b = data + pos - roff; const uint8_t *a = data + pos, *b = data + pos - roff;
int32_t l = 0; while (l < max && a[l] == b[l]) l++; int32_t l = 0; while (l < max && a[l] == b[l]) l++;
@ -913,15 +1073,60 @@ static size_t compress_block_optimal(const uint8_t *src, size_t start_pos,
int off_bytes = (m->wlog > 16) ? 3 : 2; int off_bytes = (m->wlog > 16) ? 3 : 2;
int32_t N = (int32_t)block_len; int32_t N = (int32_t)block_len;
/* DP arrays indexed by offset-from-base [0..N]. */ /* DP arrays indexed by offset-from-base [0..N].
* SPRINT 128: prep[i] is the wire rep-offset history of the best
* path reaching position i (zstd-btopt-style approximation: paths
* that lose on price but would carry better reps are dropped).
* prep[0] = {0,0,0} because the SEQ encoder and decoder both reset
* their rep state at every block boundary. */
int32_t *price = (int32_t *)malloc(sizeof(int32_t) * (N + 1)); int32_t *price = (int32_t *)malloc(sizeof(int32_t) * (N + 1));
int32_t *plen = (int32_t *)malloc(sizeof(int32_t) * (N + 1)); int32_t *plen = (int32_t *)malloc(sizeof(int32_t) * (N + 1));
uint32_t *poff = (uint32_t *)malloc(sizeof(uint32_t) * (N + 1)); uint32_t *poff = (uint32_t *)malloc(sizeof(uint32_t) * (N + 1));
uint32_t (*prep)[3] = (uint32_t (*)[3])malloc(sizeof(uint32_t[3]) * (N + 1));
opt_cand_t *cands = (opt_cand_t *)malloc(sizeof(opt_cand_t) * VV_OPT_MAX_CAND); opt_cand_t *cands = (opt_cand_t *)malloc(sizeof(opt_cand_t) * VV_OPT_MAX_CAND);
if (!price || !plen || !poff || !cands) { free(price); free(plen); free(poff); free(cands); return 0; } if (!price || !plen || !poff || !prep || !cands) { free(price); free(plen); free(poff); free(prep); free(cands); return 0; }
for (int32_t i = 0; i <= N; i++) { price[i] = VV_OPT_PRICE_INF; plen[i] = 0; poff[i] = 0; } for (int32_t i = 0; i <= N; i++) { price[i] = VV_OPT_PRICE_INF; plen[i] = 0; poff[i] = 0; }
price[0] = 0; price[0] = 0;
prep[0][0] = prep[0][1] = prep[0][2] = 0;
/* SPRINT 129/130: entropy-aware per-byte literal prices for this
* block. The distribution that matters is the RESIDUAL literal
* stream (bytes a parse leaves uncovered), not the raw block the
* raw histogram is dominated by exactly the repetitive content
* that matches will remove. A depth-4 greedy prepass on a private
* throwaway matcher (no shared-state pollution, ~1% of the DP's
* runtime) estimates that stream; its token output is histogrammed
* and discarded. Falls back to the raw-block histogram if the
* prepass cannot run. */
int32_t lit_bits[256];
{
uint32_t hist[256];
memset(hist, 0, sizeof(hist));
size_t nlit = 0;
matcher_t mp;
if (matcher_init(&mp, m->wlog, 4)) {
mp.accel = 2;
mp.max_match = m->max_match;
size_t pcap = block_len + block_len / 255 + 1024;
uint8_t *ptok = (uint8_t *)malloc(pcap);
if (ptok) {
size_t pcsz = compress_block(src, start_pos, block_len, ptok,
pcap, &mp, VV_MODE_ULTRA_FAST, min_match);
if (pcsz > 0)
nlit = tok_lit_hist(ptok, pcsz, off_bytes, hist);
free(ptok);
}
matcher_free(&mp);
}
if (nlit == 0) {
/* Prepass unavailable or block fully covered: raw fallback. */
memset(hist, 0, sizeof(hist));
for (int32_t i = 0; i < N; i++) hist[src[base + i]]++;
nlit = (size_t)N;
}
opt_build_lit_prices_from_hist(hist, nlit, lit_bits);
}
/* Forward DP. We also must keep the matcher hash chains populated as we /* Forward DP. We also must keep the matcher hash chains populated as we
* advance, so matches reference earlier positions correctly. We insert * advance, so matches reference earlier positions correctly. We insert
@ -944,13 +1149,16 @@ static size_t compress_block_optimal(const uint8_t *src, size_t start_pos,
} }
int32_t ip = base + i; int32_t ip = base + i;
/* literal edge */ /* literal edge (literals leave the rep history unchanged) */
int32_t lp = price[i] + opt_lit_price(); int32_t lp = price[i] + lit_bits[src[ip]];
if (lp < price[i + 1]) { price[i + 1] = lp; plen[i + 1] = 1; poff[i + 1] = 0; } if (lp < price[i + 1]) {
price[i + 1] = lp; plen[i + 1] = 1; poff[i + 1] = 0;
prep[i + 1][0] = prep[i][0]; prep[i + 1][1] = prep[i][1]; prep[i + 1][2] = prep[i][2];
}
/* match edges */ /* match edges */
if (ip + min_match <= end) { if (ip + min_match <= end) {
int nc = opt_collect(m, src, ip, end, cands); int nc = opt_collect(m, src, ip, end, cands, prep[i]);
/* Find the longest candidate. */ /* Find the longest candidate. */
int32_t best_len = 0; uint32_t best_off = 0; int32_t best_len = 0; uint32_t best_off = 0;
for (int c = 0; c < nc; c++) { for (int c = 0; c < nc; c++) {
@ -966,9 +1174,12 @@ static size_t compress_block_optimal(const uint8_t *src, size_t start_pos,
* the whole match. */ * the whole match. */
int32_t use = best_len; int32_t use = best_len;
if (i + use > N) use = N - i; if (i + use > N) use = N - i;
int32_t np = price[i] + opt_match_price(m, best_off, use); int32_t np = price[i] + opt_match_price(prep[i], best_off, use);
int32_t j = i + use; int32_t j = i + use;
if (np < price[j]) { price[j] = np; plen[j] = use; poff[j] = best_off; } if (np < price[j]) {
price[j] = np; plen[j] = use; poff[j] = best_off;
opt_rep_push(prep[j], prep[i], best_off);
}
/* Insert boundary positions only (match-skip heuristic), /* Insert boundary positions only (match-skip heuristic),
* then jump the DP cursor to the match end. */ * then jump the DP cursor to the match end. */
int32_t end5 = end - 5; int32_t end5 = end - 5;
@ -985,9 +1196,12 @@ static size_t compress_block_optimal(const uint8_t *src, size_t start_pos,
if (i + mlen > N) mlen = N - i; if (i + mlen > N) mlen = N - i;
if (mlen < min_match) continue; if (mlen < min_match) continue;
for (int32_t L = mlen; L >= min_match; L--) { for (int32_t L = mlen; L >= min_match; L--) {
int32_t np = price[i] + opt_match_price(m, moff, L); int32_t np = price[i] + opt_match_price(prep[i], moff, L);
int32_t j = i + L; int32_t j = i + L;
if (np < price[j]) { price[j] = np; plen[j] = L; poff[j] = moff; } if (np < price[j]) {
price[j] = np; plen[j] = L; poff[j] = moff;
opt_rep_push(prep[j], prep[i], moff);
}
if (L > min_match + 8 && L < mlen) L = min_match + 9; if (L > min_match + 8 && L < mlen) L = min_match + 9;
} }
} }
@ -1000,7 +1214,7 @@ static size_t compress_block_optimal(const uint8_t *src, size_t start_pos,
/* Worst case every position is a literal: N entries. */ /* Worst case every position is a literal: N entries. */
int32_t *seq_len = (int32_t *)malloc(sizeof(int32_t) * (N + 1)); int32_t *seq_len = (int32_t *)malloc(sizeof(int32_t) * (N + 1));
uint32_t *seq_off = (uint32_t *)malloc(sizeof(uint32_t) * (N + 1)); uint32_t *seq_off = (uint32_t *)malloc(sizeof(uint32_t) * (N + 1));
if (!seq_len || !seq_off) { free(price); free(plen); free(poff); free(cands); free(seq_len); free(seq_off); return 0; } if (!seq_len || !seq_off) { free(price); free(plen); free(poff); free(prep); free(cands); free(seq_len); free(seq_off); return 0; }
int32_t ns = 0, cur = N; int32_t ns = 0, cur = N;
while (cur > 0) { while (cur > 0) {
int32_t L = plen[cur]; int32_t L = plen[cur];
@ -1021,7 +1235,7 @@ static size_t compress_block_optimal(const uint8_t *src, size_t start_pos,
size_t ll = (size_t)(src + pos - lit_start); size_t ll = (size_t)(src + pos - lit_start);
size_t needed = 1 + (ll >= 15 ? ll / 255 + 2 : 0) + ll + 2 + ((size_t)L / 255 + 2); size_t needed = 1 + (ll >= 15 ? ll / 255 + 2 : 0) + ll + 2 + ((size_t)L / 255 + 2);
if ((size_t)(op - dst) + needed > dst_cap) { if ((size_t)(op - dst) + needed > dst_cap) {
free(price); free(plen); free(poff); free(cands); free(seq_len); free(seq_off); free(price); free(plen); free(poff); free(prep); free(cands); free(seq_len); free(seq_off);
return 0; return 0;
} }
op += emit_seq(op, lit_start, ll, (size_t)L, O, off_bytes, min_match); op += emit_seq(op, lit_start, ll, (size_t)L, O, off_bytes, min_match);
@ -1035,13 +1249,13 @@ static size_t compress_block_optimal(const uint8_t *src, size_t start_pos,
size_t ll = (size_t)(src + end - lit_start); size_t ll = (size_t)(src + end - lit_start);
size_t needed = 1 + (ll >= 15 ? ll / 255 + 2 : 0) + ll; size_t needed = 1 + (ll >= 15 ? ll / 255 + 2 : 0) + ll;
if ((size_t)(op - dst) + needed > dst_cap) { if ((size_t)(op - dst) + needed > dst_cap) {
free(price); free(plen); free(poff); free(cands); free(seq_len); free(seq_off); free(price); free(plen); free(poff); free(prep); free(cands); free(seq_len); free(seq_off);
return 0; return 0;
} }
op += emit_seq(op, lit_start, ll, 0, 0, off_bytes, min_match); op += emit_seq(op, lit_start, ll, 0, 0, off_bytes, min_match);
} }
free(price); free(plen); free(poff); free(cands); free(seq_len); free(seq_off); free(price); free(plen); free(poff); free(prep); free(cands); free(seq_len); free(seq_off);
return (size_t)(op - dst); return (size_t)(op - dst);
} }
@ -1062,10 +1276,12 @@ static size_t compress_block(const uint8_t *src, size_t start_pos, size_t block_
int32_t end = (int32_t)(start_pos + block_len); int32_t end = (int32_t)(start_pos + block_len);
const uint8_t *lit_start = src + start_pos; const uint8_t *lit_start = src + start_pos;
int off_bytes = (m->wlog > 16) ? 3 : 2; int off_bytes = (m->wlog > 16) ? 3 : 2;
uint32_t failures = 0; /* consecutive no-match positions (for --accel skip) */ uint32_t failures = 0; /* consecutive no-match positions (for accel skip) */
uint32_t nmatch = 0; /* matches found in this block (early-RAW bail) */
while (pos < end - min_match) { while (pos < end - min_match) {
int32_t mlen = 0, moff = 0; int32_t mlen = 0, moff = 0;
int pos_inserted = 0;
/* ─── Step 1: Try rep-match (free, no hash lookup) ─── */ /* ─── Step 1: Try rep-match (free, no hash lookup) ─── */
int32_t rep_idx = -1; int32_t rep_idx = -1;
@ -1133,6 +1349,7 @@ static size_t compress_block(const uint8_t *src, size_t start_pos, size_t block_
pos + 1 < end - min_match) { pos + 1 < end - min_match) {
/* Check pos+1 */ /* Check pos+1 */
matcher_insert(m, src, pos, end); matcher_insert(m, src, pos, end);
pos_inserted = 1;
int32_t noff = 0; int32_t noff = 0;
int32_t nlen = chain_match(m, src, pos + 1, end, &noff); int32_t nlen = chain_match(m, src, pos + 1, end, &noff);
@ -1191,6 +1408,7 @@ static size_t compress_block(const uint8_t *src, size_t start_pos, size_t block_
int rhs = moff_bits * (nlen + 1); int rhs = moff_bits * (nlen + 1);
if (lhs < rhs) { if (lhs < rhs) {
pos++; pos++;
pos_inserted = 0; /* the inserted position is now pos-1 */
mlen = nlen; moff = noff; mlen = nlen; moff = noff;
rep_idx = nri; /* may have shifted from explicit→rep or vice versa */ rep_idx = nri; /* may have shifted from explicit→rep or vice versa */
@ -1237,17 +1455,22 @@ static size_t compress_block(const uint8_t *src, size_t start_pos, size_t block_
* *
* Saves ~3 instructions per insert. Measured +4% encode * Saves ~3 instructions per insert. Measured +4% encode
* speedup on Silesia fast mode (Sprint 29). */ * speedup on Silesia fast mode (Sprint 29). */
/* SPRINT 124: when the lazy probe already inserted pos and
* we did not shift, start at pos+1 re-inserting pos would
* put a self-duplicate link in the chain, lengthening every
* future walk through that bucket. */
int32_t ins_first = pos + (pos_inserted ? 1 : 0);
if (mlen >= 16) { if (mlen >= 16) {
/* Long match: only insert boundary positions */ /* Long match: only insert boundary positions */
int32_t end5 = end - 5; int32_t end5 = end - 5;
for (int32_t j = pos; j < pos + 3 && j <= end5; j++) for (int32_t j = ins_first; j < pos + 3 && j <= end5; j++)
matcher_insert_fast(m, src, j); matcher_insert_fast(m, src, j);
for (int32_t j = pos + mlen - 3; j < pos + mlen && j <= end5; j++) for (int32_t j = pos + mlen - 3; j < pos + mlen && j <= end5; j++)
matcher_insert_fast(m, src, j); matcher_insert_fast(m, src, j);
} else { } else {
/* Short match: insert all positions */ /* Short match: insert all positions */
int32_t end5 = end - 5; int32_t end5 = end - 5;
for (int32_t j = pos; j < pos + mlen && j <= end5; j++) for (int32_t j = ins_first; j < pos + mlen && j <= end5; j++)
matcher_insert_fast(m, src, j); matcher_insert_fast(m, src, j);
} }
@ -1255,14 +1478,28 @@ static size_t compress_block(const uint8_t *src, size_t start_pos, size_t block_
pos += mlen; pos += mlen;
lit_start = src + pos; lit_start = src + pos;
failures = 0; /* matched: reset the no-match run */ failures = 0; /* matched: reset the no-match run */
nmatch++;
} else { } else {
matcher_insert(m, src, pos, end); if (!pos_inserted) matcher_insert(m, src, pos, end);
/* --accel: skip ahead over unmatchable regions. accel==0 keeps /* Accel: skip ahead over unmatchable regions. accel==0 keeps
* the byte-identical default (advance 1). The skipped positions * the byte-identical old default (advance 1). The skipped
* are not hashed/inserted and simply become literals. */ * positions are not hashed/inserted and simply become
* literals. SPRINT 124: balanced/extreme cap the stride at 8
* on sparse-match data (struct-of-floats) an unbounded
* ramp skips over match starts and costs double-digit ratio;
* fast mode keeps the full lz4-style ramp. */
if (m->accel) { if (m->accel) {
pos += 1 + (int32_t)(((uint32_t)failures * m->accel) >> 6); uint32_t step = 1 + (((uint32_t)failures * m->accel) >> 6);
if (mode >= VV_MODE_BALANCED && step > 8) step = 8;
pos += (int32_t)step;
failures++; failures++;
/* Early RAW bail: 128 KB into the block with zero
* matches means this block is going raw anyway (csz
* would exceed braw). Returning 0 makes the caller
* emit a RAW block without paying for the rest of the
* parse or the literal memcpys. */
if (nmatch == 0 && pos - (int32_t)start_pos >= (1 << 17))
return 0;
} else { } else {
pos++; pos++;
} }
@ -1375,25 +1612,55 @@ static size_t extract_literals(
* - dst/dst_cap: output buffer * - dst/dst_cap: output buffer
* *
* Returns bytes written to dst on success, or 0 on overflow. */ * Returns bytes written to dst on success, or 0 on overflow. */
/* SPRINT 124: high-watermark tracking for the secure-zero scrub.
* Scrubbing full buffer capacities (~4 MB) per vv_compress call cost
* up to 14% of encode wall on fast inputs; only bytes actually written
* can hold plaintext, so tracking write watermarks preserves the
* Sprint 117 security property at a fraction of the cost. */
typedef struct {
size_t tmp, lit, stripped, ent_front, ent_back;
} scrub_wm_t;
static inline void wm_max(size_t *wm, size_t used) {
if (used > *wm) *wm = used;
}
static size_t emit_block(const uint8_t *src, size_t block_start, size_t braw, static size_t emit_block(const uint8_t *src, size_t block_start, size_t braw,
int last, matcher_t *m, vv_mode_t mode, uint8_t wlog, int last, matcher_t *m, vv_mode_t mode, uint8_t wlog,
uint8_t *tmp, size_t tcap, uint8_t *tmp, size_t tcap,
uint8_t *lit_buf, size_t lit_cap, uint8_t *lit_buf, size_t lit_cap,
uint8_t *stripped, uint8_t *ent_buf, size_t ent_cap, uint8_t *stripped, uint8_t *ent_buf, size_t ent_cap,
uint8_t *dst, size_t dst_cap, int min_match, uint8_t *dst, size_t dst_cap, int min_match,
int compat_v246_5) { int compat_v246_5, scrub_wm_t *wm) {
uint8_t *op = dst; uint8_t *op = dst;
/* SPRINT 42/43 RATIO PROGRAM: extreme mode uses the whole-block optimal /* SPRINT 42/43 RATIO PROGRAM: extreme mode uses the whole-block optimal
* parser; balanced/fast keep greedy/lazy. csz==0 (overflow/alloc) flows * parser; balanced/fast keep greedy/lazy. csz==0 (overflow/alloc) flows
* into the raw-store branch below. */ * into the raw-store branch below.
*
* SPRINT 124: on format-v2 (binary-detected) input, extreme uses the
* deep greedy/lazy parser instead. The optimal DP prices every match
* at full log2(offset) cost it has no rep-offset model so on
* rep-heavy record data (struct-of-floats, sensor logs) it loses
* 15-20% ratio to the rep-aware greedy path, and on incompressible
* binary it pays a full O(N·depth) DP just to store raw (the greedy
* path has skip acceleration and an early-RAW bail). Text-like input
* keeps the optimal parser, where it wins 3-11% over greedy. */
size_t csz; size_t csz;
if (mode >= VV_MODE_EXTREME) int v2_block = (min_match < (int)VV_MIN_MATCH);
if (mode >= VV_MODE_EXTREME && !v2_block)
csz = compress_block_optimal(src, block_start, braw, tmp, tcap, m, min_match); csz = compress_block_optimal(src, block_start, braw, tmp, tcap, m, min_match);
else else
csz = compress_block(src, block_start, braw, tmp, tcap, m, mode, min_match); csz = compress_block(src, block_start, braw, tmp, tcap, m, mode, min_match);
if (wm) wm_max(&wm->tmp, csz);
if (csz == 0 || csz >= braw) { /* SPRINT 124: in balanced/extreme, a token stream slightly larger
* than raw can still win AFTER entropy coding on low-match data
* (struct-of-floats, sensor logs) nearly all the compression comes
* from the entropy stage over literals, not from matches. Only the
* entropy-less fast path must reject csz >= braw outright. */
size_t raw_gate = (mode >= VV_MODE_BALANCED) ? braw + braw / 8 : braw;
if (csz == 0 || csz >= raw_gate) {
/* Incompressible: store raw */ /* Incompressible: store raw */
if ((size_t)(op - dst) + 4 + braw > dst_cap) return 0; if ((size_t)(op - dst) + 4 + braw > dst_cap) return 0;
uint32_t bh = vv_bh_pack(VV_BLOCK_RAW, last, (uint32_t)braw); uint32_t bh = vv_bh_pack(VV_BLOCK_RAW, last, (uint32_t)braw);
@ -1423,8 +1690,9 @@ static size_t emit_block(const uint8_t *src, size_t block_start, size_t braw,
seq_block_sz = 4 + 3 + 1 + seq_len; seq_block_sz = 4 + 3 + 1 + seq_len;
seq_valid = 1; seq_valid = 1;
} }
if (wm) wm_max(&wm->ent_front, seq_len);
/* Path B: literal-only entropy ('I' or 'C') */ /* Path B: literal-only entropy ('I' or 'A') */
size_t stripped_len = 0; size_t stripped_len = 0;
size_t lit_count = 0; size_t lit_count = 0;
uint8_t *ent_buf2 = ent_buf + ent_cap / 2; uint8_t *ent_buf2 = ent_buf + ent_cap / 2;
@ -1433,97 +1701,49 @@ static size_t emit_block(const uint8_t *src, size_t block_start, size_t braw,
uint8_t ent_tag = 0; uint8_t ent_tag = 0;
size_t ent_block_sz = (size_t)-1; size_t ent_block_sz = (size_t)-1;
int try_path_b = 1; /* Path B gate (v2.53.3, revised SPRINT 124): Path B has a
/* PERF / dead-code prune (v2.53.3): Path B (literal-only 'I'/'C' * measured 0% win rate against Path A (SEQ) on real inputs
* entropy) has a measured 0% win rate against Path A (SEQ) across * SEQ codes the same literals at least as small while also
* all real inputs tested (text, binary, logs, CSV) SEQ always * coding the matches. Run it only when SEQ failed or produced
* codes the same literals at least as small while also coding the * weak output (>= 7/8 of raw). Path B is v1-only (its stripped
* matches. Path B can only conceivably win on a block where SEQ * tokens carry v1 matchlen bias), so on the v2 path skip the
* failed to find structure (its compressed size approaches raw). * work entirely the result could never be emitted.
* So skip Path B's extract_literals + redundant ANS encodes *
* whenever SEQ is valid and already beats raw by a clear margin * SPRINT 124: the CTX (order-1) coder is gone from this path.
* (seq_block_sz < braw*7/8). On blocks where SEQ does not compress * It ran exactly when SEQ was weak low-redundancy binary
* (>= braw*7/8) Path B still runs, preserving the only case it * where it burned 50% of encode wall (sensors-class inputs)
* could win. Verified byte-identical on all 12 Silesia (balanced + * and, per the Sprint 53 measurements, never won a block. */
* extreme) and on binary/log/CSV; the ratio gate guards against any int try_path_b = !use_v2 && (!seq_valid ||
* regression. This removes redundant per-block work; it is a seq_block_sz >= (braw * 7 / 8));
* code-cleanliness change, not a measurable speedup (Path B was not
* the encode bottleneck that is the depth-24 chain walk). */
if (seq_valid && seq_block_sz < (braw * 7 / 8))
try_path_b = 0;
if (mode == VV_MODE_BALANCED && seq_valid && seq_block_sz < (braw / 3)) {
/* SPRINT 29 (revised in v2.15): always try Path B in BALANCED
* mode, comparing both costs and picking the smaller. The
* earlier "skip Path B if seq compressed >3:1" heuristic
* (added in Sprint 28 for speed) saved ~30% encode time but
* hurt ratio on text-heavy data Silesia dickens/reymont
* showed Path B's 'C' tag would have produced 5-10% smaller
* output but never got the chance.
*
* v2.15 trade-off: encoder is ~25% slower in BALANCED mode
* but ratio improves measurably on text. Decode speed is
* unaffected (decoder doesn't care which tag was chosen).
*
* In ULTRA_FAST/FAST modes the original skip remains in
* effect because those modes are throughput-priority. */
(void)try_path_b;
}
if (try_path_b) { if (try_path_b) {
lit_count = extract_literals(tmp, csz, lit_buf, lit_cap, lit_count = extract_literals(tmp, csz, lit_buf, lit_cap,
stripped, &stripped_len, off_bytes); stripped, &stripped_len, off_bytes);
if (wm) {
wm_max(&wm->lit, lit_count);
wm_max(&wm->stripped, stripped_len);
}
if (lit_count > 0) { if (lit_count > 0) {
/* SPRINT 53: skip the expensive CTX (order-1 context) vva_error_t aerr = vva_encode4(lit_buf, lit_count,
* path when sequence coding is already winning by a ent_buf2, ent_cap2, &ent_len);
* big margin. Profile data across 7 fixtures (text, if (aerr == VVA_OK) ent_tag = VV_ENTROPY_ANS4;
* json, source, 4 ELF binaries) showed CTX wins 0/16
* attempts the CTX coder has never actually beaten
* SEQ on these workloads, but burned 20% of encode
* time building per-context ANS tables that were
* always discarded.
*
* Heuristic: skip CTX when seq_block_sz already does
* better than 2:1 compression (seq_block_sz < braw/2).
* Path A (SEQ) essentially never loses to Path B (CTX)
* when the LZ matcher found strong matches. CTX only
* matters for low-redundancy data where SEQ produces
* close-to-raw output exactly the case where
* seq_block_sz braw/2.
*
* Falls back to ANS4 / ANS as literal coders in the
* unchanged code below. These are ~10× cheaper than
* CTX to build. Net encode-time savings measured in
* SPRINT 53 CHANGELOG entry.
*
* Security/correctness: this is purely an encoder
* heuristic. Decoder is unchanged. Output wire format
* still meets spec. Worst case on a pathological
* input where CTX would have won: we produce slightly
* larger output via ANS4 or ANS. Ratio gate guards
* against any real regression. */
int skip_ctx = seq_valid && seq_block_sz < (braw * 4 / 5);
if (!skip_ctx && mode >= VV_MODE_BALANCED && lit_count >= 4096) {
vva_error_t aerr = vva_encode_ctx(lit_buf, lit_count,
ent_buf2, ent_cap2, &ent_len);
if (aerr == VVA_OK) ent_tag = VV_ENTROPY_CTX;
}
if (!ent_tag) { if (!ent_tag) {
vva_error_t aerr = vva_encode4(lit_buf, lit_count, aerr = vva_encode(lit_buf, lit_count,
ent_buf2, ent_cap2, &ent_len); ent_buf2, ent_cap2, &ent_len);
if (aerr == VVA_OK) ent_tag = VV_ENTROPY_ANS4;
}
if (!ent_tag) {
vva_error_t aerr = vva_encode(lit_buf, lit_count,
ent_buf2, ent_cap2, &ent_len);
if (aerr == VVA_OK) ent_tag = VV_ENTROPY_ANS; if (aerr == VVA_OK) ent_tag = VV_ENTROPY_ANS;
} }
if (ent_tag) { if (ent_tag) {
ent_block_sz = 4 + 3 + 1 + 2 + 2 + ent_len + stripped_len; ent_block_sz = 4 + 3 + 1 + 2 + 2 + ent_len + stripped_len;
} }
if (wm) wm_max(&wm->ent_back, ent_len);
} }
} }
size_t raw_block_sz = 4 + 3 + csz; size_t raw_block_sz = 4 + 3 + csz;
/* Raw-store block size: with the relaxed raw_gate above, csz may
* exceed braw, so every candidate must also beat plain storage. */
size_t store_sz = 4 + braw;
if (raw_block_sz > store_sz) raw_block_sz = store_sz;
if (seq_valid && seq_block_sz <= ent_block_sz && seq_block_sz < raw_block_sz) { if (seq_valid && seq_block_sz <= ent_block_sz && seq_block_sz < raw_block_sz) {
if ((size_t)(op - dst) + seq_block_sz > dst_cap) return 0; if ((size_t)(op - dst) + seq_block_sz > dst_cap) return 0;
@ -1554,21 +1774,20 @@ static size_t emit_block(const uint8_t *src, size_t block_start, size_t braw,
op[0] = (uint8_t)(ent_len); op[1] = (uint8_t)(ent_len >> 8); op += 2; op[0] = (uint8_t)(ent_len); op[1] = (uint8_t)(ent_len >> 8); op += 2;
memcpy(op, ent_buf2, ent_len); op += ent_len; memcpy(op, ent_buf2, ent_len); op += ent_len;
memcpy(op, stripped, stripped_len); op += stripped_len; memcpy(op, stripped, stripped_len); op += stripped_len;
} else if (!use_v2) { } else if (!use_v2 && csz < braw) {
/* Plain VV_BLOCK_COMPRESSED carries raw v1-format tokens. /* Plain VV_BLOCK_COMPRESSED carries raw v1-format tokens.
* For v2, we must not emit these the decoder would * For v2, we must not emit these the decoder would
* reconstruct matchlen with +4 instead of +3. Fall to RAW * reconstruct matchlen with +4 instead of +3. Guarded on
* block instead (handled below via "else" when raw_block_sz * csz < braw because the relaxed raw_gate can let a token
* is smaller). We reach this branch only when the previous * stream slightly larger than raw reach this point. */
* conditions all failed AND we're NOT v2. */ if ((size_t)(op - dst) + 4 + 3 + csz > dst_cap) return 0;
if ((size_t)(op - dst) + raw_block_sz > dst_cap) return 0;
uint32_t bh = vv_bh_pack(VV_BLOCK_COMPRESSED, last, (uint32_t)braw); uint32_t bh = vv_bh_pack(VV_BLOCK_COMPRESSED, last, (uint32_t)braw);
memcpy(op, &bh, 4); op += 4; memcpy(op, &bh, 4); op += 4;
op[0] = (uint8_t)(csz); op[1] = (uint8_t)(csz >> 8); op[2] = (uint8_t)(csz >> 16); op[0] = (uint8_t)(csz); op[1] = (uint8_t)(csz >> 8); op[2] = (uint8_t)(csz >> 16);
op += 3; op += 3;
memcpy(op, tmp, csz); op += csz; memcpy(op, tmp, csz); op += csz;
} else { } else {
/* v2 path, sequence coding didn't fit/help: emit RAW. */ /* Nothing beat plain storage: emit RAW. */
if ((size_t)(op - dst) + 4 + braw > dst_cap) return 0; if ((size_t)(op - dst) + 4 + braw > dst_cap) return 0;
uint32_t bh = vv_bh_pack(VV_BLOCK_RAW, last, (uint32_t)braw); uint32_t bh = vv_bh_pack(VV_BLOCK_RAW, last, (uint32_t)braw);
memcpy(op, &bh, 4); op += 4; memcpy(op, &bh, 4); op += 4;
@ -1696,32 +1915,55 @@ int64_t vv_compress_inner(const uint8_t *src, size_t src_len,
size_t sz16 = 0, sz20 = 0; size_t sz16 = 0, sz20 = 0;
/* SPRINT 93 audit: matcher_init can fail; if it does, skip /* SPRINT 93 audit: matcher_init can fail; if it does, skip
* the trial (this path is a perf-tuning probe falling * the trial (this path is a perf-tuning probe falling
* back to default wlog is safe). */ * back to default wlog is safe).
* SPRINT 124: trials run with accel=2 so incompressible
* inputs no longer pay two full 128 KB parses just to
* decide "store raw". Both trials use the same accel, so
* the 16-vs-20 comparison stays apples-to-apples. */
if (matcher_init(&m16, 16, 4)) { if (matcher_init(&m16, 16, 4)) {
m16.accel = 2;
sz16 = compress_block(src, 0, trial_len, trial_buf, trial_cap, &m16, VV_MODE_ULTRA_FAST, VV_MIN_MATCH); sz16 = compress_block(src, 0, trial_len, trial_buf, trial_cap, &m16, VV_MODE_ULTRA_FAST, VV_MIN_MATCH);
matcher_free(&m16); matcher_free(&m16);
} }
matcher_t m20; matcher_t m20;
if (matcher_init(&m20, 20, 4)) { if (matcher_init(&m20, 20, 4)) {
m20.accel = 2;
sz20 = compress_block(src, 0, trial_len, trial_buf, trial_cap, &m20, VV_MODE_ULTRA_FAST, VV_MIN_MATCH); sz20 = compress_block(src, 0, trial_len, trial_buf, trial_cap, &m20, VV_MODE_ULTRA_FAST, VV_MIN_MATCH);
matcher_free(&m20); matcher_free(&m20);
} }
free(trial_buf); free(trial_buf);
if (sz20 > 0 && sz16 > 0 && sz20 < (sz16 * 97 / 100)) wlog = 20; if (sz20 > 0 && sz16 > 0 && sz20 < (sz16 * 97 / 100)) wlog = 20;
/* Binary-like detection: best trial ratio < 2:1 */ /* Binary-like detection: best trial ratio < 2:1. A zero
* size means the early-RAW bail fired maximally
* incompressible, so binary-like by definition. */
size_t best_sz = (sz20 > 0 && sz20 < sz16) ? sz20 : sz16; size_t best_sz = (sz20 > 0 && sz20 < sz16) ? sz20 : sz16;
if (best_sz > 0 && best_sz * 2 > trial_len) enable_hash4 = 1; if (best_sz == 0 || best_sz * 2 > trial_len) enable_hash4 = 1;
} }
} }
/* SPRINT 124: adaptive format v2 (decided here because the window
* overrides below must not fire for v2-routed input). min_match=3
* ('T' blocks) is a measured 14%+ ratio win on struct-of-floats/
* record binary and 2-3% on ELF, while slightly HURTING text/JSON
* ratio and decode speed (more, shorter sequences). Auto-enable
* exactly where it wins: binary-detected inputs. Suppressed by
* the compat flag because 'T' blocks require a v2.33.0+ decoder.
* Explicit opts->format_v2 still forces it for any input. */
int use_v2_fmt = opts->format_v2 ||
(enable_hash4 && opts->mode >= VV_MODE_BALANCED &&
!opts->compat_v246_5_decoder);
/* SPRINT 67: size-based wlog override. The trial above often /* SPRINT 67: size-based wlog override. The trial above often
* misses wins that only become visible past the 128 KB trial * misses wins that only become visible past the 128 KB trial
* boundary (long-range refs in multi-MB files). Override to * boundary (long-range refs in multi-MB files). Override to
* wlog=18 for files 3 MB when the trial left wlog at 16. */ * wlog=18 for files 3 MB when the trial left wlog at 16.
* SPRINT 124: not for v2-routed (binary) input the greedy
* parser regresses badly on rep-heavy data with large windows
* (diverse far offsets break rep streaks and bloat OF codes). */
if (opts->window_log == 0 && opts->mode >= VV_MODE_BALANCED && if (opts->window_log == 0 && opts->mode >= VV_MODE_BALANCED &&
wlog == 16 && src_len >= 3145728) { !use_v2_fmt && wlog == 16 && src_len >= 3145728) {
wlog = 18; wlog = 18;
} }
@ -1747,7 +1989,11 @@ int64_t vv_compress_inner(const uint8_t *src, size_t src_len,
* Memory at wlog=24: chain[wsz]+hash4_chain[wsz] = 2*4*16M = 128 MB * Memory at wlog=24: chain[wsz]+hash4_chain[wsz] = 2*4*16M = 128 MB
* matcher. Acceptable for extreme ("max ratio, will wait"). */ * matcher. Acceptable for extreme ("max ratio, will wait"). */
if (opts->window_log == 0 && opts->mode >= VV_MODE_EXTREME && if (opts->window_log == 0 && opts->mode >= VV_MODE_EXTREME &&
src_len > (1u << 20)) { !use_v2_fmt && src_len > (1u << 20)) {
/* SPRINT 124: v2-routed (binary) extreme input uses the greedy
* parser (no rep model in the optimal DP), and greedy + large
* window is a measured 15-30% ratio LOSS on rep-heavy data
* keep the trial-chosen window there. */
uint8_t want = 20; uint8_t want = 20;
uint64_t s = src_len; uint64_t s = src_len;
while ((1ull << want) < s && want < 24) want++; while ((1ull << want) < s && want < 24) want++;
@ -1781,18 +2027,29 @@ int64_t vv_compress_inner(const uint8_t *src, size_t src_len,
* by the balanced/extreme window-selection trial above stay at * by the balanced/extreme window-selection trial above stay at
* single_probe==0 and produce bit-identical trial sizes. */ * single_probe==0 and produce bit-identical trial sizes. */
m.single_probe = (opts->mode == VV_MODE_ULTRA_FAST) ? 1 : 0; m.single_probe = (opts->mode == VV_MODE_ULTRA_FAST) ? 1 : 0;
m.accel = opts->accel > 64 ? 64 : opts->accel; /* SPRINT 124: accel defaults ON. opts->accel == 0 now means "auto":
* fast mode gets the lz4-style ramp (2 step 1 + failures/32),
* balanced/extreme a gentle one (1 step 1 + failures/64, capped
* at 8 inside compress_block). This is what turns 1 MB of random
* bytes from a 24 ns/byte full-parse crawl into a near-memcpy RAW
* store. Explicit --accel values are honored unchanged. */
{
uint32_t eff_accel = opts->accel;
if (eff_accel == 0)
eff_accel = (opts->mode >= VV_MODE_BALANCED) ? 1 : 2;
m.accel = eff_accel > 64 ? 64 : eff_accel;
}
m.no_rep = opts->no_rep ? 1 : 0; m.no_rep = opts->no_rep ? 1 : 0;
/* Format v2 cap applies to EVERY match emitted from this matcher, /* Format v2 cap applies to EVERY match emitted from this matcher,
* not just those produced via hash3. Set unconditionally when * not just those produced via hash3. Set unconditionally when
* opts.format_v2 is active. */ * the v2 format is active. */
if (opts->format_v2) { if (use_v2_fmt) {
matcher_set_format_v2(&m); matcher_set_format_v2(&m);
} }
/* Hash3 enablement is a separate, adaptive decision. Only fires /* Hash3 enablement is a separate, adaptive decision. Only fires
* on binary-like data (enable_hash4) where length-3 matches * on binary-like data (enable_hash4) where length-3 matches
* actually help. On text/JSON it stays off to avoid regressions. */ * actually help. On text/JSON it stays off to avoid regressions. */
if (opts->format_v2 && enable_hash4) { if (use_v2_fmt && enable_hash4) {
if (!matcher_enable_hash3(&m)) { if (!matcher_enable_hash3(&m)) {
matcher_free(&m); matcher_free(&m);
return VV_ERR_NOMEM; return VV_ERR_NOMEM;
@ -1817,7 +2074,14 @@ int64_t vv_compress_inner(const uint8_t *src, size_t src_len,
* output. For small one-shot calls this avoids ~3 MB of wasted * output. For small one-shot calls this avoids ~3 MB of wasted
* allocation and page-faulting every call. */ * allocation and page-faulting every call. */
lit_cap = block_bound; lit_cap = block_bound;
ent_cap = vva_bound(block_bound); /* SPRINT 124 (latent-corruption fix): ent_buf is shared by Path A
* (SEQ, writes at ent_buf[0..]) and Path B (literal entropy,
* writes at ent_buf + ent_cap/2). SEQ output on weak blocks can
* reach vva_bound(braw) with ent_cap == vva_bound the halves
* OVERLAP and Path B silently clobbers SEQ's tail before the
* winner is chosen. Size the buffer so each half holds a full
* vva_bound worth of output. */
ent_cap = 2 * vva_bound(block_bound);
lit_buf = (uint8_t *)malloc(lit_cap); lit_buf = (uint8_t *)malloc(lit_cap);
stripped = (uint8_t *)malloc(tcap); stripped = (uint8_t *)malloc(tcap);
ent_buf = (uint8_t *)malloc(ent_cap); ent_buf = (uint8_t *)malloc(ent_cap);
@ -1836,10 +2100,12 @@ int64_t vv_compress_inner(const uint8_t *src, size_t src_len,
memcpy(op, &bh, 4); op += 4; memcpy(op, &bh, 4); op += 4;
} }
/* Format v2: when opts->format_v2 is set, encode with min_match=3. /* Format v2 (explicit or adaptive): encode with min_match=3.
* Produces 'T'-tagged ENTROPY blocks which only v2.33.0+ decoders * Produces 'T'-tagged ENTROPY blocks which only v2.33.0+ decoders
* can read. Closes the real-binary compression gap vs gzip-9. */ * can read. Closes the real-binary compression gap vs gzip-9. */
int min_match = opts->format_v2 ? 3 : (int)VV_MIN_MATCH; int min_match = use_v2_fmt ? 3 : (int)VV_MIN_MATCH;
scrub_wm_t wm = {0, 0, 0, 0, 0};
while (remaining > 0) { while (remaining > 0) {
size_t braw = remaining > VV_MAX_BLOCK_SIZE ? VV_MAX_BLOCK_SIZE : remaining; size_t braw = remaining > VV_MAX_BLOCK_SIZE ? VV_MAX_BLOCK_SIZE : remaining;
@ -1850,7 +2116,7 @@ int64_t vv_compress_inner(const uint8_t *src, size_t src_len,
tmp, tcap, lit_buf, lit_cap, tmp, tcap, lit_buf, lit_cap,
stripped, ent_buf, ent_cap, stripped, ent_buf, ent_cap,
op, dst_cap - (size_t)(op - dst), min_match, op, dst_cap - (size_t)(op - dst), min_match,
opts->compat_v246_5_decoder); opts->compat_v246_5_decoder, &wm);
if (written == 0) { if (written == 0) {
free(lit_buf); free(stripped); free(ent_buf); free(lit_buf); free(stripped); free(ent_buf);
free(tmp); matcher_free(&m); free(tmp); matcher_free(&m);
@ -1861,11 +2127,19 @@ int64_t vv_compress_inner(const uint8_t *src, size_t src_len,
} }
/* Sprint 117: scrub plaintext-derived working buffers before free /* Sprint 117: scrub plaintext-derived working buffers before free
* to prevent heap-residue leak (defense in depth). */ * to prevent heap-residue leak (defense in depth).
vv_secure_zero(tmp, tcap); * SPRINT 124: scrub only up to each buffer's write watermark
if (lit_buf) vv_secure_zero(lit_buf, lit_cap); * bytes beyond it were never written and cannot hold plaintext. */
if (stripped) vv_secure_zero(stripped, tcap); vv_secure_zero(tmp, wm.tmp < tcap ? wm.tmp : tcap);
if (ent_buf) vv_secure_zero(ent_buf, ent_cap); if (lit_buf) vv_secure_zero(lit_buf, wm.lit < lit_cap ? wm.lit : lit_cap);
if (stripped) vv_secure_zero(stripped, wm.stripped < tcap ? wm.stripped : tcap);
if (ent_buf) {
vv_secure_zero(ent_buf, wm.ent_front < ent_cap ? wm.ent_front : ent_cap);
size_t back_cap = ent_cap - ent_cap / 2;
if (wm.ent_back)
vv_secure_zero(ent_buf + ent_cap / 2,
wm.ent_back < back_cap ? wm.ent_back : back_cap);
}
free(lit_buf); free(stripped); free(ent_buf); free(lit_buf); free(stripped); free(ent_buf);
free(tmp); free(tmp);
@ -1986,8 +2260,13 @@ vv_cstream_t *vv_cstream_create(const vv_options_t *opts) {
ctx->tmp = (uint8_t *)malloc(ctx->tcap); ctx->tmp = (uint8_t *)malloc(ctx->tcap);
ctx->lit_cap = VV_MAX_BLOCK_SIZE; ctx->lit_cap = VV_MAX_BLOCK_SIZE;
ctx->lit_buf = (uint8_t *)malloc(ctx->lit_cap); ctx->lit_buf = (uint8_t *)malloc(ctx->lit_cap);
ctx->stripped = (uint8_t *)malloc(ctx->lit_cap); /* SPRINT 124: stripped tokens can slightly exceed the raw block
ctx->ent_cap = vva_bound(VV_MAX_BLOCK_SIZE); * size now that emit_block lets csz [braw, braw*9/8) reach the
* entropy stage size like tmp, not like lit_buf. */
ctx->stripped = (uint8_t *)malloc(ctx->tcap);
/* SPRINT 124: 2× so Path A (front half) and Path B (back half)
* can never overlap see the matching fix in vv_compress_inner. */
ctx->ent_cap = 2 * vva_bound(VV_MAX_BLOCK_SIZE);
ctx->ent_buf = (uint8_t *)malloc(ctx->ent_cap); ctx->ent_buf = (uint8_t *)malloc(ctx->ent_cap);
/* Source window = 2 × window_size so a full block of input can /* Source window = 2 × window_size so a full block of input can
@ -2014,7 +2293,7 @@ void vv_cstream_destroy(vv_cstream_t *ctx) {
* encrypted output. All are scrubbed to prevent heap-residue leak. */ * encrypted output. All are scrubbed to prevent heap-residue leak. */
if (ctx->tmp) vv_secure_zero(ctx->tmp, ctx->tcap); if (ctx->tmp) vv_secure_zero(ctx->tmp, ctx->tcap);
if (ctx->lit_buf) vv_secure_zero(ctx->lit_buf, ctx->lit_cap); if (ctx->lit_buf) vv_secure_zero(ctx->lit_buf, ctx->lit_cap);
if (ctx->stripped) vv_secure_zero(ctx->stripped, ctx->lit_cap); if (ctx->stripped) vv_secure_zero(ctx->stripped, ctx->tcap);
if (ctx->ent_buf) vv_secure_zero(ctx->ent_buf, ctx->ent_cap); if (ctx->ent_buf) vv_secure_zero(ctx->ent_buf, ctx->ent_cap);
if (ctx->src_buf) vv_secure_zero(ctx->src_buf, ctx->src_cap); if (ctx->src_buf) vv_secure_zero(ctx->src_buf, ctx->src_cap);
free(ctx->tmp); free(ctx->lit_buf); free(ctx->stripped); free(ctx->ent_buf); free(ctx->tmp); free(ctx->lit_buf); free(ctx->stripped); free(ctx->ent_buf);
@ -2168,7 +2447,8 @@ int vv_cstream_compress_chunk(vv_cstream_t *ctx,
ctx->lit_buf, ctx->lit_cap, ctx->lit_buf, ctx->lit_cap,
ctx->stripped, ctx->ent_buf, ctx->ent_cap, ctx->stripped, ctx->ent_buf, ctx->ent_cap,
op, cap_left, stream_min_match, op, cap_left, stream_min_match,
ctx->opts.compat_v246_5_decoder); ctx->opts.compat_v246_5_decoder,
NULL /* stream scrubs full caps at destroy */);
if (block_sz == 0) return VV_ERR_OVERFLOW; if (block_sz == 0) return VV_ERR_OVERFLOW;
op += block_sz; cap_left -= block_sz; op += block_sz; cap_left -= block_sz;
} }

View file

@ -85,8 +85,23 @@ static inline void br_init(br_t *r, const uint8_t *src, size_t len) {
r->bits = 0; r->nbits = 0; r->src = src; r->pos = 0; r->len = len; r->bits = 0; r->nbits = 0; r->src = src; r->pos = 0; r->len = len;
} }
/* Refill: load bytes until accumulator is full (≥56 bits) */ /* Refill: load bytes until accumulator is full (≥56 bits).
* SPRINT 124: bulk 8-byte fast path. The byte-at-a-time loop was up
* to 7 dependent load-shift-or iterations firing every 3-4 symbols
* per stream measured as the top cost of Huffman literal decode.
* One unaligned 8-byte load + mask absorbs the same bytes; the tail
* (<8 bytes left) keeps the exact byte loop. */
static inline void br_refill(br_t *r) { static inline void br_refill(br_t *r) {
if (r->pos + 8 <= r->len) {
unsigned absorbed = (63u - (unsigned)r->nbits) >> 3; /* 0..7 */
uint64_t chunk;
memcpy(&chunk, r->src + r->pos, 8);
chunk &= ((uint64_t)1 << (absorbed * 8)) - 1;
r->bits |= chunk << r->nbits;
r->pos += absorbed;
r->nbits += (int)(absorbed * 8);
return;
}
while (r->nbits <= 56 && r->pos < r->len) { while (r->nbits <= 56 && r->pos < r->len) {
r->bits |= (uint64_t)r->src[r->pos++] << r->nbits; r->bits |= (uint64_t)r->src[r->pos++] << r->nbits;
r->nbits += 8; r->nbits += 8;
@ -767,13 +782,69 @@ vvh_error_t vvh_decode4(const uint8_t *src, size_t src_len,
} \ } \
} while (0) } while (0)
/* Variant without the per-symbol refill check, for rounds where a
* bulk refill has already guaranteed enough bits (see below). */
#define DEC_ONE_NR(R, OUT) do { \
uint32_t peek = br_peek(&(R), VVH_DECODE_BITS); \
uint32_t entry = dec->table[peek]; \
int sym = (int)(entry & 0xFF); \
int len = (int)((entry >> 8) & 0xF); \
if (VV_LIKELY(len > 0)) { \
br_consume(&(R), len); \
(OUT) = (uint8_t)sym; \
} else { \
int found = 0; \
for (int s = 0; s < dec->slow_count; s++) { \
int slen = dec->slow_len[s]; \
uint32_t mask = (1u << slen) - 1; \
if ((br_peek(&(R), slen) & mask) == dec->slow_code[s]) { \
br_consume(&(R), slen); \
(OUT) = dec->slow_sym[s]; \
found = 1; \
break; \
} \
} \
if (!found) { free(dec); return VVH_ERR_CORRUPT; } \
} \
} while (0)
/* ─── 7. Hot loop: decode 4 symbols per iteration ─── */ /* ─── 7. Hot loop: decode 4 symbols per iteration ─── */
/* Each iteration's 4 decodes are fully independent — different /* Each iteration's 4 decodes are fully independent — different
* readers, different table peeks, different output positions. * readers, different table peeks, different output positions.
* Modern OoO engines can pipeline 4 independent decode chains * Modern OoO engines can pipeline 4 independent decode chains
* achieving ~1.8-2.2× speedup over single-stream. */ * achieving ~1.8-2.2× speedup over single-stream.
*
* SPRINT 127: refill-hoisted fast rounds. One bulk refill per lane
* guarantees >= 56 accumulator bits (its 8-byte fast path applies
* whenever pos + 8 <= len, which the loop guard checks per lane),
* and three symbols consume at most 3 x VVH_MAX_CODE_LEN = 45 bits
* so each round decodes 3 symbols per lane (12 outputs) with a
* single refill branch per lane instead of one per symbol. Bit
* consumption and decode order are identical to the per-symbol
* loop; corrupt input still bottoms out at the same slow-path
* check, and nbits cannot underflow (56 - 45 >= 0). The tail and
* the last rounds fall back to the checked DEC_ONE loop. */
size_t out_idx = 0; size_t out_idx = 0;
for (size_t i = 0; i < Q; i++) { size_t i = 0;
while (i + 3 <= Q &&
r0.pos + 8 <= r0.len && r1.pos + 8 <= r1.len &&
r2.pos + 8 <= r2.len && r3.pos + 8 <= r3.len) {
br_refill(&r0); br_refill(&r1); br_refill(&r2); br_refill(&r3);
for (int k = 0; k < 3; k++) {
uint8_t y0, y1, y2, y3;
DEC_ONE_NR(r0, y0);
DEC_ONE_NR(r1, y1);
DEC_ONE_NR(r2, y2);
DEC_ONE_NR(r3, y3);
dst[out_idx + 0] = y0;
dst[out_idx + 1] = y1;
dst[out_idx + 2] = y2;
dst[out_idx + 3] = y3;
out_idx += 4;
}
i += 3;
}
for (; i < Q; i++) {
uint8_t y0, y1, y2, y3; uint8_t y0, y1, y2, y3;
DEC_ONE(r0, y0); DEC_ONE(r0, y0);
DEC_ONE(r1, y1); DEC_ONE(r1, y1);
@ -793,6 +864,7 @@ vvh_error_t vvh_decode4(const uint8_t *src, size_t src_len,
if (tail >= 3) { uint8_t y; DEC_ONE(r2, y); dst[out_idx++] = y; } if (tail >= 3) { uint8_t y; DEC_ONE(r2, y); dst[out_idx++] = y; }
#undef DEC_ONE #undef DEC_ONE
#undef DEC_ONE_NR
/* Total bytes consumed: header + stream-size header + all 4 streams */ /* Total bytes consumed: header + stream-size header + all 4 streams */
*src_consumed = streams_off + s0 + s1 + s2 + s3; *src_consumed = streams_off + s0 + s1 + s2 + s3;

View file

@ -118,11 +118,33 @@ uint16_t zupt_resolve_auto_codec(void) {
} }
static uint32_t auto_block_size(int level) { static uint32_t auto_block_size(int level) {
if (level <= 2) return 131072; /* The block IS the codec's LZ window: matches never cross a block
if (level <= 4) return 131072; * boundary, so a small block throttles the "large-window extreme"
if (level <= 6) return 262144; * parser (512 KiB gave text 3.75x where a whole-file window gives
if (level <= 7) return 262144; * 7.6x measured on codec 2.65.0). Higher levels therefore get a
return 524288; * larger block. Trade-offs held in mind: (a) block size also sets
* --dedup granularity, so the speed-first low levels (where dedup is
* most used) stay small; and (b) extreme's optimal DP is ~O(block),
* so the extreme block is bounded at 8 MiB 16 MiB bought only a few
* more percent of ratio for ~2.5x the encode time, not worth it as a
* default (raise it explicitly with -b for archival runs). Decode
* speed and memory are unaffected by block size. */
if (level <= 2) return 131072; /* fast: speed + MT + dedup granularity */
if (level <= 4) return 1u << 20; /* 1 MiB */
if (level <= 6) return 2u << 20; /* 2 MiB */
if (level <= 7) return 4u << 20; /* 4 MiB balanced: ~free, big ratio win */
return 8u << 20; /* 8 MiB extreme: large usable window */
}
/* Block size when --dedup is active. Dedup detects duplicate BLOCKS, so a
* large block almost never finds a duplicate (an 8 MiB block rarely repeats
* byte-exactly), collapsing the dedup ratio to 1.0x directly opposed to the
* large-window compression goal, which they share the one block_size knob for.
* With --dedup the user has chosen block-level dup detection, so pick a small
* block that actually finds repeats (256 KiB is the classic dedup granularity;
* finer than that costs index memory for little gain on real backups). */
static uint32_t auto_block_size_dedup(int level) {
return level <= 2 ? 131072u : 262144u;
} }
void zupt_format_size(uint64_t b, char *buf, size_t cap) { void zupt_format_size(uint64_t b, char *buf, size_t cap) {
if (b < 1024) snprintf(buf, cap, "%llu B", (unsigned long long)b); if (b < 1024) snprintf(buf, cap, "%llu B", (unsigned long long)b);
@ -768,7 +790,7 @@ zupt_error_t zupt_compress_files(const char *output_path,
const char **disk_paths, const char **disk_paths,
int num_files, int num_files,
zupt_options_t *opts) { zupt_options_t *opts) {
if (opts->block_size == 0) opts->block_size = auto_block_size(opts->level); if (opts->block_size == 0) opts->block_size = opts->dedup ? auto_block_size_dedup(opts->level) : auto_block_size(opts->level);
/* Resolve AUTO codec before compression */ /* Resolve AUTO codec before compression */
if (opts->codec_id == ZUPT_CODEC_AUTO) if (opts->codec_id == ZUPT_CODEC_AUTO)
@ -1307,7 +1329,7 @@ zupt_error_t zupt_compress_solid(const char *output_path,
const char **disk_paths, const char **disk_paths,
int num_files, int num_files,
zupt_options_t *opts) { zupt_options_t *opts) {
if (opts->block_size == 0) opts->block_size = auto_block_size(opts->level); if (opts->block_size == 0) opts->block_size = opts->dedup ? auto_block_size_dedup(opts->level) : auto_block_size(opts->level);
if (opts->block_size < 524288) opts->block_size = 524288; if (opts->block_size < 524288) opts->block_size = 524288;
/* Resolve AUTO codec before compression */ /* Resolve AUTO codec before compression */