feat: add encoder tuning harness
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -1,35 +1,92 @@
|
||||
.PHONY: hevc-nvenc x265 av1
|
||||
.PHONY: hevc-nvenc x265 x265-master av1 av1-master
|
||||
|
||||
FFMPEG=ffmpeg
|
||||
IN_FMT=rgba
|
||||
|
||||
RESOLUTION=1920x1080
|
||||
RESOLUTION=3840x2160
|
||||
FPS=60
|
||||
OUT=out.mp4
|
||||
|
||||
FFMPEG_CMD=$(FFMPEG) -f rawvideo -pix_fmt $(IN_FMT) -s $(RESOLUTION) -framerate $(FPS) -i - -vf "scale=out_color_matrix=bt709:out_range=tv:flags=accurate_rnd+full_chroma_int+bitexact,format=p010le"
|
||||
# 10 s GOP: fractal zooms have no scene cuts, so keyframes only cost bits.
|
||||
KEYINT=$(shell awk 'BEGIN { print int($(FPS) * 10 + 0.5) }')
|
||||
|
||||
FFMPEG_IN=$(FFMPEG) -f rawvideo -pix_fmt $(IN_FMT) -s $(RESOLUTION) -framerate $(FPS) -i -
|
||||
SCALE=scale=out_color_matrix=bt709:out_range=tv:flags=accurate_rnd+full_chroma_int+bitexact
|
||||
# setparams tags the frames themselves: encoders read colour info from the
|
||||
# filter output, so the -color_* options alone left primaries/transfer unset.
|
||||
TAG=setparams=colorspace=bt709:color_primaries=bt709:color_trc=bt709:range=tv
|
||||
COLOR_TAGS=-colorspace bt709 -color_primaries bt709 -color_trc bt709 -color_range tv
|
||||
|
||||
# Settings and CRFs come from tools/encode/ (see its README for the numbers).
|
||||
# Tiers: `x265`/`av1` = delivery (VMAF ~95 mean, >= 90 worst frame, smallest
|
||||
# file); `*-master` = upload master (>= 97 worst frame, size secondary).
|
||||
# The binding test clip is a faint, low-contrast Julia texture: at a CRF where
|
||||
# dense filaments already score 99, both encoders smear it, so it sets the CRF.
|
||||
|
||||
# preset medium: slow scores better at equal bitrate but runs ~2.4x slower,
|
||||
# far outside the speed budget.
|
||||
# aq-mode=3 (vs 4): 4 loses at matched bitrate on gradients and faint texture.
|
||||
# no-sao: SAO lowers VMAF at the same bitrate (smooths fine filaments).
|
||||
# merange=92: 4K zoom motion is large; ~2% smaller at no speed cost.
|
||||
# rect=1: ~7% smaller at equal or better VMAF (about 20% slower).
|
||||
# rdoq-level=2 + psy-rdoq=1: keeps the faint texture; the binding clip then
|
||||
# needs CRF ~17 instead of ~15, a smaller file overall.
|
||||
X265_PARAMS=keyint=$(KEYINT):min-keyint=$(FPS):aq-mode=3:no-sao=1:merange=92:rect=1:rdoq-level=2:psy-rdoq=1.0
|
||||
|
||||
# preset 6: fits the speed budget (p5 is ~20% slower).
|
||||
# tune=0 (VQ; ffmpeg defaults to PSNR): much better worst frames on faint texture.
|
||||
# enable-variance-boost=1: more bits to flat/low-contrast blocks; faint Julia
|
||||
# texture goes from p1 ~85 to ~93 at CRF 32.
|
||||
# enable-qm=1 (default qm-min): 3-11% smaller at equal VMAF; qm-min=0 caps
|
||||
# quality on faint texture and hurts filaments.
|
||||
# film-grain=0: synthetic grain would add noise the renders don't have.
|
||||
# tf and overlays stay default: disabling tf costs +50% bits and adds banding.
|
||||
# scd=1: untested on its own; SVT-AV1 4.2 says it won't insert keyframes at
|
||||
# scene changes anyway, so it's likely a no-op here.
|
||||
SVT_PARAMS=film-grain=0:keyint=$(KEYINT):scd=1:tune=0:enable-variance-boost=1:enable-qm=1
|
||||
|
||||
hevc-nvenc:
|
||||
$(FFMPEG_CMD) \
|
||||
$(FFMPEG_IN) -vf "$(SCALE),format=p010le" \
|
||||
-c:v hevc_nvenc -preset p7 -tune hq -rc vbr -cq 14 -b:v 0 -maxrate 200M -bufsize 400M \
|
||||
-profile:v main10 -pix_fmt p010le \
|
||||
-spatial-aq 1 -aq-strength 6 -temporal-aq 1 \
|
||||
-rc-lookahead 32 -bf 4 -b_ref_mode middle -multipass fullres \
|
||||
-colorspace bt709 -color_primaries bt709 -color_trc bt709 -color_range tv \
|
||||
$(COLOR_TAGS) \
|
||||
-tag:v hvc1 -movflags +faststart \
|
||||
$(OUT)
|
||||
|
||||
# Delivery: CRF 17 (binding clip: VMAF 94.8 mean, 92.8 worst frame).
|
||||
x265:
|
||||
$(FFMPEG_CMD) \
|
||||
-c:v libx265 -crf 18 -preset medium -pix_fmt yuv420p10le \
|
||||
-x265-params "aq-mode=3:no-sao=1" \
|
||||
-colorspace bt709 -color_primaries bt709 -color_trc bt709 -color_range tv \
|
||||
$(FFMPEG_IN) -vf "$(SCALE),format=yuv420p10le,$(TAG)" \
|
||||
-c:v libx265 -profile:v main10 -preset medium -crf 17 \
|
||||
-x265-params "$(X265_PARAMS)" \
|
||||
$(COLOR_TAGS) \
|
||||
-tag:v hvc1 -movflags +faststart \
|
||||
$(OUT)
|
||||
|
||||
# Master: CRF 12 (binding clip: 97.6 worst frame).
|
||||
x265-master:
|
||||
$(FFMPEG_IN) -vf "$(SCALE),format=yuv420p10le,$(TAG)" \
|
||||
-c:v libx265 -profile:v main10 -preset medium -crf 12 \
|
||||
-x265-params "$(X265_PARAMS)" \
|
||||
$(COLOR_TAGS) \
|
||||
-tag:v hvc1 -movflags +faststart \
|
||||
$(OUT)
|
||||
|
||||
# Delivery: CRF 32 (binding clip: VMAF 95.0 mean, 93.4 worst frame).
|
||||
av1:
|
||||
$(FFMPEG_CMD) \
|
||||
-c:v libsvtav1 -crf 24 -preset 6 -pix_fmt yuv420p10le \
|
||||
-svtav1-params "film-grain=0:enable-overlays=0" \
|
||||
-colorspace bt709 -color_primaries bt709 -color_trc bt709 -color_range tv \
|
||||
$(FFMPEG_IN) -vf "$(SCALE),format=yuv420p10le,$(TAG)" \
|
||||
-c:v libsvtav1 -preset 6 -crf 32 \
|
||||
-svtav1-params "$(SVT_PARAMS)" \
|
||||
$(COLOR_TAGS) \
|
||||
-movflags +faststart \
|
||||
$(OUT)
|
||||
|
||||
# Master: CRF 17 (binding clip: ~97.1 worst frame, interpolated from CRF 14/20).
|
||||
av1-master:
|
||||
$(FFMPEG_IN) -vf "$(SCALE),format=yuv420p10le,$(TAG)" \
|
||||
-c:v libsvtav1 -preset 6 -crf 17 \
|
||||
-svtav1-params "$(SVT_PARAMS)" \
|
||||
$(COLOR_TAGS) \
|
||||
-movflags +faststart \
|
||||
$(OUT)
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
clips/
|
||||
work/
|
||||
results/
|
||||
@@ -0,0 +1,126 @@
|
||||
# Encoder tuning
|
||||
|
||||
The x265 and SVT-AV1 settings in the `Makefile` come from this harness. It
|
||||
encodes a few lossless 4K60 test clips with a list of configs, then scores
|
||||
every encode against its source with VMAF (4K model) and CAMBI (banding).
|
||||
|
||||
```sh
|
||||
tools/encode/make_clips.sh # render clips/ once (~20 min on a GTX 1650)
|
||||
tools/encode/sweep.sh --tag x265 configs/x265-axes.txt
|
||||
tools/encode/sweep.sh --clips julia --keep --tag try my-configs.txt
|
||||
tools/encode/ladder.py results/x265-ladder.tsv # CRF / Mb/s where each family meets the targets
|
||||
```
|
||||
|
||||
Needs ffmpeg with libx265, libsvtav1 and libvmaf, the libvmaf models in
|
||||
`/usr/share/model` (`vmaf_4k_v0.6.1.json`, `vmaf_4k_v0.6.1neg.json`), and
|
||||
python3. `clips/`, `work/` and `results/` are git-ignored.
|
||||
|
||||
## Clips
|
||||
|
||||
`make_clips.sh` pipes `mandelbrot --headless --antialias` through the same
|
||||
RGBA → bt709 tv-range 10-bit conversion as the `Makefile` into lossless FFV1,
|
||||
so the clips are exactly what the encoders see and VMAF measures the encoder
|
||||
alone. Each is 4 s of 3840×2160 at 60 fps (`--size`/`--seconds` for quick tests).
|
||||
|
||||
| clip | content | what it stresses |
|
||||
|---|---|---|
|
||||
| `filaments` | seahorse valley, 1e-8 → 3e-15 | dense detail under fast scaling motion |
|
||||
| `bands` | minibrot at 3.4e-21, 1e-18 → 1.1e-20 | smooth concentric gradients: banding, blocking |
|
||||
| `julia` | embedded Julia set at 2e-30, slow zoom | low-contrast fine texture on a bright background |
|
||||
|
||||
`julia` turned out to be the hardest by far. At a CRF where the other two
|
||||
already score VMAF 99, both encoders smear its faint speckles, so it decides
|
||||
every tier's CRF.
|
||||
|
||||
## Configs and results
|
||||
|
||||
A config line is `name | ffmpeg output args`. `sweep.sh` encodes each clip,
|
||||
times it, and appends one row per encode to `results/<tag>.tsv`:
|
||||
|
||||
| column | meaning |
|
||||
|---|---|
|
||||
| `fps` | encode speed, wall clock, all cores |
|
||||
| `mbps` | bitrate |
|
||||
| `vmaf`, `vmaf_p1`, `vmaf_min` | VMAF 4K mean, 1st percentile, min |
|
||||
| `vmaf_neg` | VMAF NEG: like VMAF, but doesn't reward sharpening |
|
||||
| `cambi`, `cambi_max` | banding, mean and worst frame (0 = none, above ~5 is visible) |
|
||||
| `psnr_y` | luma PSNR |
|
||||
|
||||
VMAF uses every other frame (`n_subsample=2`). With 4 s clips that's 120
|
||||
frames, so `vmaf_p1` is the worst sampled frame. Both inputs are retimed by
|
||||
frame index before scoring. The MKV clips store 60 fps timestamps in whole
|
||||
milliseconds, and pairing by timestamp compared every third frame with its
|
||||
neighbour.
|
||||
|
||||
`ladder.py` groups configs into families by stripping a trailing `-CRF`
|
||||
(`x265-rect-16` → `x265-rect`). It interpolates, per clip, the CRF and bitrate
|
||||
where each target is met. A tier has one CRF for all content, so the lowest of
|
||||
those CRFs binds it.
|
||||
|
||||
The configs files are in the order they were run:
|
||||
|
||||
- `speed.txt`: preset speed pass (which presets fit the time budget).
|
||||
- `crf-probe.txt`: coarse CRF ranges, to see where VMAF 95–97 land.
|
||||
- `x265-axes.txt`, `av1-axes.txt`: one option at a time from a baseline.
|
||||
- `*-ladder.txt`, `*-ladder2.txt`: the promising combinations at several CRFs,
|
||||
to compare them at matched bitrate rather than at a fixed CRF.
|
||||
|
||||
## Results (Ryzen 7 3700X, 8C/16T; ffmpeg 9.0, x265 4.3, SVT-AV1 4.2.0)
|
||||
|
||||
Tier targets: **delivery** = VMAF mean ≈ 95 and worst frame ≥ 90 on every
|
||||
clip, smallest file. **Master** = worst frame ≥ 97 on every clip. `julia` binds
|
||||
every tier. "Interp" rows are interpolated between the two ladder rungs around
|
||||
that CRF (log bitrate, linear VMAF). Everything else was measured.
|
||||
|
||||
| target | CRF | clip | Mb/s | fps | VMAF mean | p1 = min | CAMBI mean / max | |
|
||||
|---|---|---|---|---|---|---|---|---|
|
||||
| `x265` | 17 | bands | 279 | 2.4 | 99.93 | 98.81 | 1.53 / 6.00 | interp 16–20 |
|
||||
| | | filaments | 241 | 3.2 | 99.95 | 98.63 | 2.54 / 8.92 | interp 16–20 |
|
||||
| | | julia | 68 | 6.1 | 94.81 | 92.75 | 0 / 0 | interp 16–20 |
|
||||
| `x265-master` | 12 | bands | 466 | 1.8 | 100.00 | 100.00 | 1.58 / 6.19 | |
|
||||
| | | filaments | 342 | 2.8 | 100.00 | 99.99 | 2.65 / 9.49 | |
|
||||
| | | julia | 171 | 3.6 | 98.36 | 97.58 | 0 / 0 | |
|
||||
| `av1` | 32 | bands | 180 | 6.0 | 99.95 | 97.07 | 1.47 / 4.70 | |
|
||||
| | | filaments | 192 | 7.7 | 99.90 | 97.04 | 2.28 / 7.14 | |
|
||||
| | | julia | 50 | 7.4 | 95.03 | 93.39 | 0 / 0 | |
|
||||
| `av1-master` | 17 | bands | 451 | 6.9 | 100.00 | 99.49 | 1.23 / 4.87 | interp 14–20 |
|
||||
| | | filaments | 380 | 8.6 | 100.00 | 99.50 | 2.40 / 7.67 | interp 14–20 |
|
||||
| | | julia | 152 | 9.0 | 98.09 | 97.12 | 0 / 0 | interp 14–20 |
|
||||
|
||||
SVT-AV1 wins both tiers: the delivery files are 20–35% smaller than
|
||||
x265's, and it encodes 1.2–2.5× faster. CAMBI max is the same at every CRF on
|
||||
`filaments` (6–9), so it comes from the content, not the encoder.
|
||||
|
||||
What the sweeps showed (details as comments in the `Makefile`):
|
||||
|
||||
- **SVT-AV1:** `tune=0` + `enable-variance-boost=1` + `enable-qm=1` is the
|
||||
combination that fixes faint texture. With the defaults, `julia` scores a
|
||||
worst frame of 85 at CRF 26. The temporal filter (`enable-tf`) was the
|
||||
suspect for smeared filaments, but turning it off or weakening it cost
|
||||
35–50% more bits for lower VMAF and more banding. `qm-min=0` caps quality.
|
||||
`sharpness`, `ac-bias` and `enable-overlays` did nothing measurable.
|
||||
- **x265:** `rect`, `merange=92` and `rdoq-level=2:psy-rdoq=1` help. SAO hurts.
|
||||
`aq-mode=4`, `deblock=-1`, `psy-rd=1`, `me=star`, `rc-lookahead=60` and
|
||||
`bframes=8` didn't help at matched bitrate or cost too much speed.
|
||||
- The worst SVT frames without variance boost sit in the clip's last
|
||||
mini-GOP, so part of that p1 is an artefact of a 4 s clip.
|
||||
|
||||
**Speed budget.** It was 3–5 fps on an i7-1165G7 (4C/8T). Pinning this machine
|
||||
to 4C/8T (`taskset -c 0-3,8-11`) as a proxy gives 1.5× (SVT) to 1.8× (x265)
|
||||
for the full 16 threads, so here the budget is about 5–9 fps. SVT preset 6
|
||||
fits at both tiers. x265 `medium` with these options doesn't: at the low CRFs
|
||||
`julia` forces, it runs at 2–6 fps. This SVT-AV1 package also reports
|
||||
`(debug)`: it's built without `NDEBUG`, but it's optimised and it's what
|
||||
ffmpeg links.
|
||||
|
||||
**Not yet measured:** x265 `-preset fast` with these options (the speed fix
|
||||
for x265) and `aq-strength=1.2` (`configs/x265-ladder2.txt`). Also a
|
||||
confirming encode at `av1-master`'s interpolated CRF 17.
|
||||
|
||||
## Tips
|
||||
|
||||
- Keep the machine idle while sweeping: `fps` is wall clock time.
|
||||
- A config that changes bitrate at a fixed CRF can't be judged from one row.
|
||||
Put it on a ladder and compare with `ladder.py`.
|
||||
- `--keep` keeps the encodes in `work/`, e.g. to extract frames to compare by eye:
|
||||
`ffmpeg -i work/julia--x.mp4 -vf "select=eq(n\,120),crop=480:270:1680:945" -frames:v 1 x.png`.
|
||||
@@ -0,0 +1,16 @@
|
||||
# SVT-AV1 4.2 per-axis sweep: each line changes one thing from av1-base.
|
||||
# Base = pass-1 settings (preset 6, no film grain) plus a 10 s GOP and scene-cut detection.
|
||||
av1-base | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 30 -svtav1-params film-grain=0:keyint=600:scd=1
|
||||
# ffmpeg's libsvtav1 defaults to tune=1 (PSNR); 0 is the visual-quality tune
|
||||
av1-tune0 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 30 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0
|
||||
# Alt-ref temporal filter: prime suspect for smeared filaments during zooms
|
||||
av1-tf0 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 30 -svtav1-params film-grain=0:keyint=600:scd=1:enable-tf=0
|
||||
av1-tfs1 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 30 -svtav1-params film-grain=0:keyint=600:scd=1:tf-strength=1
|
||||
# Variance boost: more bits to flat, low-variance blocks (gradient banding)
|
||||
av1-vb | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 30 -svtav1-params film-grain=0:keyint=600:scd=1:enable-variance-boost=1
|
||||
av1-vb3 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 30 -svtav1-params film-grain=0:keyint=600:scd=1:enable-variance-boost=1:variance-boost-strength=3
|
||||
# Quantisation matrices, deblocking sharpness, AC bias (texture), overlays
|
||||
av1-qm | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 30 -svtav1-params film-grain=0:keyint=600:scd=1:enable-qm=1:qm-min=0
|
||||
av1-sharp1 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 30 -svtav1-params film-grain=0:keyint=600:scd=1:sharpness=1
|
||||
av1-acb1 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 30 -svtav1-params film-grain=0:keyint=600:scd=1:ac-bias=1.0
|
||||
av1-ov1 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 30 -svtav1-params film-grain=0:keyint=600:scd=1:enable-overlays=1
|
||||
@@ -0,0 +1,19 @@
|
||||
# SVT-AV1 candidate ladders: compare combinations at matched bitrate (results/av1-ladder.tsv)
|
||||
av1-c0-14 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 14 -svtav1-params film-grain=0:keyint=600:scd=1
|
||||
av1-c0-20 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 20 -svtav1-params film-grain=0:keyint=600:scd=1
|
||||
av1-c0-26 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 26 -svtav1-params film-grain=0:keyint=600:scd=1
|
||||
av1-qm-14 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 14 -svtav1-params film-grain=0:keyint=600:scd=1:enable-qm=1:qm-min=0
|
||||
av1-qm-20 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 20 -svtav1-params film-grain=0:keyint=600:scd=1:enable-qm=1:qm-min=0
|
||||
av1-qm-26 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 26 -svtav1-params film-grain=0:keyint=600:scd=1:enable-qm=1:qm-min=0
|
||||
av1-tune0-14 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 14 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0
|
||||
av1-tune0-20 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 20 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0
|
||||
av1-tune0-26 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 26 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0
|
||||
av1-vb-14 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 14 -svtav1-params film-grain=0:keyint=600:scd=1:enable-variance-boost=1
|
||||
av1-vb-20 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 20 -svtav1-params film-grain=0:keyint=600:scd=1:enable-variance-boost=1
|
||||
av1-vb-26 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 26 -svtav1-params film-grain=0:keyint=600:scd=1:enable-variance-boost=1
|
||||
av1-qm-vb-14 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 14 -svtav1-params film-grain=0:keyint=600:scd=1:enable-qm=1:qm-min=0:enable-variance-boost=1
|
||||
av1-qm-vb-20 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 20 -svtav1-params film-grain=0:keyint=600:scd=1:enable-qm=1:qm-min=0:enable-variance-boost=1
|
||||
av1-qm-vb-26 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 26 -svtav1-params film-grain=0:keyint=600:scd=1:enable-qm=1:qm-min=0:enable-variance-boost=1
|
||||
av1-qm-vb-tune0-14 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 14 -svtav1-params film-grain=0:keyint=600:scd=1:enable-qm=1:qm-min=0:enable-variance-boost=1:tune=0
|
||||
av1-qm-vb-tune0-20 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 20 -svtav1-params film-grain=0:keyint=600:scd=1:enable-qm=1:qm-min=0:enable-variance-boost=1:tune=0
|
||||
av1-qm-vb-tune0-26 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 26 -svtav1-params film-grain=0:keyint=600:scd=1:enable-qm=1:qm-min=0:enable-variance-boost=1:tune=0
|
||||
@@ -0,0 +1,12 @@
|
||||
# SVT-AV1 round 2: tune0+VB without QM, default qm-min, delivery-range CRFs (results/av1-ladder2.tsv)
|
||||
av1-vb-tune0-14 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 14 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0:enable-variance-boost=1
|
||||
av1-vb-tune0-20 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 20 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0:enable-variance-boost=1
|
||||
av1-vb-tune0-26 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 26 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0:enable-variance-boost=1
|
||||
av1-vb-tune0-32 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 32 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0:enable-variance-boost=1
|
||||
av1-qm8-vb-tune0-14 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 14 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0:enable-qm=1:enable-variance-boost=1
|
||||
av1-qm8-vb-tune0-20 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 20 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0:enable-qm=1:enable-variance-boost=1
|
||||
av1-qm8-vb-tune0-26 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 26 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0:enable-qm=1:enable-variance-boost=1
|
||||
av1-qm8-vb-tune0-32 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 32 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0:enable-qm=1:enable-variance-boost=1
|
||||
av1-qm-vb-tune0-32 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 32 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0:enable-qm=1:qm-min=0:enable-variance-boost=1
|
||||
av1-qm-vb-tune0-38 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 38 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0:enable-qm=1:qm-min=0:enable-variance-boost=1
|
||||
av1-tune0-10 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 10 -svtav1-params film-grain=0:keyint=600:scd=1:tune=0
|
||||
@@ -0,0 +1,9 @@
|
||||
# Coarse CRF probe: where do VMAF 95-97 land on each clip? (pass-1 baselines)
|
||||
x265-crf22 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 22 -x265-params log-level=error:aq-mode=3:no-sao=1
|
||||
x265-crf26 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 26 -x265-params log-level=error:aq-mode=3:no-sao=1
|
||||
x265-crf30 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 30 -x265-params log-level=error:aq-mode=3:no-sao=1
|
||||
x265-crf34 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 34 -x265-params log-level=error:aq-mode=3:no-sao=1
|
||||
av1-crf32 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 32 -svtav1-params film-grain=0
|
||||
av1-crf38 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 38 -svtav1-params film-grain=0
|
||||
av1-crf44 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 44 -svtav1-params film-grain=0
|
||||
av1-crf50 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 50 -svtav1-params film-grain=0
|
||||
@@ -0,0 +1,8 @@
|
||||
# Pass 1: which presets fit the 3-5 fps budget at 4K on this machine.
|
||||
x265-fast | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset fast -crf 18 -x265-params log-level=error:aq-mode=3:no-sao=1
|
||||
x265-medium | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 18 -x265-params log-level=error:aq-mode=3:no-sao=1
|
||||
x265-slow | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset slow -crf 18 -x265-params log-level=error:aq-mode=3:no-sao=1
|
||||
av1-p5 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 5 -crf 26 -svtav1-params film-grain=0
|
||||
av1-p6 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 6 -crf 26 -svtav1-params film-grain=0
|
||||
av1-p7 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 7 -crf 26 -svtav1-params film-grain=0
|
||||
av1-p8 | -pix_fmt yuv420p10le -c:v libsvtav1 -preset 8 -crf 26 -svtav1-params film-grain=0
|
||||
@@ -0,0 +1,20 @@
|
||||
# x265 per-axis sweep: each line changes one thing from x265-base.
|
||||
# Base = pass-1 settings (preset medium, aq-mode 3, no SAO) plus a 10 s GOP.
|
||||
x265-base | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1
|
||||
# Adaptive quantisation
|
||||
x265-aq4 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=4:no-sao=1
|
||||
x265-aqs08 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:aq-strength=0.8:no-sao=1
|
||||
x265-aqs12 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:aq-strength=1.2:no-sao=1
|
||||
# In-loop filters
|
||||
x265-sao | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3
|
||||
x265-selsao2 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:selective-sao=2
|
||||
x265-deblk-1 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:deblock=-1,-1
|
||||
# Psychovisual: medium has rdoq-level 0, so psy-rdoq needs rdoq-level 2 to do anything
|
||||
x265-psyrdoq | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:rdoq-level=2:psy-rdoq=1.0
|
||||
x265-psyrd1 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:psy-rd=1.0
|
||||
# Motion: zooms are scaling, not translation
|
||||
x265-star | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:me=star
|
||||
x265-mer92 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92
|
||||
x265-bf8 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:bframes=8
|
||||
x265-la60 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:rc-lookahead=60
|
||||
x265-rect | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:rect=1
|
||||
@@ -0,0 +1,13 @@
|
||||
# x265 candidate ladders: compare combinations at matched bitrate (results/x265-ladder.tsv)
|
||||
x265-c0-12 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 12 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92
|
||||
x265-c0-16 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 16 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92
|
||||
x265-c0-20 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92
|
||||
x265-rect-12 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 12 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1
|
||||
x265-rect-16 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 16 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1
|
||||
x265-rect-20 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1
|
||||
x265-rect-psyrdoq-12 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 12 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1:rdoq-level=2:psy-rdoq=1.0
|
||||
x265-rect-psyrdoq-16 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 16 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1:rdoq-level=2:psy-rdoq=1.0
|
||||
x265-rect-psyrdoq-20 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1:rdoq-level=2:psy-rdoq=1.0
|
||||
x265-rect-aq4-12 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 12 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=4:no-sao=1:merange=92:rect=1
|
||||
x265-rect-aq4-16 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 16 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=4:no-sao=1:merange=92:rect=1
|
||||
x265-rect-aq4-20 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=4:no-sao=1:merange=92:rect=1
|
||||
@@ -0,0 +1,7 @@
|
||||
# x265 round 2: preset fast for speed, aq-strength 1.2 for julia (results/x265-ladder2.tsv)
|
||||
x265-fast-rp-12 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset fast -crf 12 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1:rdoq-level=2:psy-rdoq=1.0
|
||||
x265-fast-rp-16 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset fast -crf 16 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1:rdoq-level=2:psy-rdoq=1.0
|
||||
x265-fast-rp-20 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset fast -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1:rdoq-level=2:psy-rdoq=1.0
|
||||
x265-rp-aqs12-12 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 12 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1:rdoq-level=2:psy-rdoq=1.0:aq-strength=1.2
|
||||
x265-rp-aqs12-16 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 16 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1:rdoq-level=2:psy-rdoq=1.0:aq-strength=1.2
|
||||
x265-rp-aqs12-20 | -pix_fmt yuv420p10le -c:v libx265 -profile:v main10 -preset medium -crf 20 -x265-params log-level=error:keyint=600:min-keyint=60:aq-mode=3:no-sao=1:merange=92:rect=1:rdoq-level=2:psy-rdoq=1.0:aq-strength=1.2
|
||||
Executable
+62
@@ -0,0 +1,62 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Where does each config family meet a VMAF target? Reads sweep.sh TSVs.
|
||||
|
||||
tools/encode/ladder.py results/x265-ladder.tsv [--target vmaf:95 --target p1:90 ...]
|
||||
|
||||
A family is a config name minus its trailing -CRF (x265-rect-16 -> x265-rect;
|
||||
x265-crf16 -> x265).
|
||||
For each family and clip, log(Mb/s) and CRF are interpolated linearly in the
|
||||
chosen metric between the two ladder rungs that bracket the target. A tier
|
||||
uses one CRF for everything, so the clip needing the lowest CRF binds it; that
|
||||
CRF is printed last as "all". A dash means the ladder doesn't reach the target.
|
||||
"""
|
||||
import argparse
|
||||
import csv
|
||||
import math
|
||||
import re
|
||||
from collections import defaultdict
|
||||
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("tsv", nargs="+")
|
||||
ap.add_argument("--target", action="append",
|
||||
help="metric:value, metric one of vmaf, p1, min, neg (default vmaf:95, p1:90, p1:97)")
|
||||
args = ap.parse_args()
|
||||
targets = [(m, float(v)) for m, v in (t.split(":") for t in (args.target or ["vmaf:95", "p1:90", "p1:97"]))]
|
||||
col = {"vmaf": "vmaf", "p1": "vmaf_p1", "min": "vmaf_min", "neg": "vmaf_neg"}
|
||||
|
||||
fam = defaultdict(lambda: defaultdict(list)) # family -> clip -> [(crf, row)]
|
||||
for path in args.tsv:
|
||||
for r in csv.DictReader(open(path), delimiter="\t"):
|
||||
m = re.fullmatch(r"(.*)-(?:crf)?(\d+)", r["config"])
|
||||
if m:
|
||||
fam[m[1]][r["clip"]].append((int(m[2]), r))
|
||||
|
||||
|
||||
def solve(pts, key, target):
|
||||
"""(crf, mbps, fps) where metric `key` crosses `target`, or None."""
|
||||
pts = sorted(pts, key=lambda p: p[0]) # quality falls as CRF rises
|
||||
for (c0, a), (c1, b) in zip(pts, pts[1:]):
|
||||
q0, q1 = float(a[key]), float(b[key])
|
||||
if q0 >= target >= q1 and q0 != q1:
|
||||
t = (q0 - target) / (q0 - q1)
|
||||
lr = math.log(float(a["mbps"])) * (1 - t) + math.log(float(b["mbps"])) * t
|
||||
fps = float(a["fps"]) * (1 - t) + float(b["fps"]) * t
|
||||
return c0 + t * (c1 - c0), math.exp(lr), fps
|
||||
return None
|
||||
|
||||
|
||||
for metric, target in targets:
|
||||
print(f"\n## {metric} >= {target:g} (CRF @ Mb/s, fps)")
|
||||
clips = sorted({c for f in fam.values() for c in f})
|
||||
print(f"{'family':24}" + "".join(f"{c:>22}" for c in clips) + f"{'all (binding clip)':>24}")
|
||||
for f, per in fam.items():
|
||||
cells, worst = [], None
|
||||
for c in clips:
|
||||
s = solve(per.get(c, []), col[metric], target)
|
||||
cells.append(f"{s[0]:5.1f} @ {s[1]:6.1f}, {s[2]:4.1f}" if s else "-")
|
||||
if s is None:
|
||||
worst = "missing"
|
||||
elif worst != "missing" and (worst is None or s[0] < worst[0]):
|
||||
worst = (s[0], c)
|
||||
w = "-" if worst in (None, "missing") else f"CRF {worst[0]:4.1f} ({worst[1]})"
|
||||
print(f"{f:24}" + "".join(f"{x:>22}" for x in cells) + f"{w:>24}")
|
||||
Executable
+70
@@ -0,0 +1,70 @@
|
||||
#!/usr/bin/env bash
|
||||
# Render the lossless 4K60 test clips that sweep.sh encodes.
|
||||
#
|
||||
# Each clip goes through the same RGBA -> bt709 tv-range 10-bit conversion as
|
||||
# the Makefile, then into lossless FFV1. The clips are therefore exactly what the
|
||||
# encoders see, so VMAF against them measures the encoder alone.
|
||||
#
|
||||
# tools/encode/make_clips.sh [--size WxH] [--seconds S] [--force] [clip...]
|
||||
set -euo pipefail
|
||||
|
||||
here=$(cd "$(dirname "$0")" && pwd)
|
||||
root=$(cd "$here/../.." && pwd)
|
||||
out="$here/clips"
|
||||
size=3840x2160
|
||||
seconds=4
|
||||
fps=60
|
||||
force=0
|
||||
only=()
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case $1 in
|
||||
--size) size=$2; shift 2 ;;
|
||||
--seconds) seconds=$2; shift 2 ;;
|
||||
--force) force=1; shift ;;
|
||||
-h|--help) sed -n '2,9p' "$0"; exit 0 ;;
|
||||
*) only+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
|
||||
# Seahorse valley, minibrot at 3.7e-41 (tools/deep-zoom/LOCATIONS.md).
|
||||
SH_RE=-0.77568376800905379745347652613832924487504096622022
|
||||
SH_IM=0.13646736829469012473375311880735014411233827361594
|
||||
# Seahorse valley, minibrot at 3.4e-21 (period 429, `find_deep.py seahorse 10`).
|
||||
# Same look as the 1e-40 one for this purpose, but ~10x faster to render.
|
||||
MB_RE=-0.775683767855941232535147681616
|
||||
MB_IM=0.136467368300490586441979944945
|
||||
|
||||
# name | mandelbrot flags. Zoom speeds bracket a typical 60 s dive (~0.7 decade/s).
|
||||
clips=(
|
||||
# Worst case: dense filaments, 1.5 decades/s of scaling motion.
|
||||
"filaments|--view=$SH_RE,$SH_IM,1e-8 --to-view=$SH_RE,$SH_IM,3e-15 --linear"
|
||||
# Smooth concentric bands closing in on the minibrot: banding/blocking risk.
|
||||
"bands|--view=$MB_RE,$MB_IM,1e-18 --to-view=$MB_RE,$MB_IM,1.1e-20 --linear"
|
||||
# Embedded Julia set, slow drift: fine static texture, psy/detail retention.
|
||||
"julia|--view=$SH_RE,$SH_IM,2e-30 --to-view=$SH_RE,$SH_IM,1e-30 --linear"
|
||||
)
|
||||
|
||||
[[ -x $root/target/release/mandelbrot ]] || cargo build --release --manifest-path "$root/Cargo.toml"
|
||||
mkdir -p "$out"
|
||||
w=${size%x*}
|
||||
h=${size#*x}
|
||||
|
||||
for entry in "${clips[@]}"; do
|
||||
name=${entry%%|*}
|
||||
flags=${entry#*|}
|
||||
if [[ ${#only[@]} -gt 0 && ! " ${only[*]} " =~ " $name " ]]; then continue; fi
|
||||
dst="$out/$name.mkv"
|
||||
if [[ -f $dst && $force = 0 ]]; then echo "skip $name (exists, --force to redo)"; continue; fi
|
||||
echo "render $name ($size, ${seconds}s @ ${fps}fps)"
|
||||
# shellcheck disable=SC2086
|
||||
"$root/target/release/mandelbrot" --headless --antialias $flags \
|
||||
--width "$w" --height "$h" --fps "$fps" --duration "$seconds" --export-path - |
|
||||
ffmpeg -hide_banner -loglevel error -y \
|
||||
-f rawvideo -pix_fmt rgba -s "$size" -framerate "$fps" -i - \
|
||||
-vf "scale=out_color_matrix=bt709:out_range=tv:flags=accurate_rnd+full_chroma_int+bitexact,format=yuv420p10le" \
|
||||
-c:v ffv1 -level 3 -slices 16 -g 1 \
|
||||
-colorspace bt709 -color_primaries bt709 -color_trc bt709 -color_range tv \
|
||||
"$dst.tmp.mkv"
|
||||
mv "$dst.tmp.mkv" "$dst"
|
||||
done
|
||||
Executable
+89
@@ -0,0 +1,89 @@
|
||||
#!/usr/bin/env bash
|
||||
# Encode every test clip with every config, then score each encode.
|
||||
#
|
||||
# tools/encode/sweep.sh [--clips a,b] [--keep] [--tag NAME] CONFIG_FILE...
|
||||
#
|
||||
# A config line is `name | ffmpeg output args` (blank lines and # comments are
|
||||
# skipped). The args go between `-i clip.mkv` and the output file, e.g.
|
||||
# x265-crf18 | -c:v libx265 -preset medium -crf 18 -x265-params aq-mode=3
|
||||
# Each row of results/<tag>.tsv has encode fps, bitrate, VMAF 4K (mean, 1st
|
||||
# percentile, min), VMAF-NEG mean, CAMBI banding (mean, max) and PSNR-Y.
|
||||
set -euo pipefail
|
||||
|
||||
here=$(cd "$(dirname "$0")" && pwd)
|
||||
clipdir="$here/clips"
|
||||
model=/usr/share/model
|
||||
export SVT_LOG=1 # errors only
|
||||
clips=""
|
||||
keep=0
|
||||
tag=$(date +%Y%m%d-%H%M%S)
|
||||
configs=()
|
||||
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case $1 in
|
||||
--clips) clips=$2; shift 2 ;;
|
||||
--keep) keep=1; shift ;;
|
||||
--tag) tag=$2; shift 2 ;;
|
||||
-h|--help) sed -n '2,11p' "$0"; exit 0 ;;
|
||||
*) configs+=("$1"); shift ;;
|
||||
esac
|
||||
done
|
||||
[[ ${#configs[@]} -gt 0 ]] || { echo "usage: $0 [--clips a,b] [--keep] [--tag NAME] CONFIG_FILE..." >&2; exit 1; }
|
||||
|
||||
if [[ -z $clips ]]; then
|
||||
clips=$(cd "$clipdir" && ls ./*.mkv | sed 's|^\./||; s|\.mkv$||' | paste -sd,)
|
||||
fi
|
||||
[[ -n $clips ]] || { echo "no clips in $clipdir: run make_clips.sh first" >&2; exit 1; }
|
||||
|
||||
mkdir -p "$here/results" "$here/work"
|
||||
tsv="$here/results/$tag.tsv"
|
||||
[[ -f $tsv ]] || printf 'clip\tconfig\tfps\tmbps\tvmaf\tvmaf_p1\tvmaf_min\tvmaf_neg\tcambi\tcambi_max\tpsnr_y\n' > "$tsv"
|
||||
|
||||
IFS=, read -ra clip_list <<< "$clips"
|
||||
for cfg in "${configs[@]}"; do
|
||||
while IFS= read -r line || [[ -n $line ]]; do
|
||||
[[ $line =~ ^[[:space:]]*(#|$) ]] && continue
|
||||
name=$(sed 's/[[:space:]]*|.*//' <<< "$line")
|
||||
args=$(sed 's/^[^|]*|[[:space:]]*//' <<< "$line")
|
||||
for clip in "${clip_list[@]}"; do
|
||||
ref="$clipdir/$clip.mkv"
|
||||
enc="$here/work/$clip--$name.mp4"
|
||||
log="$here/work/$clip--$name.json"
|
||||
frames=$(ffprobe -v error -count_packets -select_streams v:0 \
|
||||
-show_entries stream=nb_read_packets -of csv=p=0 "$ref")
|
||||
printf '%-10s %-40s ' "$clip" "$name"
|
||||
|
||||
t0=$(date +%s.%N)
|
||||
# shellcheck disable=SC2086
|
||||
ffmpeg -hide_banner -loglevel error -y -i "$ref" $args -an "$enc" < /dev/null
|
||||
t1=$(date +%s.%N)
|
||||
|
||||
# 4K model for both VMAF variants; CAMBI flags banding, which VMAF
|
||||
# barely sees and which is the main risk on smooth fractal gradients.
|
||||
# Both streams are retimed by frame index: MKV stores 60 fps in whole
|
||||
# ms, and pairing by PTS matched every third frame with its neighbour.
|
||||
ffmpeg -hide_banner -loglevel error -i "$enc" -i "$ref" -lavfi \
|
||||
"[0:v]settb=1/60,setpts=N[d];[1:v]settb=1/60,setpts=N[r];[d][r]libvmaf=log_fmt=json:log_path=$log:n_threads=$(nproc):n_subsample=2:model='path=$model/vmaf_4k_v0.6.1.json\\:name=vmaf|path=$model/vmaf_4k_v0.6.1neg.json\\:name=neg':feature='name=cambi|name=psnr'" \
|
||||
-f null - < /dev/null
|
||||
|
||||
row=$(python3 - "$log" "$enc" "$frames" "$t0" "$t1" <<'PY'
|
||||
import json, os, sys
|
||||
log, enc, frames, t0, t1 = sys.argv[1], sys.argv[2], int(sys.argv[3]), float(sys.argv[4]), float(sys.argv[5])
|
||||
fr = json.load(open(log))["frames"]
|
||||
col = lambda k: sorted(f["metrics"][k] for f in fr)
|
||||
v = col("vmaf")
|
||||
p1 = v[max(0, int(0.01 * len(v)) - 1)] if len(v) >= 100 else v[0]
|
||||
mean = lambda xs: sum(xs) / len(xs)
|
||||
cambi = col("cambi")
|
||||
fps = frames / (t1 - t0)
|
||||
mbps = os.path.getsize(enc) * 8 / (frames / 60) / 1e6
|
||||
print(f"{fps:.2f}\t{mbps:.1f}\t{mean(v):.2f}\t{p1:.2f}\t{v[0]:.2f}\t{mean(col('neg')):.2f}\t{mean(cambi):.2f}\t{cambi[-1]:.2f}\t{mean(col('psnr_y')):.2f}")
|
||||
PY
|
||||
)
|
||||
printf '%s\t%s\t%s\n' "$clip" "$name" "$row" >> "$tsv"
|
||||
awk -F'\t' '{printf "%6s fps %7s Mb/s vmaf %s p1 %s min %s neg %s cambi %s/%s psnr %s\n",$1,$2,$3,$4,$5,$6,$7,$8,$9}' <<< "$row"
|
||||
[[ $keep = 1 ]] || rm -f "$enc"
|
||||
done
|
||||
done < "$cfg"
|
||||
done
|
||||
echo "results: $tsv"
|
||||
Reference in New Issue
Block a user