-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathopen-transcode.py
More file actions
executable file
·9527 lines (8561 loc) · 431 KB
/
Copy pathopen-transcode.py
File metadata and controls
executable file
·9527 lines (8561 loc) · 431 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
#!/usr/bin/env python3
"""OpenTranscode — open-source batch video transcoder (single file).
The whole application lives in this one module: title scrubbing, codec
and container profiles (AV1 / VP9 / x265 / Theora with its inverted
quality-scale wrapper), the QThread encoder worker (CPU / NVENC /
hybrid), the distro-aware environment probe, the from-git rebuild
builder, and the PySide6 media-console UI. Sections are laid out in
dependency order with external imports hoisted to the header.
Runs with nothing but Python + PySide6 on the box — no installation,
no package directory. The inlined build_parser / main / launch_gui
provide library/CLI access; executed directly, the script opens the
GUI and ignores arguments.
Author: Jeremy Anderson — https://dcos.net — info@dcos.net
License: AGPL-3.0
"""
__version__ = "4.11.0"
__author__ = "Jeremy Anderson"
__website__ = "https://dcos.net"
__email__ = "info@dcos.net"
__license__ = "AGPL-3.0"
import argparse
from collections.abc import Callable
import ctypes
from dataclasses import dataclass
from dataclasses import dataclass, field
import hashlib
import io
import json
import math
import os
from pathlib import Path
import platform
import re
import shutil
import signal
import site
import subprocess
import sys
import tempfile
import threading
import time
from PySide6.QtCore import QThread, Signal
from PySide6.QtCore import Qt, QTimer, Slot
from PySide6.QtCore import Qt, Signal, QPointF, QRectF
from PySide6.QtGui import (
QFont, QColor, QPainter, QPen, QBrush,
QRadialGradient, QFontMetrics,
)
from PySide6.QtGui import QFont, QPalette, QColor
from PySide6.QtWidgets import (
QApplication, QMainWindow, QWidget, QVBoxLayout, QHBoxLayout,
QLabel, QLineEdit, QPushButton, QComboBox, QCheckBox,
QTextEdit, QFileDialog, QGroupBox, QStatusBar, QMessageBox,
QStyleFactory,
)
from PySide6.QtWidgets import QWidget
# ════════════════════════════════════════════════════════════════════════════
# ═══ title_clean ═══
# ════════════════════════════════════════════════════════════════════════════
"""Title-artifact scrubbing for output filenames.
A release title describes the SOURCE encode (`Movie.x264.1080p.WEBRip.
x265-GRP.mkv`). After this tool re-encodes it, every codec and container
token in that title is wrong: the file is now whatever VIDEO/AUDIO/
CONTAINER say (typically AV1/Opus/MKV — Theora/OGV and the other
profiles equally). Carrying the stale tags forward misdescribes every
output in the archive.
``clean_title`` strips three artifact classes from a source stem:
1. Codec/container tokens — video (x264 … theora), audio (aac …
vorbis) and container (avi, webm, …), matched case-insensitively
on token boundaries only, so `MP4Box`, `Aviator` or `H264file`
survive untouched.
2. This app's own output suffixes — ``_archived`` and ``_<w>x<h>``
resolution markers, so re-encodes never stack suffixes.
3. Separator residue — runs left behind where a token was removed
collapse to their first separator, emptied bracket pairs go away,
and leading/trailing separators are trimmed.
Pure function, no I/O, no internal dependencies — the scrub runs as one
compiled regex per artifact class (the token alternation is built from
the table, longest-first, so `svt-av1` matches before its `av1`
substring). Falls back to the unmodified stem when scrubbing would
empty it, so a file literally named `x264.mkv` still gets a valid
output name.
"""
# Token table — single source of truth for what a title may not claim.
TITLE_ARTIFACT_TOKENS: frozenset[str] = frozenset({
# Video codecs (legacy and current — the title must never claim one)
"x264", "h264", "h.264", "avc",
"x265", "h265", "h.265", "hevc",
"xvid", "divx", "theora", "dirac",
"vp8", "vp9", "mpeg2", "mpeg4",
"av1", "svt-av1", "svtav1",
"vc1", "vc-1",
# Audio codecs (output audio is the AUDIO profile's choice)
"aac", "he-aac", "ac3", "eac3", "dts", "dtshd", "truehd",
"atmos", "mp3", "flac", "opus", "vorbis",
# Containers (output container is the CONTAINER profile's choice)
"avi", "webm", "wmv", "mp4", "m4v", "mkv", "mov", "mpg", "mpeg",
"ts", "vob", "ogv", "ogg", "flv", "m2ts",
})
_SEPARATOR_RUN = r"[._\- \[\]]+"
# A token matches only as a standalone word: the character before it
# must not be alphanumeric (releases bracket or dot their tags), and the
# character after it must not be either. Alternation is longest-first,
# so overlapping tokens (`svt-av1` ⊃ `av1`, `he-aac` ⊃ `aac`) resolve
# to the full tag.
_TOKEN_RE: re.Pattern[str] = re.compile(
r"(?i)(?<![a-z0-9])("
+ "|".join(sorted(map(re.escape, TITLE_ARTIFACT_TOKENS), key=len, reverse=True))
+ r")(?![a-z0-9])"
)
# This app's own suffix artifacts at end-of-stem: `_<w>x<h>_archived`,
# `_archived`, or a bare `_<w>x<h>` marker.
_SUFFIX_RE: re.Pattern[str] = re.compile(r"_\d+x\d+_archived$|_archived$|_\d+x\d+$")
# Residue after token removal.
_COLLAPSE_RE: re.Pattern[str] = re.compile(r"([._\- \[\]])" + _SEPARATOR_RUN)
_EMPTY_BRACKETS_RE: re.Pattern[str] = re.compile(r"\[\s*\]")
_EDGE_RE: re.Pattern[str] = re.compile(r"^[._\- \[\]]+|[._\- \[\]]+$")
def clean_title(stem: str) -> str:
"""Return *stem* with codec/container/app-artifact tokens removed."""
scrubbed = _SUFFIX_RE.sub("", stem)
scrubbed = _TOKEN_RE.sub("", scrubbed)
scrubbed = _EMPTY_BRACKETS_RE.sub("", scrubbed)
scrubbed = _COLLAPSE_RE.sub(r"\1", scrubbed)
scrubbed = _EDGE_RE.sub("", scrubbed).strip()
return scrubbed or stem
# ════════════════════════════════════════════════════════════════════════════
# ═══ codec_profiles ═══
# ════════════════════════════════════════════════════════════════════════════
"""Codec / audio / container profile tables and helpers.
Data-driven configuration that replaces the v1 if/else codec chains.
Pure data + pure functions — no PySide6, no I/O, no internal package
dependencies. Safe to import from any context (incl. unit tests and
the CLI --version path).
"""
# ──────────────────────────────────────────────
# CONFIG-DRIVEN PROFILES (replaces all if/else chains)
# ──────────────────────────────────────────────
@dataclass
class VideoCodecProfile:
label: str # Display name in combo box
av1an_encoder: str # Encoder name passed to --encoder
ffmpeg_encoder: str # Encoder name for pure-ffmpeg fallback (e.g. "libsvtav1")
container: str # Default container extension (mkv or webm)
crf_range: tuple[int, int] # (min, max) valid CRF values
default_crf: int
# (crf, preset) -> av1an --video-params string. Passed to SvtAv1EncApp /
# vpxenc / x265 as a CLI invocation, so ONLY CLI-accepted flags may
# appear here. Thread capping lives in ffmpeg_vargs_fn (where
# libsvtav1 is invoked as a library and accepts -threads) and in
# EncoderWorker's --workers count (av1an's chunk-parallel knob).
params_fn: Callable[[int, int], str]
ffmpeg_vargs_fn: Callable[[int, int], list[str]] # (crf, preset) -> ffmpeg -c:v args
presets: list[str] # Human-readable preset labels
preset_map: dict[str, int] # label -> internal preset value
# v4.3.0: the codec_name ffprobe returns for files encoded with this
# profile. Used by _output_already_encoded() to detect skip-existing.
# av1 → "av1", vp9 → "vp9", hevc → "hevc". Verified against ffprobe
# output for each encoder; this is the codec_name field in the video
# stream's JSON, NOT the encoder_name (which would be "libsvtav1" etc).
ffprobe_codec_name: str = ""
# v4.6.0: hardware (NVENC) counterpart for this codec family. Empty
# string = no hardware encoder exists for this family (VP9 has no
# NVENC encoder). The GPU path is ffmpeg-only (av1an cannot drive
# NVENC); EncoderWorker.resolve_gpu_encoder() only selects it when a
# functional probe proved the encoder works on this system. The
# ffprobe codec_name is IDENTICAL to the CPU encoder's (hevc_nvenc
# also produces "hevc"), so skip-existing detection works across
# GPU/CPU re-encodes of the same family.
gpu_encoder: str = ""
# (crf, preset) -> ffmpeg args for the NVENC encoder. Mirrors
# ffmpeg_vargs_fn. None when gpu_encoder is empty.
gpu_vargs_fn: Callable[[int, int], list[str]] | None = None
# v4.8.0: GPU-profile support. *gpu_family* is the codec family key
# used by GpuProfile.encoders ("av1"/"hevc"/"vp9"); *gpu_encoders_by_api*
# maps a hardware API (nvenc/vaapi/qsv) to this profile's ffmpeg
# encoder for that API. resolve_gpu_encoder() picks the entry matching
# the selected GPU profile.
gpu_family: str = ""
gpu_encoders_by_api: dict[str, str] = field(default_factory=dict)
# v4.9.0: True for codec families av1an cannot drive at all (Theora —
# no av1an encoder exists). The UI forces the ffmpeg path for these
# regardless of the av1an toggle, greys the combo entry out when
# ffmpeg lacks the library, and resolve_gpu_encoder() stays on CPU
# (empty gpu_family). params_fn must never be called on this path.
ffmpeg_only: bool = False
@dataclass
class AudioProfile:
label: str
params: list[str] # Tokens passed to --audio-params (joined with space)
# v3 (OTC-012, SEI CERT STR09-C): the ffmpeg audio encoder name this
# profile depends on, e.g. "libopus", "libvorbis", "flac", "libiamf".
# Used by _check_combo_compatibility and _disable_unavailable_codecs
# to look up the encoder directly in EnvProbe.ffmpeg_libs — replacing
# the v2 substring match (`"libiamf" in ap.params`) which would
# falsely match a hypothetical `-libiamf-mode` argument.
# Empty string means "no ffmpeg encoder dependency" (rare; only used
# by passthrough profiles that don't transcode audio).
ffmpeg_encoder_name: str = ""
# v4.3.0: the codec_name ffprobe returns for files encoded with this
# profile. Used by _output_already_encoded() to detect skip-existing.
# opus → "opus", vorbis → "vorbis", flac → "flac", iamf → "iamf".
ffprobe_codec_name: str = ""
@dataclass
class ContainerProfile:
label: str
ext: str # e.g. "mkv", "webm"
def _av1_params(crf: int, preset: int) -> str:
"""SVT-AV1 encoder params for av1an's --video-params.
av1an splits the --video-params value by whitespace (``split_whitespace()``)
and passes each resulting token as a separate argument to SvtAv1EncApp.
Therefore the string must contain space-separated ``--flag value`` pairs
that SvtAv1EncApp can parse natively.
Colon-separated ``key=value:key=value`` does NOT work because there are
no whitespace boundaries for av1an to split on — the entire string reaches
SvtAv1EncApp as one opaque argument, producing:
``Maybe missing spacing between tokens``.
Thread capping is NOT injected here. SvtAv1EncApp (the standalone CLI
av1an invokes per-chunk) uses `--lp N` (logical processors), not
`--threads N`. Thread capping is handled via av1an's `--workers` flag
(chunk-parallel count) and via `-threads` in the ffmpeg fallback path
(where libsvtav1 is a library and accepts it).
"""
return f"--preset {preset} --crf {crf} --keyint 240"
def _vp9_params(crf: int, preset: int) -> str:
"""VP9 encoder params for av1an's --video-params.
av1an splits by whitespace, so we use space-separated --flag=value tokens
that vpxenc parses natively.
"""
cpu_used = max(0, 8 - preset)
return f"--end-usage=q --cq-level={crf} --cpu-used={cpu_used}"
def _x265_params(crf: int, preset: int) -> str:
"""x265 encoder params for av1an's --video-params.
av1an splits by whitespace, so we use space-separated --flag value tokens
that x265 parses natively.
"""
return f"--crf {crf} --preset {preset}"
def _svtav1_ffmpeg_args(crf: int, preset: int) -> list[str]:
"""FFmpeg args for SVT-AV1 (maps av1an preset=0..8 → svtav1 -preset 0..13)."""
# av1an preset range 0-8 maps to SVT-AV1 preset range 0-13
# Scale roughly: 8→0, 6→4, 4→7, 2→10
svt_preset = max(0, min(13, round((8 - preset) * 13 / 8)))
return ["-c:v", "libsvtav1", "-preset", str(svt_preset), "-crf", str(crf),
"-pix_fmt", "yuv420p10le", "-g", "240"]
def _vp9_ffmpeg_args(crf: int, preset: int) -> list[str]:
"""FFmpeg args for VP9 (maps av1an cpu-used 0..8 → -cpu-used 0..8)."""
cpu_used = max(0, min(8, preset))
return ["-c:v", "libvpx-vp9", "-crf", str(crf), "-b:v", "0",
"-cpu-used", str(cpu_used), "-pix_fmt", "yuv420p", "-g", "240",
"-row-mt", "1", "-tiles", "2x2"]
def _x265_ffmpeg_args(crf: int, preset: int) -> list[str]:
"""FFmpeg args for x265 (maps av1an preset 5..10 → x265 -preset)."""
# av1an x265 preset range 5-10 maps to x265 preset names
preset_names = {5: "slow", 7: "medium", 9: "fast", 10: "faster"}
p = preset_names.get(preset, "medium")
return ["-c:v", "libx265", "-preset", p, "-crf", str(crf),
"-pix_fmt", "yuv420p10le", "-g", "240"]
# ── v4.6.0: NVENC (hardware) vargs ──
# NVENC quality control: -rc vbr + -cq N + -b:v 0 is the constant-quality
# mode that maps most closely to the CPU encoders' CRF (cq ≈ crf for HEVC
# and AV1 within ~±3). -b:v 0 removes the default bitrate cap so -cq
# actually governs quality. Presets are p1 (fastest) .. p7 (slowest/best)
# on all current NVENC generations; the legacy "slow/medium/fast" aliases
# are deprecated.
#
# Pixel format: 8-bit yuv420p. Pascal-generation cards (GTX 10xx) run
# HEVC Main10 at roughly half throughput, and the archival targets here
# are 8-bit phone/BluRay sources — 8-bit keeps the GPU path at full
# speed. ffmpeg auto-converts 10-bit sources to yuv420p.
def _nvenc_preset(preset: int) -> str:
"""Map the CPU preset tiers (lower value = slower/better) to NVENC
p-presets. CPU preset values across profiles are 0..10 with 0/5 =
slowest quality tiers; NVENC is fast enough that even p7 outruns any
CPU encoder, so the whole range compresses to p3..p7."""
if preset <= 6:
return "p7" # "Slow" tier → best NVENC quality
if preset <= 8:
return "p5" # "Medium" tier
return "p4" # "Fast"/"Faster" tiers
def _hevc_nvenc_args(crf: int, preset: int) -> list[str]:
"""FFmpeg args for hevc_nvenc (x265/HEVC family hardware encoder)."""
return ["-c:v", "hevc_nvenc", "-preset", _nvenc_preset(preset),
"-tune", "hq", "-rc", "vbr", "-cq", str(crf), "-b:v", "0",
"-pix_fmt", "yuv420p", "-g", "240"]
def _h264_nvenc_args(crf: int, preset: int) -> list[str]:
"""FFmpeg args for h264_nvenc (hardware H.264 — compatibility target)."""
return ["-c:v", "h264_nvenc", "-preset", _nvenc_preset(preset),
"-tune", "hq", "-rc", "vbr", "-cq", str(crf), "-b:v", "0",
"-pix_fmt", "yuv420p", "-g", "240"]
def _av1_nvenc_args(crf: int, preset: int) -> list[str]:
"""FFmpeg args for av1_nvenc (AV1 family hardware encoder, RTX 40+)."""
return ["-c:v", "av1_nvenc", "-preset", _nvenc_preset(preset),
"-tune", "hq", "-rc", "vbr", "-cq", str(crf), "-b:v", "0",
"-pix_fmt", "yuv420p", "-g", "240"]
# ── v4.9.0: Theora (OGV) — the inverted quality-scale wrapper ──
#
# libtheora grades quality with -q:v 0..31 where HIGHER is better — the
# exact opposite of the CRF knob shared by every other codec family
# (lower = better). Rather than bolt a parallel quality system onto the
# UI, the worker and the skip-existing flow (all of which speak CRF),
# Theora keeps crf_range=(18, 52) exactly like AV1/VP9, and its encoder
# args are built through a translation wrapper that maps the knob value
# onto the q-scale. Verified empirically against this project's ffmpeg
# (libtheora via `ffmpeg -h encoder=libtheora` + size sweep): bitrate
# rises monotonically with -q:v and saturates from ~31 upward, so the
# map targets 0..31.
#
# Speed: libtheora has no x26x-style preset ladder, only -speed_level
# 0..2 (0 = slowest/best). The PRESET combo maps onto that directly.
# Pixel format: Theora is 4:2:0-only, so yuv420p is forced (10-bit and
# 422/444 sources are converted by ffmpeg automatically).
THEORA_Q_RANGE: tuple[int, int] = (0, 31)
def theora_quality_from_crf(crf: int, crf_lo: int = 18, crf_hi: int = 52) -> int:
"""Translate a shared-CRF-knob position into a libtheora -q:v value.
Pure linear inversion with input clamping: the knob's best position
(crf_lo) maps to the q-scale's best (31), the knob's worst (crf_hi)
to 0. Every 2 knob steps (the knob's snap tick) drops q by ~1.8, so
the default knob value of 26 lands at q 24 — high-quality archival
territory, consistent with the other families' 4.8.2 defaults.
"""
lo, hi = THEORA_Q_RANGE
crf = max(crf_lo, min(crf_hi, crf))
t = (crf - crf_lo) / (crf_hi - crf_lo)
return round(hi - t * (hi - lo))
def _theora_av1an_forbidden(crf: int, preset: int) -> str:
"""params_fn guard — av1an has no Theora encoder.
The UI forces the ffmpeg path for ``ffmpeg_only`` profiles before the
chunk-parallel branch can ever run, so reaching this function means a
regression; raise loudly instead of feeding av1an a ``--encoder ""``
that fails deep in the chunk pipeline with an opaque CLI error.
"""
raise RuntimeError(
"Theora has no av1an encoder — encode via the ffmpeg path "
"(profile is ffmpeg_only; chunk-parallel must be disabled for it)"
)
def _theora_ffmpeg_args(crf: int, preset: int) -> list[str]:
"""FFmpeg args for libtheora (v4.9.0 OGV output target)."""
speed = max(0, min(2, preset)) # -speed_level 0..2, clamp unknown presets
return [
"-c:v", "libtheora",
"-q:v", str(theora_quality_from_crf(crf)),
"-speed_level", str(speed),
"-pix_fmt", "yuv420p",
]
VIDEO_CODECS: list[VideoCodecProfile] = [
VideoCodecProfile(
label="AV1 (SVT-AV1)",
av1an_encoder="svt_av1",
ffmpeg_encoder="libsvtav1",
container="mkv",
crf_range=(18, 52),
# v4.8.2: CRF 32 delivers xvid-tier quality on grainy sources
# (blocky shadows, smeared detail — observed in hybrid GPU+CPU
# runs). 26 is the archival default; every -6 CRF buys roughly
# +50% bitrate.
default_crf=26,
params_fn=_av1_params,
ffmpeg_vargs_fn=_svtav1_ffmpeg_args,
presets=["Slow (8)", "Medium (6)", "Fast (4)", "Faster (2)"],
preset_map={"Slow (8)": 8, "Medium (6)": 6, "Fast (4)": 4, "Faster (2)": 2},
ffprobe_codec_name="av1", # v4.3.0: skip-existing detection
# v4.6.0: av1_nvenc exists only on RTX 40+ (Ada) cards; on Pascal
# (GTX 10xx) the functional probe fails and auto falls back to
# the SVT-AV1 CPU encoder.
gpu_encoder="av1_nvenc",
gpu_vargs_fn=_av1_nvenc_args,
gpu_family="av1",
gpu_encoders_by_api={"nvenc": "av1_nvenc", "qsv": "av1_qsv",
"vaapi": "av1_vaapi"},
),
VideoCodecProfile(
label="VP9",
av1an_encoder="vpx",
ffmpeg_encoder="libvpx-vp9",
container="webm",
crf_range=(18, 52),
# v4.8.2: cq-level is a 0-63 scale for libvpx-vp9; 32 sat mid-scale
# and matched the same "sloppy" reports as AV1's old 32.
default_crf=28,
params_fn=_vp9_params,
ffmpeg_vargs_fn=_vp9_ffmpeg_args,
presets=["Slow (0)", "Medium (2)", "Fast (4)", "Faster (6)"],
preset_map={"Slow (0)": 0, "Medium (2)": 2, "Fast (4)": 4, "Faster (6)": 6},
ffprobe_codec_name="vp9", # v4.3.0: skip-existing detection
# v4.8.0: VP9 has no NVENC encoder; VAAPI (AMD/older Intel) can
# encode it on some cards.
gpu_family="vp9",
gpu_encoders_by_api={"vaapi": "vp9_vaapi"},
),
VideoCodecProfile(
label="x265 (HEVC)",
av1an_encoder="x265",
ffmpeg_encoder="libx265",
container="mkv",
crf_range=(18, 40),
# v4.8.2: 28 is ffmpeg's own default — fine for casual use, soft for
# an archival target. 24 is the high-quality tier (≈ +60% bitrate).
default_crf=24,
params_fn=_x265_params,
ffmpeg_vargs_fn=_x265_ffmpeg_args,
presets=["Slow (5)", "Medium (7)", "Fast (9)", "Faster (10)"],
preset_map={"Slow (5)": 5, "Medium (7)": 7, "Fast (9)": 9, "Faster (10)": 10},
ffprobe_codec_name="hevc", # v4.3.0: skip-existing detection
# v4.6.0: hevc_nvenc works on every NVENC generation since Maxwell
# GM206 (incl. the GTX 1070) — this is the family that benefits
# most from GPU mode.
gpu_encoder="hevc_nvenc",
gpu_vargs_fn=_hevc_nvenc_args,
gpu_family="hevc",
gpu_encoders_by_api={"nvenc": "hevc_nvenc", "qsv": "hevc_qsv",
"vaapi": "hevc_vaapi"},
),
# v4.9.0: Theora — the legacy open-source pairing (Ogg Theora +
# Vorbis), kept alive for compatibility with old players/portals.
# ffmpeg-only: av1an has no Theora encoder, no GPU exists for it, and
# its -q:v quality scale is inverted relative to CRF (see the wrapper
# section above). Canonical container is OGV; MKV can also hold it
# (soft warning), MP4/WebM cannot (hard block — see compat rules).
VideoCodecProfile(
label="Theora (OGV)",
av1an_encoder="",
ffmpeg_encoder="libtheora",
container="ogv",
crf_range=(18, 52),
default_crf=26, # → -q:v 24 via theora_quality_from_crf
params_fn=_theora_av1an_forbidden,
ffmpeg_vargs_fn=_theora_ffmpeg_args,
presets=["Best (0)", "Medium (1)", "Fast (2)"],
preset_map={"Best (0)": 0, "Medium (1)": 1, "Fast (2)": 2},
ffprobe_codec_name="theora", # skip-existing detection
gpu_family="", # no hardware Theora encoder on any API
ffmpeg_only=True,
),
]
AUDIO_PROFILES: list[AudioProfile] = [
AudioProfile(label="Opus (96k)", params=["-c:a", "libopus", "-b:a", "96k"],
ffmpeg_encoder_name="libopus", ffprobe_codec_name="opus"),
AudioProfile(label="Opus (128k)", params=["-c:a", "libopus", "-b:a", "128k"],
ffmpeg_encoder_name="libopus", ffprobe_codec_name="opus"),
AudioProfile(label="Opus (64k)", params=["-c:a", "libopus", "-b:a", "64k"],
ffmpeg_encoder_name="libopus", ffprobe_codec_name="opus"),
AudioProfile(label="Vorbis (128k)", params=["-c:a", "libvorbis", "-b:a", "128k"],
ffmpeg_encoder_name="libvorbis", ffprobe_codec_name="vorbis"),
AudioProfile(label="Vorbis (192k)", params=["-c:a", "libvorbis", "-b:a", "192k"],
ffmpeg_encoder_name="libvorbis", ffprobe_codec_name="vorbis"),
AudioProfile(label="FLAC (lossless)", params=["-c:a", "flac"],
ffmpeg_encoder_name="flac", ffprobe_codec_name="flac"),
# IAMF — AOMedia Immersive Audio Model and Formats (RFC 9454 family).
# Built on Opus internally; requires ffmpeg compiled with --enable-libiamf.
# CANNOT be muxed into MKV/WebM — must use the MP4 container (see below).
# The -strict experimental flag is harmless on ffmpeg builds where libiamf
# is already stable, and required on builds where it's still flagged
# experimental, so we always pass it for forward compatibility.
AudioProfile(
label="IAMF (128k)",
params=["-c:a", "libiamf", "-b:a", "128k", "-strict", "experimental"],
ffmpeg_encoder_name="libiamf",
ffprobe_codec_name="iamf",
),
]
CONTAINER_PROFILES: list[ContainerProfile] = [
ContainerProfile(label="MKV (Matroska)", ext="mkv"),
ContainerProfile(label="WebM", ext="webm"),
# MP4 is required for IAMF audio (MKV/WebM cannot mux the IAMF codec).
# Also useful as a more universally compatible output container.
ContainerProfile(label="MP4", ext="mp4"),
# v4.9.0: OGV — the Ogg Theora container. Video must be Theora (the
# Ogg muxer rejects every other video codec this app offers); audio
# can be Vorbis, Opus or FLAC (all standard in Ogg), but not IAMF.
ContainerProfile(label="OGV (Ogg)", ext="ogv"),
]
# ──────────────────────────────────────────────────────────────────────────────
# FFMPEG_LIB_KEY_MAP — single source of truth (OTC-007, SEI CERT MSC04-C).
#
# Maps the `ffmpeg_encoder` field of a VideoCodecProfile (e.g. "libsvtav1",
# "libvpx-vp9") to the corresponding key in EnvProbe.ffmpeg_libs (which is
# populated by _probe_ffmpeg_libs()).
#
# v2 had this map duplicated in three call sites:
# - _ffmpeg_fallback_encode (around line 2075)
# - _probe_and_init status bar (around line 4309)
# - _handle_vs_incompat fallback check (around line 4532)
# Adding a new codec required updating all three in sync — a classic
# MSC04-C violation. v3 hoists it to one module-level constant.
# ──────────────────────────────────────────────────────────────────────────────
FFMPEG_LIB_KEY_MAP: dict[str, str] = {
"libsvtav1": "libsvtav1",
"libaom-av1": "libaom",
"libvpx-vp9": "libvpx",
"libx265": "libx265",
# v4.6.0: hardware encoders. These keys are populated by
# _probe_ffmpeg_libs() alongside the software encoders, and — unlike
# the compiled-in check — EncoderWorker additionally gates the GPU
# path on env.gpu.functional (a real encode smoke test), because a
# ffmpeg build can list an NVENC encoder that the installed driver
# cannot open (NVENC API version mismatch).
"hevc_nvenc": "hevc_nvenc",
"h264_nvenc": "h264_nvenc",
"av1_nvenc": "av1_nvenc",
# v4.8.0: hardware APIs for AMD (VAAPI) and Intel (QSV) profiles.
"hevc_vaapi": "hevc_vaapi",
"h264_vaapi": "h264_vaapi",
"av1_vaapi": "av1_vaapi",
"vp9_vaapi": "vp9_vaapi",
"hevc_qsv": "hevc_qsv",
"h264_qsv": "h264_qsv",
"av1_qsv": "av1_qsv",
# v4.9.0: Theora (OGV target). Probed alongside the software encoders
# by _probe_ffmpeg_libs(); a missing libtheora greys out the codec.
"libtheora": "libtheora",
}
def ffmpeg_lib_key_for(ffmpeg_encoder: str) -> str:
"""Look up the ffmpeg_libs key for a given ffmpeg encoder name.
Returns the encoder name itself if no mapping is known — this preserves
forward compatibility with encoders added after this map was last
updated (the caller's .get() will then return False, which is the
safe default for an unknown encoder).
"""
return FFMPEG_LIB_KEY_MAP.get(ffmpeg_encoder, ffmpeg_encoder)
# ── Resolution presets ──
# Aspect ratios:
# Standard 16:9 -> w/h = 1.778
# Wide 21:9 -> w/h = 2.333
# Ultrawide 32:9 -> w/h = 3.556
@dataclass
class ResolutionProfile:
label: str # Display label in dropdown, e.g. "1080p Wide (2560x1080)"
category: str # Grouping key: "standard", "wide", "ultrawide", "original"
width: int | None # None for "original" (no scaling)
height: int | None # None for "original"
aspect_label: str # "16:9", "21:9", "32:9", "Source"
RESOLUTION_PRESETS: list[ResolutionProfile] = [
# ── Original (no scaling) ──
ResolutionProfile("Original (No Scaling)", "original", None, None, "Source"),
# ── Standard 16:9 ──
ResolutionProfile("480p ( 854x 480)", "standard", 854, 480, "16:9"),
ResolutionProfile("720p (1280x 720)", "standard", 1280, 720, "16:9"),
ResolutionProfile("1080p (1920x1080)", "standard", 1920, 1080, "16:9"),
ResolutionProfile("2K (2560x1440)", "standard", 2560, 1440, "16:9"),
ResolutionProfile("4K (3840x2160)", "standard", 3840, 2160, "16:9"),
# ── Wide 21:9 ──
ResolutionProfile("480p Wide ( 854x 366)", "wide", 854, 366, "21:9"),
ResolutionProfile("720p Wide (1280x 549)", "wide", 1280, 549, "21:9"),
ResolutionProfile("1080p Wide (2560x1080)", "wide", 2560, 1080, "21:9"),
ResolutionProfile("2K Wide (3440x1440)", "wide", 3440, 1440, "21:9"),
ResolutionProfile("4K Wide (5120x2160)", "wide", 5120, 2160, "21:9"),
# ── Ultrawide 32:9 ──
ResolutionProfile("480p UW (1706x 480)", "ultrawide", 1706, 480, "32:9"),
ResolutionProfile("1080p UW (3840x1080)", "ultrawide", 3840, 1080, "32:9"),
ResolutionProfile("2K UW (5120x1440)", "ultrawide", 5120, 1440, "32:9"),
ResolutionProfile("4K UW (7680x2160)", "ultrawide", 7680, 2160, "32:9"),
]
SUBTITLE_OPTIONS = [
("None", None),
("English", "eng"),
]
# v4.8.2: added .vob (DVD rips, MPEG-PS), .xvid (AVI/ASP rips) and .ogv
# (Ogg/Theora archives) — legacy sources exactly what this tool exists to
# re-encode; leaving them out of the filter meant they were silently skipped.
DEFAULT_INPUT_EXTENSIONS = {".mp4", ".mkv", ".avi", ".mov", ".ts", ".m4v", ".flv", ".wmv", ".webm", ".mpg", ".mpeg", ".vob", ".xvid", ".ogv"}
# ════════════════════════════════════════════════════════════════════════════
# ═══ cpu_topology ═══
# ════════════════════════════════════════════════════════════════════════════
"""CPU topology detection (physical cores, not hyperthreads).
Reads /sys/devices/system/cpu/* and falls back to ``lscpu``. Pure
stdlib; no internal package dependencies.
"""
# ──────────────────────────────────────────────
# CPU TOPOLOGY (physical cores, not hyperthreads)
# ──────────────────────────────────────────────
@dataclass
class CpuTopology:
physical_cores: int
logical_threads: int
threads_per_core: int
model_name: str
def _read_sysfs_cores() -> (tuple[int, int]) | None:
"""
Read /sys/devices/system/cpu/cpu*/topology/ to count unique
(physical_package_id, core_id) pairs — i.e. physical cores.
Returns (physical_cores, logical_threads) or None.
"""
cpu_base = Path("/sys/devices/system/cpu")
if not cpu_base.exists():
return None
unique_cores: set[tuple[str, str]] = set()
logical = 0
for cpu_dir in sorted(cpu_base.glob("cpu[0-9]*")):
core_id_file = cpu_dir / "topology" / "core_id"
pkg_id_file = cpu_dir / "topology" / "physical_package_id"
if core_id_file.exists() and pkg_id_file.exists():
try:
pkg = pkg_id_file.read_text().strip()
core = core_id_file.read_text().strip()
unique_cores.add((pkg, core))
logical += 1
except (OSError, ValueError):
# OSError: file vanished/permission; ValueError: UnicodeDecodeError
pass
if unique_cores and logical:
return (len(unique_cores), logical)
return None
def _read_lscpu_cores() -> (tuple[int, int]) | None:
"""Fallback: parse lscpu -p=CORE,SOCKET for unique physical cores."""
if not shutil.which("lscpu"):
return None
try:
res = subprocess.run(
["lscpu", "-p=CORE,SOCKET"],
capture_output=True, text=True, timeout=5,
)
lines = [l.strip() for l in res.stdout.strip().splitlines() if l.strip() and not l.startswith("#")]
if lines:
unique = set(lines)
return (len(unique), len(lines))
except (OSError, subprocess.SubprocessError):
pass
return None
def detect_cpu_topology() -> CpuTopology:
"""
Detect physical CPU topology. Prefers /sys filesystem, falls back
to lscpu, then estimates from os.cpu_count().
"""
logical = os.cpu_count() or 1
physical = logical
# Try /sys first (most reliable)
result = _read_sysfs_cores()
if result:
physical, logical = result
else:
# Try lscpu
result = _read_lscpu_cores()
if result:
physical, logical = result
else:
# Estimate: assume 2 threads/core if cpu_count > 2 and is even
if logical > 2 and logical % 2 == 0:
physical = logical // 2
tpc = logical // physical if physical > 0 else 1
# Try to get CPU model name
model = "Unknown CPU"
model_file = Path("/proc/cpuinfo")
if model_file.exists():
for line in model_file.read_text(errors="replace").splitlines():
if line.startswith("model name"):
model = line.split(":", 1)[1].strip()
break
else:
# Non-x86 / non-Linux: try lscpu
if shutil.which("lscpu"):
try:
res = subprocess.run(["lscpu"], capture_output=True, text=True, timeout=5)
for line in res.stdout.splitlines():
if "Model name" in line:
model = line.split(":", 1)[1].strip()
break
except (OSError, subprocess.SubprocessError):
pass
return CpuTopology(
physical_cores=physical,
logical_threads=logical,
threads_per_core=tpc,
model_name=model,
)
# ════════════════════════════════════════════════════════════════════════════
# ═══ distro_probe ═══
# ════════════════════════════════════════════════════════════════════════════
"""Linux distro detection and per-distro profile registry.
Replaces the v1 250-line if/elif chain with a tuple-of-dataclasses
table (``DISTRO_REGISTRY``). Adding a new distro is a one-row change.
Pure stdlib; no internal package dependencies.
"""
# ──────────────────────────────────────────────
# DISTRO DETECTION & PROFILES
# ──────────────────────────────────────────────
@dataclass
class DistroProfile:
family: str # Canonical family: arch, debian, redhat, suse, nixos, unknown
name: str # Pretty name: "Arch Linux", "Fedora 40", etc.
version_id: str # e.g. "40", "15.6", "24.05"
pkg_manager: str # e.g. "pacman", "dnf", "zypper", "apt", "nix"
install_cmd_template: str # e.g. "sudo pacman -S {packages}"
binary_extra_paths: list[str] # Distro-specific dirs to search for binaries
av1an_known_encoder_names: list[str] # Names this distro's av1an build may accept
ffmpeg_pkg: str # Package name providing ffmpeg
av1an_pkg: str # Package name providing av1an
notes: str # Distro-specific quirks worth showing the user
# Runtime dependency packages (key = generic name, value = distro package name)
dep_pkgs: dict[str, str] = field(default_factory=dict)
# Binaries that av1an invokes directly (not via ffmpeg)
encoder_binaries: dict[str, list[str]] = field(default_factory=dict)
# VSScript package name — on most distros this is bundled into 'vapoursynth',
# but Debian/Ubuntu split it into a separate -script-dev package.
# If set, this takes priority over dep_pkgs["vapoursynth"] for the VS check.
vsscript_pkg: str = ""
def _read_os_release() -> dict[str, str]:
"""Parse /etc/os-release into a dict. Falls back to empty dict."""
os_release = Path("/etc/os-release")
fallback = Path("/usr/lib/os-release")
target = os_release if os_release.exists() else fallback
if not target.exists():
return {}
data = {}
for line in target.read_text(encoding="utf-8", errors="replace").splitlines():
line = line.strip()
if "=" in line and not line.startswith("#"):
key, _, val = line.partition("=")
data[key.strip()] = val.strip().strip('"')
return data
# ──────────────────────────────────────────────────────────────────────────────
# DISTRO_REGISTRY — data-driven distro detection (v3, OTC-014).
#
# v1/v2 had a 250-line if/elif chain in detect_distro() with one branch per
# distro family. Each branch constructed a DistroProfile with mostly-identical
# fields — a classic SEI CERT MSC04-C violation (no single source of truth).
#
# v3 collapses the chain into a tuple-of-dicts table. Each entry has:
# ids: tuple of distro_id strings that match this family
# id_likes: tuple of ID_LIKE substrings that also match this family
# family: canonical family name
# pkg_manager: package manager binary name
# install_cmd: template with {packages} placeholder
# extra_paths: list of distro-specific binary search paths
# dep_pkgs: map of generic name -> distro package name
# notes: distro-specific quirks string
# vsscript_pkg: (optional) separate VSScript package name
#
# Adding a new distro is now a single-table-row change — no code modification.
# The encoder_binaries field is identical across all distros and lives in the
# function body (it's the same dict literal every time).
# ──────────────────────────────────────────────────────────────────────────────
# encoder_binaries is identical for every distro — define once.
_ENCODER_BINARIES: dict[str, list[str]] = {
"svt_av1": ["SvtAv1EncApp", "svt_av1"],
"vpx": ["vpxenc"],
"x265": ["x265"],
}
# Common av1an encoder names known across distros.
_AV1AN_KNOWN_ENCODERS: list[str] = ["svt_av1", "svt", "aom", "rav1e", "vpx", "x265"]
@dataclass(frozen=True)
class _DistroEntry:
"""One row in the DISTRO_REGISTRY table."""
ids: tuple[str, ...] # exact distro_id matches
id_likes: tuple[str, ...] # ID_LIKE substring matches
family: str
pkg_manager: str
install_cmd: str # template with {packages}
extra_paths: tuple[str, ...]
dep_pkgs: dict[str, str]
notes: str
vsscript_pkg: str = ""
DISTRO_REGISTRY: tuple[_DistroEntry, ...] = (
_DistroEntry(
ids=("arch", "manjaro", "endeavouros", "garuda", "cachyos"),
id_likes=("arch",),
family="arch",
pkg_manager="pacman",
install_cmd="sudo pacman -S {packages}",
extra_paths=("/usr/bin", "/usr/local/bin", "~/.local/bin", "~/.cargo/bin"),
dep_pkgs={
"vapoursynth": "vapoursynth",
"svt-av1": "svt-av1",
"x265": "x265",
"vpx": "libvpx",
"opus": "libopus",
"vorbis": "libvorbis",
"flac": "flac",
},
notes=(
"Arch/Manjaro: av1an is in the AUR (yay -S av1an) or community repo. "
"SVT-AV1 encoder name is typically 'svt_av1'. "
"Cargo-installed av1an may live in ~/.cargo/bin."
),
),
_DistroEntry(
ids=("fedora",),
id_likes=("fedora",),
family="redhat",
pkg_manager="dnf",
install_cmd="sudo dnf install {packages}",
extra_paths=("/usr/bin", "/usr/local/bin", "~/.cargo/bin"),
dep_pkgs={
"vapoursynth": "vapoursynth",
"svt-av1": "svt-av1",
"x265": "x265",
"vpx": "libvpx-tools",
"opus": "opus",
"vorbis": "libvorbis",
"flac": "flac",
},
notes=(
"Fedora: av1an may require COPR enablement first: "
"sudo dnf copr enable sergiomb/av1an (or build from source). "
"SVT-AV1 is in the main repos as 'svt-av1'. "
"Ensure RPM Fusion is enabled for full codec support."
),
),
_DistroEntry(
ids=("rhel", "centos", "rocky", "almalinux", "ol"),
id_likes=("rhel", "centos"),
family="redhat",
# RHEL-family: dnf if present, fall back to yum
pkg_manager="", # resolved at runtime in detect_distro()
install_cmd="", # resolved at runtime in detect_distro()
extra_paths=("/usr/bin", "/usr/local/bin", "~/.cargo/bin"),
dep_pkgs={
"vapoursynth": "vapoursynth",
"svt-av1": "svt-av1",
"x265": "x265",
"vpx": "libvpx-tools",
"opus": "opus",
"vorbis": "libvorbis",
"flac": "flac",
},
notes=(
"RHEL/CentOS/Rocky/Alma: av1an is NOT in default repos. "
"Options: (1) cargo install av1an, (2) build from GitHub source, "
"(3) use pre-built binary from releases. "
"Enable EPEL + RPM Fusion for FFmpeg codec support."
),
),
_DistroEntry(
ids=("opensuse-leap", "opensuse-tumbleweed", "sles"),
id_likes=("suse",),
family="suse",
pkg_manager="zypper",
install_cmd="sudo zypper install {packages}",
extra_paths=("/usr/bin", "/usr/local/bin", "~/.cargo/bin"),
dep_pkgs={
"vapoursynth": "vapoursynth",
"svt-av1": "svt-av1",
"x265": "x265",
"vpx": "libvpx",
"opus": "libopus",
"vorbis": "libvorbis",
"flac": "flac",
},
notes=(
"openSUSE: av1an may be available via OBS (Open Build Service). "
"Check: https://build.opensuse.org/package/show/multimedia:apps/av1an. "
"Packman repo provides FFmpeg with full codec support."
),
),
_DistroEntry(
ids=("nixos",),
id_likes=("nixos",),
family="nixos",
pkg_manager="nix",
install_cmd="nix-shell -p {packages}",
extra_paths=("/run/current-system/sw/bin", "~/.nix-profile/bin"),
dep_pkgs={
"vapoursynth": "vapoursynth",
"svt-av1": "svt-av1",
"x265": "x265",
"vpx": "libvpx",
"opus": "opus",
"vorbis": "libvorbis",
"flac": "flac",
},
notes=(
"NixOS: Use 'nix-shell -p ffmpeg av1an' or add to configuration.nix. "
"Binaries live under /run/current-system/sw/bin or ~/.nix-profile/bin. "
"av1an CLI flags may differ from other distros depending on the nixpkgs channel."
),
),
_DistroEntry(
ids=("debian", "ubuntu", "linuxmint", "pop"),
id_likes=("debian",),
family="debian",
pkg_manager="apt",
install_cmd="sudo apt install {packages}",
extra_paths=("/usr/bin", "/usr/local/bin", "~/.cargo/bin"),
dep_pkgs={
"vapoursynth": "vapoursynth",
"svt-av1": "svtav1",
"x265": "x265",
"vpx": "libvpx-tools",
"opus": "libopus-dev",
"vorbis": "libvorbis-dev",
"flac": "flac",
},
notes=(
"Debian/Ubuntu: av1an is in the repos (apt install av1an). "
"Debian repo builds may use 'svt' as encoder name instead of 'svt_av1'. "
"VSScript is in a separate package: libvapoursynth-script-dev. "
"For newer builds, consider cargo install av1an."
),