authorgravatar for git@paperclover.netclover caruso <git@paperclover.net> 2026-06-13 15:02:40-07:00
committergravatar for git@paperclover.netclover caruso <git@paperclover.net> 2026-09-12 13:37:22-07:00
log4166956deaeadc0fb5443e3475b4da5fd97b0a74
treeb58e908121dbc3cf5d3d74e32f7b96e15fc2577b
parentc6666b14acb9d62bb70b34275156feb553e400f8
signature Signed by SSH key SHA256:xbd+BjjhyBfwk7GVoURf9Yx0gzDerHbvYv7SddNWmAs

feat: yt dlp changes


21 files changed, 1776 insertions(+), 135 deletions(-)

.gitignore+1
...@@ -2,3 +2,4 @@...@@ -2,3 +2,4 @@
2/.apps2/.apps
3/certs3/certs
4/sitegen4/sitegen
5config/yt-upscaler/realesr-general-x4v3.onnx
compose.yaml+65-15
...@@ -117,6 +117,7 @@ services:...@@ -117,6 +117,7 @@ services:
117 - "${STORE_ROOT}:/w"117 - "${STORE_ROOT}:/w"
118 - "${APP_ROOT}/copyparty:/cfg"118 - "${APP_ROOT}/copyparty:/cfg"
119 - ./config/copyparty.conf:/cfg/copyparty.conf:ro119 - ./config/copyparty.conf:/cfg/copyparty.conf:ro
120 - ./config/copyparty-hooks:/hooks:ro
120 healthcheck:121 healthcheck:
121 test: ["CMD-SHELL", "wget -qO /dev/null http://127.0.0.1:80/?h"]122 test: ["CMD-SHELL", "wget -qO /dev/null http://127.0.0.1:80/?h"]
122 interval: 30s123 interval: 30s
...@@ -152,7 +153,7 @@ services:...@@ -152,7 +153,7 @@ services:
152 cpus: "16.0"153 cpus: "16.0"
153 restart: unless-stopped154 restart: unless-stopped
154 volumes:155 volumes:
155 - "$CLOVER_ROOT/Documents/Config/paperclover:/data:rw"156 - "$CLOVER_ROOT/Documents/Config/paper clover:/data:rw"
156 # rw: the indexer scrubs exif location data in place157 # rw: the indexer scrubs exif location data in place
157 - "$CLOVER_ROOT/Published:/published:rw"158 - "$CLOVER_ROOT/Published:/published:rw"
158 healthcheck:159 healthcheck:
...@@ -246,7 +247,7 @@ services:...@@ -246,7 +247,7 @@ services:
246 context: https://tangled.org/tangled.org/3lqs6zdi4nt22.git247 context: https://tangled.org/tangled.org/3lqs6zdi4nt22.git
247 dockerfile: Dockerfile248 dockerfile: Dockerfile
248 args:249 args:
249 TAG: "${KNOT_IMAGE_TAG:-v1.14.0-alpha}"250 TAG: "${KNOT_IMAGE_TAG:-v1.15.0-alpha}"
250 UID: "${USER_ID:?}"251 UID: "${USER_ID:?}"
251 GID: "${GROUP_ID:?}"252 GID: "${GROUP_ID:?}"
252 pull_policy: build253 pull_policy: build
...@@ -285,7 +286,7 @@ services:...@@ -285,7 +286,7 @@ services:
285 context: ./config/spindle286 context: ./config/spindle
286 dockerfile: Dockerfile287 dockerfile: Dockerfile
287 args:288 args:
288 TAG: "${SPINDLE_IMAGE_TAG:-v1.14.0-alpha}"289 TAG: "${SPINDLE_IMAGE_TAG:-v1.15.0-alpha}"
289 pull_policy: build290 pull_policy: build
290 depends_on:291 depends_on:
291 docker-in-docker:292 docker-in-docker:
...@@ -401,7 +402,7 @@ services:...@@ -401,7 +402,7 @@ services:
401 - ENABLE_SOCKS=no402 - ENABLE_SOCKS=no
402 - SOCKS_USER=admin403 - SOCKS_USER=admin
403 - SOCKS_PASS=socks404 - SOCKS_PASS=socks
404 - LAN_NETWORK=192.168.86.0/24405 - LAN_NETWORK=10.0.0.0/24
405 - NAME_SERVERS=84.200.69.80,37.235.1.174,1.1.1.1,37.235.1.177,84.200.70.40,1.0.0.1406 - NAME_SERVERS=84.200.69.80,37.235.1.174,1.1.1.1,37.235.1.177,84.200.70.40,1.0.0.1
406 - VPN_INPUT_PORTS=1234407 - VPN_INPUT_PORTS=1234
407 - VPN_OUTPUT_PORTS=5678408 - VPN_OUTPUT_PORTS=5678
...@@ -515,6 +516,9 @@ services:...@@ -515,6 +516,9 @@ services:
515 HOME: /config516 HOME: /config
516 volumes:517 volumes:
517 - ./config/yt:/config-yt:ro518 - ./config/yt:/config-yt:ro
519 # channel list + ytdl-sub config live outside the repo so they can be
520 # edited from the yt-feed web ui (Clover ▸ Documents/Config/Youtube Downloader)
521 - "${CLOVER_ROOT}/Documents/Config/Youtube Downloader:/yt-config:ro"
518 # the image hardcodes its lock file at /config, so state lives there522 # the image hardcodes its lock file at /config, so state lives there
519 - "${APP_ROOT}/ytdl-sub:/config"523 - "${APP_ROOT}/ytdl-sub:/config"
520 - "${MEDIA_ROOT}:/media"524 - "${MEDIA_ROOT}:/media"
...@@ -540,8 +544,13 @@ services:...@@ -540,8 +544,13 @@ services:
540 MAIL_FROM: "yt-feed@${HOME_DOMAIN:?}"544 MAIL_FROM: "yt-feed@${HOME_DOMAIN:?}"
541 MAIL_TO: "${ADMIN_EMAIL:?}"545 MAIL_TO: "${ADMIN_EMAIL:?}"
542 BASE_URL: "https://yt.${HOME_DOMAIN:?}"546 BASE_URL: "https://yt.${HOME_DOMAIN:?}"
547 # channel lists are edited here and read live; the editor writes to this dir
548 FEED_CONFIG: "/yt-config/feed.yaml"
549 YT_CONFIG_DIR: "/yt-config"
543 volumes:550 volumes:
544 - ./config/yt:/config-yt:ro551 - ./config/yt:/config-yt:ro
552 # editable channel lists (subscriptions.yaml, feed.yaml) — rw for the web editor
553 - "${CLOVER_ROOT}/Documents/Config/Youtube Downloader:/yt-config:rw"
545 - "${APP_ROOT}/yt-feed:/data"554 - "${APP_ROOT}/yt-feed:/data"
546 - "${MEDIA_ROOT}:/media"555 - "${MEDIA_ROOT}:/media"
547 restart: unless-stopped556 restart: unless-stopped
...@@ -550,6 +559,32 @@ services:...@@ -550,6 +559,32 @@ services:
550 net.paperclover.list.domain: yt559 net.paperclover.list.domain: yt
551 net.paperclover.list.priority: 53560 net.paperclover.list.priority: 53
552 net.paperclover.list.access: media-manage561 net.paperclover.list.access: media-manage
562 # thumbnail super-resolution (issue #6): upscales Independent thumbnails 4x
563 # with Real-ESRGAN (ONNX, cpu) so they stay sharp as jellyfin tv backdrops.
564 # separate from yt-feed because onnxruntime has no wheels for the ytdl-sub
565 # image's python 3.14; writes upscale-status.json into yt-feed's data dir.
566 yt-upscaler:
567 container_name: yt-upscaler
568 build:
569 context: config/yt-upscaler
570 dockerfile: Dockerfile
571 pull_policy: build
572 user: "$USER_ID:$GROUP_ID"
573 environment:
574 INDEP_DIR: "/media/jellyfin/Independent"
575 STATE_DIR: "/state"
576 SR_THREADS: "6"
577 volumes:
578 - "${MEDIA_ROOT}:/media"
579 - "${APP_ROOT}/yt-feed:/state"
580 deploy:
581 resources:
582 limits:
583 cpus: "8.0"
584 restart: unless-stopped
585 labels:
586 net.paperclover.list.name: YouTube Thumbnail Upscaler
587 net.paperclover.list.web: "false"
553 # language models588 # language models
554 opencode: # port 4096589 opencode: # port 4096
555 container_name: opencode590 container_name: opencode
...@@ -684,19 +719,24 @@ services:...@@ -684,19 +719,24 @@ services:
684 dawarich-app: # port 30161719 dawarich-app: # port 30161
685 image: freikin/dawarich:latest720 image: freikin/dawarich:latest
686 container_name: dawarich-app721 container_name: dawarich-app
722 networks:
723 default:
724 aliases:
725 - dawarich
687 volumes:726 volumes:
727 - "${APP_ROOT}/dawarich_tmp:/var/app/tmp"
688 - "${APP_ROOT}/dawarich_public:/var/app/public"728 - "${APP_ROOT}/dawarich_public:/var/app/public"
689 - "${APP_ROOT}/dawarich_watched:/var/app/tmp/imports/watched"729 - "${APP_ROOT}/dawarich_watched:/var/app/tmp/imports/watched"
690 - "${APP_ROOT}/dawarich_storage:/var/app/storage"730 - "${APP_ROOT}/dawarich_storage:/var/app/storage"
691 - "${APP_ROOT}/dawarich_data:/dawarich_db_data"731 - "${APP_ROOT}/dawarich_data:/dawarich_db_data"
692 entrypoint: web-entrypoint.sh732 - "${APP_ROOT}/keycloak:/shared-keys:ro"
733 # wrap the stock web-entrypoint to inject the OIDC client secret that
734 # keycloak's init.py wrote to /shared-keys/dawarich (same pattern as forgejo)
735 entrypoint: ["/bin/sh", "-c"]
693 command:736 command:
694 - bin/rails737 - |
695 - server738 export OIDC_CLIENT_SECRET="$$(cat /shared-keys/dawarich)"
696 - -p739 exec web-entrypoint.sh bin/rails server -p 30161 -b ::
697 - "30161"
698 - -b
699 - "::"
700 restart: on-failure740 restart: on-failure
701 user: "$USER_ID:$GROUP_ID"741 user: "$USER_ID:$GROUP_ID"
702 environment:742 environment:
...@@ -709,6 +749,9 @@ services:...@@ -709,6 +749,9 @@ services:
709 DATABASE_NAME: dawarich749 DATABASE_NAME: dawarich
710 MIN_MINUTES_SPENT_IN_CITY: 60750 MIN_MINUTES_SPENT_IN_CITY: 60
711 APPLICATION_HOSTS: "localhost,zenith,dawarich.${HOME_DOMAIN}"751 APPLICATION_HOSTS: "localhost,zenith,dawarich.${HOME_DOMAIN}"
752 OIDC_CLIENT_ID: dawarich
753 OIDC_ISSUER: "https://auth.${HOME_DOMAIN}/realms/master"
754 OIDC_REDIRECT_URI: "https://dawarich.${HOME_DOMAIN}/users/auth/openid_connect/callback"
712 TIME_ZONE: America/Los_Angeles755 TIME_ZONE: America/Los_Angeles
713 APPLICATION_PROTOCOL: http756 APPLICATION_PROTOCOL: http
714 PROMETHEUS_EXPORTER_ENABLED: "false"757 PROMETHEUS_EXPORTER_ENABLED: "false"
...@@ -898,19 +941,26 @@ services:...@@ -898,19 +941,26 @@ services:
898 labels:941 labels:
899 net.paperclover.list.name: DDNS942 net.paperclover.list.name: DDNS
900 net.paperclover.list.access: personal943 net.paperclover.list.access: personal
901 shale: # port probably is 80944 shale: # port 8000
902 image: astheno/shale945 image: astheno/shale@sha256:1f2144eb7a422871414fdc010f05cf99206f621cd68b71435e50da71ed6dbcde
903 container_name: shale946 container_name: shale
904 user: "0:0"947 user: "0:0"
905 volumes:948 volumes:
906 - "${APP_ROOT}/shale/repositories_mirrors:/repositories_mirrors"949 - "${APP_ROOT}/shale/repositories_mirrors:/repositories_mirrors"
907 - "${APP_ROOT}/shale/repositories_owned:/repositories_owned"950 - "${APP_ROOT}/shale/repositories_owned:/repositories_owned"
908 - "${APP_ROOT}/shale/data:/data"951 - "${APP_ROOT}/shale/data:/data"
952 restart: unless-stopped
909 environment:953 environment:
910 DOMAIN: "shale.${HOME_DOMAIN:?}"954 DOMAIN: "shale.${HOME_DOMAIN:?}"
911 OAUTH2_CLIENT: "snow sign on,https://auth.${HOME_DOMAIN:?}/realms/master|shale|${SHALE_CLIENT_SECRET:?}"955 OAUTH2_CLIENT: "oidc,auth.${HOME_DOMAIN:?}/realms/master|shale|${SHALE_CLIENT_SECRET:?}"
956 SESSION_SECRET: "${SHALE_SESSION_SECRET:?}"
912 SERVER_TITLE: "clover's git"957 SERVER_TITLE: "clover's git"
913 MIRROR_1: "sitegen,https://git.paperclover.net/clo/sitegen.git,website generator, standard library, and home of paperclover.net"958 MIRROR_1: "home-infra,https://git.paperclover.net/clo/home-infra.git,compose files and tasks for my home server"
959 MIRROR_2: "discord-name-painter,https://git.paperclover.net/clo/discord-name-painter.git,maintainance only. allows any user to set their display name color to any color, by using dynamically created roles."
960 MIRROR_3: "react-mutation,https://git.paperclover.net/clo/react-mutation.git,create async mutations with trivial optimistic updates and great error handling"
961 MIRROR_4: "markodown,https://git.paperclover.net/clo/markodown.git,alternate universe where markdown lets you write marko components inline"
962 MIRROR_5: "toolkit,https://git.paperclover.net/clo/toolkit.git,Clover's Creative Toolkit is collection of macOS software that I use to create."
963 MIRROR_6: "react-markdown,https://git.paperclover.net/clo/react-markdown.git,memoized markdown renderer for react using the unified plugin ecosystem"
914 # evil inc temporary infrastructure964 # evil inc temporary infrastructure
915 evil-forgejo: # port 3000965 evil-forgejo: # port 3000
916 container_name: evil-forgejo966 container_name: evil-forgejo
config/Caddyfile+38-1
...@@ -41,6 +41,30 @@...@@ -41,6 +41,30 @@
41}41}
4242
43# services are sorted alphabetically43# services are sorted alphabetically
44ai.{$HOME_DOMAIN} {
45 # inference api (qwen3.8-27b on the 3090). vllm serves both the openai
46 # routes and anthropic's /v1/messages, so this host answers both protocols.
47 #
48 # auth is vllm's own bearer token, NOT reverse_proxy_auth: that snippet does
49 # an oauth2 browser redirect, which every api client would choke on.
50 #
51 # vllm only accepts `Authorization: Bearer <key>`. anthropic-protocol
52 # clients (claude code) send `x-api-key: <key>` instead and get a 401.
53 # translate it so both conventions work against the same key.
54 @anthropic_auth header X-Api-Key *
55 handle @anthropic_auth {
56 reverse_proxy "http://vllm:8000" {
57 header_up Authorization "Bearer {http.request.header.X-Api-Key}"
58 # stream tokens as they are generated rather than buffering
59 flush_interval -1
60 }
61 }
62 handle {
63 reverse_proxy "http://vllm:8000" {
64 flush_interval -1
65 }
66 }
67}
44auth.{$HOME_DOMAIN} {68auth.{$HOME_DOMAIN} {
45 handle / {69 handle / {
46 redir / /apps70 redir / /apps
...@@ -96,6 +120,8 @@ file.{$HOME_DOMAIN} {...@@ -96,6 +120,8 @@ file.{$HOME_DOMAIN} {
96 header_up X-Forwarded-Uri {uri}120 header_up X-Forwarded-Uri {uri}
97 }121 }
98 }122 }
123 @share_root path /shr /shr/
124 respond @share_root 403
99 @should_auth not path /shr/*125 @should_auth not path /shr/*
100 reverse_proxy @should_auth "http://forward-auth" {126 reverse_proxy @should_auth "http://forward-auth" {
101 method GET127 method GET
...@@ -143,7 +169,7 @@ knot.{$HOME_DOMAIN} {...@@ -143,7 +169,7 @@ knot.{$HOME_DOMAIN} {
143 reverse_proxy "http://knot:5555"169 reverse_proxy "http://knot:5555"
144}170}
145shale.{$HOME_DOMAIN} {171shale.{$HOME_DOMAIN} {
146 reverse_proxy "http://shale"172 reverse_proxy "http://shale:8000"
147}173}
148spindle.{$HOME_DOMAIN} {174spindle.{$HOME_DOMAIN} {
149 reverse_proxy "http://spindle:6555"175 reverse_proxy "http://spindle:6555"
...@@ -195,6 +221,17 @@ music.{$HOME_DOMAIN} {...@@ -195,6 +221,17 @@ music.{$HOME_DOMAIN} {
195 reverse_proxy "http://navidrome"221 reverse_proxy "http://navidrome"
196 }222 }
197} 223}
224dav.{$HOME_DOMAIN} {
225 # plain webdav endpoint for clients that can't do the sso redirect dance
226 # (onenote 2007, rclone, finder) -- copyparty does its own auth here instead.
227 # strip the idp headers so a client on this vhost can't claim to be an sso user.
228 reverse_proxy "http://copyparty" {
229 header_up -User-Id
230 header_up -User-Groups
231 header_up -User-Email
232 header_up -User-Name
233 }
234}
198opencode.{$HOME_DOMAIN} {235opencode.{$HOME_DOMAIN} {
199 import reverse_proxy_auth "http://opencode:4096" admin236 import reverse_proxy_auth "http://opencode:4096" admin
200}237}
config/copyparty-hooks/reject-apple-cruft.py created+43
...@@ -0,0 +1,43 @@
1#!/usr/bin/env python3
2
3import os
4import sys
5
6_ = r"""
7reject the metadata files macos scatters over network shares
8
9the samba setup vetoes these server-side; webdav has no equivalent knob, and
10DSDontWriteNetworkStores only suppresses .DS_Store -- the ._ AppleDouble files
11(resource forks / xattrs) keep coming regardless. finder also retries a failed
12write forever, so left alone these generate thousands of PUTs per file.
13
14enabled globally in copyparty.conf as:
15 xbu: c,/hooks/reject-apple-cruft.py
16
17 xbu = execute before upload
18 c = check result, reject upload if error
19"""
20
21BAD_EXACT = {
22 ".DS_Store",
23 ".localized",
24 ".Spotlight-V100",
25 ".TemporaryItems",
26 ".Trashes",
27 ".fseventsd",
28 ".apdisk",
29 ".metadata_never_index",
30 ".metadata_never_index_unless_rootfs",
31 ".metadata_direct_scope_only",
32 ".hidden",
33}
34
35
36def main():
37 name = os.path.basename(sys.argv[1])
38 bad = name.startswith("._") or name in BAD_EXACT
39 sys.exit(1 if bad else 0)
40
41
42if __name__ == "__main__":
43 main()
config/copyparty.conf+3
...@@ -9,6 +9,9 @@...@@ -9,6 +9,9 @@
9 theme: 2 # monokai9 theme: 2 # monokai
10 name: clover's nas10 name: clover's nas
11 stats, nos-dup # enable the prometheus endpoint, but disable the dupes counter (too slow)11 stats, nos-dup # enable the prometheus endpoint, but disable the dupes counter (too slow)
12 dav-auth # webdav clients must always authenticate (windows gets confused otherwise)
13 ah-alg: argon2 # passwords in accounts.conf are argon2 hashes, never plaintext
14 xbu: c,/hooks/reject-apple-cruft.py # veto macos ._ / .DS_Store cruft (samba-style)
1215
13 # keycloak16 # keycloak
14 xff-src: lan # accept X-Forwarded-For from `lan`17 xff-src: lan # accept X-Forwarded-For from `lan`
config/keycloak/init.py+33
...@@ -194,6 +194,39 @@ def configure():...@@ -194,6 +194,39 @@ def configure():
194 client_data = kc_admin.get_client(client_id)194 client_data = kc_admin.get_client(client_id)
195 _ = Path("/shared/jellyfin").write_text(client_data["secret"])195 _ = Path("/shared/jellyfin").write_text(client_data["secret"])
196196
197 # dawarich
198 client_id = kc_admin.create_client(
199 {
200 "protocol": "openid-connect",
201 "clientId": "dawarich",
202 "name": "Dawarich",
203 "description": "",
204 "publicClient": False,
205 "authorizationServicesEnabled": False,
206 "serviceAccountsEnabled": False,
207 "implicitFlowEnabled": False,
208 "directAccessGrantsEnabled": False,
209 "standardFlowEnabled": True,
210 "frontchannelLogout": True,
211 "attributes": {
212 "saml_idp_initiated_sso_url_name": "",
213 "standard.token.exchange.enabled": False,
214 "oauth2.device.authorization.grant.enabled": False,
215 "oidc.ciba.grant.enabled": False,
216 "pkce.code.challenge.method": "",
217 "dpop.bound.access.tokens": "false",
218 "post.logout.redirect.uris": "*",
219 },
220 "alwaysDisplayInConsole": False,
221 "rootUrl": "",
222 "baseUrl": "",
223 "redirectUris": ["*"],
224 },
225 skip_exists=True,
226 )
227 client_data = kc_admin.get_client(client_id)
228 _ = Path("/shared/dawarich").write_text(client_data["secret"])
229
197 # forward auth230 # forward auth
198 client_id = kc_admin.create_client(231 client_id = kc_admin.create_client(
199 {232 {
config/spindle/Dockerfile+1-1
...@@ -1,7 +1,7 @@...@@ -1,7 +1,7 @@
1FROM golang:1.25-alpine AS builder1FROM golang:1.25-alpine AS builder
2ENV CGO_ENABLED=12ENV CGO_ENABLED=1
33
4ARG TAG="v1.14.0-alpha"4ARG TAG="v1.15.0-alpha"
55
6WORKDIR /app6WORKDIR /app
7RUN apk add --no-cache git gcc musl-dev7RUN apk add --no-cache git gcc musl-dev
config/yt-feed/app.py+674-26
...@@ -16,12 +16,15 @@ import urllib.request...@@ -16,12 +16,15 @@ import urllib.request
16import xml.etree.ElementTree as ET16import xml.etree.ElementTree as ET
17from email.message import EmailMessage17from email.message import EmailMessage
18from email.utils import formatdate18from email.utils import formatdate
19from xml.sax.saxutils import escape19from xml.sax.saxutils import escape, unescape
2020
21import yaml21import yaml
22from flask import Flask, Response, redirect, request22from flask import Flask, Response, redirect, request
2323
24FEED_CONFIG = os.environ.get("FEED_CONFIG", "/config-yt/feed.yaml")24FEED_CONFIG = os.environ.get("FEED_CONFIG", "/yt-config/feed.yaml")
25# the channel lists live outside the repo (Clover ▸ Documents/Config/Youtube
26# Downloader) so they can be edited from the web ui; this dir is mounted rw
27YT_CONFIG_DIR = os.environ.get("YT_CONFIG_DIR", "/yt-config")
25STATE_DIR = os.environ.get("STATE_DIR", "/data")28STATE_DIR = os.environ.get("STATE_DIR", "/data")
26INTERVAL = int(os.environ.get("CHECK_INTERVAL", "1800"))29INTERVAL = int(os.environ.get("CHECK_INTERVAL", "1800"))
27SMTP_HOST = os.environ["SMTP_HOST"]30SMTP_HOST = os.environ["SMTP_HOST"]
...@@ -125,6 +128,59 @@ def safe_name(name):...@@ -125,6 +128,59 @@ def safe_name(name):
125 return re.sub(r'[/\\:*?"<>|]', "-", name).strip() or "untitled"128 return re.sub(r'[/\\:*?"<>|]', "-", name).strip() or "untitled"
126129
127130
131def looks_like_url(s):
132 return bool(re.match(r"https?://", (s or "").strip(), re.I))
133
134
135def stable_key(v):
136 # the card's identity stays put across a stub resolving (id changes,
137 # stub_id doesn't) so the page can refresh a card in place
138 return v.get("stub_id") or v["id"]
139
140
141def safe_listdir(p):
142 try:
143 return os.listdir(p)
144 except OSError:
145 return []
146
147
148def nfo_fields(path, *tags):
149 try:
150 with open(path) as f:
151 txt = f.read()
152 except OSError:
153 return {t: "" for t in tags}
154 out = {}
155 for t in tags:
156 m = re.search(rf"<{t}>(.*?)</{t}>", txt, re.S)
157 out[t] = unescape(m.group(1)).strip() if m else ""
158 return out
159
160
161def under(base, path):
162 base = os.path.realpath(base)
163 path = os.path.realpath(path)
164 return path == base or path.startswith(base + os.sep)
165
166
167def set_nfo_tags(path, updates):
168 # rewrite individual <tag> values in an existing nfo, leaving the rest as-is
169 try:
170 with open(path) as f:
171 txt = f.read()
172 except OSError:
173 return
174 for k, v in updates.items():
175 new = f"<{k}>{escape(str(v))}</{k}>"
176 if re.search(rf"<{k}>.*?</{k}>", txt, re.S):
177 txt = re.sub(rf"<{k}>.*?</{k}>", new, txt, count=1, flags=re.S)
178 else:
179 txt = re.sub(r"(\n</[A-Za-z]+>\s*)\Z", f"\n {new}\\1", txt)
180 with open(path, "w") as f:
181 f.write(txt)
182
183
128# ---------------------------------------------------------------- feed poller184# ---------------------------------------------------------------- feed poller
129185
130def channel_id_for(url, cache):186def channel_id_for(url, cache):
...@@ -422,7 +478,7 @@ def indie_shows():...@@ -422,7 +478,7 @@ def indie_shows():
422PAGE = """<!doctype html>478PAGE = """<!doctype html>
423<html lang="en"><head>479<html lang="en"><head>
424<meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1">480<meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1">
425<title>yt triage</title>481<title>yt downloader</title>
426<style>482<style>
427:root {{ color-scheme: light dark; }}483:root {{ color-scheme: light dark; }}
428body {{ font-family: system-ui, sans-serif; max-width: 32rem; margin: 0 auto;484body {{ font-family: system-ui, sans-serif; max-width: 32rem; margin: 0 auto;
...@@ -447,8 +503,11 @@ button.skip {{ background: transparent; color: light-dark(#555, #aac); }}...@@ -447,8 +503,11 @@ button.skip {{ background: transparent; color: light-dark(#555, #aac); }}
447.done {{ color: light-dark(#2a7a2a, #8fd48f); }}503.done {{ color: light-dark(#2a7a2a, #8fd48f); }}
448.error {{ color: light-dark(#b03030, #f0a0a0); }}504.error {{ color: light-dark(#b03030, #f0a0a0); }}
449</style></head><body>505</style></head><body>
450<h1>yt triage <small style="font-weight:400">· {npending} pending</small></h1>506<h1>yt downloader <small style="font-weight:400">· {npending} pending</small>
451{cards}507 <span style="float:right;font-size:.85rem;font-weight:400">
508 <a href="/library">library ▸</a> · <a href="/config">channels ▸</a></span></h1>
509{banner}
510<div id="cards">{cards}</div>
452<form method="post" action="/add" class="card row">511<form method="post" action="/add" class="card row">
453 <input name="url" placeholder="paste youtube url(s)…" style="flex:3">512 <input name="url" placeholder="paste youtube url(s)…" style="flex:3">
454 <button>add</button>513 <button>add</button>
...@@ -474,11 +533,35 @@ function seasonOptions(f, show) {{...@@ -474,11 +533,35 @@ function seasonOptions(f, show) {{
474 sel.onchange = () => f.querySelector("input[name=episode]").value = (seasons[sel.value] || 0) + 1;533 sel.onchange = () => f.querySelector("input[name=episode]").value = (seasons[sel.value] || 0) + 1;
475 sel.onchange();534 sel.onchange();
476}}535}}
477document.querySelectorAll("select[name=dest]").forEach(destChanged);536function initCard(card) {{ const s = card.querySelector("select[name=dest]"); if (s) destChanged(s); }}
478setInterval(async () => {{537document.querySelectorAll("#cards .card").forEach(initCard);
479 const r = await fetch("/jobs.html");538// any real edit "dirties" a card so the background refresh won't clobber it
480 document.getElementById("jobs").innerHTML = await r.text();539const cards = document.getElementById("cards");
481}}, 4000);540for (const ev of ["input", "change"]) cards.addEventListener(ev, e => {{
541 const f = e.target.closest(".card"); if (f) f.dataset.dirty = "1";
542}}, true);
543let busy = false;
544async function refresh() {{
545 try {{
546 document.getElementById("jobs").innerHTML = await (await fetch("/jobs.html")).text();
547 }} catch (e) {{}}
548 if (busy) return; busy = true;
549 try {{
550 const tmp = document.createElement("div");
551 tmp.innerHTML = await (await fetch("/cards.html")).text();
552 const fresh = {{}};
553 tmp.querySelectorAll(".card[data-key]").forEach(c => fresh[c.dataset.key] = c);
554 cards.querySelectorAll(".card[data-key]").forEach(c => {{
555 if (!fresh[c.dataset.key] && c.dataset.dirty !== "1") c.remove();
556 }});
557 for (const key in fresh) {{
558 const have = cards.querySelector('.card[data-key="' + key + '"]');
559 if (!have) {{ cards.appendChild(fresh[key]); initCard(fresh[key]); }}
560 else if (have.dataset.dirty !== "1") {{ have.replaceWith(fresh[key]); initCard(fresh[key]); }}
561 }}
562 }} catch (e) {{}} finally {{ busy = false; }}
563}}
564setInterval(refresh, 4000);
482</script>565</script>
483</body></html>"""566</body></html>"""
484567
...@@ -489,18 +572,19 @@ def card_html(v, shows):...@@ -489,18 +572,19 @@ def card_html(v, shows):
489 for s in shows:572 for s in shows:
490 opts.append(f'<option value="indie|{escape(s)}">Indie Shows ▸ {escape(s)}</option>')573 opts.append(f'<option value="indie|{escape(s)}">Indie Shows ▸ {escape(s)}</option>')
491 opts.append('<option value="indie|__new__">Indie Shows ▸ new show…</option>')574 opts.append('<option value="indie|__new__">Indie Shows ▸ new show…</option>')
575 key = stable_key(v)
492 img = (f'<a href="{escape(v["link"])}"><img src="{escape(v["thumb"])}" alt=""></a>'576 img = (f'<a href="{escape(v["link"])}"><img src="{escape(v["thumb"])}" alt=""></a>'
493 if v.get("thumb") else "")577 if v.get("thumb") else "")
494 meta = ("resolving…" if v.get("unresolved")578 meta = ("resolving…" if v.get("unresolved")
495 else f"{escape(v.get('channel', '?'))} · {escape(v.get('published', ''))}")579 else f"{escape(v.get('channel', '?'))} · {escape(v.get('published', ''))}")
496 return f"""<form method="post" action="/ingest" class="card" id="v-{v['id']}">580 return f"""<form method="post" action="/ingest" class="card" id="v-{escape(key)}" data-key="{escape(key)}" data-dirty="0">
497 {img}581 {img}
498 <p class="title">{escape(v['title'])}</p>582 <p class="title">{escape(v['title'])}</p>
499 <p class="meta">{meta}</p>583 <p class="meta">{meta}</p>
500 <input type="hidden" name="vid" value="{escape(v['id'])}">584 <input type="hidden" name="vid" value="{escape(key)}">
501 <select name="dest" onchange="destChanged(this)">{''.join(opts)}</select>585 <select name="dest" onchange="destChanged(this)">{''.join(opts)}</select>
502 <input class="new-show" name="new_show" placeholder="new show name" style="display:none">586 <input class="new-show" name="new_show" placeholder="new show name" style="display:none">
503 <input class="indie-fields" name="ep_title" value="{escape(v['title'])}" placeholder="episode title">587 <input class="indie-fields" name="ep_title" placeholder="{escape(v['title'])}">
504 <div class="indie-fields row">588 <div class="indie-fields row">
505 <select name="season"></select>589 <select name="season"></select>
506 <input name="episode" type="number" min="1" title="episode #">590 <input name="episode" type="number" min="1" title="episode #">
...@@ -530,20 +614,31 @@ def jobs_html():...@@ -530,20 +614,31 @@ def jobs_html():
530 return "\n".join(out)614 return "\n".join(out)
531615
532616
617def render_cards(pending, shows):
618 if not pending:
619 return "<p style='opacity:.6'>nothing pending. enjoy the silence.</p>"
620 return "\n".join(card_html(v, shows) for v in reversed(list(pending.values())))
621
622
533@app.get("/")623@app.get("/")
534def index():624def index():
535 with state_lock:625 with state_lock:
536 pending = load_json("pending.json", {})626 pending = load_json("pending.json", {})
537 shows = indie_shows()627 shows = indie_shows()
538 cards = "\n".join(card_html(v, shows) for v in reversed(list(pending.values())))628 banner = ("<div class='card error'>youtube has bot-walled this ip — downloads "
539 if not pending:629 "are paused and will resume automatically once the wall lifts "
540 cards = "<p style='opacity:.6'>nothing pending. enjoy the silence.</p>"630 "(probed every few hours). queueing still works.</div>"
541 if wall_active():631 if wall_active() else "")
542 cards = ("<div class='card error'>youtube has bot-walled this ip — downloads "632 return PAGE.format(npending=len(pending), banner=banner,
543 "are paused and will resume automatically once the wall lifts "633 cards=render_cards(pending, shows),
544 "(probed every few hours). queueing still works.</div>" + cards)634 jobs=jobs_html(), shows_json=json.dumps(shows))
545 return PAGE.format(npending=len(pending), cards=cards,635
546 jobs=jobs_html(), shows_json=json.dumps(indie_shows()))636
637@app.get("/cards.html")
638def cards_partial():
639 with state_lock:
640 pending = load_json("pending.json", {})
641 return Response(render_cards(pending, indie_shows()), mimetype="text/html")
547642
548643
549@app.get("/jobs.html")644@app.get("/jobs.html")
...@@ -567,9 +662,14 @@ def retry():...@@ -567,9 +662,14 @@ def retry():
567662
568@app.post("/skip")663@app.post("/skip")
569def skip():664def skip():
665 vid = request.form["vid"]
570 with state_lock:666 with state_lock:
571 pending = load_json("pending.json", {})667 pending = load_json("pending.json", {})
572 pending.pop(request.form["vid"], None)668 # a manually-added stub gets re-keyed to the real video id once it
669 # resolves, but the card still submits the stub_id — match on either
670 if vid not in pending:
671 vid = next((k for k, p in pending.items() if p.get("stub_id") == vid), vid)
672 pending.pop(vid, None)
573 save_json("pending.json", pending)673 save_json("pending.json", pending)
574 return redirect("/")674 return redirect("/")
575675
...@@ -613,10 +713,15 @@ def ingest():...@@ -613,10 +713,15 @@ def ingest():
613 if not show:713 if not show:
614 return Response("missing show name", 400, mimetype="text/plain")714 return Response("missing show name", 400, mimetype="text/plain")
615 # a custom episode title; left equal to the card title means "use the715 # a custom episode title; left equal to the card title means "use the
616 # video title" (which, for a still-resolving stub, arrives later)716 # video title" (which, for a still-resolving stub, arrives later). guard
717 # against a url leaking in: the field is prefilled with the stub title,
718 # which for an unresolved paste IS the url — if the stub then resolves
719 # before submit, that url would otherwise be taken as a real title.
617 ep_title = request.form.get("ep_title", "").strip()720 ep_title = request.form.get("ep_title", "").strip()
618 job.update(dest="indie", show=show,721 if not ep_title or ep_title == v["title"] or ep_title == v.get("link") \
619 ep_title=ep_title if ep_title and ep_title != v["title"] else "",722 or looks_like_url(ep_title):
723 ep_title = ""
724 job.update(dest="indie", show=show, ep_title=ep_title,
620 season=int(request.form["season"]), episode=int(request.form["episode"]))725 season=int(request.form["season"]), episode=int(request.form["episode"]))
621 job["dest_label"] = f"{show} S{job['season']:02d}E{job['episode']:02d}"726 job["dest_label"] = f"{show} S{job['season']:02d}E{job['episode']:02d}"
622 elif dest == "independent":727 elif dest == "independent":
...@@ -633,6 +738,549 @@ def ingest():...@@ -633,6 +738,549 @@ def ingest():
633 return redirect("/")738 return redirect("/")
634739
635740
741# ---------------------------------------------------------------- library edit
742
743def library_entries():
744 # every ingested video across the two managed libraries, with its current
745 # title (from the nfo, falling back to the filename) for search + rename
746 out = []
747 for show in sorted(safe_listdir(INDIE_DIR)):
748 show_path = os.path.join(INDIE_DIR, show)
749 if not os.path.isdir(show_path):
750 continue
751 for sub in sorted(safe_listdir(show_path)):
752 sm = re.fullmatch(r"Season (\d+)", sub)
753 if not sm:
754 continue
755 sdir = os.path.join(show_path, sub)
756 for f in sorted(safe_listdir(sdir)):
757 if not f.lower().endswith(VIDEO_EXTS):
758 continue
759 base, _ = os.path.splitext(f)
760 em = re.match(r"S(\d+)E(\d+) - (.*)", base)
761 season, episode = int(sm.group(1)), int(em.group(2)) if em else 1
762 nf = nfo_fields(os.path.join(sdir, base + ".nfo"), "title", "plot")
763 title = nf["title"] or (em.group(3) if em else base)
764 out.append({
765 "type": "indie", "rel": os.path.relpath(os.path.join(sdir, f), "/media"),
766 "ctx": f"{show} · S{season}E{episode}", "title": title,
767 "link": nf["plot"] if looks_like_url(nf["plot"]) else "",
768 "season": season, "episode": episode})
769 for ch in sorted(safe_listdir(INDEP_DIR)):
770 cdir = os.path.join(INDEP_DIR, ch)
771 if not os.path.isdir(cdir):
772 continue
773 for f in sorted(safe_listdir(cdir)):
774 if not f.lower().endswith(VIDEO_EXTS):
775 continue
776 base, _ = os.path.splitext(f)
777 dm = re.match(r"(\d{4}-\d{2}-\d{2}) - (.*)", base)
778 nf = nfo_fields(os.path.join(cdir, base + ".nfo"), "title", "plot")
779 title = nf["title"] or (dm.group(2) if dm else base)
780 out.append({
781 "type": "independent", "rel": os.path.relpath(os.path.join(cdir, f), "/media"),
782 "ctx": f"{ch} · {dm.group(1) if dm else ''}", "title": title,
783 "link": nf["plot"] if looks_like_url(nf["plot"]) else ""})
784 return out
785
786
787LIB_PAGE = """<!doctype html>
788<html lang="en"><head>
789<meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1">
790<title>yt downloader · library</title>
791<style>
792:root {{ color-scheme: light dark; }}
793body {{ font-family: system-ui, sans-serif; max-width: 36rem; margin: 0 auto;
794 padding: 1rem 1rem 3rem; background: light-dark(#e8eefa, #152c42);
795 color: light-dark(#000, #fff); }}
796h1 {{ font-size: 1.2rem; font-weight: 500; }}
797.row {{ background: light-dark(#fff, #1e3a55); border-radius: 10px;
798 padding: 0.6rem 0.7rem; margin-bottom: 0.5rem; }}
799.ctx {{ font-size: 0.78rem; color: light-dark(#555, #aac); margin-bottom: 0.3rem; }}
800input {{ box-sizing: border-box; padding: 0.4rem; border-radius: 7px;
801 border: 1px solid light-dark(#bbb, #456);
802 background: light-dark(#fff, #152c42); color: inherit; }}
803.line {{ display: flex; gap: 0.4rem; align-items: center; }}
804.line .t {{ flex: 1; min-width: 0; }}
805.line .n {{ width: 3.2rem; flex: none; }}
806button {{ padding: 0.4rem 0.8rem; border-radius: 7px; border: none; cursor: pointer;
807 background: #1a46cd; color: #fff; }}
808#q {{ width: 100%; margin-bottom: 0.8rem; padding: 0.55rem; }}
809</style></head><body>
810<h1>yt library <small style="font-weight:400">· {n} videos</small>
811 <a href="/" style="float:right;font-size:.85rem;font-weight:400">◂ home</a></h1>
812<input id="q" placeholder="search…" autofocus>
813{rows}
814<script>
815const q = document.getElementById("q");
816q.addEventListener("input", () => {{
817 const t = q.value.toLowerCase();
818 document.querySelectorAll(".row").forEach(r =>
819 r.style.display = r.dataset.search.includes(t) ? "" : "none");
820}});
821</script>
822</body></html>"""
823
824
825def lib_row(e):
826 rel = escape(e["rel"], {'"': "&quot;"})
827 search = escape((e["ctx"] + " " + e["title"]).lower(), {'"': "&quot;"})
828 if e["type"] == "indie":
829 fields = (f'<input class="n" name="season" type="number" min="1" value="{e["season"]}" title="season">'
830 f'<input class="n" name="episode" type="number" min="1" value="{e["episode"]}" title="episode">')
831 else:
832 fields = ""
833 link = e.get("link")
834 open_link = (f' · <a href="{escape(link, {chr(34): "&quot;"})}" target="_blank" '
835 f'rel="noopener">↗ open</a>' if link else "")
836 return f"""<form method="post" action="/rename" class="row" data-search="{search}">
837 <div class="ctx">{escape(e['ctx'])}{open_link}</div>
838 <input type="hidden" name="rel" value="{rel}">
839 <div class="line">
840 <input class="t" name="title" value="{escape(e['title'], {'"': '&quot;'})}">
841 {fields}
842 <button>save</button>
843 </div>
844</form>"""
845
846
847@app.get("/library")
848def library():
849 entries = library_entries()
850 rows = "\n".join(lib_row(e) for e in entries) or "<p style='opacity:.6'>library is empty.</p>"
851 return LIB_PAGE.format(n=len(entries), rows=rows)
852
853
854@app.post("/rename")
855def rename():
856 rel = request.form.get("rel", "")
857 new_title = request.form.get("title", "").strip()
858 full = os.path.realpath(os.path.join("/media", rel))
859 indie = under(INDIE_DIR, full)
860 if not (indie or under(INDEP_DIR, full)):
861 return Response("path not allowed", 403, mimetype="text/plain")
862 if not os.path.isfile(full):
863 return Response("file not found", 404, mimetype="text/plain")
864 if not new_title:
865 return Response("title required", 400, mimetype="text/plain")
866 d, fname = os.path.split(full)
867 old_base, _ = os.path.splitext(fname)
868 if indie:
869 season = int(request.form.get("season", 1))
870 episode = int(request.form.get("episode", 1))
871 new_base = f"S{season:02d}E{episode:02d} - {safe_name(new_title)}"
872 nfo_updates = {"title": new_title, "season": season, "episode": episode}
873 else:
874 dm = re.match(r"(\d{4}-\d{2}-\d{2}) - ", old_base)
875 new_base = (f"{dm.group(1)} - " if dm else "") + safe_name(new_title)
876 nfo_updates = {"title": new_title}
877 if new_base != old_base:
878 for f in os.listdir(d):
879 # rename the video and every sidecar sharing its basename stem
880 # (.nfo, .jpg, -thumb.jpg, .info.json, …) in lockstep
881 if f.startswith(old_base):
882 suffix = f[len(old_base):]
883 if suffix.startswith(".") or suffix.startswith("-thumb"):
884 os.rename(os.path.join(d, f), os.path.join(d, new_base + suffix))
885 set_nfo_tags(os.path.join(d, new_base + ".nfo"), nfo_updates)
886 log(f"renamed {old_base!r} → {new_base!r}")
887 return redirect("/library")
888
889
890# -------------------------------------------------------------- channel config
891# /config → friendly gui over the auto-download channel list
892# /config/raw → codemirror editor for the raw yaml files (advanced)
893
894SUBS_FILE = "subscriptions.yaml"
895SUB_PRESET_DEFAULT = "Jellyfin TV Show by Date | only-after | flat-videos"
896SUB_SECTION_DEFAULT = "= Independent Creators"
897# per-channel rule keys the gui understands; order = the "add rule" menu order.
898# anything else in a channel entry is left for the raw editor.
899RULE_FIELDS = ["download_after", "title_include_keywords", "title_exclude_keywords",
900 "description_include_keywords", "description_exclude_keywords"]
901RAW_LABELS = {"subscriptions.yaml": "auto-download channels",
902 "feed.yaml": "notify-me channels"}
903
904
905def list_yaml_configs():
906 return sorted(f for f in safe_listdir(YT_CONFIG_DIR)
907 if f.endswith((".yaml", ".yml")))
908
909
910def write_text_atomic(path, text):
911 tmp = path + ".tmp"
912 with open(tmp, "w") as f:
913 f.write(text)
914 os.replace(tmp, path)
915
916
917def parse_subscriptions():
918 # flatten subscriptions.yaml down to a plain channel list for the gui,
919 # remembering the wrapping structure so it can be rebuilt verbatim on save
920 try:
921 with open(os.path.join(YT_CONFIG_DIR, SUBS_FILE)) as f:
922 data = yaml.safe_load(f) or {}
923 except (OSError, yaml.YAMLError):
924 data = {}
925 overrides = (data.get("__preset__") or {}).get("overrides") or {}
926 preset_key = next((k for k in data if k != "__preset__"), SUB_PRESET_DEFAULT)
927 section = data.get(preset_key) if isinstance(data.get(preset_key), dict) else {}
928 section_key = next(iter(section), SUB_SECTION_DEFAULT) if section else SUB_SECTION_DEFAULT
929 chmap = section.get(section_key) if isinstance(section.get(section_key), dict) else {}
930 channels = []
931 for key, val in (chmap or {}).items():
932 name = key[1:].strip() if key.startswith("~") else key
933 if isinstance(val, str):
934 channels.append({"name": name, "url": val, "rules": {}})
935 elif isinstance(val, dict):
936 rules = {k: val[k] for k in RULE_FIELDS if k in val}
937 channels.append({"name": name, "url": val.get("url", ""), "rules": rules})
938 return {"preset_key": preset_key, "section_key": section_key,
939 "overrides": overrides, "channels": channels}
940
941
942def build_subscriptions(channels):
943 # rebuild the file from the gui's channel list, preserving the preset
944 # wrapper + global overrides; a channel with no rules stays the compact
945 # "name: url" form, one with rules becomes the "~name:" dict form
946 cur = parse_subscriptions()
947 chmap = {}
948 for ch in channels:
949 name = (ch.get("name") or "").strip()
950 url = (ch.get("url") or "").strip()
951 if not name or not url:
952 continue
953 clean = {}
954 for field in RULE_FIELDS:
955 v = (ch.get("rules") or {}).get(field)
956 if isinstance(v, str):
957 v = v.strip()
958 if isinstance(v, list):
959 v = [str(x).strip() for x in v if str(x).strip()]
960 if v in (None, "", [], {}):
961 continue
962 clean[field] = v
963 if clean:
964 chmap["~" + name] = {"url": url, **clean}
965 else:
966 chmap[name] = url
967 out = {}
968 if cur["overrides"]:
969 out["__preset__"] = {"overrides": cur["overrides"]}
970 out[cur["preset_key"]] = {cur["section_key"]: chmap}
971 return yaml.dump(out, sort_keys=False, allow_unicode=True, width=4096,
972 default_flow_style=False)
973
974
975@app.get("/api/subscriptions")
976def api_subscriptions_get():
977 return {"channels": parse_subscriptions()["channels"]}
978
979
980@app.post("/api/subscriptions")
981def api_subscriptions_save():
982 payload = request.get_json(silent=True) or {}
983 channels = payload.get("channels")
984 if not isinstance(channels, list):
985 return {"error": "expected a channels list"}, 400
986 text = build_subscriptions(channels)
987 try:
988 yaml.safe_load(text)
989 except yaml.YAMLError as e:
990 return {"error": f"could not save: {e}"}, 400
991 write_text_atomic(os.path.join(YT_CONFIG_DIR, SUBS_FILE), text)
992 n = sum(1 for c in channels if (c.get("name") or "").strip()
993 and (c.get("url") or "").strip())
994 log(f"subscriptions.yaml saved via gui ({n} channels)")
995 return {"ok": True, "count": n}
996
997
998@app.get("/api/configs")
999def api_configs_get():
1000 files = []
1001 for name in list_yaml_configs():
1002 try:
1003 with open(os.path.join(YT_CONFIG_DIR, name)) as f:
1004 files.append({"name": name, "label": RAW_LABELS.get(name, ""),
1005 "body": f.read()})
1006 except OSError:
1007 continue
1008 return {"files": files}
1009
1010
1011@app.post("/api/config")
1012def api_config_save():
1013 payload = request.get_json(silent=True) or {}
1014 name, body = payload.get("name", ""), payload.get("body", "")
1015 if name not in list_yaml_configs():
1016 return {"error": "unknown file"}, 400
1017 try:
1018 yaml.safe_load(body)
1019 except yaml.YAMLError as e:
1020 return {"error": str(e)}, 400
1021 write_text_atomic(os.path.join(YT_CONFIG_DIR, name), body)
1022 log(f"config saved via raw editor: {name}")
1023 return {"ok": True}
1024
1025
1026GUI_PAGE = """<!doctype html>
1027<html lang="en"><head>
1028<meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1">
1029<title>yt downloader · channels</title>
1030<style>
1031:root { color-scheme: light dark; }
1032body { font-family: system-ui, sans-serif; max-width: 38rem; margin: 0 auto;
1033 padding: 1rem 1rem 6rem; background: light-dark(#e8eefa, #152c42);
1034 color: light-dark(#000, #fff); }
1035h1 { font-size: 1.2rem; font-weight: 500; }
1036h1 a { float: right; font-size: .85rem; font-weight: 400; }
1037.sub { font-size: .85rem; color: light-dark(#555, #aac); margin: -.3rem 0 1rem; }
1038.card { background: light-dark(#fff, #1e3a55); border-radius: 12px;
1039 padding: .7rem .8rem; margin-bottom: .7rem; }
1040input, select { box-sizing: border-box; padding: .42rem .5rem; border-radius: 8px;
1041 border: 1px solid light-dark(#bbb, #456); font-size: .9rem;
1042 background: light-dark(#fff, #122739); color: inherit; }
1043.chrow { display: flex; gap: .4rem; align-items: center; }
1044.chrow .name { flex: 2; min-width: 0; font-weight: 600; }
1045.chrow .url { flex: 3; min-width: 0; }
1046.x { flex: none; width: 2rem; padding: .3rem 0; background: transparent;
1047 color: light-dark(#999, #88a); border: none; cursor: pointer; font-size: 1.2rem; }
1048.rule { display: flex; gap: .4rem; align-items: center; flex-wrap: wrap;
1049 margin: .45rem 0 0; padding: .4rem .5rem; border-radius: 8px;
1050 background: light-dark(#eef2fb, #16304880); }
1051.rlabel { font-size: .82rem; color: light-dark(#445, #bcd); flex: none; }
1052.rule input[type=date] { flex: none; }
1053.rule .wordin { flex: 1; min-width: 6rem; }
1054.chips { display: flex; gap: .3rem; flex-wrap: wrap; }
1055.chip { display: inline-flex; align-items: center; gap: .3rem; font-size: .8rem;
1056 background: light-dark(#d8e0f5, #28507a); padding: .12rem .2rem .12rem .5rem;
1057 border-radius: 99px; }
1058.chip button { background: none; border: none; color: inherit; cursor: pointer;
1059 font-size: .95rem; line-height: 1; padding: 0 .15rem; opacity: .7; }
1060.rremove { flex: none; background: transparent; border: none; cursor: pointer;
1061 color: light-dark(#a55, #f0a0a0); font-size: .78rem; }
1062.addrule select { margin-top: .45rem; font-size: .82rem; color: light-dark(#456, #9bf);
1063 background: transparent; border: 1px dashed light-dark(#aab, #567); }
1064.add-channel { width: 100%; padding: .6rem; border-radius: 10px; border: 1px dashed
1065 light-dark(#9ab, #567); background: transparent; color: light-dark(#345, #bcd);
1066 cursor: pointer; font-size: .95rem; }
1067.bar { position: fixed; left: 0; right: 0; bottom: 0; display: flex; gap: .8rem;
1068 align-items: center; padding: .7rem 1rem;
1069 background: light-dark(#dde6f7ee, #112338ee); backdrop-filter: blur(6px);
1070 border-top: 1px solid light-dark(#ccd, #244260); }
1071#save { padding: .55rem 1.4rem; border-radius: 9px; border: none; cursor: pointer;
1072 background: #1a46cd; color: #fff; font-size: .95rem; }
1073#status { font-size: .85rem; color: light-dark(#456, #bcd); }
1074.bar a { margin-left: auto; font-size: .85rem; }
1075</style></head><body>
1076<h1>channels <a href="/">◂ home</a></h1>
1077<p class="sub">videos from these channels download automatically.</p>
1078<div id="list"></div>
1079<button class="add-channel" id="add">+ add channel</button>
1080<div class="bar"><button id="save">save</button><span id="status"></span>
1081 <a href="/config/raw">edit raw yaml ▸</a></div>
1082<script>
1083const RULES = [
1084 {key:"download_after", label:"Backlog", kind:"backlog"},
1085 {key:"title_include_keywords", label:"Only if title contains", kind:"words"},
1086 {key:"title_exclude_keywords", label:"Skip if title contains", kind:"words"},
1087 {key:"description_include_keywords", label:"Only if description contains", kind:"words"},
1088 {key:"description_exclude_keywords", label:"Skip if description contains", kind:"words"},
1089];
1090const byKey = Object.fromEntries(RULES.map(r => [r.key, r]));
1091let channels = [];
1092const list = document.getElementById("list");
1093const statusEl = document.getElementById("status");
1094
1095const ymd2date = s => (s && s.length === 8) ? s.slice(0,4)+"-"+s.slice(4,6)+"-"+s.slice(6,8) : "";
1096const date2ymd = s => s ? s.replaceAll("-", "") : "";
1097
1098function el(tag, props, ...kids) {
1099 const e = Object.assign(document.createElement(tag), props || {});
1100 for (const k of kids) e.append(k);
1101 return e;
1102}
1103
1104function render() {
1105 list.innerHTML = "";
1106 channels.forEach((ch, i) => list.append(card(ch, i)));
1107}
1108
1109function card(ch, i) {
1110 const name = el("input", {className:"name", value:ch.name||"", placeholder:"channel name"});
1111 name.oninput = () => ch.name = name.value;
1112 const url = el("input", {className:"url", value:ch.url||"", placeholder:"youtube.com/@handle"});
1113 url.oninput = () => ch.url = url.value;
1114 const x = el("button", {className:"x", type:"button", textContent:"×",
1115 title:"remove channel", onclick:() => { channels.splice(i,1); render(); }});
1116 const c = el("div", {className:"card"}, el("div", {className:"chrow"}, name, url, x));
1117 for (const key of Object.keys(ch.rules || {})) c.append(ruleRow(ch, key));
1118 c.append(addRule(ch));
1119 return c;
1120}
1121
1122function ruleRow(ch, key) {
1123 const meta = byKey[key];
1124 const row = el("div", {className:"rule"}, el("span", {className:"rlabel", textContent:meta.label}));
1125 if (meta.kind === "backlog") {
1126 const cur = ch.rules[key] || "19700101";
1127 const sel = el("select");
1128 sel.append(el("option", {value:"all", textContent:"entire history"}),
1129 el("option", {value:"date", textContent:"since a date"}));
1130 const date = el("input", {type:"date"});
1131 const sync = () => {
1132 if (sel.value === "all") { ch.rules[key] = "19700101"; date.style.display = "none"; }
1133 else { date.style.display = ""; ch.rules[key] = date2ymd(date.value); }
1134 };
1135 if (cur === "19700101") { sel.value = "all"; date.style.display = "none"; }
1136 else { sel.value = "date"; date.value = ymd2date(cur); }
1137 sel.onchange = sync; date.oninput = sync;
1138 row.append(sel, date);
1139 } else {
1140 const chips = el("div", {className:"chips"});
1141 (ch.rules[key] || []).forEach((w, wi) => chips.append(chip(ch.rules[key], wi)));
1142 const inp = el("input", {className:"wordin", placeholder:"type a word, press enter"});
1143 inp.onkeydown = e => {
1144 if (e.key === "Enter") {
1145 e.preventDefault();
1146 const v = inp.value.trim();
1147 if (v) { (ch.rules[key] = ch.rules[key] || []).push(v); render(); }
1148 }
1149 };
1150 row.append(chips, inp);
1151 }
1152 row.append(el("button", {className:"rremove", type:"button", textContent:"remove",
1153 onclick:() => { delete ch.rules[key]; render(); }}));
1154 return row;
1155}
1156
1157function chip(arr, i) {
1158 return el("span", {className:"chip", textContent:arr[i]},
1159 el("button", {type:"button", textContent:"×", onclick:() => { arr.splice(i,1); render(); }}));
1160}
1161
1162function addRule(ch) {
1163 const avail = RULES.filter(r => !(r.key in (ch.rules || {})));
1164 const bar = el("div", {className:"addrule"});
1165 if (!avail.length) return bar;
1166 const sel = el("select");
1167 sel.append(el("option", {value:"", textContent:"+ add rule…"}));
1168 for (const r of avail) sel.append(el("option", {value:r.key, textContent:r.label}));
1169 sel.onchange = () => {
1170 if (!sel.value) return;
1171 ch.rules = ch.rules || {};
1172 ch.rules[sel.value] = byKey[sel.value].kind === "backlog" ? "19700101" : [];
1173 render();
1174 };
1175 return bar.append(sel), bar;
1176}
1177
1178document.getElementById("add").onclick = () => {
1179 channels.push({name:"", url:"", rules:{}});
1180 render();
1181 const last = list.lastChild && list.lastChild.querySelector("input.name");
1182 if (last) last.focus();
1183};
1184
1185document.getElementById("save").onclick = async () => {
1186 statusEl.textContent = "saving…";
1187 try {
1188 const r = await fetch("/api/subscriptions", {method:"POST",
1189 headers:{"content-type":"application/json"}, body:JSON.stringify({channels})});
1190 const j = await r.json().catch(() => ({}));
1191 statusEl.textContent = r.ok ? `saved ✓ · ${j.count} channels` : `error: ${j.error || r.status}`;
1192 } catch (e) { statusEl.textContent = "error: " + e; }
1193 setTimeout(() => statusEl.textContent = "", 5000);
1194};
1195
1196(async () => {
1197 const r = await fetch("/api/subscriptions");
1198 channels = (await r.json()).channels || [];
1199 render();
1200})();
1201</script>
1202</body></html>"""
1203
1204
1205RAW_PAGE = """<!doctype html>
1206<html lang="en"><head>
1207<meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1">
1208<title>yt downloader · raw yaml</title>
1209<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/codemirror/5.65.16/codemirror.min.css">
1210<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/codemirror/5.65.16/theme/material-darker.min.css">
1211<style>
1212:root { color-scheme: light dark; }
1213body { font-family: system-ui, sans-serif; max-width: 46rem; margin: 0 auto;
1214 padding: 1rem 1rem 4rem; background: light-dark(#e8eefa, #152c42);
1215 color: light-dark(#000, #fff); }
1216h1 { font-size: 1.2rem; font-weight: 500; }
1217h1 a { float: right; font-size: .85rem; font-weight: 400; }
1218h2 { font-size: 1rem; margin: 1.6rem 0 .1rem; }
1219h2 small { font-weight: 400; color: light-dark(#667, #9ab); }
1220.CodeMirror { height: auto; border-radius: 10px; border: 1px solid light-dark(#bbb, #345);
1221 font-size: .85rem; }
1222.CodeMirror-scroll { min-height: 12rem; max-height: 28rem; }
1223.act { display: flex; gap: .8rem; align-items: center; margin-top: .4rem; }
1224button { padding: .45rem 1.1rem; border-radius: 8px; border: none; cursor: pointer;
1225 background: #1a46cd; color: #fff; font-size: .9rem; }
1226.msg { font-size: .85rem; }
1227.msg.bad { color: light-dark(#b03030, #f0a0a0); white-space: pre-wrap;
1228 font-family: ui-monospace, monospace; }
1229.msg.ok { color: light-dark(#2a7a2a, #8fd48f); }
1230</style></head><body>
1231<h1>raw yaml <a href="/config">◂ channels</a></h1>
1232<div id="files">loading…</div>
1233<script src="https://cdnjs.cloudflare.com/ajax/libs/codemirror/5.65.16/codemirror.min.js"></script>
1234<script src="https://cdnjs.cloudflare.com/ajax/libs/codemirror/5.65.16/mode/yaml/yaml.min.js"></script>
1235<script>
1236const dark = matchMedia("(prefers-color-scheme: dark)").matches;
1237(async () => {
1238 const files = (await (await fetch("/api/configs")).json()).files || [];
1239 const root = document.getElementById("files");
1240 root.innerHTML = "";
1241 for (const f of files) {
1242 const h = document.createElement("h2");
1243 h.innerHTML = f.name + (f.label ? ' <small>· ' + f.label + '</small>' : '');
1244 const ta = document.createElement("textarea");
1245 ta.value = f.body;
1246 const act = document.createElement("div"); act.className = "act";
1247 const btn = document.createElement("button"); btn.textContent = "save " + f.name;
1248 const msg = document.createElement("span"); msg.className = "msg";
1249 act.append(btn, msg);
1250 root.append(h, ta, act);
1251 const cm = CodeMirror.fromTextArea(ta, {mode:"yaml", lineNumbers:true,
1252 theme: dark ? "material-darker" : "default", viewportMargin: Infinity});
1253 btn.onclick = async () => {
1254 msg.textContent = "saving…"; msg.className = "msg";
1255 const r = await fetch("/api/config", {method:"POST",
1256 headers:{"content-type":"application/json"},
1257 body: JSON.stringify({name: f.name, body: cm.getValue()})});
1258 const j = await r.json().catch(() => ({}));
1259 if (r.ok) { msg.textContent = "saved ✓"; msg.className = "msg ok"; }
1260 else { msg.textContent = j.error || ("error " + r.status); msg.className = "msg bad"; }
1261 };
1262 }
1263})();
1264</script>
1265</body></html>"""
1266
1267
1268@app.get("/config")
1269def config_page():
1270 return Response(GUI_PAGE, mimetype="text/html")
1271
1272
1273@app.get("/config/raw")
1274def config_raw():
1275 return Response(RAW_PAGE, mimetype="text/html")
1276
1277
1278@app.get("/api/upscale")
1279def api_upscale_status():
1280 # progress is written by the separate yt-upscaler service into our data dir
1281 return load_json("upscale-status.json", {"enabled": False})
1282
1283
636if __name__ == "__main__":1284if __name__ == "__main__":
637 # jobs interrupted by a container restart pick up where they left off1285 # jobs interrupted by a container restart pick up where they left off
638 with state_lock:1286 with state_lock:
config/yt-upscaler/Dockerfile created+8
...@@ -0,0 +1,8 @@
1# standalone thumbnail super-resolution worker. kept off the ytdl-sub base
2# image because that image is on python 3.14, which onnxruntime has no wheels
3# for yet; this pins a supported python and stays tiny (no yt-dlp/ffmpeg).
4FROM python:3.12-slim
5RUN pip install --no-cache-dir onnxruntime numpy pillow
6COPY realesr-general-x4v3.onnx /app/realesr-general-x4v3.onnx
7COPY upscale.py /app/upscale.py
8CMD ["python3", "-u", "/app/upscale.py"]
config/yt-upscaler/upscale.py created+194
...@@ -0,0 +1,194 @@
1#!/usr/bin/env python3
2# yt-upscaler: keeps the Independent library's thumbnails sharp on a TV.
3# youtube only stores ~720p thumbnails, which jellyfin then stretches across
4# the whole screen as the item backdrop (soft + blocky). this re-fetches the
5# highest-res thumbnail from the image CDN (not bot-walled) and runs it through
6# Real-ESRGAN general-x4v3 (exported to ONNX, so no pickle is ever loaded) at
7# 4x on CPU, writing a crisp <video>.webp that jellyfin uses for card + backdrop.
8import io
9import json
10import os
11import re
12import time
13import urllib.request
14
15import numpy as np
16import onnxruntime as ort
17from PIL import Image
18
19INDEP_DIR = os.environ.get("INDEP_DIR", "/media/jellyfin/Independent")
20STATE_DIR = os.environ.get("STATE_DIR", "/state")
21MODEL = os.environ.get("SR_MODEL", "/app/realesr-general-x4v3.onnx")
22MIN_H = int(os.environ.get("SR_MIN_H", "1400")) # >= this tall ⇒ already done
23CAP_H = int(os.environ.get("SR_CAP_H", "2160")) # cap output height (4k-ish)
24TILE = int(os.environ.get("SR_TILE", "256")) # tile size to bound memory
25THREADS = int(os.environ.get("SR_THREADS", "4")) # leave cpu for other services
26SCAN_INTERVAL = int(os.environ.get("SR_SCAN_INTERVAL", "1800"))
27VIDEO_EXTS = (".webm", ".mp4", ".mkv")
28UA = {"User-Agent": "Mozilla/5.0 (yt-upscaler; +https://paperclover.net)"}
29
30status = {"enabled": True, "running": False, "done": 0, "total": 0,
31 "current": "", "errors": 0}
32_sess = None
33
34
35def log(m):
36 print(m, flush=True)
37
38
39def write_status():
40 try:
41 tmp = os.path.join(STATE_DIR, "upscale-status.json.tmp")
42 with open(tmp, "w") as f:
43 json.dump(status, f)
44 os.replace(tmp, os.path.join(STATE_DIR, "upscale-status.json"))
45 except OSError:
46 pass
47
48
49def session():
50 global _sess
51 if _sess is None:
52 opts = ort.SessionOptions()
53 opts.intra_op_num_threads = THREADS
54 opts.inter_op_num_threads = 1
55 _sess = ort.InferenceSession(MODEL, sess_options=opts,
56 providers=["CPUExecutionProvider"])
57 return _sess
58
59
60def upscale(img, scale=4, pad=8):
61 # tiled 4x SR; overlap each tile by `pad` px and crop it back to hide seams
62 arr = np.asarray(img, dtype=np.float32) / 255.0
63 h, w, _ = arr.shape
64 out = np.zeros((h * scale, w * scale, 3), dtype=np.float32)
65 sess = session()
66 for y in range(0, h, TILE):
67 for x in range(0, w, TILE):
68 y0, x0 = max(0, y - pad), max(0, x - pad)
69 y1, x1 = min(h, y + TILE + pad), min(w, x + TILE + pad)
70 patch = arr[y0:y1, x0:x1].transpose(2, 0, 1)[None]
71 res = sess.run(None, {"input": patch})[0][0].transpose(1, 2, 0)
72 th, tw = min(TILE, h - y) * scale, min(TILE, w - x) * scale
73 ty, tx = (y - y0) * scale, (x - x0) * scale
74 out[y * scale:y * scale + th, x * scale:x * scale + tw] = \
75 res[ty:ty + th, tx:tx + tw]
76 return Image.fromarray((out.clip(0, 1) * 255).round().astype("uint8"))
77
78
79def yt_id(base):
80 ij = base + ".info.json"
81 if os.path.exists(ij):
82 try:
83 with open(ij) as f:
84 vid = json.load(f).get("id")
85 if vid:
86 return vid
87 except (OSError, json.JSONDecodeError):
88 pass
89 try:
90 with open(base + ".nfo") as f:
91 m = re.search(r"<plot>(.*?)</plot>", f.read(), re.S)
92 if m:
93 u = re.search(r"(?:v=|youtu\.be/)([A-Za-z0-9_-]{11})", m.group(1))
94 return u.group(1) if u else None
95 except OSError:
96 pass
97 return None
98
99
100def fetch_cdn_webp(vid):
101 for q in ("maxresdefault", "sddefault", "hqdefault"):
102 try:
103 req = urllib.request.Request(
104 f"https://i.ytimg.com/vi_webp/{vid}/{q}.webp", headers=UA)
105 with urllib.request.urlopen(req, timeout=30) as r:
106 data = r.read()
107 if len(data) > 1000:
108 return Image.open(io.BytesIO(data)).convert("RGB")
109 except Exception:
110 continue
111 return None
112
113
114def img_height(path):
115 try:
116 with Image.open(path) as im:
117 return im.height
118 except Exception:
119 return 0
120
121
122def enhance(video_path):
123 base = os.path.splitext(video_path)[0]
124 webp, jpg = base + ".webp", base + ".jpg"
125 if os.path.exists(webp) and img_height(webp) >= MIN_H:
126 return "skip"
127 src = None
128 vid = yt_id(base)
129 if vid:
130 src = fetch_cdn_webp(vid)
131 if src is None:
132 for p in (webp, jpg):
133 if os.path.exists(p):
134 src = Image.open(p).convert("RGB")
135 break
136 if src is None:
137 return "no-source"
138 up = upscale(src)
139 if up.height > CAP_H:
140 up = up.resize((round(up.width * CAP_H / up.height), CAP_H), Image.LANCZOS)
141 tmp = webp + ".tmp"
142 up.save(tmp, "WEBP", quality=92, method=6)
143 os.replace(tmp, webp)
144 if os.path.exists(jpg):
145 os.remove(jpg)
146 return "done"
147
148
149def independent_videos():
150 out = []
151 for ch in sorted(os.listdir(INDEP_DIR)) if os.path.isdir(INDEP_DIR) else []:
152 cdir = os.path.join(INDEP_DIR, ch)
153 if not os.path.isdir(cdir):
154 continue
155 for f in sorted(os.listdir(cdir)):
156 if f.lower().endswith(VIDEO_EXTS):
157 out.append(os.path.join(cdir, f))
158 return out
159
160
161def main():
162 log(f"yt-upscaler started (model={MODEL}, threads={THREADS})")
163 while True:
164 vids = independent_videos()
165 pending = [v for v in vids
166 if img_height(os.path.splitext(v)[0] + ".webp") < MIN_H]
167 status.update(total=len(vids), done=len(vids) - len(pending),
168 running=bool(pending), current="")
169 write_status()
170 if pending:
171 log(f"upscaling {len(pending)} thumbnail(s)…")
172 for v in pending:
173 status["current"] = os.path.basename(v)
174 write_status()
175 t = time.time()
176 try:
177 r = enhance(v)
178 if r == "done":
179 log(f"upscaled ({time.time()-t:.0f}s): {os.path.basename(v)}")
180 elif r != "skip":
181 log(f"{r}: {os.path.basename(v)}")
182 except Exception as e:
183 status["errors"] += 1
184 log(f"failed: {os.path.basename(v)}: {e}")
185 status["done"] += 1
186 write_status()
187 time.sleep(1) # be polite to the rest of the box
188 status.update(running=False, current="")
189 write_status()
190 time.sleep(SCAN_INTERVAL)
191
192
193if __name__ == "__main__":
194 main()
config/yt/archive-loop.sh+1-1
...@@ -3,7 +3,7 @@...@@ -3,7 +3,7 @@
3# ("sign in to confirm you're not a bot"), back off for a whole day instead3# ("sign in to confirm you're not a bot"), back off for a whole day instead
4# of hammering it every cycle, which prolongs the wall.4# of hammering it every cycle, which prolongs the wall.
5while true; do5while true; do
6 ytdl-sub --config /config-yt/config.yaml sub /config-yt/subscriptions.yaml 2>&1 | tee /tmp/last-pass.log6 ytdl-sub --config /config-yt/config.yaml sub /yt-config/subscriptions.yaml 2>&1 | tee /tmp/last-pass.log
7 if grep -q "confirm you.re not a bot" /tmp/last-pass.log; then7 if grep -q "confirm you.re not a bot" /tmp/last-pass.log; then
8 echo "[archive-loop] bot wall detected; sleeping 24h"8 echo "[archive-loop] bot wall detected; sleeping 24h"
9 sleep 864009 sleep 86400
config/yt/config.yaml+14
...@@ -26,3 +26,17 @@ presets:...@@ -26,3 +26,17 @@ presets:
26 episode_file_path: "{episode_file_name_sanitized}"26 episode_file_path: "{episode_file_name_sanitized}"
27 episode_file_name: "{upload_date_standardized} - {file_title}"27 episode_file_name: "{upload_date_standardized} - {file_title}"
28 thumbnail_file_name: "{episode_file_path}.jpg"28 thumbnail_file_name: "{episode_file_path}.jpg"
29 # sponsorblock: cut paid sponsor reads, self-promo (merch/patreon), and
30 # like/subscribe reminders out of newly downloaded videos. intros, outros,
31 # and all real content are kept. applies to new downloads only (the archive
32 # remembers what's already fetched). segment data is crowd-sourced from the
33 # sponsorblock api, which isn't affected by youtube bot walls.
34 chapters:
35 sponsorblock_categories:
36 - sponsor
37 - selfpromo
38 - interaction
39 remove_sponsorblock_categories:
40 - sponsor
41 - selfpromo
42 - interaction
config/yt/feed.yaml deleted-22
...@@ -1,22 +0,0 @@
1# channels whose new videos land in the triage queue at yt.<domain> for
2# manual sorting into shows/music/creators (home-infra issue #6). each also
3# sends a notification email with a link to the review page.
4# adding a channel is one line; any youtube channel url or @handle url works.
5channels:
6 "ArrowType": "https://www.youtube.com/@ArrowType"
7 "SethBling": "https://www.youtube.com/@SethBling"
8 "Voidstar": "https://www.youtube.com/@voidstar-digital"
9 "MallBat": "https://www.youtube.com/@mallbat"
10 "Early Eyes": "https://www.youtube.com/@earlyeyes"
11 "Otaku-Vs": "https://www.youtube.com/@OtakuVs"
12 "Something Witty Entertainment": "https://www.youtube.com/@SWE"
13 "Ethan Niser": "https://www.youtube.com/@ethanniser"
14 "V3rb": "https://www.youtube.com/@VerbDoesStuff"
15 "dyc3": "https://www.youtube.com/@rollthedyc3"
16 "Technology Connections": "https://www.youtube.com/@TechnologyConnections"
17 "jan Misali": "https://www.youtube.com/@HBMmaster"
18 "awe": https://www.youtube.com/@whyawe
19 "JJBlair": "https://www.youtube.com/@JJBlairrecording"
20 "2 Mello": "https://www.youtube.com/@2mello"
21 "XavierWolf": "https://www.youtube.com/@xavierwolfy"
22 "chaosyumi": "https://www.youtube.com/@willburtz"
config/yt/subscriptions.yaml deleted-69
...@@ -1,69 +0,0 @@
1# channels that are archived automatically (home-infra issue #6).
2# every upload lands in media/jellyfin/Independent/<Channel Name>/ with
3# .nfo metadata + thumbnails so jellyfin shows each channel as a series.
4#
5# to add a channel: one line under the preset, then
6# sh sync.sh && sh docker.sh restart ytdl-sub
7# (or just sync and wait for the next 6h pass)
8
9__preset__:
10 overrides:
11 tv_show_directory: "/media/jellyfin/Independent"
12 # default backlog policy: new uploads only. set download_after on a
13 # channel (the "~name" form) to backfill from a date, or to "19700101"
14 # for the entire backlog.
15 download_after: "20260611"
16
17Jellyfin TV Show by Date | only-after | flat-videos:
18 = Independent Creators:
19 # full archive
20 "~Retro Game Mechanics Explained":
21 url: "https://www.youtube.com/@RGMechEx"
22 download_after: "19700101"
23 # skip videos whose title contains any of these (case-insensitive
24 # substrings). works on any channel entry. for videos already
25 # downloaded, just delete the files — the download archive remembers
26 # them and won't re-fetch.
27 title_exclude_keywords:
28 - "q&a session"
29 - "channel trailer"
30 - "launching memberships"
31 - "subscriber milestone"
32 "~Franco Citera":
33 url: "https://www.youtube.com/@francocitera"
34 download_after: "19700101"
35 # partial backlog
36 "~bill wurtz":
37 url: "https://www.youtube.com/@billwurtz"
38 download_after: "20260401" # 'i'm going off the map' onward
39 "~Coffeezilla":
40 url: "https://www.youtube.com/@Coffeezilla"
41 download_after: "20260609" # 'I Found The $200,000 Missing Lego' onward
42 "~classic j":
43 url: "https://www.youtube.com/@classicj7094"
44 download_after: "20240801"
45 "~JUNIA":
46 url: "https://www.youtube.com/@butterflywife"
47 download_after: "19700101"
48 "~hbomberguy":
49 url: https://www.youtube.com/@hbomberguy
50 download_after: "20190210"
51 # new uploads only (default policy)
52 "t3ssel8r": "https://www.youtube.com/@t3ssel8r"
53 "Nes": "https://www.youtube.com/@nesorion6"
54 "4096": "https://www.youtube.com/@4096"
55 "MegaLag": "https://www.youtube.com/@MegaLag"
56 "Stuff Made Here": "https://www.youtube.com/@StuffMadeHere"
57 "mali potka": "https://www.youtube.com/@malipotka4294"
58 "Michael Reeves": "https://www.youtube.com/@MichaelReeves"
59 "Patrick Foley": "https://www.youtube.com/@patrickfoley489"
60 "orchard phobia": "https://www.youtube.com/@orchardphobia"
61 "Captain Disillusion": "https://www.youtube.com/@CaptainDisillusion"
62 "mnmira": "https://www.youtube.com/@mnmira"
63 "EmpLemon": "https://www.youtube.com/@EmperorLemon"
64 "andMo'": "https://www.youtube.com/@andMo"
65 "~CGP Grey":
66 url: "https://www.youtube.com/@CGPGrey"
67 # members-only preview posts can't download and abort the queue
68 title_exclude_keywords:
69 - "early preview for bonnie bees"
generate-env.sh+1
...@@ -79,6 +79,7 @@ template() {...@@ -79,6 +79,7 @@ template() {
7979
80 section "shale"80 section "shale"
81 add "SHALE_CLIENT_SECRET" ""81 add "SHALE_CLIENT_SECRET" ""
82 add "SHALE_SESSION_SECRET" "$(hex_secret 32)"
8283
83 section "evil infra"84 section "evil infra"
84 add "EVIL_FORGEJO_SERVER_LFS_JWT_SECRET" "$(forgejo generate secret LFS_JWT_SECRET)"85 add "EVIL_FORGEJO_SERVER_LFS_JWT_SECRET" "$(forgejo generate secret LFS_JWT_SECRET)"
vllm/compose.agent.yaml created+147
...@@ -0,0 +1,147 @@
1# AGENT / MULTI-STREAM PROFILE -- for claude code and anything that fires
2# concurrent requests. see compose.patched.yaml for the 262k single-stream one.
3#
4# ./up.sh -f compose.agent.yaml up -d
5#
6# why a separate profile at all: on 24gb, 262k context and robust concurrency
7# are mutually exclusive. the long-context profile sits at 23.6/24.5gb with ONE
8# stream -- a second concurrent request has no activation headroom and the
9# engine dies with a CUDA OOM. measured, not theorised.
10#
11# THE PATCHED STACK -- runs on the CURRENT driver (550). no truenas upgrade
12# needed: all 13 patches are pure-python against vllm 0.27.1 and apply cleanly
13# to the -cu129 image (verified: 13/13 in sequence, vllm still imports).
14#
15# ./up.sh -f compose.patched.yaml up -d
16#
17# only one of vllm / vllm-patched can run at a time -- the model fills the card.
18# this service takes the `vllm` network alias so caddy's reverse_proxy keeps
19# working either way.
20#
21# what it unlocks over compose.yaml: quantized embeddings + MTP (~1.75gb ->
22# more context), int8 kv for spec-decode, the hybrid kv-group cap fix, the mtp
23# draft vocab (+10%), DFlash2, and KVarN 4/2-bit kv.
24name: vllm-agent
25
26services:
27 vllm-agent:
28 container_name: vllm-agent
29 image: vllm/vllm-openai:v0.27.1-cu129
30 restart: unless-stopped
31 ipc: host
32 # same two driver-550 workarounds as the stock config
33 tmpfs:
34 - /usr/local/cuda/compat
35 networks:
36 home-infra:
37 aliases:
38 - vllm
39 volumes:
40 - "${APP_ROOT}/vllm/models:/models:rw"
41 - "${APP_ROOT}/qwen38-stack:/stack:ro"
42 - "${APP_ROOT}/vllm/cache-patched:/cache"
43 - "./patched-entrypoint.sh:/patched-entrypoint.sh:ro"
44 environment:
45 HF_HUB_OFFLINE: "1"
46 VLLM_API_KEY: "${VLLM_API_KEY}"
47 NVIDIA_DISABLE_REQUIRE: "1"
48 PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True"
49 HOME: "/cache"
50 VLLM_NO_USAGE_STATS: "1"
51 entrypoint: ["/patched-entrypoint.sh"]
52 deploy:
53 resources:
54 reservations:
55 devices:
56 - driver: nvidia
57 count: 1
58 capabilities: [gpu]
59 command:
60 - "--model"
61 - "/models/Qwen3.8-27B-W4A16-AutoRound"
62 - "--served-model-name"
63 - "qwen3.8-27b"
64 - "--host"
65 - "0.0.0.0"
66 - "--port"
67 - "8000"
68 # 128k. NOTE this is prompt + max_tokens, not prompt alone: clients
69 # reserve their output budget against it. claude code sends a fixed
70 # max_tokens=32768, so at 98304 the usable prompt was only 65536 and it
71 # failed by ONE token on a 65537-token prompt. at 131072 the usable
72 # prompt is 98304. the kv pool (137,625) already covered this, so the
73 # raise is free -- no extra vram, no loss of concurrency headroom.
74 - "--max-model-len"
75 - "131072"
76 # 0.90, not 0.97: concurrent requests need transient activation memory.
77 # this ~1.7gb of slack is the whole point of this profile.
78 - "--gpu-memory-utilization"
79 - "0.90"
80 # MUST be explicit. left unset, vllm auto-sizes kv to the whole budget and
81 # then OOMs during cuda graph capture (it asks for 800mb it does not have)
82 # and restart-loops, recompiling each time. weights are now ~15.1gb after
83 # the lm_head/embed/mtp requant, +1.8gb peak activation, so ~5gb is what
84 # is actually free for kv once graphs are paid for.
85 # 4.5gb not 5gb: at 5gb the pool is 300k tokens but a ~250k-token request
86 # has no room left for transient GDN state + activations and the engine
87 # dies with a 24mb OOM. 4.5gb still yields >262k tokens of pool and keeps
88 # ~0.5gb of transient headroom.
89 # scheduler fairness: the auto-chosen step budget is 2048, which a single
90 # long prefill consumes entirely -- a concurrent small request (e.g. an
91 # agent's classifier call) then advances ~1 token per slow step and times
92 # out. a larger budget plus a per-prefill cap leaves room in the same
93 # step for other requests' decodes.
94 - "--max-num-batched-tokens"
95 - "4096"
96 - "--long-prefill-token-threshold"
97 - "1024"
98 - "--kv-cache-memory"
99 - "3221225472"
100 # KVarN sizes an fp16 "tail pool" from max_num_seqs -- it capped 256->233
101 # on its own and still OOM'd. we are single-user, so 8 concurrent slots is
102 # plenty and it frees several gb of tail pool for actual kv capacity.
103 # 8, not 16: KVarN sizes an fp16 tail pool from max_num_seqs, so doubling
104 # this OOMs on top of the larger step budget. 8 concurrent streams is
105 # ample for an agent client (main call + classifier + a couple of tools).
106 - "--max-num-seqs"
107 - "8"
108 - "--compilation-config"
109 - '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
110 # KVarN: 4-bit keys / 2-bit values, ~4x more tokens per byte than fp8.
111 # this is what takes context from ~123k to the model's native 262k on the
112 # same 5gb pool. lossy -- verify with a needle test, not just a boot.
113 - "--kv-cache-dtype"
114 - "kvarn_k4v2_g128"
115 - "--mamba-cache-mode"
116 - "align"
117 - "--reasoning-parser"
118 - "qwen3"
119 # bound thinking by default. the chat template defaults reasoning_effort
120 # to *xhigh* when the client sends nothing, and claude code sends nothing.
121 # at xhigh a hard prompt burns >8k tokens inside the <think> block, hits
122 # max_tokens before emitting </think>, and returns a thinking block with
123 # NO text block at all -- i.e. an empty answer. measured: xhigh at 8192
124 # output = unusable; medium and low both finish cleanly with real answers.
125 - "--default-chat-template-kwargs"
126 - '{"reasoning_effort":"medium"}'
127 - "--enable-auto-tool-choice"
128 - "--tool-call-parser"
129 - "qwen3_coder"
130 - "--enable-prefix-caching"
131 - "--speculative-config"
132 # adaptive speculation -- exactly the "fast alone, scales under load"
133 # behaviour: full 3-token drafting at batch 1, tapering to none past 4
134 # streams where rejected drafts are just wasted compute that could be
135 # serving real tokens.
136 - '{"method":"mtp","num_speculative_tokens":3,"num_speculative_tokens_per_batch_size":[[1,1,3],[2,2,2],[3,4,1],[5,256,0]]}'
137 healthcheck:
138 test: ["CMD-SHELL", "python3 -c \"import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health')\""]
139 interval: 30s
140 timeout: 10s
141 retries: 3
142 start_period: 20m
143
144networks:
145 home-infra:
146 external: true
147 name: home-infra_default
vllm/compose.patched.yaml created+111
...@@ -0,0 +1,111 @@
1# LONG-CONTEXT PROFILE -- single stream. see compose.agent.yaml for the
2# concurrency-tuned profile (claude code / multi-stream).
3#
4# THE PATCHED STACK -- runs on the CURRENT driver (550). no truenas upgrade
5# needed: all 13 patches are pure-python against vllm 0.27.1 and apply cleanly
6# to the -cu129 image (verified: 13/13 in sequence, vllm still imports).
7#
8# ./up.sh -f compose.patched.yaml up -d
9#
10# only one of vllm / vllm-patched can run at a time -- the model fills the card.
11# this service takes the `vllm` network alias so caddy's reverse_proxy keeps
12# working either way.
13#
14# what it unlocks over compose.yaml: quantized embeddings + MTP (~1.75gb ->
15# more context), int8 kv for spec-decode, the hybrid kv-group cap fix, the mtp
16# draft vocab (+10%), DFlash2, and KVarN 4/2-bit kv.
17name: vllm-patched
18
19services:
20 vllm-patched:
21 container_name: vllm-patched
22 image: vllm/vllm-openai:v0.27.1-cu129
23 restart: unless-stopped
24 ipc: host
25 # same two driver-550 workarounds as the stock config
26 tmpfs:
27 - /usr/local/cuda/compat
28 networks:
29 home-infra:
30 aliases:
31 - vllm
32 volumes:
33 - "${APP_ROOT}/vllm/models:/models:rw"
34 - "${APP_ROOT}/qwen38-stack:/stack:ro"
35 - "${APP_ROOT}/vllm/cache-patched:/cache"
36 - "./patched-entrypoint.sh:/patched-entrypoint.sh:ro"
37 environment:
38 HF_HUB_OFFLINE: "1"
39 VLLM_API_KEY: "${VLLM_API_KEY}"
40 NVIDIA_DISABLE_REQUIRE: "1"
41 PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True"
42 HOME: "/cache"
43 VLLM_NO_USAGE_STATS: "1"
44 entrypoint: ["/patched-entrypoint.sh"]
45 deploy:
46 resources:
47 reservations:
48 devices:
49 - driver: nvidia
50 count: 1
51 capabilities: [gpu]
52 command:
53 - "--model"
54 - "/models/Qwen3.8-27B-W4A16-AutoRound"
55 - "--served-model-name"
56 - "qwen3.8-27b"
57 - "--host"
58 - "0.0.0.0"
59 - "--port"
60 - "8000"
61 - "--max-model-len"
62 - "262144"
63 - "--gpu-memory-utilization"
64 - "0.97"
65 # MUST be explicit. left unset, vllm auto-sizes kv to the whole budget and
66 # then OOMs during cuda graph capture (it asks for 800mb it does not have)
67 # and restart-loops, recompiling each time. weights are now ~15.1gb after
68 # the lm_head/embed/mtp requant, +1.8gb peak activation, so ~5gb is what
69 # is actually free for kv once graphs are paid for.
70 # 4.5gb not 5gb: at 5gb the pool is 300k tokens but a ~250k-token request
71 # has no room left for transient GDN state + activations and the engine
72 # dies with a 24mb OOM. 4.5gb still yields >262k tokens of pool and keeps
73 # ~0.5gb of transient headroom.
74 - "--kv-cache-memory"
75 - "4831838208"
76 # KVarN sizes an fp16 "tail pool" from max_num_seqs -- it capped 256->233
77 # on its own and still OOM'd. we are single-user, so 8 concurrent slots is
78 # plenty and it frees several gb of tail pool for actual kv capacity.
79 # 8, not 16: KVarN sizes an fp16 tail pool from max_num_seqs, so doubling
80 # this OOMs on top of the larger step budget. 8 concurrent streams is
81 # ample for an agent client (main call + classifier + a couple of tools).
82 - "--max-num-seqs"
83 - "8"
84 - "--compilation-config"
85 - '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
86 # KVarN: 4-bit keys / 2-bit values, ~4x more tokens per byte than fp8.
87 # this is what takes context from ~123k to the model's native 262k on the
88 # same 5gb pool. lossy -- verify with a needle test, not just a boot.
89 - "--kv-cache-dtype"
90 - "kvarn_k4v2_g128"
91 - "--mamba-cache-mode"
92 - "align"
93 - "--reasoning-parser"
94 - "qwen3"
95 - "--enable-auto-tool-choice"
96 - "--tool-call-parser"
97 - "qwen3_coder"
98 - "--enable-prefix-caching"
99 - "--speculative-config"
100 - '{"method":"mtp","num_speculative_tokens":3}'
101 healthcheck:
102 test: ["CMD-SHELL", "python3 -c \"import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health')\""]
103 interval: 30s
104 timeout: 10s
105 retries: 3
106 start_period: 20m
107
108networks:
109 home-infra:
110 external: true
111 name: home-infra_default
vllm/compose.yaml created+125
...@@ -0,0 +1,125 @@
1# qwen3.8-27b inference endpoint, single rtx 3090 (ampere, 24gb)
2#
3# separate compose project, but joins home-infra_default so caddy can reach it
4# at http://vllm:8000. run through ./up.sh -- interpolation vars come from the
5# parent .env, which only exists on the nas.
6name: vllm
7
8services:
9 vllm: # port 8000
10 container_name: vllm
11 # pinned: the patched stack we want to benchmark against targets 0.27.1
12 # exactly, so keeping stock on the same base keeps that an apples-to-apples
13 # comparison rather than a version diff.
14 #
15 # cu129, NOT the default tag: the default is built on cuda 13, which needs
16 # driver 580+. truenas 25.04 ships 550.142 (cuda 12.4), so cuda 13 is a
17 # hard no. cuda 12.9 runs on 550 via cuda minor-version compatibility.
18 image: vllm/vllm-openai:v0.27.1-cu129
19 restart: unless-stopped
20 # vllm's workers talk over shared memory; the default 64mb shm is not enough
21 ipc: host
22 # the cu129 image ships nvidia's forward-compat libcuda (575) under
23 # /usr/local/cuda/compat. that path is datacenter-only -- on a geforce card
24 # it fails with cuda error 804 ("forward compatibility attempted on non
25 # supported HW") before the engine ever loads. masking the dir with an
26 # empty tmpfs forces the real 550 driver to be used, which is what we want:
27 # cuda 12.9 on a 12.4 driver is covered by minor-version compatibility.
28 tmpfs:
29 - /usr/local/cuda/compat
30 networks:
31 - home-infra
32 volumes:
33 - "${APP_ROOT}/vllm/models:/models:ro"
34 - "${APP_ROOT}/vllm/cache:/root/.cache"
35 environment:
36 # weights are already on disk; never let it reach for the hub at boot
37 HF_HUB_OFFLINE: "1"
38 # bearer token for the openai-compatible api
39 VLLM_API_KEY: "${VLLM_API_KEY}"
40 # nvidia-container-cli gates on the image's declared cuda>=12.9 against
41 # the driver's reported 12.4 and refuses to start. the gate is stricter
42 # than reality -- minor-version compat covers this -- so bypass it.
43 # revisit if the truenas driver ever moves to 580+.
44 NVIDIA_DISABLE_REQUIRE: "1"
45 # weights nearly fill the card; reduce allocator fragmentation
46 PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True"
47 deploy:
48 resources:
49 reservations:
50 devices:
51 - driver: nvidia
52 count: 1
53 capabilities: [gpu]
54 # list form on purpose: --speculative-config takes json, and the folded
55 # string form would mangle the quoting.
56 command:
57 - "--model"
58 - "/models/Qwen3.8-27B-W4A16-AutoRound"
59 - "--served-model-name"
60 - "qwen3.8-27b"
61 - "--host"
62 - "0.0.0.0"
63 - "--port"
64 - "8000"
65 # 19.5gb of weights on a 24gb card leaves little for kv. start
66 # conservative so it boots, then walk this up while watching the kv pool
67 # size vllm prints at startup.
68 # lm_head was requantized to int8 (see readme), which freed ~1.3gb of
69 # weights and bought this jump. codex spends ~10k tokens on its system
70 # prompt + tool defs before any of your code, so headroom matters.
71 # raised to just under the 105,151-token kv pool. this costs no memory --
72 # max-model-len only caps a single request, so the only price is max
73 # concurrency dropping to ~1.0x, which is irrelevant for single-user use.
74 - "--max-model-len"
75 - "102400"
76 # nothing else shares this gpu, so take almost all of it
77 - "--gpu-memory-utilization"
78 - "0.97"
79 # cuda graphs matter enormously here: without them, kernel-launch
80 # overhead dominates decode and throughput drops to ~37 tok/s. rather
81 # than --enforce-eager, cap the kv cache and spend the freed vram on
82 # decode-only graph capture. note vllm silently downgrades this to
83 # PIECEWISE because flashinfer + spec-decode cannot do full decode
84 # graphs; piecewise still gets us most of the win.
85 - "--kv-cache-memory"
86 - "4831838208"
87 - "--compilation-config"
88 - '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
89 # text-only serving: do not reserve multimodal buffers. the vision tower
90 # weights still load, we just never budget for image/video inputs.
91 - "--limit-mm-per-prompt"
92 - '{"image":0,"video":0}'
93 # only 16 of 64 layers use full attention (the rest are linear), so the
94 # kv cache is far smaller than a normal 27b -- fp8 storage stretches it
95 # further at no measurable quality cost.
96 - "--kv-cache-dtype"
97 - "fp8"
98 # required by the qwen3.8 gated-deltanet / mtp serving path
99 - "--mamba-cache-mode"
100 - "align"
101 - "--reasoning-parser"
102 - "qwen3"
103 - "--enable-auto-tool-choice"
104 - "--tool-call-parser"
105 - "qwen3_coder"
106 - "--enable-prefix-caching"
107 # multi-token prediction: the single biggest speed lever available in
108 # stock vllm. ~46 tok/s -> ~114 tok/s single-stream.
109 - "--speculative-config"
110 - '{"method":"mtp","num_speculative_tokens":3}'
111 healthcheck:
112 test: ["CMD-SHELL", "python3 -c \"import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health')\""]
113 interval: 30s
114 timeout: 10s
115 retries: 3
116 # weight load + cuda graph capture is slow on first boot
117 start_period: 15m
118 labels:
119 net.paperclover.list.name: Qwen3.8 27B
120 net.paperclover.list.web: "false"
121
122networks:
123 home-infra:
124 external: true
125 name: home-infra_default
vllm/patched-entrypoint.sh created+36
...@@ -0,0 +1,36 @@
1#!/bin/bash
2# Runs the syv-ai patch stack against the official vllm image's site-packages at
3# container start, then serves. This exists because `docker build` is not
4# available to us here -- patching in the entrypoint gets the same result without
5# baking an image. Patch application is idempotent and takes seconds; the slow
6# parts (FlashInfer JIT, torch.compile) cache into /cache.
7#
8# IMPORTANT: this runs against the -cu129 image, NOT the cuda-13 one the repo's
9# Dockerfile builds. All 13 patches are pure-python patches against vllm 0.27.1
10# and apply cleanly to the cu129 build, so the whole patch stack works on
11# driver 550 -- no truenas upgrade needed. The repo's cuda-13 requirement comes
12# from `pip install vllm` pulling torch+cu130 in a fresh build, not from the
13# patches themselves.
14set -e
15STACK=${STACK_DIR:-/stack}
16SP=$(python3 -c 'import vllm, os; print(os.path.dirname(vllm.__file__))' | tail -n1)
17echo "vllm site-packages: $SP"
18
19if [ ! -f "$SP/.syvai-patched" ]; then
20 echo "== applying $(ls "$STACK"/patches/*.patch | wc -l) patches"
21 for p in "$STACK"/patches/*.patch; do
22 echo "-- $(basename "$p")"
23 patch -p1 -d "$SP" < "$p"
24 done
25 echo "== kvarn/install.sh"
26 # PY override: the script defaults to $REPO/venv/bin/python (their venv
27 # layout); in the official image vllm lives in the system interpreter.
28 # /stack is read-only, so copy out first -- install.sh writes into its own dir.
29 cp -r "$STACK" /stack-rw
30 ( cd /stack-rw && PY=python3 bash kvarn/install.sh )
31 touch "$SP/.syvai-patched"
32else
33 echo "== patches already applied"
34fi
35
36exec python3 -m vllm.entrypoints.openai.api_server "$@"
vllm/readme.md created+275
...@@ -0,0 +1,275 @@
1# qwen3.8-27b on the 3090
2
3openai-compatible inference endpoint, stock vllm, served at
4`https://ai.{$HOME_DOMAIN}` and as `http://vllm:8000` to other containers.
5
6auth is vllm's own bearer token (`VLLM_API_KEY` in the nas `.env`). the
7`reverse_proxy_auth` caddy snippet is deliberately NOT used here: it does an
8oauth2 browser redirect, which api clients cannot follow.
9
10## why this model / quant
11
12- `dbirks/Qwen3.8-27B-W4A16-AutoRound`, ~19.5gb on disk, in `${APP_ROOT}/vllm/models`.
13- **4-bit, not 5-bit.** 28b params against a hard 24gb wall: 5-bit weights land
14 near 23gb and leave nothing for kv. the autoround 4-bit measures within the
15 confidence interval of bf16 on gsm8k / humaneval / mmlu-pro, so 5-bit has no
16 accuracy left to buy back.
17- **no fp8 / nvfp4.** the 3090 is ampere: no fp8 tensor cores (ada+), no fp4
18 (blackwell), and 8-bit does not fit regardless.
19
20## three ampere/truenas gotchas, all load-bearing
21
221. **cuda 13 is impossible here.** the default `vllm/vllm-openai` tag is built
23 on cuda 13, which needs driver 580+. truenas 25.04 ships 550.142 (cuda 12.4).
24 hence the `-cu129` tag: cuda 12.9 runs on a 12.4 driver under minor-version
25 compatibility.
262. **`NVIDIA_DISABLE_REQUIRE=1`.** nvidia-container-cli hard-gates on the
27 image's declared `cuda>=12.9` vs the driver's reported 12.4 and refuses to
28 start. the gate is stricter than reality.
293. **the compat tmpfs mask.** the cu129 image ships nvidia's forward-compat
30 libcuda (575) at `/usr/local/cuda/compat`. that path is datacenter-only; on
31 a geforce card it dies with cuda error 804 before the engine loads. masking
32 the dir with an empty tmpfs forces the real 550 driver to be used.
33
34## measured
35
36| config | single-stream | context |
37|---|---|---|
38| `--enforce-eager` (no cuda graphs) | 37 tok/s | 16k |
39| piecewise cuda graphs | 51-61 tok/s | 49k |
40| + int8 lm_head | 63-73 tok/s | 82k |
41| + max-model-len to pool limit | 71 tok/s | 102k |
42| + patch stack, int8 embed/mtp | 70.6 tok/s | 123k |
43| + mtp draft vocab | 73.1 tok/s | 123k |
44| **+ KVarN 4/2-bit kv (current)** | **89.8 tok/s** | **262k** |
45
46that is the model's full native context on one 3090, at ~2.2x the throughput we
47started with and ~2.2x claude opus's ~40 tok/s.
48
49**KVarN made it faster, not slower.** 4-bit keys / 2-bit values means far less
50memory traffic per attention step, and decode is bandwidth-bound -- so the
51lossy cache buys speed *and* context at once. that was not the expected result.
52
53verification (KVarN is lossy, so booting proves nothing):
54
55| check | result |
56|---|---|
57| needle @ 50,917 tok | found |
58| needle @ 151,381 tok | found |
59| needle @ 248,234 tok | found, engine survived |
60| 6-item reasoning battery | 5/6 no-think, 6/6 with thinking |
61
62the one no-think miss ("gives away a third, then eats 2") answers correctly
63with `enable_thinking` on, so it is a sampling artifact, not kv corruption.
64
65prefill: ~900 tok/s up to 150k, dropping to ~620 tok/s at 250k. a 250k prompt
66costs ~400s of prefill -- prefix caching is what makes long sessions usable.
67
68### the int8 lm_head step
69
70the published quant leaves lm_head in bf16: a 2.37 GiB matrix over a 248k
71vocab, re-read every single decode step. requantizing it to int8 group-128
72(`prepare/quant_lm_head.py` from the syv-ai repo, cpu-only, in place, keeps
73`.bak`) freed 1.3 gb and bought ~20% throughput -- more than the +12% the repo
74documents. round-trip error 0.64%; spot checks stayed correct.
75
76**it loads on stock vllm.** that was not obvious: the repo's docs/optimizations
77table marks every optimization "requires patch", but that is wrong for this
78one. it is plain compressed-tensors pack-quantized output.
79
80a zfs snapshot `storage1/apps@pre-requant` predates the change.
81
82memory at rest: ~16.5gb weights, ~1.8gb peak activation, 4.5gb kv cache
83(105,151 tokens) at max-model-len 81920 -> 1.28x concurrency.
84
85checkpoint breakdown (why there is still headroom), by safetensors header:
86
87| family | GiB | dtype |
88|---|---|---|
89| LM body | 11.73 | int4, already quantized |
90| embeddings | 2.37 | bf16 -- still unquantized |
91| lm_head | 2.37 -> 1.19 | now int8 |
92| vision tower | 0.86 | bf16, loaded but unused (text-only serving) |
93| MTP module | 0.79 | bf16 -- still unquantized |
94
95## running
96
97 ./up.sh up -d
98 ./up.sh logs -f # first boot is slow: weight load + graph capture
99 ./up.sh down
100
101## two profiles: pick by workload
102
103on 24gb, max context and robust concurrency are **mutually exclusive**. the
104long-context profile sits at 23.6/24.5gb with ONE stream; a second concurrent
105request has no activation headroom and the engine dies with a CUDA OOM. that is
106measured, not theorised -- an 8-stream ladder killed it.
107
108| | `compose.patched.yaml` | `compose.agent.yaml` |
109|---|---|---|
110| context | 262,144 | 131,072 |
111| single stream | ~90 tok/s | ~83-86 tok/s |
112| 8 streams | dies | 44.5 tok/s each, 237 aggregate |
113| gpu-mem-util | 0.97 | 0.90 (the headroom IS the feature) |
114| use for | one-shot long analysis | claude code, anything concurrent |
115
116 ./up.sh -f compose.agent.yaml up -d # multi-stream (default choice)
117 ./up.sh -f compose.patched.yaml up -d # 262k single-stream
118
119### concurrency ladder (agent profile, measured)
120
121| streams | per-stream | aggregate |
122|---|---|---|
123| 1 | 82.5 | 82.5 |
124| 2 | 49.3 | 98.7 |
125| 4 | 58.9 | 233.0 |
126| 8 | 44.5 | 236.8 |
127
128per-stream stays above claude opus's ~40 tok/s even at 8 concurrent.
129
130### how vllm actually scales
131
132continuous batching: every scheduler step the engine picks up to
133`max_num_batched_tokens` tokens across *all* in-flight requests, so streams
134share the gpu rather than queueing. kv is paged and allocated on demand, so the
135"Maximum concurrency 1.14x" line is a worst-case figure (all requests at full
136length), NOT an admission limit -- short requests coexist fine. the real caps
137are `max_num_seqs` (8) and free kv blocks.
138
139`num_speculative_tokens_per_batch_size` gives "fast alone, scales under load":
140`[[1,1,3],[2,2,2],[3,4,1],[5,256,0]]` = full 3-token drafting at batch 1,
141tapering to none past 4 streams, where rejected drafts are just wasted compute.
142
143### the starvation trap
144
145at the auto-chosen `max_num_batched_tokens=2048`, one long prefill consumes the
146entire step budget and a concurrent small request advances ~1 token per slow
147step. a 0.44s request became **7.27s** behind an 18k-token prefill -- which is
148why agent classifier calls time out. `--max-num-batched-tokens 4096` plus
149`--long-prefill-token-threshold 1024` caps how much of a step any single
150prefill may take, bringing that to 3.06s.
151
152### max-model-len counts OUTPUT too
153
154`--max-model-len` bounds prompt + `max_tokens`, not the prompt alone. claude
155code sends a fixed `max_tokens=32768`, so at 98304 the usable prompt was only
15665536 -- and it failed on a 65,537-token prompt by exactly one token. at 131072
157the usable prompt is 98304. if you want more, lower the client's output
158reservation (`CLAUDE_CODE_MAX_OUTPUT_TOKENS`) rather than raising vram.
159
160claude code talks to `/v1/messages` (the anthropic API), which vllm serves
161alongside the openai routes.
162
163## using it from claude code
164
165`~/Desktop/claude-qwen` sets `ANTHROPIC_BASE_URL` at this host and points every
166model alias at `qwen3.8-27b`. two things it must get right:
167
168- `CLAUDE_CODE_MAX_CONTEXT_TOKENS` / `CLAUDE_CODE_MAX_OUTPUT_TOKENS` must match
169 the running profile. these are 131072 / 8192, giving 122,880 usable prompt.
170- claude code talks the anthropic protocol to `/v1/messages`, which vllm serves.
171
172two gotchas, both fixed server-side:
173
1741. **`x-api-key` vs bearer.** vllm's `--api-key` only accepts
175 `Authorization: Bearer`; anthropic-protocol clients send `x-api-key` and got
176 a flat 401. the caddy vhost now translates `x-api-key` into a bearer header,
177 so both conventions work against the same key. unauthenticated still 401s.
1782. **thinking defaulted to xhigh and returned EMPTY answers.** the chat
179 template defaults `reasoning_effort` to xhigh when the client sends nothing,
180 and claude code sends nothing. at xhigh a hard prompt burns >8k tokens inside
181 the `<think>` block, hits max_tokens before emitting `</think>`, and comes
182 back as a thinking block with **no text block at all**. measured at 8192
183 output: xhigh -> unusable, medium and low -> clean answers. the agent profile
184 now sets `--default-chat-template-kwargs '{"reasoning_effort":"medium"}'`.
185
186## using it from codex
187
188`~/.codex/config.toml` holds the provider, `~/.codex/local.config.toml` the
189profile (codex 0.149 split these; a `[profiles.x]` table in config.toml is now
190rejected). run with `codex --profile local`.
191
192 export VLLM_API_KEY=... # same value as the nas .env
193 codex --profile local
194
195three gotchas:
196- `wire_api = "responses"` is mandatory -- codex dropped chat-completions in
197 feb 2026. vllm 0.27.1 does serve `/v1/responses`, so this works.
198- `codex exec` reads stdin by default and will hang forever looking like a
199 model problem. redirect it: `codex exec ... < /dev/null`.
200- codex burns ~10k tokens on its system prompt and tool definitions before any
201 of your code, which is why max-model-len is 48k rather than 32k.
202
203`--enable-auto-tool-choice --tool-call-parser qwen3_coder` are what make the
204agent loop work; without them codex can read but never act.
205
206## tuning levers
207
2081. `--kv-cache-memory` (currently 3.2gb) trades context against cuda graph
209 memory. dropping graphs entirely costs ~40% throughput, so do not.
2102. `--max-model-len` is 49152 against a 65,179-token pool. the model natively
211 supports 262k; getting anywhere near that needs vram freed elsewhere.
2123. `num_speculative_tokens` is 3, the measured default. read
213 `vllm:spec_decode_num_{accepted,draft}_tokens_total` from /metrics before
214 changing it -- throughput alone cannot tell a working drafter from one that
215 loaded and got ignored.
216
217## the patched stack does NOT need a truenas upgrade
218
219this was the session's biggest wrong turn, so it is worth stating plainly. the
220repo's Dockerfile is `FROM nvidia/cuda:13.0.1` and `vllm==0.27.1` pulls torch
2212.13+cu130, so it looks like the patch stack requires driver 580+ and therefore
222a truenas 25.10 upgrade. it does not. **all 13 patches are pure-python patches
223against vllm 0.27.1 and apply cleanly, in sequence, to the `-cu129` image** --
224verified 13/13 with vllm still importing. the cuda 13 dependency comes from
225building vllm from scratch, not from the patches.
226
227so `compose.patched.yaml` runs the entire patch stack on driver 550.
228`patched-entrypoint.sh` applies the patches to site-packages at container start
229(we have no `docker build` here), which costs a few seconds and is idempotent.
230
231a truenas 25.10 upgrade is still worth doing eventually -- it ships driver
232580.173.02 / cuda 13, which would let us drop `NVIDIA_DISABLE_REQUIRE` and the
233`/usr/local/cuda/compat` tmpfs mask -- but it buys no capability we do not
234already have.
235
236## IMPORTANT: the checkpoint is now patched-stack-only
237
238`quant_embed.py` wrote packed int8 embeddings, which stock vllm cannot load
239(that is what `patches/qwen3_5-embed-quant.patch` exists for). **`compose.yaml`
240will no longer start against this model dir.** to go back to stock, restore
241`storage1/apps@pre-embed-quant` or the `.bak` files the prepare scripts left.
242
243zfs snapshots, oldest first: `@pre-requant` (before any requant),
244`@pre-embed-quant`, `@pre-draftvocab`.
245
246## two settings that are load-bearing and non-obvious
247
248- **`--kv-cache-memory` must be explicit.** left unset, vllm auto-sizes kv to
249 the whole budget, then OOMs during cuda graph capture and restart-loops,
250 recompiling every time. it is set to 4.5gb, not 5gb: at 5gb the pool is 300k
251 tokens but a ~250k request has no room for transient GDN state and the engine
252 dies on a 24mb allocation.
253- **`--max-num-seqs 8`.** KVarN sizes an fp16 "tail pool" from max_num_seqs; at
254 the default it capped itself to 233 and still OOM'd. we are single-user, so 8
255 slots frees gigabytes for actual kv capacity.
256
257## what is still on the table
258
259- **DFlash2 drafter** (+3-10% throughput) is a real trade, not free: the
260 quantized drafter is ~1.19gb, and at ~60k pool tokens per gb that costs
261 ~72k tokens of context -- dropping max-model-len from 262k to roughly 195k.
262 262,144 is the model's *native* context, so spending a quarter of it for ~5%
263 speed is probably the wrong side of the trade. left off deliberately.
264- **dropping the vision tower** (0.86gb, still bf16, never used since we serve
265 text-only) is the way to get DFlash2 *and* keep 262k. it means checkpoint
266 surgery on a Qwen3VL config, which is the riskiest remaining step.
267- **truenas 25.10** to drop the two driver-550 workarounds.
268
269## the gap to ~114 tok/s
270
271the syv-ai/qwen38-27b-rtx3090 stack reports 114-133 tok/s single-stream. the
272difference is not magic: it requantizes lm_head, embeddings and the mtp module
273to int4, which frees enough vram for full cuda graphs plus a 66.7k kv pool, and
274it patches vllm 0.27.1 for dflash2 block drafting. that is the benchmark
275target for the alternate stack -- same 0.27.1 base, so the comparison is clean.
vllm/up.sh created+6
...@@ -0,0 +1,6 @@
1#!/bin/sh
2# compose wrapper: this project lives in a subfolder but its interpolation
3# vars (APP_ROOT, VLLM_API_KEY) live in the parent .env, which is nas-only.
4set -e
5cd "$(dirname "$0")"
6exec sudo docker compose --env-file ../.env "$@"