| author | |
| committer | |
| log | 4166956deaeadc0fb5443e3475b4da5fd97b0a74 |
| tree | b58e908121dbc3cf5d3d74e32f7b96e15fc2577b |
| parent | c6666b14acb9d62bb70b34275156feb553e400f8 |
| signature | Signed by SSH key SHA256:xbd+BjjhyBfwk7GVoURf9Yx0gzDerHbvYv7SddNWmAs |
21 files changed, 1776 insertions(+), 135 deletions(-)
.gitignore+1| ... | ... | @@ -2,3 +2,4 @@ |
| 2 | 2 | /.apps |
| 3 | 3 | /certs |
| 4 | 4 | /sitegen |
| 5 | config/yt-upscaler/realesr-general-x4v3.onnx |
compose.yaml+65-15| ... | ... | @@ -117,6 +117,7 @@ services: |
| 117 | 117 | - "${STORE_ROOT}:/w" |
| 118 | 118 | - "${APP_ROOT}/copyparty:/cfg" |
| 119 | 119 | - ./config/copyparty.conf:/cfg/copyparty.conf:ro |
| 120 | - ./config/copyparty-hooks:/hooks:ro | |
| 120 | 121 | healthcheck: |
| 121 | 122 | test: ["CMD-SHELL", "wget -qO /dev/null http://127.0.0.1:80/?h"] |
| 122 | 123 | interval: 30s |
| ... | ... | @@ -152,7 +153,7 @@ services: |
| 152 | 153 | cpus: "16.0" |
| 153 | 154 | restart: unless-stopped |
| 154 | 155 | volumes: |
| 155 | - "$CLOVER_ROOT/Documents/Config/paperclover:/data:rw" | |
| 156 | - "$CLOVER_ROOT/Documents/Config/paper clover:/data:rw" | |
| 156 | 157 | # rw: the indexer scrubs exif location data in place |
| 157 | 158 | - "$CLOVER_ROOT/Published:/published:rw" |
| 158 | 159 | healthcheck: |
| ... | ... | @@ -246,7 +247,7 @@ services: |
| 246 | 247 | context: https://tangled.org/tangled.org/3lqs6zdi4nt22.git |
| 247 | 248 | dockerfile: Dockerfile |
| 248 | 249 | args: |
| 249 | TAG: "${KNOT_IMAGE_TAG:-v1.14.0-alpha}" | |
| 250 | TAG: "${KNOT_IMAGE_TAG:-v1.15.0-alpha}" | |
| 250 | 251 | UID: "${USER_ID:?}" |
| 251 | 252 | GID: "${GROUP_ID:?}" |
| 252 | 253 | pull_policy: build |
| ... | ... | @@ -285,7 +286,7 @@ services: |
| 285 | 286 | context: ./config/spindle |
| 286 | 287 | dockerfile: Dockerfile |
| 287 | 288 | args: |
| 288 | TAG: "${SPINDLE_IMAGE_TAG:-v1.14.0-alpha}" | |
| 289 | TAG: "${SPINDLE_IMAGE_TAG:-v1.15.0-alpha}" | |
| 289 | 290 | pull_policy: build |
| 290 | 291 | depends_on: |
| 291 | 292 | docker-in-docker: |
| ... | ... | @@ -401,7 +402,7 @@ services: |
| 401 | 402 | - ENABLE_SOCKS=no |
| 402 | 403 | - SOCKS_USER=admin |
| 403 | 404 | - SOCKS_PASS=socks |
| 404 | - LAN_NETWORK=192.168.86.0/24 | |
| 405 | - LAN_NETWORK=10.0.0.0/24 | |
| 405 | 406 | - NAME_SERVERS=84.200.69.80,37.235.1.174,1.1.1.1,37.235.1.177,84.200.70.40,1.0.0.1 |
| 406 | 407 | - VPN_INPUT_PORTS=1234 |
| 407 | 408 | - VPN_OUTPUT_PORTS=5678 |
| ... | ... | @@ -515,6 +516,9 @@ services: |
| 515 | 516 | HOME: /config |
| 516 | 517 | volumes: |
| 517 | 518 | - ./config/yt:/config-yt:ro |
| 519 | # channel list + ytdl-sub config live outside the repo so they can be | |
| 520 | # edited from the yt-feed web ui (Clover ▸ Documents/Config/Youtube Downloader) | |
| 521 | - "${CLOVER_ROOT}/Documents/Config/Youtube Downloader:/yt-config:ro" | |
| 518 | 522 | # the image hardcodes its lock file at /config, so state lives there |
| 519 | 523 | - "${APP_ROOT}/ytdl-sub:/config" |
| 520 | 524 | - "${MEDIA_ROOT}:/media" |
| ... | ... | @@ -540,8 +544,13 @@ services: |
| 540 | 544 | MAIL_FROM: "yt-feed@${HOME_DOMAIN:?}" |
| 541 | 545 | MAIL_TO: "${ADMIN_EMAIL:?}" |
| 542 | 546 | BASE_URL: "https://yt.${HOME_DOMAIN:?}" |
| 547 | # channel lists are edited here and read live; the editor writes to this dir | |
| 548 | FEED_CONFIG: "/yt-config/feed.yaml" | |
| 549 | YT_CONFIG_DIR: "/yt-config" | |
| 543 | 550 | volumes: |
| 544 | 551 | - ./config/yt:/config-yt:ro |
| 552 | # editable channel lists (subscriptions.yaml, feed.yaml) — rw for the web editor | |
| 553 | - "${CLOVER_ROOT}/Documents/Config/Youtube Downloader:/yt-config:rw" | |
| 545 | 554 | - "${APP_ROOT}/yt-feed:/data" |
| 546 | 555 | - "${MEDIA_ROOT}:/media" |
| 547 | 556 | restart: unless-stopped |
| ... | ... | @@ -550,6 +559,32 @@ services: |
| 550 | 559 | net.paperclover.list.domain: yt |
| 551 | 560 | net.paperclover.list.priority: 53 |
| 552 | 561 | net.paperclover.list.access: media-manage |
| 562 | # thumbnail super-resolution (issue #6): upscales Independent thumbnails 4x | |
| 563 | # with Real-ESRGAN (ONNX, cpu) so they stay sharp as jellyfin tv backdrops. | |
| 564 | # separate from yt-feed because onnxruntime has no wheels for the ytdl-sub | |
| 565 | # image's python 3.14; writes upscale-status.json into yt-feed's data dir. | |
| 566 | yt-upscaler: | |
| 567 | container_name: yt-upscaler | |
| 568 | build: | |
| 569 | context: config/yt-upscaler | |
| 570 | dockerfile: Dockerfile | |
| 571 | pull_policy: build | |
| 572 | user: "$USER_ID:$GROUP_ID" | |
| 573 | environment: | |
| 574 | INDEP_DIR: "/media/jellyfin/Independent" | |
| 575 | STATE_DIR: "/state" | |
| 576 | SR_THREADS: "6" | |
| 577 | volumes: | |
| 578 | - "${MEDIA_ROOT}:/media" | |
| 579 | - "${APP_ROOT}/yt-feed:/state" | |
| 580 | deploy: | |
| 581 | resources: | |
| 582 | limits: | |
| 583 | cpus: "8.0" | |
| 584 | restart: unless-stopped | |
| 585 | labels: | |
| 586 | net.paperclover.list.name: YouTube Thumbnail Upscaler | |
| 587 | net.paperclover.list.web: "false" | |
| 553 | 588 | # language models |
| 554 | 589 | opencode: # port 4096 |
| 555 | 590 | container_name: opencode |
| ... | ... | @@ -684,19 +719,24 @@ services: |
| 684 | 719 | dawarich-app: # port 30161 |
| 685 | 720 | image: freikin/dawarich:latest |
| 686 | 721 | container_name: dawarich-app |
| 722 | networks: | |
| 723 | default: | |
| 724 | aliases: | |
| 725 | - dawarich | |
| 687 | 726 | volumes: |
| 727 | - "${APP_ROOT}/dawarich_tmp:/var/app/tmp" | |
| 688 | 728 | - "${APP_ROOT}/dawarich_public:/var/app/public" |
| 689 | 729 | - "${APP_ROOT}/dawarich_watched:/var/app/tmp/imports/watched" |
| 690 | 730 | - "${APP_ROOT}/dawarich_storage:/var/app/storage" |
| 691 | 731 | - "${APP_ROOT}/dawarich_data:/dawarich_db_data" |
| 692 | entrypoint: web-entrypoint.sh | |
| 732 | - "${APP_ROOT}/keycloak:/shared-keys:ro" | |
| 733 | # wrap the stock web-entrypoint to inject the OIDC client secret that | |
| 734 | # keycloak's init.py wrote to /shared-keys/dawarich (same pattern as forgejo) | |
| 735 | entrypoint: ["/bin/sh", "-c"] | |
| 693 | 736 | command: |
| 694 | - bin/rails | |
| 695 | - server | |
| 696 | - -p | |
| 697 | - "30161" | |
| 698 | - -b | |
| 699 | - "::" | |
| 737 | - | | |
| 738 | export OIDC_CLIENT_SECRET="$$(cat /shared-keys/dawarich)" | |
| 739 | exec web-entrypoint.sh bin/rails server -p 30161 -b :: | |
| 700 | 740 | restart: on-failure |
| 701 | 741 | user: "$USER_ID:$GROUP_ID" |
| 702 | 742 | environment: |
| ... | ... | @@ -709,6 +749,9 @@ services: |
| 709 | 749 | DATABASE_NAME: dawarich |
| 710 | 750 | MIN_MINUTES_SPENT_IN_CITY: 60 |
| 711 | 751 | APPLICATION_HOSTS: "localhost,zenith,dawarich.${HOME_DOMAIN}" |
| 752 | OIDC_CLIENT_ID: dawarich | |
| 753 | OIDC_ISSUER: "https://auth.${HOME_DOMAIN}/realms/master" | |
| 754 | OIDC_REDIRECT_URI: "https://dawarich.${HOME_DOMAIN}/users/auth/openid_connect/callback" | |
| 712 | 755 | TIME_ZONE: America/Los_Angeles |
| 713 | 756 | APPLICATION_PROTOCOL: http |
| 714 | 757 | PROMETHEUS_EXPORTER_ENABLED: "false" |
| ... | ... | @@ -898,19 +941,26 @@ services: |
| 898 | 941 | labels: |
| 899 | 942 | net.paperclover.list.name: DDNS |
| 900 | 943 | net.paperclover.list.access: personal |
| 901 | shale: # port probably is 80 | |
| 902 | image: astheno/shale | |
| 944 | shale: # port 8000 | |
| 945 | image: astheno/shale@sha256:1f2144eb7a422871414fdc010f05cf99206f621cd68b71435e50da71ed6dbcde | |
| 903 | 946 | container_name: shale |
| 904 | 947 | user: "0:0" |
| 905 | 948 | volumes: |
| 906 | 949 | - "${APP_ROOT}/shale/repositories_mirrors:/repositories_mirrors" |
| 907 | 950 | - "${APP_ROOT}/shale/repositories_owned:/repositories_owned" |
| 908 | 951 | - "${APP_ROOT}/shale/data:/data" |
| 952 | restart: unless-stopped | |
| 909 | 953 | environment: |
| 910 | 954 | DOMAIN: "shale.${HOME_DOMAIN:?}" |
| 911 | OAUTH2_CLIENT: "snow sign on,https://auth.${HOME_DOMAIN:?}/realms/master|shale|${SHALE_CLIENT_SECRET:?}" | |
| 955 | OAUTH2_CLIENT: "oidc,auth.${HOME_DOMAIN:?}/realms/master|shale|${SHALE_CLIENT_SECRET:?}" | |
| 956 | SESSION_SECRET: "${SHALE_SESSION_SECRET:?}" | |
| 912 | 957 | SERVER_TITLE: "clover's git" |
| 913 | MIRROR_1: "sitegen,https://git.paperclover.net/clo/sitegen.git,website generator, standard library, and home of paperclover.net" | |
| 958 | MIRROR_1: "home-infra,https://git.paperclover.net/clo/home-infra.git,compose files and tasks for my home server" | |
| 959 | MIRROR_2: "discord-name-painter,https://git.paperclover.net/clo/discord-name-painter.git,maintainance only. allows any user to set their display name color to any color, by using dynamically created roles." | |
| 960 | MIRROR_3: "react-mutation,https://git.paperclover.net/clo/react-mutation.git,create async mutations with trivial optimistic updates and great error handling" | |
| 961 | MIRROR_4: "markodown,https://git.paperclover.net/clo/markodown.git,alternate universe where markdown lets you write marko components inline" | |
| 962 | MIRROR_5: "toolkit,https://git.paperclover.net/clo/toolkit.git,Clover's Creative Toolkit is collection of macOS software that I use to create." | |
| 963 | MIRROR_6: "react-markdown,https://git.paperclover.net/clo/react-markdown.git,memoized markdown renderer for react using the unified plugin ecosystem" | |
| 914 | 964 | # evil inc temporary infrastructure |
| 915 | 965 | evil-forgejo: # port 3000 |
| 916 | 966 | container_name: evil-forgejo |
config/Caddyfile+38-1| ... | ... | @@ -41,6 +41,30 @@ |
| 41 | 41 | } |
| 42 | 42 | |
| 43 | 43 | # services are sorted alphabetically |
| 44 | ai.{$HOME_DOMAIN} { | |
| 45 | 	# inference api (qwen3.8-27b on the 3090). vllm serves both the openai | |
| 46 | 	# routes and anthropic's /v1/messages, so this host answers both protocols. | |
| 47 | 	# | |
| 48 | 	# auth is vllm's own bearer token, NOT reverse_proxy_auth: that snippet does | |
| 49 | 	# an oauth2 browser redirect, which every api client would choke on. | |
| 50 | 	# | |
| 51 | 	# vllm only accepts `Authorization: Bearer <key>`. anthropic-protocol | |
| 52 | 	# clients (claude code) send `x-api-key: <key>` instead and get a 401. | |
| 53 | 	# translate it so both conventions work against the same key. | |
| 54 | 	@anthropic_auth header X-Api-Key * | |
| 55 | 	handle @anthropic_auth { | |
| 56 | 		reverse_proxy "http://vllm:8000" { | |
| 57 | 			header_up Authorization "Bearer {http.request.header.X-Api-Key}" | |
| 58 | 			# stream tokens as they are generated rather than buffering | |
| 59 | 			flush_interval -1 | |
| 60 | 		} | |
| 61 | 	} | |
| 62 | 	handle { | |
| 63 | 		reverse_proxy "http://vllm:8000" { | |
| 64 | 			flush_interval -1 | |
| 65 | 		} | |
| 66 | 	} | |
| 67 | } | |
| 44 | 68 | auth.{$HOME_DOMAIN} { |
| 45 | 69 | handle / { |
| 46 | 70 | redir / /apps |
| ... | ... | @@ -96,6 +120,8 @@ file.{$HOME_DOMAIN} { |
| 96 | 120 | 			header_up X-Forwarded-Uri {uri} |
| 97 | 121 | 		} |
| 98 | 122 | 	} |
| 123 | @share_root path /shr /shr/ | |
| 124 | respond @share_root 403 | |
| 99 | 125 | @should_auth not path /shr/* |
| 100 | 126 | reverse_proxy @should_auth "http://forward-auth" { |
| 101 | 127 | method GET |
| ... | ... | @@ -143,7 +169,7 @@ knot.{$HOME_DOMAIN} { |
| 143 | 169 | 	reverse_proxy "http://knot:5555" |
| 144 | 170 | } |
| 145 | 171 | shale.{$HOME_DOMAIN} { |
| 146 | 	reverse_proxy "http://shale" | |
| 172 | 	reverse_proxy "http://shale:8000" | |
| 147 | 173 | } |
| 148 | 174 | spindle.{$HOME_DOMAIN} { |
| 149 | 175 | 	reverse_proxy "http://spindle:6555" |
| ... | ... | @@ -195,6 +221,17 @@ music.{$HOME_DOMAIN} { |
| 195 | 221 | reverse_proxy "http://navidrome" |
| 196 | 222 | } |
| 197 | 223 | } |
| 224 | dav.{$HOME_DOMAIN} { | |
| 225 | 	# plain webdav endpoint for clients that can't do the sso redirect dance | |
| 226 | 	# (onenote 2007, rclone, finder) -- copyparty does its own auth here instead. | |
| 227 | 	# strip the idp headers so a client on this vhost can't claim to be an sso user. | |
| 228 | 	reverse_proxy "http://copyparty" { | |
| 229 | 		header_up -User-Id | |
| 230 | 		header_up -User-Groups | |
| 231 | 		header_up -User-Email | |
| 232 | 		header_up -User-Name | |
| 233 | 	} | |
| 234 | } | |
| 198 | 235 | opencode.{$HOME_DOMAIN} { |
| 199 | 236 | 	import reverse_proxy_auth "http://opencode:4096" admin |
| 200 | 237 | } |
config/copyparty-hooks/reject-apple-cruft.py created+43| ... | ... | @@ -0,0 +1,43 @@ |
| 1 | #!/usr/bin/env python3 | |
| 2 | ||
| 3 | import os | |
| 4 | import sys | |
| 5 | ||
| 6 | _ = r""" | |
| 7 | reject the metadata files macos scatters over network shares | |
| 8 | ||
| 9 | the samba setup vetoes these server-side; webdav has no equivalent knob, and | |
| 10 | DSDontWriteNetworkStores only suppresses .DS_Store -- the ._ AppleDouble files | |
| 11 | (resource forks / xattrs) keep coming regardless. finder also retries a failed | |
| 12 | write forever, so left alone these generate thousands of PUTs per file. | |
| 13 | ||
| 14 | enabled globally in copyparty.conf as: | |
| 15 | xbu: c,/hooks/reject-apple-cruft.py | |
| 16 | ||
| 17 | xbu = execute before upload | |
| 18 | c = check result, reject upload if error | |
| 19 | """ | |
| 20 | ||
| 21 | BAD_EXACT = { | |
| 22 | ".DS_Store", | |
| 23 | ".localized", | |
| 24 | ".Spotlight-V100", | |
| 25 | ".TemporaryItems", | |
| 26 | ".Trashes", | |
| 27 | ".fseventsd", | |
| 28 | ".apdisk", | |
| 29 | ".metadata_never_index", | |
| 30 | ".metadata_never_index_unless_rootfs", | |
| 31 | ".metadata_direct_scope_only", | |
| 32 | ".hidden", | |
| 33 | } | |
| 34 | ||
| 35 | ||
| 36 | def main(): | |
| 37 | name = os.path.basename(sys.argv[1]) | |
| 38 | bad = name.startswith("._") or name in BAD_EXACT | |
| 39 | sys.exit(1 if bad else 0) | |
| 40 | ||
| 41 | ||
| 42 | if __name__ == "__main__": | |
| 43 | main() |
config/copyparty.conf+3| ... | ... | @@ -9,6 +9,9 @@ |
| 9 | 9 | theme: 2 # monokai |
| 10 | 10 | name: clover's nas |
| 11 | 11 | stats, nos-dup # enable the prometheus endpoint, but disable the dupes counter (too slow) |
| 12 | dav-auth # webdav clients must always authenticate (windows gets confused otherwise) | |
| 13 | ah-alg: argon2 # passwords in accounts.conf are argon2 hashes, never plaintext | |
| 14 | xbu: c,/hooks/reject-apple-cruft.py # veto macos ._ / .DS_Store cruft (samba-style) | |
| 12 | 15 | |
| 13 | 16 | # keycloak |
| 14 | 17 | xff-src: lan # accept X-Forwarded-For from `lan` |
config/keycloak/init.py+33| ... | ... | @@ -194,6 +194,39 @@ def configure(): |
| 194 | 194 | client_data = kc_admin.get_client(client_id) |
| 195 | 195 | _ = Path("/shared/jellyfin").write_text(client_data["secret"]) |
| 196 | 196 | |
| 197 | # dawarich | |
| 198 | client_id = kc_admin.create_client( | |
| 199 | { | |
| 200 | "protocol": "openid-connect", | |
| 201 | "clientId": "dawarich", | |
| 202 | "name": "Dawarich", | |
| 203 | "description": "", | |
| 204 | "publicClient": False, | |
| 205 | "authorizationServicesEnabled": False, | |
| 206 | "serviceAccountsEnabled": False, | |
| 207 | "implicitFlowEnabled": False, | |
| 208 | "directAccessGrantsEnabled": False, | |
| 209 | "standardFlowEnabled": True, | |
| 210 | "frontchannelLogout": True, | |
| 211 | "attributes": { | |
| 212 | "saml_idp_initiated_sso_url_name": "", | |
| 213 | "standard.token.exchange.enabled": False, | |
| 214 | "oauth2.device.authorization.grant.enabled": False, | |
| 215 | "oidc.ciba.grant.enabled": False, | |
| 216 | "pkce.code.challenge.method": "", | |
| 217 | "dpop.bound.access.tokens": "false", | |
| 218 | "post.logout.redirect.uris": "*", | |
| 219 | }, | |
| 220 | "alwaysDisplayInConsole": False, | |
| 221 | "rootUrl": "", | |
| 222 | "baseUrl": "", | |
| 223 | "redirectUris": ["*"], | |
| 224 | }, | |
| 225 | skip_exists=True, | |
| 226 | ) | |
| 227 | client_data = kc_admin.get_client(client_id) | |
| 228 | _ = Path("/shared/dawarich").write_text(client_data["secret"]) | |
| 229 | ||
| 197 | 230 | # forward auth |
| 198 | 231 | client_id = kc_admin.create_client( |
| 199 | 232 | { |
config/spindle/Dockerfile+1-1| ... | ... | @@ -1,7 +1,7 @@ |
| 1 | 1 | FROM golang:1.25-alpine AS builder |
| 2 | 2 | ENV CGO_ENABLED=1 |
| 3 | 3 | |
| 4 | ARG TAG="v1.14.0-alpha" | |
| 4 | ARG TAG="v1.15.0-alpha" | |
| 5 | 5 | |
| 6 | 6 | WORKDIR /app |
| 7 | 7 | RUN apk add --no-cache git gcc musl-dev |
config/yt-feed/app.py+674-26| ... | ... | @@ -16,12 +16,15 @@ import urllib.request |
| 16 | 16 | import xml.etree.ElementTree as ET |
| 17 | 17 | from email.message import EmailMessage |
| 18 | 18 | from email.utils import formatdate |
| 19 | from xml.sax.saxutils import escape | |
| 19 | from xml.sax.saxutils import escape, unescape | |
| 20 | 20 | |
| 21 | 21 | import yaml |
| 22 | 22 | from flask import Flask, Response, redirect, request |
| 23 | 23 | |
| 24 | FEED_CONFIG = os.environ.get("FEED_CONFIG", "/config-yt/feed.yaml") | |
| 24 | FEED_CONFIG = os.environ.get("FEED_CONFIG", "/yt-config/feed.yaml") | |
| 25 | # the channel lists live outside the repo (Clover ▸ Documents/Config/Youtube | |
| 26 | # Downloader) so they can be edited from the web ui; this dir is mounted rw | |
| 27 | YT_CONFIG_DIR = os.environ.get("YT_CONFIG_DIR", "/yt-config") | |
| 25 | 28 | STATE_DIR = os.environ.get("STATE_DIR", "/data") |
| 26 | 29 | INTERVAL = int(os.environ.get("CHECK_INTERVAL", "1800")) |
| 27 | 30 | SMTP_HOST = os.environ["SMTP_HOST"] |
| ... | ... | @@ -125,6 +128,59 @@ def safe_name(name): |
| 125 | 128 | return re.sub(r'[/\\:*?"<>|]', "-", name).strip() or "untitled" |
| 126 | 129 | |
| 127 | 130 | |
| 131 | def looks_like_url(s): | |
| 132 | return bool(re.match(r"https?://", (s or "").strip(), re.I)) | |
| 133 | ||
| 134 | ||
| 135 | def stable_key(v): | |
| 136 | # the card's identity stays put across a stub resolving (id changes, | |
| 137 | # stub_id doesn't) so the page can refresh a card in place | |
| 138 | return v.get("stub_id") or v["id"] | |
| 139 | ||
| 140 | ||
| 141 | def safe_listdir(p): | |
| 142 | try: | |
| 143 | return os.listdir(p) | |
| 144 | except OSError: | |
| 145 | return [] | |
| 146 | ||
| 147 | ||
| 148 | def nfo_fields(path, *tags): | |
| 149 | try: | |
| 150 | with open(path) as f: | |
| 151 | txt = f.read() | |
| 152 | except OSError: | |
| 153 | return {t: "" for t in tags} | |
| 154 | out = {} | |
| 155 | for t in tags: | |
| 156 | m = re.search(rf"<{t}>(.*?)</{t}>", txt, re.S) | |
| 157 | out[t] = unescape(m.group(1)).strip() if m else "" | |
| 158 | return out | |
| 159 | ||
| 160 | ||
| 161 | def under(base, path): | |
| 162 | base = os.path.realpath(base) | |
| 163 | path = os.path.realpath(path) | |
| 164 | return path == base or path.startswith(base + os.sep) | |
| 165 | ||
| 166 | ||
| 167 | def set_nfo_tags(path, updates): | |
| 168 | # rewrite individual <tag> values in an existing nfo, leaving the rest as-is | |
| 169 | try: | |
| 170 | with open(path) as f: | |
| 171 | txt = f.read() | |
| 172 | except OSError: | |
| 173 | return | |
| 174 | for k, v in updates.items(): | |
| 175 | new = f"<{k}>{escape(str(v))}</{k}>" | |
| 176 | if re.search(rf"<{k}>.*?</{k}>", txt, re.S): | |
| 177 | txt = re.sub(rf"<{k}>.*?</{k}>", new, txt, count=1, flags=re.S) | |
| 178 | else: | |
| 179 | txt = re.sub(r"(\n</[A-Za-z]+>\s*)\Z", f"\n {new}\\1", txt) | |
| 180 | with open(path, "w") as f: | |
| 181 | f.write(txt) | |
| 182 | ||
| 183 | ||
| 128 | 184 | # ---------------------------------------------------------------- feed poller |
| 129 | 185 | |
| 130 | 186 | def channel_id_for(url, cache): |
| ... | ... | @@ -422,7 +478,7 @@ def indie_shows(): |
| 422 | 478 | PAGE = """<!doctype html> |
| 423 | 479 | <html lang="en"><head> |
| 424 | 480 | <meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1"> |
| 425 | <title>yt triage</title> | |
| 481 | <title>yt downloader</title> | |
| 426 | 482 | <style> |
| 427 | 483 | :root {{ color-scheme: light dark; }} |
| 428 | 484 | body {{ font-family: system-ui, sans-serif; max-width: 32rem; margin: 0 auto; |
| ... | ... | @@ -447,8 +503,11 @@ button.skip {{ background: transparent; color: light-dark(#555, #aac); }} |
| 447 | 503 | .done {{ color: light-dark(#2a7a2a, #8fd48f); }} |
| 448 | 504 | .error {{ color: light-dark(#b03030, #f0a0a0); }} |
| 449 | 505 | </style></head><body> |
| 450 | <h1>yt triage <small style="font-weight:400">· {npending} pending</small></h1> | |
| 451 | {cards} | |
| 506 | <h1>yt downloader <small style="font-weight:400">· {npending} pending</small> | |
| 507 | <span style="float:right;font-size:.85rem;font-weight:400"> | |
| 508 | <a href="/library">library ▸</a> · <a href="/config">channels ▸</a></span></h1> | |
| 509 | {banner} | |
| 510 | <div id="cards">{cards}</div> | |
| 452 | 511 | <form method="post" action="/add" class="card row"> |
| 453 | 512 | <input name="url" placeholder="paste youtube url(s)…" style="flex:3"> |
| 454 | 513 | <button>add</button> |
| ... | ... | @@ -474,11 +533,35 @@ function seasonOptions(f, show) {{ |
| 474 | 533 | sel.onchange = () => f.querySelector("input[name=episode]").value = (seasons[sel.value] || 0) + 1; |
| 475 | 534 | sel.onchange(); |
| 476 | 535 | }} |
| 477 | document.querySelectorAll("select[name=dest]").forEach(destChanged); | |
| 478 | setInterval(async () => {{ | |
| 479 | const r = await fetch("/jobs.html"); | |
| 480 | document.getElementById("jobs").innerHTML = await r.text(); | |
| 481 | }}, 4000); | |
| 536 | function initCard(card) {{ const s = card.querySelector("select[name=dest]"); if (s) destChanged(s); }} | |
| 537 | document.querySelectorAll("#cards .card").forEach(initCard); | |
| 538 | // any real edit "dirties" a card so the background refresh won't clobber it | |
| 539 | const cards = document.getElementById("cards"); | |
| 540 | for (const ev of ["input", "change"]) cards.addEventListener(ev, e => {{ | |
| 541 | const f = e.target.closest(".card"); if (f) f.dataset.dirty = "1"; | |
| 542 | }}, true); | |
| 543 | let busy = false; | |
| 544 | async function refresh() {{ | |
| 545 | try {{ | |
| 546 | document.getElementById("jobs").innerHTML = await (await fetch("/jobs.html")).text(); | |
| 547 | }} catch (e) {{}} | |
| 548 | if (busy) return; busy = true; | |
| 549 | try {{ | |
| 550 | const tmp = document.createElement("div"); | |
| 551 | tmp.innerHTML = await (await fetch("/cards.html")).text(); | |
| 552 | const fresh = {{}}; | |
| 553 | tmp.querySelectorAll(".card[data-key]").forEach(c => fresh[c.dataset.key] = c); | |
| 554 | cards.querySelectorAll(".card[data-key]").forEach(c => {{ | |
| 555 | if (!fresh[c.dataset.key] && c.dataset.dirty !== "1") c.remove(); | |
| 556 | }}); | |
| 557 | for (const key in fresh) {{ | |
| 558 | const have = cards.querySelector('.card[data-key="' + key + '"]'); | |
| 559 | if (!have) {{ cards.appendChild(fresh[key]); initCard(fresh[key]); }} | |
| 560 | else if (have.dataset.dirty !== "1") {{ have.replaceWith(fresh[key]); initCard(fresh[key]); }} | |
| 561 | }} | |
| 562 | }} catch (e) {{}} finally {{ busy = false; }} | |
| 563 | }} | |
| 564 | setInterval(refresh, 4000); | |
| 482 | 565 | </script> |
| 483 | 566 | </body></html>""" |
| 484 | 567 | |
| ... | ... | @@ -489,18 +572,19 @@ def card_html(v, shows): |
| 489 | 572 | for s in shows: |
| 490 | 573 | opts.append(f'<option value="indie|{escape(s)}">Indie Shows ▸ {escape(s)}</option>') |
| 491 | 574 | opts.append('<option value="indie|__new__">Indie Shows ▸ new show…</option>') |
| 575 | key = stable_key(v) | |
| 492 | 576 | img = (f'<a href="{escape(v["link"])}"><img src="{escape(v["thumb"])}" alt=""></a>' |
| 493 | 577 | if v.get("thumb") else "") |
| 494 | 578 | meta = ("resolving…" if v.get("unresolved") |
| 495 | 579 | else f"{escape(v.get('channel', '?'))} · {escape(v.get('published', ''))}") |
| 496 | return f"""<form method="post" action="/ingest" class="card" id="v-{v['id']}"> | |
| 580 | return f"""<form method="post" action="/ingest" class="card" id="v-{escape(key)}" data-key="{escape(key)}" data-dirty="0"> | |
| 497 | 581 | {img} |
| 498 | 582 | <p class="title">{escape(v['title'])}</p> |
| 499 | 583 | <p class="meta">{meta}</p> |
| 500 | <input type="hidden" name="vid" value="{escape(v['id'])}"> | |
| 584 | <input type="hidden" name="vid" value="{escape(key)}"> | |
| 501 | 585 | <select name="dest" onchange="destChanged(this)">{''.join(opts)}</select> |
| 502 | 586 | <input class="new-show" name="new_show" placeholder="new show name" style="display:none"> |
| 503 | <input class="indie-fields" name="ep_title" value="{escape(v['title'])}" placeholder="episode title"> | |
| 587 | <input class="indie-fields" name="ep_title" placeholder="{escape(v['title'])}"> | |
| 504 | 588 | <div class="indie-fields row"> |
| 505 | 589 | <select name="season"></select> |
| 506 | 590 | <input name="episode" type="number" min="1" title="episode #"> |
| ... | ... | @@ -530,20 +614,31 @@ def jobs_html(): |
| 530 | 614 | return "\n".join(out) |
| 531 | 615 | |
| 532 | 616 | |
| 617 | def render_cards(pending, shows): | |
| 618 | if not pending: | |
| 619 | return "<p style='opacity:.6'>nothing pending. enjoy the silence.</p>" | |
| 620 | return "\n".join(card_html(v, shows) for v in reversed(list(pending.values()))) | |
| 621 | ||
| 622 | ||
| 533 | 623 | @app.get("/") |
| 534 | 624 | def index(): |
| 535 | 625 | with state_lock: |
| 536 | 626 | pending = load_json("pending.json", {}) |
| 537 | 627 | shows = indie_shows() |
| 538 | cards = "\n".join(card_html(v, shows) for v in reversed(list(pending.values()))) | |
| 539 | if not pending: | |
| 540 | cards = "<p style='opacity:.6'>nothing pending. enjoy the silence.</p>" | |
| 541 | if wall_active(): | |
| 542 | cards = ("<div class='card error'>youtube has bot-walled this ip — downloads " | |
| 543 | "are paused and will resume automatically once the wall lifts " | |
| 544 | "(probed every few hours). queueing still works.</div>" + cards) | |
| 545 | return PAGE.format(npending=len(pending), cards=cards, | |
| 546 | jobs=jobs_html(), shows_json=json.dumps(indie_shows())) | |
| 628 | banner = ("<div class='card error'>youtube has bot-walled this ip — downloads " | |
| 629 | "are paused and will resume automatically once the wall lifts " | |
| 630 | "(probed every few hours). queueing still works.</div>" | |
| 631 | if wall_active() else "") | |
| 632 | return PAGE.format(npending=len(pending), banner=banner, | |
| 633 | cards=render_cards(pending, shows), | |
| 634 | jobs=jobs_html(), shows_json=json.dumps(shows)) | |
| 635 | ||
| 636 | ||
| 637 | @app.get("/cards.html") | |
| 638 | def cards_partial(): | |
| 639 | with state_lock: | |
| 640 | pending = load_json("pending.json", {}) | |
| 641 | return Response(render_cards(pending, indie_shows()), mimetype="text/html") | |
| 547 | 642 | |
| 548 | 643 | |
| 549 | 644 | @app.get("/jobs.html") |
| ... | ... | @@ -567,9 +662,14 @@ def retry(): |
| 567 | 662 | |
| 568 | 663 | @app.post("/skip") |
| 569 | 664 | def skip(): |
| 665 | vid = request.form["vid"] | |
| 570 | 666 | with state_lock: |
| 571 | 667 | pending = load_json("pending.json", {}) |
| 572 | pending.pop(request.form["vid"], None) | |
| 668 | # a manually-added stub gets re-keyed to the real video id once it | |
| 669 | # resolves, but the card still submits the stub_id — match on either | |
| 670 | if vid not in pending: | |
| 671 | vid = next((k for k, p in pending.items() if p.get("stub_id") == vid), vid) | |
| 672 | pending.pop(vid, None) | |
| 573 | 673 | save_json("pending.json", pending) |
| 574 | 674 | return redirect("/") |
| 575 | 675 | |
| ... | ... | @@ -613,10 +713,15 @@ def ingest(): |
| 613 | 713 | if not show: |
| 614 | 714 | return Response("missing show name", 400, mimetype="text/plain") |
| 615 | 715 | # a custom episode title; left equal to the card title means "use the |
| 616 | # video title" (which, for a still-resolving stub, arrives later) | |
| 716 | # video title" (which, for a still-resolving stub, arrives later). guard | |
| 717 | # against a url leaking in: the field is prefilled with the stub title, | |
| 718 | # which for an unresolved paste IS the url — if the stub then resolves | |
| 719 | # before submit, that url would otherwise be taken as a real title. | |
| 617 | 720 | ep_title = request.form.get("ep_title", "").strip() |
| 618 | job.update(dest="indie", show=show, | |
| 619 | ep_title=ep_title if ep_title and ep_title != v["title"] else "", | |
| 721 | if not ep_title or ep_title == v["title"] or ep_title == v.get("link") \ | |
| 722 | or looks_like_url(ep_title): | |
| 723 | ep_title = "" | |
| 724 | job.update(dest="indie", show=show, ep_title=ep_title, | |
| 620 | 725 | season=int(request.form["season"]), episode=int(request.form["episode"])) |
| 621 | 726 | job["dest_label"] = f"{show} S{job['season']:02d}E{job['episode']:02d}" |
| 622 | 727 | elif dest == "independent": |
| ... | ... | @@ -633,6 +738,549 @@ def ingest(): |
| 633 | 738 | return redirect("/") |
| 634 | 739 | |
| 635 | 740 | |
| 741 | # ---------------------------------------------------------------- library edit | |
| 742 | ||
| 743 | def library_entries(): | |
| 744 | # every ingested video across the two managed libraries, with its current | |
| 745 | # title (from the nfo, falling back to the filename) for search + rename | |
| 746 | out = [] | |
| 747 | for show in sorted(safe_listdir(INDIE_DIR)): | |
| 748 | show_path = os.path.join(INDIE_DIR, show) | |
| 749 | if not os.path.isdir(show_path): | |
| 750 | continue | |
| 751 | for sub in sorted(safe_listdir(show_path)): | |
| 752 | sm = re.fullmatch(r"Season (\d+)", sub) | |
| 753 | if not sm: | |
| 754 | continue | |
| 755 | sdir = os.path.join(show_path, sub) | |
| 756 | for f in sorted(safe_listdir(sdir)): | |
| 757 | if not f.lower().endswith(VIDEO_EXTS): | |
| 758 | continue | |
| 759 | base, _ = os.path.splitext(f) | |
| 760 | em = re.match(r"S(\d+)E(\d+) - (.*)", base) | |
| 761 | season, episode = int(sm.group(1)), int(em.group(2)) if em else 1 | |
| 762 | nf = nfo_fields(os.path.join(sdir, base + ".nfo"), "title", "plot") | |
| 763 | title = nf["title"] or (em.group(3) if em else base) | |
| 764 | out.append({ | |
| 765 | "type": "indie", "rel": os.path.relpath(os.path.join(sdir, f), "/media"), | |
| 766 | "ctx": f"{show} · S{season}E{episode}", "title": title, | |
| 767 | "link": nf["plot"] if looks_like_url(nf["plot"]) else "", | |
| 768 | "season": season, "episode": episode}) | |
| 769 | for ch in sorted(safe_listdir(INDEP_DIR)): | |
| 770 | cdir = os.path.join(INDEP_DIR, ch) | |
| 771 | if not os.path.isdir(cdir): | |
| 772 | continue | |
| 773 | for f in sorted(safe_listdir(cdir)): | |
| 774 | if not f.lower().endswith(VIDEO_EXTS): | |
| 775 | continue | |
| 776 | base, _ = os.path.splitext(f) | |
| 777 | dm = re.match(r"(\d{4}-\d{2}-\d{2}) - (.*)", base) | |
| 778 | nf = nfo_fields(os.path.join(cdir, base + ".nfo"), "title", "plot") | |
| 779 | title = nf["title"] or (dm.group(2) if dm else base) | |
| 780 | out.append({ | |
| 781 | "type": "independent", "rel": os.path.relpath(os.path.join(cdir, f), "/media"), | |
| 782 | "ctx": f"{ch} · {dm.group(1) if dm else ''}", "title": title, | |
| 783 | "link": nf["plot"] if looks_like_url(nf["plot"]) else ""}) | |
| 784 | return out | |
| 785 | ||
| 786 | ||
| 787 | LIB_PAGE = """<!doctype html> | |
| 788 | <html lang="en"><head> | |
| 789 | <meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1"> | |
| 790 | <title>yt downloader · library</title> | |
| 791 | <style> | |
| 792 | :root {{ color-scheme: light dark; }} | |
| 793 | body {{ font-family: system-ui, sans-serif; max-width: 36rem; margin: 0 auto; | |
| 794 | padding: 1rem 1rem 3rem; background: light-dark(#e8eefa, #152c42); | |
| 795 | color: light-dark(#000, #fff); }} | |
| 796 | h1 {{ font-size: 1.2rem; font-weight: 500; }} | |
| 797 | .row {{ background: light-dark(#fff, #1e3a55); border-radius: 10px; | |
| 798 | padding: 0.6rem 0.7rem; margin-bottom: 0.5rem; }} | |
| 799 | .ctx {{ font-size: 0.78rem; color: light-dark(#555, #aac); margin-bottom: 0.3rem; }} | |
| 800 | input {{ box-sizing: border-box; padding: 0.4rem; border-radius: 7px; | |
| 801 | border: 1px solid light-dark(#bbb, #456); | |
| 802 | background: light-dark(#fff, #152c42); color: inherit; }} | |
| 803 | .line {{ display: flex; gap: 0.4rem; align-items: center; }} | |
| 804 | .line .t {{ flex: 1; min-width: 0; }} | |
| 805 | .line .n {{ width: 3.2rem; flex: none; }} | |
| 806 | button {{ padding: 0.4rem 0.8rem; border-radius: 7px; border: none; cursor: pointer; | |
| 807 | background: #1a46cd; color: #fff; }} | |
| 808 | #q {{ width: 100%; margin-bottom: 0.8rem; padding: 0.55rem; }} | |
| 809 | </style></head><body> | |
| 810 | <h1>yt library <small style="font-weight:400">· {n} videos</small> | |
| 811 | <a href="/" style="float:right;font-size:.85rem;font-weight:400">◂ home</a></h1> | |
| 812 | <input id="q" placeholder="search…" autofocus> | |
| 813 | {rows} | |
| 814 | <script> | |
| 815 | const q = document.getElementById("q"); | |
| 816 | q.addEventListener("input", () => {{ | |
| 817 | const t = q.value.toLowerCase(); | |
| 818 | document.querySelectorAll(".row").forEach(r => | |
| 819 | r.style.display = r.dataset.search.includes(t) ? "" : "none"); | |
| 820 | }}); | |
| 821 | </script> | |
| 822 | </body></html>""" | |
| 823 | ||
| 824 | ||
| 825 | def lib_row(e): | |
| 826 | rel = escape(e["rel"], {'"': "&quot;"}) | |
| 827 | search = escape((e["ctx"] + " " + e["title"]).lower(), {'"': "&quot;"}) | |
| 828 | if e["type"] == "indie": | |
| 829 | fields = (f'<input class="n" name="season" type="number" min="1" value="{e["season"]}" title="season">' | |
| 830 | f'<input class="n" name="episode" type="number" min="1" value="{e["episode"]}" title="episode">') | |
| 831 | else: | |
| 832 | fields = "" | |
| 833 | link = e.get("link") | |
| 834 | open_link = (f' · <a href="{escape(link, {chr(34): "&quot;"})}" target="_blank" ' | |
| 835 | f'rel="noopener">↗ open</a>' if link else "") | |
| 836 | return f"""<form method="post" action="/rename" class="row" data-search="{search}"> | |
| 837 | <div class="ctx">{escape(e['ctx'])}{open_link}</div> | |
| 838 | <input type="hidden" name="rel" value="{rel}"> | |
| 839 | <div class="line"> | |
| 840 | <input class="t" name="title" value="{escape(e['title'], {'"': '&quot;'})}"> | |
| 841 | {fields} | |
| 842 | <button>save</button> | |
| 843 | </div> | |
| 844 | </form>""" | |
| 845 | ||
| 846 | ||
| 847 | @app.get("/library") | |
| 848 | def library(): | |
| 849 | entries = library_entries() | |
| 850 | rows = "\n".join(lib_row(e) for e in entries) or "<p style='opacity:.6'>library is empty.</p>" | |
| 851 | return LIB_PAGE.format(n=len(entries), rows=rows) | |
| 852 | ||
| 853 | ||
| 854 | @app.post("/rename") | |
| 855 | def rename(): | |
| 856 | rel = request.form.get("rel", "") | |
| 857 | new_title = request.form.get("title", "").strip() | |
| 858 | full = os.path.realpath(os.path.join("/media", rel)) | |
| 859 | indie = under(INDIE_DIR, full) | |
| 860 | if not (indie or under(INDEP_DIR, full)): | |
| 861 | return Response("path not allowed", 403, mimetype="text/plain") | |
| 862 | if not os.path.isfile(full): | |
| 863 | return Response("file not found", 404, mimetype="text/plain") | |
| 864 | if not new_title: | |
| 865 | return Response("title required", 400, mimetype="text/plain") | |
| 866 | d, fname = os.path.split(full) | |
| 867 | old_base, _ = os.path.splitext(fname) | |
| 868 | if indie: | |
| 869 | season = int(request.form.get("season", 1)) | |
| 870 | episode = int(request.form.get("episode", 1)) | |
| 871 | new_base = f"S{season:02d}E{episode:02d} - {safe_name(new_title)}" | |
| 872 | nfo_updates = {"title": new_title, "season": season, "episode": episode} | |
| 873 | else: | |
| 874 | dm = re.match(r"(\d{4}-\d{2}-\d{2}) - ", old_base) | |
| 875 | new_base = (f"{dm.group(1)} - " if dm else "") + safe_name(new_title) | |
| 876 | nfo_updates = {"title": new_title} | |
| 877 | if new_base != old_base: | |
| 878 | for f in os.listdir(d): | |
| 879 | # rename the video and every sidecar sharing its basename stem | |
| 880 | # (.nfo, .jpg, -thumb.jpg, .info.json, …) in lockstep | |
| 881 | if f.startswith(old_base): | |
| 882 | suffix = f[len(old_base):] | |
| 883 | if suffix.startswith(".") or suffix.startswith("-thumb"): | |
| 884 | os.rename(os.path.join(d, f), os.path.join(d, new_base + suffix)) | |
| 885 | set_nfo_tags(os.path.join(d, new_base + ".nfo"), nfo_updates) | |
| 886 | log(f"renamed {old_base!r} → {new_base!r}") | |
| 887 | return redirect("/library") | |
| 888 | ||
| 889 | ||
| 890 | # -------------------------------------------------------------- channel config | |
| 891 | # /config → friendly gui over the auto-download channel list | |
| 892 | # /config/raw → codemirror editor for the raw yaml files (advanced) | |
| 893 | ||
| 894 | SUBS_FILE = "subscriptions.yaml" | |
| 895 | SUB_PRESET_DEFAULT = "Jellyfin TV Show by Date | only-after | flat-videos" | |
| 896 | SUB_SECTION_DEFAULT = "= Independent Creators" | |
| 897 | # per-channel rule keys the gui understands; order = the "add rule" menu order. | |
| 898 | # anything else in a channel entry is left for the raw editor. | |
| 899 | RULE_FIELDS = ["download_after", "title_include_keywords", "title_exclude_keywords", | |
| 900 | "description_include_keywords", "description_exclude_keywords"] | |
| 901 | RAW_LABELS = {"subscriptions.yaml": "auto-download channels", | |
| 902 | "feed.yaml": "notify-me channels"} | |
| 903 | ||
| 904 | ||
| 905 | def list_yaml_configs(): | |
| 906 | return sorted(f for f in safe_listdir(YT_CONFIG_DIR) | |
| 907 | if f.endswith((".yaml", ".yml"))) | |
| 908 | ||
| 909 | ||
| 910 | def write_text_atomic(path, text): | |
| 911 | tmp = path + ".tmp" | |
| 912 | with open(tmp, "w") as f: | |
| 913 | f.write(text) | |
| 914 | os.replace(tmp, path) | |
| 915 | ||
| 916 | ||
| 917 | def parse_subscriptions(): | |
| 918 | # flatten subscriptions.yaml down to a plain channel list for the gui, | |
| 919 | # remembering the wrapping structure so it can be rebuilt verbatim on save | |
| 920 | try: | |
| 921 | with open(os.path.join(YT_CONFIG_DIR, SUBS_FILE)) as f: | |
| 922 | data = yaml.safe_load(f) or {} | |
| 923 | except (OSError, yaml.YAMLError): | |
| 924 | data = {} | |
| 925 | overrides = (data.get("__preset__") or {}).get("overrides") or {} | |
| 926 | preset_key = next((k for k in data if k != "__preset__"), SUB_PRESET_DEFAULT) | |
| 927 | section = data.get(preset_key) if isinstance(data.get(preset_key), dict) else {} | |
| 928 | section_key = next(iter(section), SUB_SECTION_DEFAULT) if section else SUB_SECTION_DEFAULT | |
| 929 | chmap = section.get(section_key) if isinstance(section.get(section_key), dict) else {} | |
| 930 | channels = [] | |
| 931 | for key, val in (chmap or {}).items(): | |
| 932 | name = key[1:].strip() if key.startswith("~") else key | |
| 933 | if isinstance(val, str): | |
| 934 | channels.append({"name": name, "url": val, "rules": {}}) | |
| 935 | elif isinstance(val, dict): | |
| 936 | rules = {k: val[k] for k in RULE_FIELDS if k in val} | |
| 937 | channels.append({"name": name, "url": val.get("url", ""), "rules": rules}) | |
| 938 | return {"preset_key": preset_key, "section_key": section_key, | |
| 939 | "overrides": overrides, "channels": channels} | |
| 940 | ||
| 941 | ||
| 942 | def build_subscriptions(channels): | |
| 943 | # rebuild the file from the gui's channel list, preserving the preset | |
| 944 | # wrapper + global overrides; a channel with no rules stays the compact | |
| 945 | # "name: url" form, one with rules becomes the "~name:" dict form | |
| 946 | cur = parse_subscriptions() | |
| 947 | chmap = {} | |
| 948 | for ch in channels: | |
| 949 | name = (ch.get("name") or "").strip() | |
| 950 | url = (ch.get("url") or "").strip() | |
| 951 | if not name or not url: | |
| 952 | continue | |
| 953 | clean = {} | |
| 954 | for field in RULE_FIELDS: | |
| 955 | v = (ch.get("rules") or {}).get(field) | |
| 956 | if isinstance(v, str): | |
| 957 | v = v.strip() | |
| 958 | if isinstance(v, list): | |
| 959 | v = [str(x).strip() for x in v if str(x).strip()] | |
| 960 | if v in (None, "", [], {}): | |
| 961 | continue | |
| 962 | clean[field] = v | |
| 963 | if clean: | |
| 964 | chmap["~" + name] = {"url": url, **clean} | |
| 965 | else: | |
| 966 | chmap[name] = url | |
| 967 | out = {} | |
| 968 | if cur["overrides"]: | |
| 969 | out["__preset__"] = {"overrides": cur["overrides"]} | |
| 970 | out[cur["preset_key"]] = {cur["section_key"]: chmap} | |
| 971 | return yaml.dump(out, sort_keys=False, allow_unicode=True, width=4096, | |
| 972 | default_flow_style=False) | |
| 973 | ||
| 974 | ||
| 975 | @app.get("/api/subscriptions") | |
| 976 | def api_subscriptions_get(): | |
| 977 | return {"channels": parse_subscriptions()["channels"]} | |
| 978 | ||
| 979 | ||
| 980 | @app.post("/api/subscriptions") | |
| 981 | def api_subscriptions_save(): | |
| 982 | payload = request.get_json(silent=True) or {} | |
| 983 | channels = payload.get("channels") | |
| 984 | if not isinstance(channels, list): | |
| 985 | return {"error": "expected a channels list"}, 400 | |
| 986 | text = build_subscriptions(channels) | |
| 987 | try: | |
| 988 | yaml.safe_load(text) | |
| 989 | except yaml.YAMLError as e: | |
| 990 | return {"error": f"could not save: {e}"}, 400 | |
| 991 | write_text_atomic(os.path.join(YT_CONFIG_DIR, SUBS_FILE), text) | |
| 992 | n = sum(1 for c in channels if (c.get("name") or "").strip() | |
| 993 | and (c.get("url") or "").strip()) | |
| 994 | log(f"subscriptions.yaml saved via gui ({n} channels)") | |
| 995 | return {"ok": True, "count": n} | |
| 996 | ||
| 997 | ||
| 998 | @app.get("/api/configs") | |
| 999 | def api_configs_get(): | |
| 1000 | files = [] | |
| 1001 | for name in list_yaml_configs(): | |
| 1002 | try: | |
| 1003 | with open(os.path.join(YT_CONFIG_DIR, name)) as f: | |
| 1004 | files.append({"name": name, "label": RAW_LABELS.get(name, ""), | |
| 1005 | "body": f.read()}) | |
| 1006 | except OSError: | |
| 1007 | continue | |
| 1008 | return {"files": files} | |
| 1009 | ||
| 1010 | ||
| 1011 | @app.post("/api/config") | |
| 1012 | def api_config_save(): | |
| 1013 | payload = request.get_json(silent=True) or {} | |
| 1014 | name, body = payload.get("name", ""), payload.get("body", "") | |
| 1015 | if name not in list_yaml_configs(): | |
| 1016 | return {"error": "unknown file"}, 400 | |
| 1017 | try: | |
| 1018 | yaml.safe_load(body) | |
| 1019 | except yaml.YAMLError as e: | |
| 1020 | return {"error": str(e)}, 400 | |
| 1021 | write_text_atomic(os.path.join(YT_CONFIG_DIR, name), body) | |
| 1022 | log(f"config saved via raw editor: {name}") | |
| 1023 | return {"ok": True} | |
| 1024 | ||
| 1025 | ||
| 1026 | GUI_PAGE = """<!doctype html> | |
| 1027 | <html lang="en"><head> | |
| 1028 | <meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1"> | |
| 1029 | <title>yt downloader · channels</title> | |
| 1030 | <style> | |
| 1031 | :root { color-scheme: light dark; } | |
| 1032 | body { font-family: system-ui, sans-serif; max-width: 38rem; margin: 0 auto; | |
| 1033 | padding: 1rem 1rem 6rem; background: light-dark(#e8eefa, #152c42); | |
| 1034 | color: light-dark(#000, #fff); } | |
| 1035 | h1 { font-size: 1.2rem; font-weight: 500; } | |
| 1036 | h1 a { float: right; font-size: .85rem; font-weight: 400; } | |
| 1037 | .sub { font-size: .85rem; color: light-dark(#555, #aac); margin: -.3rem 0 1rem; } | |
| 1038 | .card { background: light-dark(#fff, #1e3a55); border-radius: 12px; | |
| 1039 | padding: .7rem .8rem; margin-bottom: .7rem; } | |
| 1040 | input, select { box-sizing: border-box; padding: .42rem .5rem; border-radius: 8px; | |
| 1041 | border: 1px solid light-dark(#bbb, #456); font-size: .9rem; | |
| 1042 | background: light-dark(#fff, #122739); color: inherit; } | |
| 1043 | .chrow { display: flex; gap: .4rem; align-items: center; } | |
| 1044 | .chrow .name { flex: 2; min-width: 0; font-weight: 600; } | |
| 1045 | .chrow .url { flex: 3; min-width: 0; } | |
| 1046 | .x { flex: none; width: 2rem; padding: .3rem 0; background: transparent; | |
| 1047 | color: light-dark(#999, #88a); border: none; cursor: pointer; font-size: 1.2rem; } | |
| 1048 | .rule { display: flex; gap: .4rem; align-items: center; flex-wrap: wrap; | |
| 1049 | margin: .45rem 0 0; padding: .4rem .5rem; border-radius: 8px; | |
| 1050 | background: light-dark(#eef2fb, #16304880); } | |
| 1051 | .rlabel { font-size: .82rem; color: light-dark(#445, #bcd); flex: none; } | |
| 1052 | .rule input[type=date] { flex: none; } | |
| 1053 | .rule .wordin { flex: 1; min-width: 6rem; } | |
| 1054 | .chips { display: flex; gap: .3rem; flex-wrap: wrap; } | |
| 1055 | .chip { display: inline-flex; align-items: center; gap: .3rem; font-size: .8rem; | |
| 1056 | background: light-dark(#d8e0f5, #28507a); padding: .12rem .2rem .12rem .5rem; | |
| 1057 | border-radius: 99px; } | |
| 1058 | .chip button { background: none; border: none; color: inherit; cursor: pointer; | |
| 1059 | font-size: .95rem; line-height: 1; padding: 0 .15rem; opacity: .7; } | |
| 1060 | .rremove { flex: none; background: transparent; border: none; cursor: pointer; | |
| 1061 | color: light-dark(#a55, #f0a0a0); font-size: .78rem; } | |
| 1062 | .addrule select { margin-top: .45rem; font-size: .82rem; color: light-dark(#456, #9bf); | |
| 1063 | background: transparent; border: 1px dashed light-dark(#aab, #567); } | |
| 1064 | .add-channel { width: 100%; padding: .6rem; border-radius: 10px; border: 1px dashed | |
| 1065 | light-dark(#9ab, #567); background: transparent; color: light-dark(#345, #bcd); | |
| 1066 | cursor: pointer; font-size: .95rem; } | |
| 1067 | .bar { position: fixed; left: 0; right: 0; bottom: 0; display: flex; gap: .8rem; | |
| 1068 | align-items: center; padding: .7rem 1rem; | |
| 1069 | background: light-dark(#dde6f7ee, #112338ee); backdrop-filter: blur(6px); | |
| 1070 | border-top: 1px solid light-dark(#ccd, #244260); } | |
| 1071 | #save { padding: .55rem 1.4rem; border-radius: 9px; border: none; cursor: pointer; | |
| 1072 | background: #1a46cd; color: #fff; font-size: .95rem; } | |
| 1073 | #status { font-size: .85rem; color: light-dark(#456, #bcd); } | |
| 1074 | .bar a { margin-left: auto; font-size: .85rem; } | |
| 1075 | </style></head><body> | |
| 1076 | <h1>channels <a href="/">◂ home</a></h1> | |
| 1077 | <p class="sub">videos from these channels download automatically.</p> | |
| 1078 | <div id="list"></div> | |
| 1079 | <button class="add-channel" id="add">+ add channel</button> | |
| 1080 | <div class="bar"><button id="save">save</button><span id="status"></span> | |
| 1081 | <a href="/config/raw">edit raw yaml ▸</a></div> | |
| 1082 | <script> | |
| 1083 | const RULES = [ | |
| 1084 | {key:"download_after", label:"Backlog", kind:"backlog"}, | |
| 1085 | {key:"title_include_keywords", label:"Only if title contains", kind:"words"}, | |
| 1086 | {key:"title_exclude_keywords", label:"Skip if title contains", kind:"words"}, | |
| 1087 | {key:"description_include_keywords", label:"Only if description contains", kind:"words"}, | |
| 1088 | {key:"description_exclude_keywords", label:"Skip if description contains", kind:"words"}, | |
| 1089 | ]; | |
| 1090 | const byKey = Object.fromEntries(RULES.map(r => [r.key, r])); | |
| 1091 | let channels = []; | |
| 1092 | const list = document.getElementById("list"); | |
| 1093 | const statusEl = document.getElementById("status"); | |
| 1094 | ||
| 1095 | const ymd2date = s => (s && s.length === 8) ? s.slice(0,4)+"-"+s.slice(4,6)+"-"+s.slice(6,8) : ""; | |
| 1096 | const date2ymd = s => s ? s.replaceAll("-", "") : ""; | |
| 1097 | ||
| 1098 | function el(tag, props, ...kids) { | |
| 1099 | const e = Object.assign(document.createElement(tag), props || {}); | |
| 1100 | for (const k of kids) e.append(k); | |
| 1101 | return e; | |
| 1102 | } | |
| 1103 | ||
| 1104 | function render() { | |
| 1105 | list.innerHTML = ""; | |
| 1106 | channels.forEach((ch, i) => list.append(card(ch, i))); | |
| 1107 | } | |
| 1108 | ||
| 1109 | function card(ch, i) { | |
| 1110 | const name = el("input", {className:"name", value:ch.name||"", placeholder:"channel name"}); | |
| 1111 | name.oninput = () => ch.name = name.value; | |
| 1112 | const url = el("input", {className:"url", value:ch.url||"", placeholder:"youtube.com/@handle"}); | |
| 1113 | url.oninput = () => ch.url = url.value; | |
| 1114 | const x = el("button", {className:"x", type:"button", textContent:"×", | |
| 1115 | title:"remove channel", onclick:() => { channels.splice(i,1); render(); }}); | |
| 1116 | const c = el("div", {className:"card"}, el("div", {className:"chrow"}, name, url, x)); | |
| 1117 | for (const key of Object.keys(ch.rules || {})) c.append(ruleRow(ch, key)); | |
| 1118 | c.append(addRule(ch)); | |
| 1119 | return c; | |
| 1120 | } | |
| 1121 | ||
| 1122 | function ruleRow(ch, key) { | |
| 1123 | const meta = byKey[key]; | |
| 1124 | const row = el("div", {className:"rule"}, el("span", {className:"rlabel", textContent:meta.label})); | |
| 1125 | if (meta.kind === "backlog") { | |
| 1126 | const cur = ch.rules[key] || "19700101"; | |
| 1127 | const sel = el("select"); | |
| 1128 | sel.append(el("option", {value:"all", textContent:"entire history"}), | |
| 1129 | el("option", {value:"date", textContent:"since a date"})); | |
| 1130 | const date = el("input", {type:"date"}); | |
| 1131 | const sync = () => { | |
| 1132 | if (sel.value === "all") { ch.rules[key] = "19700101"; date.style.display = "none"; } | |
| 1133 | else { date.style.display = ""; ch.rules[key] = date2ymd(date.value); } | |
| 1134 | }; | |
| 1135 | if (cur === "19700101") { sel.value = "all"; date.style.display = "none"; } | |
| 1136 | else { sel.value = "date"; date.value = ymd2date(cur); } | |
| 1137 | sel.onchange = sync; date.oninput = sync; | |
| 1138 | row.append(sel, date); | |
| 1139 | } else { | |
| 1140 | const chips = el("div", {className:"chips"}); | |
| 1141 | (ch.rules[key] || []).forEach((w, wi) => chips.append(chip(ch.rules[key], wi))); | |
| 1142 | const inp = el("input", {className:"wordin", placeholder:"type a word, press enter"}); | |
| 1143 | inp.onkeydown = e => { | |
| 1144 | if (e.key === "Enter") { | |
| 1145 | e.preventDefault(); | |
| 1146 | const v = inp.value.trim(); | |
| 1147 | if (v) { (ch.rules[key] = ch.rules[key] || []).push(v); render(); } | |
| 1148 | } | |
| 1149 | }; | |
| 1150 | row.append(chips, inp); | |
| 1151 | } | |
| 1152 | row.append(el("button", {className:"rremove", type:"button", textContent:"remove", | |
| 1153 | onclick:() => { delete ch.rules[key]; render(); }})); | |
| 1154 | return row; | |
| 1155 | } | |
| 1156 | ||
| 1157 | function chip(arr, i) { | |
| 1158 | return el("span", {className:"chip", textContent:arr[i]}, | |
| 1159 | el("button", {type:"button", textContent:"×", onclick:() => { arr.splice(i,1); render(); }})); | |
| 1160 | } | |
| 1161 | ||
| 1162 | function addRule(ch) { | |
| 1163 | const avail = RULES.filter(r => !(r.key in (ch.rules || {}))); | |
| 1164 | const bar = el("div", {className:"addrule"}); | |
| 1165 | if (!avail.length) return bar; | |
| 1166 | const sel = el("select"); | |
| 1167 | sel.append(el("option", {value:"", textContent:"+ add rule…"})); | |
| 1168 | for (const r of avail) sel.append(el("option", {value:r.key, textContent:r.label})); | |
| 1169 | sel.onchange = () => { | |
| 1170 | if (!sel.value) return; | |
| 1171 | ch.rules = ch.rules || {}; | |
| 1172 | ch.rules[sel.value] = byKey[sel.value].kind === "backlog" ? "19700101" : []; | |
| 1173 | render(); | |
| 1174 | }; | |
| 1175 | return bar.append(sel), bar; | |
| 1176 | } | |
| 1177 | ||
| 1178 | document.getElementById("add").onclick = () => { | |
| 1179 | channels.push({name:"", url:"", rules:{}}); | |
| 1180 | render(); | |
| 1181 | const last = list.lastChild && list.lastChild.querySelector("input.name"); | |
| 1182 | if (last) last.focus(); | |
| 1183 | }; | |
| 1184 | ||
| 1185 | document.getElementById("save").onclick = async () => { | |
| 1186 | statusEl.textContent = "saving…"; | |
| 1187 | try { | |
| 1188 | const r = await fetch("/api/subscriptions", {method:"POST", | |
| 1189 | headers:{"content-type":"application/json"}, body:JSON.stringify({channels})}); | |
| 1190 | const j = await r.json().catch(() => ({})); | |
| 1191 | statusEl.textContent = r.ok ? `saved ✓ · ${j.count} channels` : `error: ${j.error || r.status}`; | |
| 1192 | } catch (e) { statusEl.textContent = "error: " + e; } | |
| 1193 | setTimeout(() => statusEl.textContent = "", 5000); | |
| 1194 | }; | |
| 1195 | ||
| 1196 | (async () => { | |
| 1197 | const r = await fetch("/api/subscriptions"); | |
| 1198 | channels = (await r.json()).channels || []; | |
| 1199 | render(); | |
| 1200 | })(); | |
| 1201 | </script> | |
| 1202 | </body></html>""" | |
| 1203 | ||
| 1204 | ||
| 1205 | RAW_PAGE = """<!doctype html> | |
| 1206 | <html lang="en"><head> | |
| 1207 | <meta charset="utf-8"><meta name="viewport" content="width=device-width, initial-scale=1"> | |
| 1208 | <title>yt downloader · raw yaml</title> | |
| 1209 | <link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/codemirror/5.65.16/codemirror.min.css"> | |
| 1210 | <link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/codemirror/5.65.16/theme/material-darker.min.css"> | |
| 1211 | <style> | |
| 1212 | :root { color-scheme: light dark; } | |
| 1213 | body { font-family: system-ui, sans-serif; max-width: 46rem; margin: 0 auto; | |
| 1214 | padding: 1rem 1rem 4rem; background: light-dark(#e8eefa, #152c42); | |
| 1215 | color: light-dark(#000, #fff); } | |
| 1216 | h1 { font-size: 1.2rem; font-weight: 500; } | |
| 1217 | h1 a { float: right; font-size: .85rem; font-weight: 400; } | |
| 1218 | h2 { font-size: 1rem; margin: 1.6rem 0 .1rem; } | |
| 1219 | h2 small { font-weight: 400; color: light-dark(#667, #9ab); } | |
| 1220 | .CodeMirror { height: auto; border-radius: 10px; border: 1px solid light-dark(#bbb, #345); | |
| 1221 | font-size: .85rem; } | |
| 1222 | .CodeMirror-scroll { min-height: 12rem; max-height: 28rem; } | |
| 1223 | .act { display: flex; gap: .8rem; align-items: center; margin-top: .4rem; } | |
| 1224 | button { padding: .45rem 1.1rem; border-radius: 8px; border: none; cursor: pointer; | |
| 1225 | background: #1a46cd; color: #fff; font-size: .9rem; } | |
| 1226 | .msg { font-size: .85rem; } | |
| 1227 | .msg.bad { color: light-dark(#b03030, #f0a0a0); white-space: pre-wrap; | |
| 1228 | font-family: ui-monospace, monospace; } | |
| 1229 | .msg.ok { color: light-dark(#2a7a2a, #8fd48f); } | |
| 1230 | </style></head><body> | |
| 1231 | <h1>raw yaml <a href="/config">◂ channels</a></h1> | |
| 1232 | <div id="files">loading…</div> | |
| 1233 | <script src="https://cdnjs.cloudflare.com/ajax/libs/codemirror/5.65.16/codemirror.min.js"></script> | |
| 1234 | <script src="https://cdnjs.cloudflare.com/ajax/libs/codemirror/5.65.16/mode/yaml/yaml.min.js"></script> | |
| 1235 | <script> | |
| 1236 | const dark = matchMedia("(prefers-color-scheme: dark)").matches; | |
| 1237 | (async () => { | |
| 1238 | const files = (await (await fetch("/api/configs")).json()).files || []; | |
| 1239 | const root = document.getElementById("files"); | |
| 1240 | root.innerHTML = ""; | |
| 1241 | for (const f of files) { | |
| 1242 | const h = document.createElement("h2"); | |
| 1243 | h.innerHTML = f.name + (f.label ? ' <small>· ' + f.label + '</small>' : ''); | |
| 1244 | const ta = document.createElement("textarea"); | |
| 1245 | ta.value = f.body; | |
| 1246 | const act = document.createElement("div"); act.className = "act"; | |
| 1247 | const btn = document.createElement("button"); btn.textContent = "save " + f.name; | |
| 1248 | const msg = document.createElement("span"); msg.className = "msg"; | |
| 1249 | act.append(btn, msg); | |
| 1250 | root.append(h, ta, act); | |
| 1251 | const cm = CodeMirror.fromTextArea(ta, {mode:"yaml", lineNumbers:true, | |
| 1252 | theme: dark ? "material-darker" : "default", viewportMargin: Infinity}); | |
| 1253 | btn.onclick = async () => { | |
| 1254 | msg.textContent = "saving…"; msg.className = "msg"; | |
| 1255 | const r = await fetch("/api/config", {method:"POST", | |
| 1256 | headers:{"content-type":"application/json"}, | |
| 1257 | body: JSON.stringify({name: f.name, body: cm.getValue()})}); | |
| 1258 | const j = await r.json().catch(() => ({})); | |
| 1259 | if (r.ok) { msg.textContent = "saved ✓"; msg.className = "msg ok"; } | |
| 1260 | else { msg.textContent = j.error || ("error " + r.status); msg.className = "msg bad"; } | |
| 1261 | }; | |
| 1262 | } | |
| 1263 | })(); | |
| 1264 | </script> | |
| 1265 | </body></html>""" | |
| 1266 | ||
| 1267 | ||
| 1268 | @app.get("/config") | |
| 1269 | def config_page(): | |
| 1270 | return Response(GUI_PAGE, mimetype="text/html") | |
| 1271 | ||
| 1272 | ||
| 1273 | @app.get("/config/raw") | |
| 1274 | def config_raw(): | |
| 1275 | return Response(RAW_PAGE, mimetype="text/html") | |
| 1276 | ||
| 1277 | ||
| 1278 | @app.get("/api/upscale") | |
| 1279 | def api_upscale_status(): | |
| 1280 | # progress is written by the separate yt-upscaler service into our data dir | |
| 1281 | return load_json("upscale-status.json", {"enabled": False}) | |
| 1282 | ||
| 1283 | ||
| 636 | 1284 | if __name__ == "__main__": |
| 637 | 1285 | # jobs interrupted by a container restart pick up where they left off |
| 638 | 1286 | with state_lock: |
config/yt-upscaler/Dockerfile created+8| ... | ... | @@ -0,0 +1,8 @@ |
| 1 | # standalone thumbnail super-resolution worker. kept off the ytdl-sub base | |
| 2 | # image because that image is on python 3.14, which onnxruntime has no wheels | |
| 3 | # for yet; this pins a supported python and stays tiny (no yt-dlp/ffmpeg). | |
| 4 | FROM python:3.12-slim | |
| 5 | RUN pip install --no-cache-dir onnxruntime numpy pillow | |
| 6 | COPY realesr-general-x4v3.onnx /app/realesr-general-x4v3.onnx | |
| 7 | COPY upscale.py /app/upscale.py | |
| 8 | CMD ["python3", "-u", "/app/upscale.py"] |
config/yt-upscaler/upscale.py created+194| ... | ... | @@ -0,0 +1,194 @@ |
| 1 | #!/usr/bin/env python3 | |
| 2 | # yt-upscaler: keeps the Independent library's thumbnails sharp on a TV. | |
| 3 | # youtube only stores ~720p thumbnails, which jellyfin then stretches across | |
| 4 | # the whole screen as the item backdrop (soft + blocky). this re-fetches the | |
| 5 | # highest-res thumbnail from the image CDN (not bot-walled) and runs it through | |
| 6 | # Real-ESRGAN general-x4v3 (exported to ONNX, so no pickle is ever loaded) at | |
| 7 | # 4x on CPU, writing a crisp <video>.webp that jellyfin uses for card + backdrop. | |
| 8 | import io | |
| 9 | import json | |
| 10 | import os | |
| 11 | import re | |
| 12 | import time | |
| 13 | import urllib.request | |
| 14 | ||
| 15 | import numpy as np | |
| 16 | import onnxruntime as ort | |
| 17 | from PIL import Image | |
| 18 | ||
| 19 | INDEP_DIR = os.environ.get("INDEP_DIR", "/media/jellyfin/Independent") | |
| 20 | STATE_DIR = os.environ.get("STATE_DIR", "/state") | |
| 21 | MODEL = os.environ.get("SR_MODEL", "/app/realesr-general-x4v3.onnx") | |
| 22 | MIN_H = int(os.environ.get("SR_MIN_H", "1400")) # >= this tall ⇒ already done | |
| 23 | CAP_H = int(os.environ.get("SR_CAP_H", "2160")) # cap output height (4k-ish) | |
| 24 | TILE = int(os.environ.get("SR_TILE", "256")) # tile size to bound memory | |
| 25 | THREADS = int(os.environ.get("SR_THREADS", "4")) # leave cpu for other services | |
| 26 | SCAN_INTERVAL = int(os.environ.get("SR_SCAN_INTERVAL", "1800")) | |
| 27 | VIDEO_EXTS = (".webm", ".mp4", ".mkv") | |
| 28 | UA = {"User-Agent": "Mozilla/5.0 (yt-upscaler; +https://paperclover.net)"} | |
| 29 | ||
| 30 | status = {"enabled": True, "running": False, "done": 0, "total": 0, | |
| 31 | "current": "", "errors": 0} | |
| 32 | _sess = None | |
| 33 | ||
| 34 | ||
| 35 | def log(m): | |
| 36 | print(m, flush=True) | |
| 37 | ||
| 38 | ||
| 39 | def write_status(): | |
| 40 | try: | |
| 41 | tmp = os.path.join(STATE_DIR, "upscale-status.json.tmp") | |
| 42 | with open(tmp, "w") as f: | |
| 43 | json.dump(status, f) | |
| 44 | os.replace(tmp, os.path.join(STATE_DIR, "upscale-status.json")) | |
| 45 | except OSError: | |
| 46 | pass | |
| 47 | ||
| 48 | ||
| 49 | def session(): | |
| 50 | global _sess | |
| 51 | if _sess is None: | |
| 52 | opts = ort.SessionOptions() | |
| 53 | opts.intra_op_num_threads = THREADS | |
| 54 | opts.inter_op_num_threads = 1 | |
| 55 | _sess = ort.InferenceSession(MODEL, sess_options=opts, | |
| 56 | providers=["CPUExecutionProvider"]) | |
| 57 | return _sess | |
| 58 | ||
| 59 | ||
| 60 | def upscale(img, scale=4, pad=8): | |
| 61 | # tiled 4x SR; overlap each tile by `pad` px and crop it back to hide seams | |
| 62 | arr = np.asarray(img, dtype=np.float32) / 255.0 | |
| 63 | h, w, _ = arr.shape | |
| 64 | out = np.zeros((h * scale, w * scale, 3), dtype=np.float32) | |
| 65 | sess = session() | |
| 66 | for y in range(0, h, TILE): | |
| 67 | for x in range(0, w, TILE): | |
| 68 | y0, x0 = max(0, y - pad), max(0, x - pad) | |
| 69 | y1, x1 = min(h, y + TILE + pad), min(w, x + TILE + pad) | |
| 70 | patch = arr[y0:y1, x0:x1].transpose(2, 0, 1)[None] | |
| 71 | res = sess.run(None, {"input": patch})[0][0].transpose(1, 2, 0) | |
| 72 | th, tw = min(TILE, h - y) * scale, min(TILE, w - x) * scale | |
| 73 | ty, tx = (y - y0) * scale, (x - x0) * scale | |
| 74 | out[y * scale:y * scale + th, x * scale:x * scale + tw] = \ | |
| 75 | res[ty:ty + th, tx:tx + tw] | |
| 76 | return Image.fromarray((out.clip(0, 1) * 255).round().astype("uint8")) | |
| 77 | ||
| 78 | ||
| 79 | def yt_id(base): | |
| 80 | ij = base + ".info.json" | |
| 81 | if os.path.exists(ij): | |
| 82 | try: | |
| 83 | with open(ij) as f: | |
| 84 | vid = json.load(f).get("id") | |
| 85 | if vid: | |
| 86 | return vid | |
| 87 | except (OSError, json.JSONDecodeError): | |
| 88 | pass | |
| 89 | try: | |
| 90 | with open(base + ".nfo") as f: | |
| 91 | m = re.search(r"<plot>(.*?)</plot>", f.read(), re.S) | |
| 92 | if m: | |
| 93 | u = re.search(r"(?:v=|youtu\.be/)([A-Za-z0-9_-]{11})", m.group(1)) | |
| 94 | return u.group(1) if u else None | |
| 95 | except OSError: | |
| 96 | pass | |
| 97 | return None | |
| 98 | ||
| 99 | ||
| 100 | def fetch_cdn_webp(vid): | |
| 101 | for q in ("maxresdefault", "sddefault", "hqdefault"): | |
| 102 | try: | |
| 103 | req = urllib.request.Request( | |
| 104 | f"https://i.ytimg.com/vi_webp/{vid}/{q}.webp", headers=UA) | |
| 105 | with urllib.request.urlopen(req, timeout=30) as r: | |
| 106 | data = r.read() | |
| 107 | if len(data) > 1000: | |
| 108 | return Image.open(io.BytesIO(data)).convert("RGB") | |
| 109 | except Exception: | |
| 110 | continue | |
| 111 | return None | |
| 112 | ||
| 113 | ||
| 114 | def img_height(path): | |
| 115 | try: | |
| 116 | with Image.open(path) as im: | |
| 117 | return im.height | |
| 118 | except Exception: | |
| 119 | return 0 | |
| 120 | ||
| 121 | ||
| 122 | def enhance(video_path): | |
| 123 | base = os.path.splitext(video_path)[0] | |
| 124 | webp, jpg = base + ".webp", base + ".jpg" | |
| 125 | if os.path.exists(webp) and img_height(webp) >= MIN_H: | |
| 126 | return "skip" | |
| 127 | src = None | |
| 128 | vid = yt_id(base) | |
| 129 | if vid: | |
| 130 | src = fetch_cdn_webp(vid) | |
| 131 | if src is None: | |
| 132 | for p in (webp, jpg): | |
| 133 | if os.path.exists(p): | |
| 134 | src = Image.open(p).convert("RGB") | |
| 135 | break | |
| 136 | if src is None: | |
| 137 | return "no-source" | |
| 138 | up = upscale(src) | |
| 139 | if up.height > CAP_H: | |
| 140 | up = up.resize((round(up.width * CAP_H / up.height), CAP_H), Image.LANCZOS) | |
| 141 | tmp = webp + ".tmp" | |
| 142 | up.save(tmp, "WEBP", quality=92, method=6) | |
| 143 | os.replace(tmp, webp) | |
| 144 | if os.path.exists(jpg): | |
| 145 | os.remove(jpg) | |
| 146 | return "done" | |
| 147 | ||
| 148 | ||
| 149 | def independent_videos(): | |
| 150 | out = [] | |
| 151 | for ch in sorted(os.listdir(INDEP_DIR)) if os.path.isdir(INDEP_DIR) else []: | |
| 152 | cdir = os.path.join(INDEP_DIR, ch) | |
| 153 | if not os.path.isdir(cdir): | |
| 154 | continue | |
| 155 | for f in sorted(os.listdir(cdir)): | |
| 156 | if f.lower().endswith(VIDEO_EXTS): | |
| 157 | out.append(os.path.join(cdir, f)) | |
| 158 | return out | |
| 159 | ||
| 160 | ||
| 161 | def main(): | |
| 162 | log(f"yt-upscaler started (model={MODEL}, threads={THREADS})") | |
| 163 | while True: | |
| 164 | vids = independent_videos() | |
| 165 | pending = [v for v in vids | |
| 166 | if img_height(os.path.splitext(v)[0] + ".webp") < MIN_H] | |
| 167 | status.update(total=len(vids), done=len(vids) - len(pending), | |
| 168 | running=bool(pending), current="") | |
| 169 | write_status() | |
| 170 | if pending: | |
| 171 | log(f"upscaling {len(pending)} thumbnail(s)…") | |
| 172 | for v in pending: | |
| 173 | status["current"] = os.path.basename(v) | |
| 174 | write_status() | |
| 175 | t = time.time() | |
| 176 | try: | |
| 177 | r = enhance(v) | |
| 178 | if r == "done": | |
| 179 | log(f"upscaled ({time.time()-t:.0f}s): {os.path.basename(v)}") | |
| 180 | elif r != "skip": | |
| 181 | log(f"{r}: {os.path.basename(v)}") | |
| 182 | except Exception as e: | |
| 183 | status["errors"] += 1 | |
| 184 | log(f"failed: {os.path.basename(v)}: {e}") | |
| 185 | status["done"] += 1 | |
| 186 | write_status() | |
| 187 | time.sleep(1) # be polite to the rest of the box | |
| 188 | status.update(running=False, current="") | |
| 189 | write_status() | |
| 190 | time.sleep(SCAN_INTERVAL) | |
| 191 | ||
| 192 | ||
| 193 | if __name__ == "__main__": | |
| 194 | main() |
config/yt/archive-loop.sh+1-1| ... | ... | @@ -3,7 +3,7 @@ |
| 3 | 3 | # ("sign in to confirm you're not a bot"), back off for a whole day instead |
| 4 | 4 | # of hammering it every cycle, which prolongs the wall. |
| 5 | 5 | while true; do |
| 6 | ytdl-sub --config /config-yt/config.yaml sub /config-yt/subscriptions.yaml 2>&1 | tee /tmp/last-pass.log | |
| 6 | ytdl-sub --config /config-yt/config.yaml sub /yt-config/subscriptions.yaml 2>&1 | tee /tmp/last-pass.log | |
| 7 | 7 | if grep -q "confirm you.re not a bot" /tmp/last-pass.log; then |
| 8 | 8 | echo "[archive-loop] bot wall detected; sleeping 24h" |
| 9 | 9 | sleep 86400 |
config/yt/config.yaml+14| ... | ... | @@ -26,3 +26,17 @@ presets: |
| 26 | 26 | episode_file_path: "{episode_file_name_sanitized}" |
| 27 | 27 | episode_file_name: "{upload_date_standardized} - {file_title}" |
| 28 | 28 | thumbnail_file_name: "{episode_file_path}.jpg" |
| 29 | # sponsorblock: cut paid sponsor reads, self-promo (merch/patreon), and | |
| 30 | # like/subscribe reminders out of newly downloaded videos. intros, outros, | |
| 31 | # and all real content are kept. applies to new downloads only (the archive | |
| 32 | # remembers what's already fetched). segment data is crowd-sourced from the | |
| 33 | # sponsorblock api, which isn't affected by youtube bot walls. | |
| 34 | chapters: | |
| 35 | sponsorblock_categories: | |
| 36 | - sponsor | |
| 37 | - selfpromo | |
| 38 | - interaction | |
| 39 | remove_sponsorblock_categories: | |
| 40 | - sponsor | |
| 41 | - selfpromo | |
| 42 | - interaction |
config/yt/feed.yaml deleted-22| ... | ... | @@ -1,22 +0,0 @@ |
| 1 | # channels whose new videos land in the triage queue at yt.<domain> for | |
| 2 | # manual sorting into shows/music/creators (home-infra issue #6). each also | |
| 3 | # sends a notification email with a link to the review page. | |
| 4 | # adding a channel is one line; any youtube channel url or @handle url works. | |
| 5 | channels: | |
| 6 | "ArrowType": "https://www.youtube.com/@ArrowType" | |
| 7 | "SethBling": "https://www.youtube.com/@SethBling" | |
| 8 | "Voidstar": "https://www.youtube.com/@voidstar-digital" | |
| 9 | "MallBat": "https://www.youtube.com/@mallbat" | |
| 10 | "Early Eyes": "https://www.youtube.com/@earlyeyes" | |
| 11 | "Otaku-Vs": "https://www.youtube.com/@OtakuVs" | |
| 12 | "Something Witty Entertainment": "https://www.youtube.com/@SWE" | |
| 13 | "Ethan Niser": "https://www.youtube.com/@ethanniser" | |
| 14 | "V3rb": "https://www.youtube.com/@VerbDoesStuff" | |
| 15 | "dyc3": "https://www.youtube.com/@rollthedyc3" | |
| 16 | "Technology Connections": "https://www.youtube.com/@TechnologyConnections" | |
| 17 | "jan Misali": "https://www.youtube.com/@HBMmaster" | |
| 18 | "awe": https://www.youtube.com/@whyawe | |
| 19 | "JJBlair": "https://www.youtube.com/@JJBlairrecording" | |
| 20 | "2 Mello": "https://www.youtube.com/@2mello" | |
| 21 | "XavierWolf": "https://www.youtube.com/@xavierwolfy" | |
| 22 | "chaosyumi": "https://www.youtube.com/@willburtz" |
config/yt/subscriptions.yaml deleted-69| ... | ... | @@ -1,69 +0,0 @@ |
| 1 | # channels that are archived automatically (home-infra issue #6). | |
| 2 | # every upload lands in media/jellyfin/Independent/<Channel Name>/ with | |
| 3 | # .nfo metadata + thumbnails so jellyfin shows each channel as a series. | |
| 4 | # | |
| 5 | # to add a channel: one line under the preset, then | |
| 6 | # sh sync.sh && sh docker.sh restart ytdl-sub | |
| 7 | # (or just sync and wait for the next 6h pass) | |
| 8 | ||
| 9 | __preset__: | |
| 10 | overrides: | |
| 11 | tv_show_directory: "/media/jellyfin/Independent" | |
| 12 | # default backlog policy: new uploads only. set download_after on a | |
| 13 | # channel (the "~name" form) to backfill from a date, or to "19700101" | |
| 14 | # for the entire backlog. | |
| 15 | download_after: "20260611" | |
| 16 | ||
| 17 | Jellyfin TV Show by Date | only-after | flat-videos: | |
| 18 | = Independent Creators: | |
| 19 | # full archive | |
| 20 | "~Retro Game Mechanics Explained": | |
| 21 | url: "https://www.youtube.com/@RGMechEx" | |
| 22 | download_after: "19700101" | |
| 23 | # skip videos whose title contains any of these (case-insensitive | |
| 24 | # substrings). works on any channel entry. for videos already | |
| 25 | # downloaded, just delete the files — the download archive remembers | |
| 26 | # them and won't re-fetch. | |
| 27 | title_exclude_keywords: | |
| 28 | - "q&a session" | |
| 29 | - "channel trailer" | |
| 30 | - "launching memberships" | |
| 31 | - "subscriber milestone" | |
| 32 | "~Franco Citera": | |
| 33 | url: "https://www.youtube.com/@francocitera" | |
| 34 | download_after: "19700101" | |
| 35 | # partial backlog | |
| 36 | "~bill wurtz": | |
| 37 | url: "https://www.youtube.com/@billwurtz" | |
| 38 | download_after: "20260401" # 'i'm going off the map' onward | |
| 39 | "~Coffeezilla": | |
| 40 | url: "https://www.youtube.com/@Coffeezilla" | |
| 41 | download_after: "20260609" # 'I Found The $200,000 Missing Lego' onward | |
| 42 | "~classic j": | |
| 43 | url: "https://www.youtube.com/@classicj7094" | |
| 44 | download_after: "20240801" | |
| 45 | "~JUNIA": | |
| 46 | url: "https://www.youtube.com/@butterflywife" | |
| 47 | download_after: "19700101" | |
| 48 | "~hbomberguy": | |
| 49 | url: https://www.youtube.com/@hbomberguy | |
| 50 | download_after: "20190210" | |
| 51 | # new uploads only (default policy) | |
| 52 | "t3ssel8r": "https://www.youtube.com/@t3ssel8r" | |
| 53 | "Nes": "https://www.youtube.com/@nesorion6" | |
| 54 | "4096": "https://www.youtube.com/@4096" | |
| 55 | "MegaLag": "https://www.youtube.com/@MegaLag" | |
| 56 | "Stuff Made Here": "https://www.youtube.com/@StuffMadeHere" | |
| 57 | "mali potka": "https://www.youtube.com/@malipotka4294" | |
| 58 | "Michael Reeves": "https://www.youtube.com/@MichaelReeves" | |
| 59 | "Patrick Foley": "https://www.youtube.com/@patrickfoley489" | |
| 60 | "orchard phobia": "https://www.youtube.com/@orchardphobia" | |
| 61 | "Captain Disillusion": "https://www.youtube.com/@CaptainDisillusion" | |
| 62 | "mnmira": "https://www.youtube.com/@mnmira" | |
| 63 | "EmpLemon": "https://www.youtube.com/@EmperorLemon" | |
| 64 | "andMo'": "https://www.youtube.com/@andMo" | |
| 65 | "~CGP Grey": | |
| 66 | url: "https://www.youtube.com/@CGPGrey" | |
| 67 | # members-only preview posts can't download and abort the queue | |
| 68 | title_exclude_keywords: | |
| 69 | - "early preview for bonnie bees" |
generate-env.sh+1| ... | ... | @@ -79,6 +79,7 @@ template() { |
| 79 | 79 | |
| 80 | 80 | section "shale" |
| 81 | 81 | add "SHALE_CLIENT_SECRET" "" |
| 82 | add "SHALE_SESSION_SECRET" "$(hex_secret 32)" | |
| 82 | 83 | |
| 83 | 84 | section "evil infra" |
| 84 | 85 | add "EVIL_FORGEJO_SERVER_LFS_JWT_SECRET" "$(forgejo generate secret LFS_JWT_SECRET)" |
vllm/compose.agent.yaml created+147| ... | ... | @@ -0,0 +1,147 @@ |
| 1 | # AGENT / MULTI-STREAM PROFILE -- for claude code and anything that fires | |
| 2 | # concurrent requests. see compose.patched.yaml for the 262k single-stream one. | |
| 3 | # | |
| 4 | # ./up.sh -f compose.agent.yaml up -d | |
| 5 | # | |
| 6 | # why a separate profile at all: on 24gb, 262k context and robust concurrency | |
| 7 | # are mutually exclusive. the long-context profile sits at 23.6/24.5gb with ONE | |
| 8 | # stream -- a second concurrent request has no activation headroom and the | |
| 9 | # engine dies with a CUDA OOM. measured, not theorised. | |
| 10 | # | |
| 11 | # THE PATCHED STACK -- runs on the CURRENT driver (550). no truenas upgrade | |
| 12 | # needed: all 13 patches are pure-python against vllm 0.27.1 and apply cleanly | |
| 13 | # to the -cu129 image (verified: 13/13 in sequence, vllm still imports). | |
| 14 | # | |
| 15 | # ./up.sh -f compose.patched.yaml up -d | |
| 16 | # | |
| 17 | # only one of vllm / vllm-patched can run at a time -- the model fills the card. | |
| 18 | # this service takes the `vllm` network alias so caddy's reverse_proxy keeps | |
| 19 | # working either way. | |
| 20 | # | |
| 21 | # what it unlocks over compose.yaml: quantized embeddings + MTP (~1.75gb -> | |
| 22 | # more context), int8 kv for spec-decode, the hybrid kv-group cap fix, the mtp | |
| 23 | # draft vocab (+10%), DFlash2, and KVarN 4/2-bit kv. | |
| 24 | name: vllm-agent | |
| 25 | ||
| 26 | services: | |
| 27 | vllm-agent: | |
| 28 | container_name: vllm-agent | |
| 29 | image: vllm/vllm-openai:v0.27.1-cu129 | |
| 30 | restart: unless-stopped | |
| 31 | ipc: host | |
| 32 | # same two driver-550 workarounds as the stock config | |
| 33 | tmpfs: | |
| 34 | - /usr/local/cuda/compat | |
| 35 | networks: | |
| 36 | home-infra: | |
| 37 | aliases: | |
| 38 | - vllm | |
| 39 | volumes: | |
| 40 | - "${APP_ROOT}/vllm/models:/models:rw" | |
| 41 | - "${APP_ROOT}/qwen38-stack:/stack:ro" | |
| 42 | - "${APP_ROOT}/vllm/cache-patched:/cache" | |
| 43 | - "./patched-entrypoint.sh:/patched-entrypoint.sh:ro" | |
| 44 | environment: | |
| 45 | HF_HUB_OFFLINE: "1" | |
| 46 | VLLM_API_KEY: "${VLLM_API_KEY}" | |
| 47 | NVIDIA_DISABLE_REQUIRE: "1" | |
| 48 | PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" | |
| 49 | HOME: "/cache" | |
| 50 | VLLM_NO_USAGE_STATS: "1" | |
| 51 | entrypoint: ["/patched-entrypoint.sh"] | |
| 52 | deploy: | |
| 53 | resources: | |
| 54 | reservations: | |
| 55 | devices: | |
| 56 | - driver: nvidia | |
| 57 | count: 1 | |
| 58 | capabilities: [gpu] | |
| 59 | command: | |
| 60 | - "--model" | |
| 61 | - "/models/Qwen3.8-27B-W4A16-AutoRound" | |
| 62 | - "--served-model-name" | |
| 63 | - "qwen3.8-27b" | |
| 64 | - "--host" | |
| 65 | - "0.0.0.0" | |
| 66 | - "--port" | |
| 67 | - "8000" | |
| 68 | # 128k. NOTE this is prompt + max_tokens, not prompt alone: clients | |
| 69 | # reserve their output budget against it. claude code sends a fixed | |
| 70 | # max_tokens=32768, so at 98304 the usable prompt was only 65536 and it | |
| 71 | # failed by ONE token on a 65537-token prompt. at 131072 the usable | |
| 72 | # prompt is 98304. the kv pool (137,625) already covered this, so the | |
| 73 | # raise is free -- no extra vram, no loss of concurrency headroom. | |
| 74 | - "--max-model-len" | |
| 75 | - "131072" | |
| 76 | # 0.90, not 0.97: concurrent requests need transient activation memory. | |
| 77 | # this ~1.7gb of slack is the whole point of this profile. | |
| 78 | - "--gpu-memory-utilization" | |
| 79 | - "0.90" | |
| 80 | # MUST be explicit. left unset, vllm auto-sizes kv to the whole budget and | |
| 81 | # then OOMs during cuda graph capture (it asks for 800mb it does not have) | |
| 82 | # and restart-loops, recompiling each time. weights are now ~15.1gb after | |
| 83 | # the lm_head/embed/mtp requant, +1.8gb peak activation, so ~5gb is what | |
| 84 | # is actually free for kv once graphs are paid for. | |
| 85 | # 4.5gb not 5gb: at 5gb the pool is 300k tokens but a ~250k-token request | |
| 86 | # has no room left for transient GDN state + activations and the engine | |
| 87 | # dies with a 24mb OOM. 4.5gb still yields >262k tokens of pool and keeps | |
| 88 | # ~0.5gb of transient headroom. | |
| 89 | # scheduler fairness: the auto-chosen step budget is 2048, which a single | |
| 90 | # long prefill consumes entirely -- a concurrent small request (e.g. an | |
| 91 | # agent's classifier call) then advances ~1 token per slow step and times | |
| 92 | # out. a larger budget plus a per-prefill cap leaves room in the same | |
| 93 | # step for other requests' decodes. | |
| 94 | - "--max-num-batched-tokens" | |
| 95 | - "4096" | |
| 96 | - "--long-prefill-token-threshold" | |
| 97 | - "1024" | |
| 98 | - "--kv-cache-memory" | |
| 99 | - "3221225472" | |
| 100 | # KVarN sizes an fp16 "tail pool" from max_num_seqs -- it capped 256->233 | |
| 101 | # on its own and still OOM'd. we are single-user, so 8 concurrent slots is | |
| 102 | # plenty and it frees several gb of tail pool for actual kv capacity. | |
| 103 | # 8, not 16: KVarN sizes an fp16 tail pool from max_num_seqs, so doubling | |
| 104 | # this OOMs on top of the larger step budget. 8 concurrent streams is | |
| 105 | # ample for an agent client (main call + classifier + a couple of tools). | |
| 106 | - "--max-num-seqs" | |
| 107 | - "8" | |
| 108 | - "--compilation-config" | |
| 109 | - '{"cudagraph_mode":"FULL_DECODE_ONLY"}' | |
| 110 | # KVarN: 4-bit keys / 2-bit values, ~4x more tokens per byte than fp8. | |
| 111 | # this is what takes context from ~123k to the model's native 262k on the | |
| 112 | # same 5gb pool. lossy -- verify with a needle test, not just a boot. | |
| 113 | - "--kv-cache-dtype" | |
| 114 | - "kvarn_k4v2_g128" | |
| 115 | - "--mamba-cache-mode" | |
| 116 | - "align" | |
| 117 | - "--reasoning-parser" | |
| 118 | - "qwen3" | |
| 119 | # bound thinking by default. the chat template defaults reasoning_effort | |
| 120 | # to *xhigh* when the client sends nothing, and claude code sends nothing. | |
| 121 | # at xhigh a hard prompt burns >8k tokens inside the <think> block, hits | |
| 122 | # max_tokens before emitting </think>, and returns a thinking block with | |
| 123 | # NO text block at all -- i.e. an empty answer. measured: xhigh at 8192 | |
| 124 | # output = unusable; medium and low both finish cleanly with real answers. | |
| 125 | - "--default-chat-template-kwargs" | |
| 126 | - '{"reasoning_effort":"medium"}' | |
| 127 | - "--enable-auto-tool-choice" | |
| 128 | - "--tool-call-parser" | |
| 129 | - "qwen3_coder" | |
| 130 | - "--enable-prefix-caching" | |
| 131 | - "--speculative-config" | |
| 132 | # adaptive speculation -- exactly the "fast alone, scales under load" | |
| 133 | # behaviour: full 3-token drafting at batch 1, tapering to none past 4 | |
| 134 | # streams where rejected drafts are just wasted compute that could be | |
| 135 | # serving real tokens. | |
| 136 | - '{"method":"mtp","num_speculative_tokens":3,"num_speculative_tokens_per_batch_size":[[1,1,3],[2,2,2],[3,4,1],[5,256,0]]}' | |
| 137 | healthcheck: | |
| 138 | test: ["CMD-SHELL", "python3 -c \"import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health')\""] | |
| 139 | interval: 30s | |
| 140 | timeout: 10s | |
| 141 | retries: 3 | |
| 142 | start_period: 20m | |
| 143 | ||
| 144 | networks: | |
| 145 | home-infra: | |
| 146 | external: true | |
| 147 | name: home-infra_default |
vllm/compose.patched.yaml created+111| ... | ... | @@ -0,0 +1,111 @@ |
| 1 | # LONG-CONTEXT PROFILE -- single stream. see compose.agent.yaml for the | |
| 2 | # concurrency-tuned profile (claude code / multi-stream). | |
| 3 | # | |
| 4 | # THE PATCHED STACK -- runs on the CURRENT driver (550). no truenas upgrade | |
| 5 | # needed: all 13 patches are pure-python against vllm 0.27.1 and apply cleanly | |
| 6 | # to the -cu129 image (verified: 13/13 in sequence, vllm still imports). | |
| 7 | # | |
| 8 | # ./up.sh -f compose.patched.yaml up -d | |
| 9 | # | |
| 10 | # only one of vllm / vllm-patched can run at a time -- the model fills the card. | |
| 11 | # this service takes the `vllm` network alias so caddy's reverse_proxy keeps | |
| 12 | # working either way. | |
| 13 | # | |
| 14 | # what it unlocks over compose.yaml: quantized embeddings + MTP (~1.75gb -> | |
| 15 | # more context), int8 kv for spec-decode, the hybrid kv-group cap fix, the mtp | |
| 16 | # draft vocab (+10%), DFlash2, and KVarN 4/2-bit kv. | |
| 17 | name: vllm-patched | |
| 18 | ||
| 19 | services: | |
| 20 | vllm-patched: | |
| 21 | container_name: vllm-patched | |
| 22 | image: vllm/vllm-openai:v0.27.1-cu129 | |
| 23 | restart: unless-stopped | |
| 24 | ipc: host | |
| 25 | # same two driver-550 workarounds as the stock config | |
| 26 | tmpfs: | |
| 27 | - /usr/local/cuda/compat | |
| 28 | networks: | |
| 29 | home-infra: | |
| 30 | aliases: | |
| 31 | - vllm | |
| 32 | volumes: | |
| 33 | - "${APP_ROOT}/vllm/models:/models:rw" | |
| 34 | - "${APP_ROOT}/qwen38-stack:/stack:ro" | |
| 35 | - "${APP_ROOT}/vllm/cache-patched:/cache" | |
| 36 | - "./patched-entrypoint.sh:/patched-entrypoint.sh:ro" | |
| 37 | environment: | |
| 38 | HF_HUB_OFFLINE: "1" | |
| 39 | VLLM_API_KEY: "${VLLM_API_KEY}" | |
| 40 | NVIDIA_DISABLE_REQUIRE: "1" | |
| 41 | PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" | |
| 42 | HOME: "/cache" | |
| 43 | VLLM_NO_USAGE_STATS: "1" | |
| 44 | entrypoint: ["/patched-entrypoint.sh"] | |
| 45 | deploy: | |
| 46 | resources: | |
| 47 | reservations: | |
| 48 | devices: | |
| 49 | - driver: nvidia | |
| 50 | count: 1 | |
| 51 | capabilities: [gpu] | |
| 52 | command: | |
| 53 | - "--model" | |
| 54 | - "/models/Qwen3.8-27B-W4A16-AutoRound" | |
| 55 | - "--served-model-name" | |
| 56 | - "qwen3.8-27b" | |
| 57 | - "--host" | |
| 58 | - "0.0.0.0" | |
| 59 | - "--port" | |
| 60 | - "8000" | |
| 61 | - "--max-model-len" | |
| 62 | - "262144" | |
| 63 | - "--gpu-memory-utilization" | |
| 64 | - "0.97" | |
| 65 | # MUST be explicit. left unset, vllm auto-sizes kv to the whole budget and | |
| 66 | # then OOMs during cuda graph capture (it asks for 800mb it does not have) | |
| 67 | # and restart-loops, recompiling each time. weights are now ~15.1gb after | |
| 68 | # the lm_head/embed/mtp requant, +1.8gb peak activation, so ~5gb is what | |
| 69 | # is actually free for kv once graphs are paid for. | |
| 70 | # 4.5gb not 5gb: at 5gb the pool is 300k tokens but a ~250k-token request | |
| 71 | # has no room left for transient GDN state + activations and the engine | |
| 72 | # dies with a 24mb OOM. 4.5gb still yields >262k tokens of pool and keeps | |
| 73 | # ~0.5gb of transient headroom. | |
| 74 | - "--kv-cache-memory" | |
| 75 | - "4831838208" | |
| 76 | # KVarN sizes an fp16 "tail pool" from max_num_seqs -- it capped 256->233 | |
| 77 | # on its own and still OOM'd. we are single-user, so 8 concurrent slots is | |
| 78 | # plenty and it frees several gb of tail pool for actual kv capacity. | |
| 79 | # 8, not 16: KVarN sizes an fp16 tail pool from max_num_seqs, so doubling | |
| 80 | # this OOMs on top of the larger step budget. 8 concurrent streams is | |
| 81 | # ample for an agent client (main call + classifier + a couple of tools). | |
| 82 | - "--max-num-seqs" | |
| 83 | - "8" | |
| 84 | - "--compilation-config" | |
| 85 | - '{"cudagraph_mode":"FULL_DECODE_ONLY"}' | |
| 86 | # KVarN: 4-bit keys / 2-bit values, ~4x more tokens per byte than fp8. | |
| 87 | # this is what takes context from ~123k to the model's native 262k on the | |
| 88 | # same 5gb pool. lossy -- verify with a needle test, not just a boot. | |
| 89 | - "--kv-cache-dtype" | |
| 90 | - "kvarn_k4v2_g128" | |
| 91 | - "--mamba-cache-mode" | |
| 92 | - "align" | |
| 93 | - "--reasoning-parser" | |
| 94 | - "qwen3" | |
| 95 | - "--enable-auto-tool-choice" | |
| 96 | - "--tool-call-parser" | |
| 97 | - "qwen3_coder" | |
| 98 | - "--enable-prefix-caching" | |
| 99 | - "--speculative-config" | |
| 100 | - '{"method":"mtp","num_speculative_tokens":3}' | |
| 101 | healthcheck: | |
| 102 | test: ["CMD-SHELL", "python3 -c \"import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health')\""] | |
| 103 | interval: 30s | |
| 104 | timeout: 10s | |
| 105 | retries: 3 | |
| 106 | start_period: 20m | |
| 107 | ||
| 108 | networks: | |
| 109 | home-infra: | |
| 110 | external: true | |
| 111 | name: home-infra_default |
vllm/compose.yaml created+125| ... | ... | @@ -0,0 +1,125 @@ |
| 1 | # qwen3.8-27b inference endpoint, single rtx 3090 (ampere, 24gb) | |
| 2 | # | |
| 3 | # separate compose project, but joins home-infra_default so caddy can reach it | |
| 4 | # at http://vllm:8000. run through ./up.sh -- interpolation vars come from the | |
| 5 | # parent .env, which only exists on the nas. | |
| 6 | name: vllm | |
| 7 | ||
| 8 | services: | |
| 9 | vllm: # port 8000 | |
| 10 | container_name: vllm | |
| 11 | # pinned: the patched stack we want to benchmark against targets 0.27.1 | |
| 12 | # exactly, so keeping stock on the same base keeps that an apples-to-apples | |
| 13 | # comparison rather than a version diff. | |
| 14 | # | |
| 15 | # cu129, NOT the default tag: the default is built on cuda 13, which needs | |
| 16 | # driver 580+. truenas 25.04 ships 550.142 (cuda 12.4), so cuda 13 is a | |
| 17 | # hard no. cuda 12.9 runs on 550 via cuda minor-version compatibility. | |
| 18 | image: vllm/vllm-openai:v0.27.1-cu129 | |
| 19 | restart: unless-stopped | |
| 20 | # vllm's workers talk over shared memory; the default 64mb shm is not enough | |
| 21 | ipc: host | |
| 22 | # the cu129 image ships nvidia's forward-compat libcuda (575) under | |
| 23 | # /usr/local/cuda/compat. that path is datacenter-only -- on a geforce card | |
| 24 | # it fails with cuda error 804 ("forward compatibility attempted on non | |
| 25 | # supported HW") before the engine ever loads. masking the dir with an | |
| 26 | # empty tmpfs forces the real 550 driver to be used, which is what we want: | |
| 27 | # cuda 12.9 on a 12.4 driver is covered by minor-version compatibility. | |
| 28 | tmpfs: | |
| 29 | - /usr/local/cuda/compat | |
| 30 | networks: | |
| 31 | - home-infra | |
| 32 | volumes: | |
| 33 | - "${APP_ROOT}/vllm/models:/models:ro" | |
| 34 | - "${APP_ROOT}/vllm/cache:/root/.cache" | |
| 35 | environment: | |
| 36 | # weights are already on disk; never let it reach for the hub at boot | |
| 37 | HF_HUB_OFFLINE: "1" | |
| 38 | # bearer token for the openai-compatible api | |
| 39 | VLLM_API_KEY: "${VLLM_API_KEY}" | |
| 40 | # nvidia-container-cli gates on the image's declared cuda>=12.9 against | |
| 41 | # the driver's reported 12.4 and refuses to start. the gate is stricter | |
| 42 | # than reality -- minor-version compat covers this -- so bypass it. | |
| 43 | # revisit if the truenas driver ever moves to 580+. | |
| 44 | NVIDIA_DISABLE_REQUIRE: "1" | |
| 45 | # weights nearly fill the card; reduce allocator fragmentation | |
| 46 | PYTORCH_CUDA_ALLOC_CONF: "expandable_segments:True" | |
| 47 | deploy: | |
| 48 | resources: | |
| 49 | reservations: | |
| 50 | devices: | |
| 51 | - driver: nvidia | |
| 52 | count: 1 | |
| 53 | capabilities: [gpu] | |
| 54 | # list form on purpose: --speculative-config takes json, and the folded | |
| 55 | # string form would mangle the quoting. | |
| 56 | command: | |
| 57 | - "--model" | |
| 58 | - "/models/Qwen3.8-27B-W4A16-AutoRound" | |
| 59 | - "--served-model-name" | |
| 60 | - "qwen3.8-27b" | |
| 61 | - "--host" | |
| 62 | - "0.0.0.0" | |
| 63 | - "--port" | |
| 64 | - "8000" | |
| 65 | # 19.5gb of weights on a 24gb card leaves little for kv. start | |
| 66 | # conservative so it boots, then walk this up while watching the kv pool | |
| 67 | # size vllm prints at startup. | |
| 68 | # lm_head was requantized to int8 (see readme), which freed ~1.3gb of | |
| 69 | # weights and bought this jump. codex spends ~10k tokens on its system | |
| 70 | # prompt + tool defs before any of your code, so headroom matters. | |
| 71 | # raised to just under the 105,151-token kv pool. this costs no memory -- | |
| 72 | # max-model-len only caps a single request, so the only price is max | |
| 73 | # concurrency dropping to ~1.0x, which is irrelevant for single-user use. | |
| 74 | - "--max-model-len" | |
| 75 | - "102400" | |
| 76 | # nothing else shares this gpu, so take almost all of it | |
| 77 | - "--gpu-memory-utilization" | |
| 78 | - "0.97" | |
| 79 | # cuda graphs matter enormously here: without them, kernel-launch | |
| 80 | # overhead dominates decode and throughput drops to ~37 tok/s. rather | |
| 81 | # than --enforce-eager, cap the kv cache and spend the freed vram on | |
| 82 | # decode-only graph capture. note vllm silently downgrades this to | |
| 83 | # PIECEWISE because flashinfer + spec-decode cannot do full decode | |
| 84 | # graphs; piecewise still gets us most of the win. | |
| 85 | - "--kv-cache-memory" | |
| 86 | - "4831838208" | |
| 87 | - "--compilation-config" | |
| 88 | - '{"cudagraph_mode":"FULL_DECODE_ONLY"}' | |
| 89 | # text-only serving: do not reserve multimodal buffers. the vision tower | |
| 90 | # weights still load, we just never budget for image/video inputs. | |
| 91 | - "--limit-mm-per-prompt" | |
| 92 | - '{"image":0,"video":0}' | |
| 93 | # only 16 of 64 layers use full attention (the rest are linear), so the | |
| 94 | # kv cache is far smaller than a normal 27b -- fp8 storage stretches it | |
| 95 | # further at no measurable quality cost. | |
| 96 | - "--kv-cache-dtype" | |
| 97 | - "fp8" | |
| 98 | # required by the qwen3.8 gated-deltanet / mtp serving path | |
| 99 | - "--mamba-cache-mode" | |
| 100 | - "align" | |
| 101 | - "--reasoning-parser" | |
| 102 | - "qwen3" | |
| 103 | - "--enable-auto-tool-choice" | |
| 104 | - "--tool-call-parser" | |
| 105 | - "qwen3_coder" | |
| 106 | - "--enable-prefix-caching" | |
| 107 | # multi-token prediction: the single biggest speed lever available in | |
| 108 | # stock vllm. ~46 tok/s -> ~114 tok/s single-stream. | |
| 109 | - "--speculative-config" | |
| 110 | - '{"method":"mtp","num_speculative_tokens":3}' | |
| 111 | healthcheck: | |
| 112 | test: ["CMD-SHELL", "python3 -c \"import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health')\""] | |
| 113 | interval: 30s | |
| 114 | timeout: 10s | |
| 115 | retries: 3 | |
| 116 | # weight load + cuda graph capture is slow on first boot | |
| 117 | start_period: 15m | |
| 118 | labels: | |
| 119 | net.paperclover.list.name: Qwen3.8 27B | |
| 120 | net.paperclover.list.web: "false" | |
| 121 | ||
| 122 | networks: | |
| 123 | home-infra: | |
| 124 | external: true | |
| 125 | name: home-infra_default |
vllm/patched-entrypoint.sh created+36| ... | ... | @@ -0,0 +1,36 @@ |
| 1 | #!/bin/bash | |
| 2 | # Runs the syv-ai patch stack against the official vllm image's site-packages at | |
| 3 | # container start, then serves. This exists because `docker build` is not | |
| 4 | # available to us here -- patching in the entrypoint gets the same result without | |
| 5 | # baking an image. Patch application is idempotent and takes seconds; the slow | |
| 6 | # parts (FlashInfer JIT, torch.compile) cache into /cache. | |
| 7 | # | |
| 8 | # IMPORTANT: this runs against the -cu129 image, NOT the cuda-13 one the repo's | |
| 9 | # Dockerfile builds. All 13 patches are pure-python patches against vllm 0.27.1 | |
| 10 | # and apply cleanly to the cu129 build, so the whole patch stack works on | |
| 11 | # driver 550 -- no truenas upgrade needed. The repo's cuda-13 requirement comes | |
| 12 | # from `pip install vllm` pulling torch+cu130 in a fresh build, not from the | |
| 13 | # patches themselves. | |
| 14 | set -e | |
| 15 | STACK=${STACK_DIR:-/stack} | |
| 16 | SP=$(python3 -c 'import vllm, os; print(os.path.dirname(vllm.__file__))' | tail -n1) | |
| 17 | echo "vllm site-packages: $SP" | |
| 18 | ||
| 19 | if [ ! -f "$SP/.syvai-patched" ]; then | |
| 20 | echo "== applying $(ls "$STACK"/patches/*.patch | wc -l) patches" | |
| 21 | for p in "$STACK"/patches/*.patch; do | |
| 22 | echo "-- $(basename "$p")" | |
| 23 | patch -p1 -d "$SP" < "$p" | |
| 24 | done | |
| 25 | echo "== kvarn/install.sh" | |
| 26 | # PY override: the script defaults to $REPO/venv/bin/python (their venv | |
| 27 | # layout); in the official image vllm lives in the system interpreter. | |
| 28 | # /stack is read-only, so copy out first -- install.sh writes into its own dir. | |
| 29 | cp -r "$STACK" /stack-rw | |
| 30 | ( cd /stack-rw && PY=python3 bash kvarn/install.sh ) | |
| 31 | touch "$SP/.syvai-patched" | |
| 32 | else | |
| 33 | echo "== patches already applied" | |
| 34 | fi | |
| 35 | ||
| 36 | exec python3 -m vllm.entrypoints.openai.api_server "$@" |
vllm/readme.md created+275| ... | ... | @@ -0,0 +1,275 @@ |
| 1 | # qwen3.8-27b on the 3090 | |
| 2 | ||
| 3 | openai-compatible inference endpoint, stock vllm, served at | |
| 4 | `https://ai.{$HOME_DOMAIN}` and as `http://vllm:8000` to other containers. | |
| 5 | ||
| 6 | auth is vllm's own bearer token (`VLLM_API_KEY` in the nas `.env`). the | |
| 7 | `reverse_proxy_auth` caddy snippet is deliberately NOT used here: it does an | |
| 8 | oauth2 browser redirect, which api clients cannot follow. | |
| 9 | ||
| 10 | ## why this model / quant | |
| 11 | ||
| 12 | - `dbirks/Qwen3.8-27B-W4A16-AutoRound`, ~19.5gb on disk, in `${APP_ROOT}/vllm/models`. | |
| 13 | - **4-bit, not 5-bit.** 28b params against a hard 24gb wall: 5-bit weights land | |
| 14 | near 23gb and leave nothing for kv. the autoround 4-bit measures within the | |
| 15 | confidence interval of bf16 on gsm8k / humaneval / mmlu-pro, so 5-bit has no | |
| 16 | accuracy left to buy back. | |
| 17 | - **no fp8 / nvfp4.** the 3090 is ampere: no fp8 tensor cores (ada+), no fp4 | |
| 18 | (blackwell), and 8-bit does not fit regardless. | |
| 19 | ||
| 20 | ## three ampere/truenas gotchas, all load-bearing | |
| 21 | ||
| 22 | 1. **cuda 13 is impossible here.** the default `vllm/vllm-openai` tag is built | |
| 23 | on cuda 13, which needs driver 580+. truenas 25.04 ships 550.142 (cuda 12.4). | |
| 24 | hence the `-cu129` tag: cuda 12.9 runs on a 12.4 driver under minor-version | |
| 25 | compatibility. | |
| 26 | 2. **`NVIDIA_DISABLE_REQUIRE=1`.** nvidia-container-cli hard-gates on the | |
| 27 | image's declared `cuda>=12.9` vs the driver's reported 12.4 and refuses to | |
| 28 | start. the gate is stricter than reality. | |
| 29 | 3. **the compat tmpfs mask.** the cu129 image ships nvidia's forward-compat | |
| 30 | libcuda (575) at `/usr/local/cuda/compat`. that path is datacenter-only; on | |
| 31 | a geforce card it dies with cuda error 804 before the engine loads. masking | |
| 32 | the dir with an empty tmpfs forces the real 550 driver to be used. | |
| 33 | ||
| 34 | ## measured | |
| 35 | ||
| 36 | | config | single-stream | context | | |
| 37 | |---|---|---| | |
| 38 | | `--enforce-eager` (no cuda graphs) | 37 tok/s | 16k | | |
| 39 | | piecewise cuda graphs | 51-61 tok/s | 49k | | |
| 40 | | + int8 lm_head | 63-73 tok/s | 82k | | |
| 41 | | + max-model-len to pool limit | 71 tok/s | 102k | | |
| 42 | | + patch stack, int8 embed/mtp | 70.6 tok/s | 123k | | |
| 43 | | + mtp draft vocab | 73.1 tok/s | 123k | | |
| 44 | | **+ KVarN 4/2-bit kv (current)** | **89.8 tok/s** | **262k** | | |
| 45 | ||
| 46 | that is the model's full native context on one 3090, at ~2.2x the throughput we | |
| 47 | started with and ~2.2x claude opus's ~40 tok/s. | |
| 48 | ||
| 49 | **KVarN made it faster, not slower.** 4-bit keys / 2-bit values means far less | |
| 50 | memory traffic per attention step, and decode is bandwidth-bound -- so the | |
| 51 | lossy cache buys speed *and* context at once. that was not the expected result. | |
| 52 | ||
| 53 | verification (KVarN is lossy, so booting proves nothing): | |
| 54 | ||
| 55 | | check | result | | |
| 56 | |---|---| | |
| 57 | | needle @ 50,917 tok | found | | |
| 58 | | needle @ 151,381 tok | found | | |
| 59 | | needle @ 248,234 tok | found, engine survived | | |
| 60 | | 6-item reasoning battery | 5/6 no-think, 6/6 with thinking | | |
| 61 | ||
| 62 | the one no-think miss ("gives away a third, then eats 2") answers correctly | |
| 63 | with `enable_thinking` on, so it is a sampling artifact, not kv corruption. | |
| 64 | ||
| 65 | prefill: ~900 tok/s up to 150k, dropping to ~620 tok/s at 250k. a 250k prompt | |
| 66 | costs ~400s of prefill -- prefix caching is what makes long sessions usable. | |
| 67 | ||
| 68 | ### the int8 lm_head step | |
| 69 | ||
| 70 | the published quant leaves lm_head in bf16: a 2.37 GiB matrix over a 248k | |
| 71 | vocab, re-read every single decode step. requantizing it to int8 group-128 | |
| 72 | (`prepare/quant_lm_head.py` from the syv-ai repo, cpu-only, in place, keeps | |
| 73 | `.bak`) freed 1.3 gb and bought ~20% throughput -- more than the +12% the repo | |
| 74 | documents. round-trip error 0.64%; spot checks stayed correct. | |
| 75 | ||
| 76 | **it loads on stock vllm.** that was not obvious: the repo's docs/optimizations | |
| 77 | table marks every optimization "requires patch", but that is wrong for this | |
| 78 | one. it is plain compressed-tensors pack-quantized output. | |
| 79 | ||
| 80 | a zfs snapshot `storage1/apps@pre-requant` predates the change. | |
| 81 | ||
| 82 | memory at rest: ~16.5gb weights, ~1.8gb peak activation, 4.5gb kv cache | |
| 83 | (105,151 tokens) at max-model-len 81920 -> 1.28x concurrency. | |
| 84 | ||
| 85 | checkpoint breakdown (why there is still headroom), by safetensors header: | |
| 86 | ||
| 87 | | family | GiB | dtype | | |
| 88 | |---|---|---| | |
| 89 | | LM body | 11.73 | int4, already quantized | | |
| 90 | | embeddings | 2.37 | bf16 -- still unquantized | | |
| 91 | | lm_head | 2.37 -> 1.19 | now int8 | | |
| 92 | | vision tower | 0.86 | bf16, loaded but unused (text-only serving) | | |
| 93 | | MTP module | 0.79 | bf16 -- still unquantized | | |
| 94 | ||
| 95 | ## running | |
| 96 | ||
| 97 | ./up.sh up -d | |
| 98 | ./up.sh logs -f # first boot is slow: weight load + graph capture | |
| 99 | ./up.sh down | |
| 100 | ||
| 101 | ## two profiles: pick by workload | |
| 102 | ||
| 103 | on 24gb, max context and robust concurrency are **mutually exclusive**. the | |
| 104 | long-context profile sits at 23.6/24.5gb with ONE stream; a second concurrent | |
| 105 | request has no activation headroom and the engine dies with a CUDA OOM. that is | |
| 106 | measured, not theorised -- an 8-stream ladder killed it. | |
| 107 | ||
| 108 | | | `compose.patched.yaml` | `compose.agent.yaml` | | |
| 109 | |---|---|---| | |
| 110 | | context | 262,144 | 131,072 | | |
| 111 | | single stream | ~90 tok/s | ~83-86 tok/s | | |
| 112 | | 8 streams | dies | 44.5 tok/s each, 237 aggregate | | |
| 113 | | gpu-mem-util | 0.97 | 0.90 (the headroom IS the feature) | | |
| 114 | | use for | one-shot long analysis | claude code, anything concurrent | | |
| 115 | ||
| 116 | ./up.sh -f compose.agent.yaml up -d # multi-stream (default choice) | |
| 117 | ./up.sh -f compose.patched.yaml up -d # 262k single-stream | |
| 118 | ||
| 119 | ### concurrency ladder (agent profile, measured) | |
| 120 | ||
| 121 | | streams | per-stream | aggregate | | |
| 122 | |---|---|---| | |
| 123 | | 1 | 82.5 | 82.5 | | |
| 124 | | 2 | 49.3 | 98.7 | | |
| 125 | | 4 | 58.9 | 233.0 | | |
| 126 | | 8 | 44.5 | 236.8 | | |
| 127 | ||
| 128 | per-stream stays above claude opus's ~40 tok/s even at 8 concurrent. | |
| 129 | ||
| 130 | ### how vllm actually scales | |
| 131 | ||
| 132 | continuous batching: every scheduler step the engine picks up to | |
| 133 | `max_num_batched_tokens` tokens across *all* in-flight requests, so streams | |
| 134 | share the gpu rather than queueing. kv is paged and allocated on demand, so the | |
| 135 | "Maximum concurrency 1.14x" line is a worst-case figure (all requests at full | |
| 136 | length), NOT an admission limit -- short requests coexist fine. the real caps | |
| 137 | are `max_num_seqs` (8) and free kv blocks. | |
| 138 | ||
| 139 | `num_speculative_tokens_per_batch_size` gives "fast alone, scales under load": | |
| 140 | `[[1,1,3],[2,2,2],[3,4,1],[5,256,0]]` = full 3-token drafting at batch 1, | |
| 141 | tapering to none past 4 streams, where rejected drafts are just wasted compute. | |
| 142 | ||
| 143 | ### the starvation trap | |
| 144 | ||
| 145 | at the auto-chosen `max_num_batched_tokens=2048`, one long prefill consumes the | |
| 146 | entire step budget and a concurrent small request advances ~1 token per slow | |
| 147 | step. a 0.44s request became **7.27s** behind an 18k-token prefill -- which is | |
| 148 | why agent classifier calls time out. `--max-num-batched-tokens 4096` plus | |
| 149 | `--long-prefill-token-threshold 1024` caps how much of a step any single | |
| 150 | prefill may take, bringing that to 3.06s. | |
| 151 | ||
| 152 | ### max-model-len counts OUTPUT too | |
| 153 | ||
| 154 | `--max-model-len` bounds prompt + `max_tokens`, not the prompt alone. claude | |
| 155 | code sends a fixed `max_tokens=32768`, so at 98304 the usable prompt was only | |
| 156 | 65536 -- and it failed on a 65,537-token prompt by exactly one token. at 131072 | |
| 157 | the usable prompt is 98304. if you want more, lower the client's output | |
| 158 | reservation (`CLAUDE_CODE_MAX_OUTPUT_TOKENS`) rather than raising vram. | |
| 159 | ||
| 160 | claude code talks to `/v1/messages` (the anthropic API), which vllm serves | |
| 161 | alongside the openai routes. | |
| 162 | ||
| 163 | ## using it from claude code | |
| 164 | ||
| 165 | `~/Desktop/claude-qwen` sets `ANTHROPIC_BASE_URL` at this host and points every | |
| 166 | model alias at `qwen3.8-27b`. two things it must get right: | |
| 167 | ||
| 168 | - `CLAUDE_CODE_MAX_CONTEXT_TOKENS` / `CLAUDE_CODE_MAX_OUTPUT_TOKENS` must match | |
| 169 | the running profile. these are 131072 / 8192, giving 122,880 usable prompt. | |
| 170 | - claude code talks the anthropic protocol to `/v1/messages`, which vllm serves. | |
| 171 | ||
| 172 | two gotchas, both fixed server-side: | |
| 173 | ||
| 174 | 1. **`x-api-key` vs bearer.** vllm's `--api-key` only accepts | |
| 175 | `Authorization: Bearer`; anthropic-protocol clients send `x-api-key` and got | |
| 176 | a flat 401. the caddy vhost now translates `x-api-key` into a bearer header, | |
| 177 | so both conventions work against the same key. unauthenticated still 401s. | |
| 178 | 2. **thinking defaulted to xhigh and returned EMPTY answers.** the chat | |
| 179 | template defaults `reasoning_effort` to xhigh when the client sends nothing, | |
| 180 | and claude code sends nothing. at xhigh a hard prompt burns >8k tokens inside | |
| 181 | the `<think>` block, hits max_tokens before emitting `</think>`, and comes | |
| 182 | back as a thinking block with **no text block at all**. measured at 8192 | |
| 183 | output: xhigh -> unusable, medium and low -> clean answers. the agent profile | |
| 184 | now sets `--default-chat-template-kwargs '{"reasoning_effort":"medium"}'`. | |
| 185 | ||
| 186 | ## using it from codex | |
| 187 | ||
| 188 | `~/.codex/config.toml` holds the provider, `~/.codex/local.config.toml` the | |
| 189 | profile (codex 0.149 split these; a `[profiles.x]` table in config.toml is now | |
| 190 | rejected). run with `codex --profile local`. | |
| 191 | ||
| 192 | export VLLM_API_KEY=... # same value as the nas .env | |
| 193 | codex --profile local | |
| 194 | ||
| 195 | three gotchas: | |
| 196 | - `wire_api = "responses"` is mandatory -- codex dropped chat-completions in | |
| 197 | feb 2026. vllm 0.27.1 does serve `/v1/responses`, so this works. | |
| 198 | - `codex exec` reads stdin by default and will hang forever looking like a | |
| 199 | model problem. redirect it: `codex exec ... < /dev/null`. | |
| 200 | - codex burns ~10k tokens on its system prompt and tool definitions before any | |
| 201 | of your code, which is why max-model-len is 48k rather than 32k. | |
| 202 | ||
| 203 | `--enable-auto-tool-choice --tool-call-parser qwen3_coder` are what make the | |
| 204 | agent loop work; without them codex can read but never act. | |
| 205 | ||
| 206 | ## tuning levers | |
| 207 | ||
| 208 | 1. `--kv-cache-memory` (currently 3.2gb) trades context against cuda graph | |
| 209 | memory. dropping graphs entirely costs ~40% throughput, so do not. | |
| 210 | 2. `--max-model-len` is 49152 against a 65,179-token pool. the model natively | |
| 211 | supports 262k; getting anywhere near that needs vram freed elsewhere. | |
| 212 | 3. `num_speculative_tokens` is 3, the measured default. read | |
| 213 | `vllm:spec_decode_num_{accepted,draft}_tokens_total` from /metrics before | |
| 214 | changing it -- throughput alone cannot tell a working drafter from one that | |
| 215 | loaded and got ignored. | |
| 216 | ||
| 217 | ## the patched stack does NOT need a truenas upgrade | |
| 218 | ||
| 219 | this was the session's biggest wrong turn, so it is worth stating plainly. the | |
| 220 | repo's Dockerfile is `FROM nvidia/cuda:13.0.1` and `vllm==0.27.1` pulls torch | |
| 221 | 2.13+cu130, so it looks like the patch stack requires driver 580+ and therefore | |
| 222 | a truenas 25.10 upgrade. it does not. **all 13 patches are pure-python patches | |
| 223 | against vllm 0.27.1 and apply cleanly, in sequence, to the `-cu129` image** -- | |
| 224 | verified 13/13 with vllm still importing. the cuda 13 dependency comes from | |
| 225 | building vllm from scratch, not from the patches. | |
| 226 | ||
| 227 | so `compose.patched.yaml` runs the entire patch stack on driver 550. | |
| 228 | `patched-entrypoint.sh` applies the patches to site-packages at container start | |
| 229 | (we have no `docker build` here), which costs a few seconds and is idempotent. | |
| 230 | ||
| 231 | a truenas 25.10 upgrade is still worth doing eventually -- it ships driver | |
| 232 | 580.173.02 / cuda 13, which would let us drop `NVIDIA_DISABLE_REQUIRE` and the | |
| 233 | `/usr/local/cuda/compat` tmpfs mask -- but it buys no capability we do not | |
| 234 | already have. | |
| 235 | ||
| 236 | ## IMPORTANT: the checkpoint is now patched-stack-only | |
| 237 | ||
| 238 | `quant_embed.py` wrote packed int8 embeddings, which stock vllm cannot load | |
| 239 | (that is what `patches/qwen3_5-embed-quant.patch` exists for). **`compose.yaml` | |
| 240 | will no longer start against this model dir.** to go back to stock, restore | |
| 241 | `storage1/apps@pre-embed-quant` or the `.bak` files the prepare scripts left. | |
| 242 | ||
| 243 | zfs snapshots, oldest first: `@pre-requant` (before any requant), | |
| 244 | `@pre-embed-quant`, `@pre-draftvocab`. | |
| 245 | ||
| 246 | ## two settings that are load-bearing and non-obvious | |
| 247 | ||
| 248 | - **`--kv-cache-memory` must be explicit.** left unset, vllm auto-sizes kv to | |
| 249 | the whole budget, then OOMs during cuda graph capture and restart-loops, | |
| 250 | recompiling every time. it is set to 4.5gb, not 5gb: at 5gb the pool is 300k | |
| 251 | tokens but a ~250k request has no room for transient GDN state and the engine | |
| 252 | dies on a 24mb allocation. | |
| 253 | - **`--max-num-seqs 8`.** KVarN sizes an fp16 "tail pool" from max_num_seqs; at | |
| 254 | the default it capped itself to 233 and still OOM'd. we are single-user, so 8 | |
| 255 | slots frees gigabytes for actual kv capacity. | |
| 256 | ||
| 257 | ## what is still on the table | |
| 258 | ||
| 259 | - **DFlash2 drafter** (+3-10% throughput) is a real trade, not free: the | |
| 260 | quantized drafter is ~1.19gb, and at ~60k pool tokens per gb that costs | |
| 261 | ~72k tokens of context -- dropping max-model-len from 262k to roughly 195k. | |
| 262 | 262,144 is the model's *native* context, so spending a quarter of it for ~5% | |
| 263 | speed is probably the wrong side of the trade. left off deliberately. | |
| 264 | - **dropping the vision tower** (0.86gb, still bf16, never used since we serve | |
| 265 | text-only) is the way to get DFlash2 *and* keep 262k. it means checkpoint | |
| 266 | surgery on a Qwen3VL config, which is the riskiest remaining step. | |
| 267 | - **truenas 25.10** to drop the two driver-550 workarounds. | |
| 268 | ||
| 269 | ## the gap to ~114 tok/s | |
| 270 | ||
| 271 | the syv-ai/qwen38-27b-rtx3090 stack reports 114-133 tok/s single-stream. the | |
| 272 | difference is not magic: it requantizes lm_head, embeddings and the mtp module | |
| 273 | to int4, which frees enough vram for full cuda graphs plus a 66.7k kv pool, and | |
| 274 | it patches vllm 0.27.1 for dflash2 block drafting. that is the benchmark | |
| 275 | target for the alternate stack -- same 0.27.1 base, so the comparison is clean. |
vllm/up.sh created+6| ... | ... | @@ -0,0 +1,6 @@ |
| 1 | #!/bin/sh | |
| 2 | # compose wrapper: this project lives in a subfolder but its interpolation | |
| 3 | # vars (APP_ROOT, VLLM_API_KEY) live in the parent .env, which is nas-only. | |
| 4 | set -e | |
| 5 | cd "$(dirname "$0")" | |
| 6 | exec sudo docker compose --env-file ../.env "$@" |