| author | |
| committer | |
| log | 189fed0e01939f288c29955e66b5a4c8713abc2b |
| tree | c7299045f3d04bd16b876ea710b23554ab8a1944 |
| parent | 3c2e26a1d96a4046f45ebc1a19ee87dda5651e64 |
| signature | Signed by SSH key SHA256:52mNGHRsVFBDED9IAX5pe+LRWUefqTbxEReunq21QvU |
Install the physical NixOS configuration, preserve the original host identity and TPM unlock, and serve the dashboard at snowglobe.paperclover.net. Import verified PostgreSQL and app state before activation, merge personal Forgejo histories into Shale with original privacy, and map the reorganized Media folders into the retained applications.
Filter excluded services before discovery, evaluation, and release copying. Preserve original Samba account hashes and OIDC issuer identity, retain every PDS actor store, and verify application copies and database contents against their sources.
Validation: physical ZFS permission rehearsal and shared-group probes; full Keycloak and Dawarich database fingerprints; all 19 Forgejo object/ref checks plus smart HTTP serving; file checksums and SQLite integrity across retained personal apps; live service health and public HTTPS checks.
Assisted-by: gpt-6.1-sol43 files changed, 1411 insertions(+), 98 deletions(-)
config/excluded-services.json created+4| ... | ... | @@ -0,0 +1,4 @@ |
| 1 | [ | |
| 2 | "evil-forgejo", | |
| 3 | "evil-hedgedoc" | |
| 4 | ] |
config/site.pkl+8| ... | ... | @@ -12,6 +12,14 @@ cloverUid: Int = (read?("prop:cloverUid") ?? read?("env:STUDIO_CLOVER_UID") ?? " |
| 12 | 12 | cloverGid: Int = (read?("prop:cloverGid") ?? read?("env:STUDIO_CLOVER_GID") ?? "3000").toInt() |
| 13 | 13 | cloverReadOnly: Boolean = (read?("prop:cloverReadOnly") ?? read?("env:STUDIO_CLOVER_READ_ONLY") ?? "false") == "true" |
| 14 | 14 | mediaRoot: String = read?("prop:mediaRoot") ?? read?("env:STUDIO_MEDIA_ROOT") ?? "\(cloverRoot)/Media" |
| 15 | jellyfinFolders: Mapping<String, String> = new { | |
| 16 | ["Anime"] = "Anime" | |
| 17 | ["Movies"] = "Movies" | |
| 18 | ["Shows"] = "Shows" | |
| 19 | ["Indie Shows"] = "Indie Shows" | |
| 20 | ["Independent"] = "Videos/Independent" | |
| 21 | ["Paper Clover"] = "Videos/Paper Clover" | |
| 22 | } | |
| 15 | 23 | mediaReadOnly: Boolean = (read?("prop:mediaReadOnly") ?? read?("env:STUDIO_MEDIA_READ_ONLY") ?? "false") == "true" |
| 16 | 24 | tlsInternal: Boolean = (read?("prop:tlsInternal") ?? read?("env:STUDIO_TLS_INTERNAL") ?? "false") == "true" || domain.endsWith(".test") |
| 17 | 25 | preview: Boolean = (read?("prop:preview") ?? "false") == "true" |
dashboard/server/youtube-worker.py+5-5| ... | ... | @@ -29,13 +29,13 @@ SMTP_USER = os.environ.get("SMTP_USER", "") |
| 29 | 29 | SMTP_PASS = os.environ.get("SMTP_PASS", "") |
| 30 | 30 | MAIL_FROM = os.environ.get("MAIL_FROM", f"yt-feed@{os.environ.get('STUDIO_DOMAIN', 'studio.test')}") |
| 31 | 31 | MAIL_TO = os.environ.get("MAIL_TO", os.environ.get("STUDIO_OWNER_EMAIL", "account@paperclover.net")) |
| 32 | BASE_URL = os.environ.get("BASE_URL", f"https://globe.{os.environ.get('STUDIO_DOMAIN', 'studio.test')}/youtube").rstrip("/") | |
| 32 | BASE_URL = os.environ.get("BASE_URL", f"https://snowglobe.{os.environ.get('STUDIO_DOMAIN', 'studio.test')}/youtube").rstrip("/") | |
| 33 | 33 | READ_ONLY = os.environ.get("STUDIO_MEDIA_READ_ONLY") == "true" |
| 34 | 34 | |
| 35 | 35 | MEDIA_ROOT = os.environ.get("STUDIO_YT_MEDIA", "/srv/clover/Media") |
| 36 | INDIE_DIR = os.path.join(MEDIA_ROOT, "jellyfin/Indie Shows") | |
| 37 | INDEP_DIR = os.path.join(MEDIA_ROOT, "jellyfin/Independent") | |
| 38 | MUSIC_DIR = os.path.join(MEDIA_ROOT, "music_intake") | |
| 36 | INDIE_DIR = os.path.join(MEDIA_ROOT, "Indie Shows") | |
| 37 | INDEP_DIR = os.path.join(MEDIA_ROOT, "Videos/Independent") | |
| 38 | MUSIC_DIR = os.path.join(MEDIA_ROOT, "Intake - Music") | |
| 39 | 39 | VIDEO_EXTS = (".webm", ".mp4", ".mkv") |
| 40 | 40 | |
| 41 | 41 | ATOM = "{http://www.w3.org/2005/Atom}" |
| ... | ... | @@ -614,7 +614,7 @@ def command(name, args): |
| 614 | 614 | ep_title=choice["title"], |
| 615 | 615 | dest_label=f"{choice['show']} S{choice['season']:02d}E{choice['episode']:02d}") |
| 616 | 616 | else: |
| 617 | job["dest_label"] = "Independent" if choice["dest"] == "independent" else "music_intake" | |
| 617 | job["dest_label"] = "Independent" if choice["dest"] == "independent" else "Intake - Music" | |
| 618 | 618 | jobs = load_json("jobs.json", []) |
| 619 | 619 | jobs.insert(0, job) |
| 620 | 620 | save_json("jobs.json", jobs[:50]) |
dashboard/src/core.rs+95-5| ... | ... | @@ -1,6 +1,7 @@ |
| 1 | 1 | use crate::*; |
| 2 | 2 | use futures::{StreamExt, stream}; |
| 3 | 3 | use sha1::{Digest, Sha1}; |
| 4 | use std::path::Path; | |
| 4 | 5 | |
| 5 | 6 | pub async fn nomad(app: &App, path: &str) -> Result<Value> { |
| 6 | 7 | let value = nomad_raw(app, path).await?; |
| ... | ... | @@ -221,12 +222,25 @@ pub async fn service_file(app: &App, id: &str) -> Result<PathBuf> { |
| 221 | 222 | if !valid_id(id) { |
| 222 | 223 | return Err(Error::new(400, "Invalid service ID.")); |
| 223 | 224 | } |
| 225 | let excluded = excluded_services(&app.repo).await?; | |
| 226 | if excluded.iter().any(|service| service == id) { | |
| 227 | return Err(Error::new( | |
| 228 | 404, | |
| 229 | format!("No service is named {id}. Pick one from the sidebar."), | |
| 230 | )); | |
| 231 | } | |
| 224 | 232 | let direct = app.repo.join("service").join(id).join("service.pkl"); |
| 225 | 233 | if tokio::fs::try_exists(&direct).await? { |
| 226 | 234 | return Ok(direct); |
| 227 | 235 | } |
| 228 | 236 | let mut dirs = tokio::fs::read_dir(app.repo.join("service")).await?; |
| 229 | 237 | while let Some(dir) = dirs.next_entry().await? { |
| 238 | if excluded | |
| 239 | .iter() | |
| 240 | .any(|service| dir.file_name() == service.as_str()) | |
| 241 | { | |
| 242 | continue; | |
| 243 | } | |
| 230 | 244 | let file = dir.path().join(format!("{id}.pkl")); |
| 231 | 245 | if tokio::fs::try_exists(&file).await? { |
| 232 | 246 | return Ok(file); |
| ... | ... | @@ -438,15 +452,58 @@ async fn issues(app: Arc<App>) -> Result<Value> { |
| 438 | 452 | Ok(json!(issues)) |
| 439 | 453 | } |
| 440 | 454 | |
| 455 | async fn excluded_services(repo: &Path) -> Result<Vec<String>> { | |
| 456 | Ok(serde_json::from_slice( | |
| 457 | &tokio::fs::read(repo.join("config/excluded-services.json")).await?, | |
| 458 | )?) | |
| 459 | } | |
| 460 | ||
| 461 | async fn launcher_module(repo: &Path) -> Result<String> { | |
| 462 | let excluded = excluded_services(repo).await?; | |
| 463 | let mut files = Vec::new(); | |
| 464 | let mut directories = tokio::fs::read_dir(repo.join("service")).await?; | |
| 465 | while let Some(directory) = directories.next_entry().await? { | |
| 466 | let directory_name = directory.file_name().to_string_lossy().into_owned(); | |
| 467 | if excluded.contains(&directory_name) || !directory.file_type().await?.is_dir() { | |
| 468 | continue; | |
| 469 | } | |
| 470 | let mut entries = tokio::fs::read_dir(directory.path()).await?; | |
| 471 | while let Some(entry) = entries.next_entry().await? { | |
| 472 | let path = entry.path(); | |
| 473 | if !path.extension().is_some_and(|extension| extension == "pkl") { | |
| 474 | continue; | |
| 475 | } | |
| 476 | let name = if entry.file_name() == "service.pkl" { | |
| 477 | directory_name.clone() | |
| 478 | } else { | |
| 479 | path.file_stem().unwrap().to_string_lossy().into_owned() | |
| 480 | }; | |
| 481 | if !excluded.contains(&name) { | |
| 482 | files.push(path); | |
| 483 | } | |
| 484 | } | |
| 485 | } | |
| 486 | files.sort(); | |
| 487 | let mut module = String::new(); | |
| 488 | let mut services = Vec::new(); | |
| 489 | for (index, path) in files.iter().enumerate() { | |
| 490 | let uri = serde_json::to_string(&url::Url::from_file_path(path).unwrap().as_str())?; | |
| 491 | module.push_str(&format!("import {uri} as service{index}\n")); | |
| 492 | services.push(format!("{uri}, service{index}")); | |
| 493 | } | |
| 494 | module.push_str(&format!("local services = Map({})\n", services.join(", "))); | |
| 495 | module.push_str(r#"output { renderer = new JsonRenderer { omitNullProperties = false } | |
| 496 | value = services.filter((_, s) -> s.enabled).mapValues((_, s) -> | |
| 497 | let (route = (s.containers?.toMap()?.values ?? List()).add(s.container).filterNonNull().map((task) -> task.http).filterNonNull().findOrNull((http) -> http.hostname != null)) | |
| 498 | new Dynamic { name = s.meta.name; tagline = s.meta.tagline; hostname = route?.hostname; access = route?.authRole }) }"#); | |
| 499 | Ok(module) | |
| 500 | } | |
| 501 | ||
| 441 | 502 | async fn launcher(app: Arc<App>) -> Result<Arc<Document>> { |
| 442 | 503 | let real = tokio::fs::canonicalize(&app.repo).await?; |
| 443 | 504 | let state = app.clone(); |
| 444 | 505 | app.cache.get(format!("launcher:{}",real.display()),Duration::from_secs(365*86400),move || async move { |
| 445 | let module = format!(r#"import* "file://{}/service/*/*.pkl" as services | |
| 446 | output {{ renderer = new JsonRenderer {{ omitNullProperties = false }} | |
| 447 | value = services.toMap().filter((_, s) -> s.enabled).mapValues((_, s) -> | |
| 448 | let (route = (s.containers?.toMap()?.values ?? List()).add(s.container).filterNonNull().map((task) -> task.http).filterNonNull().findOrNull((http) -> http.hostname != null)) | |
| 449 | new Dynamic {{ name = s.meta.name; tagline = s.meta.tagline; hostname = route?.hostname; access = route?.authRole }}) }}"#,real.display()); | |
| 506 | let module = launcher_module(&real).await?; | |
| 450 | 507 | let value: Value = serde_json::from_slice(&command("pkl",&["eval","-"],Some(module.as_bytes())).await?)?; |
| 451 | 508 | let mut apps = Vec::new(); |
| 452 | 509 | for (file, value) in value.as_object().into_iter().flat_map(|v| v.iter()) { |
| ... | ... | @@ -779,6 +836,39 @@ async fn definition(app: Arc<App>, id: &str) -> Result<Value> { |
| 779 | 836 | #[cfg(test)] |
| 780 | 837 | mod tests { |
| 781 | 838 | use super::*; |
| 839 | #[tokio::test] | |
| 840 | async fn launcher_excludes_directories_and_grouped_definitions_before_import() { | |
| 841 | let root = std::env::temp_dir().join(format!("studio-launcher-{}", uuid::Uuid::new_v4())); | |
| 842 | tokio::fs::create_dir_all(root.join("config")).await.unwrap(); | |
| 843 | for directory in ["retired", "personal"] { | |
| 844 | tokio::fs::create_dir_all(root.join("service").join(directory)) | |
| 845 | .await | |
| 846 | .unwrap(); | |
| 847 | } | |
| 848 | tokio::fs::write(root.join("config/excluded-services.json"), r#"["retired"]"#) | |
| 849 | .await | |
| 850 | .unwrap(); | |
| 851 | for file in [ | |
| 852 | "retired/service.pkl", | |
| 853 | "personal/retired.pkl", | |
| 854 | "personal/service.pkl", | |
| 855 | "personal/fixture.pkl", | |
| 856 | ] { | |
| 857 | tokio::fs::write(root.join("service").join(file), "unread fixture") | |
| 858 | .await | |
| 859 | .unwrap(); | |
| 860 | } | |
| 861 | let module = launcher_module(&root).await.unwrap(); | |
| 862 | assert!(!module.contains("retired")); | |
| 863 | assert!(!module.contains("import*")); | |
| 864 | assert!(module.contains("/personal/service.pkl")); | |
| 865 | assert!(module.contains("/personal/fixture.pkl")); | |
| 866 | assert_eq!( | |
| 867 | module.lines().filter(|line| line.starts_with("import ")).count(), | |
| 868 | 2, | |
| 869 | ); | |
| 870 | tokio::fs::remove_dir_all(root).await.unwrap(); | |
| 871 | } | |
| 782 | 872 | #[test] |
| 783 | 873 | fn health_precedence() { |
| 784 | 874 | let job = json!({"Stop":false}); |
dashboard/src/main.rs+1-1| ... | ... | @@ -386,7 +386,7 @@ async fn main() -> std::result::Result<(), Box<dyn std::error::Error>> { |
| 386 | 386 | }); |
| 387 | 387 | let origin = env( |
| 388 | 388 | "STUDIO_PUBLIC_ORIGIN", |
| 389 | &format!("https://globe.{}", env("STUDIO_DOMAIN", "studio.test")), | |
| 389 | &format!("https://snowglobe.{}", env("STUDIO_DOMAIN", "studio.test")), | |
| 390 | 390 | ); |
| 391 | 391 | let app = Arc::new(App { |
| 392 | 392 | mcp: mcp::Store::new(&PathBuf::from(env("STUDIO_DATA_DIR", "data")), &origin) |
dashboard/src/shale.rs+1-1| ... | ... | @@ -587,7 +587,7 @@ pub async fn manage( |
| 587 | 587 | )?; |
| 588 | 588 | let issuer = url::Url::parse(&env( |
| 589 | 589 | "STUDIO_KEYCLOAK_URL", |
| 590 | &format!("https://keycloak.{}", env("STUDIO_DOMAIN", "studio.test")), | |
| 590 | &format!("https://auth.{}", env("STUDIO_DOMAIN", "studio.test")), | |
| 591 | 591 | ))?; |
| 592 | 592 | let parameters = fields(authorization.query().unwrap_or_default())?; |
| 593 | 593 | if authorization.origin() != issuer.origin() |
nixos/configuration.nix+5-4| ... | ... | @@ -158,8 +158,8 @@ in |
| 158 | 158 | STUDIO_INDEX_DIR = "/data/index"; |
| 159 | 159 | STUDIO_INTERNAL_URL = "https://dashboard.internal.${config.environment.variables.STUDIO_DOMAIN}:${toString internalPort}"; |
| 160 | 160 | STUDIO_FILES_URL = "https://file.${config.environment.variables.STUDIO_DOMAIN}"; |
| 161 | STUDIO_PUBLIC_ORIGIN = "https://globe.${config.environment.variables.STUDIO_DOMAIN}"; | |
| 162 | STUDIO_KEYCLOAK_URL = "https://keycloak.${config.environment.variables.STUDIO_DOMAIN}"; | |
| 161 | STUDIO_PUBLIC_ORIGIN = "https://snowglobe.${config.environment.variables.STUDIO_DOMAIN}"; | |
| 162 | STUDIO_KEYCLOAK_URL = "https://auth.${config.environment.variables.STUDIO_DOMAIN}"; | |
| 163 | 163 | STUDIO_JELLYFIN_URL = "https://jelly.${config.environment.variables.STUDIO_DOMAIN}"; |
| 164 | 164 | STUDIO_SHALE_URL = "https://shale.${config.environment.variables.STUDIO_DOMAIN}"; |
| 165 | 165 | STUDIO_PUBLISHED_ROOT = "/srv/clover/Published"; |
| ... | ... | @@ -207,7 +207,7 @@ in |
| 207 | 207 | --mount=type=bind,src=/srv/clover,dst=/srv/clover,bind-nonrecursive,bind-propagation=rslave,ro="$STUDIO_CLOVER_READ_ONLY" \ |
| 208 | 208 | --mount=type=bind,src=/srv/clover/Media,dst=/srv/clover/Media,bind-nonrecursive,bind-propagation=rslave,ro="$STUDIO_MEDIA_READ_ONLY" \ |
| 209 | 209 | ${lib.concatMapStringsSep " " (name: lib.escapeShellArg "--env=${name}") (builtins.attrNames environment)} \ |
| 210 | ${lib.concatMapStringsSep " " (name: lib.escapeShellArg "--add-host=${name}.${config.environment.variables.STUDIO_DOMAIN}:host-gateway") [ "dashboard.internal" "keycloak" "jelly" "db" "shale" ]} \ | |
| 210 | ${lib.concatMapStringsSep " " (name: lib.escapeShellArg "--add-host=${name}.${config.environment.variables.STUDIO_DOMAIN}:host-gateway") [ "dashboard.internal" "auth" "keycloak" "jelly" "db" "shale" ]} \ | |
| 211 | 211 | "''${optional[@]}" ${lib.escapeShellArg "${dashboard.image.imageName}:${dashboard.image.imageTag}"} |
| 212 | 212 | ''; |
| 213 | 213 | ExecStop = "${pkgs.podman}/bin/podman stop --ignore --time=8 studio-dashboard"; |
| ... | ... | @@ -248,7 +248,8 @@ in |
| 248 | 248 | "d /srv/prod 0755 root root -" |
| 249 | 249 | "d /srv/staging 0755 root root -" |
| 250 | 250 | "d /srv/vm 0755 root root -" |
| 251 | "A+ /srv/prod/yt-feed/data - - - - g:chloe:rwX,d:g:chloe:rwx" | |
| 251 | "d /srv/logs 0770 clo chloe -" | |
| 252 | "A+ /srv/prod/yt-feed/data - - - - g:chloe:rwX,m::rwX,d:g:chloe:rwx,d:m::rwx" | |
| 252 | 253 | "d /var/lib/studio 0700 root root -" |
| 253 | 254 | "d /var/lib/studio/dashboard 0700 studio-dashboard studio-dashboard -" |
| 254 | 255 | "d /var/lib/studio/routes 0700 root root -" |
nixos/dashboard.nix+6-1| ... | ... | @@ -4,7 +4,12 @@ let |
| 4 | 4 | youtubePython = python3.withPackages (packages: [ packages.pyyaml ]); |
| 5 | 5 | source = lib.fileset.toSource { |
| 6 | 6 | root = ../.; |
| 7 | fileset = lib.fileset.unions [ ../config ../service ]; | |
| 7 | fileset = lib.fileset.unions [ | |
| 8 | ../config | |
| 9 | (lib.fileset.difference ../service (lib.fileset.unions | |
| 10 | (builtins.filter builtins.pathExists (map (name: ../service + "/${name}") | |
| 11 | (builtins.fromJSON (builtins.readFile ../config/excluded-services.json)))))) | |
| 12 | ]; | |
| 8 | 13 | }; |
| 9 | 14 | web = stdenv.mkDerivation (finalAttrs: { |
| 10 | 15 | pname = "home-dashboard-web"; |
nixos/hardware-configuration.nix+17-1| ... | ... | @@ -1 +1,17 @@ |
| 1 | throw "Generate nixos/hardware-configuration.nix from the NixOS installer after partitioning Zenith's NVMe" | |
| 1 | { config, lib, modulesPath, ... }: | |
| 2 | { | |
| 3 | imports = [ (modulesPath + "/installer/scan/not-detected.nix") ]; | |
| 4 | boot.initrd.availableKernelModules = [ "nvme" "xhci_pci" "ahci" "usbhid" "usb_storage" "sd_mod" ]; | |
| 5 | boot.kernelModules = [ "kvm-amd" ]; | |
| 6 | fileSystems."/" = { | |
| 7 | device = "/dev/disk/by-uuid/ac86c41d-e621-4d6f-894d-adf26c421f06"; | |
| 8 | fsType = "ext4"; | |
| 9 | }; | |
| 10 | fileSystems."/boot" = { | |
| 11 | device = "/dev/disk/by-uuid/F4BE-0109"; | |
| 12 | fsType = "vfat"; | |
| 13 | options = [ "fmask=0077" "dmask=0077" ]; | |
| 14 | }; | |
| 15 | nixpkgs.hostPlatform = lib.mkDefault "x86_64-linux"; | |
| 16 | hardware.cpu.amd.updateMicrocode = lib.mkDefault config.hardware.enableRedistributableFirmware; | |
| 17 | } |
nixos/zenith.nix+41-11| ... | ... | @@ -1,6 +1,15 @@ |
| 1 | 1 | { lib, pkgs, ... }: |
| 2 | 2 | let |
| 3 | 3 | adminKey = lib.fileContents ../config/admin.pub; |
| 4 | pool = "globe"; | |
| 5 | encryptionRoots = { | |
| 6 | apps = "apps"; | |
| 7 | clover = "clover"; | |
| 8 | media = "clover/Media"; | |
| 9 | logs = "logs"; | |
| 10 | prod = "prod"; | |
| 11 | staging = "staging"; | |
| 12 | }; | |
| 4 | 13 | in |
| 5 | 14 | { |
| 6 | 15 | imports = [ ./zenith-network.nix ]; |
| ... | ... | @@ -10,26 +19,47 @@ in |
| 10 | 19 | networking.firewall.allowedTCPPorts = [ 80 443 ]; |
| 11 | 20 | networking.firewall.interfaces.enp4s0.allowedTCPPorts = [ 445 ]; |
| 12 | 21 | |
| 13 | boot.loader.grub.enable = true; | |
| 14 | boot.loader.grub.device = "/dev/disk/by-id/nvme-WD_Blue_SN5000_500GB_24261Z806200"; | |
| 22 | boot.loader.systemd-boot.enable = true; | |
| 23 | boot.loader.efi.canTouchEfiVariables = true; | |
| 15 | 24 | boot.zfs.forceImportRoot = false; |
| 16 | boot.zfs.extraPools = [ "storage1" ]; | |
| 25 | boot.zfs.extraPools = [ pool ]; | |
| 17 | 26 | boot.zfs.requestEncryptionCredentials = false; |
| 27 | security.tpm2.enable = true; | |
| 28 | ||
| 29 | systemd.services.studio-zfs-unlock = { | |
| 30 | requires = [ "zfs-import-${pool}.service" "dev-tpmrm0.device" ]; | |
| 31 | after = [ "zfs-import-${pool}.service" "dev-tpmrm0.device" ]; | |
| 32 | before = [ "zfs-mount.service" ]; | |
| 33 | requiredBy = [ "zfs-mount.service" ]; | |
| 34 | unitConfig.DefaultDependencies = false; | |
| 35 | serviceConfig = { | |
| 36 | Type = "oneshot"; | |
| 37 | RemainAfterExit = true; | |
| 38 | LoadCredentialEncrypted = lib.mapAttrsToList | |
| 39 | (name: _: "zfs-${name}:/etc/credstore.encrypted/zfs-${name}") encryptionRoots; | |
| 40 | }; | |
| 41 | script = lib.concatStrings (lib.mapAttrsToList (name: dataset: '' | |
| 42 | if test "$(${pkgs.zfs}/bin/zfs get -H -o value keystatus ${pool}/${dataset})" = unavailable; then | |
| 43 | ${pkgs.zfs}/bin/zfs load-key -L "file://$CREDENTIALS_DIRECTORY/zfs-${name}" ${pool}/${dataset} | |
| 44 | fi | |
| 45 | '') encryptionRoots); | |
| 46 | }; | |
| 18 | 47 | |
| 19 | 48 | services.xserver.videoDrivers = [ "nvidia" ]; |
| 49 | hardware.graphics.enable = true; | |
| 20 | 50 | hardware.nvidia.open = false; |
| 21 | 51 | hardware.nvidia.nvidiaPersistenced = true; |
| 22 | 52 | hardware.nvidia.nvidiaSettings = false; |
| 23 | 53 | |
| 24 | 54 | systemd.services.nomad = { |
| 25 | requires = [ "zfs-import-storage1.service" ]; | |
| 26 | after = [ "zfs-import-storage1.service" "zfs-mount.service" ]; | |
| 55 | requires = [ "zfs-import-${pool}.service" ]; | |
| 56 | after = [ "zfs-import-${pool}.service" "zfs-mount.service" ]; | |
| 27 | 57 | preStart = '' |
| 28 | test "$(${pkgs.util-linux}/bin/findmnt -n -o SOURCE --mountpoint /srv)" = storage1 | |
| 29 | test "$(${pkgs.util-linux}/bin/findmnt -n -o SOURCE --mountpoint /srv/clover)" = storage1/clover | |
| 30 | test "$(${pkgs.util-linux}/bin/findmnt -n -o SOURCE --mountpoint /srv/clover/Media)" = storage1/clover/Media | |
| 31 | test "$(${pkgs.util-linux}/bin/findmnt -n -o SOURCE --mountpoint /srv/prod)" = storage1/prod | |
| 32 | test "$(${pkgs.util-linux}/bin/findmnt -n -o SOURCE --mountpoint /srv/staging)" = storage1/staging | |
| 58 | test "$(${pkgs.util-linux}/bin/findmnt -n -o SOURCE --mountpoint /srv)" = ${pool} | |
| 59 | test "$(${pkgs.util-linux}/bin/findmnt -n -o SOURCE --mountpoint /srv/clover)" = ${pool}/clover | |
| 60 | test "$(${pkgs.util-linux}/bin/findmnt -n -o SOURCE --mountpoint /srv/clover/Media)" = ${pool}/clover/Media | |
| 61 | test "$(${pkgs.util-linux}/bin/findmnt -n -o SOURCE --mountpoint /srv/prod)" = ${pool}/prod | |
| 62 | test "$(${pkgs.util-linux}/bin/findmnt -n -o SOURCE --mountpoint /srv/staging)" = ${pool}/staging | |
| 33 | 63 | ''; |
| 34 | 64 | }; |
| 35 | 65 | |
| ... | ... | @@ -42,7 +72,7 @@ in |
| 42 | 72 | extraGroups = [ "wheel" ]; |
| 43 | 73 | openssh.authorizedKeys.keys = [ adminKey ]; |
| 44 | 74 | }; |
| 45 | environment.variables.STUDIO_POOL = "storage1"; | |
| 75 | environment.variables.STUDIO_POOL = pool; | |
| 46 | 76 | environment.variables.STUDIO_DOMAIN = "paperclover.net"; |
| 47 | 77 | |
| 48 | 78 | system.stateVersion = "26.05"; |
service/copyparty/prepare.py-3| ... | ... | @@ -9,9 +9,6 @@ import sys |
| 9 | 9 | |
| 10 | 10 | data = json.load(sys.stdin) |
| 11 | 11 | root = Path(data["hostRoot"]) / "cfg" |
| 12 | logs = Path(data["hostRoot"]) / "w/logs" | |
| 13 | logs.mkdir(parents=True, exist_ok=True) | |
| 14 | os.chown(logs, data["uid"], data["uid"]) | |
| 15 | 12 | service = Path(data["hostRoot"]).name |
| 16 | 13 | secret = json.loads(subprocess.check_output( |
| 17 | 14 | ["nomad", "var", "get", "-out=json", f"nomad/jobs/{service}"], |
service/copyparty/service.pkl+1| ... | ... | @@ -37,5 +37,6 @@ container { |
| 37 | 37 | ["/w/clover"] { src = site.cloverRoot; readOnly = site.cloverReadOnly } |
| 38 | 38 | ["/w/clover/Media"] { src = site.mediaRoot; readOnly = site.mediaReadOnly } |
| 39 | 39 | ["/w/media"] { src = site.mediaRoot; readOnly = site.mediaReadOnly } |
| 40 | ["/w/logs"] { src = "\(site.root)/logs"; readOnly = site.preview || site.cloverReadOnly } | |
| 40 | 41 | } |
| 41 | 42 | } |
service/jellyfin/configure.py+1-3| ... | ... | @@ -6,13 +6,11 @@ import time |
| 6 | 6 | import urllib.error |
| 7 | 7 | import urllib.parse |
| 8 | 8 | import urllib.request |
| 9 | from pathlib import Path | |
| 10 | 9 | |
| 11 | 10 | |
| 12 | 11 | config = json.load(sys.stdin) |
| 13 | 12 | base = f"https://{config['host']}" |
| 14 | ca = Path("/var/lib/caddy/.local/share/caddy/pki/authorities/local/root.crt") | |
| 15 | context = ssl.create_default_context(cafile=str(ca) if ca.exists() else None) | |
| 13 | context = ssl.create_default_context(cafile="/var/lib/studio/ca-bundle.crt") | |
| 16 | 14 | |
| 17 | 15 | |
| 18 | 16 | def request(path, data=None, token=None): |
service/jellyfin/service.pkl+3-1| ... | ... | @@ -41,7 +41,9 @@ container { |
| 41 | 41 | volumes { |
| 42 | 42 | ["/config"] {} |
| 43 | 43 | ["/cache"] {} |
| 44 | ["/media"] { src = site.mediaRoot; readOnly = true } | |
| 44 | for (name, folder in site.jellyfinFolders) { | |
| 45 | ["/media/jellyfin/\(name)"] { src = "\(site.mediaRoot)/\(folder)"; readOnly = true } | |
| 46 | } | |
| 45 | 47 | ["/config/config/branding.xml"] { config = "branding.xml" } |
| 46 | 48 | ["/etc/ssl/certs/ca-certificates.crt"] { |
| 47 | 49 | src = "/var/lib/studio/ca-bundle.crt" |
service/keycloak/service.pkl+3-1| ... | ... | @@ -1,6 +1,7 @@ |
| 1 | 1 | extends "../../config/Service.pkl" |
| 2 | 2 | |
| 3 | 3 | import "../../config/Service.pkl" as service |
| 4 | import "../../config/site.pkl" as site | |
| 4 | 5 | import "../postgres/service.pkl" as postgres |
| 5 | 6 | |
| 6 | 7 | class OpenIDClient extends service.Requirement { |
| ... | ... | @@ -32,7 +33,8 @@ container { |
| 32 | 33 | |
| 33 | 34 | http { |
| 34 | 35 | containerPort = 8080 |
| 35 | subdomain = "keycloak" | |
| 36 | subdomain = "auth" | |
| 37 | plainHostnames { "keycloak.\(site.domain)" } | |
| 36 | 38 | checkPath = "/realms/master" |
| 37 | 39 | } |
| 38 | 40 |
service/navidrome/service.pkl+1-1| ... | ... | @@ -19,7 +19,7 @@ container { |
| 19 | 19 | |
| 20 | 20 | volumes { |
| 21 | 21 | ["/data"] {} |
| 22 | ["/music"] { src = "\(site.mediaRoot)/music"; readOnly = true } | |
| 22 | ["/music"] { src = "\(site.mediaRoot)/Music"; readOnly = true } | |
| 23 | 23 | } |
| 24 | 24 | |
| 25 | 25 | env { |
service/postgres/pgadmin.pkl+4-2| ... | ... | @@ -1,6 +1,7 @@ |
| 1 | 1 | amends "../../config/Service.pkl" |
| 2 | 2 | |
| 3 | 3 | import "service.pkl" as postgres |
| 4 | import "../../config/site.pkl" as site | |
| 4 | 5 | |
| 5 | 6 | meta { name = "pgAdmin" } |
| 6 | 7 | healthyDeadline = "15m" |
| ... | ... | @@ -30,11 +31,12 @@ container { |
| 30 | 31 | } |
| 31 | 32 | |
| 32 | 33 | env { |
| 33 | ["PGADMIN_DEFAULT_EMAIL"] = "pgadmin4@pgadmin.org" | |
| 34 | ["PGADMIN_DEFAULT_EMAIL"] = site.ownerEmail | |
| 35 | ["PGADMIN_CONFIG_DESKTOP_USER"] = "\"\(site.ownerEmail)\"" | |
| 34 | 36 | ["PGADMIN_DEFAULT_PASSWORD"] = "${secret.own.password}" |
| 35 | 37 | ["PGADMIN_CONFIG_SERVER_MODE"] = "False" |
| 36 | 38 | ["PGADMIN_CONFIG_MASTER_PASSWORD_REQUIRED"] = "False" |
| 37 | ["PGADMIN_CUSTOM_CONFIG_DISTRO_FILE"] = "/var/lib/pgadmin/config_distro.py" | |
| 39 | ["PGADMIN_CUSTOM_CONFIG_DISTRO_FILE"] = "/tmp/config_distro.py" | |
| 38 | 40 | ["PGADMIN_DISABLE_POSTFIX"] = "True" |
| 39 | 41 | ["PGADMIN_LISTEN_PORT"] = "5050" |
| 40 | 42 | ["PGADMIN_REPLACE_SERVERS_ON_STARTUP"] = "True" |
service/postgres/pgadmin/start.sh+1-1| ... | ... | @@ -1,7 +1,7 @@ |
| 1 | 1 | #!/bin/sh |
| 2 | 2 | set -eu |
| 3 | 3 | |
| 4 | storage=/var/lib/pgadmin/storage/pgadmin4_pgadmin.org | |
| 4 | storage=/var/lib/pgadmin/storage/$(printf '%s' "$PGADMIN_DEFAULT_EMAIL" | tr '@' '_') | |
| 5 | 5 | mkdir -p "$storage" |
| 6 | 6 | printf '%s:%s:*:postgres:%s\n' "$POSTGRES_HOST" "$POSTGRES_PORT" "$POSTGRES_PASSWORD" >"$storage/pgpass" |
| 7 | 7 | chmod 600 "$storage/pgpass" |
service/qbittorrent/prepare.py+2-2| ... | ... | @@ -14,8 +14,8 @@ for directory in (path.parent.parent, path.parent): |
| 14 | 14 | lines = path.read_text().splitlines() if path.exists() else [] |
| 15 | 15 | |
| 16 | 16 | for section, key, value in ( |
| 17 | ("BitTorrent", "Session\\DefaultSavePath", "/data/media/seedbox"), | |
| 18 | ("BitTorrent", "Session\\TempPath", "/data/media/seedbox"), | |
| 17 | ("BitTorrent", "Session\\DefaultSavePath", "/data/media/Seedbox"), | |
| 18 | ("BitTorrent", "Session\\TempPath", "/data/media/Seedbox"), | |
| 19 | 19 | ("Preferences", "WebUI\\AuthSubnetWhitelist", "10.88.0.1/32"), |
| 20 | 20 | ("Preferences", "WebUI\\AuthSubnetWhitelistEnabled", "true"), |
| 21 | 21 | ("Preferences", "WebUI\\HostHeaderValidation", "false"), |
service/qbittorrent/service.pkl+6-1| ... | ... | @@ -35,7 +35,12 @@ container { |
| 35 | 35 | volumes { |
| 36 | 36 | ["/config"] {} |
| 37 | 37 | ["/data/tmp"] {} |
| 38 | ["/data/media"] { src = site.mediaRoot; readOnly = site.mediaReadOnly } | |
| 38 | for (name in List("Seedbox", "torrent")) { | |
| 39 | ["/data/media/\(name)"] { src = "\(site.mediaRoot)/Seedbox"; readOnly = site.mediaReadOnly } | |
| 40 | } | |
| 41 | for (name, folder in site.jellyfinFolders) { | |
| 42 | ["/data/media/jellyfin/\(name)"] { src = "\(site.mediaRoot)/\(folder)"; readOnly = site.mediaReadOnly } | |
| 43 | } | |
| 39 | 44 | ["/config/openvpn/profile.ovpn"] { config = "openvpn/profile.ovpn" } |
| 40 | 45 | } |
| 41 | 46 | tmpfs { "/config/openvpn:size=1m,mode=0700" } |
service/samba/service.pkl+2-4| ... | ... | @@ -29,7 +29,7 @@ container { |
| 29 | 29 | env { |
| 30 | 30 | // Finder writes as Clover's owner; Copyparty writes through the shared group. |
| 31 | 31 | ["UID_clo"] = site.cloverUid.toString() |
| 32 | ["ACCOUNT_clo"] = "${secret.own.password}" | |
| 32 | ["ACCOUNT_clo"] = "${secret.own.account}" | |
| 33 | 33 | ["AVAHI_DISABLE"] = "1" |
| 34 | 34 | ["WSDD2_DISABLE"] = "1" |
| 35 | 35 | ["SAMBA_CONF_WORKGROUP"] = "PAPER_CLOVER" |
| ... | ... | @@ -39,6 +39,4 @@ container { |
| 39 | 39 | } |
| 40 | 40 | } |
| 41 | 41 | |
| 42 | secrets { | |
| 43 | ["password"] {} | |
| 44 | } | |
| 42 | requiredSecrets { "account" } |
service/youtube/yt-feed.pkl+1| ... | ... | @@ -28,6 +28,7 @@ container { |
| 28 | 28 | } |
| 29 | 29 | env { |
| 30 | 30 | ["STATE_DIR"] = "/data" |
| 31 | ["INDEP_DIR"] = "/media/\(site.jellyfinFolders["Independent"])" | |
| 31 | 32 | ["SR_THREADS"] = "4" |
| 32 | 33 | } |
| 33 | 34 | } |
service/youtube/ytdl-sub.pkl+6-3| ... | ... | @@ -30,9 +30,12 @@ container { |
| 30 | 30 | src = "\(site.cloverRoot)/Documents/Config/Youtube Downloader" |
| 31 | 31 | readOnly = true |
| 32 | 32 | } |
| 33 | ["/media"] { | |
| 34 | src = if (passive) null else site.mediaRoot | |
| 35 | readOnly = false | |
| 33 | ["/media"] {} | |
| 34 | when (!passive) { | |
| 35 | ["/media/jellyfin/Independent"] { | |
| 36 | src = "\(site.mediaRoot)/\(site.jellyfinFolders["Independent"])" | |
| 37 | readOnly = site.mediaReadOnly | |
| 38 | } | |
| 36 | 39 | } |
| 37 | 40 | } |
| 38 | 41 |
tools/check-legacy-handoff.sh+12-2| ... | ... | @@ -8,7 +8,17 @@ printf '%s\n' "$handoff" | grep -Eq '^/mnt/storage1/apps/studio-handoff/[0-9]{8} |
| 8 | 8 | echo 'Invalid legacy handoff directory' >&2 |
| 9 | 9 | exit 1 |
| 10 | 10 | } |
| 11 | ssh -p "$port" "$target" "set -eu; test -f '$handoff/manifest.json'; test \"\$(findmnt -n -o SOURCE --mountpoint /mnt/storage1/apps)\" = storage1/apps; test \"\$(findmnt -n -o FSTYPE --mountpoint /mnt/storage1/apps)\" = zfs; test \"\$(zfs get -H -o value encryption storage1/apps)\" = aes-256-gcm" || { | |
| 12 | echo 'Legacy handoff or mounted storage1/apps dataset is unavailable' >&2 | |
| 11 | ssh -p "$port" "$target" " | |
| 12 | set -eu | |
| 13 | test -f '$handoff/manifest.json' | |
| 14 | pool=\$(findmnt -n -o SOURCE --mountpoint /srv) | |
| 15 | test \"\$(findmnt -n -o FSTYPE --mountpoint /srv)\" = zfs | |
| 16 | zpool list \"\$pool\" >/dev/null | |
| 17 | apps=\$(findmnt -n -o SOURCE --mountpoint /mnt/storage1/apps) | |
| 18 | test \"\$apps\" = \"\$pool/apps\" | |
| 19 | test \"\$(findmnt -n -o FSTYPE --mountpoint /mnt/storage1/apps)\" = zfs | |
| 20 | test \"\$(zfs get -H -o value encryption \"\$apps\")\" = aes-256-gcm | |
| 21 | " || { | |
| 22 | echo 'Legacy handoff or encrypted apps dataset is unavailable' >&2 | |
| 13 | 23 | exit 1 |
| 14 | 24 | } |
tools/configure-arr.py+41| ... | ... | @@ -6,6 +6,8 @@ from urllib.parse import urlsplit, urlunsplit |
| 6 | 6 | import urllib.request |
| 7 | 7 | import xml.etree.ElementTree as ET |
| 8 | 8 | |
| 9 | from studio import load | |
| 10 | ||
| 9 | 11 | |
| 10 | 12 | data = json.load(sys.stdin) |
| 11 | 13 | service = data["serviceId"] |
| ... | ... | @@ -44,6 +46,10 @@ allocation = allocations[0] |
| 44 | 46 | base = f"http://{allocation['Address']}:{allocation['Port']}/api/v3/downloadclient" |
| 45 | 47 | headers = {"X-Api-Key": api_key} |
| 46 | 48 | clients = json.load(urllib.request.urlopen(urllib.request.Request(base, headers=headers), timeout=10)) |
| 49 | previous_hosts = { | |
| 50 | field.get("value") for client in clients if client["implementation"] == "QBittorrent" | |
| 51 | for field in client["fields"] if field["name"] == "host" | |
| 52 | } | |
| 47 | 53 | qbittorrent_host, qbittorrent_port = internal_endpoint("qbittorrent") |
| 48 | 54 | if not any(client["implementation"] == "QBittorrent" for client in clients): |
| 49 | 55 | schemas = json.load(urllib.request.urlopen(urllib.request.Request(base + "/schema", headers=headers), timeout=10)) |
| ... | ... | @@ -85,6 +91,41 @@ for client in clients: |
| 85 | 91 | ) |
| 86 | 92 | urllib.request.urlopen(request, timeout=15).close() |
| 87 | 93 | |
| 94 | local_volumes = load(service.split("-preview-", 1)[0], {})["containers"]["app"]["volumes"] | |
| 95 | remote_volumes = load("qbittorrent", {})["containers"]["app"]["volumes"] | |
| 96 | mapping_base = base.replace("/downloadclient", "/remotepathmapping") | |
| 97 | mappings = json.load(urllib.request.urlopen(urllib.request.Request(mapping_base, headers=headers), timeout=10)) | |
| 98 | for remote_path, remote_volume in remote_volumes.items(): | |
| 99 | if not remote_volume.get("src"): | |
| 100 | continue | |
| 101 | source = Path(remote_volume["src"]) | |
| 102 | destinations = [ | |
| 103 | str(Path(local_path) / source.relative_to(volume["src"])) | |
| 104 | for local_path, volume in local_volumes.items() | |
| 105 | if volume.get("src") and source.is_relative_to(volume["src"]) | |
| 106 | ] | |
| 107 | if not destinations: | |
| 108 | continue | |
| 109 | if len(destinations) != 1: | |
| 110 | raise ValueError(f"{service} has ambiguous mounts for qBittorrent's {remote_path}") | |
| 111 | local_path = destinations[0] | |
| 112 | if remote_path == local_path: | |
| 113 | continue | |
| 114 | expected = {"host": qbittorrent_host, "remotePath": remote_path + "/", "localPath": local_path + "/"} | |
| 115 | existing = next((mapping for mapping in mappings | |
| 116 | if mapping["remotePath"].rstrip("/") == remote_path | |
| 117 | and mapping["host"] in previous_hosts | {qbittorrent_host}), None) | |
| 118 | if existing and all(existing[key] == value for key, value in expected.items()): | |
| 119 | continue | |
| 120 | body = {**existing, **expected} if existing else expected | |
| 121 | request = urllib.request.Request( | |
| 122 | f"{mapping_base}/{existing['id']}" if existing else mapping_base, | |
| 123 | data=json.dumps(body).encode(), | |
| 124 | headers={**headers, "Content-Type": "application/json"}, | |
| 125 | method="PUT" if existing else "POST", | |
| 126 | ) | |
| 127 | urllib.request.urlopen(request, timeout=15).close() | |
| 128 | ||
| 88 | 129 | jackett_host, jackett_port = internal_endpoint("jackett") |
| 89 | 130 | indexer_base = base.replace("/downloadclient", "/indexer") |
| 90 | 131 | indexers = json.load(urllib.request.urlopen(urllib.request.Request(indexer_base, headers=headers), timeout=10)) |
tools/dawarich-migration.md+18-4| ... | ... | @@ -1,8 +1,22 @@ |
| 1 | # Dawarich preview and fresh production start | |
| 1 | # Dawarich migration | |
| 2 | 2 | |
| 3 | Zenith's separate Redis dump is 86,557 bytes and belongs to Dawarich, not Snow Globe's standalone Redis service. DB 0 contains Dawarich cache and track-generation keys; DB 1 held 106 queued `Family::Invitations::CleanupJob` entries at inspection. The old image's nightly schedule sends them to `family`, while its worker listens to `families`; the live `family` queue had grown to 107 and `families` was empty on 2026-09-27. The pinned Snow Globe image schedules that job on `families`, and all 14 of its scheduled queues appear in the worker's queue list. The new Zenith installation starts with fresh PostgreSQL, storage, and Redis; the owner will handle application data migration. The offline PostgreSQL handoff excludes Dawarich. The existing VM production database still holds the copied 22,666 points used for testing. | |
| 3 | The 2026-10-04 offline handoff includes Dawarich's PostgreSQL 18 database, with a verified restore and hashes for every table, sequence, and large object. Restore it before starting Dawarich. Keep the original Rails `SECRET_KEY_BASE` and copy `/var/app/storage` into the managed service dataset before the first application boot. | |
| 4 | 4 | |
| 5 | Zenith's live Dawarich database is PostgreSQL 18 with PostGIS. The VM's Postgres service is also PostgreSQL 18 with PostGIS. The tested transfer used a consistent custom-format dump made with `pg_dump -Fc --no-owner --no-acl --exclude-extension=postgis`, then restored it into a **separate stage database** after creating the PostGIS extension there. It did not write to Zenith or the VM's production database. | |
| 5 | The production database import on 2026-10-04 passed all 63 exported relation and large-object fingerprints before any Dawarich process started. Evidence is `/var/lib/studio/dawarich-import-7hw6v9w1/verified.json`. PostGIS retains its 8,500 built-in spatial references; PostgreSQL extension dumps contain only custom reference rows. Removing the built-ins changes geometry JSON serialization as well as breaking reference lookup. The importer preserves built-ins and replaces only explicitly dumped custom SRIDs. | |
| 6 | ||
| 7 | [import-personal-postgres.py](import-personal-postgres.py) reads the allocated production database from `nomad/jobs/dawarich/inputs/database`. The target database and `svc_dawarich` role must already exist. For initial production migration, keep both Nomad jobs stopped and use an isolated PostGIS initializer with `--network none`, mounting the final PostgreSQL data directory: | |
| 8 | ||
| 9 | ```sh | |
| 10 | python3 tools/import-personal-postgres.py dawarich \ | |
| 11 | /mnt/storage1/apps/studio-handoff/20261004T230507Z-e75419 \ | |
| 12 | --container PERSONAL_POSTGRES_INITIALIZER | |
| 13 | ``` | |
| 14 | ||
| 15 | The importer saves a private target backup under `/var/lib/studio/dawarich-import-*`, restores application objects as the allocated owner, preserves PostGIS reference data, and compares the result with the export's content hashes. It leaves Dawarich stopped. Stop and remove the initializer before deploying the production Postgres job. For a later import into an already running production Postgres allocation, omit `--container`; Dawarich must still be stopped. | |
| 16 | ||
| 17 | The historical preview results below concern the rehearsal VM. Zenith's separate Redis dump was 86,557 bytes and belonged to Dawarich. DB 0 contained caches and track-generation keys; DB 1 held 106 queued `Family::Invitations::CleanupJob` entries. The old image's nightly schedule sent them to `family`, while its worker listened to `families`; that queue reached 107 entries on 2026-09-27. The pinned image schedules this job on `families`, and all 14 of its scheduled queues appear in the worker's queue list. | |
| 18 | ||
| 19 | The rehearsal's source and target used PostgreSQL 18 with PostGIS. The tested transfer used a consistent custom-format dump made with `pg_dump -Fc --no-owner --no-acl --exclude-extension=postgis`, then restored it into a **separate stage database** after creating the PostGIS extension there. It did not write to Zenith or the VM's production database. | |
| 6 | 20 | |
| 7 | 21 | The Rails `SECRET_KEY_BASE` must stay the same as the source value so encrypted records remain readable. The old `/var/app/storage` contained one file; its checksum matched the staged copy. The generated `/var/app/public` assets and `/var/app/tmp` cache were not imported. The stage used its own Keycloak client and database; its two imported user records remained intact. |
| 8 | 22 | |
| ... | ... | @@ -14,4 +28,4 @@ The corrected deployment is `dawarich-preview-1e4dac4a` at `https://dawarich-pre |
| 14 | 28 | |
| 15 | 29 | [import-dawarich.sh](import-dawarich.sh) repeats the preview restore from the latest read-only Zenith app snapshot and copies the small storage directory. It compares the source and target `SECRET_KEY_BASE` by hash, backs up the target ZFS dataset and database, restores the source PostgreSQL dump with PostGIS, and checks point/user counts before starting the preview. The 2026-09-26 run completed successfully: 22,666 points and two users survived migrations, HTTPS health returned 200 from this Mac with certificate verification, Nomad marked the job healthy, and the worker had no `start_at` error. |
| 16 | 30 | |
| 17 | The old-data importer remains useful for VM previews while Zenith is running. It is not part of the same-machine production cutover. | |
| 31 | The shell importer requires the old running source and serves VM previews. The Python importer uses the retained offline handoff for the same-machine production cutover. |
tools/deploy-test.py+8| ... | ... | @@ -27,6 +27,9 @@ class MainExport(unittest.TestCase): |
| 27 | 27 | else: |
| 28 | 28 | path.mkdir() |
| 29 | 29 | (path / "fixture").write_text("committed") |
| 30 | (repo / "config/excluded-services.json").write_text('["retired"]\n') | |
| 31 | (repo / "service/retired").mkdir() | |
| 32 | (repo / "service/retired/private").write_text("must not be exported") | |
| 30 | 33 | executable = repo / "tools/fixture" |
| 31 | 34 | executable.chmod(0o755) |
| 32 | 35 | for args in [["describe", "-m", "Test main"], ["bookmark", "set", "main"]]: |
| ... | ... | @@ -38,6 +41,8 @@ class MainExport(unittest.TestCase): |
| 38 | 41 | original = subprocess.run |
| 39 | 42 | |
| 40 | 43 | def transport(argv, **kwargs): |
| 44 | if argv[0] == "jj" and "show" in argv and argv[-1].startswith("service/retired/"): | |
| 45 | raise AssertionError("read excluded source") | |
| 41 | 46 | if argv[0] == "ssh": |
| 42 | 47 | return subprocess.CompletedProcess(argv, 1) |
| 43 | 48 | if argv[0] == "rsync" and "-e" in argv: |
| ... | ... | @@ -62,6 +67,9 @@ class MainExport(unittest.TestCase): |
| 62 | 67 | self.assertEqual((stage / "service/fixture").read_text(), "unfinished") |
| 63 | 68 | self.assertEqual((stage / "service/new").read_text(), "new working file") |
| 64 | 69 | self.assertNotIn("main", json.loads((stage / ".studio-release.json").read_text())) |
| 70 | self.assertFalse((main / "service/retired").exists()) | |
| 71 | self.assertFalse((stage / "service/retired").exists()) | |
| 72 | self.assertFalse(any("retired" in relative.parts for _, relative in release.files(repo))) | |
| 65 | 73 | subprocess.run(["jj", "bookmark", "set", "main"], cwd=repo, check=True, capture_output=True) |
| 66 | 74 | with self.assertRaisesRegex(ValueError, "commit description"): |
| 67 | 75 | deploy.upload(main=True) |
tools/deploy.py+10-1| ... | ... | @@ -11,7 +11,7 @@ import sys |
| 11 | 11 | import tempfile |
| 12 | 12 | import time |
| 13 | 13 | |
| 14 | from release import SOURCES, tree_digest | |
| 14 | from release import SOURCES, excluded_services, tree_digest | |
| 15 | 15 | |
| 16 | 16 | |
| 17 | 17 | REPO = Path(__file__).resolve().parent.parent |
| ... | ... | @@ -43,6 +43,7 @@ def sync_manager(release): |
| 43 | 43 | |
| 44 | 44 | |
| 45 | 45 | def upload(main=False): |
| 46 | excluded = excluded_services(REPO) | |
| 46 | 47 | with tempfile.TemporaryDirectory() as temporary: |
| 47 | 48 | snapshot = Path(temporary) |
| 48 | 49 | revision = None |
| ... | ... | @@ -64,6 +65,11 @@ def upload(main=False): |
| 64 | 65 | for entry in entries: |
| 65 | 66 | name, kind, executable = json.loads(entry) |
| 66 | 67 | relative = Path(name) |
| 68 | if relative.parts[:1] == ("service",) and ( | |
| 69 | set(relative.parts[1:]) & excluded or | |
| 70 | relative.suffix == ".pkl" and relative.stem in excluded | |
| 71 | ): | |
| 72 | continue | |
| 67 | 73 | if kind != "file" or relative.is_absolute() or ".." in relative.parts: |
| 68 | 74 | raise ValueError(f"unsupported main release file: {name}") |
| 69 | 75 | destination = snapshot / relative |
| ... | ... | @@ -79,6 +85,9 @@ def upload(main=False): |
| 79 | 85 | "--exclude=.identities.lock", "--exclude=identities.pending", "--exclude=._*", |
| 80 | 86 | "--exclude=dashboard/node_modules", "--exclude=dashboard/dist", |
| 81 | 87 | "--exclude=dashboard/.cache", "--exclude=dashboard/data", "--exclude=dashboard/target", |
| 88 | *(f"--exclude=/service/{name}/" for name in sorted(excluded)), | |
| 89 | *(f"--exclude=/service/*/{name}.pkl" for name in sorted(excluded)), | |
| 90 | *(f"--exclude=/service/*/{name}/" for name in sorted(excluded)), | |
| 82 | 91 | *(str(REPO / path) for path in SOURCES), |
| 83 | 92 | str(snapshot) + "/"], check=True, |
| 84 | 93 | ) |
tools/export-offline-postgres.py created+220| ... | ... | @@ -0,0 +1,220 @@ |
| 1 | #!/usr/bin/env python3 | |
| 2 | import argparse | |
| 3 | from contextlib import contextmanager | |
| 4 | from datetime import datetime, timezone | |
| 5 | import hashlib | |
| 6 | import json | |
| 7 | import os | |
| 8 | from pathlib import Path | |
| 9 | import secrets | |
| 10 | import shutil | |
| 11 | import subprocess | |
| 12 | import tempfile | |
| 13 | ||
| 14 | ||
| 15 | def run(command, *, user=None, output=None): | |
| 16 | result = subprocess.run(command, stdout=output or subprocess.PIPE, | |
| 17 | stderr=subprocess.PIPE, user=user, group=user, | |
| 18 | extra_groups=[] if user is not None else None) | |
| 19 | if result.returncode: | |
| 20 | with (destination / 'diagnostics.log').open('ab') as log: | |
| 21 | log.write(result.stderr) | |
| 22 | raise RuntimeError(f'{Path(command[0]).name} failed; see private diagnostics.log') | |
| 23 | return result.stdout | |
| 24 | ||
| 25 | ||
| 26 | def sha256(path): | |
| 27 | digest = hashlib.sha256() | |
| 28 | with path.open('rb') as source: | |
| 29 | for block in iter(lambda: source.read(1024 * 1024), b''): | |
| 30 | digest.update(block) | |
| 31 | return digest.hexdigest() | |
| 32 | ||
| 33 | ||
| 34 | def tree_hash(path): | |
| 35 | digest = hashlib.sha256() | |
| 36 | for entry in sorted(path.rglob('*')): | |
| 37 | digest.update(str(entry.relative_to(path)).encode() + b'\0') | |
| 38 | if entry.is_symlink(): | |
| 39 | raise ValueError(f'External cluster link requires separate backup: {entry}') | |
| 40 | if entry.is_file(): | |
| 41 | digest.update(bytes.fromhex(sha256(entry))) | |
| 42 | return digest.hexdigest() | |
| 43 | ||
| 44 | ||
| 45 | def identifier(value): | |
| 46 | return '"' + value.replace('"', '""') + '"' | |
| 47 | ||
| 48 | ||
| 49 | @contextmanager | |
| 50 | def server(runtime, folder, *, source=None): | |
| 51 | folder.mkdir(mode=0o700) | |
| 52 | data = folder / 'data' | |
| 53 | socket = folder / 'socket' | |
| 54 | socket.mkdir(mode=0o700) | |
| 55 | os.chown(folder, 65534, 65534) | |
| 56 | os.chown(socket, 65534, 65534) | |
| 57 | if source: | |
| 58 | run(['rsync', '-aH', '--numeric-ids', str(source) + '/', str(data) + '/']) | |
| 59 | (data / 'postmaster.pid').unlink(missing_ok=True) | |
| 60 | for entry in [data, *data.rglob('*')]: | |
| 61 | os.chown(entry, 65534, 65534) | |
| 62 | else: | |
| 63 | run([str(runtime / 'bin/initdb'), '-D', str(data), '-U', 'infra2_backup_verifier', | |
| 64 | '--auth=trust', '--locale=C.UTF-8'], user=65534) | |
| 65 | config = folder / 'postgresql.conf' | |
| 66 | config.write_text("listen_addresses = ''\nssl = off\nlogging_collector = off\n" | |
| 67 | "shared_preload_libraries = ''\n" | |
| 68 | f"unix_socket_directories = '{socket}'\n" | |
| 69 | f"hba_file = '{folder / 'pg_hba.conf'}'\n") | |
| 70 | (folder / 'pg_hba.conf').write_text('local all all trust\n') | |
| 71 | os.chown(config, 65534, 65534) | |
| 72 | os.chown(folder / 'pg_hba.conf', 65534, 65534) | |
| 73 | control = [str(runtime / 'bin/pg_ctl'), '-D', str(data)] | |
| 74 | try: | |
| 75 | run(control + ['-l', str(folder / 'server.log'), '-o', f'-c config_file={config}', | |
| 76 | '-w', 'start'], user=65534) | |
| 77 | yield socket | |
| 78 | finally: | |
| 79 | if (data / 'postmaster.pid').exists(): | |
| 80 | run(control + ['-m', 'fast', '-w', 'stop'], user=65534) | |
| 81 | ||
| 82 | ||
| 83 | def sql(runtime, socket, database, query, *, role='postgres'): | |
| 84 | return run([str(runtime / 'bin/psql'), '-X', '-h', str(socket), '-U', role, | |
| 85 | '-d', database, '-v', 'ON_ERROR_STOP=1', '-At', '-c', query]) | |
| 86 | ||
| 87 | ||
| 88 | def fingerprint(runtime, socket, database): | |
| 89 | relations = json.loads(sql(runtime, socket, database, """ | |
| 90 | SELECT coalesce(json_agg(x ORDER BY n, c), '[]') FROM ( | |
| 91 | SELECT n.nspname n, c.relname c, c.relkind k | |
| 92 | FROM pg_class c JOIN pg_namespace n ON n.oid=c.relnamespace | |
| 93 | WHERE n.nspname NOT LIKE 'pg_%' AND n.nspname <> 'information_schema' | |
| 94 | AND c.relkind IN ('r','m','S') | |
| 95 | ) x | |
| 96 | """)) | |
| 97 | result = {} | |
| 98 | for relation in relations: | |
| 99 | table = identifier(relation['n']) + '.' + identifier(relation['c']) | |
| 100 | query = (f'SELECT last_value,is_called FROM {table}' if relation['k'] == 'S' else | |
| 101 | f"COPY (SELECT row_to_json(t)::text FROM {table} t " | |
| 102 | 'ORDER BY row_to_json(t)::text COLLATE "C") TO STDOUT') | |
| 103 | content = sql(runtime, socket, database, query) | |
| 104 | result[table] = {'rows': len(content.splitlines()), | |
| 105 | 'sha256': hashlib.sha256(content).hexdigest()} | |
| 106 | for name, query in { | |
| 107 | 'large_objects': 'COPY (SELECT loid,pageno,encode(data,\'hex\') FROM pg_largeobject ORDER BY loid,pageno) TO STDOUT', | |
| 108 | 'large_object_owners': 'COPY (SELECT oid,pg_get_userbyid(lomowner),lomacl::text FROM pg_largeobject_metadata ORDER BY oid) TO STDOUT', | |
| 109 | }.items(): | |
| 110 | content = sql(runtime, socket, database, query) | |
| 111 | result[name] = {'rows': len(content.splitlines()), | |
| 112 | 'sha256': hashlib.sha256(content).hexdigest()} | |
| 113 | return result | |
| 114 | ||
| 115 | ||
| 116 | parser = argparse.ArgumentParser(description='Export and restore-check stopped PostgreSQL clusters from private copies.') | |
| 117 | parser.add_argument('--cluster', nargs=2, action='append', required=True, metavar=('SOURCE', 'RUNTIME')) | |
| 118 | parser.add_argument('--locale-archive', type=Path, required=True) | |
| 119 | args = parser.parse_args() | |
| 120 | if os.geteuid() != 0: | |
| 121 | parser.error('Run as root on the installer') | |
| 122 | os.umask(0o077) | |
| 123 | os.environ['LOCALE_ARCHIVE'] = str(args.locale_archive) | |
| 124 | os.environ.pop('PGOPTIONS', None) | |
| 125 | root = Path('/mnt/storage1/apps/studio-handoff') | |
| 126 | mounted = subprocess.check_output(['findmnt', '-n', '-o', 'SOURCE,FSTYPE', '--mountpoint', str(root.parent)], text=True).split() | |
| 127 | if len(mounted) != 2 or mounted[1] != 'zfs' or not mounted[0].endswith('/apps'): | |
| 128 | parser.error('Expected retained apps dataset mounted at /mnt/storage1/apps') | |
| 129 | if subprocess.run(['pgrep', '-x', 'postgres'], stdout=subprocess.DEVNULL).returncode == 0: | |
| 130 | parser.error('Stop PostgreSQL before taking an offline copy') | |
| 131 | handoff_id = datetime.now(timezone.utc).strftime('%Y%m%dT%H%M%SZ-') + secrets.token_hex(3) | |
| 132 | root.mkdir(mode=0o700, exist_ok=True) | |
| 133 | destination = root / ('.pending-' + handoff_id) | |
| 134 | destination.mkdir(mode=0o700) | |
| 135 | manifest = {'id': handoff_id, 'clusters': {}, 'databases': {}} | |
| 136 | work = Path(tempfile.mkdtemp(prefix='infra2-pg-export-', dir='/run')) | |
| 137 | os.chown(work, 65534, 65534) | |
| 138 | try: | |
| 139 | for source_arg, runtime_arg in args.cluster: | |
| 140 | source, runtime = Path(source_arg).resolve(), Path(runtime_arg).resolve() | |
| 141 | if not source.is_relative_to(root.parent) or not (source / 'PG_VERSION').is_file(): | |
| 142 | raise ValueError('Source must be a retained cluster inside the apps dataset') | |
| 143 | major = (source / 'PG_VERSION').read_text().strip() | |
| 144 | version = run([str(runtime / 'bin/postgres'), '--version']).decode().strip() | |
| 145 | if version.split()[-1].split('.')[0] != major or major in manifest['clusters']: | |
| 146 | raise ValueError('Runtime major must match its unique source cluster') | |
| 147 | before = tree_hash(source) | |
| 148 | output = destination / ('pg' + major) | |
| 149 | output.mkdir(mode=0o700) | |
| 150 | run(['tar', '-C', str(source), '-czf', str(output / 'cluster.tar.gz'), '.']) | |
| 151 | cluster = {'source': str(source), 'runtime': version, 'sourceSha256': before, 'databases': {}} | |
| 152 | with server(runtime, work / ('source' + major), source=source) as socket: | |
| 153 | databases = json.loads(sql(runtime, socket, 'postgres', """ | |
| 154 | SELECT json_agg(x ORDER BY name) FROM ( | |
| 155 | SELECT datname name, datallowconn connect, datistemplate template, | |
| 156 | pg_get_userbyid(datdba) owner FROM pg_database WHERE datname <> 'template0' | |
| 157 | ) x | |
| 158 | """)) | |
| 159 | tablespaces = json.loads(sql(runtime, socket, 'postgres', "SELECT coalesce(json_agg(spcname), '[]') FROM pg_tablespace WHERE spcname NOT IN ('pg_default','pg_global')")) | |
| 160 | if tablespaces: | |
| 161 | raise ValueError('Custom tablespaces require an explicit restore layout') | |
| 162 | with (output / 'globals.sql').open('wb') as file: | |
| 163 | run([str(runtime / 'bin/pg_dumpall'), '-h', str(socket), '-U', 'postgres', '--globals-only'], output=file) | |
| 164 | for index, database in enumerate(databases): | |
| 165 | name = database['name'] | |
| 166 | if not database['connect']: | |
| 167 | sql(runtime, socket, 'postgres', f'ALTER DATABASE {identifier(name)} ALLOW_CONNECTIONS true') | |
| 168 | dump = output / f'{index:03d}.dump' | |
| 169 | with dump.open('wb') as file: | |
| 170 | run([str(runtime / 'bin/pg_dump'), '-h', str(socket), '-U', 'postgres', '-d', name, | |
| 171 | '-Fc', '--create'], output=file) | |
| 172 | entry = {**database, 'file': str(dump.relative_to(destination)), 'bytes': dump.stat().st_size, | |
| 173 | 'sha256': sha256(dump), 'contents': fingerprint(runtime, socket, name)} | |
| 174 | cluster['databases'][name] = entry | |
| 175 | if name in {'evil-forgejo', 'evil-hedgedoc'} and major == '18': | |
| 176 | query = ('SELECT (SELECT count(*) FROM "user"), (SELECT count(*) FROM repository)' if name == 'evil-forgejo' else | |
| 177 | 'SELECT (SELECT count(*) FROM "Notes"), (SELECT count(*) FROM "Users"), (SELECT count(*) FROM "Revisions"), (SELECT count(*) FROM "Authors")') | |
| 178 | entry['counts'] = [int(n) for n in sql(runtime, socket, name, query).decode().strip().split('|')] | |
| 179 | if tree_hash(source) != before: | |
| 180 | raise RuntimeError('Original cluster changed during export; export is not publishable') | |
| 181 | with server(runtime, work / ('restore' + major)) as socket: | |
| 182 | bootstrap = 'infra2_backup_verifier' | |
| 183 | sql(runtime, socket, 'postgres', f'CREATE DATABASE {bootstrap}', role=bootstrap) | |
| 184 | run([str(runtime / 'bin/psql'), '-X', '-h', str(socket), '-U', bootstrap, '-d', bootstrap, | |
| 185 | '-v', 'ON_ERROR_STOP=1', '-f', str(output / 'globals.sql')]) | |
| 186 | sql(runtime, socket, bootstrap, 'ALTER DATABASE template1 IS_TEMPLATE false', role=bootstrap) | |
| 187 | for name, entry in cluster['databases'].items(): | |
| 188 | run([str(runtime / 'bin/pg_restore'), '-h', str(socket), '-U', bootstrap, '-d', bootstrap, | |
| 189 | '--clean', '--if-exists', '--create', '--exit-on-error', str(destination / entry['file'])]) | |
| 190 | if fingerprint(runtime, socket, name) != entry['contents']: | |
| 191 | raise RuntimeError(f'Restored data differs: PostgreSQL {major} database {name}') | |
| 192 | print(f'PostgreSQL {major}: restored and verified {name}', flush=True) | |
| 193 | cluster['restoreVerified'] = True | |
| 194 | manifest['clusters'][major] = cluster | |
| 195 | for name in ('evil-forgejo', 'evil-hedgedoc'): | |
| 196 | entry = manifest['clusters']['18']['databases'][name] | |
| 197 | (destination / (name + '.dump')).symlink_to(entry['file']) | |
| 198 | manifest['databases'][name] = {key: entry[key] for key in ('bytes', 'sha256', 'counts')} | |
| 199 | manifest['databases'][name]['file'] = name + '.dump' | |
| 200 | keys = {'lfs_jwt': 'EVIL_FORGEJO_SERVER_LFS_JWT_SECRET', 'oauth_jwt': 'EVIL_FORGEJO_OAUTH2_JWT_SECRET', | |
| 201 | 'security_key': 'EVIL_FORGEJO_SECURITY_SECRET_KEY', 'internal_token': 'EVIL_FORGEJO_SECURITY_INTERNAL_TOKEN', | |
| 202 | 'anubis_key': 'ANUBIS_PRIVATE_KEY', 'mailer_address': 'MAILER_ADDRESS', | |
| 203 | 'mailer_username': 'MAILER_USERNAME', 'mailer_password': 'MAILER_PASSWORD'} | |
| 204 | legacy_env = (root.parent / 'home-infra/.env').read_text().splitlines() | |
| 205 | fingerprints = {} | |
| 206 | for target, key in keys.items(): | |
| 207 | values = [line.split('=', 1)[1] for line in legacy_env if line.startswith(key + '=')] | |
| 208 | if len(values) != 1 or not values[0]: | |
| 209 | raise ValueError(f'Legacy secret unavailable: {key}') | |
| 210 | fingerprints[target] = hashlib.sha256(values[0].encode()).hexdigest() | |
| 211 | manifest['databases']['evil-forgejo']['secretSha256'] = fingerprints | |
| 212 | manifest['files'] = {str(path.relative_to(destination)): {'bytes': path.stat().st_size, 'sha256': sha256(path)} | |
| 213 | for path in destination.rglob('*') if path.is_file() and not path.is_symlink()} | |
| 214 | (destination / 'manifest.json').write_text(json.dumps(manifest, indent=2) + '\n') | |
| 215 | final = root / handoff_id | |
| 216 | destination.rename(final) | |
| 217 | os.sync() | |
| 218 | print(f'Verified PostgreSQL handoff: {final}', flush=True) | |
| 219 | finally: | |
| 220 | shutil.rmtree(work) |
tools/import-forgejo-to-shale.py created+240| ... | ... | @@ -0,0 +1,240 @@ |
| 1 | #!/usr/bin/env python3 | |
| 2 | import argparse | |
| 3 | from datetime import datetime, timezone | |
| 4 | import hashlib | |
| 5 | import json | |
| 6 | import os | |
| 7 | from pathlib import Path | |
| 8 | import re | |
| 9 | import secrets | |
| 10 | import shutil | |
| 11 | import sqlite3 | |
| 12 | import tempfile | |
| 13 | import urllib.error | |
| 14 | import urllib.request | |
| 15 | ||
| 16 | ||
| 17 | SOURCE = Path('/mnt/storage1/apps/forgejo/git/repositories') | |
| 18 | ACCESS = ('access_git_webui', 'access_git_proto', 'access_git_proto_push', 'access_issues', | |
| 19 | 'access_issues_submit', 'access_issues_comment', 'access_readme') | |
| 20 | ||
| 21 | ||
| 22 | def sha256(path): | |
| 23 | digest = hashlib.sha256() | |
| 24 | with path.open('rb') as source: | |
| 25 | for block in iter(lambda: source.read(1024 * 1024), b''): | |
| 26 | digest.update(block) | |
| 27 | return digest.hexdigest() | |
| 28 | ||
| 29 | ||
| 30 | def refs(repository): | |
| 31 | result = {} | |
| 32 | packed = repository / 'packed-refs' | |
| 33 | if packed.exists(): | |
| 34 | for line in packed.read_text().splitlines(): | |
| 35 | if line and line[0] not in '#^': | |
| 36 | value, name = line.split(' ') | |
| 37 | result[name] = value | |
| 38 | for path in (repository / 'refs').rglob('*'): | |
| 39 | if path.is_symlink(): | |
| 40 | raise ValueError('Repository refs contain a link. Resolve it before importing') | |
| 41 | if path.is_file(): | |
| 42 | result[str(path.relative_to(repository))] = path.read_text().strip() | |
| 43 | for name, value in result.items(): | |
| 44 | if not name.startswith('refs/') or any(part in ('', '.', '..') for part in name.split('/')): | |
| 45 | raise ValueError('Repository has an unsafe ref name. Review its retained refs') | |
| 46 | if not re.fullmatch('[0-9a-f]{40}', value): | |
| 47 | raise ValueError('Repository has a symbolic or unsupported ref. Resolve it before importing') | |
| 48 | return result | |
| 49 | ||
| 50 | ||
| 51 | def stopped(): | |
| 52 | token = os.environ.get('NOMAD_TOKEN') or Path('/var/lib/studio/nomad.token').read_text().strip() | |
| 53 | def read(path): | |
| 54 | request = urllib.request.Request('http://127.0.0.1:4646/v1/' + path, | |
| 55 | headers={'X-Nomad-Token': token}) | |
| 56 | with urllib.request.urlopen(request, timeout=10) as response: | |
| 57 | return json.load(response) | |
| 58 | try: | |
| 59 | job = read('job/shale') | |
| 60 | except urllib.error.HTTPError as error: | |
| 61 | if error.code == 404: | |
| 62 | return | |
| 63 | raise | |
| 64 | if not job.get('Stop') or any(item['ClientStatus'] in ('pending', 'running') | |
| 65 | for item in read('job/shale/allocations')): | |
| 66 | raise ValueError('Stop Shale and wait for its allocations before importing') | |
| 67 | ||
| 68 | ||
| 69 | def main(): | |
| 70 | parser = argparse.ArgumentParser(description='Import retained personal Forgejo histories into stopped Shale.') | |
| 71 | parser.add_argument('--target', type=Path, required=True, help='Copied Shale service root') | |
| 72 | parser.add_argument('--metadata', type=Path, required=True, help='Personal Forgejo repository metadata JSON') | |
| 73 | args = parser.parse_args() | |
| 74 | if os.geteuid() != 0: | |
| 75 | parser.error('Run as root on Zenith') | |
| 76 | os.umask(0o077) | |
| 77 | target = args.target.resolve() | |
| 78 | retained = Path('/mnt/storage1/apps') | |
| 79 | if target.is_relative_to(retained) or retained.is_relative_to(target): | |
| 80 | raise ValueError('Target overlaps retained app data. Use the managed Shale copy') | |
| 81 | database = target / 'data/astheno.shale.db' | |
| 82 | if not database.is_file(): | |
| 83 | raise ValueError('Shale database is missing. Import its retained state first') | |
| 84 | metadata = json.loads(args.metadata.read_text()) | |
| 85 | required = {'owner_name', 'lower_name', 'name', 'is_private', 'default_branch', 'is_empty', 'is_mirror'} | |
| 86 | if not isinstance(metadata, list) or any(not isinstance(row, dict) or not required.issubset(row) for row in metadata): | |
| 87 | raise ValueError('Forgejo metadata is incomplete. Export the documented personal repository fields') | |
| 88 | if any(not re.fullmatch(r'[A-Za-z0-9_.-]+', row[field]) or row[field] in ('.', '..') | |
| 89 | for row in metadata for field in ('owner_name', 'lower_name', 'name')): | |
| 90 | raise ValueError('Repository metadata has unsafe names. Review the personal Forgejo export') | |
| 91 | indexed = {(row['owner_name'], row['lower_name']): row for row in metadata} | |
| 92 | if len(indexed) != len(metadata) or any(type(row[field]) is not bool for row in metadata | |
| 93 | for field in ('is_private', 'is_empty', 'is_mirror')): | |
| 94 | raise ValueError('Forgejo metadata has duplicate repositories or missing visibility') | |
| 95 | sources = sorted(SOURCE.glob('*/*.git')) | |
| 96 | if not sources: | |
| 97 | raise ValueError('No retained personal Forgejo repositories were found') | |
| 98 | if any(path.is_file() for path in (SOURCE.parent / 'lfs').rglob('*')): | |
| 99 | raise ValueError('Forgejo has LFS data. Preserve its serving path before migrating repositories') | |
| 100 | for source in sources: | |
| 101 | if source.is_symlink() or (source.parent.name, source.name[:-4]) not in indexed: | |
| 102 | raise ValueError('A retained repository has no verified metadata. Export its personal Forgejo database entry') | |
| 103 | if any((source / folder).is_symlink() for folder in ('objects', 'refs')): | |
| 104 | raise ValueError('Repository storage contains a link. Preserve its contents before importing') | |
| 105 | if (source / 'shallow').exists() or (source / 'objects/info/alternates').exists(): | |
| 106 | raise ValueError('Repository history depends on external objects. Preserve them before importing') | |
| 107 | if any((source / 'objects/pack').glob('*.promisor')): | |
| 108 | raise ValueError('Repository has promised objects. Retrieve the complete history before importing') | |
| 109 | expected_paths = {(row['owner_name'], row['lower_name']) for row in metadata if not row['is_empty']} | |
| 110 | if not expected_paths.issubset({(source.parent.name, source.name[:-4]) for source in sources}): | |
| 111 | raise ValueError('A nonempty Forgejo repository is missing from retained storage') | |
| 112 | stopped() | |
| 113 | evidence = Path(tempfile.mkdtemp(prefix='shale-forgejo-import-', dir='/var/lib/studio')) | |
| 114 | db = sqlite3.connect(database) | |
| 115 | db.row_factory = sqlite3.Row | |
| 116 | if db.execute('PRAGMA integrity_check').fetchone()[0] != 'ok': | |
| 117 | raise ValueError('Shale database failed integrity checking. Recover the retained copy') | |
| 118 | owners = db.execute("SELECT id FROM users WHERE name='snow'").fetchall() | |
| 119 | if len(owners) != 1: | |
| 120 | raise ValueError('Shale needs one existing snow account. Preserve its original identity before importing') | |
| 121 | owner = owners[0]['id'] | |
| 122 | stat = database.stat() | |
| 123 | with sqlite3.connect(evidence / 'before.db') as backup: | |
| 124 | db.backup(backup) | |
| 125 | existing = {row['name'].casefold(): dict(row) for row in db.execute( | |
| 126 | 'SELECT id,uuid,owner,name,' + ','.join(ACCESS) + ' FROM repositories')} | |
| 127 | records = [] | |
| 128 | now = datetime.now(timezone.utc).isoformat(timespec='seconds') | |
| 129 | try: | |
| 130 | for key, row in indexed.items(): | |
| 131 | name = row['name'] if key[0] == 'clo' else key[0] + '/' + row['name'] | |
| 132 | entry = existing.get(name.casefold()) | |
| 133 | if entry and entry['owner'] != owner: | |
| 134 | raise ValueError('Repository belongs to another Shale account. Review its ownership before merging') | |
| 135 | if entry and row['is_private']: | |
| 136 | for field in ACCESS: | |
| 137 | db.execute(f"UPDATE repositories SET {field}='private' WHERE id=? AND {field} IN ('public','unlisted')", (entry['id'],)) | |
| 138 | db.commit() | |
| 139 | db.execute('BEGIN IMMEDIATE') | |
| 140 | for key, row in sorted(indexed.items()): | |
| 141 | source = SOURCE / key[0] / (key[1] + '.git') | |
| 142 | name = row['name'] if key[0] == 'clo' else key[0] + '/' + row['name'] | |
| 143 | entry = existing.get(name.casefold()) | |
| 144 | if entry: | |
| 145 | directory = target / 'repositories_owned' / entry['uuid'] | |
| 146 | if not directory.is_dir(): | |
| 147 | raise ValueError('An existing Shale repository directory is missing. Recover it before merging') | |
| 148 | else: | |
| 149 | alphabet = '0123456789ABCDEFGHJKMNPQRSTVWXYZ' | |
| 150 | number = secrets.randbits(128) | |
| 151 | repo_id = ''.join(alphabet[(number >> (5 * position)) & 31] for position in range(25, -1, -1)) | |
| 152 | directory = target / 'repositories_owned' / repo_id | |
| 153 | directory.mkdir(mode=0o700) | |
| 154 | os.chown(directory, stat.st_uid, stat.st_gid) | |
| 155 | for folder in ('objects', 'refs'): | |
| 156 | (directory / folder).mkdir(mode=0o700) | |
| 157 | os.chown(directory / folder, stat.st_uid, stat.st_gid) | |
| 158 | (directory / 'config').write_text('[core]\nrepositoryformatversion = 0\nbare = true\n') | |
| 159 | access = 'private' if row['is_private'] else 'public' | |
| 160 | cursor = db.execute( | |
| 161 | 'INSERT INTO repositories(uuid,owner,created_on,name,description,' + ','.join(ACCESS) + ',last_updated) ' | |
| 162 | 'VALUES(?,?,?,?,?,?,?,?,?,?,?,?,?)', | |
| 163 | (repo_id, owner, now, name, '', access, access, 'private', 'off', 'off', 'off', access, now)) | |
| 164 | entry = {'id': cursor.lastrowid, 'uuid': repo_id, 'name': name} | |
| 165 | existing[name.casefold()] = entry | |
| 166 | previous = refs(directory) | |
| 167 | original = refs(source) if source.exists() else {} | |
| 168 | objects = {} | |
| 169 | for path in sorted((source / 'objects').rglob('*')) if source.exists() else (): | |
| 170 | if path.is_symlink(): | |
| 171 | raise ValueError('Repository objects contain a link. Preserve its contents before importing') | |
| 172 | relative = path.relative_to(source / 'objects') | |
| 173 | if not path.is_file() or relative.parts[0] == 'info': | |
| 174 | continue | |
| 175 | checksum = sha256(path) | |
| 176 | destination = directory / 'objects' / relative | |
| 177 | destination.parent.mkdir(mode=0o700, parents=True, exist_ok=True) | |
| 178 | if destination.exists(): | |
| 179 | if sha256(destination) != checksum: | |
| 180 | raise ValueError('Object files differ under the same name. Keep Shale stopped and review the copies') | |
| 181 | else: | |
| 182 | shutil.copyfile(path, destination) | |
| 183 | os.chmod(destination, 0o600) | |
| 184 | os.chown(destination, stat.st_uid, stat.st_gid) | |
| 185 | if sha256(destination) != checksum: | |
| 186 | raise RuntimeError('Copied objects failed verification. Keep Shale stopped') | |
| 187 | objects[str(relative)] = checksum | |
| 188 | mapped = {} | |
| 189 | combined = dict(previous) | |
| 190 | for reference, value in original.items(): | |
| 191 | destination = reference | |
| 192 | if destination in combined and combined[destination] != value: | |
| 193 | category = 'heads' if reference.startswith('refs/heads/') else 'tags' if reference.startswith('refs/tags/') else None | |
| 194 | destination = ('refs/' + category + '/forgejo/' + key[0] + '/' + reference.split('/', 2)[2] | |
| 195 | if category else 'refs/forgejo/' + key[0] + '/' + reference[5:]) | |
| 196 | if destination in combined and combined[destination] != value: | |
| 197 | destination += '-'+value | |
| 198 | path = directory / destination | |
| 199 | path.parent.mkdir(mode=0o700, parents=True, exist_ok=True) | |
| 200 | path.write_text(value + '\n') | |
| 201 | os.chown(path, stat.st_uid, stat.st_gid) | |
| 202 | combined[destination] = value | |
| 203 | mapped[reference] = destination | |
| 204 | if not (directory / 'HEAD').exists(): | |
| 205 | desired = 'refs/heads/' + row['default_branch'] | |
| 206 | if original and desired not in original: | |
| 207 | desired = next((reference for reference in sorted(original) if reference.startswith('refs/heads/')), desired) | |
| 208 | (directory / 'HEAD').write_text('ref: ' + desired + '\n') | |
| 209 | os.chown(directory / 'HEAD', stat.st_uid, stat.st_gid) | |
| 210 | os.chown(directory / 'config', stat.st_uid, stat.st_gid) | |
| 211 | final = refs(directory) | |
| 212 | if any(final.get(reference) != value for reference, value in previous.items()) or any( | |
| 213 | final.get(mapped[reference]) != value for reference, value in original.items() | |
| 214 | ): | |
| 215 | raise RuntimeError('Ref verification failed. Keep Shale stopped and review the retained refs') | |
| 216 | if source.exists() and refs(source) != original: | |
| 217 | raise RuntimeError('Source refs changed during import. Keep Shale stopped') | |
| 218 | for parent in (directory / 'objects', directory / 'refs'): | |
| 219 | for path in parent.rglob('*'): | |
| 220 | if path.is_dir(): | |
| 221 | os.chown(path, stat.st_uid, stat.st_gid) | |
| 222 | records.append({'source': '/'.join(key), 'target': entry['name'], 'uuid': entry['uuid'], | |
| 223 | 'private': row['is_private'], 'sourceMirror': row['is_mirror'], 'objects': objects, | |
| 224 | 'originalRefs': original, 'refMapping': mapped, 'previousRefs': previous, | |
| 225 | 'head': (directory / 'HEAD').read_text().strip()}) | |
| 226 | stopped() | |
| 227 | if db.execute('PRAGMA foreign_key_check').fetchall(): | |
| 228 | raise RuntimeError('Shale database references are invalid. Keep it stopped') | |
| 229 | db.commit() | |
| 230 | finally: | |
| 231 | db.close() | |
| 232 | (evidence / 'repositories.json').write_text(json.dumps(records, indent=2) + '\n') | |
| 233 | print(f'Imported and verified {len(records)} personal repositories. Evidence and previous database: {evidence}') | |
| 234 | ||
| 235 | ||
| 236 | if __name__ == '__main__': | |
| 237 | try: | |
| 238 | main() | |
| 239 | except (ValueError, RuntimeError, OSError, sqlite3.Error, urllib.error.HTTPError) as error: | |
| 240 | raise SystemExit(str(error)) from None |
tools/import-personal-files.py created+114| ... | ... | @@ -0,0 +1,114 @@ |
| 1 | #!/usr/bin/env python3 | |
| 2 | import argparse | |
| 3 | import json | |
| 4 | import os | |
| 5 | from pathlib import Path | |
| 6 | import sqlite3 | |
| 7 | import subprocess | |
| 8 | import time | |
| 9 | import urllib.error | |
| 10 | import urllib.request | |
| 11 | ||
| 12 | ||
| 13 | parser = argparse.ArgumentParser(description="Copy retained app files on the new host before starting the service.") | |
| 14 | parser.add_argument("service", choices=["tailscale", "ddns-updater", "copyparty", "ytdl-sub", "pds"]) | |
| 15 | ||
| 16 | ||
| 17 | def main(): | |
| 18 | service = parser.parse_args().service | |
| 19 | if os.geteuid() != 0: | |
| 20 | raise SystemExit("Run this importer as root on the production host.") | |
| 21 | token = Path("/var/lib/studio/nomad.token").read_text().strip() | |
| 22 | request = urllib.request.Request( | |
| 23 | "http://127.0.0.1:4646/v1/job/" + service, | |
| 24 | headers={"X-Nomad-Token": token}, | |
| 25 | ) | |
| 26 | try: | |
| 27 | with urllib.request.urlopen(request, timeout=10) as response: | |
| 28 | job = json.load(response) | |
| 29 | if not job.get("Stop"): | |
| 30 | raise SystemExit("Stop " + service + " before copying its files.") | |
| 31 | except urllib.error.HTTPError as error: | |
| 32 | if error.code != 404: | |
| 33 | raise | |
| 34 | request.full_url += "/allocations" | |
| 35 | try: | |
| 36 | with urllib.request.urlopen(request, timeout=10) as response: | |
| 37 | allocations = json.load(response) | |
| 38 | except urllib.error.HTTPError as error: | |
| 39 | if error.code != 404: | |
| 40 | raise | |
| 41 | allocations = [] | |
| 42 | if any(item["ClientStatus"] not in ("complete", "failed", "lost") for item in allocations): | |
| 43 | raise SystemExit("Wait for " + service + " allocations to stop before copying its files.") | |
| 44 | ||
| 45 | root = Path("/srv/prod") / service | |
| 46 | dataset = subprocess.check_output(["findmnt", "-n", "-o", "SOURCE", "--mountpoint", str(root)], text=True).strip() | |
| 47 | if dataset != "globe/prod/" + service: | |
| 48 | raise SystemExit("Mount globe/prod/" + service + " at " + str(root) + " before copying files.") | |
| 49 | uid = 0 if service == "tailscale" else json.loads(Path("/var/lib/studio/identities.json").read_text())[service] | |
| 50 | directories = { | |
| 51 | "tailscale": ("tailscale", "var/lib/tailscale"), | |
| 52 | "ddns-updater": ("ddns-updater", "updater/data"), | |
| 53 | "copyparty": ("copyparty/copyparty", "cfg/copyparty"), | |
| 54 | "ytdl-sub": ("ytdl-sub", "config"), | |
| 55 | "pds": ("pds", "pds"), | |
| 56 | } | |
| 57 | source_name, target_name = directories[service] | |
| 58 | source = Path("/mnt/storage1/apps") / source_name | |
| 59 | target = root / target_name | |
| 60 | if not source.is_dir(): | |
| 61 | raise SystemExit("Retained app files are missing at " + str(source)) | |
| 62 | if service == "tailscale": | |
| 63 | json.loads((source / "tailscaled.state").read_text()) | |
| 64 | elif service == "ddns-updater": | |
| 65 | json.loads((source / "updates.json").read_text()) | |
| 66 | elif service == "copyparty": | |
| 67 | for database in source.glob("*.db"): | |
| 68 | with sqlite3.connect("file:" + str(database) + "?mode=ro", uri=True) as connection: | |
| 69 | if connection.execute("PRAGMA integrity_check").fetchone()[0] != "ok": | |
| 70 | raise SystemExit("Check the retained Copyparty database before copying " + database.name) | |
| 71 | if not (source / "shares.db").is_file(): | |
| 72 | raise SystemExit("Restore the retained Copyparty shares.db before importing.") | |
| 73 | elif service == "pds": | |
| 74 | for name in ("account", "sequencer", "did_cache"): | |
| 75 | if not (source / (name + ".sqlite")).is_file(): | |
| 76 | raise SystemExit("Restore the retained PDS " + name + " database before importing.") | |
| 77 | if list(source.rglob("*-wal")) or list(source.rglob("*-shm")): | |
| 78 | raise SystemExit("Back up all PDS SQLite databases, including actor stores, before importing WAL files.") | |
| 79 | for database in source.rglob("*.sqlite"): | |
| 80 | with sqlite3.connect("file:" + str(database) + "?mode=ro&immutable=1", uri=True) as connection: | |
| 81 | if connection.execute("PRAGMA integrity_check").fetchone()[0] != "ok": | |
| 82 | raise SystemExit("Check the retained PDS database before importing " + database.name) | |
| 83 | ||
| 84 | snapshot = dataset + "@before-files-import-" + str(time.time_ns()) | |
| 85 | subprocess.run(["zfs", "snapshot", snapshot], check=True) | |
| 86 | target.mkdir(parents=True, exist_ok=True) | |
| 87 | exclusions = ["--exclude=*.lock", "--exclude=/.ytdl-sub-lock", "--exclude=/.cache/", "--exclude=/work/"] | |
| 88 | subprocess.run(["rsync", "-aH", "--delete", *exclusions, str(source) + "/", str(target) + "/"], check=True) | |
| 89 | drift = subprocess.check_output([ | |
| 90 | "rsync", "-rlnc", "--delete", "--out-format=%n", *exclusions, str(source) + "/", str(target) + "/", | |
| 91 | ], text=True) | |
| 92 | if drift: | |
| 93 | raise SystemExit("Copied files differ from the retained source. Restore " + snapshot + " before retrying.") | |
| 94 | os.chown(target, uid, uid) | |
| 95 | for directory, children, files in os.walk(target): | |
| 96 | for name in children + files: | |
| 97 | os.chown(Path(directory) / name, uid, uid, follow_symlinks=False) | |
| 98 | if service == "tailscale": | |
| 99 | os.chmod(target / "tailscaled.state", 0o600) | |
| 100 | elif service == "copyparty": | |
| 101 | for database in target.glob("*.db"): | |
| 102 | with sqlite3.connect("file:" + str(database) + "?mode=ro", uri=True) as connection: | |
| 103 | if connection.execute("PRAGMA integrity_check").fetchone()[0] != "ok": | |
| 104 | raise SystemExit("Copied Copyparty database needs recovery from " + snapshot) | |
| 105 | elif service == "pds": | |
| 106 | for database in target.rglob("*.sqlite"): | |
| 107 | with sqlite3.connect("file:" + str(database) + "?mode=ro&immutable=1", uri=True) as connection: | |
| 108 | if connection.execute("PRAGMA integrity_check").fetchone()[0] != "ok": | |
| 109 | raise SystemExit("Copied PDS database needs recovery from " + snapshot) | |
| 110 | print(json.dumps({"service": service, "snapshot": snapshot, "copied": True})) | |
| 111 | ||
| 112 | ||
| 113 | if __name__ == "__main__": | |
| 114 | main() |
tools/import-personal-postgres.py created+215| ... | ... | @@ -0,0 +1,215 @@ |
| 1 | #!/usr/bin/env python3 | |
| 2 | import argparse | |
| 3 | import hashlib | |
| 4 | import json | |
| 5 | import os | |
| 6 | from pathlib import Path | |
| 7 | import re | |
| 8 | import subprocess | |
| 9 | import tempfile | |
| 10 | import urllib.error | |
| 11 | import urllib.request | |
| 12 | ||
| 13 | ||
| 14 | def identifier(value): | |
| 15 | return '"' + value.replace('"', '""') + '"' | |
| 16 | ||
| 17 | ||
| 18 | def digest(path): | |
| 19 | result = hashlib.sha256() | |
| 20 | with path.open('rb') as stream: | |
| 21 | for block in iter(lambda: stream.read(1024 * 1024), b''): | |
| 22 | result.update(block) | |
| 23 | return result.hexdigest() | |
| 24 | ||
| 25 | ||
| 26 | def fingerprint(sql, database): | |
| 27 | relations = json.loads(sql(database, """ | |
| 28 | SELECT coalesce(json_agg(x ORDER BY n, c), '[]') FROM ( | |
| 29 | SELECT n.nspname n, c.relname c, c.relkind k | |
| 30 | FROM pg_class c JOIN pg_namespace n ON n.oid=c.relnamespace | |
| 31 | WHERE n.nspname NOT LIKE 'pg_%' AND n.nspname <> 'information_schema' | |
| 32 | AND c.relkind IN ('r','m','S') | |
| 33 | ) x | |
| 34 | """)) | |
| 35 | result = {} | |
| 36 | for relation in relations: | |
| 37 | table = identifier(relation['n']) + '.' + identifier(relation['c']) | |
| 38 | query = (f'SELECT last_value,is_called FROM {table}' if relation['k'] == 'S' else | |
| 39 | f'COPY (SELECT row_to_json(t)::text FROM {table} t ' | |
| 40 | 'ORDER BY row_to_json(t)::text COLLATE "C") TO STDOUT') | |
| 41 | content = sql(database, query) | |
| 42 | result[table] = {'rows': len(content.splitlines()), | |
| 43 | 'sha256': hashlib.sha256(content).hexdigest()} | |
| 44 | content = sql(database, "COPY (SELECT loid,pageno,encode(data,'hex') " | |
| 45 | 'FROM pg_largeobject ORDER BY loid,pageno) TO STDOUT') | |
| 46 | result['large_objects'] = {'rows': len(content.splitlines()), | |
| 47 | 'sha256': hashlib.sha256(content).hexdigest()} | |
| 48 | return result | |
| 49 | ||
| 50 | ||
| 51 | def main(): | |
| 52 | parser = argparse.ArgumentParser(description='Restore a personal database before starting its service.') | |
| 53 | parser.add_argument('service', choices=('keycloak', 'dawarich')) | |
| 54 | parser.add_argument('handoff', type=Path) | |
| 55 | parser.add_argument('--container', help='Restore inside the isolated initializer before deploying Postgres') | |
| 56 | args = parser.parse_args() | |
| 57 | if os.geteuid() != 0: | |
| 58 | parser.error('Run as root on Zenith') | |
| 59 | os.umask(0o077) | |
| 60 | handoff = args.handoff.resolve() | |
| 61 | cluster = json.loads((handoff / 'manifest.json').read_text())['clusters']['18'] | |
| 62 | if not cluster.get('restoreVerified'): | |
| 63 | raise ValueError('PostgreSQL 18 backup is unverified. Use the verified handoff') | |
| 64 | entry = cluster['databases'][args.service] | |
| 65 | dump = (handoff / entry['file']).resolve() | |
| 66 | if not dump.is_relative_to(handoff / 'pg18') or dump.suffix != '.dump': | |
| 67 | raise ValueError('Dump is outside the PostgreSQL 18 handoff') | |
| 68 | if dump.stat().st_size != entry['bytes'] or digest(dump) != entry['sha256']: | |
| 69 | raise ValueError('Backup checksum differs. Recover the verified handoff before importing') | |
| 70 | ||
| 71 | token = os.environ.get('NOMAD_TOKEN') or Path('/var/lib/studio/nomad.token').read_text().strip() | |
| 72 | ||
| 73 | def nomad(path): | |
| 74 | request = urllib.request.Request('http://127.0.0.1:4646/v1/' + path, | |
| 75 | headers={'X-Nomad-Token': token}) | |
| 76 | with urllib.request.urlopen(request, timeout=10) as response: | |
| 77 | return json.load(response) | |
| 78 | ||
| 79 | def stopped(service): | |
| 80 | try: | |
| 81 | job = nomad('job/' + service) | |
| 82 | except urllib.error.HTTPError as error: | |
| 83 | if error.code == 404: | |
| 84 | return | |
| 85 | raise | |
| 86 | if not job.get('Stop') or any( | |
| 87 | item['ClientStatus'] in ('pending', 'running') | |
| 88 | for item in nomad('job/' + service + '/allocations') | |
| 89 | ): | |
| 90 | raise ValueError(f'Stop {service} and wait for its allocations before importing') | |
| 91 | ||
| 92 | stopped(args.service) | |
| 93 | connection = nomad('var/nomad/jobs/' + args.service + '/inputs/database')['Items'] | |
| 94 | database, owner = connection['name'], connection['username'] | |
| 95 | expected_database = 'keycloak_next' if args.service == 'keycloak' else 'dawarich' | |
| 96 | if database != expected_database or owner != 'svc_' + database: | |
| 97 | raise ValueError('Database variable differs from the production service. Allocate its database first') | |
| 98 | if args.container: | |
| 99 | stopped('postgres') | |
| 100 | else: | |
| 101 | allocations = [item['ID'] for item in nomad('job/postgres/allocations') | |
| 102 | if item['ClientStatus'] == 'running' and item['DesiredStatus'] == 'run'] | |
| 103 | if len(allocations) != 1: | |
| 104 | raise ValueError('Postgres needs exactly one running allocation') | |
| 105 | folder = Path(tempfile.mkdtemp(prefix=args.service + '-import-', dir='/var/lib/studio')) | |
| 106 | diagnostics = folder / 'diagnostics.log' | |
| 107 | podman = ['podman'] | |
| 108 | ||
| 109 | def run(command, *, source=None, output=None): | |
| 110 | with diagnostics.open('ab') as errors: | |
| 111 | result = subprocess.run(command, input=source if isinstance(source, bytes) else None, | |
| 112 | stdin=source if source is not None and not isinstance(source, bytes) else None, | |
| 113 | stdout=output or subprocess.PIPE, stderr=errors) | |
| 114 | if result.returncode: | |
| 115 | raise RuntimeError(f'{Path(command[0]).name} failed. See {diagnostics}') | |
| 116 | return result.stdout | |
| 117 | ||
| 118 | if args.container: | |
| 119 | container = json.loads(run(podman + ['inspect', args.container]))[0] | |
| 120 | if not container['State']['Running'] or container['HostConfig']['NetworkMode'] != 'none': | |
| 121 | raise ValueError('Initializer must be running with --network none') | |
| 122 | container_id = container['Id'] | |
| 123 | else: | |
| 124 | containers = [parts[0] for line in run(podman + ['ps', '--format', '{{.ID}} {{.Names}}']).decode().splitlines() | |
| 125 | if len(parts := line.split()) == 2 and parts[1].endswith(allocations[0])] | |
| 126 | if len(containers) != 1: | |
| 127 | raise ValueError('Postgres allocation container is unavailable') | |
| 128 | container_id = containers[0] | |
| 129 | execute = podman + ['exec', '-i', container_id] | |
| 130 | ||
| 131 | def sql(db, query): | |
| 132 | return run(execute + ['psql', '-X', '-U', 'postgres', '-d', db, | |
| 133 | '-At', '-v', 'ON_ERROR_STOP=1'], source=(query + '\n').encode()) | |
| 134 | ||
| 135 | current_owner = sql('postgres', f"SELECT pg_get_userbyid(datdba) FROM pg_database WHERE datname='{database}'").decode().strip() | |
| 136 | if current_owner != owner: | |
| 137 | raise ValueError('Production database is missing or has another owner. Allocate it with the Postgres provider') | |
| 138 | with dump.open('rb') as source: | |
| 139 | listing = run(execute + ['pg_restore', '--list'], source=source).decode() | |
| 140 | extensions = re.findall(r'^\d+; \d+ \d+ EXTENSION - (\S+) ', listing, re.MULTILINE) | |
| 141 | if set(extensions) - {'plpgsql', 'postgis'} or ('postgis' in extensions) != (args.service == 'dawarich'): | |
| 142 | raise ValueError('Backup extensions differ from the service. Review the personal database before importing') | |
| 143 | extension_data, application = [], [] | |
| 144 | for line in listing.splitlines(): | |
| 145 | if re.match(r'^\d+; \d+ \d+ (EXTENSION |COMMENT - EXTENSION |SCHEMA - public )', line): | |
| 146 | continue | |
| 147 | if re.match(r'^\d+; \d+ \d+ TABLE DATA public spatial_ref_sys ', line): | |
| 148 | extension_data.append(line) | |
| 149 | else: | |
| 150 | application.append(line) | |
| 151 | contents = {name: value for name, value in entry['contents'].items() if name != 'large_object_owners'} | |
| 152 | restore_list = '/tmp/' + folder.name + '.list' | |
| 153 | run(execute + ['tee', restore_list], source=('\n'.join(application) + '\n').encode()) | |
| 154 | try: | |
| 155 | stopped(args.service) | |
| 156 | if args.container: | |
| 157 | stopped('postgres') | |
| 158 | if sql('postgres', f"SELECT count(*) FROM pg_stat_activity WHERE datname='{database}'").strip() != b'0': | |
| 159 | raise ValueError('Database has connected clients. Stop them before importing') | |
| 160 | backup = folder / 'before.dump' | |
| 161 | with backup.open('wb') as output: | |
| 162 | run(execute + ['pg_dump', '-U', 'postgres', '-Fc', database], output=output) | |
| 163 | if not backup.stat().st_size: | |
| 164 | raise RuntimeError('Target backup is empty. Import stopped') | |
| 165 | sql('postgres', f'DROP DATABASE {database};') | |
| 166 | sql('postgres', f'CREATE DATABASE {database} OWNER {owner};') | |
| 167 | sql(database, f'ALTER SCHEMA public OWNER TO {owner};') | |
| 168 | if args.service == 'dawarich': | |
| 169 | sql(database, 'CREATE EXTENSION postgis;') | |
| 170 | with dump.open('rb') as source: | |
| 171 | run(execute + ['pg_restore', '-U', 'postgres', '-d', database, '--no-owner', '--no-acl', | |
| 172 | '--role=' + owner, '--single-transaction', '--exit-on-error', | |
| 173 | '--use-list=' + restore_list], source=source) | |
| 174 | if extension_data: | |
| 175 | run(execute + ['tee', restore_list], source=('\n'.join(extension_data) + '\n').encode()) | |
| 176 | with dump.open('rb') as source: | |
| 177 | extension_sql = run(execute + ['pg_restore', '-U', 'postgres', '--data-only', '--no-owner', | |
| 178 | '--no-acl', '--use-list=' + restore_list, '--file=-'], source=source) | |
| 179 | srids = [] | |
| 180 | copying = False | |
| 181 | for line in extension_sql.splitlines(): | |
| 182 | if line.startswith(b'COPY public.spatial_ref_sys ('): | |
| 183 | copying = True | |
| 184 | elif copying and line == b'\\.': | |
| 185 | copying = False | |
| 186 | elif copying: | |
| 187 | srids.append(int(line.split(b'\t', 1)[0])) | |
| 188 | # PostGIS dumps include custom SRIDs; the extension supplies built-in definitions. | |
| 189 | if srids: | |
| 190 | sql(database, 'DELETE FROM public.spatial_ref_sys WHERE srid IN (' + ','.join(map(str, srids)) + ');') | |
| 191 | with dump.open('rb') as source: | |
| 192 | run(execute + ['pg_restore', '-U', 'postgres', '-d', database, '--no-owner', '--no-acl', | |
| 193 | '--data-only', '--single-transaction', '--exit-on-error', | |
| 194 | '--use-list=' + restore_list], source=source) | |
| 195 | actual = fingerprint(sql, database) | |
| 196 | if actual != contents: | |
| 197 | changed = sorted(name for name in actual.keys() | contents.keys() if actual.get(name) != contents.get(name)) | |
| 198 | (folder / 'differences.json').write_text(json.dumps(changed, indent=2) + '\n') | |
| 199 | raise RuntimeError(f'Restored data differs. Keep {args.service} stopped and review {folder}') | |
| 200 | if dump.stat().st_size != entry['bytes'] or digest(dump) != entry['sha256']: | |
| 201 | raise RuntimeError('Backup changed during import. Keep the service stopped') | |
| 202 | (folder / 'verified.json').write_text(json.dumps({ | |
| 203 | 'service': args.service, 'database': database, 'source': str(dump), | |
| 204 | 'sha256': entry['sha256'], 'contents': actual, | |
| 205 | }, indent=2) + '\n') | |
| 206 | print(f'Imported and verified {args.service}. Previous database: {backup}') | |
| 207 | finally: | |
| 208 | run(execute + ['rm', '-f', restore_list]) | |
| 209 | ||
| 210 | ||
| 211 | if __name__ == '__main__': | |
| 212 | try: | |
| 213 | main() | |
| 214 | except (ValueError, RuntimeError, OSError, KeyError, urllib.error.HTTPError) as error: | |
| 215 | raise SystemExit(str(error)) from None |
tools/import-qbittorrent.sh+2-2| ... | ... | @@ -21,8 +21,8 @@ if [[ $offline == false ]]; then |
| 21 | 21 | source_running=$(ssh "$source_host" 'sudo -n docker inspect -f "{{.State.Running}}" qbittorrent') |
| 22 | 22 | [[ $source_running == false ]] || { echo 'Stop Zenith qBittorrent before importing its profile' >&2; exit 1; } |
| 23 | 23 | fi |
| 24 | "${remote[@]}" 'test "$(findmnt -n -o FSTYPE --mountpoint /srv/clover/Media)" = zfs; test -d /srv/clover/Media/seedbox; test -L /srv/clover/Media/torrent; test "$(readlink /srv/clover/Media/torrent)" = seedbox' || { | |
| 25 | echo 'Mount Media and establish the seedbox/torrent alias before starting qBittorrent' >&2; exit 1; | |
| 24 | "${remote[@]}" 'test "$(findmnt -n -o FSTYPE --mountpoint /srv/clover/Media)" = zfs; test -d /srv/clover/Media/Seedbox; test -L /srv/clover/Media/torrent; test /srv/clover/Media/Seedbox -ef /srv/clover/Media/torrent' || { | |
| 25 | echo 'Mount Media and point torrent at Seedbox before importing qBittorrent' >&2; exit 1; | |
| 26 | 26 | } |
| 27 | 27 | |
| 28 | 28 | response=$("${remote[@]}" 'curl -s -w "\n%{http_code}" -H "X-Nomad-Token: $(cat /var/lib/studio/nomad.token)" http://127.0.0.1:4646/v1/job/qbittorrent') |
tools/import-yt-feed.sh+2-2| ... | ... | @@ -25,8 +25,8 @@ if [[ -n ${STUDIO_LEGACY_HANDOFF:-} ]]; then |
| 25 | 25 | fi |
| 26 | 26 | if [[ $instance == yt-feed ]]; then |
| 27 | 27 | root=/srv/prod/yt-feed |
| 28 | "${remote[@]}" 'test "$(findmnt -n -o FSTYPE --mountpoint /srv/clover/Media)" = zfs; test -d /srv/clover/Media/music-intake; test -L /srv/clover/Media/music_intake; test "$(readlink /srv/clover/Media/music_intake)" = music-intake' || { | |
| 29 | echo 'Mount Media and establish the music-intake alias before importing YouTube Triage' >&2 | |
| 28 | "${remote[@]}" 'test "$(findmnt -n -o FSTYPE --mountpoint /srv/clover/Media)" = zfs; test -d "/srv/clover/Media/Intake - Music"; test -d "/srv/clover/Media/Indie Shows"; test -d /srv/clover/Media/Videos/Independent' || { | |
| 29 | echo 'Mount Media with Intake - Music, Indie Shows, and Videos/Independent before importing YouTube Triage' >&2 | |
| 30 | 30 | exit 1 |
| 31 | 31 | } |
| 32 | 32 | if [[ $offline == false ]]; then |
tools/jellyfin-migration.md+2| ... | ... | @@ -9,3 +9,5 @@ A full preview restore used the existing read-only `storage1/apps@hourly-2026-09 |
| 9 | 9 | Production import (`STUDIO_DEPLOY_HOST=<new-host> STUDIO_DEPLOY_PORT=22 sh tools/import-jellyfin.sh jellyfin`) uses the stopped live source, requires the new Jellyfin job stopped and the real Media ZFS filesystem mounted, snapshots the destination, verifies the copy, assigns Jellyfin's allocated UID, and leaves the new job stopped for cutover. During the preview boot, a three-second Nomad health probe timed out under load and Caddy briefly removed the stage route. The probe now allows 15 seconds; the updated preview deployment passed. The imported Jellyfin settings also automatically installed newer AniList and Intro Skipper plugins after startup. |
| 10 | 10 | |
| 11 | 11 | After a same-machine OS replacement, set `STUDIO_LEGACY_HANDOFF` to the [offline handoff](legacy-handoff.md) directory. The production importer then copies the old app metadata directly between mounted datasets on the new host, without a Mac scratch copy or old Docker daemon. The 3.97 TiB Media dataset remains a ZFS mount and is not copied. |
| 12 | ||
| 13 | The service binds the reorganized host folders at their original `/media/jellyfin/...` container paths using the shared [config/site.pkl](../config/site.pkl) mapping. Both existing library settings and new-instance setup keep those paths; no database path rewrite is needed. See the [current service layout](media-cutover.md#service-paths-for-the-reorganized-tree). This follows [Jellyfin's migration guidance](https://jellyfin.org/docs/general/administration/migrate/) to preserve the paths seen by the application. |
tools/legacy-handoff.md+20-3| ... | ... | @@ -1,10 +1,27 @@ |
| 1 | 1 | # Same-machine PostgreSQL handoff |
| 2 | 2 | |
| 3 | `storage1/apps` is an encrypted ZFS root. Set its mountpoint explicitly to `/mnt/storage1/apps` before changing the pool root, then keep it mounted there after replacing Zenith with NixOS; [media-cutover.md](media-cutover.md) gives the command order. The old app directories and `/mnt/storage1/apps/home-infra/.env` remain there; the new `/srv/prod` root is separate. | |
| 3 | The retained apps dataset is an encrypted ZFS root. Keep its mountpoint explicitly at `/mnt/storage1/apps` after renaming the pool to `globe` and replacing Zenith with NixOS; [media-cutover.md](media-cutover.md) gives the command order. The old app directories and `/mnt/storage1/apps/home-infra/.env` remain there; the new `/srv/prod` root is separate. | |
| 4 | 4 | |
| 5 | At cutover, stop the old apps being migrated, leaving old PostgreSQL running long enough to export. Run `bash tools/export-legacy-postgres.sh` before replacing the OS. Zenith grants the `clo` account passwordless `sudo docker` access for the export; the script checks the exact `storage1/apps` mount and Compose's HedgeDoc label before writing. It writes custom-format dumps and a manifest into a private `studio-handoff/<timestamp>-<suffix>` directory on `storage1/apps`, checks evil.inc Forgejo and HedgeDoc remain stopped, and prints the directory path. Dawarich starts with a fresh database and Redis queue, so its old database is omitted. Snapshot `storage1/apps` after all old writers are stopped and the exporter finishes, then keep the encrypted pool and handoff path intact. The exporter does not stop apps itself. | |
| 5 | If TrueNAS is still running, stop the old apps being migrated, leaving old PostgreSQL running long enough to export. Run `bash tools/export-legacy-postgres.sh` before replacing the OS. Zenith grants the `clo` account passwordless `sudo docker` access for the export; this online exporter requires the original `storage1/apps` mount and exports Forgejo and HedgeDoc only. It does not stop apps itself. | |
| 6 | 6 | |
| 7 | After NixOS has imported and unlocked `storage1/apps`, set `STUDIO_DEPLOY_HOST`, `STUDIO_DEPLOY_PORT`, and `STUDIO_LEGACY_HANDOFF` to that printed directory. The importers require the exact encrypted `storage1/apps` ZFS mount and a stopped destination job. HedgeDoc and evil.inc Forgejo verify dump hashes and source table counts before restoring their PostgreSQL databases. Shale, Jellyfin, Navidrome, PDS, qBittorrent, Sonarr, Radarr, Jackett, and YouTube Triage can copy their retained app data without the old Docker daemon. Each importer prints its pre-import backup or ZFS snapshot. | |
| 7 | If already booted into the installer, [export-offline-postgres.py](export-offline-postgres.py) exports every connectable database and all roles from both retained clusters. It requires all PostgreSQL servers to be stopped, matching-major runtimes with the source extensions installed, and `glibcLocales`. It starts private RAM copies over private Unix sockets, saves complete physical archives including `template0`, checks the original file hashes remain unchanged, and restores every logical dump into fresh temporary clusters. Verification compares every user table's contents, sequence state, and large objects. The originals are never started. Run as root on the installer, with the runtime paths supplied by the pinned Nix builds: | |
| 8 | ||
| 9 | ```sh | |
| 10 | python3 tools/export-offline-postgres.py \ | |
| 11 | --cluster /mnt/storage1/apps/pg_data "$postgres17_runtime" \ | |
| 12 | --cluster /mnt/storage1/apps/pg_data18/18/docker "$postgres18_runtime" \ | |
| 13 | --locale-archive "$glibc_locales/lib/locale/locale-archive" | |
| 14 | ``` | |
| 15 | ||
| 16 | It publishes a private `studio-handoff/<timestamp>-<suffix>` directory only after successful restores. Its `pg17` and `pg18` directories contain all database dumps, `globals.sql`, and `cluster.tar.gz`; the manifest records their hashes and restore evidence. Root-level Forgejo and HedgeDoc dump links select the PG18 exports for the existing importers. Dawarich's old database is retained in the full export even though its new service starts fresh. Copy the complete handoff directory, preserving these links, to a private off-server archive and verify the file hashes before replacing the OS. `globals.sql` contains role password hashes; protect the archive like the recovery keys. Failed exports remain under `.pending-*` and cannot be selected by the importers. | |
| 17 | ||
| 18 | The October 4 installer export restored and verified nine databases and eleven roles from each cluster. Its complete archive is saved on the Mac at `/Users/clo/Library/Application Support/Zenith Recovery/postgres-backups/20261004T230507Z-e75419.tar`; every archived file hash and both physical cluster contents were checked against the server manifest. The production handoff is: | |
| 19 | ||
| 20 | ```sh | |
| 21 | export STUDIO_LEGACY_HANDOFF=/mnt/storage1/apps/studio-handoff/20261004T230507Z-e75419 | |
| 22 | ``` | |
| 23 | ||
| 24 | After NixOS has imported and unlocked the apps dataset, set `STUDIO_DEPLOY_HOST`, `STUDIO_DEPLOY_PORT`, and `STUDIO_LEGACY_HANDOFF` to that printed directory. The importers require the mounted pool's encrypted apps dataset at `/mnt/storage1/apps` and a stopped destination job. HedgeDoc and evil.inc Forgejo verify dump hashes and source table counts before restoring their PostgreSQL databases. Shale, Jellyfin, Navidrome, PDS, qBittorrent, Sonarr, Radarr, Jackett, and YouTube Triage can copy their retained app data without the old Docker daemon. Each importer prints its pre-import backup or ZFS snapshot. | |
| 8 | 25 | |
| 9 | 26 | `import-legacy-secrets.sh` reads the retained old `.env` from the new host when `STUDIO_LEGACY_HANDOFF` is set. It sends only the requested keys into Snow Globe's Nomad variables; it does not print their values. The handoff directory must exist on the target. Without that variable, the script continues reading from the old Zenith host for a two-host migration. |
| 10 | 27 |
tools/media-cutover.md+171-24| ... | ... | @@ -1,6 +1,6 @@ |
| 1 | 1 | # Media dataset cutover |
| 2 | 2 | |
| 3 | Zenith currently mounts `storage1/media` at `/mnt/storage1/media`. It is one 3.97 TiB ZFS filesystem containing `jellyfin`, `music`, `music_intake`, and `torrent`; `jellyfin` and `torrent` report the same filesystem device. The dataset has one snapshot with no clones. The VM reads it through a read-only mount at `/srv/clover/Media` without copying its contents. | |
| 3 | Before cutover, TrueNAS mounted `storage1/media` at `/mnt/storage1/media`. It is one 3.97 TiB ZFS filesystem containing `jellyfin`, `music`, `music_intake`, and `torrent`; `jellyfin` and `torrent` report the same filesystem device. The dataset had one snapshot with no clones at inspection. The VM reads it through a read-only mount at `/srv/clover/Media` without copying its contents. | |
| 4 | 4 | |
| 5 | 5 | The Mac mounted the VM's `media` SMB share over Tailscale, listed the source folders, and received a write error for a test file. The `clover` share is writable in the VM; both shares use the same `clo` account. |
| 6 | 6 | |
| ... | ... | @@ -8,45 +8,192 @@ Zenith's NFSv4 ACLs need a separate [permission cutover](storage-acl-cutover.md) |
| 8 | 8 | |
| 9 | 9 | The new layout needs that same dataset at `/srv/clover/Media`. A disposable VM test moved a ZFS filesystem beneath a different parent with `zfs rename`, changed its mountpoint, and confirmed that its snapshot, inode, and hardlinks survived. A second test renamed an independent encrypted root beneath another encrypted root; it remained its own encryption root. These test the ZFS operations, not the live Zenith cutover. |
| 10 | 10 | |
| 11 | At cutover, stop the old media writers and share services, take a named ZFS snapshot, then move the existing filesystems after their mountpoints are no longer busy: | |
| 11 | Run the layout commands as root in the live installer, one step at a time. Stop on any error. The old writers and share services must be stopped; booting the installer leaves the old TrueNAS services stopped. | |
| 12 | ||
| 13 | ### Reopen the inspected pool for writes | |
| 14 | ||
| 15 | The recovery inspection imported `storage1` read-only with an alternate root under `/run/infra2-inspection`. Export releases that import; it does not delete the pool. Reimport the verified data pool by its numeric ID, without an alternate root. `-N` leaves every dataset unmounted, `cachefile=none` avoids recording this temporary import, and `-f` permits the installer to take over the pool last used by TrueNAS. This ID identifies the four-disk data pool, not the NVMe's `boot-pool`. | |
| 16 | ||
| 17 | ```sh | |
| 18 | zpool export storage1 | |
| 19 | zpool import -N -f -d /dev/disk/by-id -o cachefile=none 8944904502819631621 | |
| 20 | zpool get readonly,altroot storage1 | |
| 21 | ``` | |
| 22 | ||
| 23 | The last command must show `readonly=off` and `altroot=-` before continuing. Leave `boot-pool` alone. | |
| 24 | ||
| 25 | ### Record identity and take a snapshot | |
| 26 | ||
| 27 | Keep this shell open: `media_guid` records the dataset's identity for the check after renaming. The recursive snapshot captures all existing datasets' file state before cutover. It does not undo dataset names or mountpoint properties. | |
| 28 | ||
| 29 | ```sh | |
| 30 | media_guid=$(zfs get -H -o value guid storage1/media) | |
| 31 | zfs snapshot -r storage1@before-infra2-layout | |
| 32 | ``` | |
| 33 | ||
| 34 | ### Rename the pool to globe | |
| 35 | ||
| 36 | The top-level dataset has the pool's name. Rename it by exporting and importing the pool with a new name, not with `zfs rename`. These commands also work midway through the mountpoint changes below, provided the datasets remain unmounted. | |
| 37 | ||
| 38 | ```sh | |
| 39 | zpool export storage1 | |
| 40 | zpool import -N -d /dev/disk/by-id -o cachefile=none 8944904502819631621 globe | |
| 41 | zpool get guid,readonly,altroot globe | |
| 42 | zfs list -r -o name,mountpoint,mounted globe | |
| 43 | ``` | |
| 44 | ||
| 45 | Expect the same pool GUID `8944904502819631621`, `readonly=off`, and `altroot=-`. Every dataset and snapshot prefix becomes `globe/`; their GUIDs, encryption keys, and snapshots are retained. The explicit `/srv` mountpoint and pinned `/mnt/storage1/apps` path remain unchanged. The production configuration uses `globe` for imports and service startup checks. See [OpenZFS pool import](https://openzfs.github.io/openzfs-docs/man/master/8/zpool-import.8.html). | |
| 46 | ||
| 47 | From here onward, use `globe` in ZFS arguments. Keep `/mnt/storage1/apps` as a filesystem path: the legacy importers still read it there. Existing recovery exports label their keys with the original `storage1/...` names; those same keys now unlock the corresponding `globe/...` datasets. | |
| 48 | ||
| 49 | ### Set the layout | |
| 50 | ||
| 51 | `mountpoint` controls where a dataset appears in the filesystem. `rename` changes its name and parent in the ZFS hierarchy. The `-u` flags update metadata without mounting datasets. Encrypted datasets can be renamed while locked; these commands do not change their keys or copy their file contents. See [OpenZFS rename](https://openzfs.github.io/openzfs-docs/man/master/8/zfs-rename.8.html) and [mountpoint properties](https://openzfs.github.io/openzfs-docs/man/master/7/zfsprops.7.html). | |
| 52 | ||
| 53 | | Command | Effect | | |
| 54 | | --- | --- | | |
| 55 | | `zfs set -u mountpoint=/mnt/storage1/apps globe/apps` | Pin the retained app tree at the path used by the legacy importers, before moving the pool root. | | |
| 56 | | `zfs set -u mountpoint=/srv globe` | Move the pool's root mountpoint; default-mountpoint children such as backups, logs, and mirrors follow it. | | |
| 57 | | `zfs set -u mountpoint=/srv/clover globe/clover` | Set Clover's mount location. Its dataset name stays `globe/clover`. | | |
| 58 | | `zfs rename -u globe/media globe/clover/Media` | Move the existing Media dataset beneath Clover and capitalize its name. Its snapshots stay attached to the same dataset. | | |
| 59 | | `zfs set -u mountpoint=/srv/clover/Media globe/clover/Media` | Set Media's mount location. It remains an independent encryption root with its original key. | | |
| 60 | ||
| 61 | The destination dataset must not already exist, and the `Media` directory inside Clover must be absent or empty before mounting there. The live inspection checked both on 2026-10-04: the dataset and directory were absent. | |
| 62 | ||
| 63 | ### Verify before mounting | |
| 64 | ||
| 65 | ```sh | |
| 66 | test "$(zfs get -H -o value guid globe/clover/Media)" = "$media_guid" | |
| 67 | zfs list -o name,mountpoint,mounted globe globe/apps globe/clover globe/clover/Media | |
| 68 | zfs get encryptionroot globe/apps globe/clover globe/clover/Media | |
| 69 | zfs list -t snapshot -r -o name globe/clover/Media | |
| 70 | ``` | |
| 71 | ||
| 72 | The identity check must succeed. Expect these mountpoint properties, with `mounted=no` throughout: | |
| 73 | ||
| 74 | ```text | |
| 75 | globe /srv | |
| 76 | globe/apps /mnt/storage1/apps | |
| 77 | globe/clover /srv/clover | |
| 78 | globe/clover/Media /srv/clover/Media | |
| 79 | ``` | |
| 80 | ||
| 81 | Media's `encryptionroot` must be `globe/clover/Media`, not `globe/clover`; apps and Clover must remain their own encryption roots. Media's original snapshots should now use its new dataset prefix, alongside `@before-infra2-layout`. Loading keys and mounting the filesystems follow this metadata check; permission conversion is covered in [storage-acl-cutover.md](storage-acl-cutover.md). Preserve the old NVMe until the PostgreSQL handoff in [legacy-handoff.md](legacy-handoff.md) is verified. | |
| 82 | ||
| 83 | A disposable pool with an altroot reproduced the default child's move when the parent mountpoint changed, then kept the old child path after an explicit mountpoint was set. After loading keys and mounting, Clover and Media must resolve to distinct ZFS filesystems before starting Snow Globe; `tools/studio.py pool` enforces this. Check a known hardlinked download/library pair by device and inode, then verify Jellyfin and Navidrome library paths inside their containers. | |
| 84 | ||
| 85 | The root change also relocates other default-mountpoint children: `backup` (including the 2.07 TB `backup/sandwich`), `mirrors` (266 GB), `agent`, `homes`, `logs`, and `.ix-virt`. They remain datasets under `/srv` until deliberately renamed; the move does not copy their data. `ix-apps` has a local `/mnt/.ix-apps` mountpoint and `.system` uses `legacy`, so neither follows the root. Preserve these datasets during cutover. | |
| 86 | ||
| 87 | ### Remove retired app clones and the rehearsal store | |
| 88 | ||
| 89 | The five `apps-hourly-…-clone` entries are cloned filesystems, distinct from the original `globe/apps@hourly-…` snapshots. The rehearsal store is `globe/apps/snowglobe-rehearsal-20261002` (59.8 GiB at inspection). The live installer inspection on 2026-10-04 confirmed that all six are unmounted, and `zfs destroy -nvr` previews succeeded: each would delete only the named filesystem and its own `@before-infra2-layout` snapshot. | |
| 90 | ||
| 91 | These commands permanently remove those six retired filesystems and their snapshots. They preserve `globe/apps`, its original hourly snapshots, and the other datasets' cutover snapshots. Run one command at a time and stop on an error. To preview a command again, replace `-vr` with `-nvr`; lowercase `-r` includes the target's children and snapshots. Do not use uppercase `-R`, which also destroys dependent clones outside the target hierarchy. See [OpenZFS destroy](https://openzfs.github.io/openzfs-docs/man/master/8/zfs-destroy.8.html). | |
| 92 | ||
| 93 | ```sh | |
| 94 | zfs destroy -vr globe/apps-hourly-2026-09-06_21-00-clone | |
| 95 | zfs destroy -vr globe/apps-hourly-2026-09-08_23-00-clone | |
| 96 | zfs destroy -vr globe/apps-hourly-2026-09-11_07-00-clone | |
| 97 | zfs destroy -vr globe/apps-hourly-2026-09-11_12-00-clone | |
| 98 | zfs destroy -vr globe/apps-hourly-2026-09-12_14-00-clone | |
| 99 | zfs destroy -vr globe/apps/snowglobe-rehearsal-20261002 | |
| 100 | ``` | |
| 101 | ||
| 102 | Check the remaining app datasets and source snapshots afterward: | |
| 103 | ||
| 104 | ```sh | |
| 105 | zfs list -r -o name,used,mountpoint globe/apps | |
| 106 | zfs list -t snapshot -d 1 -o name globe/apps | |
| 107 | ``` | |
| 108 | ||
| 109 | ### Unlock and mount Clover and Media | |
| 110 | ||
| 111 | Clover and Media are independent encryption roots, so each needs its original key. The recovery export labels these `storage1/clover` and `storage1/media`, respectively. For the installer commands below, the recovered keys must be available as `/run/infra2-keys/clover.hex` and `/run/infra2-keys/media.hex`, with directory mode `0700` and file mode `0600`, owned by root. These temporary files disappear at reboot; recover them from the private archive again if needed. Keep keys out of this repository and terminal output. | |
| 112 | ||
| 113 | Run one command at a time and stop on errors. Mount the pool root if needed, then verify that `/srv` is its actual mountpoint: | |
| 114 | ||
| 115 | ```sh | |
| 116 | if [ "$(zfs get -H -o value mounted globe)" = no ]; then zfs mount globe; fi | |
| 117 | test "$(findmnt -n -o SOURCE --mountpoint /srv)" = globe | |
| 118 | ``` | |
| 119 | ||
| 120 | Load the two keys, mount Clover before its Media child, then remove the temporary key files after both mounts succeed. `-L` overrides the key source for this command without changing the datasets' persistent `keylocation=prompt`. Loading keys alone does not mount them; see [OpenZFS load-key](https://openzfs.github.io/openzfs-docs/man/master/8/zfs-load-key.8.html). Use ordinary [ZFS mounts](https://openzfs.github.io/openzfs-docs/man/master/8/zfs-mount.8.html), without overlay options; stop if the Media mountpoint is nonempty. | |
| 121 | ||
| 122 | ```sh | |
| 123 | zfs load-key -L file:///run/infra2-keys/clover.hex globe/clover | |
| 124 | zfs load-key -L file:///run/infra2-keys/media.hex globe/clover/Media | |
| 125 | zfs mount globe/clover | |
| 126 | zfs mount globe/clover/Media | |
| 127 | rm /run/infra2-keys/clover.hex /run/infra2-keys/media.hex | |
| 128 | test "$(findmnt -n -o SOURCE --mountpoint /srv/clover)" = globe/clover | |
| 129 | test "$(findmnt -n -o SOURCE --mountpoint /srv/clover/Media)" = globe/clover/Media | |
| 130 | findmnt -rn -o SOURCE,TARGET,FSTYPE --mountpoint /srv/clover/Media | |
| 131 | ``` | |
| 132 | ||
| 133 | The last command must show `globe/clover/Media /srv/clover/Media zfs`. This mounts only Clover and Media beneath the pool root. Continue to the folder renames below after these checks succeed. Root can perform these renames; service access still requires the separate ACL cutover. | |
| 134 | ||
| 135 | ### Unlock retained apps for the legacy handoff | |
| 136 | ||
| 137 | The original apps key is labeled `storage1/apps` in the recovery export. Prepare it as `/run/infra2-keys/apps.hex` with the same root-only permissions as the Clover and Media keys. Unlock and mount the retained dataset at its pinned legacy path: | |
| 12 | 138 | |
| 13 | 139 | ```sh |
| 14 | zfs snapshot storage1/media@before-studio-cutover | |
| 15 | zfs set mountpoint=/mnt/storage1/apps storage1/apps | |
| 16 | zfs set mountpoint=/srv storage1 | |
| 17 | zfs set mountpoint=/srv/clover storage1/clover | |
| 18 | zfs rename storage1/media storage1/clover/Media | |
| 19 | zfs set mountpoint=/srv/clover/Media storage1/clover/Media | |
| 20 | test "$(findmnt -n -o SOURCE --mountpoint /mnt/storage1/apps)" = storage1/apps | |
| 21 | test "$(findmnt -n -o SOURCE --mountpoint /srv)" = storage1 | |
| 22 | test "$(findmnt -n -o SOURCE --mountpoint /srv/clover)" = storage1/clover | |
| 23 | test "$(findmnt -n -o SOURCE --mountpoint /srv/clover/Media)" = storage1/clover/Media | |
| 140 | zfs load-key -L file:///run/infra2-keys/apps.hex globe/apps | |
| 141 | zfs mount globe/apps | |
| 142 | rm /run/infra2-keys/apps.hex | |
| 143 | test "$(findmnt -n -o SOURCE --mountpoint /mnt/storage1/apps)" = globe/apps | |
| 144 | findmnt -rn -o SOURCE,TARGET,FSTYPE --mountpoint /mnt/storage1/apps | |
| 24 | 145 | ``` |
| 25 | 146 | |
| 26 | The snapshot may be taken before reinstalling; run the mountpoint changes after NixOS imports the pool without TrueNAS's `/mnt` altroot. `storage1/apps` currently has a *default* mountpoint; setting it locally before changing the pool root preserves `/mnt/storage1/apps` for the import handoff. A disposable pool with an altroot reproduced the default child's move when the parent mountpoint changed, then kept the old child path after an explicit mountpoint was set. The pool keeps its `storage1` name and uses `/srv` as its root mount. Clover and Media must resolve to distinct ZFS filesystems before starting Snow Globe; `tools/studio.py pool` enforces this. This move does not copy 3.97 TiB of files. Check a known hardlinked download/library pair by device and inode after the move, then verify Jellyfin and Navidrome library paths inside their containers. | |
| 147 | Expect `globe/apps /mnt/storage1/apps zfs`. Keep this path intact for the [legacy handoff](legacy-handoff.md); mounting the dataset does not start the old apps. | |
| 148 | ||
| 149 | ### Keys after installation | |
| 150 | ||
| 151 | The recovery keys remain in the private recovery archive on the Mac. The `/run/infra2-keys` files above are temporary transfer copies; after `load-key`, ZFS holds the loaded key material in memory, so deleting those files does not relock mounted datasets. Reboot loses the loaded keys. The datasets retain `keylocation=prompt` for manual recovery; the installed target loads TPM-encrypted credentials before mounting them. Nomad's mount checks prevent it from starting against locked data. | |
| 152 | ||
| 153 | Zenith's enabled TPM 2.0 (`/dev/tpm0` and `/dev/tpmrm0`) sealed and decrypted all six encryption-root keys during installation preparation on 2026-10-04. A systemd service test successfully loaded the encrypted credentials and unlocked the previously locked logs dataset. [zenith-hardware.md](zenith-hardware.md) records their recovery locations and the installed boot service; the root names and credential paths have one home in [zenith.nix](../nixos/zenith.nix). | |
| 154 | ||
| 155 | [TPM-encrypted systemd credentials](https://github.com/systemd/systemd/blob/main/man/systemd-creds.xml) keep encrypted key blobs on the NVMe and supply decrypted credentials in RAM. These credentials bind to the original TPM and PCR 7. Secure Boot is disabled, so this supplies convenient automatic unlock and protection for detached drives, without establishing a trusted boot chain. Firmware changes affecting PCR 7 or a TPM reset can require resealing the credentials from the recovery keys. | |
| 156 | ||
| 157 | Manual unlock over root SSH remains the recovery path: recover the relevant key to a private `/run` file, use `zfs load-key -L file:///run/<key-file> <dataset>`, remove that file, then mount the dataset. Keep the off-server keys when replacing hardware; encrypted blobs from the old TPM cannot unlock on a replacement machine. | |
| 27 | 158 | |
| 28 | The root change also relocates other default-mountpoint children: `backup` (including the 2.07 TB `backup/sandwich`), `mirrors` (266 GB), `agent`, `homes`, `logs`, `.ix-virt`, and five old `apps` clones. They remain datasets under `/srv` until deliberately renamed; the move does not copy their data. `ix-apps` has a local `/mnt/.ix-apps` mountpoint and `.system` uses `legacy`, so neither follows the root. Preserve these datasets during cutover. | |
| 159 | ### Rename the media folders | |
| 160 | ||
| 161 | Clover and Media both report `casesensitivity=insensitive` on the live pool. Names differing only by capitalization resolve to the same entry. For a case-only folder rename, use an unused temporary name; ordinary `mv music Music` interprets the existing destination directory as a request to move `music` inside itself. From the Media directory: | |
| 162 | ||
| 163 | ```sh | |
| 164 | test ! -e .music-case-rename && test ! -L .music-case-rename && | |
| 165 | mv -T -- music .music-case-rename && | |
| 166 | mv -T -- .music-case-rename Music | |
| 167 | ``` | |
| 168 | ||
| 169 | Lowercase configured paths still resolve after a case-only rename; do not create a separate `music -> Music` or `seedbox -> Seedbox` alias, because those names collide on this filesystem. The distinct `torrent` alias below preserves the old host path. qBittorrent also receives a container mount at `/data/media/torrent` pointing to `Seedbox`, so its saved torrent paths remain valid. | |
| 29 | 170 | |
| 30 | 171 | Zenith's qBittorrent `BT_backup` is 21 MB. A checksum-verified copy contained 544 complete bencoded resume files. Their `save_path` values include 498 at `/data/media/torrent`, three in its subfolders, and 43 in Jellyfin folders; 543 also set `qBt-downloadPath` to `/data/media/torrent`. Keep these saved paths working by renaming the real media folder and leaving a relative compatibility symlink, after the media dataset is mounted at its new path and the old writers are stopped: |
| 31 | 172 | |
| 32 | 173 | ```sh |
| 33 | 174 | test -d /srv/clover/Media/torrent |
| 34 | test ! -e /srv/clover/Media/seedbox | |
| 35 | mv /srv/clover/Media/torrent /srv/clover/Media/seedbox | |
| 36 | ln -s seedbox /srv/clover/Media/torrent | |
| 37 | test "$(stat -c %d:%i /srv/clover/Media/seedbox)" = "$(stat -Lc %d:%i /srv/clover/Media/torrent)" | |
| 175 | test ! -e /srv/clover/Media/Seedbox | |
| 176 | mv /srv/clover/Media/torrent /srv/clover/Media/Seedbox | |
| 177 | ln -s Seedbox /srv/clover/Media/torrent | |
| 178 | test "$(stat -c %d:%i /srv/clover/Media/Seedbox)" = "$(stat -Lc %d:%i /srv/clover/Media/torrent)" | |
| 38 | 179 | ``` |
| 39 | 180 | |
| 40 | The same-dataset rename preserves existing hardlinks. New downloads use `/data/media/seedbox`; saved torrents continue to resolve through `/data/media/torrent`. Rename the existing music intake folder after stopping YouTube Triage, then keep its current container path working through a relative symlink: | |
| 181 | The same-dataset rename preserves existing hardlinks. New downloads use `/data/media/Seedbox`; saved torrents continue to resolve through `/data/media/torrent`. If `Seedbox` and the `torrent` alias already exist, skip the rename and verify their device/inode identity. Rename the existing music intake folder after stopping YouTube Triage, then keep its current container path working through a relative symlink: | |
| 41 | 182 | |
| 42 | 183 | ```sh |
| 43 | 184 | test -d /srv/clover/Media/music_intake |
| 44 | test ! -e /srv/clover/Media/music-intake | |
| 45 | mv /srv/clover/Media/music_intake /srv/clover/Media/music-intake | |
| 46 | ln -s music-intake /srv/clover/Media/music_intake | |
| 185 | test ! -e '/srv/clover/Media/Intake - Music' | |
| 186 | mv /srv/clover/Media/music_intake '/srv/clover/Media/Intake - Music' | |
| 187 | ln -s 'Intake - Music' /srv/clover/Media/music_intake | |
| 47 | 188 | ``` |
| 48 | 189 | |
| 49 | YouTube Archiver keeps its download archive JSON beside the videos in `Media/jellyfin/Independent`, not in its `/config` cache ([upstream archive behavior](https://github.com/jmbannon/ytdl-sub/wiki/5.-Optimizing-Your-First-Config)). Zenith has 12 archive files there (52,345 bytes), and all 12 are visible through the VM's read-only media mount. Preserve these files with the media dataset; the old app directory contains only a lock, cache, and empty work directory. | |
| 190 | ### Service paths for the reorganized tree | |
| 191 | ||
| 192 | The service definitions read the current capitalized host folders. Jellyfin and qBittorrent retain their saved paths inside the containers through individual bind mounts; the host needs no `jellyfin` compatibility directory. The shared folder mapping is defined once in [config/site.pkl](../config/site.pkl), under `jellyfinFolders`. Jellyfin mounts its entries beneath `/media/jellyfin`; qBittorrent mounts them beneath `/data/media/jellyfin` and mounts `Seedbox` at both `/data/media/Seedbox` and `/data/media/torrent`. | |
| 193 | ||
| 194 | Navidrome mounts `Music` at `/music`. Jellyfin and Navidrome mounts remain read-only; qBittorrent follows `site.mediaReadOnly`, so read-only rehearsals stay read-only. Keeping the recorded Jellyfin paths follows its [migration guidance](https://jellyfin.org/docs/general/administration/migrate/) and avoids rewriting its database paths or library settings. | |
| 195 | ||
| 196 | Before reorganization, YouTube Archiver kept its download archive JSON beside the videos in `Media/jellyfin/Independent`, not in its `/config` cache ([upstream archive behavior](https://github.com/jmbannon/ytdl-sub/wiki/5.-Optimizing-Your-First-Config)). The source inspection found 12 archive files there (52,345 bytes), and all 12 were visible through the VM's read-only media mount. That folder now lives at `Media/Videos/Independent`; preserve its archive files with the videos. The old app directory contained only a lock, cache, and empty work directory. | |
| 50 | 197 | |
| 51 | 198 | The complete 33 MB qBittorrent config was also copied into a disposable ZFS clone in the VM. The pinned container started with `--network none` and no media mount; its API loaded all 544 torrents. All reported `missingFiles`, as expected without `/data/media`. One resume file without `qBt-savePath` adopted the new default `/data/media/seedbox`; the other 543 retained their saved paths. The container, clone, and copied config were removed after the test. |
| 52 | 199 | |
| ... | ... | @@ -54,7 +201,7 @@ A second disposable restore copied Zenith's qBittorrent profile into the VM and |
| 54 | 201 | |
| 55 | 202 | The old resume files are owned by UID 3000, while Snow Globe assigns qBittorrent UID 3114. [import-qbittorrent.sh](import-qbittorrent.sh) copies the stopped old profile into the stopped Snow Globe service's dataset, snapshots the destination, checks the copy, changes ownership, applies the new save path, and starts the service. It refuses to run while Zenith's qBittorrent is active. In a disposable VM restore, UID 3114 loaded all 544 imported torrents without a media mount; all correctly reported `missingFiles`. With the same read-only media alias and legacy symlink as above, all 544 reached 100% progress and none remained `missingFiles`. |
| 56 | 203 | |
| 57 | The production importer now requires the real Media ZFS mount and `torrent -> seedbox` alias before it starts qBittorrent. For a same-machine OS replacement, set `STUDIO_LEGACY_HANDOFF` to the [offline handoff](legacy-handoff.md) directory; the profile copy then reads the retained old apps dataset without old Docker. | |
| 204 | The production importer requires the real Media ZFS mount and a `torrent` symlink resolving to the same directory as `Seedbox` before it starts qBittorrent. It compares device/inode identity, so `torrent -> Seedbox/` is accepted. For a same-machine OS replacement, set `STUDIO_LEGACY_HANDOFF` to the [offline handoff](legacy-handoff.md) directory; the profile copy then reads the retained old apps dataset without old Docker. | |
| 58 | 205 | |
| 59 | 206 | The PIA credentials are the two lines in `~/pia.txt` on the Mac (username, then password). With `STUDIO_DEPLOY_HOST` and `STUDIO_DEPLOY_PORT` set for production, import them through `python3 tools/deploy.py secrets qbittorrent --file ~/pia.txt --key vpn_user --key vpn_pass` before deploying qBittorrent. The CLI requires a private file, sends its contents to the target over SSH, and does not print the values. |
| 60 | 207 |
tools/personal-cutover.md created+45| ... | ... | @@ -0,0 +1,45 @@ |
| 1 | # Personal service cutover | |
| 2 | ||
| 3 | The installed host retains the original encrypted app tree at `/mnt/storage1/apps`. Clover and its Media child are already mounted at `/srv/clover` and `/srv/clover/Media`. Copy each app's state into its managed production dataset, verify it, and only then start that app. Old app trees stay untouched. Personal Forgejo repository migration and PostgreSQL restoration are separate from these file imports. | |
| 4 | ||
| 5 | ```sh | |
| 6 | export STUDIO_DEPLOY_HOST=root@10.0.0.1 | |
| 7 | export STUDIO_DEPLOY_PORT=22 | |
| 8 | export STUDIO_LEGACY_HANDOFF=/mnt/storage1/apps/studio-handoff/20261004T230507Z-e75419 | |
| 9 | ``` | |
| 10 | ||
| 11 | Provision managed service datasets and identities before import. Register a stopped production Nomad job when required: render HCL, parse with `nomad job run -output`, set the returned `Job.Stop=true`, then POST `/v1/jobs` and confirm zero allocations. Do not deploy an empty app to create its dataset. A missing job is accepted by the Shale, qBittorrent, ARR, YouTube, and personal-files importers. Jellyfin, Navidrome, and Jackett importers require `Stop=true` on an existing job. | |
| 12 | ||
| 13 | | App | Retained source → managed destination | Import and pre-start proof | | |
| 14 | | --- | --- | --- | | |
| 15 | | Shale | `apps/shale/{data,repositories_owned,repositories_mirrors}` → `/srv/prod/shale/` | `bash tools/import-shale.sh shale`; three directory checksums and SQLite integrity. Forgejo-only repositories still need import into Shale. | | |
| 16 | | Jellyfin | `apps/jellyfin/` → `/srv/prod/jellyfin/config/` | `sh tools/import-jellyfin.sh jellyfin`; excludes caches/logs/transcodes; verifies both SQLite databases and copy checksums. | | |
| 17 | | Navidrome | `apps/navidrone/` → `/srv/prod/navidrome/data/` | `sh tools/import-navidrome.sh navidrome`; SQLite integrity, checksum, and retained `snow` administrator. The old directory is spelled `navidrone`. | | |
| 18 | | qBittorrent | `apps/qbittorrent/qBittorrent/` → `/srv/prod/qbittorrent/config/qBittorrent/` | `bash tools/import-qbittorrent.sh`; verifies resume-file count and checksum, prepares config, **then starts the service**. Load PIA secrets first. | | |
| 19 | | PDS | `apps/pds/` → `/srv/prod/pds/pds/` | `ssh root@10.0.0.1 'python3 - pds' < tools/import-personal-files.py`; preserve original JWT/admin/PLC/mail secrets first. Checks every SQLite database, including actor stores, and copies actor keys/blocks. Refuses WAL files requiring a consistent SQLite backup. | | |
| 20 | | Sonarr / Radarr | `apps/<app>/` → `/srv/prod/<app>/config/` | `bash tools/import-arr.sh <app>`; consistent SQLite backup, integrity and hash; excludes logs/PIDs. Configure current internal endpoints after startup. | | |
| 21 | | Jackett | `apps/jackett/` → `/srv/prod/jackett/config/Jackett/` | `sh tools/import-jackett.sh jackett`; retains API key and indexer JSON, verifies checksums. | | |
| 22 | | YouTube state | `apps/yt-feed/` → `/srv/prod/yt-feed/data/` | `bash tools/import-yt-feed.sh yt-feed`; validates JSON and checksums; stops and restarts dashboard's worker around import. Import mail secrets before activating worker. | | |
| 23 | | Clover DB | `/srv/clover/.zfs/snapshot/<snapshot>/Documents/Config/paper clover/` → `/srv/prod/clover-source-of-truth/data/` | `STUDIO_MIGRATION_SNAPSHOT=<snapshot> bash tools/import-source-data.sh`; direct host-local tar stream, tree digest and SQLite integrity. Import original API key. | | |
| 24 | | Tailscale / DDNS / Copyparty / YouTube Archiver / PDS | Sources and mount destinations are defined in [import-personal-files.py](import-personal-files.py). | Run the host-side importer below once per app. No app starts; it checks stopped allocations, snapshots destination, copies, checksum-verifies, and assigns its UID. | | |
| 25 | ||
| 26 | ```sh | |
| 27 | ssh root@10.0.0.1 'python3 - tailscale' < tools/import-personal-files.py | |
| 28 | ssh root@10.0.0.1 'python3 - ddns-updater' < tools/import-personal-files.py | |
| 29 | ssh root@10.0.0.1 'python3 - copyparty' < tools/import-personal-files.py | |
| 30 | ssh root@10.0.0.1 'python3 - ytdl-sub' < tools/import-personal-files.py | |
| 31 | ``` | |
| 32 | ||
| 33 | Tailscale must retain `tailscaled.state` to preserve its node identity; the copy remains root-owned and mode `0600`. The PDS importer copies every actor store, not only its three root SQLite files; `import-pds.sh` excludes nested SQLite files and is unsuitable for this offline cutover. DDNS retains update history; its provider configuration comes from imported Cloudflare secrets and the service's generated `CONFIG`. Copyparty's active state is the **nested** old `copyparty/copyparty/` directory, including `shares.db`, sessions, IdP database, and salts; it stays nested under the new `/cfg/copyparty`. The [container sets `XDG_CONFIG_HOME=/cfg`](https://github.com/9001/copyparty/blob/hovudstraum/scripts/docker/Dockerfile.ac), and Copyparty appends its app directory. The outer legacy policy config is not copied: `prepare.py` generates the current policy. Its share database contained 17 shares and three selected-file records at the October 5 UTC inspection. Per-volume histories already on Clover/Media stay on those datasets. | |
| 34 | ||
| 35 | YouTube Archiver's 12 retained download-archive files stay beside videos in `Media/Videos/Independent`; cache and stale locks in its old app directory are disposable. Its retained subscriptions specify `/media/jellyfin/Independent`, so the service binds that container path to the reorganized folder. The upscaler reads `/media/Videos/Independent` directly. Dashboard's worker reads the actual `Indie Shows`, `Videos/Independent`, and `Intake - Music` folders; it needs no media symlinks. | |
| 36 | ||
| 37 | Samba serves existing Clover/Media directly and needs only its account secret. Intake likewise uses `/srv/clover/Documents/Intake` directly and needs the retained API key. VictoriaMetrics, VictoriaLogs, and VictoriaTraces had no legacy directories at their personal app paths; start new stores. OpenSpeedTest, FlareSolverr, and Forward Auth have no retained application file state. Keycloak and Dawarich require their personal PostgreSQL restores before activation; Dawarich's file-volume migration belongs with its database importer. | |
| 38 | ||
| 39 | Activate telemetry and restored PostgreSQL first, then restored Keycloak and Forward Auth. Start restored Shale and complete Git repository migration. After personal files/secrets are verified, enable Copyparty, Samba, PDS, Navidrome, Jellyfin, Intake, and Clover DB. Bring up Jackett and qBittorrent before Sonarr/Radarr. Start restored YouTube state, dashboard worker, and Archiver after checking final writable media paths. DDNS starts after wildcard routing is ready; preserved Tailscale state can start once its own copy passes. | |
| 40 | ||
| 41 | Do not reuse preview imports for production: those branches can start jobs, use old `storage1` snapshot names, or intentionally disable writers. qBittorrent's production importer starts its job at the end; YouTube's importer restarts dashboard and may resume retained queued work immediately. All other production file importers listed above leave the imported app stopped. | |
| 42 | ||
| 43 | October 5 UTC execution evidence is stored root-only under `/var/lib/studio/migrations/personal-cutover-20261004`; the original app snapshot is `globe/apps@personal-cutover-20261004`. The manual no-start qBittorrent copy retained 550 saved torrents, YouTube retained six valid JSON state files, Copyparty retained 17 shares, PDS retained all four SQLite databases, and pgAdmin passed SQLite integrity. Dawarich retained its storage file, empty watched-import directory, and 95,396-byte Redis dump. YouTube state needs explicit access ACL masks (`m::rwX`) and default masks (`d:m::rwx`) so dashboard and upscaler can both write. | |
| 44 | ||
| 45 | Before ARR activation, the copied databases received a separate `before-arr-path-cutover-*` snapshot. Sonarr translated two root folders and 89 series paths; Radarr translated one root folder and 56 movie paths from `/data/media/jellyfin/<folder>` to `/data/media/<folder>`. All other application table values matched their pre-change hashes. Seven download-path mappings per app are derived from qBittorrent and ARR volume definitions, including `torrent` → `Seedbox` and the saved Jellyfin aliases; [configure-arr.py](configure-arr.py) updates their host alongside the download client. [Servarr maps paths only for the matching download-client host](https://github.com/Servarr/Wiki/blob/master/sonarr/settings.md). All mapped root destinations passed service UID/GID 3000 read/search checks. Three Sonarr series directories were already absent in the pre-layout Media snapshot; the other 86 and all 56 Radarr movie directories resolve. Private proofs are `<app>-path-cutover.json`, `<app>-mapped-directory-proof.json`, and `sonarr-missing-directory-origin.json` in the evidence directory above. |
tools/release.py+14-1| ... | ... | @@ -20,10 +20,23 @@ RELEASE_ID = re.compile(r"[0-9a-f]{16}\Z") |
| 20 | 20 | STAGE_ID = re.compile(r"[a-z][a-z0-9-]*\Z") |
| 21 | 21 | |
| 22 | 22 | |
| 23 | def excluded_services(root): | |
| 24 | path = root / "config/excluded-services.json" | |
| 25 | return set(json.loads(path.read_text())) if path.exists() else set() | |
| 26 | ||
| 27 | ||
| 23 | 28 | def files(root): |
| 29 | excluded = excluded_services(root) | |
| 24 | 30 | for name in SOURCES: |
| 25 | 31 | source = root / name |
| 26 | paths = source.rglob("*") if source.is_dir() else [source] | |
| 32 | if name == "service" and source.is_dir(): | |
| 33 | paths = [] | |
| 34 | for directory, directories, filenames in os.walk(source): | |
| 35 | directories[:] = [entry for entry in directories if entry not in excluded] | |
| 36 | paths.extend(Path(directory) / entry for entry in [*directories, *filenames] | |
| 37 | if not (entry.endswith(".pkl") and entry[:-4] in excluded)) | |
| 38 | else: | |
| 39 | paths = source.rglob("*") if source.is_dir() else [source] | |
| 27 | 40 | for path in sorted(paths): |
| 28 | 41 | relative = path.relative_to(root) |
| 29 | 42 | if relative.parts[:2] in ( |
tools/router.py+1-1| ... | ... | @@ -231,7 +231,7 @@ def render(token): |
| 231 | 231 | if traces: |
| 232 | 232 | lines += ["http://127.0.0.1:10428 {", " bind 127.0.0.1", |
| 233 | 233 | *proxy(" ".join(sorted(traces["upstreams"])), " "), "}"] |
| 234 | dashboard_host = "globe." + os.environ["STUDIO_DOMAIN"] | |
| 234 | dashboard_host = "snowglobe." + os.environ["STUDIO_DOMAIN"] | |
| 235 | 235 | dashboard_port = int(os.environ["STUDIO_DASHBOARD_PORT"]) |
| 236 | 236 | if not HOST.fullmatch(dashboard_host) or not 1 <= dashboard_port <= 65535: |
| 237 | 237 | raise ValueError("invalid dashboard route") |
tools/shale-migration.md+36| ... | ... | @@ -1,5 +1,41 @@ |
| 1 | 1 | # Shale migration |
| 2 | 2 | |
| 3 | Shale is the sole Git server on Zenith. Personal Forgejo is retired. Retained Forgejo repositories and database exports remain migration sources; copying the existing Shale directory does not import repositories that only existed in Forgejo. The legacy multi-server SSH router is not part of the target setup. | |
| 4 | ||
| 5 | The 2026-10-04 personal inventory found 12 Shale owned repositories, no Shale mirrors, and 19 retained Forgejo bare repositories: 18 under `clo` and one under `nix`. Forgejo's retained LFS directory contains no files. Complete both imports before Shale's first production activation. | |
| 6 | ||
| 7 | The production data migration on 2026-10-04 completed before activating either production Shale or Postgres. The retained Shale copy passed checksum and SQLite checks, then all 19 Forgejo histories imported with verified object-file hashes and ref mappings. The result contains 21 repository records. All 11 private Forgejo sources retain private access; existing Shale identities, tokens, and sessions survive. The original session secret is preserved in the Nomad variable and its private recovery copy. Production import evidence is `/var/lib/studio/shale-forgejo-import-ldfl9zuy/repositories.json`; the earlier service dataset state is `globe/prod/shale@before-shale-import-1791160403-42496`. | |
| 8 | ||
| 9 | After copying the retained Shale state into `/srv/prod/shale`, [import-forgejo-to-shale.py](import-forgejo-to-shale.py) imports the additional histories while Shale stays stopped. Its metadata JSON comes only from the separately restored **personal** Forgejo PostgreSQL database: | |
| 10 | ||
| 11 | ```sql | |
| 12 | SELECT coalesce(json_agg(x), '[]') FROM ( | |
| 13 | SELECT r.id, r.owner_id, u.lower_name AS owner_name, | |
| 14 | r.lower_name, r.name, r.is_private, r.default_branch, | |
| 15 | r.is_empty, r.is_mirror | |
| 16 | FROM repository r JOIN "user" u ON u.id = r.owner_id | |
| 17 | ORDER BY u.lower_name, r.lower_name | |
| 18 | ) x; | |
| 19 | ``` | |
| 20 | ||
| 21 | ```sh | |
| 22 | python3 tools/import-forgejo-to-shale.py \ | |
| 23 | --target /srv/prod/shale \ | |
| 24 | --metadata /run/personal-forgejo-repositories.json | |
| 25 | ``` | |
| 26 | ||
| 27 | The importer preserves existing Shale identities and repository records. Clover's repositories retain their names; the other namespace becomes `nix/config`. It copies and hashes Git objects without copying Forgejo hooks or configuration. Missing branches, tags, pull refs, and notes retain their names. A divergent branch or tag is retained under `refs/heads/forgejo/<owner>/...` or `refs/tags/forgejo/<owner>/...`; other conflicting refs use `refs/forgejo/<owner>/...`. Existing Shale refs retain their values, including newer histories already migrated from old mirrors. Former Forgejo mirrors become owned histories, with their original mirror status recorded in the private evidence. That directory contains every original ref, its destination, all copied object hashes, and the previous SQLite database. | |
| 28 | ||
| 29 | New repository access follows verified Forgejo visibility, with pushes restricted to the owner and issue submission disabled. When private Forgejo history joins an existing repository, public or unlisted permissions are tightened to private **before objects are copied**; existing `off` permissions remain off. An import failure must leave Shale stopped. Restoring only the previous database can re-expose imported private objects through its old public permissions, so recovery must preserve the tightened access or restore the complete service dataset. | |
| 30 | ||
| 31 | The existing `snow` account's original OIDC provider and subject must survive the cutover. Keep the canonical `auth.paperclover.net/realms/master` issuer. Shale caches repository metadata and access at startup, so finish SQLite changes before starting the application. An isolated pinned-image test proved direct registration of a new repository: private anonymous web access returned 404 and Git discovery returned 401; after public permissions and a restart, its page and Git advertisement returned 200 with the original branch object ID. The test used copied data, `--network none`, and no published ports. | |
| 32 | ||
| 33 | A full-data test imported all 19 retained Forgejo histories into a disposable copy of the existing Shale state, yielding 21 repository records. Its conservative fixture metadata marked every incoming repository private. Every copied object file and every source ref passed verification. The pinned application then denied anonymous access to a new private repository and a formerly public collision, while preserving an unaffected public repository. On that isolated copy, public test permissions proved HTTP pages and Git advertisements for `bgds`, `home-infra`, and `nix/config`; `home-infra` advertised the original Forgejo `main` and `vllm`. An actual smart HTTP fetch returned a 53-object pack with a verified pack checksum. Production visibility must come from the restored Forgejo metadata, not this test fixture. | |
| 34 | ||
| 35 | After migration and authenticated browser checks, push the new infra `main` to `https://shale.paperclover.net/home-infra.git` with a Shale personal access token. The existing `home-infra` record and UUID are retained; the imported Forgejo branches remain available alongside the tested final `main`. | |
| 36 | ||
| 37 | The October 4 transport check inspected the image pinned in [service.pkl](../service/shale/service.pkl) in disposable containers without mounting real app data. Its embedded Git endpoint and account settings use HTTP and personal access tokens; no SSH listener, authorized-key interface, or forced-command handler was found. The [official installation](https://astheno.software/shale/installation/) and [configuration reference](https://astheno.software/shale/reference/environment/) also expose HTTP serving and OAuth login without SSH configuration. A `git` account must either use a Shale-aware SSH bridge or await native SSH support. Direct filesystem Git commands would bypass Shale's authorization. | |
| 38 | ||
| 3 | 39 | Zenith's Shale app directory contains a small SQLite database and 419 MB of owned repositories. `bash tools/import-shale.sh shale-preview-4eea0e3b` copied `data`, `repositories_owned`, and `repositories_mirrors` opaquely from the read-only `storage1/apps@hourly-2026-09-26_05-00` snapshot. It verified checksums and SQLite integrity, then restarted the preview. Both sides had 11 top-level owned repository directories; the preview had one healthy Nomad allocation and returned HTTPS 200. Repository contents were not inspected. |
| 4 | 40 | |
| 5 | 41 | For the production copy, stop Zenith's Shale container and the Snow Globe Shale job, set `STUDIO_DEPLOY_HOST` and `STUDIO_DEPLOY_PORT` for the new host, then run `bash tools/import-shale.sh shale`. The importer checks both jobs remain stopped, snapshots the destination dataset, verifies all three copied directories, and leaves Snow Globe stopped. Start the new job after the copy, check the SQLite state and a known login through the new Keycloak client, then switch the public route. The destination snapshot printed by the importer remains available for recovery. |
tools/studio.py+14-5| ... | ... | @@ -15,6 +15,7 @@ import tempfile |
| 15 | 15 | import time |
| 16 | 16 | |
| 17 | 17 | from data import cloned_postgres, dataset_for |
| 18 | from release import excluded_services | |
| 18 | 19 | |
| 19 | 20 | |
| 20 | 21 | REPO = Path(__file__).resolve().parent.parent |
| ... | ... | @@ -30,12 +31,18 @@ def command(*args, capture=False, **kwargs): |
| 30 | 31 | |
| 31 | 32 | |
| 32 | 33 | def service_files(): |
| 34 | excluded = excluded_services(REPO) | |
| 33 | 35 | files = {} |
| 34 | for path in SERVICES.glob("*/*.pkl"): | |
| 35 | name = path.parent.name if path.name == "service.pkl" else path.stem | |
| 36 | if name in files: | |
| 37 | raise ValueError(f"duplicate service definition: {name}") | |
| 38 | files[name] = path | |
| 36 | for directory in SERVICES.iterdir(): | |
| 37 | if directory.name in excluded or not directory.is_dir(): | |
| 38 | continue | |
| 39 | for path in directory.glob("*.pkl"): | |
| 40 | name = directory.name if path.name == "service.pkl" else path.stem | |
| 41 | if name in excluded: | |
| 42 | continue | |
| 43 | if name in files: | |
| 44 | raise ValueError(f"duplicate service definition: {name}") | |
| 45 | files[name] = path | |
| 39 | 46 | return files |
| 40 | 47 | |
| 41 | 48 | |
| ... | ... | @@ -918,6 +925,8 @@ def main(): |
| 918 | 925 | parser.error("allocate requires a service name") |
| 919 | 926 | if args.key and args.mode != "secrets": |
| 920 | 927 | parser.error("--key is available only for secrets") |
| 928 | if args.name and args.mode != "destroy" and args.name not in service_files(): | |
| 929 | raise ValueError(f"unknown service: {args.name}") | |
| 921 | 930 | properties = { |
| 922 | 931 | key: value |
| 923 | 932 | for key, value in (("domain", args.base_domain), ("root", args.root), ("pool", args.pool), ("mediaRoot", args.media_root)) |
tools/zenith-hardware.md+12-2| ... | ... | @@ -1,5 +1,7 @@ |
| 1 | 1 | # Zenith source host inventory |
| 2 | 2 | |
| 3 | NixOS was installed onto the NVMe on October 4, 2026 and passed two boots over SSH at the existing addresses. The final boot reported `running` with zero failed units: the data pool imported healthy, all six encrypted roots unlocked and mounted automatically through the TPM credentials, and Nomad, Caddy, SSH, the host service, and NVIDIA persistence started successfully. Original dataset GUIDs and all PostgreSQL handoff file hashes matched their pre-install records. The RTX 3090 uses driver 595.71.05; `hardware.graphics.enable` supplies the driver library link needed by its persistence daemon. The historical TrueNAS inventory below describes the earlier layout. | |
| 4 | ||
| 3 | 5 | Read-only inspection on 2026-09-27 found a BIOS-booted Ryzen 9 5950X host with a Realtek RTL8111/8168/8411 NIC (`r8169`), RTX 3090, one 500 GB WD Blue SN5000 NVMe boot disk, and four 8 TB WD80EFPX disks. The NVMe is `/dev/disk/by-id/nvme-WD_Blue_SN5000_500GB_24261Z806200`; its GPT has a 1 MB BIOS boot partition, a 512 MB EFI partition, and a TrueNAS `boot-pool` partition with the running root at `boot-pool/ROOT/25.04.2.4`. The four other disks form the healthy `storage1` RAIDZ1 pool. The VM's UEFI boot configuration does not describe this host. |
| 4 | 6 | |
| 5 | 7 | `enp4s0` has static `10.0.0.1/24` and `192.168.0.1/24` addresses, default gateway `10.0.0.2`, and resolvers `1.1.1.1` and `1.0.0.1`. The NixOS target retains these; DHCP would not preserve the NAS address. |
| ... | ... | @@ -8,12 +10,20 @@ Read-only inspection on 2026-09-27 found a BIOS-booted Ryzen 9 5950X host with a |
| 8 | 10 | |
| 9 | 11 | On 2026-09-27, Zenith ran OpenZFS 2.3.0 and reported `storage1` healthy. The pinned NixOS VM ran OpenZFS 2.4.4; its `zpool upgrade -v` listed every feature currently enabled or active on `storage1`, including encryption and block cloning. This checks feature support, not an actual import of the four-disk pool. |
| 10 | 12 | |
| 11 | TrueNAS currently snapshots Clover hourly for one week, daily for four weeks, and monthly for two years; apps hourly for one week and daily for one month. Media has no scheduled snapshot task. NixOS does not yet replace this retention policy. The pinned NixOS `services.zfs.autoSnapshot` module has global retention counts, so it cannot express these different dataset schedules. Its `services.sanoid` module supports retention per dataset; the owner is choosing the replacement policy before it is enabled. | |
| 13 | TrueNAS's previous schedule snapshotted Clover hourly for one week, daily for four weeks, and monthly for two years; apps hourly for one week and daily for one month. Media had no scheduled snapshot task. NixOS does not yet replace this retention policy. The pinned NixOS `services.zfs.autoSnapshot` module has global retention counts, so it cannot express these different dataset schedules. Its `services.sanoid` module supports retention per dataset; the owner is choosing the replacement policy before it is enabled. | |
| 14 | ||
| 15 | The October 4 installer boots in UEFI mode on the same hardware. The `zenith` target now uses systemd-boot, host ID `4fa19ccb`, the renamed `globe` pool, and `paperclover.net`. The owner completed the dataset renames in [media-cutover.md](media-cutover.md); retained apps stay at `/mnt/storage1/apps`. Encrypted `globe/prod` and `globe/staging` roots were created with POSIX ACLs for the new service volumes. Nomad's startup guard requires the real pool, Clover, Media, production, and staging mounts; deployment separately checks their ACL types. | |
| 16 | ||
| 17 | The October 4 personal ACL cutover retained `globe/clover@personal-acl-20261004` and `globe/clover/Media@personal-acl-20261004`. Snapshot clones first passed read/write tests for UIDs 3000, 3106, 3114, and 3116 with group 3000. Both original datasets now use `acltype=posix`, `aclmode=discard`, and `aclinherit=discard`; retaining the legacy `aclmode=restricted` caused `EPERM` during rehearsal. One metadata pass processed 1,448,236 Clover entries and 31,187 Media entries without changing owner UIDs: group 3000 receives `g+rwX`, directories have setgid and inheritable group ACLs, and other access is `---`. Traversal stayed on each filesystem and did not follow symlinks. `/var/lib/studio/personal-acl-verified.json` records the completed original passes. The test clones were destroyed after verification; the original snapshots remain. | |
| 12 | 18 | |
| 13 | The `zenith` NixOS target uses the observed BIOS boot mode, NVMe disk ID, ZFS host ID, `storage1` pool, and `paperclover.net` domain. The configured host ID `4fa19ccb` matches live `hostid`, and the configured NVMe by-id path resolves to the live boot disk; the four pool members have distinct partition UUIDs and no reported read, write, or checksum errors. Generate `nixos/hardware-configuration.nix` from the installer after partitioning the NVMe; its placeholder intentionally prevents building the target before the new root filesystem is known. NixOS imports `storage1` without requesting encryption credentials during boot. Set `storage1/apps` to the explicit `/mnt/storage1/apps` mountpoint before changing the pool root, as [media-cutover.md](media-cutover.md) specifies; its current default mountpoint would otherwise move with the pool. After loading the keys and mounting the renamed datasets, start Nomad; its startup check requires `storage1` at `/srv`, `storage1/clover` at `/srv/clover`, `storage1/clover/Media` at `/srv/clover/Media`, and `storage1/prod` and `storage1/staging` at their corresponding paths. Snow Globe's preflight separately checks encryption and ACL type. Keep `storage1/apps` mounted at `/mnt/storage1/apps` through the import handoff. Dataset renames, encryption keys, ACL conversion, and the OS installation are owner-run downtime steps. | |
| 19 | The ASUS firmware TPM2 successfully sealed and decrypted every ZFS encryption-root key. [zenith.nix](../nixos/zenith.nix) loads encrypted systemd credentials from `/etc/credstore.encrypted` and runs `studio-zfs-unlock` after pool import and before `zfs-mount`. Plaintext keys remain outside the Nix store and NVMe; the running service's credentials exist in RAM. The credentials are bound to this TPM and PCR 7, so firmware changes affecting Secure Boot or clearing/replacing the TPM can require manual recovery. Secure Boot is currently disabled. The original dataset keys and TrueNAS configuration remain in the private Mac recovery archive; new production/staging keys are backed up separately at `/Users/clo/Library/Application Support/Zenith Recovery/postgres-backups/new-os-dataset-keys.json`. | |
| 20 | ||
| 21 | The NVMe installation replaces the old TrueNAS boot pool with a 1 GiB FAT32 EFI partition and an ext4 root partition occupying the remaining space. The only erase target is `/dev/disk/by-id/nvme-WD_Blue_SN5000_500GB_24261Z806200`; no HDD partition is formatted. [legacy-handoff.md](legacy-handoff.md) records the verified exports and off-server backups made before replacing the OS. The generated [hardware configuration](../nixos/hardware-configuration.nix) identifies the new root and EFI filesystems. The installed source lives at `/etc/nixos`; subsequent OS changes use `nixos-rebuild switch --flake /etc/nixos#zenith`. | |
| 14 | 22 | |
| 15 | 23 | The target authorizes this Mac's existing ED25519 key for `clo` and root SSH. Its fingerprint `SHA256:52mNGHRsVFBDED9IAX5pe+LRWUefqTbxEReunq21QvU` matches the key Zenith currently accepts from this Mac. The other three keys in Zenith's `clo` authorized-keys file are not copied into the new root account. A synthetic NixOS evaluation confirmed both accounts receive exactly this key. |
| 16 | 24 | |
| 25 | The original TrueNAS ED25519, ECDSA, and RSA SSH host keys were recovered from the configuration backup, checked against their public keys, and restored into `/etc/ssh` on NixOS. ED25519 fingerprint `SHA256:P8pRWNrL6tceq9KN41Hdb0v0kmIboMf/zFBgOtHJfm0` matches this Mac's existing `git.paperclover.net` trust entry; a fresh connection with strict host checking passed after reloading SSH. The installer keys used during installation are retained privately in `/root/infra2-installer-host-keys`. [Shale migration](shale-migration.md) records the intended Git service scope, transport check, and replacement of the legacy SSH router. | |
| 26 | ||
| 17 | 27 | [Nomad stores its server state under `data_dir`](https://developer.hashicorp.com/nomad/docs/configuration), including [variables and their encrypted secret values](https://developer.hashicorp.com/nomad/docs/concepts/variables). Snow Globe stores generated service UIDs, deployment history, backup manifests/dumps, and dashboard state under `/var/lib/studio`. Both paths are on the VM's OS disk, outside its ZFS service datasets. Nomad used `66 MB` at inspection; Snow Globe's `8.7 GB` includes an `8.5 GB` VM-only `zpool.img`, while its release backups used `275 MB` across 25 runs. Reusing or replacing Zenith's NVMe would lose the control records unless they are migrated to encrypted ZFS or backed up separately. The pinned NixOS Nomad module accepts `services.nomad.settings.data_dir = "/srv/prod/nomad"` when `dropPrivileges = false`; a synthetic target evaluation passed. The corresponding Snow Globe state mount and dataset boundary await the storage policy decision before the physical cutover. |
| 18 | 28 | |
| 19 | 29 | A `nomad operator snapshot save` on the VM produced a private, compressed 143,608-byte snapshot in `/run`. Inspection reported 46 variables, 31 jobs, and 25 ACL policies. The snapshot restored into a fresh server-only Nomad agent in an isolated network namespace: its API listed all 31 jobs and 46 variables, and a hash comparison of Shale's secret variable items matched the live server without printing the values. The temporary agent, data, and snapshot were removed; the live leader and jobs stayed healthy. This proves [Nomad's server-state snapshot and restore](https://developer.hashicorp.com/nomad/commands/operator/snapshot/restore) on the pinned version, but Snow Globe's separate UID registry and backup files still need durable storage. |