diff --git a/flake.nix b/flake.nix index 71734588367ace33c4bdb1fc313064c9a5b14be7..6e7f32171a2221bc21f5e286310e77163856c0b6 100644 --- a/flake.nix +++ b/flake.nix @@ -13,9 +13,11 @@ pkgs.nodejs_24 # runtime pkgs.deno # formatter pkgs.python3 # for font subsetting + pkgs.python3Packages.pip # for font subsetting # paperclover.net pkgs.exiftool + pkgs.imagemagick pkgs.ffmpeg-full pkgs.rsync ]; diff --git a/lib/progress.ts b/lib/progress.ts index e1b686670f408c94e05ef11537151fbdcdb0d106..49f145f399969a046f5702be48995915168aa3d0 100644 --- a/lib/progress.ts +++ b/lib/progress.ts @@ -154,9 +154,9 @@ function newNode( parent?.sortChildren && parent.children.sort(parent.sortChildren); state.signal("draw"); } - const scope = log.headlessScope(() => { - console.log("TODO"); - }); + // const scope = log.headlessScope((msg) => { + // console.log("TODO"); + // }); const binding: Node = { start(text, opts) { const [child, node] = newNode(text, opts); @@ -233,7 +233,7 @@ function newNode( mutate(); } }, - log: scope, + log: log.scoped(""), end: () => void end(), [Symbol.dispose]: () => void end(), }; diff --git a/src/bin/download-db.ts b/src/bin/download-db.ts new file mode 100644 index 0000000000000000000000000000000000000000..85f0373d1cc53f8101b2271d6b013bd5958a432e --- /dev/null +++ b/src/bin/download-db.ts @@ -0,0 +1,26 @@ +export async function main() { + await rsync.spawn({ + cwd: Path.resolve(import.meta.dirname).parent!.toString(), + args: [ + "-a", + "--progress", + "clo@paperclover.net:~/paperclover.net/.clover/cache.sqlite", + ".clover/", + ], + progress: progress.start("download file cache"), + }); + await rsync.spawn({ + cwd: Path.resolve(import.meta.dirname).parent!.toString(), + args: [ + "-a", + "--progress", + "clo@paperclover.net:~/paperclover.net/.clover/questions.sqlite", + ".clover/", + ], + progress: progress.start("download questions state"), + }); +} + +import * as progress from "lib/progress.ts"; +import * as rsync from "../file-viewer/rsync.ts"; +import { Path } from "#sitegen/path"; diff --git a/src/file-viewer/backend.tsx b/src/file-viewer/backend.tsx index 1fd159b8990d635f64e710900fe20ccb9dedf5d9..324f61fd58a0f1052e6599263deee17c0fd341ad 100644 --- a/src/file-viewer/backend.tsx +++ b/src/file-viewer/backend.tsx @@ -45,7 +45,8 @@ app.get("/file/*", async (c, next) => { // The permissions system is currently binary, using the friend auth. const permissions = FilePermissions.getByPrefix(rawFilePath); if (permissions !== 0) { - const friendAuthChallenge = requireFriendAuth(c); + console.log(rawFilePath, getForFile(rawFilePath)); + const friendAuthChallenge = requireFriendAuth(c, getForFile(rawFilePath)); if (friendAuthChallenge) return friendAuthChallenge; } @@ -292,7 +293,7 @@ import { etagMatches, serveAsset } from "#sitegen/assets"; import * as view from "#sitegen/view"; import * as mime from "lib/mime.ts"; -import { requireFriendAuth } from "@/friend-auth.ts"; +import { getForFile, requireFriendAuth } from "@/friend-auth.ts"; import { MediaFile, MediaFileKind } from "@/file-viewer/models/MediaFile.ts"; import { FilePermissions } from "@/file-viewer/models/FilePermissions.ts"; import { MediaPanel } from "@/file-viewer/views/clofi.tsx"; diff --git a/src/file-viewer/bin/file-scan.ts b/src/file-viewer/bin/file-scan.ts new file mode 100644 index 0000000000000000000000000000000000000000..80dd9e8a514784983b4fd331547f6bcbc9b5e983 --- /dev/null +++ b/src/file-viewer/bin/file-scan.ts @@ -0,0 +1,1058 @@ +// The file scanner incrementally updates an sqlite database with file +// stats. Additionally, it runs "processors" on files, which precompute +// expensive data such as running `ffprobe` on all media to get the +// duration. +// +// Processors are also used to derive compressed and optimized assets, +// which is how automatic JXL / AV1 encoding is done. Derived files are +// uploaded to the clover NAS to be pulled by VPS instances for hosting. +// +// This is the third iteration of the scanner, hence its name "scan3"; +// Remember that any software you want to be maintainable and high +// quality cannot be written with AI. +const sotToken = process.env.CLOVER_SOT_KEY; + +export async function main() { + const start = performance.now(); + using _ = log.startWidget({ + format: (now) => + `paper clover's file scanner [${((now - start) / 1000).toFixed(1)}s]`, + }); + + const promises = new async.PromiseAggregator(); + + const walkQueue = new queue.PriorityQueue(10); + + const dirsNode = progress.start("Walk Tree", { estimate: 1 }); + dirsNode.passive = true; + const fileNode = progress.start("Process File", { estimate: 0 }); + fileNode.sortChildren = (a, b) => { + const ac = a.children.length > 0 ? 1 : 0; + const bc = b.children.length > 0 ? 1 : 0; + if (ac !== bc) return bc - ac; + return a.text.localeCompare(b.text); + }; + + // Read a directory or file stat and queue up changed files. + const scanDirectory = walkQueue.wrap( + async (path: Path) => { + const publicPath = toPublicPath(path); + using node = dirsNode.start(publicPath + " - stat"); + using _ = ts.defer(() => dirsNode.inc()); + + const stat = await path.stat(); + + const mediaFile = MediaFile.getByPath(publicPath); + + if (stat.isDirectory()) { + node.text = publicPath + " - reading"; + const items = (await path.readDir()) + .filter((child) => !skipBasename(child.base)) + .map((child) => (promises.push(scanDirectory(child)), child.base)); + dirsNode.estimate += items.length; + + for (const child of mediaFile?.getChildren() ?? []) { + if (items.includes(child.basename)) continue; + const recursive = child.kind === MediaFileKind.directory + ? [child, ...child.getRecursiveFileChildren()] + : [child]; + for (const deletion of recursive) deletion.delete(); + } + + return; + } + + if ( + !mediaFile || // All processes must be performed if there is no file. + // Rerun all processors if it changed + stat.size !== mediaFile.size || + stat.mtime.getTime() !== mediaFile.date.getTime() + ) { + promises.push(updateMetadata({ path, publicPath, stat, mediaFile })); + } else { + // If the scanners changed, it may mean more processes should be run. + await queueProcessors({ + path, + stat, + mediaFile, + node: fileNode.start(publicPath.slice(1)), + }); + } + }, + ); + const updateMetadata = queue.wrap( + async ({ path, publicPath, stat, mediaFile }: UpdateMetadataArgs) => { + using errorHandler = new DisposableStack(); + const label = publicPath.slice(1); + const node = errorHandler.use(fileNode.start(label)); + + await scrubLocationMetadata(path, stat, node); + + node.text = `${label} - hashing`; + const hash = await new Promise((resolve, reject) => { + const reader = fs.createReadStream(path.toString()); + reader.on("error", reject); + + const hasher = crypto.createHash("sha1").setEncoding("hex"); + hasher.on("error", reject); + hasher.on("readable", () => resolve(hasher.read())); + + reader.pipe(hasher); + }); + let date = stat.mtime; + if ( + mediaFile && + mediaFile.date.getTime() < stat.mtime.getTime() && + Date.now() - stat.mtime.getTime() < monthMilliseconds + ) { + date = mediaFile.date; + console.warn( + `M-time on ${publicPath} was likely corrupted. ${ + formatDate( + mediaFile.date, + ) + } -> ${formatDate(stat.mtime)}`, + ); + } + mediaFile = MediaFile.createFile({ + path: publicPath, + date, + hash, + size: stat.size, + duration: mediaFile?.duration ?? 0, + dimensions: mediaFile?.dimensions ?? "", + contents: mediaFile?.contents ?? "", + }); + const parent = mediaFile.getParent(); + if (parent) parent.setProcessed(0); + + node.text = `${label}`; + await queueProcessors({ + path, + stat, + mediaFile, + node, + }); + errorHandler.move(); + }, + () => ({ priority: -1 }), + ); + const queueFileProcessor = queue.wrap(async function ({ + path, + stat, + mediaFile, + processor, + index, + after, + fileNode: innerNode, + }: ProcessJob) { + using node = innerNode.start(processor.name); + await processor.run({ + path, + stat, + mediaFile, + node, + }); + fileNode.value += 1; + mediaFile.setProcessed(mediaFile.processed | (1 << (16 + index))); + for (const dependantJob of after) { + ASSERT( + dependantJob.needs > 0, + `dependantJob.needs > 0, ${dependantJob.needs}`, + ); + dependantJob.needs -= 1; + if (dependantJob.needs == 0) { + promises.push(queueFileProcessor(dependantJob)); + } + } + }, (job) => ({ cores: job.processor.cores ?? 0 })); + + function decodeProcessors(input: string) { + return input + .split(";") + .filter(Boolean) + .map(([a, b, c]) => ({ + id: a, + hash: (UNWRAP(b).charCodeAt(0) << 8) + UNWRAP(c).charCodeAt(0), + })); + } + + async function queueProcessors({ + path, + stat, + mediaFile, + node, + }: Omit) { + using errorHandler = new DisposableStack(); + errorHandler.use(node); + + node.showEstimate = false; + node.passive = true; + + const ext = mediaFile.extensionNonEmpty.toLowerCase(); + let possible = processors.filter((p) => + p.include ? p.include.has(ext) : !p.exclude?.has(ext) + ); + if (possible.length === 0) return; + ASSERT(possible.length < 16, "too many bits"); + + const hash = possible.reduce((a, b) => a ^ b.hash, 0) | 1; + ASSERT(hash <= 0xffff, `${hash.toString(16)} has no bits above 16 set`); + let processed = mediaFile.processed; + + // If the hash has changed, migrate the bitfield over. + // This also runs when the processor hash is in it's initial 0 state. + let order: ReturnType; + try { + order = decodeProcessors(mediaFile.processors); + } catch { + // this function sucks and this system sucks i hate it. + order = []; + } + if ((processed & 0xffff) !== hash) { + const previous = order.filter( + (_, i) => (processed & (1 << (16 + i))) !== 0, + ); + processed = hash; + for (const { id, hash } of previous) { + const p = processors.find((p) => p.id === id); + if (!p) continue; + const index = possible.indexOf(p); + if (index !== -1 && p.hash === hash) processed |= 1 << (16 + index); + } + mediaFile.setProcessors( + processed, + possible + .map((p) => p.id + String.fromCharCode(p.hash >> 8, p.hash & 0xff)) + .join(";"), + ); + } else { + possible = order.map(({ id }) => + UNWRAP(possible.find((p) => p.id === id)) + ); + } + + // Queue needed processors. + const jobs: ProcessJob[] = []; + for (let i = 0, { length } = possible; i < length; i += 1) { + if ((processed & (1 << (16 + i))) === 0) { + const processor = UNWRAP(possible[i]); + const job: ProcessJob = { + path, + stat, + mediaFile, + processor, + index: i, + after: [], + needs: processor.depends.length, + fileNode: node, + }; + jobs.push(job); + if (job.needs === 0) promises.push(queueFileProcessor(job)); + } + } + node.estimate = jobs.length; + fileNode.estimate += jobs.length; + for (const job of jobs) { + for (const dependId of job.processor.depends) { + const dependJob = jobs.find((j) => j.processor.id === dependId); + if (dependJob) { + dependJob.after.push(job); + } else { + ASSERT(job.needs > 0, `job.needs !== 0, ${job.needs}`); + job.needs -= 1; + if (job.needs === 0) promises.push(queueFileProcessor(job)); + } + } + } + if (node.estimate > 0) { + errorHandler.move(); + + const parent = mediaFile.getParent(); + if (parent) parent.setProcessed(0); + } + } + + // Add the root & recursively iterate! + const rootPath = Path.resolve(root); + if (!rootPath.ifExistsSync()) { + throw new Error(`file store ${rootPath} is not mounted`); + } + promises.push(scanDirectory(rootPath)); + + await promises.all(); + fileNode.end(); + dirsNode.end(); + + // Update directory metadata + using dirMetaNode = progress.start("update directory metadata"); + const dirs = MediaFile.getDirectoriesToReindex() + .sort((a, b) => b.path.length - a.path.length); + + for (const dir of dirs) { + using _ = dirMetaNode.start(dir.path); + const children = dir.getChildren(); + + // readme.txt + const readmeContent = children.find((x) => + x.basename === "readme.txt" + )?.contents ?? ""; + + // dirsort + let dirsort: string[] | null = null; + const dirSortRaw = + children.find((x) => x.basename === ".dirsort")?.contents ?? ""; + if (dirSortRaw) { + dirsort = dirSortRaw + .split("\n") + .map((x) => x.trim()) + .filter(Boolean); + } + + // Permissions + if (children.some((x) => x.basename === ".friends")) { + FilePermissions.setPermissions(dir.path, 1); + } else { + FilePermissions.setPermissions(dir.path, 0); + } + + // Recursive stats. + let totalSize = 0; + let newestDate = new Date(0); + let allHashes = ""; + for (const child of children) { + totalSize += child.size; + allHashes += child.hash; + + if (child.basename !== "/readme.txt" && child.date > newestDate) { + newestDate = child.date; + } + } + + // Project Date + const dateFile = children.find((x) => x.basename === ".date"); + if (dateFile) { + const date = new Date(dateFile.contents); + newestDate = date; + dir.setProcessors( + 0, + JSON.stringify({ hideChildrenDates: true }), + ); + } else { + dir.setProcessors(0, ""); + } + + const dirHash = crypto + .createHash("sha1") + .update(dir.path + allHashes) + .digest("hex"); + + MediaFile.markDirectoryProcessed({ + id: dir.id, + timestamp: newestDate, + contents: readmeContent, + size: totalSize, + hash: dirHash, + dirsort, + }); + } + dirMetaNode.end(); + + // Sync to remote + if ( + ((await derived.workDir.ifExistsSync()?.readDir())?.length ?? 0) > 0 + ) { + await rsync.spawn({ + args: [ + "--links", + "--recursive", + "--times", + "--partial", + "--progress", + // "--remove-source-files", + "--delay-updates", + "--exclude=tmp.*", + derived.workDir.toString() + "/", + "clo@zenith:/mnt/storage1/clover/Documents/Config/paperclover/derived/", + ], + progress: progress.start("upload derived assets"), + cwd: process.cwd(), + }); + + await fs.removeEmptyDirectories(derived.workDir.toString()); + } else { + console.info("No new derived assets"); + } + + MediaFile.db.prepare("VACUUM").run(); + MediaFile.db.reload(); + + await rsync.spawn({ + args: [ + MediaFile.db.file, + "clo@zenith:/mnt/storage1/clover/Documents/Config/paperclover/cache.sqlite", + ], + progress: progress.start("Uploading database (source of truth)"), + cwd: process.cwd(), + }); + await rsync.spawn({ + args: [ + MediaFile.db.file, + "clo@paperclover.net:paperclover.net/.clover/cache.sqlite", + ], + progress: progress.start("Uploading database (web node)"), + }); + if (sotToken) { + const res = await fetch("https://db.paperclover.net/reload", { + method: "POST", + headers: { + Authorization: sotToken, + }, + }); + if (!res.ok) { + console.warn( + `Failed to reload remote database ${res.status} ${res.statusText}`, + ); + } + } else console.warn("Missing SOT token"); + + console.info( + "Updated file viewer index in \x1b[1m" + + ((performance.now() - start) / 1000).toFixed(1) + + "s\x1b[0m", + ); + + const { duration, count } = MediaFile.db + .prepare<[], { count: number; duration: number }>(` + select + count(*) as count, + sum(duration) as duration + from media_files + `) + .getNonNull(); + const { derivedSize, derivedCount } = MediaFile.db + .prepare<[], { derivedCount: number; derivedSize: number }>(` + select + count(*) as derivedCount, + sum(size) as derivedSize + from derived_files + `) + .getNonNull(); + const canonicalSize = UNWRAP(MediaFile.getByPath("/")).size; + + console.info(); + console.info( + "Global Stats:\n" + + `- File Count: \x1b[1m${count}\x1b[0m\n` + + `- Derived Count: \x1b[1m${derivedCount}\x1b[0m\n` + + `- Media Duration: \x1b[1m${formatDurationLong(duration)}\x1b[0m\n` + + `- Canonical Size: \x1b[1m${formatSize(canonicalSize)}\x1b[0m\n` + + `- Derived Size: \x1b[1m${formatSize(derivedSize)}\x1b[0m\n`, + ); +} + +interface Process { + name: string; + cores?: number; + enable?: boolean; + include?: Set; + exclude?: Set; + depends?: string[]; + version?: number; + /* Perform an action. */ + run(args: ProcessFileArgs): Promise; +} + +const ffprobeBin = testProgram("ffprobe", "--help"); +const ffmpegBin = testProgram("ffmpeg", "--help"); + +const ffmpegOptions = ["-hide_banner", "-loglevel", "warning"]; + +// NOTE: Never re-order the processors. Add new ones at the end. +const procDuration: Process = { + name: "calculate duration", + enable: ffprobeBin !== null, + include: rules.extsDuration, + cores: 1, + async run({ path, mediaFile }) { + const { stdout } = await subprocess.exec(ffprobeBin!, [ + "-v", + "error", + "-show_entries", + "format=duration", + "-of", + "default=noprint_wrappers=1:nokey=1", + path.toString(), + ]); + + const duration = parseFloat(stdout.trim()); + if (Number.isNaN(duration)) { + throw new Error("Could not extract duration from " + stdout); + } + mediaFile.setDuration(Math.ceil(duration)); + }, +}; + +const procDimensions: Process = { + name: "calculate dimensions", + enable: ffprobeBin != null, + include: rules.extsDimensions, + cores: 1, + async run({ path, mediaFile }) { + const { ext } = path; + + let dimensions; + + if (ext === ".svg") { + // Parse out of text data + const content = await path.read("utf-8"); + const widthMatch = content.match(/width="(\d+)"/); + const heightMatch = content.match(/height="(\d+)"/); + + if (widthMatch && heightMatch) { + dimensions = `${widthMatch[1]}x${heightMatch[1]}`; + } + } else if (rules.extsImage.has(ext)) { + // Use magick to observe streams + const { stdout } = await subprocess.exec("magick", [ + "identify", + "-auto-orient", + "-format", + "%w %h", + path.toString(), + ]); + const [w, h] = stdout.split(" ").map((x) => Number(x)); + if (w && h) { + dimensions = w + "x" + h; + } + } else { + // Use ffprobe to observe streams + const { stdout } = await subprocess.exec("ffprobe", [ + "-v", + "error", + "-select_streams", + "v:0", + "-show_entries", + "stream=width,height", + "-of", + "json", + path.toString(), + ]); + const result = JSON.parse(stdout); + const stream = result.streams[0]; + if (stream) { + dimensions = UNWRAP(stream.width) + "x" + UNWRAP(stream.height); + } + } + + mediaFile.setDimensions(dimensions ?? ""); + }, +}; + +const procLoadTextContents: Process = { + name: "load text content", + include: rules.extsReadContents, + cores: 1, + version: 2, + async run({ path, mediaFile, stat }) { + if (stat.size > 1_000_000) return; + const text = await path.read("utf-8"); + mediaFile.setContents(text); + }, +}; + +const procHighlightCode: Process = { + name: "highlight source code", + include: new Set(rules.extsCode.keys()), + cores: 1, + version: 2, + async run({ path, mediaFile, stat }) { + const language = UNWRAP( + rules.extsCode.get(path.ext.toLowerCase()), + ); + // An issue is that .ts is an overloaded extension, shared between + // 'transport stream' and 'typescript'. + // + // Filter used here is: + // - more than 1mb + // - invalid UTF-8 + if (stat.size > 1_000_000) return; + let code; + const buf = await path.read(); + try { + code = new TextDecoder("utf-8", { fatal: true }).decode(buf); + } catch (error) { + mediaFile.setContents(""); + return; + } + const content = await highlight.highlightCode(code, language); + mediaFile.setContents(content); + }, +}; + +const procImageSubsets: Process = { + name: "encode image subsets", + include: rules.extsImage, + depends: [procDimensions.name], + version: 2, + async run({ path, mediaFile, node }) { + const { width, height } = UNWRAP(mediaFile.parseDimensions()); + const targetSizes = transcodeRules.imageSizes.filter((w) => w < width); + const baseStatus = node.text; + + using stack = new DisposableStack(); + for (const size of targetSizes) { + const { w, h } = resizeDimensions(width, height, size); + for (const { ext, args } of transcodeRules.imagePresets) { + node.text = baseStatus + ` (${w}x${h}, ${ext.slice(1).toUpperCase()})`; + + stack.use( + await derived.produce({ + mediaFile, + node, + cores: 2, + subkey: `${size}${ext}`, + async producer(dir) { + await subprocess.exec(ffmpegBin!, [ + ...ffmpegOptions, + "-i", + path.toString(), + "-vf", + `scale=${w}:${h}:force_original_aspect_ratio=increase,crop=${w}:${h}`, + ...args, + dir.join(`${size}${ext}`).toString(), + ]); + }, + }), + ); + } + } + + stack.move(); + }, +}; + +const videoArgsCache = new async.OnceMap(transcodeRules.getVideoInputArgs); + +const qualityMap: Record = { + u: "ultra-high", + h: "high", + m: "medium", + l: "low", + d: "data-saving", +}; +const procVideos = transcodeRules.videoFormats.map((preset) => ({ + name: `encode av1 ${UNWRAP(qualityMap[UNWRAP(preset.id[1])])}`, + include: rules.extsVideo, + enable: ffmpegBin != null, + depends: [procDuration.name, procDimensions.name], + async run({ path, mediaFile, node }) { + if ((mediaFile.duration ?? 0) < 5) return; + if (!mediaFile.dimensions) return; + + if (mediaFile.path === "/2021/top-10000-bread/output.mp4") return; + + await derived.produce({ + mediaFile, + node, + cores: 4, + subkey: `av1-${preset.id}`, + async producer(dir) { + const input = await videoArgsCache.getOrRun(path); + const args = transcodeRules.getAv1VideoArgs( + preset, + UNWRAP(input.video, "frick on " + path + JSON.stringify(input)), + dir, + ); + + await ffmpeg.spawn({ + ffmpeg: ffmpegBin!, + progress: node, + args, + cwd: dir, + }); + }, + }); + }, +})); +const procVideoAudios = transcodeRules.audioFormats.map((preset) => ({ + name: `encode opus ${UNWRAP(qualityMap[UNWRAP(preset.id)])}`, + include: rules.extsVideo, + enable: ffmpegBin != null, + depends: [procDuration.name, procDimensions.name], + version: 2, + async run({ path, mediaFile, node }) { + if ((mediaFile.duration ?? 0) < 5) return; + if (!mediaFile.dimensions) return; + if (mediaFile.path === "/2021/top-10000-bread/output.mp4") return; + await derived.produce({ + mediaFile, + node, + cores: 4, + subkey: `opus-${preset.id}`, + async producer(dir) { + const input = await videoArgsCache.getOrRun(path); + ASSERT(input.video); + if (!input.audio) return; + const args = transcodeRules.getOpusAudioArgs(preset, input.audio, dir); + + await ffmpeg.spawn({ + ffmpeg: ffmpegBin!, + progress: node, + args, + cwd: dir, + }); + }, + }); + }, +})); +const procDash: Process = { + name: `encode mpeg-dash`, + include: rules.extsVideo, + enable: ffmpegBin != null, + depends: [...procVideos, ...procVideoAudios].map((x) => x.name), + version: 4, + async run({ path, mediaFile, node }) { + if ((mediaFile.duration ?? 0) < 5) return; + if (!mediaFile.dimensions) return; + + if (mediaFile.path === "/2021/top-10000-bread/output.mp4") return; + await derived.produce({ + mediaFile, + node, + cores: 1, + subkey: `dash-av1`, + async producer(dir) { + const input = await videoArgsCache.getOrRun(path); + const videos = transcodeRules.videoFormats.map( + (preset) => + UNWRAP(dir.parent).join( + `av1-${preset.id}`, + transcodeRules.av1FileName, + ), + ); + const audios = input.audio + ? transcodeRules.audioFormats.map( + (preset) => + UNWRAP(dir.parent).join( + `opus-${preset.id}`, + transcodeRules.opusFileName, + ), + ) + : []; + const args = transcodeRules.getMpegDashArgs( + videos, + audios, + dir, + ); + + await dir.join("d").makeDir(); + + await ffmpeg.spawn({ + ffmpeg: ffmpegBin!, + progress: node, + args, + cwd: dir, + }); + }, + }); + }, +}; +const procH264Hls: Process = { + name: `encode h.264 hls`, + include: rules.extsVideo, + enable: ffmpegBin != null, + depends: [procDuration.name, procDimensions.name], + async run({ path, mediaFile, node }) { + if ((mediaFile.duration ?? 0) < 5) return; + if (!mediaFile.dimensions) return; + + await derived.produce({ + mediaFile, + node, + cores: 8, + subkey: `hls`, + async producer(dir) { + const input = await videoArgsCache.getOrRun(path); + const args = transcodeRules.getH264HlsArgs(input, dir); + + await ffmpeg.spawn({ + ffmpeg: ffmpegBin!, + progress: node, + args, + cwd: dir, + }); + }, + }); + }, +}; + +const procCompression = [ + { name: "gzip", fn: () => zlib.createGzip({ level: 9 }) }, + { name: "zstd", fn: () => zlib.createZstdCompress() }, +].map( + ({ name, fn }) => ({ + name: `compress ${name}`, + exclude: rules.extsPreCompressed, + async run({ path, mediaFile, node }) { + await derived.produce({ + mediaFile, + node, + cores: 1, + subkey: name, + async producer(dir) { + await stream.promises.pipeline( + fs.createReadStream(path.toString()), + fn(), + fs.createWriteStream(dir.join(name).toString()), + ); + }, + }); + }, + }), +); + +const processors = [ + procDimensions, + procDuration, + procLoadTextContents, + procHighlightCode, + procImageSubsets, + ...procVideos, + ...procVideoAudios, + procDash, + ...procCompression, + procH264Hls, +].map((process, id, all) => { + const strIndex = (id: number) => String.fromCharCode("a".charCodeAt(0) + id); + return { + ...(process as Process), + id: strIndex(id), + // Create a unique key. + hash: new Uint16Array( + crypto + .createHash("sha1") + .update( + process.run.toString() + + (process.version ? String(process.version) : ""), + ) + .digest().buffer, + ).reduce((a, b) => a ^ b), + depends: (process.depends ?? []).map((depend) => { + const index = all.findIndex((p) => p.name === depend); + if (index === -1) throw new Error(`Cannot find depend '${depend}'`); + if (index === id) throw new Error(`Cannot depend on self: '${depend}'`); + return strIndex(index); + }), + }; +}); + +function resizeDimensions(w: number, h: number, desiredWidth: number) { + ASSERT(desiredWidth < w, `${desiredWidth} < ${w}`); + return { w: desiredWidth, h: Math.floor((h / w) * desiredWidth) }; +} + +interface UpdateMetadataArgs { + path: Path; + publicPath: string; + stat: fs.Stats; + mediaFile: MediaFile | null; +} + +interface ProcessFileArgs { + path: Path; + stat: fs.Stats; + mediaFile: MediaFile; + node: progress.Node; +} + +interface ProcessJob { + path: Path; + stat: fs.Stats; + mediaFile: MediaFile; + processor: (typeof processors)[0]; + index: number; + after: ProcessJob[]; + needs: number; + fileNode: progress.Node; +} + +function skipBasename(basename: string): boolean { + // dot files must be incrementally tracked + if (basename === ".dirsort") return false; + if (basename === ".friends") return false; + if (basename === ".date") return false; + + return ( + basename.startsWith(".") || + // basename.startsWith("._") || + // basename.startsWith(".tmp") || + // basename === ".DS_Store" || + basename.toLowerCase() === "thumbs.db" || + basename.toLowerCase() === "desktop.ini" + ); +} + +function toPublicPath(diskPath: Path) { + if (diskPath.toString() === root) return "/"; + return "/" + path.relative(root, diskPath.toString()).replaceAll("\\", "/"); +} + +function testProgram(name: string, helpArgument: string) { + try { + child_process.spawnSync(name, [helpArgument]); + return name; + } catch (err) { + console.warn(`Missing or corrupt executable '${name}'`); + } + return null; +} + +// Helper function to check and remove location metadata +async function scrubLocationMetadata( + path: Path, + stats: fs.Stats, + progress: progress.Ref, +): Promise { + using _ = progress.start("scrub exif metadata"); + const ext = path.ext.toLowerCase(); + if (!rules.extsScrubExif.has(ext)) return false; + + let hasLocation = false; + let args: string[] = []; + + // Check for location metadata based on file type + const tempOutput = UNWRAP(path.parent).join(`.tmp.${path.base}`); + switch (ext) { + case ".jpg": + case ".jpeg": + case ".png": + const { stdout: gpsCheck } = await subprocess.exec("exiftool", [ + "-gps:all", + path.toString(), + ]); + hasLocation = gpsCheck.trim().length > 0; + args = ["-gps:all=", path.toString(), "-o", tempOutput.toString()]; + break; + case ".mov": + case ".mp4": + const { stdout: videoCheck } = await subprocess.exec("exiftool", [ + "-ee", + "-G3", + "-s", + path.toString(), + ]); + hasLocation = videoCheck.includes("GPS") || + videoCheck.includes("Location"); + args = [ + "-gps:all=", + "-xmp:all=", + path.toString(), + "-o", + tempOutput.toString(), + ]; + break; + case ".m4a": + const { stdout: m4aCheck } = await subprocess.exec("exiftool", [ + "-ee", + "-G3", + "-s", + path.toString(), + ]); + hasLocation = m4aCheck.includes("GPS") || + m4aCheck.includes("Location") || + m4aCheck.includes("Filename") || + m4aCheck.includes("Title"); + + if (hasLocation) { + args = [ + "-gps:all=", + "-location:all=", + "-filename:all=", + "-title=", + "-m4a:all=", + path.toString(), + "-o", + tempOutput.toString(), + ]; + } + break; + } + + const accessTime = stats.atime; + const modTime = stats.mtime; + + let backup: Path | null = null; + try { + if (hasLocation) { + // Prepare a backup + const tmp = UNWRAP(path.parent).join(`.tmp.backup.${path.base}`); + await fsp.copyFile(path.toString(), tmp.toString()); + await fsp.utimes(tmp.toString(), accessTime, modTime); + backup = tmp; + + // Remove metadata + await subprocess.exec("exiftool", args); + if (!tempOutput.ifExistsSync()) { + throw new Error(`Failed to create output file: ${tempOutput}`); + } + + // Restore original timestamps + await fsp.rename(tempOutput.toString(), path.toString()); + await fsp.utimes(path.toString(), accessTime, modTime); + + // Backup is no longer needed + await fsp.unlink(backup.toString()); + + console.info( + `Scrubbed location metadata in ${path.relative(Path.resolve(root))}`, + ); + return true; + } + } catch (error) { + if (backup) { + await fsp.rename(backup.toString(), path.toString()); + } + if (fs.existsSync(tempOutput.toString())) { + await fsp.unlink(tempOutput.toString()); + } + throw error; + } + + return false; +} + +const monthMilliseconds = 30 * 24 * 60 * 60 * 1000; + +import * as async from "lib/async.ts"; +import * as fs from "#sitegen/fs"; +import * as subprocess from "lib/subprocess.ts"; +import { Path } from "#sitegen/path"; +import * as queue from "../../../lib/queue.ts"; +import * as log from "../../../lib/log.ts"; +import * as progress from "lib/progress.ts"; +import * as ts from "../../../lib/ts.ts"; + +import * as child_process from "node:child_process"; +import * as crypto from "node:crypto"; +import * as fsp from "node:fs/promises"; +import * as path from "node:path"; +import * as stream from "node:stream"; +import * as zlib from "node:zlib"; + +import { MediaFile, MediaFileKind } from "@/file-viewer/models/MediaFile.ts"; +import { FilePermissions } from "@/file-viewer/models/FilePermissions.ts"; +import { + formatDate, + formatDurationLong, + formatSize, +} from "@/file-viewer/format.ts"; +import * as rules from "@/file-viewer/rules.ts"; +import * as highlight from "@/file-viewer/highlight.ts"; +import * as ffmpeg from "lib/subprocess/ffmpeg.ts"; +import * as rsync from "@/file-viewer/rsync.ts"; +import * as derived from "@/file-viewer/models/derived.ts"; +import * as transcodeRules from "@/file-viewer/transcode-rules.ts"; + +import { rawFileRoot as root } from "../paths.ts"; +import { ASSERT, UNWRAP } from "lib/assert.ts"; diff --git a/src/file-viewer/bin/file-trim.ts b/src/file-viewer/bin/file-trim.ts new file mode 100644 index 0000000000000000000000000000000000000000..4c31f666742571b24a43c3fe4f0ce5143f0e9b25 --- /dev/null +++ b/src/file-viewer/bin/file-trim.ts @@ -0,0 +1,18 @@ +export async function main() { + const start = performance.now(); + using _ = log.startWidget({ + format: (now) => + `paper clover's file scanner [${((now - start) / 1000).toFixed(1)}s]`, + }); + + const orphaned = derived.findOrphanedRoots(); + for (const root of orphaned) { + console.info("delete " + root.key); + derived.deleteRoot(root); + } + + // TODO: delete unreferenced files +} + +import * as log from "lib/log.ts"; +import * as derived from "../models/derived.ts"; diff --git a/src/file-viewer/bin/scan3.ts b/src/file-viewer/bin/scan3.ts deleted file mode 100644 index 2269fab2d30e54e9b895c4a260531f744094ef29..0000000000000000000000000000000000000000 --- a/src/file-viewer/bin/scan3.ts +++ /dev/null @@ -1,1043 +0,0 @@ -// The file scanner incrementally updates an sqlite database with file -// stats. Additionally, it runs "processors" on files, which precompute -// expensive data such as running `ffprobe` on all media to get the -// duration. -// -// Processors are also used to derive compressed and optimized assets, -// which is how automatic JXL / AV1 encoding is done. Derived files are -// uploaded to the clover NAS to be pulled by VPS instances for hosting. -// -// This is the third iteration of the scanner, hence its name "scan3"; -// Remember that any software you want to be maintainable and high -// quality cannot be written with AI. -const sotToken = process.env.CLOVER_SOT_KEY; - -export async function main() { - const start = performance.now(); - using _ = log.startWidget({ - format: (now) => - `paper clover's scan3 [${((now - start) / 1000).toFixed(1)}s]`, - }); - - const promises = new async.PromiseAggregator(); - - const walkQueue = new queue.PriorityQueue(10); - - const dirsNode = progress.start("Walk Tree", { estimate: 1 }); - dirsNode.passive = true; - const fileNode = progress.start("Process File", { estimate: 0 }); - fileNode.sortChildren = (a, b) => { - const ac = a.children.length > 0 ? 1 : 0; - const bc = b.children.length > 0 ? 1 : 0; - if (ac !== bc) return bc - ac; - return a.text.localeCompare(b.text); - }; - - // Read a directory or file stat and queue up changed files. - const scanDirectory = walkQueue.wrap( - async (path: Path) => { - const publicPath = toPublicPath(path); - using node = dirsNode.start(publicPath + " - stat"); - using _ = ts.defer(() => dirsNode.inc()); - - const stat = await path.stat(); - - const mediaFile = MediaFile.getByPath(publicPath); - - if (stat.isDirectory()) { - node.text = publicPath + " - reading"; - const items = (await path.readDir()) - .filter((child) => !skipBasename(child.base)) - .map((child) => (promises.push(scanDirectory(child)), child.base)); - dirsNode.estimate += items.length; - - for (const child of mediaFile?.getChildren() ?? []) { - if (items.includes(child.basename)) continue; - const recursive = child.kind === MediaFileKind.directory - ? [child, ...child.getRecursiveFileChildren()] - : [child]; - for (const deletion of recursive) deletion.delete(); - } - - return; - } - - if ( - !mediaFile || // All processes must be performed if there is no file. - // Rerun all processors if it changed - stat.size !== mediaFile.size || - stat.mtime.getTime() !== mediaFile.date.getTime() - ) { - promises.push(updateMetadata({ path, publicPath, stat, mediaFile })); - } else { - // If the scanners changed, it may mean more processes should be run. - await queueProcessors({ - path, - stat, - mediaFile, - node: fileNode.start(publicPath.slice(1)), - }); - } - }, - ); - const updateMetadata = queue.wrap( - async ({ path, publicPath, stat, mediaFile }: UpdateMetadataArgs) => { - using errorHandler = new DisposableStack(); - const label = publicPath.slice(1); - const node = errorHandler.use(fileNode.start(label)); - - await scrubLocationMetadata(path, stat, node); - - node.text = `${label} - hashing`; - const hash = await new Promise((resolve, reject) => { - const reader = fs.createReadStream(path.toString()); - reader.on("error", reject); - - const hasher = crypto.createHash("sha1").setEncoding("hex"); - hasher.on("error", reject); - hasher.on("readable", () => resolve(hasher.read())); - - reader.pipe(hasher); - }); - let date = stat.mtime; - if ( - mediaFile && - mediaFile.date.getTime() < stat.mtime.getTime() && - Date.now() - stat.mtime.getTime() < monthMilliseconds - ) { - date = mediaFile.date; - console.warn( - `M-time on ${publicPath} was likely corrupted. ${ - formatDate( - mediaFile.date, - ) - } -> ${formatDate(stat.mtime)}`, - ); - } - mediaFile = MediaFile.createFile({ - path: publicPath, - date, - hash, - size: stat.size, - duration: mediaFile?.duration ?? 0, - dimensions: mediaFile?.dimensions ?? "", - contents: mediaFile?.contents ?? "", - }); - const parent = mediaFile.getParent(); - if (parent) parent.setProcessed(0); - - node.text = `${label}`; - await queueProcessors({ - path, - stat, - mediaFile, - node, - }); - errorHandler.move(); - }, - () => ({ priority: -1 }), - ); - const queueFileProcessor = queue.wrap(async function ({ - path, - stat, - mediaFile, - processor, - index, - after, - fileNode: innerNode, - }: ProcessJob) { - using node = innerNode.start(processor.name); - await processor.run({ - path, - stat, - mediaFile, - node, - }); - fileNode.value += 1; - mediaFile.setProcessed(mediaFile.processed | (1 << (16 + index))); - for (const dependantJob of after) { - ASSERT( - dependantJob.needs > 0, - `dependantJob.needs > 0, ${dependantJob.needs}`, - ); - dependantJob.needs -= 1; - if (dependantJob.needs == 0) { - promises.push(queueFileProcessor(dependantJob)); - } - } - }, (job) => ({ cores: job.processor.cores ?? 0 })); - - function decodeProcessors(input: string) { - return input - .split(";") - .filter(Boolean) - .map(([a, b, c]) => ({ - id: a, - hash: (UNWRAP(b).charCodeAt(0) << 8) + UNWRAP(c).charCodeAt(0), - })); - } - - async function queueProcessors({ - path, - stat, - mediaFile, - node, - }: Omit) { - using errorHandler = new DisposableStack(); - errorHandler.use(node); - - node.showEstimate = false; - node.passive = true; - - const ext = mediaFile.extensionNonEmpty.toLowerCase(); - let possible = processors.filter((p) => - p.include ? p.include.has(ext) : !p.exclude?.has(ext) - ); - if (possible.length === 0) return; - ASSERT(possible.length < 16, "too many bits"); - - const hash = possible.reduce((a, b) => a ^ b.hash, 0) | 1; - ASSERT(hash <= 0xffff, `${hash.toString(16)} has no bits above 16 set`); - let processed = mediaFile.processed; - - // If the hash has changed, migrate the bitfield over. - // This also runs when the processor hash is in it's initial 0 state. - let order: ReturnType; - try { - order = decodeProcessors(mediaFile.processors); - } catch { - // this function sucks and this system sucks i hate it. - order = []; - } - if ((processed & 0xffff) !== hash) { - const previous = order.filter( - (_, i) => (processed & (1 << (16 + i))) !== 0, - ); - processed = hash; - for (const { id, hash } of previous) { - const p = processors.find((p) => p.id === id); - if (!p) continue; - const index = possible.indexOf(p); - if (index !== -1 && p.hash === hash) processed |= 1 << (16 + index); - } - mediaFile.setProcessors( - processed, - possible - .map((p) => p.id + String.fromCharCode(p.hash >> 8, p.hash & 0xff)) - .join(";"), - ); - } else { - possible = order.map(({ id }) => - UNWRAP(possible.find((p) => p.id === id)) - ); - } - - // Queue needed processors. - const jobs: ProcessJob[] = []; - for (let i = 0, { length } = possible; i < length; i += 1) { - if ((processed & (1 << (16 + i))) === 0) { - const processor = UNWRAP(possible[i]); - const job: ProcessJob = { - path, - stat, - mediaFile, - processor, - index: i, - after: [], - needs: processor.depends.length, - fileNode: node, - }; - jobs.push(job); - if (job.needs === 0) promises.push(queueFileProcessor(job)); - } - } - node.estimate = jobs.length; - fileNode.estimate += jobs.length; - for (const job of jobs) { - for (const dependId of job.processor.depends) { - const dependJob = jobs.find((j) => j.processor.id === dependId); - if (dependJob) { - dependJob.after.push(job); - } else { - ASSERT(job.needs > 0, `job.needs !== 0, ${job.needs}`); - job.needs -= 1; - if (job.needs === 0) promises.push(queueFileProcessor(job)); - } - } - } - if (node.estimate > 0) { - errorHandler.move(); - - const parent = mediaFile.getParent(); - if (parent) parent.setProcessed(0); - } - } - - // Add the root & recursively iterate! - const rootPath = Path.resolve(root); - if (!rootPath.ifExistsSync()) { - throw new Error(`file store ${rootPath} is not mounted`); - } - promises.push(scanDirectory(rootPath)); - - await promises.all(); - fileNode.end(); - dirsNode.end(); - - // Update directory metadata - using dirMetaNode = progress.start("update directory metadata"); - const dirs = MediaFile.getDirectoriesToReindex() - .sort((a, b) => b.path.length - a.path.length); - - for (const dir of dirs) { - using _ = dirMetaNode.start(dir.path); - const children = dir.getChildren(); - - // readme.txt - const readmeContent = children.find((x) => - x.basename === "readme.txt" - )?.contents ?? ""; - - // dirsort - let dirsort: string[] | null = null; - const dirSortRaw = - children.find((x) => x.basename === ".dirsort")?.contents ?? ""; - if (dirSortRaw) { - dirsort = dirSortRaw - .split("\n") - .map((x) => x.trim()) - .filter(Boolean); - } - - // Permissions - if (children.some((x) => x.basename === ".friends")) { - FilePermissions.setPermissions(dir.path, 1); - } else { - FilePermissions.setPermissions(dir.path, 0); - } - - // Recursive stats. - let totalSize = 0; - let newestDate = new Date(0); - let allHashes = ""; - for (const child of children) { - totalSize += child.size; - allHashes += child.hash; - - if (child.basename !== "/readme.txt" && child.date > newestDate) { - newestDate = child.date; - } - } - - // Project Date - const dateFile = children.find((x) => x.basename === ".date"); - if (dateFile) { - const date = new Date(dateFile.contents); - newestDate = date; - dir.setProcessors( - 0, - JSON.stringify({ hideChildrenDates: true }), - ); - } else { - dir.setProcessors(0, ""); - } - - const dirHash = crypto - .createHash("sha1") - .update(dir.path + allHashes) - .digest("hex"); - - MediaFile.markDirectoryProcessed({ - id: dir.id, - timestamp: newestDate, - contents: readmeContent, - size: totalSize, - hash: dirHash, - dirsort, - }); - } - dirMetaNode.end(); - - // Sync to remote - if ( - ((await derived.workDir.ifExistsSync()?.readDir())?.length ?? 0) > 0 - ) { - await rsync.spawn({ - args: [ - "--links", - "--recursive", - "--times", - "--partial", - "--progress", - // "--remove-source-files", - "--delay-updates", - "--exclude=tmp.*", - derived.workDir.toString() + "/", - "clo@zenith:/mnt/storage1/clover/Documents/Config/paperclover/derived/", - ], - progress: progress.start("upload derived assets"), - cwd: process.cwd(), - }); - - await fs.removeEmptyDirectories(derived.workDir.toString()); - } else { - console.info("No new derived assets"); - } - - MediaFile.db.prepare("VACUUM").run(); - MediaFile.db.reload(); - - await rsync.spawn({ - args: [ - MediaFile.db.file, - "clo@zenith:/mnt/storage1/clover/Documents/Config/paperclover/cache.sqlite", - ], - progress: progress.start("Uploading database (source of truth)"), - cwd: process.cwd(), - }); - await rsync.spawn({ - args: [ - MediaFile.db.file, - "clo@paperclover.net:paperclover.net/.clover/cache.sqlite", - ], - progress: progress.start("Uploading database (web node)"), - }); - if (sotToken) { - const res = await fetch("https://db.paperclover.net/reload", { - method: "POST", - headers: { - Authorization: sotToken, - }, - }); - if (!res.ok) { - console.warn( - `Failed to reload remote database ${res.status} ${res.statusText}`, - ); - } - } else console.warn("Missing SOT token"); - - console.info( - "Updated file viewer index in \x1b[1m" + - ((performance.now() - start) / 1000).toFixed(1) + - "s\x1b[0m", - ); - - const { duration, count } = MediaFile.db - .prepare<[], { count: number; duration: number }>(` - select - count(*) as count, - sum(duration) as duration - from media_files - `) - .getNonNull(); - const { derivedSize, derivedCount } = MediaFile.db - .prepare<[], { derivedCount: number; derivedSize: number }>(` - select - count(*) as derivedCount, - sum(size) as derivedSize - from derived_files - `) - .getNonNull(); - const canonicalSize = UNWRAP(MediaFile.getByPath("/")).size; - - console.info(); - console.info( - "Global Stats:\n" + - `- File Count: \x1b[1m${count}\x1b[0m\n` + - `- Derived Count: \x1b[1m${derivedCount}\x1b[0m\n` + - `- Media Duration: \x1b[1m${formatDurationLong(duration)}\x1b[0m\n` + - `- Canonical Size: \x1b[1m${formatSize(canonicalSize)}\x1b[0m\n` + - `- Derived Size: \x1b[1m${formatSize(derivedSize)}\x1b[0m\n`, - ); -} - -interface Process { - name: string; - cores?: number; - enable?: boolean; - include?: Set; - exclude?: Set; - depends?: string[]; - version?: number; - /* Perform an action. */ - run(args: ProcessFileArgs): Promise; -} - -const ffprobeBin = testProgram("ffprobe", "--help"); -const ffmpegBin = testProgram("ffmpeg", "--help"); - -const ffmpegOptions = ["-hide_banner", "-loglevel", "warning"]; - -// NOTE: Never re-order the processors. Add new ones at the end. -const procDuration: Process = { - name: "calculate duration", - enable: ffprobeBin !== null, - include: rules.extsDuration, - cores: 1, - async run({ path, mediaFile }) { - const { stdout } = await subprocess.exec(ffprobeBin!, [ - "-v", - "error", - "-show_entries", - "format=duration", - "-of", - "default=noprint_wrappers=1:nokey=1", - path.toString(), - ]); - - const duration = parseFloat(stdout.trim()); - if (Number.isNaN(duration)) { - throw new Error("Could not extract duration from " + stdout); - } - mediaFile.setDuration(Math.ceil(duration)); - }, -}; - -const procDimensions: Process = { - name: "calculate dimensions", - enable: ffprobeBin != null, - include: rules.extsDimensions, - cores: 1, - async run({ path, mediaFile }) { - const { ext } = path; - - let dimensions; - - if (ext === ".svg") { - // Parse out of text data - const content = await path.read("utf-8"); - const widthMatch = content.match(/width="(\d+)"/); - const heightMatch = content.match(/height="(\d+)"/); - - if (widthMatch && heightMatch) { - dimensions = `${widthMatch[1]}x${heightMatch[1]}`; - } - } else { - // Use ffprobe to observe streams - const { stdout } = await subprocess.exec("ffprobe", [ - "-v", - "error", - "-select_streams", - "v:0", - "-show_entries", - "stream=width,height", - "-of", - "csv=s=x:p=0", - path.toString(), - ]); - if (stdout.includes("x")) { - dimensions = stdout.trim(); - } - } - - mediaFile.setDimensions(dimensions ?? ""); - }, -}; - -const procLoadTextContents: Process = { - name: "load text content", - include: rules.extsReadContents, - cores: 1, - version: 2, - async run({ path, mediaFile, stat }) { - if (stat.size > 1_000_000) return; - const text = await path.read("utf-8"); - mediaFile.setContents(text); - }, -}; - -const procHighlightCode: Process = { - name: "highlight source code", - include: new Set(rules.extsCode.keys()), - cores: 1, - version: 2, - async run({ path, mediaFile, stat }) { - const language = UNWRAP( - rules.extsCode.get(path.ext.toLowerCase()), - ); - // An issue is that .ts is an overloaded extension, shared between - // 'transport stream' and 'typescript'. - // - // Filter used here is: - // - more than 1mb - // - invalid UTF-8 - if (stat.size > 1_000_000) return; - let code; - const buf = await path.read(); - try { - code = new TextDecoder("utf-8", { fatal: true }).decode(buf); - } catch (error) { - mediaFile.setContents(""); - return; - } - const content = await highlight.highlightCode(code, language); - mediaFile.setContents(content); - }, -}; - -const procImageSubsets: Process = { - name: "encode image subsets", - include: rules.extsImage, - depends: [procDimensions.name], - version: 2, - async run({ path, mediaFile, node }) { - const { width, height } = UNWRAP(mediaFile.parseDimensions()); - const targetSizes = transcodeRules.imageSizes.filter((w) => w < width); - const baseStatus = node.text; - - using stack = new DisposableStack(); - for (const size of targetSizes) { - const { w, h } = resizeDimensions(width, height, size); - for (const { ext, args } of transcodeRules.imagePresets) { - node.text = baseStatus + ` (${w}x${h}, ${ext.slice(1).toUpperCase()})`; - - stack.use( - await derived.produce({ - mediaFile, - node, - cores: 2, - subkey: `${size}${ext}`, - async producer(dir) { - await subprocess.exec(ffmpegBin!, [ - ...ffmpegOptions, - "-i", - path.toString(), - "-vf", - `scale=${w}:${h}:force_original_aspect_ratio=increase,crop=${w}:${h}`, - ...args, - dir.join(`${size}${ext}`).toString(), - ]); - }, - }), - ); - } - } - - stack.move(); - }, -}; - -const videoArgsCache = new async.OnceMap(transcodeRules.getVideoInputArgs); - -const qualityMap: Record = { - u: "ultra-high", - h: "high", - m: "medium", - l: "low", - d: "data-saving", -}; -const procVideos = transcodeRules.videoFormats.map((preset) => ({ - name: `encode av1 ${UNWRAP(qualityMap[UNWRAP(preset.id[1])])}`, - include: rules.extsVideo, - enable: ffmpegBin != null, - depends: [procDuration.name, procDimensions.name], - async run({ path, mediaFile, node }) { - if ((mediaFile.duration ?? 0) < 5) return; - if (!mediaFile.dimensions) return; - - if (mediaFile.path === "/2021/top-10000-bread/output.mp4") return; - - await derived.produce({ - mediaFile, - node, - cores: 4, - subkey: `av1-${preset.id}`, - async producer(dir) { - const input = await videoArgsCache.getOrRun(path); - const args = transcodeRules.getAv1VideoArgs( - preset, - UNWRAP(input.video, "frick on " + path + JSON.stringify(input)), - dir, - ); - - await ffmpeg.spawn({ - ffmpeg: ffmpegBin!, - progress: node, - args, - cwd: dir, - }); - }, - }); - }, -})); -const procVideoAudios = transcodeRules.audioFormats.map((preset) => ({ - name: `encode opus ${UNWRAP(qualityMap[UNWRAP(preset.id)])}`, - include: rules.extsVideo, - enable: ffmpegBin != null, - depends: [procDuration.name, procDimensions.name], - version: 2, - async run({ path, mediaFile, node }) { - if ((mediaFile.duration ?? 0) < 5) return; - if (!mediaFile.dimensions) return; - if (mediaFile.path === "/2021/top-10000-bread/output.mp4") return; - await derived.produce({ - mediaFile, - node, - cores: 4, - subkey: `opus-${preset.id}`, - async producer(dir) { - const input = await videoArgsCache.getOrRun(path); - ASSERT(input.video); - if (!input.audio) return; - const args = transcodeRules.getOpusAudioArgs(preset, input.audio, dir); - - await ffmpeg.spawn({ - ffmpeg: ffmpegBin!, - progress: node, - args, - cwd: dir, - }); - }, - }); - }, -})); -const procDash: Process = { - name: `encode mpeg-dash`, - include: rules.extsVideo, - enable: ffmpegBin != null, - depends: [...procVideos, ...procVideoAudios].map((x) => x.name), - version: 4, - async run({ path, mediaFile, node }) { - if ((mediaFile.duration ?? 0) < 5) return; - if (!mediaFile.dimensions) return; - - if (mediaFile.path === "/2021/top-10000-bread/output.mp4") return; - await derived.produce({ - mediaFile, - node, - cores: 1, - subkey: `dash-av1`, - async producer(dir) { - const input = await videoArgsCache.getOrRun(path); - const videos = transcodeRules.videoFormats.map( - (preset) => - UNWRAP(dir.parent).join( - `av1-${preset.id}`, - transcodeRules.av1FileName, - ), - ); - const audios = input.audio - ? transcodeRules.audioFormats.map( - (preset) => - UNWRAP(dir.parent).join( - `opus-${preset.id}`, - transcodeRules.opusFileName, - ), - ) - : []; - const args = transcodeRules.getMpegDashArgs( - videos, - audios, - dir, - ); - - await dir.join("d").makeDir(); - - await ffmpeg.spawn({ - ffmpeg: ffmpegBin!, - progress: node, - args, - cwd: dir, - }); - }, - }); - }, -}; -const procH264Hls: Process = { - name: `encode h.264 hls`, - include: rules.extsVideo, - enable: ffmpegBin != null, - depends: [procDuration.name, procDimensions.name], - async run({ path, mediaFile, node }) { - if ((mediaFile.duration ?? 0) < 5) return; - if (!mediaFile.dimensions) return; - - await derived.produce({ - mediaFile, - node, - cores: 8, - subkey: `hls`, - async producer(dir) { - const input = await videoArgsCache.getOrRun(path); - const args = transcodeRules.getH264HlsArgs(input, dir); - - await ffmpeg.spawn({ - ffmpeg: ffmpegBin!, - progress: node, - args, - cwd: dir, - }); - }, - }); - }, -}; - -const procCompression = [ - { name: "gzip", fn: () => zlib.createGzip({ level: 9 }) }, - { name: "zstd", fn: () => zlib.createZstdCompress() }, -].map( - ({ name, fn }) => ({ - name: `compress ${name}`, - exclude: rules.extsPreCompressed, - async run({ path, mediaFile, node }) { - await derived.produce({ - mediaFile, - node, - cores: 1, - subkey: name, - async producer(dir) { - await stream.promises.pipeline( - fs.createReadStream(path.toString()), - fn(), - fs.createWriteStream(dir.join(name).toString()), - ); - }, - }); - }, - }), -); - -const processors = [ - procDimensions, - procDuration, - procLoadTextContents, - procHighlightCode, - procImageSubsets, - ...procVideos, - ...procVideoAudios, - procDash, - ...procCompression, - procH264Hls, -].map((process, id, all) => { - const strIndex = (id: number) => String.fromCharCode("a".charCodeAt(0) + id); - return { - ...(process as Process), - id: strIndex(id), - // Create a unique key. - hash: new Uint16Array( - crypto - .createHash("sha1") - .update( - process.run.toString() + - (process.version ? String(process.version) : ""), - ) - .digest().buffer, - ).reduce((a, b) => a ^ b), - depends: (process.depends ?? []).map((depend) => { - const index = all.findIndex((p) => p.name === depend); - if (index === -1) throw new Error(`Cannot find depend '${depend}'`); - if (index === id) throw new Error(`Cannot depend on self: '${depend}'`); - return strIndex(index); - }), - }; -}); - -function resizeDimensions(w: number, h: number, desiredWidth: number) { - ASSERT(desiredWidth < w, `${desiredWidth} < ${w}`); - return { w: desiredWidth, h: Math.floor((h / w) * desiredWidth) }; -} - -interface UpdateMetadataArgs { - path: Path; - publicPath: string; - stat: fs.Stats; - mediaFile: MediaFile | null; -} - -interface ProcessFileArgs { - path: Path; - stat: fs.Stats; - mediaFile: MediaFile; - node: progress.Node; -} - -interface ProcessJob { - path: Path; - stat: fs.Stats; - mediaFile: MediaFile; - processor: (typeof processors)[0]; - index: number; - after: ProcessJob[]; - needs: number; - fileNode: progress.Node; -} - -function skipBasename(basename: string): boolean { - // dot files must be incrementally tracked - if (basename === ".dirsort") return false; - if (basename === ".friends") return false; - if (basename === ".date") return false; - - return ( - basename.startsWith(".") || - // basename.startsWith("._") || - // basename.startsWith(".tmp") || - // basename === ".DS_Store" || - basename.toLowerCase() === "thumbs.db" || - basename.toLowerCase() === "desktop.ini" - ); -} - -function toPublicPath(diskPath: Path) { - if (diskPath.toString() === root) return "/"; - return "/" + path.relative(root, diskPath.toString()).replaceAll("\\", "/"); -} - -function testProgram(name: string, helpArgument: string) { - try { - child_process.spawnSync(name, [helpArgument]); - return name; - } catch (err) { - console.warn(`Missing or corrupt executable '${name}'`); - } - return null; -} - -// Helper function to check and remove location metadata -async function scrubLocationMetadata( - path: Path, - stats: fs.Stats, - progress: progress.Ref, -): Promise { - using _ = progress.start("scrub exif metadata"); - const ext = path.ext.toLowerCase(); - if (!rules.extsScrubExif.has(ext)) return false; - - let hasLocation = false; - let args: string[] = []; - - // Check for location metadata based on file type - const tempOutput = UNWRAP(path.parent).join(`.tmp.${path.base}`); - switch (ext) { - case ".jpg": - case ".jpeg": - case ".png": - const { stdout: gpsCheck } = await subprocess.exec("exiftool", [ - "-gps:all", - path.toString(), - ]); - hasLocation = gpsCheck.trim().length > 0; - args = ["-gps:all=", path.toString(), "-o", tempOutput.toString()]; - break; - case ".mov": - case ".mp4": - const { stdout: videoCheck } = await subprocess.exec("exiftool", [ - "-ee", - "-G3", - "-s", - path.toString(), - ]); - hasLocation = videoCheck.includes("GPS") || - videoCheck.includes("Location"); - args = [ - "-gps:all=", - "-xmp:all=", - path.toString(), - "-o", - tempOutput.toString(), - ]; - break; - case ".m4a": - const { stdout: m4aCheck } = await subprocess.exec("exiftool", [ - "-ee", - "-G3", - "-s", - path.toString(), - ]); - hasLocation = m4aCheck.includes("GPS") || - m4aCheck.includes("Location") || - m4aCheck.includes("Filename") || - m4aCheck.includes("Title"); - - if (hasLocation) { - args = [ - "-gps:all=", - "-location:all=", - "-filename:all=", - "-title=", - "-m4a:all=", - path.toString(), - "-o", - tempOutput.toString(), - ]; - } - break; - } - - const accessTime = stats.atime; - const modTime = stats.mtime; - - let backup: Path | null = null; - try { - if (hasLocation) { - // Prepare a backup - const tmp = UNWRAP(path.parent).join(`.tmp.backup.${path.base}`); - await fsp.copyFile(path.toString(), tmp.toString()); - await fsp.utimes(tmp.toString(), accessTime, modTime); - backup = tmp; - - // Remove metadata - await subprocess.exec("exiftool", args); - if (!tempOutput.ifExistsSync()) { - throw new Error(`Failed to create output file: ${tempOutput}`); - } - - // Restore original timestamps - await fsp.rename(tempOutput.toString(), path.toString()); - await fsp.utimes(path.toString(), accessTime, modTime); - - // Backup is no longer needed - await fsp.unlink(backup.toString()); - - console.info( - `Scrubbed location metadata in ${path.relative(Path.resolve(root))}`, - ); - return true; - } - } catch (error) { - if (backup) { - await fsp.rename(backup.toString(), path.toString()); - } - if (fs.existsSync(tempOutput.toString())) { - await fsp.unlink(tempOutput.toString()); - } - throw error; - } - - return false; -} - -const monthMilliseconds = 30 * 24 * 60 * 60 * 1000; - -import * as async from "lib/async.ts"; -import * as fs from "#sitegen/fs"; -import * as subprocess from "lib/subprocess.ts"; -import { Path } from "#sitegen/path"; -import * as queue from "../../../lib/queue.ts"; -import * as log from "../../../lib/log.ts"; -import * as progress from "lib/progress.ts"; -import * as ts from "../../../lib/ts.ts"; - -import * as child_process from "node:child_process"; -import * as crypto from "node:crypto"; -import * as fsp from "node:fs/promises"; -import * as path from "node:path"; -import * as stream from "node:stream"; -import * as zlib from "node:zlib"; - -import { MediaFile, MediaFileKind } from "@/file-viewer/models/MediaFile.ts"; -import { FilePermissions } from "@/file-viewer/models/FilePermissions.ts"; -import { - formatDate, - formatDurationLong, - formatSize, -} from "@/file-viewer/format.ts"; -import * as rules from "@/file-viewer/rules.ts"; -import * as highlight from "@/file-viewer/highlight.ts"; -import * as ffmpeg from "lib/subprocess/ffmpeg.ts"; -import * as rsync from "@/file-viewer/rsync.ts"; -import * as derived from "@/file-viewer/models/derived.ts"; -import * as transcodeRules from "@/file-viewer/transcode-rules.ts"; - -import { rawFileRoot as root } from "../paths.ts"; -import { ASSERT, UNWRAP } from "lib/assert.ts"; diff --git a/src/file-viewer/models/MediaFile.ts b/src/file-viewer/models/MediaFile.ts index 5b4a5a4dbe3d75c99f8530c7b1321c18658a32bc..006f220c5514dc418884af23e9973362b32d4bc1 100644 --- a/src/file-viewer/models/MediaFile.ts +++ b/src/file-viewer/models/MediaFile.ts @@ -224,6 +224,7 @@ export class MediaFile { return createDirectoryQuery.getNonNull({ path: filePath, parentId, + time: Date.now(), }).id; } // walk up the path until we find a directory that exists @@ -236,6 +237,7 @@ export class MediaFile { parentId = createDirectoryQuery.getNonNull({ path: current, parentId, + time: Date.now(), }).id; } // walk back down the path, creating directories as needed @@ -245,6 +247,7 @@ export class MediaFile { parentId = createDirectoryQuery.getNonNull({ path: current, parentId, + time: Date.now(), }).id; } return parentId; @@ -314,16 +317,16 @@ export interface DirConfig { // Get a directory ID by path, creating it if it doesn't exist const createDirectoryQuery = db.prepare< - [{ path: string; parentId: number | null }], + [{ path: string; parentId: number | null; time: number }], { id: number } >( /* SQL */ ` insert into media_files ( - path, parent_id, kind, timestamp, hash, size, - duration, dimensions, contents, dirsort, processed) + path, parent_id, kind, timestamp, timestamp_updated, hash, + size, duration, dimensions, contents, dirsort, processed) values ( - $path, $parentId, ${MediaFileKind.directory}, 0, '', 0, - 0, '', '', '', 0) + $path, $parentId, ${MediaFileKind.directory}, 0, $time, '', + 0, 0, '', '', '', 0) returning id; `, ); diff --git a/src/file-viewer/models/derived.ts b/src/file-viewer/models/derived.ts index cf90ef90bda5e3e13ec7e7fc22641495913198c9..10ff11985ebf44ea5af5393f38e81837d3dab443 100644 --- a/src/file-viewer/models/derived.ts +++ b/src/file-viewer/models/derived.ts @@ -165,6 +165,14 @@ export function get(file: MediaFile, subPath: string) { return null; } +export function findOrphanedRoots() { + return findOrphanedRootsQuery.array(); +} + +export function deleteRoot(root: { id: number }) { + return deleteRootQuery.run({ root: root.id }); +} + const insertRootQuery = db.prepare< [{ key: string; date: number }], { id: number } @@ -194,6 +202,16 @@ const deleteRefQuery = db.prepare<[{ root: number; file: number }]>(/* SQL */ ` delete from derived_refs where root = $root and file = $file; `); +const findOrphanedRootsQuery = db.prepare< + [], + { id: number; key: string; date: number } +>(/* SQL */ ` + select dr.id, dr.key, dr.date + from derived_roots dr + left join derived_refs refs on refs.root = dr.id + where refs.id is null +`); + import { getDb } from "#sitegen/sqlite"; import type { MediaFile } from "./MediaFile.ts"; import { Path } from "#sitegen/path"; diff --git a/src/file-viewer/rsync.ts b/src/file-viewer/rsync.ts index 535c517dfd3083ff964ee4f4ae77e099c1dce1e9..a48b853d7ed443efd19b4fd6c25f15999ed7350d 100644 --- a/src/file-viewer/rsync.ts +++ b/src/file-viewer/rsync.ts @@ -10,7 +10,7 @@ export type Line = currentFile: string; bytesTransferred: number; percentage: number; - timeElapsed: string; + timeElapsed: string | null; transferNumber: number; filesToCheck: number; totalFiles: number; @@ -91,7 +91,7 @@ export async function spawn(options: SpawnOptions) { e.args = [rsync, ...args].join(" "); e.code = code; e.signal = signal; - return e; + throw e; } } @@ -126,9 +126,9 @@ export class Parse { return { kind: "progress", currentFile: this.lastSeenFile || "", - bytesTransferred: Number(bytesStr.replaceAll(",", "")), + bytesTransferred: Number(UNWRAP(bytesStr).replaceAll(",", "")), percentage: Number(percentageStr), - timeElapsed, + timeElapsed: timeElapsed ?? null, transferNumber: this.currentTransfer, filesToCheck: toCheckStr ? this.toCheck = Number(toCheckStr) @@ -185,3 +185,4 @@ import * as readline from "node:readline"; import events from "node:events"; import * as progress from "lib/progress.ts"; import { formatSize } from "@/file-viewer/format.ts"; +import { UNWRAP } from "lib/assert.ts"; diff --git a/src/friend-auth.ts b/src/friend-auth.ts index d65c8d62ef616bb05ae79a84647179eba445b867..2aba24f2979ea222177394bba21198b19c69255c 100644 --- a/src/friend-auth.ts +++ b/src/friend-auth.ts @@ -1,75 +1,96 @@ -let friendPassword = ""; +let hardcoded = { + friendPassword: "", + perPage: {} as Record, + getForFile: (file: string) => [] as string[], +}; try { - friendPassword = require("./friends/hardcoded-password.ts").friendPassword; + hardcoded = require("./friends/hardcoded-password.ts"); } catch {} export const app = new Hono(); const cookieAge = 60 * 60 * 24 * 30; // 1 month -function checkFriendsCookie(c: Context) { +export const getForFile = hardcoded.getForFile; + +function checkFriendsCookie(c: Context, passwords: string[]) { const cookie = c.req.header("Cookie"); if (!cookie) return false; const cookies = cookie.split("; ").map((x) => x.split("=")); return cookies.some( (kv) => - kv[0].trim() === "friends_password" && - kv[1].trim() && - kv[1].trim() === friendPassword, + kv[0]!.trim() === "friends_password" && + kv[1]!.trim() && + passwords.includes(kv[1]!.trim()), ); } -export function requireFriendAuth(c: Context) { +export function requireFriendAuth( + c: Context, + passwords = [hardcoded.friendPassword], +) { + if (passwords.length === 0 || passwords[0]!.length === 0) { + return serveAsset(c, "/friends/misconfigure", 501); + } const k = c.req.query("password") || c.req.query("k"); if (k) { - if (k === friendPassword) { + if (passwords.includes(k)) { return c.body(null, 303, { - Location: "/friends", + Location: c.req.path, "Set-Cookie": `friends_password=${k}; Path=/; HttpOnly; SameSite=Strict; Max-Age=${cookieAge}`, }); } else { return c.body(null, 303, { - Location: "/friends", + Location: c.req.path, }); } } - if (checkFriendsCookie(c)) { + if (checkFriendsCookie(c, passwords)) { return undefined; } else { return serveAsset(c, "/friends/auth", 403); } } -app.get("/friends", (c) => { - const friendAuthChallenge = requireFriendAuth(c); - if (friendAuthChallenge) return friendAuthChallenge; - return serveAsset(c, "/friends", 200); -}); - let incorrectMap: Record = {}; -app.post("/friends", async (c) => { - const ip = c.header("X-Forwarded-For") ?? getConnInfo(c).remote.address ?? - "unknown"; - if (incorrectMap[ip]) { + +function friendPage(route: assets.Key) { + const expected = hardcoded.perPage[route] ?? hardcoded.friendPassword; + + app.get(route, (c) => { + const friendAuthChallenge = requireFriendAuth(c, [expected]); + if (friendAuthChallenge) return friendAuthChallenge; + return serveAsset(c, route, 200); + }); + + app.post(route, async (c) => { + const ip = c.header("X-Forwarded-For") ?? getConnInfo(c).remote.address ?? + "unknown"; + if (incorrectMap[ip]) { + return serveAsset(c, "/friends/auth/fail", 403); + } + const data = await c.req.formData(); + const k = data.get("password"); + if (k === expected) { + return c.body(null, 303, { + Location: c.req.path, + "Set-Cookie": + `friends_password=${k}; Path=/; HttpOnly; SameSite=Strict; Max-Age=${cookieAge}`, + }); + } + incorrectMap[ip] = true; + await setTimeout(2500); + incorrectMap[ip] = false; return serveAsset(c, "/friends/auth/fail", 403); - } - const data = await c.req.formData(); - const k = data.get("password"); - if (k === friendPassword) { - return c.body(null, 303, { - Location: "/friends", - "Set-Cookie": - `friends_password=${k}; Path=/; HttpOnly; SameSite=Strict; Max-Age=${cookieAge}`, - }); - } - incorrectMap[ip] = true; - await setTimeout(2500); - incorrectMap[ip] = false; - return serveAsset(c, "/friends/auth/fail", 403); -}); + }); +} + +friendPage("/friends"); +friendPage("/friends/oct25"); import { type Context, Hono } from "hono"; import { serveAsset } from "#sitegen/assets"; import { setTimeout } from "node:timers/promises"; import { getConnInfo } from "#hono/conninfo"; +import * as assets from "#sitegen/assets"; diff --git a/src/pages/friends/auth.fail.marko b/src/pages/friends/auth.fail.marko index d03b9a2c284d3f4d7dd75a856981260ed697b26b..48269abe7f87b1482a3ba8a6b32ae46ac96ead8d 100644 --- a/src/pages/friends/auth.fail.marko +++ b/src/pages/friends/auth.fail.marko @@ -3,9 +3,9 @@ export const meta = { title: "password required" };

incorrect or outdated password

please contact clover

-
+

try again

- +
diff --git a/src/pages/friends/auth.marko b/src/pages/friends/auth.marko index bea6ed3a4e17af6c2cf97395eff66aaa8f11835e..ccfe2d0cb1cd9c7fcc431d68088d078e29722ae9 100644 --- a/src/pages/friends/auth.marko +++ b/src/pages/friends/auth.marko @@ -1,9 +1,9 @@ export const meta = { title: "password required" };
-
+

what's the password

- +

(clover will give the password to you if you're a friend)

diff --git a/src/pages/friends/misconfigure.marko b/src/pages/friends/misconfigure.marko new file mode 100644 index 0000000000000000000000000000000000000000..c7e09fca74a6a9f9502f8cebff50b789380daa2a --- /dev/null +++ b/src/pages/friends/misconfigure.marko @@ -0,0 +1,10 @@ +export const meta = { title: "password required" }; + +
+
+

what's the password

+ + +

this page's password is not configured

+
+
diff --git a/src/tags/clover-video.client.ts b/src/tags/clover-video.client.ts index 547e55cc5815d020ba3c35d38a34a145627b002c..7a64b19dfe52555ec6b618053c18cec056bfc659 100644 --- a/src/tags/clover-video.client.ts +++ b/src/tags/clover-video.client.ts @@ -67,7 +67,9 @@ function hydrateVideoInner( container: HTMLElement, mode: PlaybackMode, ) { - const video = container.querySelector("video")!; + const video = container instanceof HTMLVideoElement + ? container + : container.querySelector("video")!; if (video.hasAttribute("controls")) return; let src = video.getAttribute("data-src"); if (!src) return; diff --git a/src/tags/clover-video.marko b/src/tags/clover-video.marko index 2d2a4c145ad5abb1ed338871645db5910d05c7e8..81d2d442f4adafc7f0cff6fbd9e5835b27c3a3b9 100644 --- a/src/tags/clover-video.marko +++ b/src/tags/clover-video.marko @@ -5,9 +5,10 @@ export interface Input { poster?: MediaFile | string; header?: Marko.AttrTag<{ content: Marko.Body }>; minimal?: boolean; + noFigure?: boolean } - + @@ -19,40 +20,50 @@ export interface Input { - - -
- <${header.content}/> - ${file.basename} - -
- - -