| author | |
| committer | |
| log | 12f1b96207363924cca0970b5e6cb515a6d7541a |
| tree | 6a7082de6ca703b06e9f72831b15fe16276e7c24 |
| parent | b6380f3a43e39e9571af27722d728fc17327d472 |
| signature | Signed by SSH key SHA256:cOKiuRFOeSRxne6EWgHtdQQSlBxjOXm2hOCFnCdLQbQ |
Assisted-by: Claude:claude-fable-539 files changed, 2995 insertions(+), 1200 deletions(-)
framework/lib/sqlite.ts+33-16| ... | ... | @@ -26,7 +26,7 @@ export class WrappedDatabase { |
| 26 | 26 | |
| 27 | 27 | constructor(file: string) { |
| 28 | 28 | this.file = file; |
| 29 | this.node = new DatabaseSync(file); | |
| 29 | this.node = WrappedDatabase.open(file); | |
| 30 | 30 | this.node.exec(` |
| 31 | 31 | create table if not exists clover_migrations ( |
| 32 | 32 | key text not null primary key, |
| ... | ... | @@ -35,6 +35,16 @@ export class WrappedDatabase { |
| 35 | 35 | `); |
| 36 | 36 | } |
| 37 | 37 | |
| 38 | // wal so the source of truth can serve reads + db snapshots while the | |
| 39 | // indexer writes. foreign keys are per-connection and must be re-applied | |
| 40 | // on reload. | |
| 41 | private static open(file: string) { | |
| 42 | const node = new DatabaseSync(file); | |
| 43 | node.exec(`pragma journal_mode = wal;`); | |
| 44 | node.exec(`pragma foreign_keys = on;`); | |
| 45 | return node; | |
| 46 | } | |
| 47 | ||
| 38 | 48 | // TODO: add migration support |
| 39 | 49 | // the idea is you keep `schema` as the new schema but can add |
| 40 | 50 | // migrations to the mix really easily. |
| ... | ... | @@ -60,20 +70,13 @@ export class WrappedDatabase { |
| 60 | 70 | ); |
| 61 | 71 | query = lines.map((x) => x.slice(trim)).join("\n"); |
| 62 | 72 | |
| 63 | let prepared; | |
| 64 | try { | |
| 65 | prepared = this.node.prepare(query); | |
| 66 | } catch (err) { | |
| 67 | if (err) (err as { query: string }).query = query; | |
| 68 | throw err; | |
| 69 | } | |
| 70 | const stmt = new Stmt<Args, Result>(prepared); | |
| 73 | const stmt = new Stmt<Args, Result>(this, query); | |
| 71 | 74 | this.stmts.push(stmt); |
| 72 | 75 | return stmt; |
| 73 | 76 | } |
| 74 | 77 | |
| 75 | 78 | reload() { |
| 76 | const newNode = new DatabaseSync(this.file); | |
| 79 | const newNode = WrappedDatabase.open(this.file); | |
| 77 | 80 | this.node.close(); |
| 78 | 81 | this.node = newNode; |
| 79 | 82 | for (const stmt of this.stmts) { |
| ... | ... | @@ -83,17 +86,31 @@ export class WrappedDatabase { |
| 83 | 86 | } |
| 84 | 87 | |
| 85 | 88 | export class Stmt<Args extends unknown[] = unknown[], Row = unknown> { |
| 86 | #node: StatementSync; | |
| 89 | // statements prepare lazily on first use, so that importing a model | |
| 90 | // module never requires its tables to exist yet (the migration scripts | |
| 91 | // depend on this, and it makes `reload` cheap). | |
| 92 | #db: WrappedDatabase; | |
| 93 | #lazyNode: StatementSync | null = null; | |
| 87 | 94 | #class: any | null = null; |
| 88 | 95 | query: string; |
| 89 | 96 | |
| 90 | constructor(node: StatementSync) { | |
| 91 | this.#node = node; | |
| 92 | this.query = node.sourceSQL; | |
| 97 | constructor(db: WrappedDatabase, query: string) { | |
| 98 | this.#db = db; | |
| 99 | this.query = query; | |
| 100 | } | |
| 101 | ||
| 102 | get #node(): StatementSync { | |
| 103 | if (this.#lazyNode) return this.#lazyNode; | |
| 104 | try { | |
| 105 | return this.#lazyNode = this.#db.node.prepare(this.query); | |
| 106 | } catch (err) { | |
| 107 | if (err) (err as any).query = this.query; | |
| 108 | throw err; | |
| 109 | } | |
| 93 | 110 | } |
| 94 | 111 | |
| 95 | private reload(db: DatabaseSync) { | |
| 96 | this.#node = db.prepare(this.query); | |
| 112 | private reload(_db: DatabaseSync) { | |
| 113 | this.#lazyNode = null; | |
| 97 | 114 | } |
| 98 | 115 | |
| 99 | 116 | /** Get one row */ |
lib/log.ts+19-6| ... | ... | @@ -457,11 +457,24 @@ export function createTerminalWidgetHost( |
| 457 | 457 | timer = null; |
| 458 | 458 | ASSERT(!rendering); |
| 459 | 459 | rendering = true; |
| 460 | // the finally is load-bearing: if a render callback or terminal write | |
| 461 | // throws, `rendering` must reset, or the exit path's cancel() asserts | |
| 462 | // and masks the original error. | |
| 463 | try { | |
| 464 | redrawCallbackInner(); | |
| 465 | } finally { | |
| 466 | rendering = false; | |
| 467 | } | |
| 468 | } | |
| 469 | ||
| 470 | function redrawCallbackInner() { | |
| 460 | 471 | redrawTime = (lastFlush = now()) - 0.00001; // windows time precision workaround |
| 461 | 472 | |
| 462 | 473 | // trivial path when not using widgets |
| 463 | 474 | if (!lines.length && !widgets.length) { |
| 464 | ASSERT(buffer); | |
| 475 | // a redraw can get scheduled with nothing to write (e.g. widgets torn | |
| 476 | // down before the timer fired); it is a no-op, not an error | |
| 477 | if (!buffer) return; | |
| 465 | 478 | needsToRestoreCursor = false; |
| 466 | 479 | needsToSaveCursor = false; |
| 467 | 480 | if (writeOutputTemporaryLock) { |
| ... | ... | @@ -482,7 +495,6 @@ export function createTerminalWidgetHost( |
| 482 | 495 | } |
| 483 | 496 | partialLineIndex = partialLineLength(buffer); |
| 484 | 497 | buffer = ""; |
| 485 | rendering = false; | |
| 486 | 498 | return; |
| 487 | 499 | } |
| 488 | 500 | |
| ... | ... | @@ -544,7 +556,6 @@ export function createTerminalWidgetHost( |
| 544 | 556 | if (buffer) terminal.writeOutput(buffer); |
| 545 | 557 | buffer = ""; |
| 546 | 558 | if (hasSyncStart) terminal.writeInteractive(ansi.syncEnd); |
| 547 | rendering = false; | |
| 548 | 559 | return; |
| 549 | 560 | } |
| 550 | 561 | |
| ... | ... | @@ -645,7 +656,6 @@ export function createTerminalWidgetHost( |
| 645 | 656 | needsToSaveCursor = false; |
| 646 | 657 | lines = newWidgetLines; |
| 647 | 658 | buffer = ""; |
| 648 | rendering = false; | |
| 649 | 659 | } |
| 650 | 660 | |
| 651 | 661 | function redrawSoon(ms: number) { |
| ... | ... | @@ -770,8 +780,11 @@ export function createTerminalWidgetHost( |
| 770 | 780 | }; |
| 771 | 781 | }, |
| 772 | 782 | cancel() { |
| 773 | ASSERT(!rendering, "cannot call cancel() during rendering"); | |
| 774 | flushAndClear(false); | |
| 783 | // cancel may be reached from exit handlers while a redraw is on the | |
| 784 | // stack (a crash inside a render callback unwinds through here); skip | |
| 785 | // the flush in that case and just tear down, so the original error | |
| 786 | // is the one that surfaces. | |
| 787 | if (!rendering) flushAndClear(false); | |
| 775 | 788 | widgets.splice(0, widgets.length); |
| 776 | 789 | internals.splice(0, internals.length); |
| 777 | 790 | }, |
lib/progress.test.ts+25| ... | ... | @@ -248,6 +248,30 @@ test("event encoding round trip cases", async () => { |
| 248 | 248 | }, |
| 249 | 249 | [9], |
| 250 | 250 | ]); |
| 251 | ||
| 252 | // log messages: a non-info level without scope/stack/custom previously | |
| 253 | // desynced the whole stream (the flag tests used `&&` instead of `&`), | |
| 254 | // and the message/stack-frame loops never decremented their counters | |
| 255 | await roundTrip([ | |
| 256 | 0, | |
| 257 | { | |
| 258 | k: 1 as EncodedKey, | |
| 259 | t: "encode av1", | |
| 260 | l: [ | |
| 261 | { level: "warn", text: "deprecated pixel format", time: 1768210850000 }, | |
| 262 | { level: "info", text: "hello", time: 1768210851000 }, | |
| 263 | { | |
| 264 | level: "error", | |
| 265 | text: "scoped + stacked", | |
| 266 | time: 1768210852000, | |
| 267 | scope: "ffmpeg", | |
| 268 | stack: [{ fn: "spawn", file: "ffmpeg.ts", line: 12, col: 3 }], | |
| 269 | }, | |
| 270 | ], | |
| 271 | [progress.internals.kNode]: fakeNode, | |
| 272 | }, | |
| 273 | [1], | |
| 274 | ]); | |
| 251 | 275 | }); |
| 252 | 276 | |
| 253 | 277 | function cleanProgressEvent(event: progress.StreamEvent) { |
| ... | ... | @@ -255,6 +279,7 @@ function cleanProgressEvent(event: progress.StreamEvent) { |
| 255 | 279 | typeof x === "object" && !Array.isArray(x) |
| 256 | 280 | ? testing.removeUndefinedKeys({ |
| 257 | 281 | ...x, |
| 282 | l: x.l?.map((m) => testing.removeUndefinedKeys({ ...m })), | |
| 258 | 283 | E: undefined, |
| 259 | 284 | p: undefined, |
| 260 | 285 | s: undefined, |
lib/progress.ts+36-11| ... | ... | @@ -1007,6 +1007,7 @@ export function decodeEventStream< |
| 1007 | 1007 | } |
| 1008 | 1008 | ASSERT(hasEmittedEnd, "Stream terminated early."); |
| 1009 | 1009 | } catch (err) { |
| 1010 | console.error(err); | |
| 1010 | 1011 | reader.cancel(err); |
| 1011 | 1012 | if (!signal.aborted) reject(err); |
| 1012 | 1013 | } finally { |
| ... | ... | @@ -1251,6 +1252,9 @@ class Decoder< |
| 1251 | 1252 | t: text, |
| 1252 | 1253 | v: value, |
| 1253 | 1254 | e: total, |
| 1255 | s: showTotal, | |
| 1256 | p: passive, | |
| 1257 | h: hidden, | |
| 1254 | 1258 | l: messages, |
| 1255 | 1259 | c: children, |
| 1256 | 1260 | E: estimatedTime, |
| ... | ... | @@ -1265,6 +1269,9 @@ class Decoder< |
| 1265 | 1269 | if (text) node.text = text; |
| 1266 | 1270 | if (value) node.value = value; |
| 1267 | 1271 | if (total) node.total = total; |
| 1272 | if (showTotal != null) node.showTotal = showTotal; | |
| 1273 | if (passive != null) node.passive = passive; | |
| 1274 | if (hidden != null) node.hidden = hidden; | |
| 1268 | 1275 | for (const message of messages ?? []) node.log.writeMessage(message); |
| 1269 | 1276 | if (estimatedTime) node.estimatedTime = estimatedTime; |
| 1270 | 1277 | if (formattedValue) node.valueFormatter = () => formattedValue; |
| ... | ... | @@ -1275,6 +1282,9 @@ class Decoder< |
| 1275 | 1282 | text, |
| 1276 | 1283 | value, |
| 1277 | 1284 | total, |
| 1285 | showTotal, | |
| 1286 | passive, | |
| 1287 | hidden, | |
| 1278 | 1288 | messages, |
| 1279 | 1289 | estimatedTime, |
| 1280 | 1290 | formattedValue, |
| ... | ... | @@ -1301,7 +1311,9 @@ class Decoder< |
| 1301 | 1311 | |
| 1302 | 1312 | startRecursive(key: EncodedKey, opts: PendingStart): Ref { |
| 1303 | 1313 | ASSERT(this.pendingStart.delete(key)); |
| 1304 | const parentId = UNWRAP(this.pendingParents.get(key)); | |
| 1314 | // a node whose parent linkage is missing (malformed stream) attaches at | |
| 1315 | // the root rather than killing the whole decoder | |
| 1316 | const parentId = this.pendingParents.get(key) ?? (0 as EncodedKey); | |
| 1305 | 1317 | let ref: Ref | null = parentId === 0 |
| 1306 | 1318 | ? this.target |
| 1307 | 1319 | : this.active.get(parentId) ?? null; |
| ... | ... | @@ -1311,13 +1323,14 @@ class Decoder< |
| 1311 | 1323 | UNWRAP(this.pendingStart.get(parentId)), |
| 1312 | 1324 | ); |
| 1313 | 1325 | } |
| 1314 | const { text, estimatedTime, ...options } = opts; | |
| 1326 | const { text, estimatedTime, formattedValue, ...options } = opts; | |
| 1315 | 1327 | const node = ref.start(text, { |
| 1316 | 1328 | ...options, |
| 1317 | 1329 | estimateCompletion: false, |
| 1318 | 1330 | }); |
| 1319 | 1331 | this.active.set(key, node); |
| 1320 | 1332 | node.estimatedTime = estimatedTime; |
| 1333 | if (formattedValue) node.valueFormatter = () => formattedValue; | |
| 1321 | 1334 | this.pendingParents.delete(key); |
| 1322 | 1335 | return node; |
| 1323 | 1336 | } |
| ... | ... | @@ -1327,6 +1340,9 @@ interface PendingStart { |
| 1327 | 1340 | text: string; |
| 1328 | 1341 | value?: number; |
| 1329 | 1342 | total?: number; |
| 1343 | showTotal?: boolean; | |
| 1344 | passive?: boolean; | |
| 1345 | hidden?: boolean; | |
| 1330 | 1346 | messages?: log.Message[]; |
| 1331 | 1347 | estimatedTime?: number | null; |
| 1332 | 1348 | formattedValue?: string; |
| ... | ... | @@ -1504,21 +1520,27 @@ export function decodeByteStream< |
| 1504 | 1520 | target: Ref, |
| 1505 | 1521 | ): async.Cancelable<Result> & Events<EventMap> { |
| 1506 | 1522 | let cancelled = false; |
| 1523 | let reader: stream.BufferedReader | null = null; | |
| 1507 | 1524 | return decodeEventStream<Result, EventMap>( |
| 1508 | 1525 | new ReadableStream<StreamEvent>({ |
| 1509 | 1526 | async start(controller) { |
| 1510 | using reader = new stream.BufferedReader(encoded.getReader()); | |
| 1527 | reader = new stream.BufferedReader(encoded.getReader()); | |
| 1511 | 1528 | try { |
| 1512 | while (true) controller.enqueue(await readStreamEvent(reader)); | |
| 1529 | while (reader) controller.enqueue(await readStreamEvent(reader)); | |
| 1513 | 1530 | } catch (e) { |
| 1514 | 1531 | if (!cancelled) { |
| 1515 | 1532 | reader.cancel(e); |
| 1516 | 1533 | throw e; |
| 1517 | 1534 | } |
| 1535 | } finally { | |
| 1536 | reader?.releaseLock(); | |
| 1537 | reader = null; | |
| 1518 | 1538 | } |
| 1519 | 1539 | }, |
| 1520 | 1540 | cancel(reason) { |
| 1521 | 1541 | cancelled = true; |
| 1542 | reader?.releaseLock(); | |
| 1543 | reader = null; | |
| 1522 | 1544 | encoded.cancel(reason); |
| 1523 | 1545 | }, |
| 1524 | 1546 | }), |
| ... | ... | @@ -1560,20 +1582,23 @@ async function readStreamEvent(r: stream.BufferedReader): Promise<StreamEvent> { |
| 1560 | 1582 | if (flags.changedLogs) { |
| 1561 | 1583 | l = []; |
| 1562 | 1584 | let len = await r.varUint(); |
| 1563 | while (len > 0) { | |
| 1585 | while (len-- > 0) { | |
| 1564 | 1586 | const msgFlags = await r.u8(); |
| 1565 | 1587 | const level = logLevelSerialize[msgFlags & 0b1111] ?? "info"; |
| 1566 | const hasScope = msgFlags && (1 << 5) > 0; | |
| 1567 | const hasStack = msgFlags && (1 << 6) > 0; | |
| 1568 | const hasCustom = msgFlags && (1 << 7) > 0; | |
| 1588 | // bitwise tests: `msgFlags && (1 << 5) > 0` parsed as | |
| 1589 | // `msgFlags && true`, which fabricated scope/stack/custom reads for | |
| 1590 | // any message with a non-zero level and desynced the entire stream | |
| 1591 | const hasScope = (msgFlags & (1 << 5)) !== 0; | |
| 1592 | const hasStack = (msgFlags & (1 << 6)) !== 0; | |
| 1593 | const hasCustom = (msgFlags & (1 << 7)) !== 0; | |
| 1569 | 1594 | const text = await r.stringWithLength(); |
| 1570 | 1595 | const time = await r.varUint(); |
| 1571 | 1596 | const scope = hasScope ? await r.stringWithLength() : undefined; |
| 1572 | 1597 | let stack: stack.Frame[] | undefined = undefined; |
| 1573 | 1598 | if (hasStack) { |
| 1574 | 1599 | stack = []; |
| 1575 | let len = await r.varUint(); | |
| 1576 | while (len > 0) { | |
| 1600 | let frames = await r.varUint(); | |
| 1601 | while (frames-- > 0) { | |
| 1577 | 1602 | const fn = await r.stringWithLength(); |
| 1578 | 1603 | const file = await r.stringWithLength(); |
| 1579 | 1604 | const line = await r.varUint(); |
| ... | ... | @@ -1756,7 +1781,7 @@ const global: Ref = /** @__PURE__ */ ((root = new Root()) => (attachToScreen(roo |
| 1756 | 1781 | * a {@linkcode Ref} to the global progress root. unlike referencing the |
| 1757 | 1782 | * namespace import, this value is tree-shakable. |
| 1758 | 1783 | */ |
| 1759 | export const globalRoot: Ref = { start: global.start }; | |
| 1784 | export const globalRoot: Ref = { start: (text, opts) => global.start(text, opts) }; | |
| 1760 | 1785 | |
| 1761 | 1786 | /** |
| 1762 | 1787 | * for testing. not covered by semver |
lib/queue.ts+8-1| ... | ... | @@ -176,4 +176,11 @@ function insertSorted<T extends { priority: number }>(arr: T[], item: T) { |
| 176 | 176 | declare var navigator: { hardwareConcurrency: number }; |
| 177 | 177 | const global = new PriorityQueue(navigator.hardwareConcurrency); |
| 178 | 178 | |
| 179 | import { UNWRAP } from "./assert.ts"; | |
| 179 | /** adjust the global queue's concurrency */ | |
| 180 | export function setConcurrency(n: number) { | |
| 181 | ASSERT(Number.isInteger(n) && n > 0, `setConcurrency(${n})`); | |
| 182 | global.coresRemain += n - global.concurrency; | |
| 183 | global.concurrency = n; | |
| 184 | } | |
| 185 | ||
| 186 | import { ASSERT, UNWRAP } from "./assert.ts"; |
lib/subprocess/ffmpeg.ts+3| ... | ... | @@ -11,6 +11,8 @@ export interface SpawnOptions { |
| 11 | 11 | ffmpeg?: string; |
| 12 | 12 | progress: progress.Node; |
| 13 | 13 | cwd?: string; // TODO: Path |
| 14 | /** kills the process when aborted */ | |
| 15 | signal?: AbortSignal; | |
| 14 | 16 | } |
| 15 | 17 | |
| 16 | 18 | /** |
| ... | ... | @@ -34,6 +36,7 @@ export async function spawn(options: SpawnOptions) { |
| 34 | 36 | stdio: ["ignore", "inherit", "pipe"], |
| 35 | 37 | env: { ...process.env, SVT_LOG: "2" }, |
| 36 | 38 | cwd: cwd?.toString(), |
| 39 | signal: options.signal, | |
| 37 | 40 | }); |
| 38 | 41 | const parser = new Parse(); |
| 39 | 42 | let running = true; |
run.js+1| ... | ... | @@ -81,6 +81,7 @@ console["log"] = log.log; |
| 81 | 81 | process.on("uncaughtException", (error) => { |
| 82 | 82 | console.error("Uncaught Exception"); |
| 83 | 83 | console.error(error); |
| 84 | log.getDrawLock("long"); | |
| 84 | 85 | process.exit(1); |
| 85 | 86 | }); |
| 86 | 87 |
src/backend.ts+1| ... | ... | @@ -65,3 +65,4 @@ import { type Context, Hono, type Next } from "hono"; |
| 65 | 65 | import { logger } from "hono/logger"; |
| 66 | 66 | import { trimTrailingSlash } from "hono/trailing-slash"; |
| 67 | 67 | import * as admin from "./admin.ts"; |
| 68 | import "./file-viewer/sync.ts"; |
src/bin/download-db.ts+5-10| ... | ... | @@ -1,14 +1,8 @@ |
| 1 | // Pull production state for local development. cache.sqlite comes from the | |
| 2 | // source of truth's /db route (no ssh needed); questions.sqlite is still | |
| 3 | // rsync'd off the web node since the source of truth does not own it yet. | |
| 1 | 4 | export async function main() { |
| 2 | await rsync.spawn({ | |
| 3 | cwd: Path.resolve("."), | |
| 4 | args: [ | |
| 5 | "-a", | |
| 6 | "--progress", | |
| 7 | "clo@paperclover.net:~/paperclover.net/.clover/cache.sqlite", | |
| 8 | ".clover/cache.sqlite", | |
| 9 | ], | |
| 10 | progress: progress.start("download file cache"), | |
| 11 | }); | |
| 5 | await sync.revalidate(); | |
| 12 | 6 | await rsync.spawn({ |
| 13 | 7 | cwd: Path.resolve("."), |
| 14 | 8 | args: [ |
| ... | ... | @@ -24,3 +18,4 @@ export async function main() { |
| 24 | 18 | import { Path } from "#sitegen/path"; |
| 25 | 19 | import * as progress from "@clo/lib/progress"; |
| 26 | 20 | import * as rsync from "../file-viewer/rsync.ts"; |
| 21 | import * as sync from "../file-viewer/sync.ts"; |
src/bin/tail-progress.ts created+50| ... | ... | @@ -0,0 +1,50 @@ |
| 1 | // Tail the source of truth's indexing progress from anywhere: | |
| 2 | // | |
| 3 | // node run tail-progress # production | |
| 4 | // node run tail-progress http://zenith:43201/progress # staging | |
| 5 | // | |
| 6 | // Opens a fetch to the /progress route and renders the live progress tree | |
| 7 | // in this terminal. Reconnects automatically; each (re)connect resumes from | |
| 8 | // the server's current state, since the stream encoder snapshots on attach. | |
| 9 | const defaultUrl = "https://db.paperclover.net/progress"; | |
| 10 | ||
| 11 | export async function main() { | |
| 12 | const url = process.argv[2] ?? defaultUrl; | |
| 13 | const token = process.env.CLOVER_SOT_KEY; | |
| 14 | if (!token) { | |
| 15 | console.warn("CLOVER_SOT_KEY is not set; the server will likely 401"); | |
| 16 | } | |
| 17 | ||
| 18 | let delay = 1000; | |
| 19 | while (true) { | |
| 20 | const connectedAt = Date.now(); | |
| 21 | try { | |
| 22 | const res = await fetch(url, { | |
| 23 | headers: token ? { Authorization: token } : {}, | |
| 24 | }); | |
| 25 | if (res.status === 401) { | |
| 26 | console.error("unauthorized; set CLOVER_SOT_KEY"); | |
| 27 | process.exit(1); | |
| 28 | } | |
| 29 | if (!res.ok || !res.body) { | |
| 30 | throw new Error(`server responded ${res.status} ${res.statusText}`); | |
| 31 | } | |
| 32 | console.info(`connected to ${url}`); | |
| 33 | // resolves only when the stream ends; the indexer root never ends, so | |
| 34 | // this await effectively lasts until disconnect. | |
| 35 | await progress.decodeByteStream( | |
| 36 | res.body as ReadableStream<Uint8Array>, | |
| 37 | progress.globalRoot, | |
| 38 | ); | |
| 39 | } catch (err: any) { | |
| 40 | console.warn(`disconnected: ${err?.message ?? err}`); | |
| 41 | } | |
| 42 | if (Date.now() - connectedAt > 15_000) delay = 1000; | |
| 43 | else delay = Math.min(delay * 2, 30_000); | |
| 44 | console.info(`reconnecting in ${Math.round(delay / 1000)}s...`); | |
| 45 | await async.delay(delay); | |
| 46 | } | |
| 47 | } | |
| 48 | ||
| 49 | import * as async from "@clo/lib/async"; | |
| 50 | import * as progress from "@clo/lib/progress"; |
src/file-viewer/backend.ts+3-1| ... | ... | @@ -83,7 +83,9 @@ app.get("/file/*", async (c, next) => { |
| 83 | 83 | // - Old browsers like Internet Explorer act as `?view=dl` |
| 84 | 84 | let viewMode = c.req.query("view"); |
| 85 | 85 | if (c.req.query("dl") != null) viewMode = "download"; |
| 86 | if (lofi) viewMode = file.extension === ".html" ? "embed" : "download"; | |
| 86 | if (lofi) { | |
| 87 | viewMode = file.extension.toLowerCase() === ".html" ? "embed" : "download"; | |
| 88 | } | |
| 87 | 89 | |
| 88 | 90 | if ( |
| 89 | 91 | viewMode == null && !derivedKey |
src/file-viewer/bin/file-scan.ts+36-999| ... | ... | @@ -1,408 +1,32 @@ |
| 1 | // The file scanner incrementally updates an sqlite database with file | |
| 2 | // stats. Additionally, it runs "processors" on files, which precompute | |
| 3 | // expensive data such as running `ffprobe` on all media to get the | |
| 4 | // duration. | |
| 1 | // Run one full scan of the file store from this machine, rendering progress | |
| 2 | // in the terminal. In production the indexer service inside the source of | |
| 3 | // truth server does this continuously; this wrapper exists for development | |
| 4 | // (and the nostalgia of watching it go). | |
| 5 | 5 | // |
| 6 | // Processors are also used to derive compressed and optimized assets, | |
| 7 | // which is how automatic JXL / AV1 encoding is done. Derived files are | |
| 8 | // uploaded to the clover NAS to be pulled by VPS instances for hosting. | |
| 9 | const sotToken = process.env.CLOVER_SOT_KEY; | |
| 10 | ||
| 6 | // CLOVER_FILE_RAW=/tmp/store CLOVER_FILE_DERIVED=/tmp/derived \ | |
| 7 | // CLOVER_DB=.clover node run file-scan | |
| 11 | 8 | export async function main() { |
| 12 | 9 | const start = performance.now(); |
| 13 | 10 | using _ = log.startWidget({ |
| 14 | 11 | format: ({ now }) => `paper clover's file scanner [${((now - start) / 1000).toFixed(1)}s]`, |
| 15 | 12 | }); |
| 16 | 13 | |
| 17 | const promises = new async.PromiseAggregator(); | |
| 18 | ||
| 19 | const walkQueue = new queue.PriorityQueue(10); | |
| 20 | ||
| 21 | const dirsNode = progress.start("Walk Tree", { total: 1 }); | |
| 22 | dirsNode.passive = true; | |
| 23 | const fileNode = progress.start("Process File", { total: 0 }); | |
| 24 | fileNode.sortChildren = (a, b) => { | |
| 25 | const ac = a.children.length > 0 ? 1 : 0; | |
| 26 | const bc = b.children.length > 0 ? 1 : 0; | |
| 27 | if (ac !== bc) return bc - ac; | |
| 28 | return a.text.localeCompare(b.text); | |
| 29 | }; | |
| 30 | ||
| 31 | // Read a directory or file stat and queue up changed files. | |
| 32 | const scanDirectory = walkQueue.wrap( | |
| 33 | async (path: Path) => { | |
| 34 | const publicPath = toPublicPath(path); | |
| 35 | using node = dirsNode.start(publicPath + " - stat"); | |
| 36 | using _ = ts.defer(() => dirsNode.inc()); | |
| 37 | ||
| 38 | const stat = await path.stat(); | |
| 39 | ||
| 40 | const mediaFile = MediaFile.getByPath(publicPath); | |
| 41 | ||
| 42 | if (stat.isDirectory()) { | |
| 43 | node.text = publicPath + " - reading"; | |
| 44 | const items = (await path.readDir()) | |
| 45 | .filter((child) => !skipBasename(child.base)) | |
| 46 | .map((child) => (promises.push(scanDirectory(child)), child.base)); | |
| 47 | dirsNode.total += items.length; | |
| 48 | ||
| 49 | for (const child of mediaFile?.getChildren() ?? []) { | |
| 50 | if (items.includes(child.basename)) continue; | |
| 51 | const recursive = child.kind === MediaFileKind.directory | |
| 52 | ? [child, ...child.getRecursiveFileChildren()] | |
| 53 | : [child]; | |
| 54 | for (const deletion of recursive) deletion.delete(); | |
| 55 | } | |
| 56 | ||
| 57 | return; | |
| 58 | } | |
| 59 | ||
| 60 | if ( | |
| 61 | !mediaFile // All processes must be performed if there is no file. | |
| 62 | // Rerun all processors if it changed | |
| 63 | || stat.size !== mediaFile.size | |
| 64 | || stat.mtime.getTime() !== mediaFile.date.getTime() | |
| 65 | ) { | |
| 66 | promises.push(updateMetadata({ path, publicPath, stat, mediaFile })); | |
| 67 | } else { | |
| 68 | // If the scanners changed, it may mean more processes should be run. | |
| 69 | await queueProcessors({ | |
| 70 | path, | |
| 71 | stat, | |
| 72 | mediaFile, | |
| 73 | node: fileNode.start(publicPath.slice(1)), | |
| 74 | }); | |
| 75 | } | |
| 76 | }, | |
| 77 | ); | |
| 78 | const updateMetadata = queue.wrap( | |
| 79 | async ({ path, publicPath, stat, mediaFile }: UpdateMetadataArgs) => { | |
| 80 | using errorHandler = new DisposableStack(); | |
| 81 | const label = publicPath.slice(1); | |
| 82 | const node = errorHandler.use(fileNode.start(label)); | |
| 83 | ||
| 84 | await scrubLocationMetadata(path, stat, node); | |
| 85 | ||
| 86 | node.text = `${label} - hashing`; | |
| 87 | const hash = await new Promise<string>((resolve, reject) => { | |
| 88 | const reader = fs.createReadStream(path.toString()); | |
| 89 | reader.on("error", reject); | |
| 90 | ||
| 91 | const hasher = crypto.createHash("sha1").setEncoding("hex"); | |
| 92 | hasher.on("error", reject); | |
| 93 | hasher.on("readable", () => resolve(hasher.read())); | |
| 94 | ||
| 95 | reader.pipe(hasher); | |
| 96 | }); | |
| 97 | let date = stat.mtime; | |
| 98 | if ( | |
| 99 | mediaFile | |
| 100 | && mediaFile.date.getTime() < stat.mtime.getTime() | |
| 101 | && Date.now() - stat.mtime.getTime() < monthMilliseconds | |
| 102 | ) { | |
| 103 | date = mediaFile.date; | |
| 104 | console.warn( | |
| 105 | `M-time on ${publicPath} was likely corrupted. ${ | |
| 106 | formatDate( | |
| 107 | mediaFile.date, | |
| 108 | ) | |
| 109 | } -> ${formatDate(stat.mtime)}`, | |
| 110 | ); | |
| 111 | } | |
| 112 | mediaFile = MediaFile.createFile({ | |
| 113 | path: publicPath, | |
| 114 | date, | |
| 115 | hash, | |
| 116 | size: stat.size, | |
| 117 | duration: mediaFile?.duration ?? 0, | |
| 118 | dimensions: mediaFile?.dimensions ?? "", | |
| 119 | contents: mediaFile?.contents ?? "", | |
| 120 | }); | |
| 121 | const parent = mediaFile.getParent(); | |
| 122 | if (parent) parent.setProcessed(0); | |
| 123 | ||
| 124 | node.text = `${label}`; | |
| 125 | await queueProcessors({ | |
| 126 | path, | |
| 127 | stat, | |
| 128 | mediaFile, | |
| 129 | node, | |
| 130 | }); | |
| 131 | errorHandler.move(); | |
| 132 | }, | |
| 133 | () => ({ priority: -1 }), | |
| 134 | ); | |
| 135 | const queueFileProcessor = queue.wrap(async function({ | |
| 136 | path, | |
| 137 | stat, | |
| 138 | mediaFile, | |
| 139 | processor, | |
| 140 | index, | |
| 141 | after, | |
| 142 | fileNode: innerNode, | |
| 143 | }: ProcessJob) { | |
| 144 | using node = innerNode.start(processor.name); | |
| 145 | await processor.run({ | |
| 146 | path, | |
| 147 | stat, | |
| 148 | mediaFile, | |
| 149 | node, | |
| 150 | }); | |
| 151 | fileNode.value += 1; | |
| 152 | mediaFile.setProcessed(mediaFile.processed | (1 << (16 + index))); | |
| 153 | for (const dependantJob of after) { | |
| 154 | ASSERT( | |
| 155 | dependantJob.needs > 0, | |
| 156 | `dependantJob.needs > 0, ${dependantJob.needs}`, | |
| 157 | ); | |
| 158 | dependantJob.needs -= 1; | |
| 159 | if (dependantJob.needs == 0) { | |
| 160 | promises.push(queueFileProcessor(dependantJob)); | |
| 161 | } | |
| 162 | } | |
| 163 | }, (job) => ({ cores: job.processor.cores ?? 0 })); | |
| 164 | ||
| 165 | function decodeProcessors(input: string) { | |
| 166 | return input | |
| 167 | .split(";") | |
| 168 | .filter(Boolean) | |
| 169 | .map(([a, b, c]) => ({ | |
| 170 | id: a, | |
| 171 | hash: (UNWRAP(b).charCodeAt(0) << 8) + UNWRAP(c).charCodeAt(0), | |
| 172 | })); | |
| 173 | } | |
| 174 | ||
| 175 | async function queueProcessors({ | |
| 176 | path, | |
| 177 | stat, | |
| 178 | mediaFile, | |
| 179 | node, | |
| 180 | }: Omit<ProcessFileArgs, "spin">) { | |
| 181 | using errorHandler = new DisposableStack(); | |
| 182 | errorHandler.use(node); | |
| 183 | ||
| 184 | node.showTotal = false; | |
| 185 | node.passive = true; | |
| 186 | ||
| 187 | const ext = mediaFile.extensionNonEmpty.toLowerCase(); | |
| 188 | let possible = processors.filter((p) => p.include ? p.include.has(ext) : !p.exclude?.has(ext)); | |
| 189 | if (possible.length === 0) return; | |
| 190 | ASSERT(possible.length < 16, "too many bits"); | |
| 191 | ||
| 192 | const hash = possible.reduce((a, b) => a ^ b.hash, 0) | 1; | |
| 193 | ASSERT(hash <= 0xffff, `${hash.toString(16)} has no bits above 16 set`); | |
| 194 | let processed = mediaFile.processed; | |
| 195 | ||
| 196 | // If the hash has changed, migrate the bitfield over. | |
| 197 | // This also runs when the processor hash is in it's initial 0 state. | |
| 198 | let order: ReturnType<typeof decodeProcessors>; | |
| 199 | try { | |
| 200 | order = decodeProcessors(mediaFile.processors); | |
| 201 | } catch { | |
| 202 | // this function sucks and this system sucks i hate it. | |
| 203 | order = []; | |
| 204 | } | |
| 205 | if ((processed & 0xffff) !== hash) { | |
| 206 | const previous = order.filter( | |
| 207 | (_, i) => (processed & (1 << (16 + i))) !== 0, | |
| 208 | ); | |
| 209 | processed = hash; | |
| 210 | for (const { id, hash } of previous) { | |
| 211 | const p = processors.find((p) => p.id === id); | |
| 212 | if (!p) continue; | |
| 213 | const index = possible.indexOf(p); | |
| 214 | if (index !== -1 && p.hash === hash) processed |= 1 << (16 + index); | |
| 215 | } | |
| 216 | mediaFile.setProcessors( | |
| 217 | processed, | |
| 218 | possible | |
| 219 | .map((p) => p.id + String.fromCharCode(p.hash >> 8, p.hash & 0xff)) | |
| 220 | .join(";"), | |
| 221 | ); | |
| 222 | } else { | |
| 223 | possible = order.map(({ id }) => UNWRAP(possible.find((p) => p.id === id))); | |
| 224 | } | |
| 225 | ||
| 226 | // Queue needed processors. | |
| 227 | const jobs: ProcessJob[] = []; | |
| 228 | for (let i = 0, { length } = possible; i < length; i += 1) { | |
| 229 | if ((processed & (1 << (16 + i))) === 0) { | |
| 230 | const processor = UNWRAP(possible[i]); | |
| 231 | const job: ProcessJob = { | |
| 232 | path, | |
| 233 | stat, | |
| 234 | mediaFile, | |
| 235 | processor, | |
| 236 | index: i, | |
| 237 | after: [], | |
| 238 | needs: processor.depends.length, | |
| 239 | fileNode: node, | |
| 240 | }; | |
| 241 | jobs.push(job); | |
| 242 | if (job.needs === 0) promises.push(queueFileProcessor(job)); | |
| 243 | } | |
| 244 | } | |
| 245 | node.total = jobs.length; | |
| 246 | fileNode.total += jobs.length; | |
| 247 | for (const job of jobs) { | |
| 248 | for (const dependId of job.processor.depends) { | |
| 249 | const dependJob = jobs.find((j) => j.processor.id === dependId); | |
| 250 | if (dependJob) { | |
| 251 | dependJob.after.push(job); | |
| 252 | } else { | |
| 253 | ASSERT(job.needs > 0, `job.needs !== 0, ${job.needs}`); | |
| 254 | job.needs -= 1; | |
| 255 | if (job.needs === 0) promises.push(queueFileProcessor(job)); | |
| 256 | } | |
| 257 | } | |
| 258 | } | |
| 259 | if (node.total > 0) { | |
| 260 | errorHandler.move(); | |
| 261 | ||
| 262 | const parent = mediaFile.getParent(); | |
| 263 | if (parent) parent.setProcessed(0); | |
| 264 | } | |
| 265 | } | |
| 266 | ||
| 267 | // Add the root & recursively iterate! | |
| 268 | const rootPath = Path.resolve(root); | |
| 269 | if (!rootPath.ifExistsSync()) { | |
| 270 | throw new Error(`file store ${rootPath} is not mounted`); | |
| 271 | } | |
| 272 | promises.push(scanDirectory(rootPath)); | |
| 273 | ||
| 274 | await promises.all(); | |
| 275 | fileNode.end(); | |
| 276 | dirsNode.end(); | |
| 277 | ||
| 278 | // Update directory metadata | |
| 279 | using dirMetaNode = progress.start("update directory metadata"); | |
| 280 | const dirs = MediaFile.getDirectoriesToReindex() | |
| 281 | .sort((a, b) => b.path.length - a.path.length); | |
| 282 | ||
| 283 | for (const dir of dirs) { | |
| 284 | using _ = dirMetaNode.start(dir.path); | |
| 285 | const children = dir.getChildren(); | |
| 286 | ||
| 287 | // readme.txt | |
| 288 | const readmeContent = children.find((x) => x.basename === "readme.txt")?.contents ?? ""; | |
| 289 | ||
| 290 | // dirsort | |
| 291 | let dirsort: string[] | null = null; | |
| 292 | const dirSortRaw = children.find((x) => x.basename === ".dirsort")?.contents ?? ""; | |
| 293 | if (dirSortRaw) { | |
| 294 | dirsort = dirSortRaw | |
| 295 | .split("\n") | |
| 296 | .map((x) => x.trim()) | |
| 297 | .filter(Boolean); | |
| 298 | } | |
| 299 | ||
| 300 | // Permissions | |
| 301 | if (children.some((x) => x.basename === ".friends")) { | |
| 302 | FilePermissions.setPermissions(dir.path, 1); | |
| 303 | } else { | |
| 304 | FilePermissions.setPermissions(dir.path, 0); | |
| 305 | } | |
| 306 | ||
| 307 | // Recursive stats. | |
| 308 | let totalSize = 0; | |
| 309 | let newestDate = new Date(0); | |
| 310 | let allHashes = ""; | |
| 311 | for (const child of children) { | |
| 312 | totalSize += child.size; | |
| 313 | allHashes += child.hash; | |
| 314 | ||
| 315 | if (child.basename !== "/readme.txt" && child.date > newestDate) { | |
| 316 | newestDate = child.date; | |
| 317 | } | |
| 318 | } | |
| 319 | ||
| 320 | // Project Date | |
| 321 | const dateFile = children.find((x) => x.basename === ".date"); | |
| 322 | if (dateFile) { | |
| 323 | const date = new Date(dateFile.contents); | |
| 324 | newestDate = date; | |
| 325 | dir.setProcessors( | |
| 326 | 0, | |
| 327 | JSON.stringify({ hideChildrenDates: true }), | |
| 328 | ); | |
| 329 | } else { | |
| 330 | dir.setProcessors(0, ""); | |
| 331 | } | |
| 332 | ||
| 333 | const dirHash = crypto | |
| 334 | .createHash("sha1") | |
| 335 | .update(dir.path + allHashes) | |
| 336 | .digest("hex"); | |
| 337 | ||
| 338 | MediaFile.markDirectoryProcessed({ | |
| 339 | id: dir.id, | |
| 340 | timestamp: newestDate, | |
| 341 | contents: readmeContent, | |
| 342 | size: totalSize, | |
| 343 | hash: dirHash, | |
| 344 | dirsort, | |
| 345 | }); | |
| 346 | } | |
| 347 | dirMetaNode.end(); | |
| 348 | ||
| 349 | // Sync to remote | |
| 350 | if ( | |
| 351 | ((await derived.workDir.ifExistsSync()?.readDir())?.length ?? 0) > 0 | |
| 352 | ) { | |
| 353 | await rsync.spawn({ | |
| 354 | args: [ | |
| 355 | "--links", | |
| 356 | "--recursive", | |
| 357 | "--times", | |
| 358 | "--partial", | |
| 359 | "--progress", | |
| 360 | // "--remove-source-files", | |
| 361 | "--delay-updates", | |
| 362 | "--exclude=tmp.*", | |
| 363 | derived.workDir.toString() + "/", | |
| 364 | "clo@file.paperclover.net:/mnt/storage1/clover/Documents/Config/paperclover/derived/", | |
| 365 | ], | |
| 366 | progress: progress.start("upload derived assets"), | |
| 367 | cwd: process.cwd(), | |
| 368 | }); | |
| 369 | ||
| 370 | await fs.removeEmptyDirectories(derived.workDir.toString()); | |
| 371 | } else { | |
| 372 | console.info("No new derived assets"); | |
| 373 | } | |
| 374 | ||
| 375 | MediaFile.db.prepare("VACUUM").run(); | |
| 376 | MediaFile.db.reload(); | |
| 377 | ||
| 378 | await rsync.spawn({ | |
| 379 | args: [ | |
| 380 | MediaFile.db.file, | |
| 381 | "clo@file.paperclover.net:/mnt/storage1/clover/Documents/Config/paperclover/cache.sqlite", | |
| 382 | ], | |
| 383 | progress: progress.start("Uploading database (source of truth)"), | |
| 384 | cwd: process.cwd(), | |
| 14 | const scanner = new Scanner({ | |
| 15 | root: Path.resolve(rawFileRoot), | |
| 16 | progress: progress.globalRoot, | |
| 17 | settleMs: Number(process.env.CLOVER_SETTLE_MS ?? 10_000), | |
| 385 | 18 | }); |
| 386 | await rsync.spawn({ | |
| 387 | args: [ | |
| 388 | MediaFile.db.file, | |
| 389 | "clo@paperclover.net:paperclover.net/.clover/cache.sqlite", | |
| 390 | ], | |
| 391 | progress: progress.start("Uploading database (web node)"), | |
| 392 | }); | |
| 393 | if (sotToken) { | |
| 394 | const res = await fetch("https://db.paperclover.net/reload", { | |
| 395 | method: "POST", | |
| 396 | headers: { | |
| 397 | Authorization: sotToken, | |
| 398 | }, | |
| 399 | }); | |
| 400 | if (!res.ok) { | |
| 401 | console.warn( | |
| 402 | `Failed to reload remote database ${res.status} ${res.statusText}`, | |
| 403 | ); | |
| 404 | } | |
| 405 | } else console.warn("Missing SOT token"); | |
| 19 | await scanner.sweep(); | |
| 20 | await scanner.waitIdle(); | |
| 21 | dirmeta.run(progress.globalRoot); | |
| 22 | ||
| 23 | const orphaned = derived.findOrphanedRoots(); | |
| 24 | for (const orphan of orphaned) { | |
| 25 | console.info("delete orphaned " + orphan.key); | |
| 26 | await derived.deleteRootFiles(orphan); | |
| 27 | derived.deleteRoot(orphan); | |
| 28 | } | |
| 29 | await derived.cleanAbandonedTmp(); | |
| 406 | 30 | |
| 407 | 31 | console.info( |
| 408 | 32 | "Updated file viewer index in \x1b[1m" |
| ... | ... | @@ -426,6 +50,11 @@ export async function main() { |
| 426 | 50 | from derived_files |
| 427 | 51 | `) |
| 428 | 52 | .getNonNull(); |
| 53 | const { failed } = MediaFile.db | |
| 54 | .prepare<[], { failed: number }>(` | |
| 55 | select count(*) as failed from file_processors where status = 2 | |
| 56 | `) | |
| 57 | .getNonNull(); | |
| 429 | 58 | const canonicalSize = UNWRAP(MediaFile.getByPath("/")).size; |
| 430 | 59 | |
| 431 | 60 | console.info(); |
| ... | ... | @@ -435,614 +64,22 @@ export async function main() { |
| 435 | 64 | + `- Derived Count: \x1b[1m${derivedCount}\x1b[0m\n` |
| 436 | 65 | + `- Media Duration: \x1b[1m${string.formatDurationLetters(duration)}\x1b[0m\n` |
| 437 | 66 | + `- Canonical Size: \x1b[1m${string.formatByteSize(canonicalSize)}\x1b[0m\n` |
| 438 | + `- Derived Size: \x1b[1m${string.formatByteSize(derivedSize)}\x1b[0m\n`, | |
| 439 | ); | |
| 440 | } | |
| 441 | ||
| 442 | interface Process { | |
| 443 | name: string; | |
| 444 | cores?: number; | |
| 445 | enable?: boolean; | |
| 446 | include?: Set<string>; | |
| 447 | exclude?: Set<string>; | |
| 448 | depends?: string[]; | |
| 449 | version?: number; | |
| 450 | /* Perform an action. */ | |
| 451 | run(args: ProcessFileArgs): Promise<void>; | |
| 452 | } | |
| 453 | ||
| 454 | const ffprobeBin = testProgram("ffprobe", "--help"); | |
| 455 | const ffmpegBin = testProgram("ffmpeg", "--help"); | |
| 456 | ||
| 457 | const ffmpegOptions = ["-hide_banner", "-loglevel", "warning"]; | |
| 458 | ||
| 459 | // NOTE: Never re-order the processors. Add new ones at the end. | |
| 460 | const procDuration: Process = { | |
| 461 | name: "calculate duration", | |
| 462 | enable: ffprobeBin !== null, | |
| 463 | include: rules.extsDuration, | |
| 464 | cores: 1, | |
| 465 | async run({ path, mediaFile }) { | |
| 466 | const { stdout } = await subprocess.exec(ffprobeBin!, [ | |
| 467 | "-v", | |
| 468 | "error", | |
| 469 | "-show_entries", | |
| 470 | "format=duration", | |
| 471 | "-of", | |
| 472 | "default=noprint_wrappers=1:nokey=1", | |
| 473 | path.toString(), | |
| 474 | ]); | |
| 475 | ||
| 476 | const duration = parseFloat(stdout.trim()); | |
| 477 | if (Number.isNaN(duration)) { | |
| 478 | throw new Error("Could not extract duration from " + stdout); | |
| 479 | } | |
| 480 | mediaFile.setDuration(Math.ceil(duration)); | |
| 481 | }, | |
| 482 | }; | |
| 483 | ||
| 484 | const procDimensions: Process = { | |
| 485 | name: "calculate dimensions", | |
| 486 | enable: ffprobeBin != null, | |
| 487 | include: rules.extsDimensions, | |
| 488 | cores: 1, | |
| 489 | async run({ path, mediaFile }) { | |
| 490 | const { ext } = path; | |
| 491 | ||
| 492 | let dimensions; | |
| 493 | ||
| 494 | if (ext === ".svg") { | |
| 495 | // Parse out of text data | |
| 496 | const content = await path.read("utf-8"); | |
| 497 | const widthMatch = content.match(/width="(\d+)"/); | |
| 498 | const heightMatch = content.match(/height="(\d+)"/); | |
| 499 | ||
| 500 | if (widthMatch && heightMatch) { | |
| 501 | dimensions = `${widthMatch[1]}x${heightMatch[1]}`; | |
| 502 | } | |
| 503 | } else if (rules.extsImage.has(ext)) { | |
| 504 | // Use magick to observe streams | |
| 505 | const { stdout } = await subprocess.exec("magick", [ | |
| 506 | "identify", | |
| 507 | "-auto-orient", | |
| 508 | "-format", | |
| 509 | "%w %h", | |
| 510 | path.toString(), | |
| 511 | ]); | |
| 512 | const [w, h] = stdout.split(" ").map((x) => Number(x)); | |
| 513 | if (w && h) { | |
| 514 | dimensions = w + "x" + h; | |
| 515 | } | |
| 516 | } else { | |
| 517 | // Use ffprobe to observe streams | |
| 518 | const { stdout } = await subprocess.exec("ffprobe", [ | |
| 519 | "-v", | |
| 520 | "error", | |
| 521 | "-select_streams", | |
| 522 | "v:0", | |
| 523 | "-show_entries", | |
| 524 | "stream=width,height", | |
| 525 | "-of", | |
| 526 | "json", | |
| 527 | path.toString(), | |
| 528 | ]); | |
| 529 | const result = JSON.parse(stdout); | |
| 530 | const stream = result.streams[0]; | |
| 531 | if (stream) { | |
| 532 | dimensions = UNWRAP(stream.width) + "x" + UNWRAP(stream.height); | |
| 533 | } | |
| 534 | } | |
| 535 | ||
| 536 | mediaFile.setDimensions(dimensions ?? ""); | |
| 537 | }, | |
| 538 | }; | |
| 539 | ||
| 540 | const procLoadTextContents: Process = { | |
| 541 | name: "load text content", | |
| 542 | include: rules.extsReadContents, | |
| 543 | cores: 1, | |
| 544 | version: 2, | |
| 545 | async run({ path, mediaFile, stat }) { | |
| 546 | if (stat.size > 1_000_000) return; | |
| 547 | const text = await path.read("utf-8"); | |
| 548 | mediaFile.setContents(text); | |
| 549 | }, | |
| 550 | }; | |
| 551 | ||
| 552 | const procHighlightCode: Process = { | |
| 553 | name: "highlight source code", | |
| 554 | include: new Set(rules.extsCode.keys()), | |
| 555 | cores: 1, | |
| 556 | version: 2, | |
| 557 | async run({ path, mediaFile, stat }) { | |
| 558 | const language = UNWRAP( | |
| 559 | rules.extsCode.get(path.ext.toLowerCase()), | |
| 560 | ); | |
| 561 | // An issue is that .ts is an overloaded extension, shared between | |
| 562 | // 'transport stream' and 'typescript'. | |
| 563 | // | |
| 564 | // Filter used here is: | |
| 565 | // - more than 1mb | |
| 566 | // - invalid UTF-8 | |
| 567 | if (stat.size > 1_000_000) return; | |
| 568 | let code; | |
| 569 | const buf = await path.read(); | |
| 570 | try { | |
| 571 | code = new TextDecoder("utf-8", { fatal: true }).decode(buf); | |
| 572 | } catch (error) { | |
| 573 | mediaFile.setContents(""); | |
| 574 | return; | |
| 575 | } | |
| 576 | const content = await highlight.highlightCode(code, language); | |
| 577 | mediaFile.setContents(content); | |
| 578 | }, | |
| 579 | }; | |
| 580 | ||
| 581 | const procImageSubsets: Process = { | |
| 582 | name: "encode image subsets", | |
| 583 | include: rules.extsImage, | |
| 584 | depends: [procDimensions.name], | |
| 585 | version: 3, | |
| 586 | async run({ path, mediaFile, node }) { | |
| 587 | const { width, height } = UNWRAP(mediaFile.parseDimensions()); | |
| 588 | const targetSizes = transcodeRules.imageSizes.filter((w) => w < width); | |
| 589 | const baseStatus = node.text; | |
| 590 | ||
| 591 | using stack = new DisposableStack(); | |
| 592 | for (const size of targetSizes) { | |
| 593 | const { w, h } = resizeDimensions(width, height, size); | |
| 594 | for (const { ext, args } of transcodeRules.imagePresets) { | |
| 595 | node.text = baseStatus + ` (${w}x${h}, ${ext.slice(1).toUpperCase()})`; | |
| 596 | ||
| 597 | stack.use( | |
| 598 | await derived.produce({ | |
| 599 | mediaFile, | |
| 600 | node, | |
| 601 | cores: 2, | |
| 602 | subkey: `${size}${ext}`, | |
| 603 | async producer(dir) { | |
| 604 | await subprocess.exec(ffmpegBin!, [ | |
| 605 | ...ffmpegOptions, | |
| 606 | "-i", | |
| 607 | path.toString(), | |
| 608 | "-vf", | |
| 609 | `scale=${w}:${h}:force_original_aspect_ratio=increase,crop=${w}:${h}`, | |
| 610 | ...args, | |
| 611 | dir.join(`${size}${ext}`).toString(), | |
| 612 | ]); | |
| 613 | }, | |
| 614 | }), | |
| 615 | ); | |
| 616 | } | |
| 617 | } | |
| 618 | ||
| 619 | stack.move(); | |
| 620 | }, | |
| 621 | }; | |
| 622 | ||
| 623 | const videoArgsCache = new async.OnceMap(transcodeRules.getVideoInputArgs); | |
| 624 | ||
| 625 | const qualityMap: Record<string, string> = { | |
| 626 | u: "ultra-high", | |
| 627 | h: "high", | |
| 628 | m: "medium", | |
| 629 | l: "low", | |
| 630 | d: "data-saving", | |
| 631 | }; | |
| 632 | const procVideos = transcodeRules.videoFormats.map<Process>((preset) => ({ | |
| 633 | name: `encode av1 ${UNWRAP(qualityMap[UNWRAP(preset.id[1])])}`, | |
| 634 | include: rules.extsVideo, | |
| 635 | enable: ffmpegBin != null, | |
| 636 | depends: [procDuration.name, procDimensions.name], | |
| 637 | version: 4, | |
| 638 | async run({ path, mediaFile, node }) { | |
| 639 | if ((mediaFile.duration ?? 0) < 5) return; | |
| 640 | if (!mediaFile.dimensions) return; | |
| 641 | ||
| 642 | if (mediaFile.path === "/2021/top-10000-bread/output.mp4") return; | |
| 643 | ||
| 644 | await derived.produce({ | |
| 645 | mediaFile, | |
| 646 | node, | |
| 647 | cores: 4, | |
| 648 | subkey: `av1-${preset.id}`, | |
| 649 | async producer(dir) { | |
| 650 | const input = await videoArgsCache.getOrRun(path); | |
| 651 | const dimensions = mediaFile.parseDimensions(); | |
| 652 | ASSERT(dimensions); | |
| 653 | const args = transcodeRules.getAv1VideoArgs( | |
| 654 | preset, | |
| 655 | dimensions, | |
| 656 | UNWRAP(input.video, "frick on " + path + JSON.stringify(input)), | |
| 657 | dir, | |
| 658 | ); | |
| 659 | ||
| 660 | await ffmpeg.spawn({ | |
| 661 | ffmpeg: ffmpegBin!, | |
| 662 | progress: node, | |
| 663 | args, | |
| 664 | cwd: dir.toString(), | |
| 665 | }); | |
| 666 | }, | |
| 667 | }); | |
| 668 | }, | |
| 669 | })); | |
| 670 | const procVideoAudios = transcodeRules.audioFormats.map<Process>((preset) => ({ | |
| 671 | name: `encode opus ${UNWRAP(qualityMap[UNWRAP(preset.id)])}`, | |
| 672 | include: rules.extsVideo, | |
| 673 | enable: ffmpegBin != null, | |
| 674 | depends: [procDuration.name, procDimensions.name], | |
| 675 | version: 3, | |
| 676 | async run({ path, mediaFile, node }) { | |
| 677 | if ((mediaFile.duration ?? 0) < 5) return; | |
| 678 | if (!mediaFile.dimensions) return; | |
| 679 | if (mediaFile.path === "/2021/top-10000-bread/output.mp4") return; | |
| 680 | await derived.produce({ | |
| 681 | mediaFile, | |
| 682 | node, | |
| 683 | cores: 4, | |
| 684 | subkey: `opus-${preset.id}`, | |
| 685 | async producer(dir) { | |
| 686 | const input = await videoArgsCache.getOrRun(path); | |
| 687 | ASSERT(input.video); | |
| 688 | if (!input.audio) return; | |
| 689 | const args = transcodeRules.getOpusAudioArgs(preset, input.audio, dir); | |
| 690 | ||
| 691 | await ffmpeg.spawn({ | |
| 692 | ffmpeg: ffmpegBin!, | |
| 693 | progress: node, | |
| 694 | args, | |
| 695 | cwd: dir.toString(), | |
| 696 | }); | |
| 697 | }, | |
| 698 | }); | |
| 699 | }, | |
| 700 | })); | |
| 701 | const procDash: Process = { | |
| 702 | name: `encode mpeg-dash`, | |
| 703 | include: rules.extsVideo, | |
| 704 | enable: ffmpegBin != null, | |
| 705 | depends: [...procVideos, ...procVideoAudios].map((x) => x.name), | |
| 706 | version: 7, | |
| 707 | async run({ path, mediaFile, node }) { | |
| 708 | if ((mediaFile.duration ?? 0) < 5) return; | |
| 709 | if (!mediaFile.dimensions) return; | |
| 710 | ||
| 711 | if (mediaFile.path === "/2021/top-10000-bread/output.mp4") return; | |
| 712 | await derived.produce({ | |
| 713 | mediaFile, | |
| 714 | node, | |
| 715 | cores: 1, | |
| 716 | subkey: `dash-av1`, | |
| 717 | async producer(dir) { | |
| 718 | const input = await videoArgsCache.getOrRun(path); | |
| 719 | const videos = transcodeRules.videoFormats.map( | |
| 720 | (preset) => | |
| 721 | UNWRAP(dir.parent).join( | |
| 722 | `av1-${preset.id}`, | |
| 723 | transcodeRules.av1FileName, | |
| 724 | ), | |
| 725 | ); | |
| 726 | const audios = input.audio | |
| 727 | ? transcodeRules.audioFormats.map( | |
| 728 | (preset) => | |
| 729 | UNWRAP(dir.parent).join( | |
| 730 | `opus-${preset.id}`, | |
| 731 | transcodeRules.opusFileName, | |
| 732 | ), | |
| 733 | ) | |
| 734 | : []; | |
| 735 | const args = transcodeRules.getMpegDashArgs( | |
| 736 | videos, | |
| 737 | audios, | |
| 738 | dir, | |
| 739 | ); | |
| 740 | ||
| 741 | await dir.join("d").makeDir(); | |
| 742 | ||
| 743 | await ffmpeg.spawn({ | |
| 744 | ffmpeg: ffmpegBin!, | |
| 745 | progress: node, | |
| 746 | args, | |
| 747 | cwd: dir.toString(), | |
| 748 | }); | |
| 749 | }, | |
| 750 | }); | |
| 751 | }, | |
| 752 | }; | |
| 753 | const procH264Hls: Process = { | |
| 754 | name: `encode h.264 hls`, | |
| 755 | include: rules.extsVideo, | |
| 756 | enable: ffmpegBin != null, | |
| 757 | depends: [procDuration.name, procDimensions.name], | |
| 758 | async run({ path, mediaFile, node }) { | |
| 759 | if ((mediaFile.duration ?? 0) < 5) return; | |
| 760 | if (!mediaFile.dimensions) return; | |
| 761 | ||
| 762 | await derived.produce({ | |
| 763 | mediaFile, | |
| 764 | node, | |
| 765 | cores: 8, | |
| 766 | subkey: `hls`, | |
| 767 | async producer(dir) { | |
| 768 | const input = await videoArgsCache.getOrRun(path); | |
| 769 | const args = transcodeRules.getH264HlsArgs(input, dir); | |
| 770 | ||
| 771 | await ffmpeg.spawn({ | |
| 772 | ffmpeg: ffmpegBin!, | |
| 773 | progress: node, | |
| 774 | args, | |
| 775 | cwd: dir.toString(), | |
| 776 | }); | |
| 777 | }, | |
| 778 | }); | |
| 779 | }, | |
| 780 | }; | |
| 781 | ||
| 782 | const procCompression = [ | |
| 783 | { name: "gzip", fn: () => zlib.createGzip({ level: 9 }) }, | |
| 784 | { name: "zstd", fn: () => zlib.createZstdCompress() }, | |
| 785 | ].map<Process>( | |
| 786 | ({ name, fn }) => ({ | |
| 787 | name: `compress ${name}`, | |
| 788 | exclude: rules.extsPreCompressed, | |
| 789 | async run({ path, mediaFile, node }) { | |
| 790 | await derived.produce({ | |
| 791 | mediaFile, | |
| 792 | node, | |
| 793 | cores: 1, | |
| 794 | subkey: name, | |
| 795 | async producer(dir) { | |
| 796 | await stream.promises.pipeline( | |
| 797 | fs.createReadStream(path.toString()), | |
| 798 | fn(), | |
| 799 | fs.createWriteStream(dir.join(name).toString()), | |
| 800 | ); | |
| 801 | }, | |
| 802 | }); | |
| 803 | }, | |
| 804 | }), | |
| 805 | ); | |
| 806 | ||
| 807 | const processors = [ | |
| 808 | procDimensions, | |
| 809 | procDuration, | |
| 810 | procLoadTextContents, | |
| 811 | procHighlightCode, | |
| 812 | procImageSubsets, | |
| 813 | ...procVideos, | |
| 814 | ...procVideoAudios, | |
| 815 | procDash, | |
| 816 | ...procCompression, | |
| 817 | procH264Hls, | |
| 818 | ].map((process, id, all) => { | |
| 819 | const strIndex = (id: number) => String.fromCharCode("a".charCodeAt(0) + id); | |
| 820 | return { | |
| 821 | ...(process as Process), | |
| 822 | id: strIndex(id), | |
| 823 | // Create a unique key. | |
| 824 | hash: new Uint16Array( | |
| 825 | crypto | |
| 826 | .createHash("sha1") | |
| 827 | .update( | |
| 828 | process.run.toString() | |
| 829 | + (process.version ? String(process.version) : ""), | |
| 830 | ) | |
| 831 | .digest().buffer, | |
| 832 | ).reduce((a, b) => a ^ b), | |
| 833 | depends: (process.depends ?? []).map((depend) => { | |
| 834 | const index = all.findIndex((p) => p.name === depend); | |
| 835 | if (index === -1) throw new Error(`Cannot find depend '${depend}'`); | |
| 836 | if (index === id) throw new Error(`Cannot depend on self: '${depend}'`); | |
| 837 | return strIndex(index); | |
| 838 | }), | |
| 839 | }; | |
| 840 | }); | |
| 841 | ||
| 842 | function resizeDimensions(w: number, h: number, desiredWidth: number) { | |
| 843 | ASSERT(desiredWidth < w, `${desiredWidth} < ${w}`); | |
| 844 | return { w: desiredWidth, h: Math.floor((h / w) * desiredWidth) }; | |
| 845 | } | |
| 846 | ||
| 847 | interface UpdateMetadataArgs { | |
| 848 | path: Path; | |
| 849 | publicPath: string; | |
| 850 | stat: fs.Stats; | |
| 851 | mediaFile: MediaFile | null; | |
| 852 | } | |
| 853 | ||
| 854 | interface ProcessFileArgs { | |
| 855 | path: Path; | |
| 856 | stat: fs.Stats; | |
| 857 | mediaFile: MediaFile; | |
| 858 | node: progress.Node; | |
| 859 | } | |
| 860 | ||
| 861 | interface ProcessJob { | |
| 862 | path: Path; | |
| 863 | stat: fs.Stats; | |
| 864 | mediaFile: MediaFile; | |
| 865 | processor: (typeof processors)[0]; | |
| 866 | index: number; | |
| 867 | after: ProcessJob[]; | |
| 868 | needs: number; | |
| 869 | fileNode: progress.Node; | |
| 870 | } | |
| 871 | ||
| 872 | function skipBasename(basename: string): boolean { | |
| 873 | // dot files must be incrementally tracked | |
| 874 | if (basename === ".dirsort") return false; | |
| 875 | if (basename === ".friends") return false; | |
| 876 | if (basename === ".date") return false; | |
| 877 | ||
| 878 | return ( | |
| 879 | basename.startsWith(".") | |
| 880 | // basename.startsWith("._") || | |
| 881 | // basename.startsWith(".tmp") || | |
| 882 | // basename === ".DS_Store" || | |
| 883 | || basename.toLowerCase() === "thumbs.db" | |
| 884 | || basename.toLowerCase() === "desktop.ini" | |
| 67 | + `- Derived Size: \x1b[1m${string.formatByteSize(derivedSize)}\x1b[0m\n` | |
| 68 | + (failed > 0 | |
| 69 | ? `- \x1b[31mFailed Processors: ${failed}\x1b[0m (select * from file_processors where status = 2)\n` | |
| 70 | : ""), | |
| 885 | 71 | ); |
| 886 | 72 | } |
| 887 | 73 | |
| 888 | function toPublicPath(diskPath: Path) { | |
| 889 | if (diskPath.toString() === root) return "/"; | |
| 890 | return "/" + path.relative(root, diskPath.toString()).replaceAll("\\", "/"); | |
| 891 | } | |
| 892 | ||
| 893 | function testProgram(name: string, helpArgument: string) { | |
| 894 | try { | |
| 895 | child_process.spawnSync(name, [helpArgument]); | |
| 896 | return name; | |
| 897 | } catch (err) { | |
| 898 | console.warn(`Missing or corrupt executable '${name}'`); | |
| 899 | } | |
| 900 | return null; | |
| 901 | } | |
| 902 | ||
| 903 | // Helper function to check and remove location metadata | |
| 904 | async function scrubLocationMetadata( | |
| 905 | path: Path, | |
| 906 | stats: fs.Stats, | |
| 907 | progress: progress.Ref, | |
| 908 | ): Promise<boolean> { | |
| 909 | using _ = progress.start("scrub exif metadata"); | |
| 910 | const ext = path.ext.toLowerCase(); | |
| 911 | if (!rules.extsScrubExif.has(ext)) return false; | |
| 912 | ||
| 913 | let hasLocation = false; | |
| 914 | let args: string[] = []; | |
| 915 | ||
| 916 | // Check for location metadata based on file type | |
| 917 | const tempOutput = UNWRAP(path.parent).join(`.tmp.${path.base}`); | |
| 918 | switch (ext) { | |
| 919 | case ".jpg": | |
| 920 | case ".jpeg": | |
| 921 | case ".png": | |
| 922 | const { stdout: gpsCheck } = await subprocess.exec("exiftool", [ | |
| 923 | "-gps:all", | |
| 924 | path.toString(), | |
| 925 | ]); | |
| 926 | hasLocation = gpsCheck.trim().length > 0; | |
| 927 | args = ["-gps:all=", path.toString(), "-o", tempOutput.toString()]; | |
| 928 | break; | |
| 929 | case ".mov": | |
| 930 | case ".mp4": | |
| 931 | const { stdout: videoCheck } = await subprocess.exec("exiftool", [ | |
| 932 | "-ee", | |
| 933 | "-G3", | |
| 934 | "-s", | |
| 935 | path.toString(), | |
| 936 | ]); | |
| 937 | hasLocation = videoCheck.includes("GPS") | |
| 938 | || videoCheck.includes("Location"); | |
| 939 | args = [ | |
| 940 | "-gps:all=", | |
| 941 | "-xmp:all=", | |
| 942 | path.toString(), | |
| 943 | "-o", | |
| 944 | tempOutput.toString(), | |
| 945 | ]; | |
| 946 | break; | |
| 947 | case ".m4a": | |
| 948 | const { stdout: m4aCheck } = await subprocess.exec("exiftool", [ | |
| 949 | "-ee", | |
| 950 | "-G3", | |
| 951 | "-s", | |
| 952 | path.toString(), | |
| 953 | ]); | |
| 954 | hasLocation = m4aCheck.includes("GPS") | |
| 955 | || m4aCheck.includes("Location") | |
| 956 | || m4aCheck.includes("Filename") | |
| 957 | || m4aCheck.includes("Title"); | |
| 958 | ||
| 959 | if (hasLocation) { | |
| 960 | args = [ | |
| 961 | "-gps:all=", | |
| 962 | "-location:all=", | |
| 963 | "-filename:all=", | |
| 964 | "-title=", | |
| 965 | "-m4a:all=", | |
| 966 | path.toString(), | |
| 967 | "-o", | |
| 968 | tempOutput.toString(), | |
| 969 | ]; | |
| 970 | } | |
| 971 | break; | |
| 972 | } | |
| 973 | ||
| 974 | const accessTime = stats.atime; | |
| 975 | const modTime = stats.mtime; | |
| 976 | ||
| 977 | let backup: Path | null = null; | |
| 978 | try { | |
| 979 | if (hasLocation) { | |
| 980 | // Prepare a backup | |
| 981 | const tmp = UNWRAP(path.parent).join(`.tmp.backup.${path.base}`); | |
| 982 | await fsp.copyFile(path.toString(), tmp.toString()); | |
| 983 | await fsp.utimes(tmp.toString(), accessTime, modTime); | |
| 984 | backup = tmp; | |
| 985 | ||
| 986 | // Remove metadata | |
| 987 | await subprocess.exec("exiftool", args); | |
| 988 | if (!tempOutput.ifExistsSync()) { | |
| 989 | throw new Error(`Failed to create output file: ${tempOutput}`); | |
| 990 | } | |
| 991 | ||
| 992 | // Restore original timestamps | |
| 993 | await fsp.rename(tempOutput.toString(), path.toString()); | |
| 994 | await fsp.utimes(path.toString(), accessTime, modTime); | |
| 995 | ||
| 996 | // Backup is no longer needed | |
| 997 | await fsp.unlink(backup.toString()); | |
| 998 | ||
| 999 | console.info( | |
| 1000 | `Scrubbed location metadata in ${path.relative(Path.resolve(root))}`, | |
| 1001 | ); | |
| 1002 | return true; | |
| 1003 | } | |
| 1004 | } catch (error) { | |
| 1005 | if (backup) { | |
| 1006 | await fsp.rename(backup.toString(), path.toString()); | |
| 1007 | } | |
| 1008 | if (fs.existsSync(tempOutput.toString())) { | |
| 1009 | await fsp.unlink(tempOutput.toString()); | |
| 1010 | } | |
| 1011 | throw error; | |
| 1012 | } | |
| 1013 | ||
| 1014 | return false; | |
| 1015 | } | |
| 1016 | ||
| 1017 | const monthMilliseconds = 30 * 24 * 60 * 60 * 1000; | |
| 1018 | ||
| 1019 | import * as fs from "#sitegen/fs"; | |
| 1020 | import { Path } from "#sitegen/path"; | |
| 1021 | ||
| 1022 | import * as async from "@clo/lib/async"; | |
| 1023 | 74 | import * as log from "@clo/lib/log"; |
| 1024 | 75 | import * as progress from "@clo/lib/progress"; |
| 1025 | import * as queue from "@clo/lib/queue"; | |
| 1026 | 76 | import * as string from "@clo/lib/string"; |
| 1027 | import * as subprocess from "@clo/lib/subprocess"; | |
| 1028 | import * as ts from "@clo/lib/ts"; | |
| 1029 | 77 | |
| 1030 | import * as child_process from "node:child_process"; | |
| 1031 | import * as crypto from "node:crypto"; | |
| 1032 | import * as fsp from "node:fs/promises"; | |
| 1033 | import * as path from "node:path"; | |
| 1034 | import * as stream from "node:stream"; | |
| 1035 | import * as zlib from "node:zlib"; | |
| 78 | import { Path } from "#sitegen/path"; | |
| 79 | import { UNWRAP } from "@clo/lib/assert"; | |
| 1036 | 80 | |
| 1037 | import { formatDate } from "#src/file-viewer/format.ts"; | |
| 1038 | import * as highlight from "#src/file-viewer/highlight.ts"; | |
| 81 | import * as dirmeta from "#src/file-viewer/indexer/dirmeta.ts"; | |
| 82 | import { Scanner } from "#src/file-viewer/indexer/scan.ts"; | |
| 1039 | 83 | import * as derived from "#src/file-viewer/models/derived.ts"; |
| 1040 | import { FilePermissions } from "#src/file-viewer/models/FilePermissions.ts"; | |
| 1041 | import { MediaFile, MediaFileKind } from "#src/file-viewer/models/MediaFile.ts"; | |
| 1042 | import * as rsync from "#src/file-viewer/rsync.ts"; | |
| 1043 | import * as rules from "#src/file-viewer/rules.ts"; | |
| 1044 | import * as transcodeRules from "#src/file-viewer/transcode-rules.ts"; | |
| 1045 | import * as ffmpeg from "@clo/lib/subprocess/ffmpeg"; | |
| 1046 | ||
| 1047 | import { ASSERT, UNWRAP } from "@clo/lib/assert"; | |
| 1048 | import { rawFileRoot as root } from "../paths.ts"; | |
| 84 | import { MediaFile } from "#src/file-viewer/models/MediaFile.ts"; | |
| 85 | import { rawFileRoot } from "../paths.ts"; |
src/file-viewer/bin/file-trim.ts deleted-17| ... | ... | @@ -1,17 +0,0 @@ |
| 1 | export async function main() { | |
| 2 | const start = performance.now(); | |
| 3 | using _ = log.startWidget({ | |
| 4 | format: (now) => `paper clover's file scanner [${((now - start) / 1000).toFixed(1)}s]`, | |
| 5 | }); | |
| 6 | ||
| 7 | const orphaned = derived.findOrphanedRoots(); | |
| 8 | for (const root of orphaned) { | |
| 9 | console.info("delete " + root.key); | |
| 10 | derived.deleteRoot(root); | |
| 11 | } | |
| 12 | ||
| 13 | // TODO: delete unreferenced files | |
| 14 | } | |
| 15 | ||
| 16 | import * as log from "@clo/lib/log"; | |
| 17 | import * as derived from "../models/derived.ts"; |
src/file-viewer/bin/migrate-db.ts created+331| ... | ... | @@ -0,0 +1,331 @@ |
| 1 | // One-shot migration of cache.sqlite from the old bitfield processor | |
| 2 | // tracking ("v1") to the processors/file_processors tables ("v2"). | |
| 3 | // | |
| 4 | // CLOVER_DB=.clover node run migrate-db | |
| 5 | // | |
| 6 | // There is exactly one real copy of this database; migrate it once locally, | |
| 7 | // verify the site works, then place the migrated file on all machines | |
| 8 | // alongside the new code. The old `processed` column packed a 16-bit hash of | |
| 9 | // the applicable processor set plus per-processor "ran" bits indexed into | |
| 10 | // the `processors` string, whose entries were [letter id][2-char hash of | |
| 11 | // the processor's source code]. Completions are carried over as | |
| 12 | // done-at-current-version: the first sweep after migration must re-run | |
| 13 | // NOTHING (re-encoding the entire store would take weeks). Force re-runs | |
| 14 | // later by bumping a processor's version. | |
| 15 | // | |
| 16 | // This file intentionally avoids importing the models (their prepared | |
| 17 | // statements require the new schema) and instead uses raw SQL. It imports | |
| 18 | // the registry only for processor names, versions, and applicability. | |
| 19 | ||
| 20 | // the old positional letter ids, frozen. a..q matched the old array order. | |
| 21 | const letterMap: Record<string, string> = { | |
| 22 | a: "dimensions", | |
| 23 | b: "duration", | |
| 24 | c: "text-contents", | |
| 25 | d: "highlight-code", | |
| 26 | e: "image-subsets", | |
| 27 | f: "av1-au", | |
| 28 | g: "av1-ah", | |
| 29 | h: "av1-am", | |
| 30 | i: "av1-al", | |
| 31 | j: "av1-ad", | |
| 32 | k: "opus-h", | |
| 33 | l: "opus-m", | |
| 34 | m: "opus-d", | |
| 35 | n: "dash", | |
| 36 | o: "gzip", | |
| 37 | p: "zstd", | |
| 38 | q: "h264-hls", | |
| 39 | }; | |
| 40 | ||
| 41 | export async function main() { | |
| 42 | const db = getDb("cache.sqlite"); | |
| 43 | const raw = db.node; | |
| 44 | console.info(`migrating ${db.file}`); | |
| 45 | ||
| 46 | // -- preflight -- | |
| 47 | const columns = raw.prepare(`pragma table_info(media_files)`).all() as { | |
| 48 | name: string; | |
| 49 | }[]; | |
| 50 | if (columns.length === 0) { | |
| 51 | console.error("media_files does not exist; nothing to migrate"); | |
| 52 | process.exit(1); | |
| 53 | } | |
| 54 | if (!columns.some((c) => c.name === "processed")) { | |
| 55 | console.info("already migrated (no `processed` column); nothing to do"); | |
| 56 | return; | |
| 57 | } | |
| 58 | ||
| 59 | // -- backup -- | |
| 60 | raw.exec(`pragma wal_checkpoint(truncate);`); | |
| 61 | const backup = db.file + ".pre-v2"; | |
| 62 | if (fs.existsSync(backup) && !process.argv.includes("--force")) { | |
| 63 | console.error(`backup ${backup} already exists; pass --force to continue`); | |
| 64 | process.exit(1); | |
| 65 | } | |
| 66 | // content-only copy: fs.copyFile's metadata preservation gets EPERM'd on | |
| 67 | // the NAS datasets (restrictive ACL mode), plain writes do not. | |
| 68 | await stream.promises.pipeline( | |
| 69 | fs.createReadStream(db.file), | |
| 70 | fs.createWriteStream(backup), | |
| 71 | ); | |
| 72 | console.info(`backed up to ${backup}`); | |
| 73 | ||
| 74 | const now = Date.now(); | |
| 75 | const summary = new Map<string, { done: number; pending: number }>(); | |
| 76 | for (const p of registry.processors) { | |
| 77 | summary.set(p.name, { done: 0, pending: 0 }); | |
| 78 | } | |
| 79 | ||
| 80 | raw.exec(`pragma foreign_keys = off;`); | |
| 81 | raw.exec(`begin;`); | |
| 82 | try { | |
| 83 | // -- new tables (kept in sync with models/ProcessorState.ts) -- | |
| 84 | raw.exec(/* SQL */ ` | |
| 85 | create table if not exists processors ( | |
| 86 | id integer primary key autoincrement, | |
| 87 | name text not null unique, | |
| 88 | version integer not null | |
| 89 | ); | |
| 90 | create table if not exists file_processors ( | |
| 91 | file integer not null references media_files(id) on delete cascade, | |
| 92 | processor integer not null references processors(id) on delete cascade, | |
| 93 | version integer not null, | |
| 94 | status integer not null, | |
| 95 | updated integer not null, | |
| 96 | error text, | |
| 97 | primary key (file, processor) | |
| 98 | ); | |
| 99 | create index if not exists file_processors_processor | |
| 100 | on file_processors (processor); | |
| 101 | `); | |
| 102 | // mark the table-creation key so models/ProcessorState.ts skips its DDL | |
| 103 | raw.prepare( | |
| 104 | `insert or ignore into clover_migrations (key, version) values (?, ?);`, | |
| 105 | ).run("processor_state", 1); | |
| 106 | ||
| 107 | const ids = new Map<string, number>(); | |
| 108 | const insertProcessor = raw.prepare( | |
| 109 | `insert into processors (name, version) values (?, ?) | |
| 110 | on conflict(name) do update set version = excluded.version | |
| 111 | returning id;`, | |
| 112 | ); | |
| 113 | for (const p of registry.processors) { | |
| 114 | const { id } = insertProcessor.get(p.name, p.version) as { id: number }; | |
| 115 | ids.set(p.name, id); | |
| 116 | } | |
| 117 | ||
| 118 | // -- carry over completions from the bitfield -- | |
| 119 | const files = raw.prepare( | |
| 120 | `select id, path, processed, processors from media_files where kind = 1;`, | |
| 121 | ).all() as { | |
| 122 | id: number; | |
| 123 | path: string; | |
| 124 | processed: number; | |
| 125 | processors: string; | |
| 126 | }[]; | |
| 127 | const insertState = raw.prepare( | |
| 128 | `insert or replace into file_processors | |
| 129 | (file, processor, version, status, updated, error) | |
| 130 | values (?, ?, ?, 1, ?, null);`, | |
| 131 | ); | |
| 132 | let doneRows = 0; | |
| 133 | let undecodable = 0; | |
| 134 | for (const file of files) { | |
| 135 | let entries: string[]; | |
| 136 | try { | |
| 137 | entries = decodeProcessorLetters(file.processors); | |
| 138 | } catch { | |
| 139 | undecodable += 1; | |
| 140 | continue; | |
| 141 | } | |
| 142 | for (let i = 0; i < entries.length; i += 1) { | |
| 143 | if ((file.processed & (1 << (16 + i))) === 0) continue; | |
| 144 | const name = letterMap[UNWRAP(entries[i])]; | |
| 145 | if (!name) continue; // processor no longer exists | |
| 146 | const proc = registry.byName(name); | |
| 147 | if (!proc) continue; | |
| 148 | insertState.run(file.id, UNWRAP(ids.get(name)), proc.version, now); | |
| 149 | UNWRAP(summary.get(name)).done += 1; | |
| 150 | doneRows += 1; | |
| 151 | } | |
| 152 | } | |
| 153 | ||
| 154 | // -- rebuild media_files without the bitfield columns -- | |
| 155 | raw.exec(/* SQL */ ` | |
| 156 | create table media_files_new ( | |
| 157 | id integer primary key autoincrement, | |
| 158 | parent_id integer, | |
| 159 | path text, | |
| 160 | kind integer not null, | |
| 161 | timestamp integer not null, | |
| 162 | timestamp_updated integer not null default current_timestamp, | |
| 163 | hash text not null, | |
| 164 | size integer not null, | |
| 165 | duration integer not null default 0, | |
| 166 | dimensions text not null default "", | |
| 167 | contents text not null, | |
| 168 | dirsort text, | |
| 169 | config text not null default "", | |
| 170 | dir_reindex integer not null default 0, | |
| 171 | pending integer not null default 0, | |
| 172 | foreign key (parent_id) references media_files(id) on delete cascade | |
| 173 | ); | |
| 174 | insert into media_files_new ( | |
| 175 | id, parent_id, path, kind, timestamp, timestamp_updated, hash, size, | |
| 176 | duration, dimensions, contents, dirsort, config, dir_reindex, pending) | |
| 177 | select | |
| 178 | id, parent_id, path, kind, timestamp, timestamp_updated, hash, size, | |
| 179 | duration, dimensions, contents, dirsort, | |
| 180 | case when kind = 0 and processors like '{%' then processors else '' end, | |
| 181 | case when kind = 0 and processed = 0 then 1 else 0 end, | |
| 182 | 0 | |
| 183 | from media_files; | |
| 184 | drop table media_files; | |
| 185 | alter table media_files_new rename to media_files; | |
| 186 | `); | |
| 187 | ||
| 188 | // -- collapse case duplicates -- | |
| 189 | // the file stores are case-insensitive, so two case spellings of one | |
| 190 | // path are the same physical file. the old scanner could record both; | |
| 191 | // keep the newest row (matching the new unique nocase index) and move | |
| 192 | // any children over. | |
| 193 | const dupeGroups = raw.prepare( | |
| 194 | `select group_concat(id) ids, max(id) keep, lower(path) lp | |
| 195 | from media_files group by lower(path) having count(*) > 1;`, | |
| 196 | ).all() as { ids: string; keep: number; lp: string }[]; | |
| 197 | for (const group of dupeGroups) { | |
| 198 | const drop = group.ids.split(",").map(Number) | |
| 199 | .filter((id) => id !== group.keep); | |
| 200 | console.warn( | |
| 201 | `case-duplicate rows for ${group.lp}: keeping ${group.keep}, dropping ${drop.join(", ")}`, | |
| 202 | ); | |
| 203 | const dropList = drop.join(","); | |
| 204 | raw.exec(/* SQL */ ` | |
| 205 | update media_files set parent_id = ${group.keep} | |
| 206 | where parent_id in (${dropList}); | |
| 207 | delete from file_processors where file in (${dropList}); | |
| 208 | delete from derived_refs where file in (${dropList}); | |
| 209 | delete from media_files where id in (${dropList}); | |
| 210 | `); | |
| 211 | } | |
| 212 | ||
| 213 | raw.exec(/* SQL */ ` | |
| 214 | create unique index media_files_path | |
| 215 | on media_files (path collate nocase); | |
| 216 | create index media_files_parent_id on media_files (parent_id); | |
| 217 | create index media_files_file_children on media_files (kind, path); | |
| 218 | create index media_files_dir_reindex on media_files (kind, dir_reindex); | |
| 219 | `); | |
| 220 | ||
| 221 | // -- compute `pending` from applicability minus completions -- | |
| 222 | const states = raw.prepare( | |
| 223 | `select processor, version from file_processors where file = ?;`, | |
| 224 | ); | |
| 225 | const setPending = raw.prepare( | |
| 226 | `update media_files set pending = ? where id = ?;`, | |
| 227 | ); | |
| 228 | const pendingPaths: string[] = []; | |
| 229 | for (const file of files) { | |
| 230 | const ext = extensionNonEmpty(file.path).toLowerCase(); | |
| 231 | const applicable = registry.applicableFor(ext); | |
| 232 | if (applicable.length === 0) continue; | |
| 233 | const done = new Map( | |
| 234 | (states.all(file.id) as { processor: number; version: number }[]) | |
| 235 | .map((row) => [row.processor, row.version]), | |
| 236 | ); | |
| 237 | const missing = applicable.filter( | |
| 238 | (p) => done.get(UNWRAP(ids.get(p.name))) !== p.version, | |
| 239 | ); | |
| 240 | if (missing.length === 0) continue; | |
| 241 | setPending.run(missing.length, file.id); | |
| 242 | for (const p of missing) UNWRAP(summary.get(p.name)).pending += 1; | |
| 243 | if (pendingPaths.length < 32) { | |
| 244 | pendingPaths.push( | |
| 245 | `${file.path} (${missing.map((p) => p.name).join(", ")})`, | |
| 246 | ); | |
| 247 | } | |
| 248 | } | |
| 249 | ||
| 250 | raw.exec(`commit;`); | |
| 251 | raw.exec(`pragma foreign_keys = on;`); | |
| 252 | ||
| 253 | const violations = raw.prepare(`pragma foreign_key_check;`).all(); | |
| 254 | if (violations.length > 0) { | |
| 255 | console.error("foreign key violations after migration:", violations); | |
| 256 | process.exit(1); | |
| 257 | } | |
| 258 | raw.exec(`vacuum;`); | |
| 259 | raw.exec(`pragma wal_checkpoint(truncate);`); | |
| 260 | ||
| 261 | // -- summary -- | |
| 262 | console.info(""); | |
| 263 | console.info(`migrated ${files.length} files, ${doneRows} completions`); | |
| 264 | if (undecodable) { | |
| 265 | console.warn(`${undecodable} files had undecodable processor strings`); | |
| 266 | } | |
| 267 | console.info("per-processor state (done / pending):"); | |
| 268 | for (const [name, { done, pending }] of summary) { | |
| 269 | const warn = pending > 0 && heavyProcessors.has(name) ? " <-- WILL RUN" : ""; | |
| 270 | console.info( | |
| 271 | ` ${name.padEnd(16)} ${String(done).padStart(6)} / ${String(pending).padStart(4)}${warn}`, | |
| 272 | ); | |
| 273 | } | |
| 274 | if (pendingPaths.length > 0) { | |
| 275 | console.info(""); | |
| 276 | console.info("files with pending work (first 32):"); | |
| 277 | for (const line of pendingPaths) console.info(" " + line); | |
| 278 | } else { | |
| 279 | console.info("no files have pending work; first sweep will be a no-op"); | |
| 280 | } | |
| 281 | } catch (err) { | |
| 282 | raw.exec(`rollback;`); | |
| 283 | raw.exec(`pragma foreign_keys = on;`); | |
| 284 | console.error("migration failed and was rolled back"); | |
| 285 | throw err; | |
| 286 | } | |
| 287 | } | |
| 288 | ||
| 289 | /** decode the old `processors` column into its letter ids */ | |
| 290 | function decodeProcessorLetters(input: string): string[] { | |
| 291 | return input | |
| 292 | .split(";") | |
| 293 | .filter(Boolean) | |
| 294 | .map(([a, b, c]) => { | |
| 295 | UNWRAP(b); | |
| 296 | UNWRAP(c); | |
| 297 | return UNWRAP(a); | |
| 298 | }); | |
| 299 | } | |
| 300 | ||
| 301 | /** mirror of MediaFile.extensionNonEmpty for a raw path string */ | |
| 302 | function extensionNonEmpty(filePath: string) { | |
| 303 | const basename = path.basename(filePath); | |
| 304 | const ext = path.extname(basename); | |
| 305 | if (ext === "") return basename; | |
| 306 | return ext; | |
| 307 | } | |
| 308 | ||
| 309 | // processors expensive enough that an accidental re-run is a disaster | |
| 310 | const heavyProcessors = new Set([ | |
| 311 | "image-subsets", | |
| 312 | "av1-au", | |
| 313 | "av1-ah", | |
| 314 | "av1-am", | |
| 315 | "av1-al", | |
| 316 | "av1-ad", | |
| 317 | "opus-h", | |
| 318 | "opus-m", | |
| 319 | "opus-d", | |
| 320 | "dash", | |
| 321 | "h264-hls", | |
| 322 | ]); | |
| 323 | ||
| 324 | import * as fs from "node:fs"; | |
| 325 | import * as path from "node:path"; | |
| 326 | import * as stream from "node:stream"; | |
| 327 | ||
| 328 | import { getDb } from "#sitegen/sqlite"; | |
| 329 | import { UNWRAP } from "@clo/lib/assert"; | |
| 330 | ||
| 331 | import * as registry from "#src/file-viewer/indexer/registry.ts"; |
src/file-viewer/indexer/dirmeta.ts created+108| ... | ... | @@ -0,0 +1,108 @@ |
| 1 | // Directory metadata pass: readme contents, explicit sort order, friend | |
| 2 | // permissions, recursive size/date/hash aggregates, and the `.date` | |
| 3 | // override. Driven by the `dir_reindex` flag, which the scanner sets on | |
| 4 | // parents whenever children change. Ported from the tail of file-scan.ts. | |
| 5 | const console = log.scoped("indexer"); | |
| 6 | ||
| 7 | export function run(p: progress.Ref): boolean { | |
| 8 | const dirs = MediaFile.getDirectoriesToReindex() | |
| 9 | .sort((a, b) => b.path.length - a.path.length); | |
| 10 | if (dirs.length === 0) return false; | |
| 11 | using node = p.start("update directory metadata", { total: dirs.length }); | |
| 12 | ||
| 13 | for (const dir of dirs) { | |
| 14 | using _ = node.start(dir.path); | |
| 15 | try { | |
| 16 | processDir(dir); | |
| 17 | } catch (err) { | |
| 18 | // one broken directory must not starve the rest of the pass; the | |
| 19 | // dir_reindex flag stays set, so it retries on the next trigger | |
| 20 | console.error(`directory metadata for ${dir.path} failed:`, err); | |
| 21 | } | |
| 22 | node.inc(); | |
| 23 | } | |
| 24 | console.info(`updated metadata for ${dirs.length} directories`); | |
| 25 | return true; | |
| 26 | } | |
| 27 | ||
| 28 | function processDir(dir: MediaFile) { | |
| 29 | const children = dir.getChildren(); | |
| 30 | ||
| 31 | // readme.txt | |
| 32 | const readmeContent = children.find((x) => x.basename === "readme.txt")?.contents ?? ""; | |
| 33 | ||
| 34 | // dirsort | |
| 35 | let dirsort: string[] | null = null; | |
| 36 | const dirSortRaw = children.find((x) => x.basename === ".dirsort")?.contents ?? ""; | |
| 37 | if (dirSortRaw) { | |
| 38 | dirsort = dirSortRaw | |
| 39 | .split("\n") | |
| 40 | .map((x) => x.trim()) | |
| 41 | .filter(Boolean); | |
| 42 | } | |
| 43 | ||
| 44 | // Permissions | |
| 45 | if (children.some((x) => x.basename === ".friends")) { | |
| 46 | FilePermissions.setPermissions(dir.path, 1); | |
| 47 | } else { | |
| 48 | FilePermissions.setPermissions(dir.path, 0); | |
| 49 | } | |
| 50 | ||
| 51 | // Recursive stats. | |
| 52 | let totalSize = 0; | |
| 53 | let newestDate = new Date(0); | |
| 54 | let allHashes = ""; | |
| 55 | for (const child of children) { | |
| 56 | totalSize += child.size; | |
| 57 | allHashes += child.hash; | |
| 58 | ||
| 59 | // readme.txt and hidden files don't render a date in the UI, so they | |
| 60 | // must not contribute to the directory's date either. | |
| 61 | const dateExempt = child.basename === "readme.txt" || child.basename.startsWith("."); | |
| 62 | if (!dateExempt && child.date > newestDate) { | |
| 63 | newestDate = child.date; | |
| 64 | } | |
| 65 | } | |
| 66 | ||
| 67 | // Project Date | |
| 68 | const dateFile = children.find((x) => x.basename === ".date"); | |
| 69 | if (dateFile) { | |
| 70 | const date = new Date(dateFile.contents); | |
| 71 | if (Number.isNaN(date.getTime())) { | |
| 72 | // a fresh .date file has no extracted contents until the | |
| 73 | // text-contents processor lands; leave dir_reindex set and let a | |
| 74 | // later pass pick this directory back up. a genuinely malformed | |
| 75 | // .date keeps warning here until it is fixed on disk. | |
| 76 | console.warn( | |
| 77 | `${dir.path}/.date is empty or unparseable; deferring date override`, | |
| 78 | ); | |
| 79 | return; | |
| 80 | } | |
| 81 | newestDate = date; | |
| 82 | dir.setConfig(JSON.stringify({ hideChildrenDates: true })); | |
| 83 | } else { | |
| 84 | dir.setConfig(""); | |
| 85 | } | |
| 86 | ||
| 87 | const dirHash = crypto | |
| 88 | .createHash("sha1") | |
| 89 | .update(dir.path + allHashes) | |
| 90 | .digest("hex"); | |
| 91 | ||
| 92 | MediaFile.markDirectoryProcessed({ | |
| 93 | id: dir.id, | |
| 94 | timestamp: newestDate, | |
| 95 | contents: readmeContent, | |
| 96 | size: totalSize, | |
| 97 | hash: dirHash, | |
| 98 | dirsort, | |
| 99 | }); | |
| 100 | } | |
| 101 | ||
| 102 | import * as crypto from "node:crypto"; | |
| 103 | ||
| 104 | import * as log from "@clo/lib/log"; | |
| 105 | import * as progress from "@clo/lib/progress"; | |
| 106 | ||
| 107 | import { FilePermissions } from "#src/file-viewer/models/FilePermissions.ts"; | |
| 108 | import { MediaFile } from "#src/file-viewer/models/MediaFile.ts"; |
src/file-viewer/indexer/processors/compress.ts created+35| ... | ... | @@ -0,0 +1,35 @@ |
| 1 | // pre-compressed variants for static serving. applies to everything that is | |
| 2 | // not already compressed (media containers, archives, ...). | |
| 3 | export const processors: Processor[] = [ | |
| 4 | { name: "gzip", fn: () => zlib.createGzip({ level: 9 }) }, | |
| 5 | { name: "zstd", fn: () => zlib.createZstdCompress() }, | |
| 6 | ].map(({ name, fn }) => ({ | |
| 7 | name, | |
| 8 | version: 1, | |
| 9 | title: `compress ${name}`, | |
| 10 | exclude: rules.extsPreCompressed, | |
| 11 | async run({ path, mediaFile, node, signal }) { | |
| 12 | await derived.produce({ | |
| 13 | mediaFile, | |
| 14 | node, | |
| 15 | cores: 1, | |
| 16 | subkey: name, | |
| 17 | async producer(dir) { | |
| 18 | await stream.promises.pipeline( | |
| 19 | fs.createReadStream(path.toString()), | |
| 20 | fn(), | |
| 21 | fs.createWriteStream(dir.join(name).toString()), | |
| 22 | { signal }, | |
| 23 | ); | |
| 24 | }, | |
| 25 | }); | |
| 26 | }, | |
| 27 | })); | |
| 28 | ||
| 29 | import * as fs from "node:fs"; | |
| 30 | import * as stream from "node:stream"; | |
| 31 | import * as zlib from "node:zlib"; | |
| 32 | ||
| 33 | import * as derived from "#src/file-viewer/models/derived.ts"; | |
| 34 | import * as rules from "#src/file-viewer/rules.ts"; | |
| 35 | import type { Processor } from "../registry.ts"; |
src/file-viewer/indexer/processors/media.ts created+352| ... | ... | @@ -0,0 +1,352 @@ |
| 1 | // ffmpeg/ffprobe/magick-based processors: duration, dimensions, optimized | |
| 2 | // image subsets, av1+opus+dash streaming variants, and h264 hls. bodies are | |
| 3 | // ported unchanged from the old file-scan.ts. | |
| 4 | const ffprobeBin = testProgram("ffprobe", "--help"); | |
| 5 | const ffmpegBin = testProgram("ffmpeg", "--help"); | |
| 6 | ||
| 7 | const ffmpegOptions = ["-hide_banner", "-loglevel", "warning"]; | |
| 8 | ||
| 9 | const procDuration: Processor = { | |
| 10 | name: "duration", | |
| 11 | version: 1, | |
| 12 | title: "calculate duration", | |
| 13 | enable: ffprobeBin !== null, | |
| 14 | include: rules.extsDuration, | |
| 15 | cores: 1, | |
| 16 | async run({ path, mediaFile, signal }) { | |
| 17 | const { stdout } = await subprocess.exec(ffprobeBin!, [ | |
| 18 | "-v", | |
| 19 | "error", | |
| 20 | "-show_entries", | |
| 21 | "format=duration", | |
| 22 | "-of", | |
| 23 | "default=noprint_wrappers=1:nokey=1", | |
| 24 | path.toString(), | |
| 25 | ], { signal }); | |
| 26 | ||
| 27 | const duration = parseFloat(stdout.trim()); | |
| 28 | if (Number.isNaN(duration)) { | |
| 29 | throw new Error("Could not extract duration from " + stdout); | |
| 30 | } | |
| 31 | mediaFile.setDuration(Math.ceil(duration)); | |
| 32 | }, | |
| 33 | }; | |
| 34 | ||
| 35 | const procDimensions: Processor = { | |
| 36 | name: "dimensions", | |
| 37 | version: 2, | |
| 38 | title: "calculate dimensions", | |
| 39 | enable: ffprobeBin != null, | |
| 40 | include: rules.extsDimensions, | |
| 41 | cores: 1, | |
| 42 | async run({ path, mediaFile, signal }) { | |
| 43 | const { ext } = path; | |
| 44 | ||
| 45 | let dimensions; | |
| 46 | ||
| 47 | if (ext === ".svg") { | |
| 48 | // Parse out of text data | |
| 49 | const content = await path.read("utf-8"); | |
| 50 | const widthMatch = content.match(/width="(\d+)"/); | |
| 51 | const heightMatch = content.match(/height="(\d+)"/); | |
| 52 | ||
| 53 | if (widthMatch && heightMatch) { | |
| 54 | dimensions = `${widthMatch[1]}x${heightMatch[1]}`; | |
| 55 | } | |
| 56 | } else if (rules.extsImage.has(ext)) { | |
| 57 | // Use magick to observe streams | |
| 58 | const { stdout } = await subprocess.exec("magick", [ | |
| 59 | "identify", | |
| 60 | "-auto-orient", | |
| 61 | "-format", | |
| 62 | "%w %h", | |
| 63 | path.toString(), | |
| 64 | ], { signal }); | |
| 65 | const [w, h] = stdout.split(" ").map((x) => Number(x)); | |
| 66 | if (w && h) { | |
| 67 | dimensions = w + "x" + h; | |
| 68 | } | |
| 69 | } else { | |
| 70 | // Use ffprobe to observe streams | |
| 71 | const { stdout } = await subprocess.exec("ffprobe", [ | |
| 72 | "-v", | |
| 73 | "error", | |
| 74 | "-select_streams", | |
| 75 | "v:0", | |
| 76 | "-show_entries", | |
| 77 | "stream=width,height:stream_side_data=rotation", | |
| 78 | "-of", | |
| 79 | "json", | |
| 80 | path.toString(), | |
| 81 | ], { signal }); | |
| 82 | const result = JSON.parse(stdout); | |
| 83 | const stream = result.streams[0]; | |
| 84 | if (stream) { | |
| 85 | let width = UNWRAP(stream.width); | |
| 86 | let height = UNWRAP(stream.height); | |
| 87 | // phone videos store the sensor's landscape frame plus a display | |
| 88 | // matrix; everything downstream (scale filters, aspect-ratio css) | |
| 89 | // wants the rotated display dimensions. | |
| 90 | const rotation = (stream.side_data_list ?? []) | |
| 91 | .find((s: any) => typeof s.rotation === "number")?.rotation ?? 0; | |
| 92 | if (Math.abs(rotation) % 180 === 90) { | |
| 93 | [width, height] = [height, width]; | |
| 94 | } | |
| 95 | dimensions = width + "x" + height; | |
| 96 | } | |
| 97 | } | |
| 98 | ||
| 99 | mediaFile.setDimensions(dimensions ?? ""); | |
| 100 | }, | |
| 101 | }; | |
| 102 | ||
| 103 | const procImageSubsets: Processor = { | |
| 104 | name: "image-subsets", | |
| 105 | version: 3, | |
| 106 | title: "encode image subsets", | |
| 107 | include: rules.extsImage, | |
| 108 | depends: [procDimensions.name], | |
| 109 | async run({ path, mediaFile, node, signal }) { | |
| 110 | const dims = mediaFile.parseDimensions(); | |
| 111 | // dimensions could not be extracted; nothing to scale. | |
| 112 | if (!dims) return; | |
| 113 | const { width, height } = dims; | |
| 114 | const targetSizes = transcodeRules.imageSizes.filter((w) => w < width); | |
| 115 | const baseStatus = node.text; | |
| 116 | ||
| 117 | using stack = new DisposableStack(); | |
| 118 | for (const size of targetSizes) { | |
| 119 | const { w, h } = resizeDimensions(width, height, size); | |
| 120 | for (const { ext, args } of transcodeRules.imagePresets) { | |
| 121 | node.text = baseStatus + ` (${w}x${h}, ${ext.slice(1).toUpperCase()})`; | |
| 122 | ||
| 123 | stack.use( | |
| 124 | await derived.produce({ | |
| 125 | mediaFile, | |
| 126 | node, | |
| 127 | cores: 2, | |
| 128 | subkey: `${size}${ext}`, | |
| 129 | async producer(dir) { | |
| 130 | await subprocess.exec(ffmpegBin!, [ | |
| 131 | ...ffmpegOptions, | |
| 132 | "-i", | |
| 133 | path.toString(), | |
| 134 | "-vf", | |
| 135 | `scale=${w}:${h}:force_original_aspect_ratio=increase,crop=${w}:${h}`, | |
| 136 | ...args, | |
| 137 | dir.join(`${size}${ext}`).toString(), | |
| 138 | ], { signal }); | |
| 139 | }, | |
| 140 | }), | |
| 141 | ); | |
| 142 | } | |
| 143 | } | |
| 144 | ||
| 145 | stack.move(); | |
| 146 | }, | |
| 147 | }; | |
| 148 | ||
| 149 | const videoArgsCache = new async.OnceMap(transcodeRules.getVideoInputArgs); | |
| 150 | ||
| 151 | const qualityMap: Record<string, string> = { | |
| 152 | u: "ultra-high", | |
| 153 | h: "high", | |
| 154 | m: "medium", | |
| 155 | l: "low", | |
| 156 | d: "data-saving", | |
| 157 | }; | |
| 158 | ||
| 159 | function skipVideo(mediaFile: MediaFile) { | |
| 160 | if ((mediaFile.duration ?? 0) < 5) return true; | |
| 161 | if (!mediaFile.dimensions) return true; | |
| 162 | if (rules.processDenyList.has(mediaFile.path)) return true; | |
| 163 | return false; | |
| 164 | } | |
| 165 | ||
| 166 | const procVideos = transcodeRules.videoFormats.map<Processor>((preset) => ({ | |
| 167 | name: `av1-${preset.id}`, | |
| 168 | version: 4, | |
| 169 | title: `encode av1 ${UNWRAP(qualityMap[UNWRAP(preset.id[1])])}`, | |
| 170 | include: rules.extsVideo, | |
| 171 | enable: ffmpegBin != null, | |
| 172 | depends: [procDuration.name, procDimensions.name], | |
| 173 | async run({ path, mediaFile, node, signal }) { | |
| 174 | if (skipVideo(mediaFile)) return; | |
| 175 | ||
| 176 | await derived.produce({ | |
| 177 | mediaFile, | |
| 178 | node, | |
| 179 | cores: 4, | |
| 180 | subkey: `av1-${preset.id}`, | |
| 181 | async producer(dir) { | |
| 182 | const input = await videoArgsCache.getOrRun(path); | |
| 183 | const dimensions = mediaFile.parseDimensions(); | |
| 184 | ASSERT(dimensions); | |
| 185 | const args = transcodeRules.getAv1VideoArgs( | |
| 186 | preset, | |
| 187 | dimensions, | |
| 188 | UNWRAP(input.video, "frick on " + path + JSON.stringify(input)), | |
| 189 | dir, | |
| 190 | ); | |
| 191 | ||
| 192 | await ffmpeg.spawn({ | |
| 193 | ffmpeg: ffmpegBin!, | |
| 194 | progress: node, | |
| 195 | args, | |
| 196 | cwd: dir.toString(), | |
| 197 | signal, | |
| 198 | }); | |
| 199 | }, | |
| 200 | }); | |
| 201 | }, | |
| 202 | })); | |
| 203 | ||
| 204 | const procVideoAudios = transcodeRules.audioFormats.map<Processor>((preset) => ({ | |
| 205 | name: `opus-${preset.id}`, | |
| 206 | version: 3, | |
| 207 | title: `encode opus ${UNWRAP(qualityMap[UNWRAP(preset.id)])}`, | |
| 208 | include: rules.extsVideo, | |
| 209 | enable: ffmpegBin != null, | |
| 210 | depends: [procDuration.name, procDimensions.name], | |
| 211 | async run({ path, mediaFile, node, signal }) { | |
| 212 | if (skipVideo(mediaFile)) return; | |
| 213 | await derived.produce({ | |
| 214 | mediaFile, | |
| 215 | node, | |
| 216 | cores: 4, | |
| 217 | subkey: `opus-${preset.id}`, | |
| 218 | async producer(dir) { | |
| 219 | const input = await videoArgsCache.getOrRun(path); | |
| 220 | ASSERT(input.video); | |
| 221 | if (!input.audio) return; | |
| 222 | const args = transcodeRules.getOpusAudioArgs(preset, input.audio, dir); | |
| 223 | ||
| 224 | await ffmpeg.spawn({ | |
| 225 | ffmpeg: ffmpegBin!, | |
| 226 | progress: node, | |
| 227 | args, | |
| 228 | cwd: dir.toString(), | |
| 229 | signal, | |
| 230 | }); | |
| 231 | }, | |
| 232 | }); | |
| 233 | }, | |
| 234 | })); | |
| 235 | ||
| 236 | const procDash: Processor = { | |
| 237 | name: "dash", | |
| 238 | version: 7, | |
| 239 | title: "encode mpeg-dash", | |
| 240 | include: rules.extsVideo, | |
| 241 | enable: ffmpegBin != null, | |
| 242 | depends: [...procVideos, ...procVideoAudios].map((x) => x.name), | |
| 243 | async run({ path, mediaFile, node, signal }) { | |
| 244 | if (skipVideo(mediaFile)) return; | |
| 245 | await derived.produce({ | |
| 246 | mediaFile, | |
| 247 | node, | |
| 248 | cores: 1, | |
| 249 | subkey: `dash-av1`, | |
| 250 | async producer(dir) { | |
| 251 | const input = await videoArgsCache.getOrRun(path); | |
| 252 | const videos = transcodeRules.videoFormats.map( | |
| 253 | (preset) => | |
| 254 | UNWRAP(dir.parent).join( | |
| 255 | `av1-${preset.id}`, | |
| 256 | transcodeRules.av1FileName, | |
| 257 | ), | |
| 258 | ); | |
| 259 | const audios = input.audio | |
| 260 | ? transcodeRules.audioFormats.map( | |
| 261 | (preset) => | |
| 262 | UNWRAP(dir.parent).join( | |
| 263 | `opus-${preset.id}`, | |
| 264 | transcodeRules.opusFileName, | |
| 265 | ), | |
| 266 | ) | |
| 267 | : []; | |
| 268 | const args = transcodeRules.getMpegDashArgs( | |
| 269 | videos, | |
| 270 | audios, | |
| 271 | dir, | |
| 272 | ); | |
| 273 | ||
| 274 | await dir.join("d").makeDir(); | |
| 275 | ||
| 276 | await ffmpeg.spawn({ | |
| 277 | ffmpeg: ffmpegBin!, | |
| 278 | progress: node, | |
| 279 | args, | |
| 280 | cwd: dir.toString(), | |
| 281 | signal, | |
| 282 | }); | |
| 283 | }, | |
| 284 | }); | |
| 285 | }, | |
| 286 | }; | |
| 287 | ||
| 288 | const procH264Hls: Processor = { | |
| 289 | name: "h264-hls", | |
| 290 | version: 1, | |
| 291 | title: "encode h.264 hls", | |
| 292 | include: rules.extsVideo, | |
| 293 | enable: ffmpegBin != null, | |
| 294 | depends: [procDuration.name, procDimensions.name], | |
| 295 | async run({ path, mediaFile, node, signal }) { | |
| 296 | if (skipVideo(mediaFile)) return; | |
| 297 | ||
| 298 | await derived.produce({ | |
| 299 | mediaFile, | |
| 300 | node, | |
| 301 | cores: 8, | |
| 302 | subkey: `hls`, | |
| 303 | async producer(dir) { | |
| 304 | const input = await videoArgsCache.getOrRun(path); | |
| 305 | const args = transcodeRules.getH264HlsArgs(input, dir); | |
| 306 | ||
| 307 | await ffmpeg.spawn({ | |
| 308 | ffmpeg: ffmpegBin!, | |
| 309 | progress: node, | |
| 310 | args, | |
| 311 | cwd: dir.toString(), | |
| 312 | signal, | |
| 313 | }); | |
| 314 | }, | |
| 315 | }); | |
| 316 | }, | |
| 317 | }; | |
| 318 | ||
| 319 | function resizeDimensions(w: number, h: number, desiredWidth: number) { | |
| 320 | ASSERT(desiredWidth < w, `${desiredWidth} < ${w}`); | |
| 321 | return { w: desiredWidth, h: Math.floor((h / w) * desiredWidth) }; | |
| 322 | } | |
| 323 | ||
| 324 | function testProgram(name: string, helpArgument: string) { | |
| 325 | // spawnSync does not throw on a missing binary; it reports `error` | |
| 326 | const result = child_process.spawnSync(name, [helpArgument]); | |
| 327 | if (!result.error) return name; | |
| 328 | console.warn(`Missing or corrupt executable '${name}'`); | |
| 329 | return null; | |
| 330 | } | |
| 331 | ||
| 332 | export const processors: Processor[] = [ | |
| 333 | procDimensions, | |
| 334 | procDuration, | |
| 335 | procImageSubsets, | |
| 336 | ...procVideos, | |
| 337 | ...procVideoAudios, | |
| 338 | procDash, | |
| 339 | procH264Hls, | |
| 340 | ]; | |
| 341 | ||
| 342 | import { ASSERT, UNWRAP } from "@clo/lib/assert"; | |
| 343 | import * as async from "@clo/lib/async"; | |
| 344 | import * as subprocess from "@clo/lib/subprocess"; | |
| 345 | import * as ffmpeg from "@clo/lib/subprocess/ffmpeg"; | |
| 346 | import * as child_process from "node:child_process"; | |
| 347 | ||
| 348 | import * as derived from "#src/file-viewer/models/derived.ts"; | |
| 349 | import type { MediaFile } from "#src/file-viewer/models/MediaFile.ts"; | |
| 350 | import * as rules from "#src/file-viewer/rules.ts"; | |
| 351 | import * as transcodeRules from "#src/file-viewer/transcode-rules.ts"; | |
| 352 | import type { Processor } from "../registry.ts"; |
src/file-viewer/indexer/processors/text.ts created+56| ... | ... | @@ -0,0 +1,56 @@ |
| 1 | // text-content extraction processors. `text-contents` fills | |
| 2 | // `media_files.contents` for plain text files; `highlight-code` fills it | |
| 3 | // with pre-rendered syntax highlighting html for source code. | |
| 4 | const procLoadTextContents: Processor = { | |
| 5 | name: "text-contents", | |
| 6 | version: 2, | |
| 7 | title: "load text content", | |
| 8 | include: rules.extsReadContents, | |
| 9 | cores: 1, | |
| 10 | async run({ path, mediaFile, stat }) { | |
| 11 | if (stat.size > 1_000_000) return; | |
| 12 | const text = await path.read("utf-8"); | |
| 13 | mediaFile.setContents(text); | |
| 14 | }, | |
| 15 | }; | |
| 16 | ||
| 17 | const procHighlightCode: Processor = { | |
| 18 | name: "highlight-code", | |
| 19 | version: 2, | |
| 20 | title: "highlight source code", | |
| 21 | include: new Set(rules.extsCode.keys()), | |
| 22 | cores: 1, | |
| 23 | async run({ path, mediaFile, stat }) { | |
| 24 | const language = UNWRAP( | |
| 25 | rules.extsCode.get(path.ext.toLowerCase()), | |
| 26 | ); | |
| 27 | // An issue is that .ts is an overloaded extension, shared between | |
| 28 | // 'transport stream' and 'typescript'. | |
| 29 | // | |
| 30 | // Filter used here is: | |
| 31 | // - more than 1mb | |
| 32 | // - invalid UTF-8 | |
| 33 | if (stat.size > 1_000_000) return; | |
| 34 | let code; | |
| 35 | const buf = await path.read(); | |
| 36 | try { | |
| 37 | code = new TextDecoder("utf-8", { fatal: true }).decode(buf); | |
| 38 | } catch (error) { | |
| 39 | mediaFile.setContents(""); | |
| 40 | return; | |
| 41 | } | |
| 42 | const content = await highlight.highlightCode(code, language); | |
| 43 | mediaFile.setContents(content); | |
| 44 | }, | |
| 45 | }; | |
| 46 | ||
| 47 | export const processors: Processor[] = [ | |
| 48 | procLoadTextContents, | |
| 49 | procHighlightCode, | |
| 50 | ]; | |
| 51 | ||
| 52 | import { UNWRAP } from "@clo/lib/assert"; | |
| 53 | ||
| 54 | import * as highlight from "#src/file-viewer/highlight.ts"; | |
| 55 | import * as rules from "#src/file-viewer/rules.ts"; | |
| 56 | import type { Processor } from "../registry.ts"; |
src/file-viewer/indexer/registry.ts created+100| ... | ... | @@ -0,0 +1,100 @@ |
| 1 | // The processor registry. Each processor has a stable string `name` and an | |
| 2 | // integer `version`; bumping the version makes every applicable file re-run | |
| 3 | // that processor on the next sweep. This replaces the old scheme of hashing | |
| 4 | // `run.toString()` into a bitfield (rest in peace). | |
| 5 | // | |
| 6 | // This module is intentionally free of database imports so that the one-shot | |
| 7 | // database migration can reuse the definitions and applicability rules | |
| 8 | // before the new schema exists. | |
| 9 | export interface Processor { | |
| 10 | /** stable identifier, stored in the `processors` table. never rename. */ | |
| 11 | name: string; | |
| 12 | /** bump to re-run this processor on all applicable files */ | |
| 13 | version: number; | |
| 14 | /** human-readable label for progress display */ | |
| 15 | title: string; | |
| 16 | /** approximate cores used while running, for the scheduler */ | |
| 17 | cores?: number; | |
| 18 | /** false when a required tool (ffmpeg, ...) is missing on this host */ | |
| 19 | enable?: boolean; | |
| 20 | /** if set, only these extensions apply; otherwise all but `exclude` */ | |
| 21 | include?: Set<string>; | |
| 22 | /** extensions that do not apply (only when `include` is unset) */ | |
| 23 | exclude?: Set<string>; | |
| 24 | /** processor names that must complete before this one runs */ | |
| 25 | depends?: string[]; | |
| 26 | run(args: ProcessorRunArgs): Promise<void>; | |
| 27 | } | |
| 28 | ||
| 29 | export interface ProcessorRunArgs { | |
| 30 | path: Path; | |
| 31 | stat: fs.Stats; | |
| 32 | mediaFile: MediaFile; | |
| 33 | node: progress.Node; | |
| 34 | /** aborts when the file is deleted mid-run; pass to subprocesses */ | |
| 35 | signal: AbortSignal; | |
| 36 | } | |
| 37 | ||
| 38 | export const processors: readonly Processor[] = [ | |
| 39 | ...mediaProcessors, | |
| 40 | ...textProcessors, | |
| 41 | ...compressProcessors, | |
| 42 | ]; | |
| 43 | ||
| 44 | // integrity checks, evaluated once at import | |
| 45 | { | |
| 46 | const names = new Set<string>(); | |
| 47 | for (const p of processors) { | |
| 48 | if (names.has(p.name)) throw new Error(`duplicate processor: ${p.name}`); | |
| 49 | names.add(p.name); | |
| 50 | ASSERT(Number.isInteger(p.version) && p.version >= 1, p.name); | |
| 51 | for (const depend of p.depends ?? []) { | |
| 52 | if (depend === p.name) throw new Error(`${p.name} depends on itself`); | |
| 53 | if (!processors.some((o) => o.name === depend)) { | |
| 54 | throw new Error(`${p.name} depends on unknown '${depend}'`); | |
| 55 | } | |
| 56 | } | |
| 57 | } | |
| 58 | } | |
| 59 | ||
| 60 | /** | |
| 61 | * which processors apply to a file extension. extension must come from | |
| 62 | * `MediaFile.extensionNonEmpty`, lowercased. deterministic across hosts: | |
| 63 | * `enable` and the disable list do not affect applicability, only execution, | |
| 64 | * so `media_files.pending` means the same thing everywhere. | |
| 65 | */ | |
| 66 | export function applicableFor(ext: string): Processor[] { | |
| 67 | return processors.filter((p) => p.include ? p.include.has(ext) : !p.exclude?.has(ext)); | |
| 68 | } | |
| 69 | ||
| 70 | const disabledList = new Set( | |
| 71 | (process.env.CLOVER_PROCESSORS_DISABLE ?? "") | |
| 72 | .split(",") | |
| 73 | .map((x) => x.trim()) | |
| 74 | .filter(Boolean), | |
| 75 | ); | |
| 76 | for (const name of disabledList) { | |
| 77 | if (!processors.some((p) => p.name === name)) { | |
| 78 | console.warn(`CLOVER_PROCESSORS_DISABLE: unknown processor '${name}'`); | |
| 79 | } | |
| 80 | } | |
| 81 | ||
| 82 | /** false when the host cannot or should not execute this processor */ | |
| 83 | export function canExecute(p: Processor): boolean { | |
| 84 | return p.enable !== false && !disabledList.has(p.name); | |
| 85 | } | |
| 86 | ||
| 87 | export function byName(name: string): Processor | null { | |
| 88 | return processors.find((p) => p.name === name) ?? null; | |
| 89 | } | |
| 90 | ||
| 91 | import type * as progress from "@clo/lib/progress"; | |
| 92 | import type * as fs from "node:fs"; | |
| 93 | ||
| 94 | import type { Path } from "#sitegen/path"; | |
| 95 | import type { MediaFile } from "#src/file-viewer/models/MediaFile.ts"; | |
| 96 | ||
| 97 | import { ASSERT } from "@clo/lib/assert"; | |
| 98 | import { processors as compressProcessors } from "./processors/compress.ts"; | |
| 99 | import { processors as mediaProcessors } from "./processors/media.ts"; | |
| 100 | import { processors as textProcessors } from "./processors/text.ts"; |
src/file-viewer/indexer/scan.ts created+559| ... | ... | @@ -0,0 +1,559 @@ |
| 1 | // The scanner keeps `media_files` in sync with the raw file store and | |
| 2 | // schedules processors. It is used three ways: | |
| 3 | // - a full sweep over the entire store (boot + periodic) | |
| 4 | // - a targeted scan of one path (file watcher events) | |
| 5 | // - the `file-scan` cli wrapper for development machines | |
| 6 | // | |
| 7 | // Files become visible in the database the moment their metadata row is | |
| 8 | // written; processors run afterwards and trickle their outputs in. The | |
| 9 | // `pending` column on `media_files` tells the UI that more data is coming. | |
| 10 | // `scanPath` resolves once metadata is settled; processor jobs continue in | |
| 11 | // the background on the global queue (await `waitIdle` to block on them). | |
| 12 | // | |
| 13 | // A file that is still being written (a large SMB upload, for example) is | |
| 14 | // not touched until it has been stable for `settleMs`: no hashing, no exif | |
| 15 | // scrubbing, no processors. The scanner just waits and re-stats. | |
| 16 | const console = log.scoped("indexer"); | |
| 17 | ||
| 18 | export interface ScannerOptions { | |
| 19 | /** root of the raw file store */ | |
| 20 | root: Path; | |
| 21 | /** where the two persistent progress nodes live */ | |
| 22 | progress: progress.Ref; | |
| 23 | /** notified when database contents change, for the web-node pinger */ | |
| 24 | onChange?: (kind: ChangeKind) => void; | |
| 25 | /** called whenever all queued work (walks + processors) finishes */ | |
| 26 | onIdle?: () => void; | |
| 27 | /** a file must be unmodified for this long before it is indexed */ | |
| 28 | settleMs?: number; | |
| 29 | } | |
| 30 | ||
| 31 | /** `metadata` is urgent (new/removed files); `processed` is lazy. */ | |
| 32 | export type ChangeKind = "metadata" | "processed"; | |
| 33 | ||
| 34 | export class Scanner { | |
| 35 | root: Path; | |
| 36 | onChange: (kind: ChangeKind) => void; | |
| 37 | onIdle: () => void; | |
| 38 | settleMs: number; | |
| 39 | /** processor name -> `processors` table row id */ | |
| 40 | ids: Map<string, number>; | |
| 41 | ||
| 42 | walkQueue = new queue.PriorityQueue(10); | |
| 43 | // scrub+hash runs on its own small pool, never the global queue: encode | |
| 44 | // jobs hold global cores for minutes to hours, and a saturated pool would | |
| 45 | // starve metadata updates — files must become visible immediately even | |
| 46 | // mid-encode-storm. two slots bound disk thrash while staying responsive. | |
| 47 | hashQueue = new queue.PriorityQueue(2); | |
| 48 | #inFlight = 0; | |
| 49 | #idleWaiters: (() => void)[] = []; | |
| 50 | ||
| 51 | // the entire progress display is two permanent top-level groups: what is | |
| 52 | // being indexed right now, and which processors are running. passive | |
| 53 | // nodes vanish when idle. no per-batch wrapper nodes; it does not matter | |
| 54 | // whether work came from the boot sweep, the watcher, or /scan. | |
| 55 | indexNode: progress.Node; | |
| 56 | processNode: progress.Node; | |
| 57 | ||
| 58 | constructor(options: ScannerOptions) { | |
| 59 | this.root = options.root; | |
| 60 | this.onChange = options.onChange ?? (() => {}); | |
| 61 | this.onIdle = options.onIdle ?? (() => {}); | |
| 62 | this.settleMs = options.settleMs ?? 10_000; | |
| 63 | this.ids = ProcessorState.syncRegistry(registry.processors); | |
| 64 | if (!this.root.ifExistsSync()) { | |
| 65 | throw new Error(`file store ${this.root} is not mounted`); | |
| 66 | } | |
| 67 | this.indexNode = options.progress.start("indexing files", { | |
| 68 | passive: true, | |
| 69 | showTotal: false, | |
| 70 | }); | |
| 71 | this.processNode = options.progress.start("running processors", { | |
| 72 | passive: true, | |
| 73 | showTotal: false, | |
| 74 | }); | |
| 75 | this.processNode.sortChildren = (a, b) => { | |
| 76 | const ac = a.children.length > 0 ? 1 : 0; | |
| 77 | const bc = b.children.length > 0 ? 1 : 0; | |
| 78 | if (ac !== bc) return bc - ac; | |
| 79 | return a.text.localeCompare(b.text); | |
| 80 | }; | |
| 81 | } | |
| 82 | ||
| 83 | /** scan the whole store. resolves when all metadata rows are updated. */ | |
| 84 | sweep(): Promise<void> { | |
| 85 | return this.scanPath(this.root); | |
| 86 | } | |
| 87 | ||
| 88 | /** | |
| 89 | * scan one absolute path (file or directory). resolves when the subtree's | |
| 90 | * metadata is settled; processor jobs continue in the background. | |
| 91 | */ | |
| 92 | async scanPath(path: Path): Promise<void> { | |
| 93 | const ctx: ScanContext = { | |
| 94 | promises: new async.PromiseAggregator(), | |
| 95 | }; | |
| 96 | ctx.promises.push(this.#visit(path, ctx)); | |
| 97 | await ctx.promises.all(); | |
| 98 | } | |
| 99 | ||
| 100 | /** wait for all background processor jobs to finish */ | |
| 101 | async waitIdle(): Promise<void> { | |
| 102 | if (this.#inFlight === 0) return; | |
| 103 | await new Promise<void>((resolve) => this.#idleWaiters.push(resolve)); | |
| 104 | } | |
| 105 | ||
| 106 | #beginJob() { | |
| 107 | this.#inFlight += 1; | |
| 108 | } | |
| 109 | #endJob() { | |
| 110 | ASSERT(this.#inFlight > 0); | |
| 111 | this.#inFlight -= 1; | |
| 112 | if (this.#inFlight === 0) { | |
| 113 | const waiters = this.#idleWaiters; | |
| 114 | this.#idleWaiters = []; | |
| 115 | for (const w of waiters) w(); | |
| 116 | this.onIdle(); | |
| 117 | } | |
| 118 | } | |
| 119 | ||
| 120 | #visit = this.walkQueue.wrap(async (path: Path, ctx: ScanContext) => { | |
| 121 | const publicPath = toPublicPath(this.root, path); | |
| 122 | using node = this.indexNode.start(publicPath + " - stat"); | |
| 123 | ||
| 124 | let stat: fs.Stats; | |
| 125 | try { | |
| 126 | stat = await path.stat(); | |
| 127 | } catch (err) { | |
| 128 | if (error.code(err) !== "ENOENT") throw err; | |
| 129 | // deleted; watcher events often describe files already gone | |
| 130 | this.#removePath(publicPath); | |
| 131 | return; | |
| 132 | } | |
| 133 | ||
| 134 | const mediaFile = MediaFile.getByPath(publicPath); | |
| 135 | ||
| 136 | // the row may carry a different spelling of the same path (case-only | |
| 137 | // rename: same inode, same mtime, so no metadata pass runs). adopt the | |
| 138 | // on-disk spelling — only the parent's listing knows it; both the event | |
| 139 | // path and the stored path can be stale. | |
| 140 | if (mediaFile && mediaFile.id !== 0 && mediaFile.path !== publicPath) { | |
| 141 | await this.#reconcileCase(path, mediaFile); | |
| 142 | } | |
| 143 | ||
| 144 | if (stat.isDirectory()) { | |
| 145 | node.text = publicPath + " - reading"; | |
| 146 | const items = (await path.readDir()) | |
| 147 | .filter((child) => !skipBasename(child.base)) | |
| 148 | .map((child) => ( | |
| 149 | ctx.promises.push(this.#visit(child, ctx)), child.base | |
| 150 | )); | |
| 151 | ||
| 152 | // reconcile deletions against the database. the store is | |
| 153 | // case-insensitive, so compare names folded; a case-only rename is the | |
| 154 | // same file, not a delete + create. | |
| 155 | const names = new Set(items.map((name) => name.toLowerCase())); | |
| 156 | for (const child of mediaFile?.getChildren() ?? []) { | |
| 157 | if (names.has(child.basename.toLowerCase())) continue; | |
| 158 | this.#removeFile(child); | |
| 159 | } | |
| 160 | return; | |
| 161 | } | |
| 162 | ||
| 163 | if ( | |
| 164 | !mediaFile | |
| 165 | || stat.size !== mediaFile.size | |
| 166 | || stat.mtime.getTime() !== mediaFile.date.getTime() | |
| 167 | ) { | |
| 168 | // do not hold a walk queue slot while settling/hashing. concurrent | |
| 169 | // visits of the same path (watch events racing a sweep, or two case | |
| 170 | // spellings of one physical file) share one metadata update instead | |
| 171 | // of hashing — or worse, scrubbing — the file twice. one broken file | |
| 172 | // logs and moves on; it must not abort the surrounding scan. | |
| 173 | const flightKey = publicPath.toLowerCase(); | |
| 174 | let job = this.#metaInFlight.get(flightKey); | |
| 175 | if (!job) { | |
| 176 | job = this.#updateMetadata({ path, publicPath, stat, mediaFile }) | |
| 177 | .catch((err) => console.error(`indexing ${publicPath} failed:`, err)) | |
| 178 | .finally(() => this.#metaInFlight.delete(flightKey)); | |
| 179 | this.#metaInFlight.set(flightKey, job); | |
| 180 | } | |
| 181 | ctx.promises.push(job); | |
| 182 | } else { | |
| 183 | this.#queueProcessors({ path, stat, mediaFile }); | |
| 184 | } | |
| 185 | }); | |
| 186 | ||
| 187 | #metaInFlight = new Map<string, Promise<void>>(); | |
| 188 | ||
| 189 | async #reconcileCase(path: Path, mediaFile: MediaFile) { | |
| 190 | const parent = path.parent; | |
| 191 | if (!parent) return; | |
| 192 | let names: string[]; | |
| 193 | try { | |
| 194 | names = (await parent.readDir()).map((entry) => entry.base); | |
| 195 | } catch { | |
| 196 | return; | |
| 197 | } | |
| 198 | const folded = path.base.toLowerCase(); | |
| 199 | const trueName = names.find((name) => name.toLowerCase() === folded); | |
| 200 | if (!trueName) return; | |
| 201 | const truePath = toPublicPath(this.root, parent.join(trueName)); | |
| 202 | if (truePath === mediaFile.path) return; | |
| 203 | console.info(`case rename ${mediaFile.path} -> ${truePath}`); | |
| 204 | mediaFile.updatePath(truePath); | |
| 205 | mediaFile.getParent()?.markDirReindex(); | |
| 206 | this.onChange("metadata"); | |
| 207 | } | |
| 208 | ||
| 209 | #removePath(publicPath: string) { | |
| 210 | const row = MediaFile.getByPath(publicPath); | |
| 211 | if (!row || row.id === 0) return; | |
| 212 | this.#removeFile(row); | |
| 213 | } | |
| 214 | ||
| 215 | #removeFile(file: MediaFile) { | |
| 216 | const recursive = file.kind === MediaFileKind.directory | |
| 217 | ? [file, ...file.getRecursiveFileChildren()] | |
| 218 | : [file]; | |
| 219 | for (const deletion of recursive) { | |
| 220 | deletion.delete(); | |
| 221 | // kill in-flight processors; a deleted file must not keep encoding | |
| 222 | this.#fileAborts.get(deletion.id)?.abort(); | |
| 223 | } | |
| 224 | console.info(`deleted ${file.path}`); | |
| 225 | file.getParent()?.markDirReindex(); | |
| 226 | this.onChange("metadata"); | |
| 227 | } | |
| 228 | ||
| 229 | async #updateMetadata( | |
| 230 | { path, publicPath, stat, mediaFile }: { | |
| 231 | path: Path; | |
| 232 | publicPath: string; | |
| 233 | stat: fs.Stats; | |
| 234 | mediaFile: MediaFile | null; | |
| 235 | }, | |
| 236 | ) { | |
| 237 | const label = publicPath.slice(1); | |
| 238 | using node = this.indexNode.start(label); | |
| 239 | ||
| 240 | // hold off on everything until the file has stopped changing. | |
| 241 | const settled = await this.#waitUntilSettled(path, stat, node); | |
| 242 | if (settled === null) { | |
| 243 | this.#removePath(publicPath); | |
| 244 | return; | |
| 245 | } | |
| 246 | stat = settled; | |
| 247 | ||
| 248 | node.text = `${label} - waiting to hash`; | |
| 249 | const hash = await this.hashQueue.run({ | |
| 250 | cores: 1, | |
| 251 | run: async () => { | |
| 252 | if (await scrub.scrubLocationMetadata(path, stat, node)) { | |
| 253 | stat = await path.stat(); | |
| 254 | } | |
| 255 | node.text = `${label} - hashing`; | |
| 256 | return await hashFile(path); | |
| 257 | }, | |
| 258 | }); | |
| 259 | ||
| 260 | let date = stat.mtime; | |
| 261 | if ( | |
| 262 | mediaFile | |
| 263 | && mediaFile.date.getTime() < stat.mtime.getTime() | |
| 264 | && Date.now() - stat.mtime.getTime() < monthMilliseconds | |
| 265 | ) { | |
| 266 | date = mediaFile.date; | |
| 267 | console.warn( | |
| 268 | `M-time on ${publicPath} was likely corrupted. ${formatDate(mediaFile.date)} -> ${formatDate(stat.mtime)}`, | |
| 269 | ); | |
| 270 | } | |
| 271 | ||
| 272 | const contentChanged = !mediaFile || mediaFile.hash !== hash; | |
| 273 | mediaFile = MediaFile.createFile({ | |
| 274 | path: publicPath, | |
| 275 | date, | |
| 276 | hash, | |
| 277 | size: stat.size, | |
| 278 | duration: mediaFile?.duration ?? 0, | |
| 279 | dimensions: mediaFile?.dimensions ?? "", | |
| 280 | contents: mediaFile?.contents ?? "", | |
| 281 | }); | |
| 282 | if (contentChanged) ProcessorState.invalidateFile(mediaFile.id); | |
| 283 | ||
| 284 | mediaFile.getParent()?.markDirReindex(); | |
| 285 | this.onChange("metadata"); | |
| 286 | ||
| 287 | node.text = label; | |
| 288 | this.#queueProcessors({ path, stat, mediaFile }); | |
| 289 | } | |
| 290 | ||
| 291 | async #waitUntilSettled( | |
| 292 | path: Path, | |
| 293 | stat: fs.Stats, | |
| 294 | node: progress.Node, | |
| 295 | ): Promise<fs.Stats | null> { | |
| 296 | const baseText = node.text; | |
| 297 | let polls = 0; | |
| 298 | while (true) { | |
| 299 | const age = Date.now() - stat.mtime.getTime(); | |
| 300 | if (age >= this.settleMs) break; | |
| 301 | // mtimes in the future cannot settle; the corruption guard dates them | |
| 302 | if (age < -60_000) break; | |
| 303 | node.text = `${baseText} - waiting for upload to finish`; | |
| 304 | await async.delay(Math.min(this.settleMs, 2500)); | |
| 305 | if ((polls += 1) === 240) { // roughly ten minutes | |
| 306 | console.warn(`${path} has been unstable for a long time`); | |
| 307 | } | |
| 308 | try { | |
| 309 | stat = await path.stat(); | |
| 310 | } catch (err) { | |
| 311 | if (error.code(err) === "ENOENT") return null; | |
| 312 | throw err; | |
| 313 | } | |
| 314 | } | |
| 315 | node.text = baseText; | |
| 316 | return stat; | |
| 317 | } | |
| 318 | ||
| 319 | // files whose processor pipeline is currently queued or running. a second | |
| 320 | // scan of the file (sweep racing the watcher) must not double-queue jobs | |
| 321 | // or reset `pending` mid-flight. the abort controller cancels the | |
| 322 | // pipeline's subprocesses when the file is deleted. | |
| 323 | #activeProcessing = new Set<number>(); | |
| 324 | #fileAborts = new Map<number, AbortController>(); | |
| 325 | ||
| 326 | #queueProcessors( | |
| 327 | args: { | |
| 328 | path: Path; | |
| 329 | stat: fs.Stats; | |
| 330 | mediaFile: MediaFile; | |
| 331 | }, | |
| 332 | ) { | |
| 333 | const { mediaFile } = args; | |
| 334 | if (this.#activeProcessing.has(mediaFile.id)) return; | |
| 335 | const ext = mediaFile.extensionNonEmpty.toLowerCase(); | |
| 336 | const applicable = registry.applicableFor(ext); | |
| 337 | if (applicable.length === 0) { | |
| 338 | if (mediaFile.pending !== 0) mediaFile.setPending(0); | |
| 339 | return; | |
| 340 | } | |
| 341 | ||
| 342 | const states = ProcessorState.getStates(mediaFile.id); | |
| 343 | const needed = applicable.filter((p) => { | |
| 344 | const state = states.get(UNWRAP(this.ids.get(p.name))); | |
| 345 | return !state || state.version !== p.version | |
| 346 | || state.status === ProcessorState.ProcessorStatus.failed; | |
| 347 | }); | |
| 348 | if (mediaFile.pending !== needed.length) { | |
| 349 | mediaFile.setPending(needed.length); | |
| 350 | } | |
| 351 | if (needed.length === 0) return; | |
| 352 | ||
| 353 | // a processor can only run when the host has its tools and all of its | |
| 354 | // dependencies are either previously-done or also runnable now. | |
| 355 | const runnable = new Set<registry.Processor>(); | |
| 356 | let grew = true; | |
| 357 | while (grew) { | |
| 358 | grew = false; | |
| 359 | for (const p of needed) { | |
| 360 | if (runnable.has(p) || !registry.canExecute(p)) continue; | |
| 361 | const ok = (p.depends ?? []).every((depend) => { | |
| 362 | const dep = needed.find((o) => o.name === depend); | |
| 363 | return !dep || runnable.has(dep); | |
| 364 | }); | |
| 365 | if (ok) { | |
| 366 | runnable.add(p); | |
| 367 | grew = true; | |
| 368 | } | |
| 369 | } | |
| 370 | } | |
| 371 | if (runnable.size === 0) return; | |
| 372 | ||
| 373 | this.#activeProcessing.add(mediaFile.id); | |
| 374 | const abort = new AbortController(); | |
| 375 | this.#fileAborts.set(mediaFile.id, abort); | |
| 376 | const node = this.processNode.start(mediaFile.path.slice(1), { | |
| 377 | passive: true, | |
| 378 | showTotal: false, | |
| 379 | total: runnable.size, | |
| 380 | }); | |
| 381 | // the whole pipeline is accounted upfront: counting per-started-job | |
| 382 | // would let the in-flight count transiently hit zero between a | |
| 383 | // dependency finishing and its dependants starting. | |
| 384 | for (let i = 0; i < runnable.size; i += 1) this.#beginJob(); | |
| 385 | let remaining = runnable.size; | |
| 386 | const settleJob = () => { | |
| 387 | node.value += 1; | |
| 388 | this.#endJob(); | |
| 389 | if ((remaining -= 1) === 0) { | |
| 390 | node.end(); | |
| 391 | this.#activeProcessing.delete(mediaFile.id); | |
| 392 | this.#fileAborts.delete(mediaFile.id); | |
| 393 | } | |
| 394 | }; | |
| 395 | ||
| 396 | const jobs = [...runnable].map<ProcessJob>((processor) => ({ | |
| 397 | processor, | |
| 398 | after: [], | |
| 399 | needs: 0, | |
| 400 | done: false, | |
| 401 | })); | |
| 402 | for (const job of jobs) { | |
| 403 | for (const depend of job.processor.depends ?? []) { | |
| 404 | const dependJob = jobs.find((j) => j.processor.name === depend); | |
| 405 | if (dependJob) { | |
| 406 | dependJob.after.push(job); | |
| 407 | job.needs += 1; | |
| 408 | } | |
| 409 | } | |
| 410 | } | |
| 411 | ||
| 412 | // when a job fails, everything transitively depending on it is | |
| 413 | // abandoned: still pending, retried together on the next sweep. | |
| 414 | const abandon = (job: ProcessJob) => { | |
| 415 | for (const dependant of job.after) { | |
| 416 | if (dependant.done) continue; | |
| 417 | dependant.done = true; | |
| 418 | settleJob(); | |
| 419 | mediaFile.decPending(); | |
| 420 | abandon(dependant); | |
| 421 | } | |
| 422 | }; | |
| 423 | ||
| 424 | const start = (job: ProcessJob) => { | |
| 425 | queue.run({ | |
| 426 | cores: job.processor.cores ?? 0, | |
| 427 | run: () => this.#executeJob(job.processor, args, node, abort.signal), | |
| 428 | }).then(() => { | |
| 429 | job.done = true; | |
| 430 | settleJob(); | |
| 431 | for (const dependant of job.after) { | |
| 432 | ASSERT(dependant.needs > 0); | |
| 433 | dependant.needs -= 1; | |
| 434 | if (dependant.needs === 0 && !dependant.done) start(dependant); | |
| 435 | } | |
| 436 | }, () => { | |
| 437 | job.done = true; | |
| 438 | settleJob(); | |
| 439 | abandon(job); | |
| 440 | }); | |
| 441 | }; | |
| 442 | for (const job of jobs) if (job.needs === 0) start(job); | |
| 443 | } | |
| 444 | ||
| 445 | async #executeJob( | |
| 446 | processor: registry.Processor, | |
| 447 | { path, stat, mediaFile }: { | |
| 448 | path: Path; | |
| 449 | stat: fs.Stats; | |
| 450 | mediaFile: MediaFile; | |
| 451 | }, | |
| 452 | parent: progress.Node, | |
| 453 | signal: AbortSignal, | |
| 454 | ) { | |
| 455 | // deleted while this job sat in the queue: skip silently. the delete | |
| 456 | // already cleaned up `file_processors` and `pending`. | |
| 457 | const rowExists = () => MediaFile.getByPath(mediaFile.path)?.id === mediaFile.id; | |
| 458 | if (signal.aborted || !rowExists()) return; | |
| 459 | ||
| 460 | using node = parent.start(processor.title); | |
| 461 | const id = UNWRAP(this.ids.get(processor.name)); | |
| 462 | try { | |
| 463 | await processor.run({ path, stat, mediaFile, node, signal }); | |
| 464 | if (signal.aborted || !rowExists()) return; | |
| 465 | ProcessorState.recordResult( | |
| 466 | mediaFile.id, | |
| 467 | id, | |
| 468 | processor.version, | |
| 469 | ProcessorState.ProcessorStatus.done, | |
| 470 | ); | |
| 471 | mediaFile.decPending(); | |
| 472 | this.onChange("processed"); | |
| 473 | } catch (err: any) { | |
| 474 | if (signal.aborted) { | |
| 475 | console.info(`${processor.name} aborted on ${mediaFile.path} (deleted)`); | |
| 476 | throw err; | |
| 477 | } | |
| 478 | if (rowExists()) { | |
| 479 | const message = String(err?.stack ?? err).slice(0, 4000); | |
| 480 | ProcessorState.recordResult( | |
| 481 | mediaFile.id, | |
| 482 | id, | |
| 483 | processor.version, | |
| 484 | ProcessorState.ProcessorStatus.failed, | |
| 485 | message, | |
| 486 | ); | |
| 487 | mediaFile.decPending(); | |
| 488 | this.onChange("processed"); | |
| 489 | } | |
| 490 | console.error(`${processor.name} failed on ${mediaFile.path}:`, err); | |
| 491 | throw err; | |
| 492 | } | |
| 493 | } | |
| 494 | } | |
| 495 | ||
| 496 | interface ScanContext { | |
| 497 | promises: async.PromiseAggregator; | |
| 498 | } | |
| 499 | ||
| 500 | interface ProcessJob { | |
| 501 | processor: registry.Processor; | |
| 502 | after: ProcessJob[]; | |
| 503 | needs: number; | |
| 504 | done: boolean; | |
| 505 | } | |
| 506 | ||
| 507 | export function hashFile(path: Path): Promise<string> { | |
| 508 | return new Promise<string>((resolve, reject) => { | |
| 509 | const reader = fs.createReadStream(path.toString()); | |
| 510 | reader.on("error", reject); | |
| 511 | ||
| 512 | const hasher = crypto.createHash("sha1").setEncoding("hex"); | |
| 513 | hasher.on("error", reject); | |
| 514 | hasher.on("readable", () => resolve(hasher.read())); | |
| 515 | ||
| 516 | reader.pipe(hasher); | |
| 517 | }); | |
| 518 | } | |
| 519 | ||
| 520 | export function skipBasename(basename: string): boolean { | |
| 521 | // dot files must be incrementally tracked | |
| 522 | if (basename === ".dirsort") return false; | |
| 523 | if (basename === ".friends") return false; | |
| 524 | if (basename === ".date") return false; | |
| 525 | ||
| 526 | return ( | |
| 527 | basename.startsWith(".") | |
| 528 | || basename.startsWith("tmp.") | |
| 529 | || basename.toLowerCase() === "thumbs.db" | |
| 530 | || basename.toLowerCase() === "desktop.ini" | |
| 531 | ); | |
| 532 | } | |
| 533 | ||
| 534 | export function toPublicPath(root: Path, diskPath: Path) { | |
| 535 | if (diskPath.toString() === root.toString()) return "/"; | |
| 536 | return "/" | |
| 537 | + path.relative(root.toString(), diskPath.toString()).replaceAll("\\", "/"); | |
| 538 | } | |
| 539 | ||
| 540 | const monthMilliseconds = 30 * 24 * 60 * 60 * 1000; | |
| 541 | ||
| 542 | import * as crypto from "node:crypto"; | |
| 543 | import * as fs from "node:fs"; | |
| 544 | import * as path from "node:path"; | |
| 545 | ||
| 546 | import { Path } from "#sitegen/path"; | |
| 547 | import { ASSERT, UNWRAP } from "@clo/lib/assert"; | |
| 548 | import * as async from "@clo/lib/async"; | |
| 549 | import * as error from "@clo/lib/error"; | |
| 550 | import * as log from "@clo/lib/log"; | |
| 551 | import * as progress from "@clo/lib/progress"; | |
| 552 | import * as queue from "@clo/lib/queue"; | |
| 553 | import * as ts from "@clo/lib/ts"; | |
| 554 | ||
| 555 | import { formatDate } from "#src/file-viewer/format.ts"; | |
| 556 | import { MediaFile, MediaFileKind } from "#src/file-viewer/models/MediaFile.ts"; | |
| 557 | import * as ProcessorState from "#src/file-viewer/models/ProcessorState.ts"; | |
| 558 | import * as registry from "./registry.ts"; | |
| 559 | import * as scrub from "./scrub.ts"; |
src/file-viewer/indexer/scrub.ts created+160| ... | ... | @@ -0,0 +1,160 @@ |
| 1 | // gps/location metadata removal, ported from the old file-scan.ts. runs | |
| 2 | // before hashing so the stored hash always reflects the scrubbed file. | |
| 3 | // requires the file to be fully written (the scanner's stability gate runs | |
| 4 | // first); modifies the file in place while preserving timestamps. | |
| 5 | const exiftoolBin = testProgram("exiftool"); | |
| 6 | ||
| 7 | // `-ee` on an iphone video dumps per-frame embedded metadata, which easily | |
| 8 | // exceeds execFile's default 1MB stdout buffer | |
| 9 | const execOptions = { maxBuffer: 64 * 1024 * 1024 }; | |
| 10 | ||
| 11 | export async function scrubLocationMetadata( | |
| 12 | path: Path, | |
| 13 | stats: fs.Stats, | |
| 14 | progress: progress.Ref, | |
| 15 | ): Promise<boolean> { | |
| 16 | using _ = progress.start("scrub exif metadata"); | |
| 17 | const ext = path.ext.toLowerCase(); | |
| 18 | if (!rules.extsScrubExif.has(ext)) return false; | |
| 19 | if (!exiftoolBin) { | |
| 20 | warnMissingExiftool(); | |
| 21 | return false; | |
| 22 | } | |
| 23 | ||
| 24 | let hasLocation = false; | |
| 25 | let args: string[] = []; | |
| 26 | ||
| 27 | // Check for location metadata based on file type | |
| 28 | const tempOutput = UNWRAP(path.parent).join(`.tmp.${path.base}`); | |
| 29 | switch (ext) { | |
| 30 | case ".jpg": | |
| 31 | case ".jpeg": | |
| 32 | case ".png": | |
| 33 | const { stdout: gpsCheck } = await subprocess.exec("exiftool", [ | |
| 34 | "-gps:all", | |
| 35 | path.toString(), | |
| 36 | ], execOptions); | |
| 37 | hasLocation = gpsCheck.trim().length > 0; | |
| 38 | args = ["-gps:all=", path.toString(), "-o", tempOutput.toString()]; | |
| 39 | break; | |
| 40 | case ".mov": | |
| 41 | case ".mp4": | |
| 42 | const { stdout: videoCheck } = await subprocess.exec("exiftool", [ | |
| 43 | "-ee", | |
| 44 | "-G3", | |
| 45 | "-s", | |
| 46 | path.toString(), | |
| 47 | ], execOptions); | |
| 48 | hasLocation = videoCheck.includes("GPS") | |
| 49 | || videoCheck.includes("Location"); | |
| 50 | args = [ | |
| 51 | "-gps:all=", | |
| 52 | "-xmp:all=", | |
| 53 | path.toString(), | |
| 54 | "-o", | |
| 55 | tempOutput.toString(), | |
| 56 | ]; | |
| 57 | break; | |
| 58 | case ".m4a": | |
| 59 | const { stdout: m4aCheck } = await subprocess.exec("exiftool", [ | |
| 60 | "-ee", | |
| 61 | "-G3", | |
| 62 | "-s", | |
| 63 | path.toString(), | |
| 64 | ], execOptions); | |
| 65 | hasLocation = m4aCheck.includes("GPS") | |
| 66 | || m4aCheck.includes("Location") | |
| 67 | || m4aCheck.includes("Filename") | |
| 68 | || m4aCheck.includes("Title"); | |
| 69 | ||
| 70 | if (hasLocation) { | |
| 71 | args = [ | |
| 72 | "-gps:all=", | |
| 73 | "-location:all=", | |
| 74 | "-filename:all=", | |
| 75 | "-title=", | |
| 76 | "-m4a:all=", | |
| 77 | path.toString(), | |
| 78 | "-o", | |
| 79 | tempOutput.toString(), | |
| 80 | ]; | |
| 81 | } | |
| 82 | break; | |
| 83 | } | |
| 84 | ||
| 85 | const accessTime = stats.atime; | |
| 86 | const modTime = stats.mtime; | |
| 87 | ||
| 88 | let backup: Path | null = null; | |
| 89 | try { | |
| 90 | if (hasLocation) { | |
| 91 | // Prepare a backup. content-only copy: fs.copyFile's metadata | |
| 92 | // preservation gets EPERM'd on the NAS datasets, plain writes do not. | |
| 93 | const tmp = UNWRAP(path.parent).join(`.tmp.backup.${path.base}`); | |
| 94 | await stream.promises.pipeline( | |
| 95 | fs.createReadStream(path.toString()), | |
| 96 | fs.createWriteStream(tmp.toString()), | |
| 97 | ); | |
| 98 | await fsp.utimes(tmp.toString(), accessTime, modTime); | |
| 99 | backup = tmp; | |
| 100 | ||
| 101 | // a leftover temp from a crashed run makes exiftool refuse to write | |
| 102 | await tempOutput.delete({ force: true }); | |
| 103 | ||
| 104 | // Remove metadata | |
| 105 | await subprocess.exec("exiftool", args, execOptions); | |
| 106 | if (!tempOutput.ifExistsSync()) { | |
| 107 | throw new Error(`Failed to create output file: ${tempOutput}`); | |
| 108 | } | |
| 109 | ||
| 110 | // Restore original timestamps | |
| 111 | await fsp.rename(tempOutput.toString(), path.toString()); | |
| 112 | await fsp.utimes(path.toString(), accessTime, modTime); | |
| 113 | ||
| 114 | // Backup is no longer needed | |
| 115 | await fsp.unlink(backup.toString()); | |
| 116 | ||
| 117 | console.info(`Scrubbed location metadata in ${path}`); | |
| 118 | return true; | |
| 119 | } | |
| 120 | } catch (error) { | |
| 121 | // restore is best-effort: a concurrent scrub of the same physical file | |
| 122 | // (case-insensitive store) may have already consumed the backup | |
| 123 | if (backup && fs.existsSync(backup.toString())) { | |
| 124 | await fsp.rename(backup.toString(), path.toString()); | |
| 125 | } | |
| 126 | if (fs.existsSync(tempOutput.toString())) { | |
| 127 | await fsp.unlink(tempOutput.toString()); | |
| 128 | } | |
| 129 | throw error; | |
| 130 | } | |
| 131 | ||
| 132 | return false; | |
| 133 | } | |
| 134 | ||
| 135 | let warnedMissingExiftool = false; | |
| 136 | function warnMissingExiftool() { | |
| 137 | if (warnedMissingExiftool) return; | |
| 138 | warnedMissingExiftool = true; | |
| 139 | console.warn( | |
| 140 | "exiftool is not installed; skipping gps metadata scrubbing entirely", | |
| 141 | ); | |
| 142 | } | |
| 143 | ||
| 144 | function testProgram(name: string) { | |
| 145 | // spawnSync does not throw on a missing binary; it reports `error` | |
| 146 | const result = child_process.spawnSync(name, ["-ver"]); | |
| 147 | return result.error ? null : name; | |
| 148 | } | |
| 149 | ||
| 150 | import * as child_process from "node:child_process"; | |
| 151 | import * as fs from "node:fs"; | |
| 152 | import * as fsp from "node:fs/promises"; | |
| 153 | import * as stream from "node:stream"; | |
| 154 | ||
| 155 | import { Path } from "#sitegen/path"; | |
| 156 | import { UNWRAP } from "@clo/lib/assert"; | |
| 157 | import * as progress from "@clo/lib/progress"; | |
| 158 | import * as subprocess from "@clo/lib/subprocess"; | |
| 159 | ||
| 160 | import * as rules from "#src/file-viewer/rules.ts"; |
src/file-viewer/indexer/service.ts created+297| ... | ... | @@ -0,0 +1,297 @@ |
| 1 | // The indexer service: owns the singleton progress root that all scanning | |
| 2 | // and processing operations report into, watches the file store, runs | |
| 3 | // periodic full sweeps, and pings web nodes when the database changes. | |
| 4 | // | |
| 5 | // One progress root for the lifetime of the process; it is never ended. | |
| 6 | // `GET /progress` on the source of truth attaches encoders to it, which | |
| 7 | // snapshot current state on connect and then stream deltas. | |
| 8 | const console = log.scoped("indexer"); | |
| 9 | ||
| 10 | /** the singleton progress root. */ | |
| 11 | export const root = new progress.Root(); | |
| 12 | ||
| 13 | /** bumped on every database mutation; `GET /db` uses this to invalidate */ | |
| 14 | export let generation = 0; | |
| 15 | ||
| 16 | // -- database change events -- | |
| 17 | // consumers (the /db/events sse route) subscribe here; the indexer emits a | |
| 18 | // debounced beacon after changes. anything else that swaps the database | |
| 19 | // (e.g. the /reload route) can emit through `emitDbChange`. | |
| 20 | const dbSubscribers = new Set<(generation: number) => void>(); | |
| 21 | ||
| 22 | /** subscribe to database-changed beacons. dispose to unsubscribe. */ | |
| 23 | export function subscribeDbChange(fn: (generation: number) => void): ts.Dispose { | |
| 24 | dbSubscribers.add(fn); | |
| 25 | return ts.defer(() => void dbSubscribers.delete(fn)); | |
| 26 | } | |
| 27 | ||
| 28 | /** notify all subscribers that the database changed */ | |
| 29 | export function emitDbChange() { | |
| 30 | generation += 1; | |
| 31 | for (const fn of dbSubscribers) fn(generation); | |
| 32 | } | |
| 33 | ||
| 34 | export interface ServiceOptions { | |
| 35 | /** raw file store root. default: paths.rawFileRoot */ | |
| 36 | root?: string; | |
| 37 | /** full sweep interval. default: CLOVER_SCAN_INTERVAL or 24h */ | |
| 38 | sweepIntervalMs?: number; | |
| 39 | /** stability window before indexing a changed file */ | |
| 40 | settleMs?: number; | |
| 41 | /** disable the file watcher (CLOVER_WATCH=0) */ | |
| 42 | watch?: boolean; | |
| 43 | } | |
| 44 | ||
| 45 | let started = false; | |
| 46 | ||
| 47 | export function start(options: ServiceOptions = {}) { | |
| 48 | ASSERT(!started, "indexer service started twice"); | |
| 49 | started = true; | |
| 50 | ||
| 51 | const cores = Number(process.env.CLOVER_INDEX_CORES ?? 0); | |
| 52 | if (cores > 0) queue.setConcurrency(cores); | |
| 53 | ||
| 54 | const service = new Service(options); | |
| 55 | service.init(); | |
| 56 | return service; | |
| 57 | } | |
| 58 | ||
| 59 | export class Service { | |
| 60 | scanner: Scanner; | |
| 61 | sweepIntervalMs: number; | |
| 62 | watchEnabled: boolean; | |
| 63 | watcher: fs.FSWatcher | null = null; | |
| 64 | ||
| 65 | // paths that changed according to the watcher, waiting to be scanned | |
| 66 | #dirty = new Map<string, number>(); | |
| 67 | #dirtyTimer: NodeJS.Timeout | null = null; | |
| 68 | #sweeping = false; | |
| 69 | ||
| 70 | constructor(options: ServiceOptions) { | |
| 71 | this.scanner = new Scanner({ | |
| 72 | root: Path.resolve(options.root ?? paths.rawFileRoot), | |
| 73 | progress: root, | |
| 74 | settleMs: options.settleMs | |
| 75 | ?? Number(process.env.CLOVER_SETTLE_MS ?? 10_000), | |
| 76 | onChange: (kind) => this.#onDbChange(kind), | |
| 77 | onIdle: () => this.#onIdle(), | |
| 78 | }); | |
| 79 | this.sweepIntervalMs = options.sweepIntervalMs | |
| 80 | ?? parseDuration(process.env.CLOVER_SCAN_INTERVAL ?? "24h"); | |
| 81 | this.watchEnabled = options.watch ?? process.env.CLOVER_WATCH !== "0"; | |
| 82 | } | |
| 83 | ||
| 84 | init() { | |
| 85 | void this.sweep("boot"); | |
| 86 | setInterval(() => void this.sweep("periodic"), this.sweepIntervalMs) | |
| 87 | .unref(); | |
| 88 | if (this.watchEnabled) this.#startWatcher(); | |
| 89 | console.info( | |
| 90 | `indexer service started (root: ${this.scanner.root}, ` | |
| 91 | + `sweep every ${string.formatDurationLetters(this.sweepIntervalMs / 1000)}, ` | |
| 92 | + `watch: ${this.watchEnabled})`, | |
| 93 | ); | |
| 94 | } | |
| 95 | ||
| 96 | /** run a full sweep (or a subtree scan when `subPath` is given) */ | |
| 97 | async sweep(reason: string, subPath?: string) { | |
| 98 | if (this.#sweeping && !subPath) { | |
| 99 | console.warn(`skipping ${reason} sweep; one is already running`); | |
| 100 | return; | |
| 101 | } | |
| 102 | try { | |
| 103 | if (!subPath) this.#sweeping = true; | |
| 104 | const target = subPath | |
| 105 | ? this.scanner.root.join("." + path.posix.normalize("/" + subPath)) | |
| 106 | : this.scanner.root; | |
| 107 | await this.scanner.scanPath(target); | |
| 108 | this.#runDirMeta(); | |
| 109 | if (!subPath) await this.#maintenance(); | |
| 110 | } catch (err) { | |
| 111 | console.error(`${reason} scan failed:`, err); | |
| 112 | } finally { | |
| 113 | if (!subPath) this.#sweeping = false; | |
| 114 | } | |
| 115 | } | |
| 116 | ||
| 117 | // delete derived assets whose last referencing file is gone, plus tmp | |
| 118 | // directories left behind by crashed producers. folded in from the old | |
| 119 | // `file-trim` binary. | |
| 120 | async #maintenance() { | |
| 121 | using node = this.scanner.indexNode.start("maintenance"); | |
| 122 | const orphaned = derived.findOrphanedRoots(); | |
| 123 | for (const orphan of orphaned) { | |
| 124 | node.text = `delete orphaned ${orphan.key}`; | |
| 125 | await derived.deleteRootFiles(orphan); | |
| 126 | derived.deleteRoot(orphan); | |
| 127 | } | |
| 128 | const cleaned = await derived.cleanAbandonedTmp(); | |
| 129 | if (orphaned.length + cleaned > 0) { | |
| 130 | console.info( | |
| 131 | `maintenance: ${orphaned.length} orphaned roots, ${cleaned} stale tmp dirs`, | |
| 132 | ); | |
| 133 | this.#onDbChange("processed"); | |
| 134 | } | |
| 135 | } | |
| 136 | ||
| 137 | // -- file watcher -- | |
| 138 | ||
| 139 | #startWatcher() { | |
| 140 | try { | |
| 141 | this.watcher = fs.watch( | |
| 142 | this.scanner.root.toString(), | |
| 143 | { recursive: true }, | |
| 144 | (_event, subPath) => subPath && this.#markDirty(subPath), | |
| 145 | ); | |
| 146 | this.watcher.on("error", (err) => { | |
| 147 | console.error("file watcher died, relying on periodic sweeps:", err); | |
| 148 | this.watcher = null; | |
| 149 | }); | |
| 150 | } catch (err) { | |
| 151 | console.error("file watcher unavailable, relying on sweeps:", err); | |
| 152 | this.watcher = null; | |
| 153 | } | |
| 154 | } | |
| 155 | ||
| 156 | #markDirty(subPath: string) { | |
| 157 | subPath = subPath.replaceAll("\\", "/"); | |
| 158 | // ignore events for paths the scanner would skip anyway (.DS_Store and | |
| 159 | // friends), but also for anything inside a skipped directory. | |
| 160 | if (subPath.split("/").some((part) => skipBasename(part))) return; | |
| 161 | this.#dirty.set(subPath, Date.now()); | |
| 162 | this.#dirtyTimer ??= setTimeout(() => { | |
| 163 | this.#dirtyTimer = null; | |
| 164 | void this.#flushDirty(); | |
| 165 | }, watchDebounceMs); | |
| 166 | } | |
| 167 | ||
| 168 | // targets currently being scanned. a path that keeps emitting events (a | |
| 169 | // large upload sitting in the settle gate) must not pile up concurrent | |
| 170 | // scans of itself; it is re-checked once the in-flight scan completes. | |
| 171 | #scanning = new Map<string, Promise<void>>(); | |
| 172 | ||
| 173 | async #flushDirty() { | |
| 174 | if (this.#dirty.size === 0) return; | |
| 175 | const all = [...this.#dirty.keys()]; | |
| 176 | this.#dirty.clear(); | |
| 177 | ||
| 178 | // coalesce: if a parent path is queued, skip its children. scanning is | |
| 179 | // recursive for directories, so the parent covers them. comparisons are | |
| 180 | // case-insensitive to match the file store's semantics; rename events | |
| 181 | // can spell the same physical path two ways. | |
| 182 | const sorted = all.sort(); | |
| 183 | const targets: string[] = []; | |
| 184 | for (const p of sorted) { | |
| 185 | const lp = p.toLowerCase(); | |
| 186 | const covered = targets.some((t) => { | |
| 187 | const lt = t.toLowerCase(); | |
| 188 | return lp === lt || lp.startsWith(lt + "/"); | |
| 189 | }); | |
| 190 | if (!covered) targets.push(p); | |
| 191 | } | |
| 192 | ||
| 193 | const fresh: string[] = []; | |
| 194 | for (const target of targets) { | |
| 195 | // defer paths already being scanned; the tail of the flush that owns | |
| 196 | // them re-arms the timer, which picks these back up. | |
| 197 | if (this.#scanning.has(target.toLowerCase())) { | |
| 198 | this.#dirty.set(target, Date.now()); | |
| 199 | } else fresh.push(target); | |
| 200 | } | |
| 201 | if (fresh.length === 0) return; | |
| 202 | ||
| 203 | // parallel: one file mid-upload (settling) must not block the others | |
| 204 | await Promise.all(fresh.map((target) => { | |
| 205 | const job = this.scanner | |
| 206 | .scanPath(this.scanner.root.join(target)) | |
| 207 | .catch((err) => console.error(`watch scan of ${target} failed:`, err)) | |
| 208 | .finally(() => { | |
| 209 | this.#scanning.delete(target.toLowerCase()); | |
| 210 | }); | |
| 211 | this.#scanning.set(target.toLowerCase(), job); | |
| 212 | return job; | |
| 213 | })); | |
| 214 | this.#runDirMeta(); | |
| 215 | ||
| 216 | // events that arrived during the scans (including deferred re-checks of | |
| 217 | // the paths scanned just now) get their own flush. | |
| 218 | if (this.#dirty.size > 0) { | |
| 219 | this.#dirtyTimer ??= setTimeout(() => { | |
| 220 | this.#dirtyTimer = null; | |
| 221 | void this.#flushDirty(); | |
| 222 | }, watchDebounceMs); | |
| 223 | } | |
| 224 | } | |
| 225 | ||
| 226 | // -- directory metadata -- | |
| 227 | ||
| 228 | // the dir meta pass also runs after processors finish, since readme.txt | |
| 229 | // contents arrive via the text-contents processor. | |
| 230 | #onIdle() { | |
| 231 | if (this.#dirMetaTimer) return; | |
| 232 | this.#dirMetaTimer = setTimeout(() => { | |
| 233 | this.#dirMetaTimer = null; | |
| 234 | this.#runDirMeta(); | |
| 235 | }, 1000); | |
| 236 | this.#dirMetaTimer.unref(); | |
| 237 | } | |
| 238 | #dirMetaTimer: NodeJS.Timeout | null = null; | |
| 239 | ||
| 240 | #runDirMeta() { | |
| 241 | try { | |
| 242 | if (dirmeta.run(this.scanner.indexNode)) this.#onDbChange("metadata"); | |
| 243 | } catch (err) { | |
| 244 | console.error("directory metadata pass failed:", err); | |
| 245 | } | |
| 246 | } | |
| 247 | ||
| 248 | // -- database change beacons -- | |
| 249 | // consumers subscribe via /db/events and pull /db when beaconed. metadata | |
| 250 | // changes (new files) flush fast; processor completions are batched | |
| 251 | // coarsely since they arrive in bursts during encodes. | |
| 252 | ||
| 253 | #beaconTimer: NodeJS.Timeout | null = null; | |
| 254 | #beaconDeadline = Infinity; | |
| 255 | ||
| 256 | #onDbChange(kind: ChangeKind) { | |
| 257 | const delay = kind === "metadata" ? metadataBeaconMs : processedBeaconMs; | |
| 258 | const deadline = Date.now() + delay; | |
| 259 | if (deadline < this.#beaconDeadline) { | |
| 260 | this.#beaconDeadline = deadline; | |
| 261 | if (this.#beaconTimer) clearTimeout(this.#beaconTimer); | |
| 262 | this.#beaconTimer = setTimeout(() => { | |
| 263 | this.#beaconTimer = null; | |
| 264 | this.#beaconDeadline = Infinity; | |
| 265 | emitDbChange(); | |
| 266 | }, delay); | |
| 267 | this.#beaconTimer.unref(); | |
| 268 | } | |
| 269 | } | |
| 270 | } | |
| 271 | ||
| 272 | export function parseDuration(text: string): number { | |
| 273 | const match = text.match(/^(\d+(?:\.\d+)?)\s*(ms|s|m|h|d)?$/); | |
| 274 | if (!match) throw new Error(`cannot parse duration: ${JSON.stringify(text)}`); | |
| 275 | const scale = { ms: 1, s: 1000, m: 60_000, h: 3_600_000, d: 86_400_000 }; | |
| 276 | return Number(match[1]) * scale[(match[2] ?? "ms") as keyof typeof scale]; | |
| 277 | } | |
| 278 | ||
| 279 | const watchDebounceMs = 1000; | |
| 280 | const metadataBeaconMs = 2_000; | |
| 281 | const processedBeaconMs = 30_000; | |
| 282 | ||
| 283 | import * as fs from "node:fs"; | |
| 284 | import * as path from "node:path"; | |
| 285 | ||
| 286 | import { Path } from "#sitegen/path"; | |
| 287 | import { ASSERT } from "@clo/lib/assert"; | |
| 288 | import * as log from "@clo/lib/log"; | |
| 289 | import * as progress from "@clo/lib/progress"; | |
| 290 | import * as queue from "@clo/lib/queue"; | |
| 291 | import * as string from "@clo/lib/string"; | |
| 292 | import * as ts from "@clo/lib/ts"; | |
| 293 | ||
| 294 | import * as derived from "#src/file-viewer/models/derived.ts"; | |
| 295 | import * as paths from "#src/file-viewer/paths.ts"; | |
| 296 | import * as dirmeta from "./dirmeta.ts"; | |
| 297 | import { type ChangeKind, Scanner, skipBasename } from "./scan.ts"; |
src/file-viewer/models/FilePermissions.ts+7-5| ... | ... | @@ -32,21 +32,23 @@ export class FilePermissions { |
| 32 | 32 | } |
| 33 | 33 | } |
| 34 | 34 | |
| 35 | // comparisons fold case to match the file store: a permission on | |
| 36 | // "/2026/friends" must also gate a request spelled "/2026/Friends" | |
| 35 | 37 | const getByPrefixQuery = db.prepare< |
| 36 | 38 | [prefix: string], |
| 37 | 39 | Pick<FilePermissions, "allow"> |
| 38 | 40 | >(/* SQL */ ` |
| 39 | select allow | |
| 40 | from permissions | |
| 41 | where ? glob prefix || '*' | |
| 42 | order by length(prefix) desc | |
| 41 | select allow | |
| 42 | from permissions | |
| 43 | where lower(?) glob lower(prefix) || '*' | |
| 44 | order by length(prefix) desc | |
| 43 | 45 | limit 1; |
| 44 | 46 | `); |
| 45 | 47 | const getExactQuery = db.prepare< |
| 46 | 48 | [file: string], |
| 47 | 49 | Pick<FilePermissions, "allow"> |
| 48 | 50 | >(/* SQL */ ` |
| 49 | select allow from permissions where ? == prefix | |
| 51 | select allow from permissions where ? == prefix collate nocase | |
| 50 | 52 | `); |
| 51 | 53 | |
| 52 | 54 | const insertQuery = db.prepare<[{ prefix: string; allow: number }]>(/* SQL */ ` |
src/file-viewer/models/MediaFile.ts+86-53| ... | ... | @@ -5,7 +5,7 @@ db.table( |
| 5 | 5 | create table media_files ( |
| 6 | 6 | id integer primary key autoincrement, |
| 7 | 7 | parent_id integer, |
| 8 | path text unique, | |
| 8 | path text, | |
| 9 | 9 | kind integer not null, |
| 10 | 10 | timestamp integer not null, |
| 11 | 11 | timestamp_updated integer not null default current_timestamp, |
| ... | ... | @@ -15,18 +15,20 @@ db.table( |
| 15 | 15 | dimensions text not null default "", |
| 16 | 16 | contents text not null, |
| 17 | 17 | dirsort text, |
| 18 | processed integer not null, | |
| 19 | processors text not null default "", | |
| 20 | foreign key (parent_id) references media_files(id) | |
| 18 | config text not null default "", | |
| 19 | dir_reindex integer not null default 0, | |
| 20 | pending integer not null default 0, | |
| 21 | foreign key (parent_id) references media_files(id) on delete cascade | |
| 21 | 22 | ); |
| 22 | -- index for quickly looking up files by path | |
| 23 | create index media_files_path on media_files (path); | |
| 23 | -- path lookups fold case: the underlying file stores (zfs smb datasets, | |
| 24 | -- apfs) are case-insensitive, so two case spellings are one file | |
| 25 | create unique index media_files_path on media_files (path collate nocase); | |
| 24 | 26 | -- index for quickly looking up children |
| 25 | 27 | create index media_files_parent_id on media_files (parent_id); |
| 26 | 28 | -- index for quickly looking up recursive file children |
| 27 | 29 | create index media_files_file_children on media_files (kind, path); |
| 28 | -- index for finding directories that need to be processed | |
| 29 | create index media_files_directory_processed on media_files (kind, processed); | |
| 30 | -- index for finding directories that need re-indexing | |
| 31 | create index media_files_dir_reindex on media_files (kind, dir_reindex); | |
| 30 | 32 | `, |
| 31 | 33 | ); |
| 32 | 34 | |
| ... | ... | @@ -73,14 +75,18 @@ export class MediaFile { |
| 73 | 75 | /** in bytes */ |
| 74 | 76 | size!: number; |
| 75 | 77 | /** |
| 76 | * 0 - not processed | |
| 77 | * non-zero - processed | |
| 78 | * | |
| 79 | * file: a bit-field of the processors. | |
| 80 | * directory: this is for re-indexing contents | |
| 78 | * For directories, a JSON-encoded object derived from special files | |
| 79 | * (`.date` sets hideChildrenDates). Empty string otherwise. | |
| 80 | */ | |
| 81 | config!: string; | |
| 82 | /** for directories: 1 when the metadata pass must revisit this dir */ | |
| 83 | dir_reindex!: number; | |
| 84 | /** | |
| 85 | * number of processors queued or running for this file. when zero, all | |
| 86 | * derived data (duration, dimensions, contents, derived assets) is as | |
| 87 | * complete as it will get. the UI uses this to show "still processing". | |
| 81 | 88 | */ |
| 82 | processed!: number; | |
| 83 | processors!: string; | |
| 89 | pending!: number; | |
| 84 | 90 | |
| 85 | 91 | // -- instance ops -- |
| 86 | 92 | get date() { |
| ... | ... | @@ -134,14 +140,20 @@ export class MediaFile { |
| 134 | 140 | ASSERT(result.kind === MediaFileKind.directory); |
| 135 | 141 | return result; |
| 136 | 142 | } |
| 137 | setProcessed(processed: number) { | |
| 138 | setProcessedQuery.run({ id: this.id, processed }); | |
| 139 | this.processed = processed; | |
| 143 | setConfig(config: string) { | |
| 144 | setConfigQuery.run({ id: this.id, config }); | |
| 145 | this.config = config; | |
| 146 | } | |
| 147 | markDirReindex() { | |
| 148 | markDirReindexQuery.run(this.id); | |
| 149 | this.dir_reindex = 1; | |
| 150 | } | |
| 151 | setPending(pending: number) { | |
| 152 | setPendingQuery.run({ id: this.id, pending }); | |
| 153 | this.pending = pending; | |
| 140 | 154 | } |
| 141 | setProcessors(processed: number, processors: string) { | |
| 142 | setProcessorsQuery.run({ id: this.id, processed, processors }); | |
| 143 | this.processed = processed; | |
| 144 | this.processors = processors; | |
| 155 | decPending() { | |
| 156 | this.pending = decPendingQuery.getNonNull(this.id).pending; | |
| 145 | 157 | } |
| 146 | 158 | setDuration(duration: number) { |
| 147 | 159 | setDurationQuery.run({ id: this.id, duration }); |
| ... | ... | @@ -162,6 +174,14 @@ export class MediaFile { |
| 162 | 174 | delete() { |
| 163 | 175 | deleteCascadeQuery.run({ id: this.id }); |
| 164 | 176 | } |
| 177 | /** adopt a new spelling of the same path (the file stores fold case) */ | |
| 178 | updatePath(newPath: string) { | |
| 179 | if (this.kind === MediaFileKind.directory) { | |
| 180 | updatePathPrefixQuery.run({ old: this.path + "/", new: newPath + "/" }); | |
| 181 | } | |
| 182 | updatePathQuery.run({ id: this.id, path: newPath }); | |
| 183 | this.path = newPath; | |
| 184 | } | |
| 165 | 185 | |
| 166 | 186 | // -- static ops -- |
| 167 | 187 | static getByPath(filePath: string): MediaFile | null { |
| ... | ... | @@ -179,7 +199,9 @@ export class MediaFile { |
| 179 | 199 | contents: "the file scanner has not been run yet", |
| 180 | 200 | dirsort: null, |
| 181 | 201 | size: 0, |
| 182 | processed: 1, | |
| 202 | config: "", | |
| 203 | dir_reindex: 0, | |
| 204 | pending: 0, | |
| 183 | 205 | }); |
| 184 | 206 | } |
| 185 | 207 | return null; |
| ... | ... | @@ -269,9 +291,6 @@ export class MediaFile { |
| 269 | 291 | size, |
| 270 | 292 | }); |
| 271 | 293 | } |
| 272 | static setProcessed(id: number, processed: number) { | |
| 273 | setProcessedQuery.run({ id, processed }); | |
| 274 | } | |
| 275 | 294 | static createOrUpdateDirectory(dirPath: string) { |
| 276 | 295 | const id = MediaFile.getOrPutDirectoryId(dirPath); |
| 277 | 296 | return updateDirectoryQuery.get(id); |
| ... | ... | @@ -323,15 +342,16 @@ const createDirectoryQuery = db.prepare< |
| 323 | 342 | /* SQL */ ` |
| 324 | 343 | insert into media_files ( |
| 325 | 344 | path, parent_id, kind, timestamp, timestamp_updated, hash, |
| 326 | size, duration, dimensions, contents, dirsort, processed) | |
| 345 | size, duration, dimensions, contents, dirsort, dir_reindex) | |
| 327 | 346 | values ( |
| 328 | $path, $parentId, ${MediaFileKind.directory}, 0, $time, '', | |
| 329 | 0, 0, '', '', '', 0) | |
| 347 | $path, $parentId, ${MediaFileKind.directory}, 0, $time, '', | |
| 348 | 0, 0, '', '', '', 1) | |
| 330 | 349 | returning id; |
| 331 | 350 | `, |
| 332 | 351 | ); |
| 333 | 352 | const getDirectoryIdQuery = db.prepare<[string], { id: number }>(/* SQL */ ` |
| 334 | SELECT id FROM media_files WHERE path = ? AND kind = ${MediaFileKind.directory}; | |
| 353 | SELECT id FROM media_files | |
| 354 | WHERE path = ? collate nocase AND kind = ${MediaFileKind.directory}; | |
| 335 | 355 | `); |
| 336 | 356 | const createFileQuery = db.prepare<[{ |
| 337 | 357 | path: string; |
| ... | ... | @@ -346,38 +366,41 @@ const createFileQuery = db.prepare<[{ |
| 346 | 366 | }], void>(/* SQL */ ` |
| 347 | 367 | insert into media_files ( |
| 348 | 368 | path, parent_id, kind, timestamp, timestamp_updated, hash, |
| 349 | size, duration, dimensions, contents, processed) | |
| 369 | size, duration, dimensions, contents) | |
| 350 | 370 | values ( |
| 351 | 371 | $path, $parentId, ${MediaFileKind.file}, $timestamp, $timestampUpdated, |
| 352 | $hash, $size, $duration, $dimensions, $contents, 0) | |
| 353 | on conflict(path) do update set | |
| 372 | $hash, $size, $duration, $dimensions, $contents) | |
| 373 | on conflict(path collate nocase) do update set | |
| 374 | path = excluded.path, | |
| 354 | 375 | timestamp = excluded.timestamp, |
| 355 | 376 | timestamp_updated = excluded.timestamp_updated, |
| 377 | hash = excluded.hash, | |
| 356 | 378 | duration = excluded.duration, |
| 357 | 379 | size = excluded.size, |
| 358 | contents = excluded.contents, | |
| 359 | processed = case | |
| 360 | when media_files.hash != excluded.hash then 0 | |
| 361 | else media_files.processed | |
| 362 | end | |
| 380 | contents = excluded.contents | |
| 363 | 381 | returning *; |
| 364 | 382 | `).as(MediaFile); |
| 365 | const setProcessedQuery = db.prepare<[{ | |
| 383 | const setConfigQuery = db.prepare<[{ | |
| 366 | 384 | id: number; |
| 367 | processed: number; | |
| 385 | config: string; | |
| 368 | 386 | }]>(/* SQL */ ` |
| 369 | update media_files set processed = $processed where id = $id; | |
| 387 | update media_files set config = $config where id = $id; | |
| 388 | `); | |
| 389 | const markDirReindexQuery = db.prepare<[id: number]>(/* SQL */ ` | |
| 390 | update media_files set dir_reindex = 1 where id = ?; | |
| 370 | 391 | `); |
| 371 | const setProcessorsQuery = db.prepare<[{ | |
| 392 | const setPendingQuery = db.prepare<[{ | |
| 372 | 393 | id: number; |
| 373 | processed: number; | |
| 374 | processors: string; | |
| 394 | pending: number; | |
| 375 | 395 | }]>(/* SQL */ ` |
| 376 | update media_files set | |
| 377 | processed = $processed, | |
| 378 | processors = $processors | |
| 379 | where id = $id; | |
| 396 | update media_files set pending = $pending where id = $id; | |
| 380 | 397 | `); |
| 398 | const decPendingQuery = db.prepare<[id: number], { pending: number }>( | |
| 399 | /* SQL */ ` | |
| 400 | update media_files set pending = max(0, pending - 1) where id = ? | |
| 401 | returning pending; | |
| 402 | `, | |
| 403 | ); | |
| 381 | 404 | const setDurationQuery = db.prepare<[{ |
| 382 | 405 | id: number; |
| 383 | 406 | duration: number; |
| ... | ... | @@ -397,8 +420,18 @@ const setContentsQuery = db.prepare<[{ |
| 397 | 420 | update media_files set contents = $contents where id = $id; |
| 398 | 421 | `); |
| 399 | 422 | const getByPathQuery = db.prepare<[string]>(/* SQL */ ` |
| 400 | select * from media_files where path = ?; | |
| 423 | select * from media_files where path = ? collate nocase; | |
| 401 | 424 | `).as(MediaFile); |
| 425 | const updatePathQuery = db.prepare<[{ id: number; path: string }]>(/* SQL */ ` | |
| 426 | update media_files set path = $path where id = $id; | |
| 427 | `); | |
| 428 | const updatePathPrefixQuery = db.prepare<[{ old: string; new: string }]>( | |
| 429 | /* SQL */ ` | |
| 430 | update media_files | |
| 431 | set path = $new || substr(path, length($old) + 1) | |
| 432 | where lower(substr(path, 1, length($old))) = lower($old); | |
| 433 | `, | |
| 434 | ); | |
| 402 | 435 | const markDirectoryProcessedQuery = db.prepare<[{ |
| 403 | 436 | timestamp: number; |
| 404 | 437 | contents: string; |
| ... | ... | @@ -408,7 +441,7 @@ const markDirectoryProcessedQuery = db.prepare<[{ |
| 408 | 441 | id: number; |
| 409 | 442 | }]>(/* SQL */ ` |
| 410 | 443 | update media_files set |
| 411 | processed = 1, | |
| 444 | dir_reindex = 0, | |
| 412 | 445 | timestamp = $timestamp, |
| 413 | 446 | contents = $contents, |
| 414 | 447 | dirsort = $dirsort, |
| ... | ... | @@ -417,7 +450,7 @@ const markDirectoryProcessedQuery = db.prepare<[{ |
| 417 | 450 | where id = $id; |
| 418 | 451 | `); |
| 419 | 452 | const updateDirectoryQuery = db.prepare<[id: number]>(/* SQL */ ` |
| 420 | update media_files set processed = 0 where id = ?; | |
| 453 | update media_files set dir_reindex = 1 where id = ?; | |
| 421 | 454 | `); |
| 422 | 455 | |
| 423 | 456 | const getChildrenQuery = db.prepare<[id: number]>(/* SQL */ ` |
| ... | ... | @@ -448,8 +481,8 @@ const deleteCascadeQuery = db.prepare<[{ id: number }]>(/* SQL */ ` |
| 448 | 481 | const getDirectoriesToReindexQuery = db.prepare(` |
| 449 | 482 | with recursive directory_chain as ( |
| 450 | 483 | -- base case |
| 451 | select id, parent_id, path from media_files | |
| 452 | where kind = 0 and processed = 0 | |
| 484 | select id, parent_id, path from media_files | |
| 485 | where kind = 0 and dir_reindex = 1 | |
| 453 | 486 | -- recurse to find all parents so that size/hash can be updated |
| 454 | 487 | union |
| 455 | 488 | select m.id, m.parent_id, m.path |
src/file-viewer/models/ProcessorState.ts created+146| ... | ... | @@ -0,0 +1,146 @@ |
| 1 | // Tracks which processors have run on which files. Replaces the old | |
| 2 | // `processed` bitfield + `processors` string columns on `media_files`. | |
| 3 | // | |
| 4 | // - `processors` is the registry: one row per known processor, with the | |
| 5 | // version that the current code declares. rows for processors removed | |
| 6 | // from the code are pruned (cascading their file state). | |
| 7 | // - `file_processors` holds completions only. "pending" is the absence of | |
| 8 | // a row, or a row with a stale version. failures are recorded with | |
| 9 | // status=2 + the error text, and retried on the next sweep. | |
| 10 | const db = getDb("cache.sqlite"); | |
| 11 | db.table( | |
| 12 | "processor_state", | |
| 13 | /* SQL */ ` | |
| 14 | create table if not exists processors ( | |
| 15 | id integer primary key autoincrement, | |
| 16 | name text not null unique, | |
| 17 | version integer not null | |
| 18 | ); | |
| 19 | create table if not exists file_processors ( | |
| 20 | file integer not null references media_files(id) on delete cascade, | |
| 21 | processor integer not null references processors(id) on delete cascade, | |
| 22 | version integer not null, | |
| 23 | status integer not null, | |
| 24 | updated integer not null, | |
| 25 | error text, | |
| 26 | primary key (file, processor) | |
| 27 | ); | |
| 28 | create index file_processors_processor on file_processors (processor); | |
| 29 | `, | |
| 30 | ); | |
| 31 | ||
| 32 | export enum ProcessorStatus { | |
| 33 | done = 1, | |
| 34 | failed = 2, | |
| 35 | } | |
| 36 | ||
| 37 | export interface FileProcessorRow { | |
| 38 | file: number; | |
| 39 | processor: number; | |
| 40 | version: number; | |
| 41 | status: ProcessorStatus; | |
| 42 | updated: number; | |
| 43 | error: string | null; | |
| 44 | } | |
| 45 | ||
| 46 | /** | |
| 47 | * upsert the registry and prune processors that no longer exist in code. | |
| 48 | * returns a map of processor name to its row id, used by the scanner. | |
| 49 | */ | |
| 50 | export function syncRegistry( | |
| 51 | defs: readonly { name: string; version: number }[], | |
| 52 | ): Map<string, number> { | |
| 53 | const ids = new Map<string, number>(); | |
| 54 | const tx = db.node; | |
| 55 | tx.exec("begin"); | |
| 56 | try { | |
| 57 | for (const { name, version } of defs) { | |
| 58 | const { id } = upsertProcessorQuery.getNonNull({ name, version }); | |
| 59 | ids.set(name, id); | |
| 60 | } | |
| 61 | const known = new Set(defs.map((d) => d.name)); | |
| 62 | for (const { id, name } of allProcessorsQuery.array()) { | |
| 63 | if (!known.has(name)) deleteProcessorQuery.run(id); | |
| 64 | } | |
| 65 | tx.exec("commit"); | |
| 66 | } catch (err) { | |
| 67 | tx.exec("rollback"); | |
| 68 | throw err; | |
| 69 | } | |
| 70 | return ids; | |
| 71 | } | |
| 72 | ||
| 73 | /** completion state for one file, keyed by processor row id */ | |
| 74 | export function getStates( | |
| 75 | fileId: number, | |
| 76 | ): Map<number, Pick<FileProcessorRow, "version" | "status">> { | |
| 77 | const map = new Map<number, { version: number; status: ProcessorStatus }>(); | |
| 78 | for (const row of getStatesQuery.array(fileId)) { | |
| 79 | map.set(row.processor, { version: row.version, status: row.status }); | |
| 80 | } | |
| 81 | return map; | |
| 82 | } | |
| 83 | ||
| 84 | export function recordResult( | |
| 85 | fileId: number, | |
| 86 | processorId: number, | |
| 87 | version: number, | |
| 88 | status: ProcessorStatus, | |
| 89 | error: string | null = null, | |
| 90 | ) { | |
| 91 | recordResultQuery.run({ | |
| 92 | file: fileId, | |
| 93 | processor: processorId, | |
| 94 | version, | |
| 95 | status, | |
| 96 | updated: Date.now(), | |
| 97 | error, | |
| 98 | }); | |
| 99 | } | |
| 100 | ||
| 101 | /** the file's contents changed (hash mismatch); all processors must re-run */ | |
| 102 | export function invalidateFile(fileId: number) { | |
| 103 | invalidateFileQuery.run(fileId); | |
| 104 | } | |
| 105 | ||
| 106 | // -- queries -- | |
| 107 | const upsertProcessorQuery = db.prepare< | |
| 108 | [{ name: string; version: number }], | |
| 109 | { id: number } | |
| 110 | >(/* SQL */ ` | |
| 111 | insert into processors (name, version) values ($name, $version) | |
| 112 | on conflict(name) do update set version = excluded.version | |
| 113 | returning id; | |
| 114 | `); | |
| 115 | const allProcessorsQuery = db.prepare<[], { id: number; name: string }>( | |
| 116 | /* SQL */ ` | |
| 117 | select id, name from processors; | |
| 118 | `, | |
| 119 | ); | |
| 120 | const deleteProcessorQuery = db.prepare<[id: number]>(/* SQL */ ` | |
| 121 | delete from processors where id = ?; | |
| 122 | `); | |
| 123 | const getStatesQuery = db.prepare<[file: number], FileProcessorRow>(/* SQL */ ` | |
| 124 | select * from file_processors where file = ?; | |
| 125 | `); | |
| 126 | const recordResultQuery = db.prepare<[{ | |
| 127 | file: number; | |
| 128 | processor: number; | |
| 129 | version: number; | |
| 130 | status: number; | |
| 131 | updated: number; | |
| 132 | error: string | null; | |
| 133 | }]>(/* SQL */ ` | |
| 134 | insert into file_processors (file, processor, version, status, updated, error) | |
| 135 | values ($file, $processor, $version, $status, $updated, $error) | |
| 136 | on conflict(file, processor) do update set | |
| 137 | version = excluded.version, | |
| 138 | status = excluded.status, | |
| 139 | updated = excluded.updated, | |
| 140 | error = excluded.error; | |
| 141 | `); | |
| 142 | const invalidateFileQuery = db.prepare<[file: number]>(/* SQL */ ` | |
| 143 | delete from file_processors where file = ?; | |
| 144 | `); | |
| 145 | ||
| 146 | import { getDb } from "#sitegen/sqlite"; |
src/file-viewer/models/derived.ts+59-3| ... | ... | @@ -39,7 +39,10 @@ db.table( |
| 39 | 39 | `, |
| 40 | 40 | ); |
| 41 | 41 | |
| 42 | export const workDir = Path.resolve(".clover/derived"); | |
| 42 | // derived assets are written directly into the derived file store. files are | |
| 43 | // produced into a `tmp.`-prefixed sibling directory and renamed into place, | |
| 44 | // so a crash never leaves a partially-written asset at its final path. | |
| 45 | export const workDir = Path.resolve(derivedFileRoot); | |
| 43 | 46 | |
| 44 | 47 | let ongoing = new Map<string, Promise<number>>(); |
| 45 | 48 | |
| ... | ... | @@ -63,8 +66,13 @@ export async function produce( |
| 63 | 66 | if (root) break brk; |
| 64 | 67 | const { promise, resolve, reject } = Promise.withResolvers<number>(); |
| 65 | 68 | ongoing.set(key, promise); |
| 69 | // the rejection is rethrown to this caller; concurrent callers await | |
| 70 | // through `ongoing`, but when there are none the rejection would | |
| 71 | // otherwise be unobserved | |
| 72 | promise.catch(() => {}); | |
| 73 | const finalDir = workDir.join(key); | |
| 74 | const tmp = workDir.join(`${file.hash}/tmp.${subkey}`); | |
| 66 | 75 | try { |
| 67 | const tmp = workDir.join(key); | |
| 68 | 76 | node.hidden = true; |
| 69 | 77 | const filesWithStats = await queue.run({ |
| 70 | 78 | cores, |
| ... | ... | @@ -106,6 +114,11 @@ export async function produce( |
| 106 | 114 | ); |
| 107 | 115 | }, |
| 108 | 116 | }); |
| 117 | // move into the final location before recording rows; readers only | |
| 118 | // discover the asset through the database, so this is safe. | |
| 119 | await finalDir.delete({ recursive: true, force: true }); | |
| 120 | await fsp.rename(tmp.toString(), finalDir.toString()); | |
| 121 | ||
| 109 | 122 | db.node.exec("BEGIN"); |
| 110 | 123 | try { |
| 111 | 124 | root = insertRootQuery.getNonNull({ key, date: Date.now() }).id; |
| ... | ... | @@ -126,6 +139,11 @@ export async function produce( |
| 126 | 139 | resolve(root); |
| 127 | 140 | } catch (e) { |
| 128 | 141 | if (root) deleteRootQuery.run({ root }); |
| 142 | // remove partial output from disk: the tmp dir if the producer died, | |
| 143 | // or the renamed final dir if the database writes failed (e.g. the | |
| 144 | // source file was deleted while this asset was being produced) | |
| 145 | await tmp.delete({ recursive: true, force: true }).catch(() => {}); | |
| 146 | await finalDir.delete({ recursive: true, force: true }).catch(() => {}); | |
| 129 | 147 | reject(e); |
| 130 | 148 | throw e; |
| 131 | 149 | } finally { |
| ... | ... | @@ -173,6 +191,42 @@ export function deleteRoot(root: { id: number }) { |
| 173 | 191 | return deleteRootQuery.run({ root: root.id }); |
| 174 | 192 | } |
| 175 | 193 | |
| 194 | /** remove an orphaned root's files from the derived store */ | |
| 195 | export async function deleteRootFiles(root: { key: string }) { | |
| 196 | const dir = workDir.join(root.key); | |
| 197 | await dir.delete({ recursive: true, force: true }); | |
| 198 | // reclaim the per-hash parent directory once its last root is gone | |
| 199 | // (rmdir refuses non-empty directories, which is exactly what we want) | |
| 200 | await fsp.rmdir(UNWRAP(dir.parent).toString()).catch(() => {}); | |
| 201 | } | |
| 202 | ||
| 203 | /** | |
| 204 | * delete `tmp.*` directories left behind by crashed producers. only removes | |
| 205 | * directories untouched for a day, to never race an ongoing producer. | |
| 206 | * | |
| 207 | * `tmp.*` FILES are never touched: encode intermediates inside completed | |
| 208 | * roots (`av1-au/tmp.av1.mp4`, ...) are intentionally named that way to be | |
| 209 | * excluded from `derived_files`, but the dash producer reads them across | |
| 210 | * root directories, so they must persist. | |
| 211 | */ | |
| 212 | export async function cleanAbandonedTmp(): Promise<number> { | |
| 213 | let cleaned = 0; | |
| 214 | const dayAgo = Date.now() - 24 * 60 * 60 * 1000; | |
| 215 | for (const hashDir of await workDir.readDir().catch(() => [])) { | |
| 216 | for (const sub of await hashDir.readDir().catch(() => [])) { | |
| 217 | if (!sub.base.startsWith("tmp.")) continue; | |
| 218 | try { | |
| 219 | const stat = await sub.stat(); | |
| 220 | if (!stat.isDirectory()) continue; | |
| 221 | if (stat.mtime.getTime() > dayAgo) continue; | |
| 222 | await sub.delete({ recursive: true, force: true }); | |
| 223 | cleaned += 1; | |
| 224 | } catch {} | |
| 225 | } | |
| 226 | } | |
| 227 | return cleaned; | |
| 228 | } | |
| 229 | ||
| 176 | 230 | const insertRootQuery = db.prepare< |
| 177 | 231 | [{ key: string; date: number }], |
| 178 | 232 | { id: number } |
| ... | ... | @@ -214,9 +268,11 @@ const findOrphanedRootsQuery = db.prepare< |
| 214 | 268 | |
| 215 | 269 | import { Path } from "#sitegen/path"; |
| 216 | 270 | import { getDb } from "#sitegen/sqlite"; |
| 217 | import { ASSERT } from "@clo/lib/assert"; | |
| 271 | import { ASSERT, UNWRAP } from "@clo/lib/assert"; | |
| 218 | 272 | import * as progress from "@clo/lib/progress"; |
| 219 | 273 | import * as queue from "@clo/lib/queue"; |
| 220 | 274 | import * as crypto from "node:crypto"; |
| 221 | 275 | import * as fs from "node:fs"; |
| 276 | import * as fsp from "node:fs/promises"; | |
| 277 | import { derivedFileRoot } from "../paths.ts"; | |
| 222 | 278 | import type { MediaFile } from "./MediaFile.ts"; |
src/file-viewer/rules.ts+6| ... | ... | @@ -107,6 +107,12 @@ export const extsPreCompressed = new Set([ |
| 107 | 107 | ]); |
| 108 | 108 | extsPreCompressed.delete(".svg"); |
| 109 | 109 | |
| 110 | // files that should never have media processors (transcodes) run on them, | |
| 111 | // usually because the file is intentionally cursed. | |
| 112 | export const processDenyList = new Set([ | |
| 113 | "/2021/top-10000-bread/output.mp4", | |
| 114 | ]); | |
| 115 | ||
| 110 | 116 | export function fileIcon( |
| 111 | 117 | file: Pick<MediaFile, "kind" | "basename" | "path">, |
| 112 | 118 | dirOpen?: boolean, |
src/file-viewer/sync.ts created+160| ... | ... | @@ -0,0 +1,160 @@ |
| 1 | // Keeps the web node's copy of cache.sqlite in sync with the source of | |
| 2 | // truth. The SOT serves a server-sent-events feed at /db/events; this | |
| 3 | // module holds it open and pulls /db on every beacon. The server emits a | |
| 4 | // beacon on connect, so boot, reconnect, and missed-while-disconnected all | |
| 5 | // resolve through the same path — there is no polling and no TTL. | |
| 6 | // | |
| 7 | // Pulls are ETag-guarded (no-change is a tiny 304; the etag is persisted so | |
| 8 | // a reboot does not re-download an unchanged database), land in a temp | |
| 9 | // file, and atomically rename over .clover/cache.sqlite before hot-swapping | |
| 10 | // the open database handle. | |
| 11 | const console = log.scoped("sync"); | |
| 12 | ||
| 13 | const token = process.env.CLOVER_SOT_KEY ?? null; | |
| 14 | const enabled = process.env.CLOVER_DB_SYNC !== "0" && token != null; | |
| 15 | const sotUrl = new URL(process.env.CLOVER_SOT_URL ?? "https://db.paperclover.net") | |
| 16 | .toString().replace(/\/+$/g, ""); | |
| 17 | if (!enabled) { | |
| 18 | console.warn( | |
| 19 | "database sync disabled" | |
| 20 | + (token ? " (CLOVER_DB_SYNC=0)" : " (no CLOVER_SOT_KEY)"), | |
| 21 | ); | |
| 22 | } | |
| 23 | ||
| 24 | let inFlight: Promise<void> | null = null; | |
| 25 | ||
| 26 | export function revalidate(): Promise<void> { | |
| 27 | return inFlight ??= revalidateInner().finally(() => { | |
| 28 | inFlight = null; | |
| 29 | }); | |
| 30 | } | |
| 31 | ||
| 32 | async function revalidateInner(): Promise<void> { | |
| 33 | const db = getDb("cache.sqlite"); | |
| 34 | const res = await fetch(`${sotUrl}/db`, { | |
| 35 | headers: { | |
| 36 | Authorization: token!, | |
| 37 | ...etag() ? { "If-None-Match": etag()! } : null, | |
| 38 | }, | |
| 39 | signal: AbortSignal.timeout(120_000), | |
| 40 | }); | |
| 41 | if (res.status === 304) return; | |
| 42 | if (!res.ok || !res.body) { | |
| 43 | throw new Error(`source of truth responded ${res.status} ${res.statusText}`); | |
| 44 | } | |
| 45 | ||
| 46 | const tmp = db.file + ".tmp"; | |
| 47 | await stream.promises.pipeline( | |
| 48 | stream.Readable.fromWeb(res.body as any), | |
| 49 | fs.createWriteStream(tmp), | |
| 50 | ); | |
| 51 | // fsync before the rename so a power cut cannot leave a torn file | |
| 52 | const handle = await fsp.open(tmp, "r+"); | |
| 53 | await handle.sync(); | |
| 54 | await handle.close(); | |
| 55 | await fsp.rename(tmp, db.file); | |
| 56 | db.reload(); | |
| 57 | saveEtag(res.headers.get("ETag")); | |
| 58 | console.info(`database updated (${string.formatByteSize(Number(res.headers.get("Content-Length") ?? "0"))})`); | |
| 59 | } | |
| 60 | ||
| 61 | // -- the persisted etag, stored beside the database -- | |
| 62 | let cachedEtag: string | null | undefined; | |
| 63 | function etagPath() { | |
| 64 | return getDb("cache.sqlite").file + ".etag"; | |
| 65 | } | |
| 66 | function etag(): string | null { | |
| 67 | if (cachedEtag === undefined) { | |
| 68 | try { | |
| 69 | cachedEtag = fs.readFileSync(etagPath(), "utf-8").trim() || null; | |
| 70 | } catch { | |
| 71 | cachedEtag = null; | |
| 72 | } | |
| 73 | } | |
| 74 | return cachedEtag; | |
| 75 | } | |
| 76 | function saveEtag(value: string | null) { | |
| 77 | cachedEtag = value; | |
| 78 | try { | |
| 79 | fs.writeFileSync(etagPath(), value ?? ""); | |
| 80 | } catch (err) { | |
| 81 | console.warn("could not persist database etag:", err); | |
| 82 | } | |
| 83 | } | |
| 84 | ||
| 85 | // -- the subscription -- | |
| 86 | // a hand-rolled sse reader over fetch (EventSource cannot send the | |
| 87 | // Authorization header). any event block triggers a revalidate; a failed | |
| 88 | // revalidate aborts the connection so the reconnect path retries it. | |
| 89 | async function subscribeLoop() { | |
| 90 | let backoff = 1000; | |
| 91 | while (true) { | |
| 92 | const ctrl = new AbortController(); | |
| 93 | let watchdog: NodeJS.Timeout | null = null; | |
| 94 | try { | |
| 95 | const res = await fetch(`${sotUrl}/db/events`, { | |
| 96 | headers: { Authorization: token! }, | |
| 97 | signal: ctrl.signal, | |
| 98 | }); | |
| 99 | if (!res.ok || !res.body) { | |
| 100 | throw new Error(`${res.status} ${res.statusText}`); | |
| 101 | } | |
| 102 | console.info(`subscribed to database changes from ${sotUrl}`); | |
| 103 | backoff = 1000; | |
| 104 | ||
| 105 | // the server keepalives every 25s; 90s of silence means the | |
| 106 | // connection died without a FIN | |
| 107 | const armWatchdog = () => { | |
| 108 | if (watchdog) clearTimeout(watchdog); | |
| 109 | watchdog = setTimeout( | |
| 110 | () => ctrl.abort(new Error("no keepalive")), | |
| 111 | 90_000, | |
| 112 | ); | |
| 113 | watchdog.unref(); | |
| 114 | }; | |
| 115 | armWatchdog(); | |
| 116 | const reader = res.body.getReader(); | |
| 117 | const decoder = new TextDecoder(); | |
| 118 | let buffer = ""; | |
| 119 | while (true) { | |
| 120 | const { value, done } = await reader.read(); | |
| 121 | if (done) break; | |
| 122 | armWatchdog(); | |
| 123 | buffer += decoder.decode(value, { stream: true }); | |
| 124 | const blocks = buffer.split("\n\n"); | |
| 125 | buffer = blocks.pop()!; | |
| 126 | const changed = blocks.some((block) => | |
| 127 | block.split("\n").some((line) => line.startsWith("event:") || line.startsWith("data:")) | |
| 128 | ); | |
| 129 | if (changed) { | |
| 130 | void revalidate().catch((err) => { | |
| 131 | console.error("revalidate failed:", err); | |
| 132 | ctrl.abort(new Error("revalidate failed")); | |
| 133 | }); | |
| 134 | } | |
| 135 | } | |
| 136 | throw new Error("stream ended"); | |
| 137 | } catch (err: any) { | |
| 138 | console.warn( | |
| 139 | `database subscription lost (${err?.message ?? err}); ` | |
| 140 | + `retrying in ${Math.round(backoff / 1000)}s`, | |
| 141 | ); | |
| 142 | } finally { | |
| 143 | if (watchdog) clearTimeout(watchdog); | |
| 144 | ctrl.abort(); | |
| 145 | } | |
| 146 | await new Promise<void>((resolve) => setTimeout(resolve, backoff).unref()); | |
| 147 | backoff = Math.min(backoff * 2, 60_000); | |
| 148 | } | |
| 149 | } | |
| 150 | ||
| 151 | if (enabled) void subscribeLoop(); | |
| 152 | ||
| 153 | import * as fs from "node:fs"; | |
| 154 | import * as fsp from "node:fs/promises"; | |
| 155 | import * as stream from "node:stream"; | |
| 156 | ||
| 157 | import { getDb } from "#sitegen/sqlite"; | |
| 158 | ||
| 159 | import * as log from "@clo/lib/log"; | |
| 160 | import * as string from "@clo/lib/string"; |
src/file-viewer/tags/media-dir.marko+1-1| ... | ... | @@ -27,7 +27,7 @@ static interface DirConfig { |
| 27 | 27 | return { sorted, activeFilename, readme }; |
| 28 | 28 | })()> |
| 29 | 29 | |
| 30 | <const/extraConfig: DirConfig=dir.processors ? JSON.parse(dir.processors) : {}> | |
| 30 | <const/extraConfig: DirConfig=dir.config ? JSON.parse(dir.config) : {}> | |
| 31 | 31 | |
| 32 | 32 | <define/Inner> |
| 33 | 33 | <div.content.primary> |
src/file-viewer/tags/media-panel.marko+1-1| ... | ... | @@ -100,7 +100,7 @@ static const cotyledonEndingFile = "/2024/for everone"; |
| 100 | 100 | </div> |
| 101 | 101 | |
| 102 | 102 | // content |
| 103 | <const/ext=path.extname(file.path)> | |
| 103 | <const/ext=path.extname(file.path).toLowerCase()> | |
| 104 | 104 | <if=file.path === "/"> |
| 105 | 105 | <media-root-dir dir=file ...{ activeFilename, isLast, hasCotyledonCookie }/> |
| 106 | 106 | </if> |
src/file-viewer/tags/view-audio.marko+1-1| ... | ... | @@ -25,7 +25,7 @@ export interface Input { |
| 25 | 25 | <audio controls preload="none" style="width: 100%"> |
| 26 | 26 | <source |
| 27 | 27 | src="/file" + file.path |
| 28 | type=`audio/${path.extname(file.path).substring(1)}` | |
| 28 | type=`audio/${path.extname(file.path).substring(1).toLowerCase()}` | |
| 29 | 29 | > |
| 30 | 30 | </audio> |
| 31 | 31 | //<if=lyricsFile> |
src/file-viewer/tags/view-code.marko+1-2| ... | ... | @@ -9,8 +9,7 @@ export interface Input { |
| 9 | 9 | <if=contents> |
| 10 | 10 | <pre style="white-space:pre-wrap">$!{contents}</pre> |
| 11 | 11 | </if> |
| 12 | <else if=size> | |
| 13 | ${" "}1_000_000> | |
| 12 | <else if=(size > 1_000_000)> | |
| 14 | 13 | <view-download-too-big ...input/> |
| 15 | 14 | </else> |
| 16 | 15 | <else> |
src/file-viewer/tags/view-download.marko+1-1| ... | ... | @@ -28,7 +28,7 @@ export interface Input { |
| 28 | 28 | <a href="/file" + file.path + "?dl" aria-label="download">${file.basename}</a> |
| 29 | 29 | </define> |
| 30 | 30 | |
| 31 | <const/ext=path.extname(file.basename)> | |
| 31 | <const/ext=path.extname(file.basename).toLowerCase()> | |
| 32 | 32 | <if=ext === ".blend"> |
| 33 | 33 | <p> |
| 34 | 34 | <code>.blend</code> files can be open in |
src/file-viewer/tags/view-text.marko+16-11| ... | ... | @@ -145,18 +145,23 @@ static function highlightHashComments(text: string) { |
| 145 | 145 | file: { path, basename, contents }, |
| 146 | 146 | siblings, |
| 147 | 147 | }=input> |
| 148 | <pre style={ "white-space": "pre-wrap", "max-width": "70ch" }> | |
| 148 | <if=!contents && input.file.pending> | |
| 149 | <p>this file was just added and is still being processed. come back soon!</p> | |
| 150 | </if> | |
| 151 | <else> | |
| 152 | <pre style={ "white-space": "pre-wrap", "max-width": "70ch" }> | |
| 149 | 153 | $!{ |
| 150 | path.startsWith("/2021/phoenix-write/maps") && basename === "map.txt" | |
| 151 | ? // special cased for phoenix write maps | |
| 152 | highlightHashComments(contents) | |
| 153 | : // normal | |
| 154 | highlightLinksInTextView( | |
| 155 | contents, | |
| 156 | siblings.filter((f) => f.kind === MediaFileKind.file), | |
| 157 | ) | |
| 158 | } | |
| 159 | </pre> | |
| 154 | path.startsWith("/2021/phoenix-write/maps") && basename === "map.txt" | |
| 155 | ? // special cased for phoenix write maps | |
| 156 | highlightHashComments(contents) | |
| 157 | : // normal | |
| 158 | highlightLinksInTextView( | |
| 159 | contents, | |
| 160 | siblings.filter((f) => f.kind === MediaFileKind.file), | |
| 161 | ) | |
| 162 | } | |
| 163 | </pre> | |
| 164 | </else> | |
| 160 | 165 | |
| 161 | 166 | import * as string from "@clo/lib/string"; |
| 162 | 167 | import { MediaFile, MediaFileKind } from "#src/file-viewer/models/MediaFile.ts"; |
src/source-of-truth.dockerfile+22-22| ... | ... | @@ -1,31 +1,31 @@ |
| 1 | from node:24 as builder | |
| 1 | # The source of truth server: file serving + the auto-indexer service. | |
| 2 | # No bundling step; the whole source tree plus node_modules ship in the | |
| 3 | # image and `node run` executes it the same way development does. | |
| 4 | # | |
| 5 | # trixie rather than bookworm: the dimensions processor needs the `magick` | |
| 6 | # entry point, which is ImageMagick 7 (bookworm only packages v6). | |
| 7 | from node:24-trixie | |
| 2 | 8 | |
| 3 | run apt install -y git | |
| 4 | ||
| 5 | workdir /paperclover.net | |
| 6 | copy package*.json ./ | |
| 7 | run npm ci | |
| 8 | copy . ./ | |
| 9 | run npx esbuild --bundle \ | |
| 10 | framework/backend/entry-node.ts \ | |
| 11 | --platform=node \ | |
| 12 | --format=esm \ | |
| 13 | --define:globalThis.CLOVER_SERVER_ENTRY='"#src/source-of-truth.ts"' \ | |
| 14 | --minify \ | |
| 15 | --sourcemap=linked \ | |
| 16 | --entry-names=index \ | |
| 17 | --outdir=/app | |
| 18 | ||
| 19 | from node:24 as runtime | |
| 9 | # media processors shell out to these. exiftool is `libimage-exiftool-perl` | |
| 10 | # on debian; curl is for the compose healthcheck. | |
| 11 | run apt-get update && apt-get install -y --no-install-recommends \ | |
| 12 | ffmpeg \ | |
| 13 | imagemagick \ | |
| 14 | libimage-exiftool-perl \ | |
| 15 | curl \ | |
| 16 | && rm -rf /var/lib/apt/lists/* | |
| 20 | 17 | |
| 21 | 18 | workdir /app |
| 22 | copy --from=builder /app/ ./ | |
| 19 | copy package.json package-lock.json ./ | |
| 20 | run npm ci --no-audit --no-fund | |
| 21 | copy . . | |
| 22 | # prime run.js's dependency hash so container boot never re-runs npm | |
| 23 | run node run || true | |
| 23 | 24 | |
| 24 | env PORT=43200 | |
| 25 | env PORT=80 | |
| 25 | 26 | # must be configured |
| 26 | 27 | env CLOVER_DB=/dev/null |
| 27 | 28 | env CLOVER_FILE_RAW=/dev/null |
| 28 | 29 | env CLOVER_FILE_DERIVED=/dev/null |
| 29 | 30 | |
| 30 | cmd ["node", "--enable-source-maps", "/app/index.js"] | |
| 31 | ||
| 31 | cmd ["node", "run", "source-of-truth"] |
src/source-of-truth.ts+165-3| ... | ... | @@ -22,7 +22,7 @@ |
| 22 | 22 | */ |
| 23 | 23 | const app = new Hono(); |
| 24 | 24 | export default app; |
| 25 | export const port = 4000; | |
| 25 | export const port = Number(process.env.PORT ?? 4000); | |
| 26 | 26 | |
| 27 | 27 | const token = UNWRAP(process.env.CLOVER_SOT_KEY); |
| 28 | 28 | |
| ... | ... | @@ -33,6 +33,26 @@ if (!fs.existsSync(derivedFileRoot)) { |
| 33 | 33 | throw new Error(`${derivedFileRoot} does not exist`); |
| 34 | 34 | } |
| 35 | 35 | |
| 36 | /** | |
| 37 | * `node run source-of-truth` boots the full source of truth: the indexer | |
| 38 | * service (file watcher, periodic sweeps, processors) plus this http server. | |
| 39 | * set CLOVER_INDEXER=0 for a serve-only process. | |
| 40 | */ | |
| 41 | export async function main() { | |
| 42 | if (process.env.CLOVER_INDEXER !== "0") { | |
| 43 | service = indexer.start(); | |
| 44 | } | |
| 45 | const server = await http.serve({ | |
| 46 | respond: async (request) => app.fetch(request), | |
| 47 | port, | |
| 48 | }); | |
| 49 | console.info(`source of truth live at ${server.url}`); | |
| 50 | const close = () => void server.close().then(() => process.exit(0)); | |
| 51 | process.on("SIGINT", close); | |
| 52 | process.on("SIGTERM", close); | |
| 53 | } | |
| 54 | let service: indexer.Service | null = null; | |
| 55 | ||
| 36 | 56 | // Re-use file descriptors if the same file is being read twice. |
| 37 | 57 | const fds = new Map<string, Awaitable<{ fd: number; refs: number }>>(); |
| 38 | 58 | |
| ... | ... | @@ -52,6 +72,13 @@ app.get("/file/*", async (c) => { |
| 52 | 72 | } |
| 53 | 73 | const file = MediaFile.getByPath(filePath); |
| 54 | 74 | if (!file || file.kind === MediaFileKind.directory) return c.notFound(); |
| 75 | // derived asset names are plain relative paths under the hash directory | |
| 76 | if ( | |
| 77 | derivedAsset | |
| 78 | && derivedAsset.split("/").some((part) => part === "" || part === "." || part === "..") | |
| 79 | ) { | |
| 80 | return c.notFound(); | |
| 81 | } | |
| 55 | 82 | const fullPath = derivedAsset |
| 56 | 83 | ? path.join(derivedFileRoot, file.hash, derivedAsset) |
| 57 | 84 | : path.join(rawFileRoot, file.path); |
| ... | ... | @@ -65,8 +92,8 @@ app.get("/file/*", async (c) => { |
| 65 | 92 | fds.delete(fullPath); |
| 66 | 93 | throw err; |
| 67 | 94 | }); |
| 68 | fds.set(file.path, promise); | |
| 69 | fds.set(file.path, handle = await promise); | |
| 95 | fds.set(fullPath, promise); | |
| 96 | fds.set(fullPath, handle = await promise); | |
| 70 | 97 | } |
| 71 | 98 | handle.refs += 1; |
| 72 | 99 | } catch (err: any) { |
| ... | ... | @@ -97,18 +124,153 @@ app.post("/reload", async (c) => { |
| 97 | 124 | return c.json({ error: "invalid authorization header" }, 401); |
| 98 | 125 | } |
| 99 | 126 | MediaFile.db.reload(); |
| 127 | indexer.emitDbChange(); | |
| 100 | 128 | return c.body(null, 204); |
| 101 | 129 | }); |
| 102 | 130 | |
| 131 | // a tiny server-sent-events feed of database changes. consumers (web nodes, | |
| 132 | // dev servers) hold this open and pull /db whenever a beacon arrives; the | |
| 133 | // connection itself is the subscription, so the source of truth never needs | |
| 134 | // a list of its consumers. events carry the generation, but clients can | |
| 135 | // treat any event as "something changed" — /db pulls are ETag-guarded. | |
| 136 | app.get("/db/events", (c) => { | |
| 137 | if (c.req.header("Authorization") !== token) { | |
| 138 | return c.json({ error: "invalid authorization header" }, 401); | |
| 139 | } | |
| 140 | const encoder = new TextEncoder(); | |
| 141 | let cleanup: (() => void) | null = null; | |
| 142 | const stream = new ReadableStream<Uint8Array>({ | |
| 143 | start(controller) { | |
| 144 | const send = (text: string) => controller.enqueue(encoder.encode(text)); | |
| 145 | // an event on connect makes clients revalidate immediately, covering | |
| 146 | // anything they missed while disconnected | |
| 147 | send(`retry: 5000\n\n`); | |
| 148 | send(`event: change\ndata: ${indexer.generation}\n\n`); | |
| 149 | const unsubscribe = indexer.subscribeDbChange((generation) => { | |
| 150 | send(`event: change\ndata: ${generation}\n\n`); | |
| 151 | }); | |
| 152 | // keepalive comments so proxies and clients hold the connection open | |
| 153 | const keepalive = setInterval(() => send(`: keepalive\n\n`), 25_000); | |
| 154 | keepalive.unref(); | |
| 155 | cleanup = () => { | |
| 156 | unsubscribe[Symbol.dispose](); | |
| 157 | clearInterval(keepalive); | |
| 158 | }; | |
| 159 | }, | |
| 160 | cancel() { | |
| 161 | cleanup?.(); | |
| 162 | }, | |
| 163 | }); | |
| 164 | c.header("Content-Type", "text/event-stream"); | |
| 165 | c.header("Cache-Control", "no-store"); | |
| 166 | return c.body(stream); | |
| 167 | }); | |
| 168 | ||
| 169 | // live progress of all indexing operations, as a clover progress stream. | |
| 170 | // each request snapshots the current tree state and then follows along; | |
| 171 | // connect any time with `node run tail-progress`. | |
| 172 | app.get("/progress", (c) => { | |
| 173 | if (c.req.header("Authorization") !== token) { | |
| 174 | return c.json({ error: "invalid authorization header" }, 401); | |
| 175 | } | |
| 176 | c.header("Content-Type", progress.contentType); | |
| 177 | return c.body(progress.encodeByteStream(indexer.root) as ReadableStream); | |
| 178 | }); | |
| 179 | ||
| 180 | // download a consistent snapshot of cache.sqlite. web nodes call this with | |
| 181 | // `If-None-Match` on a stale-while-revalidate schedule, and immediately | |
| 182 | // when pinged at /internal/db-changed. | |
| 183 | app.get("/db", async (c) => { | |
| 184 | if (c.req.header("Authorization") !== token) { | |
| 185 | return c.json({ error: "invalid authorization header" }, 401); | |
| 186 | } | |
| 187 | const snap = await getDbSnapshot(); | |
| 188 | if (c.req.header("If-None-Match") === snap.etag) { | |
| 189 | return c.body(null, 304); | |
| 190 | } | |
| 191 | c.header("Content-Type", "application/vnd.sqlite3"); | |
| 192 | c.header("Content-Length", String(snap.size)); | |
| 193 | c.header("ETag", snap.etag); | |
| 194 | return c.body( | |
| 195 | stream.Readable.toWeb(fs.createReadStream(snap.path)) as ReadableStream, | |
| 196 | ); | |
| 197 | }); | |
| 198 | ||
| 199 | // trigger a sweep (or a subtree scan with {"path": "/2025"}). progress is | |
| 200 | // visible on /progress; this returns immediately. | |
| 201 | app.post("/scan", async (c) => { | |
| 202 | if (c.req.header("Authorization") !== token) { | |
| 203 | return c.json({ error: "invalid authorization header" }, 401); | |
| 204 | } | |
| 205 | if (!service) return c.json({ error: "indexer is disabled" }, 409); | |
| 206 | const body = await c.req.json().catch(() => ({})); | |
| 207 | const subPath = typeof body.path === "string" ? body.path : undefined; | |
| 208 | void service.sweep("manual", subPath); | |
| 209 | return c.json({ ok: true }, 202); | |
| 210 | }); | |
| 211 | ||
| 212 | // -- database snapshotting -- | |
| 213 | // `VACUUM INTO` writes a compact, transactionally-consistent copy that does | |
| 214 | // not depend on -wal sidecar files. regenerated lazily when the indexer | |
| 215 | // reports changes, at most every 30 seconds. | |
| 216 | interface DbSnapshot { | |
| 217 | gen: number; | |
| 218 | time: number; | |
| 219 | path: string; | |
| 220 | etag: string; | |
| 221 | size: number; | |
| 222 | } | |
| 223 | let snapshot: DbSnapshot | null = null; | |
| 224 | let snapshotInFlight: Promise<DbSnapshot> | null = null; | |
| 225 | ||
| 226 | function getDbSnapshot() { | |
| 227 | const gen = indexer.generation; | |
| 228 | // a stale-generation snapshot is never served: subscribers pull exactly | |
| 229 | // once per beacon, so a 304 here would leave them out of date until the | |
| 230 | // ttl backstop. beacons are debounced upstream, which bounds how often | |
| 231 | // the vacuum can actually run. | |
| 232 | if (snapshot && snapshot.gen === gen && fs.existsSync(snapshot.path)) { | |
| 233 | return Promise.resolve(snapshot); | |
| 234 | } | |
| 235 | return snapshotInFlight ??= (async () => { | |
| 236 | try { | |
| 237 | const dir = process.env.CLOVER_DB ?? ".clover"; | |
| 238 | const file = path.join(dir, "db-snapshot.sqlite"); | |
| 239 | await fs.rm(file, { force: true }); | |
| 240 | MediaFile.db.node.exec( | |
| 241 | `vacuum into '${file.replaceAll("'", "''")}';`, | |
| 242 | ); | |
| 243 | const [hash, { size }] = await Promise.all([ | |
| 244 | hashFile(new Path(file)), | |
| 245 | fs.stat(file), | |
| 246 | ]); | |
| 247 | return snapshot = { | |
| 248 | gen, | |
| 249 | time: Date.now(), | |
| 250 | path: file, | |
| 251 | etag: `"${hash}"`, | |
| 252 | size, | |
| 253 | }; | |
| 254 | } finally { | |
| 255 | snapshotInFlight = null; | |
| 256 | } | |
| 257 | })(); | |
| 258 | } | |
| 259 | ||
| 103 | 260 | const openFile = util.promisify(fsCallbacks.open); |
| 104 | 261 | const closeFile = util.promisify(fsCallbacks.close); |
| 105 | 262 | const fstat = util.promisify(fsCallbacks.fstat); |
| 106 | 263 | |
| 107 | 264 | import * as fs from "#sitegen/fs"; |
| 265 | import { Path } from "#sitegen/path"; | |
| 266 | import { hashFile } from "#src/file-viewer/indexer/scan.ts"; | |
| 267 | import * as indexer from "#src/file-viewer/indexer/service.ts"; | |
| 108 | 268 | import { FilePermissions } from "#src/file-viewer/models/FilePermissions.ts"; |
| 109 | 269 | import { MediaFile, MediaFileKind } from "#src/file-viewer/models/MediaFile.ts"; |
| 110 | 270 | import { ASSERT, UNWRAP } from "@clo/lib/assert"; |
| 271 | import * as http from "@clo/lib/http"; | |
| 111 | 272 | import * as mime from "@clo/lib/mime"; |
| 273 | import * as progress from "@clo/lib/progress"; | |
| 112 | 274 | import { Hono } from "hono"; |
| 113 | 275 | import * as fsCallbacks from "node:fs"; |
| 114 | 276 | import * as path from "node:path"; |
src/tags/clover-media.marko+32-20| ... | ... | @@ -14,28 +14,13 @@ export interface Input { |
| 14 | 14 | <else if=extsVideo.has(file.extension.toLowerCase())> |
| 15 | 15 | <clover-video minimal noFigure ...{ file }/> |
| 16 | 16 | </else> |
| 17 | <else if=file.extension === ".gif"> | |
| 17 | <else if=file.extension.toLowerCase() === ".gif"> | |
| 18 | 18 | <img src=`/file${escapeUri(file.path)}` ...rest> |
| 19 | 19 | </else> |
| 20 | 20 | <else> |
| 21 | 21 | <const/base=`/file${escapeUri(file.path)}`> |
| 22 | <const/{ width }=UNWRAP(file.parseDimensions())> | |
| 23 | <const/targetSizes=transcodeRules.imageSizes.filter((w) => w < width)> | |
| 24 | <const/sizes=( | |
| 25 | displayWidthEstimate | |
| 26 | ? typeof displayWidthEstimate === "number" | |
| 27 | ? `${displayWidthEstimate}px` | |
| 28 | : displayWidthEstimate | |
| 29 | : `100vw` | |
| 30 | )> | |
| 31 | <picture> | |
| 32 | <for|ext| of=["jxl", "webp"]> | |
| 33 | <source | |
| 34 | type=`image/${ext}` | |
| 35 | srcset=targetSizes.map((s) => `${base}:/${s}.${ext} ${s}w`) | |
| 36 | ...{ sizes } | |
| 37 | > | |
| 38 | </for> | |
| 22 | <const/dims=file.parseDimensions()> | |
| 23 | <if=!dims || file.pending> | |
| 39 | 24 | <img |
| 40 | 25 | src=base |
| 41 | 26 | style=( |
| ... | ... | @@ -45,11 +30,38 @@ export interface Input { |
| 45 | 30 | ) |
| 46 | 31 | ...rest |
| 47 | 32 | > |
| 48 | </picture> | |
| 33 | </if> | |
| 34 | <else> | |
| 35 | <const/targetSizes=transcodeRules.imageSizes.filter((w) => w < dims.width)> | |
| 36 | <const/sizes=( | |
| 37 | displayWidthEstimate | |
| 38 | ? typeof displayWidthEstimate === "number" | |
| 39 | ? `${displayWidthEstimate}px` | |
| 40 | : displayWidthEstimate | |
| 41 | : `100vw` | |
| 42 | )> | |
| 43 | <picture> | |
| 44 | <for|ext| of=["jxl", "webp"]> | |
| 45 | <source | |
| 46 | type=`image/${ext}` | |
| 47 | srcset=targetSizes.map((s) => `${base}:/${s}.${ext} ${s}w`).join(",") | |
| 48 | ...{ sizes } | |
| 49 | > | |
| 50 | </for> | |
| 51 | <img | |
| 52 | src=base | |
| 53 | style=( | |
| 54 | align | |
| 55 | ? { "object-position": align, width: "100%", height: "100%" } | |
| 56 | : undefined | |
| 57 | ) | |
| 58 | ...rest | |
| 59 | > | |
| 60 | </picture> | |
| 61 | </else> | |
| 49 | 62 | </else> |
| 50 | 63 | |
| 51 | 64 | import * as transcodeRules from "#src/file-viewer/transcode-rules.ts"; |
| 52 | 65 | import { extsVideo } from "#src/file-viewer/rules.ts"; |
| 53 | 66 | import { MediaFile } from "#src/file-viewer/models/MediaFile.ts"; |
| 54 | import { UNWRAP } from "@clo/lib/assert"; | |
| 55 | 67 | import { escapeUri } from "#src/file-viewer/format.ts"; |
src/tags/clover-video.client.ts+55-15| ... | ... | @@ -65,21 +65,7 @@ function decidePlaybackMode(): Promise<PlaybackMode> { |
| 65 | 65 | console.warn(`does not implement av1+opus`); |
| 66 | 66 | } |
| 67 | 67 | |
| 68 | return canNativelyPlayHls().then((nativeHls): PlaybackMode => { | |
| 69 | if (nativeHls) return "hls-native"; | |
| 70 | // @ts-expect-error | |
| 71 | return (hls ??= import("/js/scripts/vendor/hls.js")) | |
| 72 | .then((hls): PlaybackMode => { | |
| 73 | // Polyfill HLS (Chrome 23, Firefox 42, IE11 on Win8) | |
| 74 | // TODO: ES modules have a minimum of Chrome 63 or Firefox 60 | |
| 75 | if (hls?.default?.isSupported?.() || hls?.isSupported()) { | |
| 76 | return "hls-polyfill"; | |
| 77 | } | |
| 78 | console.warn("does not support hls.js"); | |
| 79 | // Other browsers | |
| 80 | return "none"; | |
| 81 | }); | |
| 82 | }); | |
| 68 | return decideHlsMode(); | |
| 83 | 69 | }) |
| 84 | 70 | .catch((e): PlaybackMode => { |
| 85 | 71 | console.warn(e); |
| ... | ... | @@ -92,6 +78,28 @@ function decidePlaybackMode(): Promise<PlaybackMode> { |
| 92 | 78 | }); |
| 93 | 79 | } |
| 94 | 80 | |
| 81 | // the hls fallback decision is also needed by av1-capable browsers when a | |
| 82 | // video has an hls manifest but no dash one (dash encodes can fail or lag) | |
| 83 | function decideHlsMode(): Promise<PlaybackMode> { | |
| 84 | return canNativelyPlayHls().then( | |
| 85 | (nativeHls): PlaybackMode | Promise<PlaybackMode> => { | |
| 86 | if (nativeHls) return "hls-native"; | |
| 87 | // @ts-expect-error | |
| 88 | return (hls ??= import("/js/scripts/vendor/hls.js")) | |
| 89 | .then((hls): PlaybackMode => { | |
| 90 | // Polyfill HLS (Chrome 23, Firefox 42, IE11 on Win8) | |
| 91 | // TODO: ES modules have a minimum of Chrome 63 or Firefox 60 | |
| 92 | if (hls?.default?.isSupported?.() || hls?.isSupported()) { | |
| 93 | return "hls-polyfill"; | |
| 94 | } | |
| 95 | console.warn("does not support hls.js"); | |
| 96 | // Other browsers | |
| 97 | return "none"; | |
| 98 | }); | |
| 99 | }, | |
| 100 | ); | |
| 101 | } | |
| 102 | ||
| 95 | 103 | function hydrateVideoInner( |
| 96 | 104 | container: HTMLElement, |
| 97 | 105 | mode: PlaybackMode, |
| ... | ... | @@ -109,6 +117,38 @@ function hydrateVideoInner( |
| 109 | 117 | const dashFile = `${src}:/dash.mpd`; |
| 110 | 118 | const hlsFile = `${src}:/master.m3u8`; |
| 111 | 119 | |
| 120 | // data-streams lists which manifests exist ("dash", "hls", or both). a | |
| 121 | // manifest may be missing because the indexer has not finished the file | |
| 122 | // yet ("pending"), it skipped it (short/tiny videos, "none"), or that | |
| 123 | // encode failed. never pick a backend whose manifest would 404; fall back | |
| 124 | // to playing the original file directly. | |
| 125 | const streams = (video.getAttribute("data-streams") ?? "").split(","); | |
| 126 | const hasDash = streams.includes("dash"); | |
| 127 | const hasHls = streams.includes("hls"); | |
| 128 | if (!hasDash && !hasHls) { | |
| 129 | if (streams.includes("pending")) { | |
| 130 | const note = document.createElement("p"); | |
| 131 | note.className = "clover-video-pending"; | |
| 132 | note.textContent = "this video is still being processed, so quality selection is " | |
| 133 | + "unavailable. if it does not play, come back later :("; | |
| 134 | (container.querySelector("figcaption") ?? video.parentElement) | |
| 135 | ?.appendChild(note); | |
| 136 | } | |
| 137 | video.src = src; | |
| 138 | video.controls = true; | |
| 139 | onCloverVideoInit?.(id, video); | |
| 140 | return; | |
| 141 | } | |
| 142 | if (mode === "av1" && !hasDash) { | |
| 143 | // av1-capable browser, but only an hls manifest exists for this video | |
| 144 | decideHlsMode().then((m) => hydrateVideoInner(container, m)); | |
| 145 | return; | |
| 146 | } | |
| 147 | if ((mode === "hls-native" || mode === "hls-polyfill") && !hasHls) { | |
| 148 | // dash-only video on a browser that cannot decode av1: raw playback | |
| 149 | mode = "none"; | |
| 150 | } | |
| 151 | ||
| 112 | 152 | if (forceMode) { |
| 113 | 153 | const figcaption = container.querySelector("figcaption"); |
| 114 | 154 | if (figcaption) { |
src/tags/clover-video.marko+18-1| ... | ... | @@ -29,13 +29,29 @@ export interface Input { |
| 29 | 29 | : null |
| 30 | 30 | )> |
| 31 | 31 | |
| 32 | <const/{ width, height }=file.parseDimensions()> | |
| 32 | <const/{ width, height }=file.parseDimensions() ?? { | |
| 33 | width: 1920, | |
| 34 | height: 1080, | |
| 35 | }> | |
| 36 | <!-- which streaming manifests exist ("dash", "hls", or both). the player | |
| 37 | must only pick a backend whose manifest is actually present: dash can fail | |
| 38 | or lag behind hls for the same file. "pending" tells it to show a | |
| 39 | come-back-later note instead of silently failing --> | |
| 40 | <const/streams=( | |
| 41 | [ | |
| 42 | derived.get(file, "dash.mpd") && "dash", | |
| 43 | derived.get(file, "master.m3u8") && "hls", | |
| 44 | ] | |
| 45 | .filter(Boolean) | |
| 46 | .join(",") || (file.pending ? "pending" : "none") | |
| 47 | )> | |
| 33 | 48 | |
| 34 | 49 | <define/Inner> |
| 35 | 50 | <video |
| 36 | 51 | ...{ id } |
| 37 | 52 | class={ "clover-video": noFigure, minimal: noFigure && minimal } |
| 38 | 53 | data-src=file.path |
| 54 | data-streams=streams | |
| 39 | 55 | poster=poster ? `/file${poster.path}` : null |
| 40 | 56 | width=width |
| 41 | 57 | height=height |
| ... | ... | @@ -83,4 +99,5 @@ export interface Input { |
| 83 | 99 | client import "./clover-video.client.ts"; |
| 84 | 100 | |
| 85 | 101 | import { MediaFile } from "#src/file-viewer/models/MediaFile.ts"; |
| 102 | import * as derived from "#src/file-viewer/models/derived.ts"; | |
| 86 | 103 | import { UNWRAP, ASSERT } from "@clo/lib/assert"; |