| ... | @@ -1,1015 +0,0 @@ |
| 1 | // This file was started by AI and maintained by hand since. |
| 2 | import "@paperclover/console/inject"; |
| 3 | import { Progress } from "@paperclover/console/Progress"; |
| 4 | import { Spinner } from "@paperclover/console/Spinner"; |
| 5 | import assert from "node:assert"; |
| 6 | import { execFile } from "node:child_process"; |
| 7 | import { existsSync, Stats } from "node:fs"; |
| 8 | import * as fsp from "node:fs/promises"; |
| 9 | import * as path from "node:path"; |
| 10 | import { promisify } from "node:util"; |
| 11 | import { BlobAsset, cache, FilePermissions, MediaFile } from "../db.ts"; |
| 12 | import { formatDate, formatSize } from "./share.ts"; |
| 13 | import { highlightCode, type Language } from "./highlight.ts"; |
| 14 | |
| 15 | const execFileAsync = promisify(execFile); |
| 16 | |
| 17 | // Configuration |
| 18 | const FILE_ROOT = process.env.SCAN_FILE_ROOT; |
| 19 | if (!FILE_ROOT) { |
| 20 | throw new Error( |
| 21 | "FILE_ROOT environment variable not set (e.g. '/path/to/files')", |
| 22 | ); |
| 23 | } |
| 24 | const LOCAL_DIR = path.resolve(FILE_ROOT); |
| 25 | const DRY_RUN = process.argv.includes("--dry-run"); |
| 26 | const SHOULD_COMPRESS = true; |
| 27 | const VERBOSE = process.argv.includes("--verbose"); |
| 28 | const SHOULD_SCRUB = true; |
| 29 | const COMPRESS_STORE = process.env.COMPRESS_STORE || |
| 30 | path.join(process.cwd(), ".clover/compressed"); |
| 31 | |
| 32 | // Helper function for logging that respects verbose flag |
| 33 | function log(message: string, always = false): void { |
| 34 | if (always || VERBOSE) { |
| 35 | console.log(message); |
| 36 | } |
| 37 | } |
| 38 | |
| 39 | // File extensions that need duration metadata |
| 40 | const MEDIA_EXTENSIONS = new Set([ |
| 41 | ".mp4", |
| 42 | ".mkv", |
| 43 | ".webm", |
| 44 | ".avi", |
| 45 | ".mov", |
| 46 | ".mp3", |
| 47 | ".flac", |
| 48 | ".wav", |
| 49 | ".ogg", |
| 50 | ".m4a", |
| 51 | ]); |
| 52 | |
| 53 | // File extensions that need dimension metadata |
| 54 | const IMAGE_EXTENSIONS = new Set([ |
| 55 | ".jpg", |
| 56 | ".jpeg", |
| 57 | ".png", |
| 58 | ".gif", |
| 59 | ".webp", |
| 60 | ".avif", |
| 61 | ".heic", |
| 62 | ".svg", |
| 63 | ]); |
| 64 | |
| 65 | const VIDEO_EXTENSIONS = new Set([".mp4", ".mkv", ".webm", ".avi", ".mov"]); |
| 66 | |
| 67 | // File extensions that need metadata scrubbing |
| 68 | const SCRUB_EXTENSIONS = new Set([ |
| 69 | ".jpg", |
| 70 | ".jpeg", |
| 71 | ".png", |
| 72 | ".mov", |
| 73 | ".mp4", |
| 74 | ".m4a", |
| 75 | ]); |
| 76 | |
| 77 | const CODE_EXTENSIONS: Record<string, Language> = { |
| 78 | ".json": "json", |
| 79 | ".toml": "toml", |
| 80 | ".ts": "ts", |
| 81 | ".js": "ts", |
| 82 | ".tsx": "tsx", |
| 83 | ".jsx": "tsx", |
| 84 | ".css": "css", |
| 85 | ".py": "python", |
| 86 | ".lua": "lua", |
| 87 | ".sh": "shell", |
| 88 | ".bat": "dosbatch", |
| 89 | ".ps1": "powershell", |
| 90 | ".cmd": "dosbatch", |
| 91 | ".yaml": "yaml", |
| 92 | ".yml": "yaml", |
| 93 | ".zig": "zig", |
| 94 | ".astro": "astro", |
| 95 | ".mdx": "mdx", |
| 96 | ".xml": "xml", |
| 97 | ".jsonc": "json", |
| 98 | ".php": "php", |
| 99 | ".patch": "diff", |
| 100 | ".diff": "diff", |
| 101 | }; |
| 102 | |
| 103 | const READ_CONTENTS_EXTENSIONS = new Set([".txt", ".chat"]); |
| 104 | |
| 105 | // For files that have changed indexing logic, update the date here rescanning |
| 106 | // will reconstruct the entire file object. This way you can incrementally |
| 107 | // update new file types without having to reindex everything. |
| 108 | const lastUpdateTypes: Record<string, Date> = {}; |
| 109 | lastUpdateTypes[".lnk"] = new Date("2025-05-13 13:58:00"); |
| 110 | for (const ext in CODE_EXTENSIONS) { |
| 111 | lastUpdateTypes[ext] = new Date("2025-05-13 13:58:00"); |
| 112 | } |
| 113 | for (const ext of READ_CONTENTS_EXTENSIONS) { |
| 114 | lastUpdateTypes[ext] = new Date("2025-05-13 13:58:00"); |
| 115 | } |
| 116 | lastUpdateTypes[".diff"] = new Date("2025-05-18 13:58:00"); |
| 117 | lastUpdateTypes[".patch"] = new Date("2025-05-18 13:58:00"); |
| 118 | |
| 119 | // Helper functions for metadata extraction |
| 120 | async function calculateHash(filePath: string): Promise<string> { |
| 121 | try { |
| 122 | const hash = await execFileAsync("sha1sum", [filePath]); |
| 123 | return hash.stdout.split(" ")[0]; |
| 124 | } catch (error) { |
| 125 | console.error(`Error calculating hash for ${filePath}:`, error); |
| 126 | throw error; |
| 127 | } |
| 128 | } |
| 129 | |
| 130 | async function calculateDuration(filePath: string): Promise<number> { |
| 131 | try { |
| 132 | const ext = path.extname(filePath).toLowerCase(); |
| 133 | if (!MEDIA_EXTENSIONS.has(ext)) return 0; |
| 134 | |
| 135 | const { stdout } = await execFileAsync("ffprobe", [ |
| 136 | "-v", |
| 137 | "error", |
| 138 | "-show_entries", |
| 139 | "format=duration", |
| 140 | "-of", |
| 141 | "default=noprint_wrappers=1:nokey=1", |
| 142 | filePath, |
| 143 | ]); |
| 144 | return Math.ceil(parseFloat(stdout.trim())); |
| 145 | } catch (error) { |
| 146 | console.error(`Error calculating duration for ${filePath}:`, error); |
| 147 | return 0; // Return 0 for duration on error |
| 148 | } |
| 149 | } |
| 150 | |
| 151 | async function calculateDimensions(filePath: string): Promise<string> { |
| 152 | const ext = path.extname(filePath).toLowerCase(); |
| 153 | if (!IMAGE_EXTENSIONS.has(ext) && !VIDEO_EXTENSIONS.has(ext)) return ""; |
| 154 | |
| 155 | try { |
| 156 | if (ext === ".svg") { |
| 157 | // For SVG files, parse the file and extract width/height |
| 158 | const content = await fsp.readFile(filePath, "utf8"); |
| 159 | const widthMatch = content.match(/width="(\d+)"/); |
| 160 | const heightMatch = content.match(/height="(\d+)"/); |
| 161 | |
| 162 | if (widthMatch && heightMatch) { |
| 163 | return `${widthMatch[1]}x${heightMatch[1]}`; |
| 164 | } |
| 165 | } else if (IMAGE_EXTENSIONS.has(ext) || VIDEO_EXTENSIONS.has(ext)) { |
| 166 | // Use ffprobe for images and videos |
| 167 | const { stdout } = await execFileAsync("ffprobe", [ |
| 168 | "-v", |
| 169 | "error", |
| 170 | "-select_streams", |
| 171 | "v:0", |
| 172 | "-show_entries", |
| 173 | "stream=width,height", |
| 174 | "-of", |
| 175 | "csv=s=x:p=0", |
| 176 | filePath, |
| 177 | ]); |
| 178 | return stdout.trim(); |
| 179 | } |
| 180 | } catch (error) { |
| 181 | console.error(`Error calculating dimensions for ${filePath}:`, error); |
| 182 | } |
| 183 | |
| 184 | return ""; |
| 185 | } |
| 186 | |
| 187 | // Helper function to check and remove location metadata |
| 188 | async function scrubLocationMetadata( |
| 189 | filePath: string, |
| 190 | stats: Stats, |
| 191 | ): Promise<boolean> { |
| 192 | try { |
| 193 | const ext = path.extname(filePath).toLowerCase(); |
| 194 | if (!SCRUB_EXTENSIONS.has(ext)) return false; |
| 195 | |
| 196 | let hasLocation = false; |
| 197 | let args: string[] = []; |
| 198 | |
| 199 | // Check for location metadata based on file type |
| 200 | const tempOutput = path.join( |
| 201 | path.dirname(filePath), |
| 202 | `.tmp.${path.basename(filePath)}`, |
| 203 | ); |
| 204 | switch (ext) { |
| 205 | case ".jpg": |
| 206 | case ".jpeg": |
| 207 | case ".png": |
| 208 | // Check for GPS tags in EXIF |
| 209 | const { stdout: gpsCheck } = await execFileAsync("exiftool", [ |
| 210 | "-gps:all", |
| 211 | filePath, |
| 212 | ]); |
| 213 | hasLocation = gpsCheck.trim().length > 0; |
| 214 | args = ["-gps:all=", filePath, "-o", tempOutput]; |
| 215 | break; |
| 216 | case ".mov": |
| 217 | case ".mp4": |
| 218 | // Check for GPS metadata in video files |
| 219 | const { stdout: videoCheck } = await execFileAsync("exiftool", [ |
| 220 | "-ee", |
| 221 | "-G3", |
| 222 | "-s", |
| 223 | filePath, |
| 224 | ]); |
| 225 | hasLocation = videoCheck.includes("GPS") || |
| 226 | videoCheck.includes("Location"); |
| 227 | args = ["-gps:all=", "-xmp:all=", filePath, "-o", tempOutput]; |
| 228 | break; |
| 229 | case ".m4a": |
| 230 | // Check for location and other metadata in m4a files |
| 231 | const { stdout: m4aCheck } = await execFileAsync("exiftool", [ |
| 232 | "-ee", |
| 233 | "-G3", |
| 234 | "-s", |
| 235 | filePath, |
| 236 | ]); |
| 237 | hasLocation = m4aCheck.includes("GPS") || |
| 238 | m4aCheck.includes("Location") || |
| 239 | m4aCheck.includes("Filename") || |
| 240 | m4aCheck.includes("Title"); |
| 241 | |
| 242 | if (hasLocation) { |
| 243 | args = [ |
| 244 | "-gps:all=", |
| 245 | "-location:all=", |
| 246 | "-filename:all=", |
| 247 | "-title=", |
| 248 | "-m4a:all=", |
| 249 | filePath, |
| 250 | "-o", |
| 251 | tempOutput, |
| 252 | ]; |
| 253 | } |
| 254 | break; |
| 255 | } |
| 256 | |
| 257 | const accessTime = stats.atime; |
| 258 | const modTime = stats.mtime; |
| 259 | |
| 260 | let backup: string | null = null; |
| 261 | try { |
| 262 | if (hasLocation) { |
| 263 | if (DRY_RUN) return true; |
| 264 | |
| 265 | // Prepare a backup |
| 266 | const tmp = path.join( |
| 267 | path.dirname(filePath), |
| 268 | `.tmp.backup.${path.basename(filePath)}`, |
| 269 | ); |
| 270 | await fsp.copyFile(filePath, tmp); |
| 271 | await fsp.utimes(tmp, accessTime, modTime); |
| 272 | backup = tmp; |
| 273 | |
| 274 | // Remove metadata |
| 275 | await execFileAsync("exiftool", args); |
| 276 | if (!existsSync(tempOutput)) { |
| 277 | throw new Error(`Failed to create output file: ${tempOutput}`); |
| 278 | } |
| 279 | |
| 280 | // Restore original timestamps |
| 281 | await fsp.rename(tempOutput, filePath); |
| 282 | await fsp.utimes(filePath, accessTime, modTime); |
| 283 | |
| 284 | // Backup is no longer needed |
| 285 | await fsp.unlink(backup); |
| 286 | |
| 287 | log( |
| 288 | `Scrubbed location metadata in ${path.relative(LOCAL_DIR, filePath)}`, |
| 289 | true, |
| 290 | ); |
| 291 | return true; |
| 292 | } |
| 293 | } catch (error) { |
| 294 | if (backup) { |
| 295 | await fsp.rename(backup, filePath); |
| 296 | } |
| 297 | if (existsSync(tempOutput)) { |
| 298 | await fsp.unlink(tempOutput); |
| 299 | } |
| 300 | throw error; |
| 301 | } |
| 302 | } catch (error) { |
| 303 | console.error(`Error scrubbing metadata for ${filePath}:`, error); |
| 304 | } |
| 305 | |
| 306 | return false; |
| 307 | } |
| 308 | |
| 309 | // Queue implementation for parallel processing |
| 310 | type AsyncQueueProcessor<T> = (s: Spinner, item: T) => Promise<void>; |
| 311 | class AsyncQueue<T> { |
| 312 | private queue: T[] = []; |
| 313 | private running = 0; |
| 314 | private maxConcurrent: number; |
| 315 | private processed = 0; |
| 316 | private progress?: Progress<{ active: Spinner[] }>; |
| 317 | private name: string; |
| 318 | private estimate?: number; |
| 319 | |
| 320 | constructor(name: string, maxConcurrent: number) { |
| 321 | this.maxConcurrent = maxConcurrent; |
| 322 | this.name = name; |
| 323 | } |
| 324 | |
| 325 | setEstimate(estimate: number) { |
| 326 | this.estimate = estimate; |
| 327 | if (this.progress) { |
| 328 | this.progress.total = Math.max( |
| 329 | this.processed + this.queue.length, |
| 330 | estimate, |
| 331 | ); |
| 332 | } |
| 333 | } |
| 334 | |
| 335 | getProgress() { |
| 336 | if (!this.progress) { |
| 337 | this.progress = new Progress({ |
| 338 | spinner: null, |
| 339 | text: ({ active }) => { |
| 340 | const now = performance.now(); |
| 341 | let text = `[${this.processed}/${ |
| 342 | this.processed + this.queue.length |
| 343 | }] ${this.name}`; |
| 344 | let n = 0; |
| 345 | for (const item of active) { |
| 346 | let itemText = "- " + item.format(now); |
| 347 | text += `\n` + |
| 348 | itemText.slice(0, Math.max(0, process.stdout.columns - 1)); |
| 349 | if (n > 10) { |
| 350 | text += `\n ... + ${active.length - n} more`; |
| 351 | break; |
| 352 | } |
| 353 | n++; |
| 354 | } |
| 355 | return text; |
| 356 | }, |
| 357 | props: { |
| 358 | active: [] as Spinner[], |
| 359 | }, |
| 360 | }); |
| 361 | this.progress.total = this.estimate ?? 0; |
| 362 | this.progress.value = 0; |
| 363 | this.progress.fps = 30; |
| 364 | } |
| 365 | return this.progress; |
| 366 | } |
| 367 | |
| 368 | async add(item: T, processor: AsyncQueueProcessor<T>): Promise<void> { |
| 369 | this.queue.push(item); |
| 370 | this.getProgress().total = Math.max( |
| 371 | this.processed + this.queue.length, |
| 372 | this.estimate ?? 0, |
| 373 | ); |
| 374 | return this.processNext(processor); |
| 375 | } |
| 376 | |
| 377 | async addBatch(items: T[], processor: AsyncQueueProcessor<T>): Promise<void> { |
| 378 | this.queue.push(...items); |
| 379 | this.getProgress().total = Math.max( |
| 380 | this.processed + this.queue.length, |
| 381 | this.estimate ?? 0, |
| 382 | ); |
| 383 | return this.processNext(processor); |
| 384 | } |
| 385 | |
| 386 | private async processNext(processor: AsyncQueueProcessor<T>): Promise<void> { |
| 387 | if (this.running >= this.maxConcurrent || this.queue.length === 0) { |
| 388 | return; |
| 389 | } |
| 390 | |
| 391 | const item = this.queue.shift(); |
| 392 | if (!item) return; |
| 393 | |
| 394 | this.running++; |
| 395 | |
| 396 | try { |
| 397 | const progress = this.getProgress(); |
| 398 | |
| 399 | let itemText = ""; |
| 400 | if (typeof item === "string") { |
| 401 | itemText = item; |
| 402 | } else if (typeof item === "object" && item !== null && "path" in item) { |
| 403 | itemText = "" + item.path; |
| 404 | } else { |
| 405 | itemText = JSON.stringify(item); |
| 406 | } |
| 407 | if (itemText.startsWith(LOCAL_DIR)) { |
| 408 | itemText = path.relative(LOCAL_DIR, itemText); |
| 409 | } |
| 410 | |
| 411 | const spinner = new Spinner(itemText); |
| 412 | spinner.stop(); |
| 413 | progress.props.active.unshift(spinner); |
| 414 | await processor(spinner, item); |
| 415 | progress.props = { |
| 416 | active: progress.props.active.filter((s) => s !== spinner), |
| 417 | }; |
| 418 | this.processed++; |
| 419 | progress.value = this.processed; |
| 420 | } catch (error) { |
| 421 | console.error(`Error processing ${this.name} queue item:`, error); |
| 422 | this.processed++; |
| 423 | this.getProgress().value = this.processed; |
| 424 | } finally { |
| 425 | this.running--; |
| 426 | await this.processNext(processor); |
| 427 | } |
| 428 | } |
| 429 | |
| 430 | async waitForCompletion(): Promise<void> { |
| 431 | if (this.queue.length === 0 && this.running === 0) { |
| 432 | if (this.processed > 0) { |
| 433 | this.#success(); |
| 434 | } |
| 435 | return; |
| 436 | } |
| 437 | |
| 438 | return new Promise((resolve) => { |
| 439 | const checkInterval = setInterval(() => { |
| 440 | if (this.queue.length === 0 && this.running === 0) { |
| 441 | clearInterval(checkInterval); |
| 442 | this.#success(); |
| 443 | resolve(); |
| 444 | } |
| 445 | }, 100); |
| 446 | }); |
| 447 | } |
| 448 | |
| 449 | #success() { |
| 450 | this.getProgress().success(`${this.processed} ${this.name}`); |
| 451 | } |
| 452 | } |
| 453 | |
| 454 | function skipBasename(basename: string): boolean { |
| 455 | // dot files must be incrementally tracked |
| 456 | if (basename === ".dirsort") return true; |
| 457 | if (basename === ".friends") return true; |
| 458 | |
| 459 | return ( |
| 460 | basename.startsWith(".") || |
| 461 | basename.startsWith("._") || |
| 462 | basename.startsWith(".tmp") || |
| 463 | basename === ".DS_Store" || |
| 464 | basename.toLowerCase() === "thumbs.db" || |
| 465 | basename.toLowerCase() === "desktop.ini" |
| 466 | ); |
| 467 | } |
| 468 | |
| 469 | // File system scanner |
| 470 | class FileSystemScanner { |
| 471 | private visitedPaths = new Set<string>(); |
| 472 | private previousPaths = new Set<string>(); |
| 473 | private dirQueue = new AsyncQueue<string>("Scan Directories", 10); |
| 474 | private fileQueue = new AsyncQueue<{ path: string; stat: any }>( |
| 475 | "File metadata", |
| 476 | 20, |
| 477 | ); |
| 478 | private compressQueue: AsyncQueue<{ file: MediaFile; path: string }> | null = |
| 479 | SHOULD_COMPRESS ? new AsyncQueue("Compress Assets", 10) : null; |
| 480 | |
| 481 | private getDbPath(localPath: string): string { |
| 482 | // Convert local file system path to database path |
| 483 | const relativePath = path.relative(LOCAL_DIR, localPath); |
| 484 | return "/" + relativePath.split(path.sep).join(path.posix.sep); |
| 485 | } |
| 486 | |
| 487 | private getLocalPath(dbPath: string): string { |
| 488 | // Convert database path to local file system path |
| 489 | return path.join(LOCAL_DIR, dbPath.slice(1)); |
| 490 | } |
| 491 | |
| 492 | async scanFile(s: Spinner, filePath: string, stat: any): Promise<void> { |
| 493 | const dbPath = this.getDbPath(filePath); |
| 494 | |
| 495 | // Skip hidden files |
| 496 | const basename = path.basename(filePath); |
| 497 | if (skipBasename(basename)) { |
| 498 | return; |
| 499 | } |
| 500 | |
| 501 | this.visitedPaths.add(dbPath); |
| 502 | |
| 503 | // Get existing file info from db |
| 504 | const existingFile = MediaFile.getByPath(dbPath); |
| 505 | |
| 506 | // Determine which date to use (for date protection) |
| 507 | let dateToUse = stat.mtime; |
| 508 | const year2025Start = new Date("2025-01-01T00:00:00Z"); |
| 509 | |
| 510 | if ( |
| 511 | existingFile && |
| 512 | existingFile.date < year2025Start && |
| 513 | stat.mtime >= year2025Start |
| 514 | ) { |
| 515 | console.error( |
| 516 | `Error: ${dbPath} is ${ |
| 517 | formatDate( |
| 518 | existingFile.date, |
| 519 | ) |
| 520 | }, got modified to ${formatDate(stat.mtime)}`, |
| 521 | ); |
| 522 | dateToUse = existingFile.date; |
| 523 | } |
| 524 | |
| 525 | // Check if we need to reprocess the file |
| 526 | if (existingFile && existingFile.size === stat.size && existingFile.hash) { |
| 527 | maybe_skip: { |
| 528 | const lastUpdateDate = lastUpdateTypes[path.extname(filePath)]; |
| 529 | if (lastUpdateDate && existingFile.lastUpdateDate < lastUpdateDate) { |
| 530 | console.log( |
| 531 | `Reprocessing ${dbPath} because indexing logic changed after ${ |
| 532 | formatDate( |
| 533 | lastUpdateDate, |
| 534 | ) |
| 535 | }`, |
| 536 | ); |
| 537 | break maybe_skip; |
| 538 | } |
| 539 | |
| 540 | if (SHOULD_COMPRESS && existingFile.processed !== 2) { |
| 541 | this.compressQueue!.add( |
| 542 | { file: existingFile, path: dbPath }, |
| 543 | this.compressFile.bind(this), |
| 544 | ); |
| 545 | } |
| 546 | |
| 547 | // File hasn't changed, no need to reprocess |
| 548 | MediaFile.createFile({ |
| 549 | path: dbPath, |
| 550 | date: dateToUse, |
| 551 | hash: existingFile.hash, |
| 552 | size: stat.size, |
| 553 | duration: existingFile.duration, |
| 554 | dimensions: existingFile.dimensions, |
| 555 | content: existingFile.contents, |
| 556 | }); |
| 557 | return; |
| 558 | } |
| 559 | } |
| 560 | |
| 561 | // Process the file |
| 562 | log(`Processing file: ${dbPath}`); |
| 563 | |
| 564 | // Scrub location metadata if needed |
| 565 | if (SHOULD_SCRUB) { |
| 566 | if (await scrubLocationMetadata(filePath, stat)) { |
| 567 | // Re-stat the file in case it was modified |
| 568 | const newStat = await fsp.stat(filePath); |
| 569 | stat.size = newStat.size; |
| 570 | } |
| 571 | } |
| 572 | |
| 573 | // Extract content |
| 574 | const hash = await calculateHash(filePath); |
| 575 | let content = ""; |
| 576 | if (filePath.endsWith(".lnk")) { |
| 577 | content = (await fsp.readFile(filePath, "utf8")).trim(); |
| 578 | } |
| 579 | const language = CODE_EXTENSIONS[path.extname(filePath)]; |
| 580 | if (language) { |
| 581 | read_code: { |
| 582 | // An issue is that .ts is an overloaded extension, shared between |
| 583 | // 'transport stream' and 'typescript'. |
| 584 | // |
| 585 | // Filter used here is: |
| 586 | // - more than 1mb |
| 587 | // - invalid UTF-8 |
| 588 | if (stat.size > 1_000_000) break read_code; |
| 589 | let code; |
| 590 | const buf = await fsp.readFile(filePath); |
| 591 | try { |
| 592 | code = new TextDecoder("utf-8", { fatal: true }).decode(buf); |
| 593 | } catch (error) { |
| 594 | break read_code; |
| 595 | } |
| 596 | content = await highlightCode(code, language); |
| 597 | } |
| 598 | } |
| 599 | if (!content && READ_CONTENTS_EXTENSIONS.has(path.extname(filePath))) { |
| 600 | content = await fsp.readFile(filePath, "utf8"); |
| 601 | } |
| 602 | // End extract content |
| 603 | |
| 604 | if (hash === existingFile?.hash) { |
| 605 | MediaFile.createFile({ |
| 606 | path: dbPath, |
| 607 | date: dateToUse, |
| 608 | hash, |
| 609 | size: stat.size, |
| 610 | duration: existingFile.duration, |
| 611 | dimensions: existingFile.dimensions, |
| 612 | content, |
| 613 | }); |
| 614 | return; |
| 615 | } else if (existingFile) { |
| 616 | if (existingFile.processed === 2) { |
| 617 | if (BlobAsset.decrementOrDelete(existingFile.hash)) { |
| 618 | log( |
| 619 | `Deleted compressed asset ${existingFile.hash}.{gzip, zstd}`, |
| 620 | true, |
| 621 | ); |
| 622 | await fsp.unlink( |
| 623 | path.join( |
| 624 | COMPRESS_STORE, |
| 625 | existingFile.hash.substring(0, 2), |
| 626 | existingFile.hash + ".gz", |
| 627 | ), |
| 628 | ); |
| 629 | await fsp.unlink( |
| 630 | path.join( |
| 631 | COMPRESS_STORE, |
| 632 | existingFile.hash.substring(0, 2), |
| 633 | existingFile.hash + ".zstd", |
| 634 | ), |
| 635 | ); |
| 636 | } |
| 637 | } |
| 638 | } |
| 639 | const [duration, dimensions] = await Promise.all([ |
| 640 | calculateDuration(filePath), |
| 641 | calculateDimensions(filePath), |
| 642 | ]); |
| 643 | |
| 644 | // Update database with all metadata |
| 645 | MediaFile.createFile({ |
| 646 | path: dbPath, |
| 647 | date: dateToUse, |
| 648 | hash, |
| 649 | size: stat.size, |
| 650 | duration, |
| 651 | dimensions, |
| 652 | content, |
| 653 | }); |
| 654 | |
| 655 | if (SHOULD_COMPRESS) { |
| 656 | this.compressQueue!.add( |
| 657 | { |
| 658 | file: MediaFile.getByPath(dbPath)!, |
| 659 | path: dbPath, |
| 660 | }, |
| 661 | this.compressFile.bind(this), |
| 662 | ); |
| 663 | } |
| 664 | } |
| 665 | |
| 666 | async compressFile(s: Spinner, { file }: { file: MediaFile }): Promise<void> { |
| 667 | log(`Compressing file: ${file.path}`); |
| 668 | if (DRY_RUN) return; |
| 669 | |
| 670 | const filePath = path.join(FILE_ROOT!, file.path); |
| 671 | |
| 672 | const hash = file.hash; |
| 673 | const firstTwoChars = hash.substring(0, 2); |
| 674 | const compressDir = `${COMPRESS_STORE}/${firstTwoChars}`; |
| 675 | const compressPath = `${compressDir}/${hash}`; |
| 676 | |
| 677 | // Create directory structure if it doesn't exist |
| 678 | await fsp.mkdir(compressDir, { recursive: true }); |
| 679 | |
| 680 | // Compress the file with gzip |
| 681 | const blob = BlobAsset.putOrIncrement(hash); |
| 682 | if (blob.refs > 1) { |
| 683 | log( |
| 684 | `Skipping compression of ${filePath} because it already exists in ${compressPath}`, |
| 685 | ); |
| 686 | return; |
| 687 | } |
| 688 | // Check if already exists |
| 689 | if (existsSync(compressPath + ".gz")) { |
| 690 | file.setCompressed(true); |
| 691 | return; |
| 692 | } |
| 693 | try { |
| 694 | const gzipProcess = Bun.spawn(["gzip", "-c", filePath, "-9"], { |
| 695 | stdout: Bun.file(compressPath + ".gz"), |
| 696 | }); |
| 697 | const zstdProcess = Bun.spawn(["zstd", "-c", filePath, "-9"], { |
| 698 | stdout: Bun.file(compressPath + ".zstd"), |
| 699 | }); |
| 700 | const [gzipExited, zstdExited] = await Promise.all([ |
| 701 | gzipProcess.exited, |
| 702 | zstdProcess.exited, |
| 703 | ]); |
| 704 | assert(gzipExited === 0); |
| 705 | assert(zstdExited === 0); |
| 706 | assert(existsSync(compressPath + ".gz")); |
| 707 | assert(existsSync(compressPath + ".zstd")); |
| 708 | file.setCompressed(true); |
| 709 | } catch (error) { |
| 710 | console.error(`Error compressing file ${filePath}:`, error); |
| 711 | BlobAsset.decrementOrDelete(hash); |
| 712 | file.setCompressed(false); |
| 713 | } |
| 714 | } |
| 715 | |
| 716 | async scanDirectory(s: Spinner, dirPath: string): Promise<void> { |
| 717 | const dbPath = this.getDbPath(dirPath); |
| 718 | |
| 719 | this.visitedPaths.add(dbPath); |
| 720 | |
| 721 | // Create or update directory entry |
| 722 | log(`Scanning directory: ${dbPath}`); |
| 723 | if (!DRY_RUN) { |
| 724 | MediaFile.createOrUpdateDirectory(dbPath); |
| 725 | } |
| 726 | |
| 727 | try { |
| 728 | const entries = await fsp.readdir(dirPath, { withFileTypes: true }); |
| 729 | |
| 730 | // Process files and subdirectories |
| 731 | for (const entry of entries) { |
| 732 | const entryPath = path.join(dirPath, entry.name); |
| 733 | |
| 734 | // Skip hidden files and system files |
| 735 | if (skipBasename(entry.name)) { |
| 736 | continue; |
| 737 | } |
| 738 | |
| 739 | if (entry.isDirectory()) { |
| 740 | // Queue subdirectory for scanning |
| 741 | this.dirQueue.add(entryPath, this.scanDirectory.bind(this)); |
| 742 | } else if (entry.isFile()) { |
| 743 | // Queue file for processing |
| 744 | const stat = await fsp.stat(entryPath); |
| 745 | |
| 746 | this.fileQueue.add( |
| 747 | { path: entryPath, stat }, |
| 748 | async (s, item) => await this.scanFile(s, item.path, item.stat), |
| 749 | ); |
| 750 | } |
| 751 | } |
| 752 | } catch (error) { |
| 753 | console.error(`Error scanning directory ${dirPath}:`, error); |
| 754 | } |
| 755 | } |
| 756 | |
| 757 | async processDirectoryMetadata(dirPath: string): Promise<void> { |
| 758 | const dbPath = this.getDbPath(dirPath); |
| 759 | const dir = MediaFile.getByPath(dbPath); |
| 760 | |
| 761 | if (!dir || dir.kind !== MediaFile.Kind.directory) { |
| 762 | return; |
| 763 | } |
| 764 | |
| 765 | if (DRY_RUN) return; |
| 766 | |
| 767 | const children = dir.getChildren(); |
| 768 | |
| 769 | // Calculate directory metadata |
| 770 | let totalSize = 0; |
| 771 | let newestDate = new Date(0); |
| 772 | let allHashes = ""; |
| 773 | |
| 774 | // Check for readme.txt |
| 775 | let readmeContent = ""; |
| 776 | |
| 777 | try { |
| 778 | readmeContent = await fsp.readFile( |
| 779 | path.join(dirPath, "readme.txt"), |
| 780 | "utf8", |
| 781 | ); |
| 782 | } catch (error: any) { |
| 783 | console.info(`no readme ${dirPath}`); |
| 784 | if (error.code !== "ENOENT") { |
| 785 | console.error(`Error reading readme.txt in ${dirPath}:`, error); |
| 786 | } |
| 787 | } |
| 788 | |
| 789 | let dirsort: string[] | null = null; |
| 790 | try { |
| 791 | dirsort = (await fsp.readFile(path.join(dirPath, ".dirsort"), "utf8")) |
| 792 | .split("\n") |
| 793 | .map((x) => x.trim()) |
| 794 | .filter(Boolean); |
| 795 | } catch (error: any) { |
| 796 | if (error.code !== "ENOENT") { |
| 797 | console.error(`Error reading .dirsort in ${dirPath}:`, error); |
| 798 | } |
| 799 | } |
| 800 | |
| 801 | if (await fsp.exists(path.join(dirPath, ".friends"))) { |
| 802 | FilePermissions.setPermissions(dbPath, 1); |
| 803 | } else { |
| 804 | FilePermissions.setPermissions(dbPath, 0); |
| 805 | } |
| 806 | |
| 807 | // Process children |
| 808 | for (const child of children) { |
| 809 | totalSize += child.size; |
| 810 | allHashes += child.hash; |
| 811 | |
| 812 | // Update newest date, ignoring readme.txt |
| 813 | if (!child.path.endsWith("/readme.txt") && child.date > newestDate) { |
| 814 | newestDate = child.date; |
| 815 | } |
| 816 | } |
| 817 | |
| 818 | // Create a hash for the directory |
| 819 | const dirHash = new Bun.CryptoHasher("sha1") |
| 820 | .update(dbPath + allHashes) |
| 821 | .digest("hex"); |
| 822 | |
| 823 | // Update directory metadata |
| 824 | MediaFile.markDirectoryProcessed({ |
| 825 | id: dir.id, |
| 826 | timestamp: newestDate, |
| 827 | contents: readmeContent, |
| 828 | size: totalSize, |
| 829 | hash: dirHash, |
| 830 | dirsort, |
| 831 | }); |
| 832 | } |
| 833 | |
| 834 | async findDeletedFiles(): Promise<void> { |
| 835 | if (DRY_RUN) return; |
| 836 | |
| 837 | // Find all paths that exist in the DB but not in the filesystem |
| 838 | const deletedPaths = Array.from(this.previousPaths).filter( |
| 839 | (path) => !this.visitedPaths.has(path), |
| 840 | ); |
| 841 | |
| 842 | for (const dbPath of deletedPaths) { |
| 843 | const file = MediaFile.getByPath(dbPath); |
| 844 | if (!file) continue; |
| 845 | |
| 846 | log(`Item Deleted: ${dbPath}`, true); |
| 847 | if (file.processed === 2) { |
| 848 | if (BlobAsset.decrementOrDelete(file.hash)) { |
| 849 | log(`Deleted compressed asset ${file.hash}.{gzip, zstd}`, true); |
| 850 | await fsp.unlink( |
| 851 | path.join( |
| 852 | COMPRESS_STORE, |
| 853 | file.hash.substring(0, 2), |
| 854 | file.hash + ".gz", |
| 855 | ), |
| 856 | ); |
| 857 | await fsp.unlink( |
| 858 | path.join( |
| 859 | COMPRESS_STORE, |
| 860 | file.hash.substring(0, 2), |
| 861 | file.hash + ".zstd", |
| 862 | ), |
| 863 | ); |
| 864 | } |
| 865 | } |
| 866 | MediaFile.deleteByPath(dbPath); |
| 867 | } |
| 868 | } |
| 869 | |
| 870 | async loadPreviousPaths(): Promise<void> { |
| 871 | // Get all files and directories from the database |
| 872 | // This uses a custom query to get all paths at once |
| 873 | const getAllPathsQuery = cache |
| 874 | .prepare(`SELECT path, kind FROM media_files`) |
| 875 | .all() as { |
| 876 | path: string; |
| 877 | kind: MediaFile.Kind; |
| 878 | }[]; |
| 879 | |
| 880 | let dirs = 0; |
| 881 | let files = 0; |
| 882 | for (const row of getAllPathsQuery) { |
| 883 | this.previousPaths.add(row.path); |
| 884 | if (row.kind === MediaFile.Kind.directory) { |
| 885 | dirs++; |
| 886 | } else { |
| 887 | files++; |
| 888 | } |
| 889 | } |
| 890 | |
| 891 | this.dirQueue.setEstimate(dirs); |
| 892 | this.fileQueue.setEstimate(files); |
| 893 | |
| 894 | // log(`Loaded ${this.previousPaths.size} paths from database`, true); |
| 895 | } |
| 896 | |
| 897 | async scan(): Promise<void> { |
| 898 | log(`Starting file system scan in ${LOCAL_DIR}`, true); |
| 899 | |
| 900 | // Check if the root directory exists and is accessible |
| 901 | try { |
| 902 | const rootStat = await fsp.stat(LOCAL_DIR); |
| 903 | if (!rootStat.isDirectory()) { |
| 904 | throw new Error(`${LOCAL_DIR} is not a directory`); |
| 905 | } |
| 906 | } catch (error) { |
| 907 | console.error(`Error: Cannot access root directory ${LOCAL_DIR}`, error); |
| 908 | console.error( |
| 909 | `Aborting scan to prevent database corruption. Please check if the volume is mounted.`, |
| 910 | ); |
| 911 | process.exit(1); |
| 912 | } |
| 913 | |
| 914 | await this.loadPreviousPaths(); |
| 915 | |
| 916 | await this.dirQueue.add(LOCAL_DIR, this.scanDirectory.bind(this)); |
| 917 | |
| 918 | await this.dirQueue.waitForCompletion(); |
| 919 | await this.fileQueue.waitForCompletion(); |
| 920 | |
| 921 | await this.findDeletedFiles(); |
| 922 | |
| 923 | const allDirs = Array.from(this.visitedPaths) |
| 924 | .filter((path) => { |
| 925 | const file = MediaFile.getByPath(path); |
| 926 | return file && file.kind === MediaFile.Kind.directory; |
| 927 | }) |
| 928 | .sort((a, b) => b.length - a.length); |
| 929 | |
| 930 | const dirMetadataQueue = new AsyncQueue<string>("Directory Metadata", 10); |
| 931 | for (const dirPath of allDirs) { |
| 932 | await this.processDirectoryMetadata(this.getLocalPath(dirPath)); |
| 933 | } |
| 934 | |
| 935 | await dirMetadataQueue.waitForCompletion(); |
| 936 | |
| 937 | if (SHOULD_COMPRESS) { |
| 938 | await this.compressQueue!.waitForCompletion(); |
| 939 | } |
| 940 | |
| 941 | log("Scan completed successfully!", true); |
| 942 | } |
| 943 | } |
| 944 | |
| 945 | // Main execution |
| 946 | function showHelp() { |
| 947 | console.log(` |
| 948 | MediaFile Scanner - Index filesystem content for paperclover.net |
| 949 | |
| 950 | Environment variables: |
| 951 | FILE_ROOT Required. Path to the directory to scan |
| 952 | COMPRESS_STORE Optional. Path to store compressed files (default: .clover/compressed) |
| 953 | |
| 954 | Options: |
| 955 | --help Show this help message |
| 956 | --dry-run Don't make any changes to the database |
| 957 | --verbose Show detailed output |
| 958 | |
| 959 | Usage: |
| 960 | bun ./media/scan.ts [options] |
| 961 | |
| 962 | `); |
| 963 | process.exit(0); |
| 964 | } |
| 965 | |
| 966 | { |
| 967 | // Show help if requested |
| 968 | if (process.argv.includes("--help")) { |
| 969 | showHelp(); |
| 970 | process.exit(0); |
| 971 | } |
| 972 | |
| 973 | // Check if the root directory exists before starting |
| 974 | if (!existsSync(LOCAL_DIR)) { |
| 975 | console.error( |
| 976 | `Error: Root directory ${LOCAL_DIR} does not exist or is not accessible.`, |
| 977 | ); |
| 978 | console.error(`Please check if the volume is mounted correctly.`); |
| 979 | process.exit(1); |
| 980 | } |
| 981 | |
| 982 | const startTime = Date.now(); |
| 983 | |
| 984 | try { |
| 985 | const scanner = new FileSystemScanner(); |
| 986 | await scanner.scan(); |
| 987 | |
| 988 | const endTime = Date.now(); |
| 989 | log(`Scan completed in ${(endTime - startTime) / 1000} seconds`, true); |
| 990 | |
| 991 | const rootDir = MediaFile.getByPath("/")!; |
| 992 | const totalEntries = cache |
| 993 | .prepare(`SELECT COUNT(*) as count FROM media_files`) |
| 994 | .get() as { count: number }; |
| 995 | const totalDuration = cache |
| 996 | .prepare(`SELECT SUM(duration) as duration FROM media_files`) |
| 997 | .get() as { duration: number }; |
| 998 | console.log(); |
| 999 | console.log("Global Stats"); |
| 1000 | console.log(` Entry count: ${totalEntries.count}`); |
| 1001 | console.log(` Uncompressed size: ${formatSize(rootDir.size)}`); |
| 1002 | console.log( |
| 1003 | ` Total audio/video duration: ${ |
| 1004 | ( |
| 1005 | totalDuration.duration / |
| 1006 | 60 / |
| 1007 | 60 |
| 1008 | ).toFixed(1) |
| 1009 | } hours`, |
| 1010 | ); |
| 1011 | } catch (error) { |
| 1012 | console.error("Error during scan:", error); |
| 1013 | process.exit(1); |
| 1014 | } |
| 1015 | } |