diff --git a/backend/package.json b/backend/package.json index 4d250b9..edf1c54 100755 --- a/backend/package.json +++ b/backend/package.json @@ -19,6 +19,7 @@ "test:integration:json": "jest -c jest.integration.config.js --runInBand --json --outputFile=integration-results.json", "test:integration:cov": "jest -c jest.integration.config.js --runInBand --coverage --forceExit", "bench:hashing": "tsx scripts/bench-hash-latency.ts", + "backfill:images": "tsx scripts/backfill-image-reencode.ts", "db:test:up": "docker compose -f docker-compose.test.yml up -d", "db:test:down": "docker compose -f docker-compose.test.yml down -v", "migrate:up": "node migrate.js up", diff --git a/backend/scripts/backfill-image-reencode.ts b/backend/scripts/backfill-image-reencode.ts new file mode 100644 index 0000000..c16b048 --- /dev/null +++ b/backend/scripts/backfill-image-reencode.ts @@ -0,0 +1,160 @@ +/** + * Applies #226's re-encoding to the photos that were stored before it existed. + * + * New uploads are handled in the request path. Everything already on the volume + * still carries whatever the camera wrote, including the coordinates the photo + * was taken at, and is still served publicly. This is the other half. + * + * The transform is lossy and there is no undo, so: + * + * - It reports by default and changes nothing without --apply. + * - It is idempotent. `needsProcessing` skips a file that is already stripped + * and already within bounds, so a second run is not a second lossy pass. + * - `reencodeInPlace` writes to a temporary file and renames, so an + * interruption cannot leave a half-written image being served. + * - It never renames the stored file, so item_images.image_path stays correct + * and no database write is needed at all. + * + * Usage + * ----- + * npm run backfill:images # report only + * npm run backfill:images -- --apply # rewrite the files + * + * Point it at QA first. Compare a handful of images by eye before production, + * and take a backup that you have confirmed restores. + */ + +import sharp from 'sharp'; +import { promises as fs } from 'fs'; +import path from 'path'; +import { pool } from '../src/db'; +import { needsProcessing, reencodeInPlace } from '../src/imageProcessing'; + +const UPLOADS_DIR = process.env.UPLOADS_DIR || '/app/uploads'; +const APPLY = process.argv.includes('--apply'); + +// The stored extension is the file's real type — uploadTypes.ts derives it from +// the validated content type on the way in, so it can be trusted on the way +// back out. Anything else is a file this application would refuse to serve. +const TYPE_FOR_EXTENSION: Readonly> = { + '.jpg': 'image/jpeg', + '.png': 'image/png', + '.webp': 'image/webp' +}; + +interface Totals { + seen: number; + missing: number; + skipped: number; + unrecognised: number; + processed: number; + failed: number; + bytesBefore: number; + bytesAfter: number; +} + +/** + * One stored image: classify it, and rewrite it when it needs rewriting. + * + * Split out of `run` so that the loop reads as a loop. Every outcome is + * counted rather than thrown, because one unreadable file in a catalogue is + * not a reason to leave the rest of it exposed. + */ +async function handleRow(imagePath: string, totals: Totals): Promise { + // basename only: image_path is '/uploads/', and the directory it is + // served from is a server constant rather than part of the stored value. + const filePath = path.join(UPLOADS_DIR, path.basename(imagePath)); + const mimetype = TYPE_FOR_EXTENSION[path.extname(filePath).toLowerCase()]; + + if (!mimetype) { + console.warn(`[backfill] unrecognised extension, skipping: ${imagePath}`); + totals.unrecognised++; + return; + } + + let before: number; + try { + before = (await fs.stat(filePath)).size; + } catch { + // A row pointing at nothing is a pre-existing inconsistency. Reported + // rather than fatal: it is not this script's job to fix, and stopping + // would leave the rest of the catalogue exposed. + console.warn(`[backfill] file missing for ${imagePath}`); + totals.missing++; + return; + } + + try { + const meta = await sharp(filePath).metadata(); + + if (!needsProcessing(meta)) { + totals.skipped++; + return; + } + + if (!APPLY) { + console.info( + `[backfill] would process ${imagePath} ` + + `(${meta.width}x${meta.height}, exif ${meta.exif ? 'present' : 'absent'}, ${before} bytes)` + ); + totals.processed++; + totals.bytesBefore += before; + return; + } + + await reencodeInPlace(filePath, mimetype); + const after = (await fs.stat(filePath)).size; + // Counted only once the rewrite succeeded, so a file that threw is `failed` + // and nothing else. A row counted as both processed and failed would make + // the summary unreadable at exactly the moment it matters. + totals.processed++; + totals.bytesBefore += before; + totals.bytesAfter += after; + console.info(`[backfill] ${imagePath}: ${before} -> ${after} bytes`); + } catch (err) { + console.error(`[backfill] failed on ${imagePath}:`, err); + totals.failed++; + } +} + +async function run(): Promise { + const totals: Totals = { + seen: 0, + missing: 0, + skipped: 0, + unrecognised: 0, + processed: 0, + failed: 0, + bytesBefore: 0, + bytesAfter: 0 + }; + + const { rows } = await pool.query<{ image_path: string }>( + `SELECT image_path FROM item_images ORDER BY id` + ); + + console.info( + `[backfill] ${rows.length} image rows in ${UPLOADS_DIR}, ` + + `${APPLY ? 'APPLYING CHANGES' : 'reporting only (pass --apply to rewrite)'}` + ); + + for (const row of rows) { + totals.seen++; + await handleRow(row.image_path, totals); + } + + console.info('[backfill] done', totals); + + if (totals.failed > 0) { + // A non-zero exit so a partial run is visible to whatever invoked it, + // rather than reading as success because the summary printed. + process.exitCode = 1; + } +} + +run() + .catch((err) => { + console.error(err); + process.exitCode = 1; + }) + .finally(() => pool.end());