Both photo sweeps tracked whether they were running in a module-level variable. Correct on one replica, wrong behind a load balancer: the status poll answers from whichever process it reaches, so an idle replica reports isRunning false while another is mid-run, the UI re-enables the button, and the next POST lands elsewhere and starts a second pass over the whole library. The .whereNull() guards mean nothing is corrupted; the cost is duplicated S3/NAS I/O and an operator who cannot tell whether a job is running. Migration 189 adds one row per job. The claim is a conditional UPDATE whose affected-row count is the answer — the shape backgroundProcessor already uses to hand a photo to exactly one worker — so two replicas cannot both match. The lease is fenced on a per-claim token: taking over a stale claim does not stop the old runner, so without fencing a superseded runner finishing late cleared the new owner's flag and overwrote its result. heartbeat() reports renewal failure and the loops stop on it. Renewal runs on a timer spanning the claim through release, including the candidate query, because one hung NAS read can outlast the stale window inside a single iteration. maintenance_jobs is excluded from .picpeak archives — an archive taken mid-sweep would otherwise restore a live lease with no runner to release it. The importer filters the same set, so older archives are skipped too. Response shape is unchanged, so the frontend needs no change. Co-authored-by: Paul Nothaft <[email protected]>
This commit is contained in:
co-authored by
Paul Nothaft
parent
410b8f8f6f
commit
05e23ef1a1
@@ -0,0 +1,233 @@
|
||||
/**
|
||||
* Shared run state for the maintenance sweeps (#1181).
|
||||
*
|
||||
* The behaviour that matters here cannot be observed from one process holding
|
||||
* a module-level flag, which is exactly why the flag moved into the database.
|
||||
* A second replica is simulated the only way that is honest in a single-process
|
||||
* test: by asserting on the shared row itself, and by driving claim() twice —
|
||||
* a second caller getting null is precisely what a second replica gets.
|
||||
*/
|
||||
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const os = require('os');
|
||||
const express = require('express');
|
||||
const request = require('supertest');
|
||||
|
||||
describe('maintenance job state (#1181)', () => {
|
||||
let tmpDir; let db; let app; let jobs;
|
||||
|
||||
const dimStatus = () => request(app).get('/api/admin/photos/repair-dimensions/status');
|
||||
const capStatus = () => request(app).get('/api/admin/photos/repair-capture-dates/status');
|
||||
|
||||
beforeAll(async () => {
|
||||
tmpDir = await fs.promises.mkdtemp(path.join(os.tmpdir(), 'picpeak-mjs-'));
|
||||
process.env.NODE_ENV = 'test';
|
||||
process.env.TEST_DATABASE_PATH = path.join(tmpDir, 'data', 'db.sqlite');
|
||||
await fs.promises.mkdir(path.dirname(process.env.TEST_DATABASE_PATH), { recursive: true });
|
||||
process.env.STORAGE_PATH = path.join(tmpDir, 'storage');
|
||||
process.env.JWT_SECRET = process.env.JWT_SECRET || 'mjs-secret';
|
||||
|
||||
jest.resetModules();
|
||||
jest.doMock('../../src/middleware/auth', () => ({
|
||||
adminAuth: (req, _res, next) => { req.admin = { id: 1, username: 'tester', roleName: 'admin' }; next(); },
|
||||
}));
|
||||
jest.doMock('../../src/middleware/permissions', () => ({
|
||||
requirePermission: () => (_req, _res, next) => next(),
|
||||
}));
|
||||
jest.doMock('../../src/utils/logger', () => ({
|
||||
debug: jest.fn(), info: jest.fn(), warn: jest.fn(), error: jest.fn(),
|
||||
}));
|
||||
|
||||
({ db } = await require('./helpers/crmDb').bootCrmDb());
|
||||
jobs = require('../../src/services/maintenanceJobState');
|
||||
|
||||
app = express();
|
||||
app.use(express.json());
|
||||
app.use('/api/admin/photos', require('../../src/routes/adminPhotoDimensions'));
|
||||
}, 180000);
|
||||
|
||||
afterAll(async () => {
|
||||
if (db) await db.destroy?.();
|
||||
await fs.promises.rm(tmpDir, { recursive: true, force: true }).catch(() => {});
|
||||
});
|
||||
|
||||
beforeEach(async () => {
|
||||
await db('maintenance_jobs').update({
|
||||
is_running: false, started_at: null, heartbeat_at: null, finished_at: null, last_result: null, owner: null, claim_token: null,
|
||||
});
|
||||
});
|
||||
|
||||
test('the lease table is kept out of .picpeak archives', () => {
|
||||
// It is live state, not data. An archive taken mid-sweep would otherwise
|
||||
// carry is_running = true and a claim token owned by a process on the
|
||||
// SOURCE install; restored inside the staleness window, the target reports
|
||||
// the job as running and refuses new POSTs with no runner to release it.
|
||||
// The importer filters on this same set, so archives written before the
|
||||
// exclusion are skipped on restore too.
|
||||
const { EXCLUDED_TABLES } = require('../../src/services/picpeakExportService');
|
||||
expect(EXCLUDED_TABLES.has('maintenance_jobs')).toBe(true);
|
||||
});
|
||||
|
||||
test('the migration seeds a row for each job', async () => {
|
||||
const names = await db('maintenance_jobs').pluck('job_name');
|
||||
expect(names.sort()).toEqual(['photo_capture_date_backfill', 'photo_dimension_repair']);
|
||||
});
|
||||
|
||||
test('a second claim is refused while the first is alive', async () => {
|
||||
expect(await jobs.claim(jobs.JOB_DIMENSION_REPAIR)).toEqual(expect.any(String));
|
||||
// What a second replica's POST does. Nothing about the first claim lives in
|
||||
// this process, so this is the same question the other replica asks.
|
||||
expect(await jobs.claim(jobs.JOB_DIMENSION_REPAIR)).toBeNull();
|
||||
});
|
||||
|
||||
test('the two jobs claim independently', async () => {
|
||||
expect(await jobs.claim(jobs.JOB_DIMENSION_REPAIR)).toEqual(expect.any(String));
|
||||
expect(await jobs.claim(jobs.JOB_CAPTURE_DATE_BACKFILL)).toEqual(expect.any(String));
|
||||
});
|
||||
|
||||
test('each claim gets a distinct token', async () => {
|
||||
const first = await jobs.claim(jobs.JOB_DIMENSION_REPAIR);
|
||||
await jobs.release(jobs.JOB_DIMENSION_REPAIR, first);
|
||||
const second = await jobs.claim(jobs.JOB_DIMENSION_REPAIR);
|
||||
// Same process, same pid — so an owner string would have collided here and
|
||||
// the fencing below would be worthless.
|
||||
expect(second).not.toBe(first);
|
||||
});
|
||||
|
||||
test('a claim whose heartbeat has gone quiet can be taken over', async () => {
|
||||
expect(await jobs.claim(jobs.JOB_DIMENSION_REPAIR)).toEqual(expect.any(String));
|
||||
expect(await jobs.claim(jobs.JOB_DIMENSION_REPAIR)).toBeNull();
|
||||
|
||||
// The replica holding it was killed: no release, no further heartbeats.
|
||||
const longAgo = new Date(Date.now() - jobs.DEFAULT_STALE_MS - 60000).toISOString();
|
||||
await db('maintenance_jobs').where({ job_name: jobs.JOB_DIMENSION_REPAIR }).update({ heartbeat_at: longAgo });
|
||||
|
||||
expect(await jobs.claim(jobs.JOB_DIMENSION_REPAIR)).toEqual(expect.any(String));
|
||||
});
|
||||
|
||||
test('a superseded runner cannot renew its lease', async () => {
|
||||
const oldToken = await jobs.claim(jobs.JOB_DIMENSION_REPAIR);
|
||||
const longAgo = new Date(Date.now() - jobs.DEFAULT_STALE_MS - 60000).toISOString();
|
||||
await db('maintenance_jobs').where({ job_name: jobs.JOB_DIMENSION_REPAIR }).update({ heartbeat_at: longAgo });
|
||||
const newToken = await jobs.claim(jobs.JOB_DIMENSION_REPAIR);
|
||||
expect(newToken).toEqual(expect.any(String));
|
||||
|
||||
// The old runner is still alive and mid-loop. Its renewal must tell it so,
|
||||
// which is what makes the route loop stop instead of running alongside the
|
||||
// new owner.
|
||||
expect(await jobs.heartbeat(jobs.JOB_DIMENSION_REPAIR, oldToken)).toBe(false);
|
||||
expect(await jobs.heartbeat(jobs.JOB_DIMENSION_REPAIR, newToken)).toBe(true);
|
||||
});
|
||||
|
||||
test('a superseded runner cannot release the new owner\'s claim', async () => {
|
||||
const oldToken = await jobs.claim(jobs.JOB_CAPTURE_DATE_BACKFILL);
|
||||
const longAgo = new Date(Date.now() - jobs.DEFAULT_STALE_MS - 60000).toISOString();
|
||||
await db('maintenance_jobs').where({ job_name: jobs.JOB_CAPTURE_DATE_BACKFILL }).update({ heartbeat_at: longAgo });
|
||||
const newToken = await jobs.claim(jobs.JOB_CAPTURE_DATE_BACKFILL);
|
||||
|
||||
// The old runner finishes late and tries to write its result. Unfenced,
|
||||
// this cleared is_running under the new owner and let a THIRD sweep start.
|
||||
expect(await jobs.release(jobs.JOB_CAPTURE_DATE_BACKFILL, oldToken, { success: 999, noExif: 0, failed: 0 })).toBe(false);
|
||||
|
||||
const state = await jobs.read(jobs.JOB_CAPTURE_DATE_BACKFILL);
|
||||
expect(state.isRunning).toBe(true);
|
||||
expect(state.lastResult).toBeNull();
|
||||
// And the row is still the new owner's to release.
|
||||
expect(await jobs.release(jobs.JOB_CAPTURE_DATE_BACKFILL, newToken, { success: 1, noExif: 0, failed: 0 })).toBe(true);
|
||||
});
|
||||
|
||||
test('a stale run reads as not running, so the button comes back', async () => {
|
||||
await jobs.claim(jobs.JOB_DIMENSION_REPAIR);
|
||||
expect((await jobs.read(jobs.JOB_DIMENSION_REPAIR)).isRunning).toBe(true);
|
||||
|
||||
const longAgo = new Date(Date.now() - jobs.DEFAULT_STALE_MS - 60000).toISOString();
|
||||
await db('maintenance_jobs').where({ job_name: jobs.JOB_DIMENSION_REPAIR }).update({ heartbeat_at: longAgo });
|
||||
|
||||
// is_running is still true in the row — nothing released it — but a status
|
||||
// poll must not leave the operator staring at a job that cannot finish.
|
||||
expect((await db('maintenance_jobs').where({ job_name: jobs.JOB_DIMENSION_REPAIR }).first()).is_running).toBeTruthy();
|
||||
expect((await jobs.read(jobs.JOB_DIMENSION_REPAIR)).isRunning).toBe(false);
|
||||
});
|
||||
|
||||
test('a heartbeat keeps a long run claimed', async () => {
|
||||
const token = await jobs.claim(jobs.JOB_DIMENSION_REPAIR);
|
||||
const longAgo = new Date(Date.now() - jobs.DEFAULT_STALE_MS - 60000).toISOString();
|
||||
await db('maintenance_jobs').where({ job_name: jobs.JOB_DIMENSION_REPAIR }).update({ heartbeat_at: longAgo });
|
||||
|
||||
expect(await jobs.heartbeat(jobs.JOB_DIMENSION_REPAIR, token)).toBe(true);
|
||||
|
||||
expect(await jobs.claim(jobs.JOB_DIMENSION_REPAIR)).toBeNull();
|
||||
expect((await jobs.read(jobs.JOB_DIMENSION_REPAIR)).isRunning).toBe(true);
|
||||
});
|
||||
|
||||
test('release stores the result and read gives it back parsed', async () => {
|
||||
const token = await jobs.claim(jobs.JOB_CAPTURE_DATE_BACKFILL);
|
||||
await jobs.release(jobs.JOB_CAPTURE_DATE_BACKFILL, token, { success: 3, noExif: 2, failed: 1 });
|
||||
|
||||
const state = await jobs.read(jobs.JOB_CAPTURE_DATE_BACKFILL);
|
||||
expect(state.isRunning).toBe(false);
|
||||
expect(state.lastResult).toEqual({ success: 3, noExif: 2, failed: 1 });
|
||||
});
|
||||
|
||||
test('releasing without a result keeps the previous run visible', async () => {
|
||||
const first = await jobs.claim(jobs.JOB_CAPTURE_DATE_BACKFILL);
|
||||
await jobs.release(jobs.JOB_CAPTURE_DATE_BACKFILL, first, { success: 7, noExif: 0, failed: 0 });
|
||||
|
||||
// The "nothing to do" path: claimed, found no candidates, released. It must
|
||||
// not blank the numbers the last real run reported.
|
||||
const second = await jobs.claim(jobs.JOB_CAPTURE_DATE_BACKFILL);
|
||||
await jobs.release(jobs.JOB_CAPTURE_DATE_BACKFILL, second);
|
||||
|
||||
expect((await jobs.read(jobs.JOB_CAPTURE_DATE_BACKFILL)).lastResult).toEqual({ success: 7, noExif: 0, failed: 0 });
|
||||
});
|
||||
|
||||
test('a malformed result does not take the status endpoint down', async () => {
|
||||
await db('maintenance_jobs').where({ job_name: jobs.JOB_DIMENSION_REPAIR }).update({ last_result: 'not json' });
|
||||
const state = await jobs.read(jobs.JOB_DIMENSION_REPAIR);
|
||||
expect(state.lastResult).toBeNull();
|
||||
expect(state.isRunning).toBe(false);
|
||||
});
|
||||
|
||||
test('both status endpoints report the shared row, not process memory', async () => {
|
||||
await jobs.claim(jobs.JOB_DIMENSION_REPAIR);
|
||||
const capToken = await jobs.claim(jobs.JOB_CAPTURE_DATE_BACKFILL);
|
||||
await jobs.release(jobs.JOB_CAPTURE_DATE_BACKFILL, capToken, { success: 1, noExif: 0, failed: 0 });
|
||||
|
||||
// Written straight to the row, exactly as another replica would have.
|
||||
const dim = await dimStatus();
|
||||
expect(dim.status).toBe(200);
|
||||
expect(dim.body.isRunning).toBe(true);
|
||||
|
||||
const cap = await capStatus();
|
||||
expect(cap.status).toBe(200);
|
||||
expect(cap.body.isRunning).toBe(false);
|
||||
expect(cap.body.lastResult).toEqual({ success: 1, noExif: 0, failed: 0 });
|
||||
});
|
||||
|
||||
test('a POST is refused while another replica holds the claim', async () => {
|
||||
// The claim was taken by "another replica" — this process knows nothing
|
||||
// about it beyond the row.
|
||||
await jobs.claim(jobs.JOB_CAPTURE_DATE_BACKFILL);
|
||||
|
||||
const res = await request(app).post('/api/admin/photos/repair-capture-dates');
|
||||
expect(res.status).toBe(409);
|
||||
|
||||
const dimRes = await request(app).post('/api/admin/photos/repair-dimensions');
|
||||
// The other job is untouched by that claim, so it is free to start.
|
||||
expect(dimRes.status).toBe(200);
|
||||
});
|
||||
|
||||
test('the no-op path releases the claim it took', async () => {
|
||||
// No photos at all, so both endpoints take their "nothing to do" exit.
|
||||
await db('photos').del();
|
||||
|
||||
const res = await request(app).post('/api/admin/photos/repair-capture-dates');
|
||||
expect(res.body.count).toBe(0);
|
||||
|
||||
const row = await db('maintenance_jobs').where({ job_name: jobs.JOB_CAPTURE_DATE_BACKFILL }).first();
|
||||
expect(row.is_running).toBeFalsy();
|
||||
// ...and a second POST is therefore accepted rather than 409ing forever.
|
||||
expect((await request(app).post('/api/admin/photos/repair-capture-dates')).status).toBe(200);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,102 @@
|
||||
/**
|
||||
* PostgreSQL checks for the shared maintenance-job state (#1181).
|
||||
* Gated: runs only when PICPEAK_PG_TEST_URL points at a throwaway database, e.g.
|
||||
* PICPEAK_PG_TEST_URL="postgres://picpeak:[email protected]:7102/picpeak_mjs_test" \
|
||||
* npx jest __tests__/integration/maintenanceJobStatePg.test.js
|
||||
*
|
||||
* What SQLite cannot answer: the claim leans on comparing a `timestamp` column
|
||||
* against an ISO-8601 string, and on an UPDATE ... WHERE guard being atomic
|
||||
* under real concurrent connections. SQLite compares those strings
|
||||
* lexicographically and serialises writes anyway, so it would pass either way —
|
||||
* exactly the shape of divergence that has bitten this repo before.
|
||||
*/
|
||||
|
||||
const knex = require('knex');
|
||||
|
||||
const PG_URL = process.env.PICPEAK_PG_TEST_URL;
|
||||
const maybe = PG_URL ? describe : describe.skip;
|
||||
|
||||
maybe('maintenance job state on Postgres', () => {
|
||||
let pgDb;
|
||||
let jobs;
|
||||
const JOB = 'photo_dimension_repair';
|
||||
|
||||
beforeAll(async () => {
|
||||
pgDb = knex({ client: 'pg', connection: PG_URL, pool: { min: 0, max: 10 } });
|
||||
await pgDb.raw('DROP TABLE IF EXISTS maintenance_jobs');
|
||||
await require('../../migrations/core/189_maintenance_job_state').up(pgDb);
|
||||
|
||||
jest.resetModules();
|
||||
jest.doMock('../../src/database/db', () => ({ db: pgDb }));
|
||||
jest.doMock('../../src/utils/logger', () => ({
|
||||
debug: jest.fn(), info: jest.fn(), warn: jest.fn(), error: jest.fn(),
|
||||
}));
|
||||
jobs = require('../../src/services/maintenanceJobState');
|
||||
}, 60000);
|
||||
|
||||
afterAll(async () => {
|
||||
jest.dontMock('../../src/database/db');
|
||||
if (pgDb) await pgDb.destroy();
|
||||
});
|
||||
|
||||
beforeEach(async () => {
|
||||
await pgDb('maintenance_jobs').update({
|
||||
is_running: false, started_at: null, heartbeat_at: null, finished_at: null, last_result: null, owner: null, claim_token: null,
|
||||
});
|
||||
});
|
||||
|
||||
test('the ISO-string cutoff really compares as a timestamp, not as text', async () => {
|
||||
expect(await jobs.claim(JOB)).toEqual(expect.any(String));
|
||||
expect(await jobs.claim(JOB)).toBeNull();
|
||||
|
||||
await pgDb('maintenance_jobs').where({ job_name: JOB })
|
||||
.update({ heartbeat_at: new Date(Date.now() - jobs.DEFAULT_STALE_MS - 60000).toISOString() });
|
||||
|
||||
// If Postgres had rejected or mis-cast the ISO string this would either
|
||||
// throw or never match.
|
||||
expect(await jobs.claim(JOB)).toEqual(expect.any(String));
|
||||
|
||||
const row = await pgDb('maintenance_jobs').where({ job_name: JOB }).first();
|
||||
expect(row.heartbeat_at).toBeInstanceOf(Date);
|
||||
});
|
||||
|
||||
test('concurrent claims on real connections produce exactly one winner', async () => {
|
||||
// The whole point of the conditional UPDATE. Ten connections race; nine
|
||||
// must lose. SQLite cannot demonstrate this — it serialises writers.
|
||||
const results = await Promise.all(Array.from({ length: 10 }, () => jobs.claim(JOB)));
|
||||
expect(results.filter(Boolean)).toHaveLength(1);
|
||||
// ...and the winner holds a token nobody else can forge.
|
||||
expect(results.find(Boolean)).toEqual(expect.any(String));
|
||||
});
|
||||
|
||||
test('a released job can be re-claimed exactly once again', async () => {
|
||||
const token = await jobs.claim(JOB);
|
||||
await jobs.release(JOB, token, { success: 2, failed: 0 });
|
||||
|
||||
const results = await Promise.all(Array.from({ length: 5 }, () => jobs.claim(JOB)));
|
||||
expect(results.filter(Boolean)).toHaveLength(1);
|
||||
expect((await jobs.read(JOB)).lastResult).toEqual({ success: 2, failed: 0 });
|
||||
});
|
||||
|
||||
test('a superseded runner is fenced out on real Postgres', async () => {
|
||||
const oldToken = await jobs.claim(JOB);
|
||||
await pgDb('maintenance_jobs').where({ job_name: JOB })
|
||||
.update({ heartbeat_at: new Date(Date.now() - jobs.DEFAULT_STALE_MS - 60000).toISOString() });
|
||||
const newToken = await jobs.claim(JOB);
|
||||
|
||||
expect(await jobs.heartbeat(JOB, oldToken)).toBe(false);
|
||||
expect(await jobs.release(JOB, oldToken, { success: 999, failed: 0 })).toBe(false);
|
||||
// The new owner still holds it, with its result unwritten.
|
||||
expect((await jobs.read(JOB)).isRunning).toBe(true);
|
||||
expect(await jobs.release(JOB, newToken, { success: 4, failed: 0 })).toBe(true);
|
||||
});
|
||||
|
||||
test('read() reports a live claim as running and a stale one as not', async () => {
|
||||
await jobs.claim(JOB);
|
||||
expect((await jobs.read(JOB)).isRunning).toBe(true);
|
||||
|
||||
await pgDb('maintenance_jobs').where({ job_name: JOB })
|
||||
.update({ heartbeat_at: new Date(Date.now() - jobs.DEFAULT_STALE_MS - 1000).toISOString() });
|
||||
expect((await jobs.read(JOB)).isRunning).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,73 @@
|
||||
/**
|
||||
* Migration 189: shared run state for the maintenance sweeps (#1181).
|
||||
*
|
||||
* Both photo maintenance jobs — the dimension repair and the capture-date
|
||||
* backfill — tracked whether they were running in a module-level variable. On
|
||||
* a single-replica install that is correct. Behind a load balancer it is not:
|
||||
* the flag lives in one process, so a status poll routed to any other replica
|
||||
* answers `isRunning: false`, the UI re-enables the button, and the next POST
|
||||
* lands somewhere else and starts a second pass over the whole library. Both
|
||||
* replicas then read and parse every original off S3 or the NAS mount. The
|
||||
* `.whereNull(...)` guards on the writes mean nothing is corrupted — the cost
|
||||
* is the duplicated I/O, and an operator who cannot tell whether a job is
|
||||
* running.
|
||||
*
|
||||
* One row per job, claimed with a conditional UPDATE so the claim itself is
|
||||
* the mutual exclusion — the same UPDATE-with-guard shape backgroundProcessor
|
||||
* already uses to hand a photo to exactly one worker
|
||||
* (services/backgroundProcessor.js:110-116).
|
||||
*
|
||||
* heartbeat_at exists because a lock with no expiry is worse than no lock: a
|
||||
* replica that is OOM-killed mid-run would leave is_running = true forever and
|
||||
* no way to clear it short of editing the database. The runner touches it as
|
||||
* it goes, and a claim is allowed to take over a run whose heartbeat has gone
|
||||
* quiet. See services/maintenanceJobState.js for the read side, which reports
|
||||
* a stale run as not-running so the button comes back on its own.
|
||||
*
|
||||
* Rows are seeded here rather than created on demand so the claim is a plain
|
||||
* UPDATE with no insert race behind it.
|
||||
*/
|
||||
|
||||
const JOBS = ['photo_dimension_repair', 'photo_capture_date_backfill'];
|
||||
|
||||
exports.up = async function (knex) {
|
||||
const exists = await knex.schema.hasTable('maintenance_jobs');
|
||||
if (!exists) {
|
||||
await knex.schema.createTable('maintenance_jobs', (t) => {
|
||||
// The job's identity, not a surrogate key: there is exactly one row per
|
||||
// job and every access is by name, so the name is the primary key.
|
||||
t.string('job_name', 64).primary();
|
||||
t.boolean('is_running').notNullable().defaultTo(false);
|
||||
t.timestamp('started_at').nullable();
|
||||
t.timestamp('heartbeat_at').nullable();
|
||||
t.timestamp('finished_at').nullable();
|
||||
// JSON as text: the shape differs per job (the backfill reports a third
|
||||
// counter the dimension repair has no equivalent for) and nothing
|
||||
// queries into it, so a json column would buy nothing and cost engine
|
||||
// differences between Postgres and SQLite.
|
||||
t.text('last_result').nullable();
|
||||
// Diagnostics only — which process is holding the claim.
|
||||
t.string('owner', 128).nullable();
|
||||
// The fencing token. Unique per claim, not per process: after a stale
|
||||
// takeover the old runner may still be alive and mid-loop, and it can
|
||||
// even be the same process that re-claimed. Every write it makes is
|
||||
// scoped to the token it was handed, so a superseded runner can neither
|
||||
// renew a claim it has lost nor release one it no longer owns.
|
||||
t.string('claim_token', 64).nullable();
|
||||
});
|
||||
console.log('189: created maintenance_jobs');
|
||||
}
|
||||
|
||||
// Idempotent on re-run and safe against a table that already carries rows.
|
||||
for (const jobName of JOBS) {
|
||||
const row = await knex('maintenance_jobs').where({ job_name: jobName }).first();
|
||||
if (!row) {
|
||||
await knex('maintenance_jobs').insert({ job_name: jobName, is_running: false });
|
||||
console.log(`189: seeded job row ${jobName}`);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
exports.down = async function (knex) {
|
||||
await knex.schema.dropTableIfExists('maintenance_jobs');
|
||||
};
|
||||
@@ -7,29 +7,83 @@ const fs = require('fs').promises;
|
||||
const logger = require('../utils/logger');
|
||||
|
||||
const { resolvePhotoFilePath } = require('../services/photoResolver');
|
||||
const maintenanceJobs = require('../services/maintenanceJobState');
|
||||
|
||||
// Module-level progress state
|
||||
let repairProgress = {
|
||||
isRunning: false,
|
||||
lastResult: null
|
||||
};
|
||||
// Run state for both sweeps lives in the database, not in this process
|
||||
// (#1181). It used to be a module-level object per job, which is correct on a
|
||||
// single replica and wrong behind a load balancer: the status poll answers
|
||||
// from whichever process it reaches, so an idle replica reports isRunning
|
||||
// false while another is mid-run, the UI re-enables the button, and the next
|
||||
// POST starts a duplicate pass over the entire library.
|
||||
//
|
||||
// Two separate rows, for the same reason the two objects were separate: the
|
||||
// jobs walk the same photos but read different things out of them, and one
|
||||
// running must not block or report for the other.
|
||||
const { JOB_DIMENSION_REPAIR, JOB_CAPTURE_DATE_BACKFILL } = maintenanceJobs;
|
||||
|
||||
// Same shape, separate state: the two jobs walk the same photos but read
|
||||
// different things out of them, and one running must not block or report for
|
||||
// the other.
|
||||
let captureDateProgress = {
|
||||
isRunning: false,
|
||||
lastResult: null
|
||||
const { HEARTBEAT_INTERVAL_MS } = maintenanceJobs;
|
||||
|
||||
/**
|
||||
* Renew the lease on a timer for as long as the run holds it.
|
||||
*
|
||||
* On a timer, not between photos: a single hung read on a stalled NAS mount or
|
||||
* a slow S3 object can outlast the whole stale window inside one iteration, and
|
||||
* a renewal that only fires between photos never gets to run. The lease would
|
||||
* expire while the job was demonstrably alive, another replica would take it
|
||||
* over, and the two would walk the same rows — precisely the case the lease
|
||||
* exists to prevent. The timer also covers the candidate query, which on a
|
||||
* large library is itself slow.
|
||||
*
|
||||
* `lost()` reports whether the claim has since been taken over. The loops check
|
||||
* it between photos and stop: mid-photo interruption is not possible, so the
|
||||
* worst case is one extra row written by the old runner, and its release is
|
||||
* fenced on the token anyway.
|
||||
*/
|
||||
function startLeaseKeeper(jobName, token) {
|
||||
let lost = false;
|
||||
const timer = setInterval(async () => {
|
||||
try {
|
||||
if (!(await maintenanceJobs.heartbeat(jobName, token))) {
|
||||
lost = true;
|
||||
clearInterval(timer);
|
||||
}
|
||||
} catch (err) {
|
||||
// heartbeat() already swallows query errors and reports the claim as
|
||||
// held; this is belt-and-braces so an unexpected throw cannot kill the
|
||||
// timer callback and silently stop all renewals.
|
||||
logger.warn(`Lease renewal error for ${jobName}: ${err.message}`);
|
||||
}
|
||||
}, HEARTBEAT_INTERVAL_MS);
|
||||
// Do not hold the event loop open on account of a maintenance sweep.
|
||||
if (typeof timer.unref === 'function') timer.unref();
|
||||
|
||||
return {
|
||||
lost: () => lost,
|
||||
stop: () => clearInterval(timer),
|
||||
};
|
||||
}
|
||||
|
||||
// Repair photo dimensions (background job)
|
||||
router.post('/repair-dimensions', adminAuth, requirePermission('photos.edit'), async (req, res) => {
|
||||
try {
|
||||
if (repairProgress.isRunning) {
|
||||
// Claimed before the candidate query, not after: that query is an await,
|
||||
// and two requests arriving inside it would both read "not running" and
|
||||
// both start a pass. The claim is a conditional UPDATE, so it settles the
|
||||
// race across replicas as well as within one.
|
||||
const token = await maintenanceJobs.claim(JOB_DIMENSION_REPAIR);
|
||||
if (!token) {
|
||||
return res.status(409).json({ error: 'Repair is already running' });
|
||||
}
|
||||
// Started at the claim, not at the loop: on a large or loaded install the
|
||||
// candidate SELECT below (plus the setImmediate hop) can itself outlast the
|
||||
// stale window, and an unrenewed claim would be taken over before the sweep
|
||||
// had read its first photo. Handed to the background section, which stops
|
||||
// it; every early exit here stops it too.
|
||||
const lease = startLeaseKeeper(JOB_DIMENSION_REPAIR, token);
|
||||
|
||||
const photos = await db('photos')
|
||||
let photos;
|
||||
try {
|
||||
photos = await db('photos')
|
||||
.join('events', 'photos.event_id', 'events.id')
|
||||
.where(function () {
|
||||
this.whereNull('photos.width').orWhereNull('photos.height');
|
||||
@@ -42,8 +96,17 @@ router.post('/repair-dimensions', adminAuth, requirePermission('photos.edit'), a
|
||||
'photos.source_origin', 'photos.external_relpath', 'photos.event_id',
|
||||
'events.source_mode', 'events.external_path', 'events.slug'
|
||||
);
|
||||
} catch (err) {
|
||||
lease.stop();
|
||||
await maintenanceJobs.release(JOB_DIMENSION_REPAIR, token);
|
||||
throw err;
|
||||
}
|
||||
|
||||
if (photos.length === 0) {
|
||||
// Released with no result: nothing ran, so the numbers from the last
|
||||
// real run stay on screen rather than being blanked by a no-op.
|
||||
lease.stop();
|
||||
await maintenanceJobs.release(JOB_DIMENSION_REPAIR, token);
|
||||
return res.json({ message: 'No photos need dimension repair', count: 0 });
|
||||
}
|
||||
|
||||
@@ -54,24 +117,32 @@ router.post('/repair-dimensions', adminAuth, requirePermission('photos.edit'), a
|
||||
});
|
||||
|
||||
// Process in background
|
||||
repairProgress.isRunning = true;
|
||||
repairProgress.lastResult = null;
|
||||
|
||||
setImmediate(async () => {
|
||||
let sharp;
|
||||
try {
|
||||
sharp = require('sharp');
|
||||
} catch (err) {
|
||||
logger.error('Sharp not available for dimension repair:', err.message);
|
||||
repairProgress.isRunning = false;
|
||||
repairProgress.lastResult = { success: 0, failed: 0, error: 'Sharp not available' };
|
||||
lease.stop();
|
||||
await maintenanceJobs.release(JOB_DIMENSION_REPAIR, token, { success: 0, failed: 0, error: 'Sharp not available' });
|
||||
return;
|
||||
}
|
||||
|
||||
let successCount = 0;
|
||||
let errorCount = 0;
|
||||
let lostClaim = false;
|
||||
|
||||
// Everything below runs detached from the request, so an unexpected
|
||||
// throw has nobody to report to. Without this the claim would sit held
|
||||
// until it aged out of the staleness window, disabling the button for
|
||||
// that whole time on every replica.
|
||||
try {
|
||||
for (const photo of photos) {
|
||||
// The timer does the renewing; this only notices that it has
|
||||
// already failed, so the loop stops instead of running on beside the
|
||||
// replica that took the claim over.
|
||||
if (lease.lost()) { lostClaim = true; break; }
|
||||
|
||||
try {
|
||||
const event = { source_mode: photo.source_mode, external_path: photo.external_path, slug: photo.slug };
|
||||
let fullPath;
|
||||
@@ -115,9 +186,23 @@ router.post('/repair-dimensions', adminAuth, requirePermission('photos.edit'), a
|
||||
}
|
||||
}
|
||||
|
||||
repairProgress.isRunning = false;
|
||||
repairProgress.lastResult = { success: successCount, failed: errorCount };
|
||||
if (lostClaim) {
|
||||
// Another replica declared this run stale and took it over. It owns
|
||||
// the row now, so releasing would clear ITS flag — release() refuses
|
||||
// on the token, but there is nothing to report either way.
|
||||
logger.warn(`Dimension repair stopped: claim taken over after ${successCount} updated, ${errorCount} errors`);
|
||||
return;
|
||||
}
|
||||
await maintenanceJobs.release(JOB_DIMENSION_REPAIR, token, { success: successCount, failed: errorCount });
|
||||
logger.info(`Dimension repair complete: ${successCount} success, ${errorCount} errors`);
|
||||
} catch (err) {
|
||||
logger.error('Dimension repair aborted:', err);
|
||||
await maintenanceJobs
|
||||
.release(JOB_DIMENSION_REPAIR, token, { success: successCount, failed: errorCount, error: err.message })
|
||||
.catch(() => {});
|
||||
} finally {
|
||||
lease.stop();
|
||||
}
|
||||
});
|
||||
} catch (error) {
|
||||
logger.error('Error starting dimension repair:', error);
|
||||
@@ -146,13 +231,15 @@ router.get('/repair-dimensions/status', adminAuth, requirePermission('photos.vie
|
||||
|
||||
const total = Number(totalPhotos.count);
|
||||
const withDims = Number(withDimensions.count);
|
||||
// Read from the shared row, so this answers the same on every replica.
|
||||
const state = await maintenanceJobs.read(JOB_DIMENSION_REPAIR);
|
||||
|
||||
res.json({
|
||||
total,
|
||||
withDimensions: withDims,
|
||||
withoutDimensions: total - withDims,
|
||||
isRunning: repairProgress.isRunning,
|
||||
lastResult: repairProgress.lastResult
|
||||
isRunning: state.isRunning,
|
||||
lastResult: state.lastResult
|
||||
});
|
||||
} catch (error) {
|
||||
logger.error('Error fetching dimension repair status:', error);
|
||||
@@ -185,15 +272,18 @@ router.get('/repair-dimensions/status', adminAuth, requirePermission('photos.vie
|
||||
// exposure and predates this; it is brought in line separately in #1182.
|
||||
router.post('/repair-capture-dates', adminAuth, requirePermission('system.manage'), async (req, res) => {
|
||||
try {
|
||||
if (captureDateProgress.isRunning) {
|
||||
// Claimed here, not after the candidate query: that query is an await, and
|
||||
// two POSTs arriving inside it would both read "not running" and both start
|
||||
// a pass over the same rows. Being a conditional UPDATE, the claim settles
|
||||
// that between replicas too. Every early exit below has to release it
|
||||
// again, hence the try/catch around the query.
|
||||
const token = await maintenanceJobs.claim(JOB_CAPTURE_DATE_BACKFILL);
|
||||
if (!token) {
|
||||
return res.status(409).json({ error: 'Capture date backfill is already running' });
|
||||
}
|
||||
// Claimed here, not after the candidate query: that query is an await, and
|
||||
// two POSTs arriving inside it would both read isRunning === false and both
|
||||
// start a pass over the same rows. Every early exit below has to release it
|
||||
// again, hence the try/catch around the query.
|
||||
captureDateProgress.isRunning = true;
|
||||
captureDateProgress.lastResult = null;
|
||||
// Started at the claim so the candidate SELECT below is covered too — see
|
||||
// the dimension repair above.
|
||||
const lease = startLeaseKeeper(JOB_CAPTURE_DATE_BACKFILL, token);
|
||||
|
||||
let photos;
|
||||
try {
|
||||
@@ -228,12 +318,15 @@ router.post('/repair-capture-dates', adminAuth, requirePermission('system.manage
|
||||
'events.source_mode', 'events.external_path', 'events.slug'
|
||||
);
|
||||
} catch (err) {
|
||||
captureDateProgress.isRunning = false;
|
||||
lease.stop();
|
||||
await maintenanceJobs.release(JOB_CAPTURE_DATE_BACKFILL, token);
|
||||
throw err;
|
||||
}
|
||||
|
||||
if (photos.length === 0) {
|
||||
captureDateProgress.isRunning = false;
|
||||
// No result passed: nothing ran, so the last real run's numbers survive.
|
||||
lease.stop();
|
||||
await maintenanceJobs.release(JOB_CAPTURE_DATE_BACKFILL, token);
|
||||
return res.json({ message: 'No photos need a capture date', count: 0 });
|
||||
}
|
||||
|
||||
@@ -248,8 +341,15 @@ router.post('/repair-capture-dates', adminAuth, requirePermission('system.manage
|
||||
let successCount = 0;
|
||||
let missingCount = 0;
|
||||
let errorCount = 0;
|
||||
let lostClaim = false;
|
||||
|
||||
// Same reasoning as the dimension repair: detached from the request, so
|
||||
// an unexpected throw must not leave the claim held.
|
||||
try {
|
||||
for (const photo of photos) {
|
||||
// The timer renews; this only notices it has already failed.
|
||||
if (lease.lost()) { lostClaim = true; break; }
|
||||
|
||||
try {
|
||||
const event = { source_mode: photo.source_mode, external_path: photo.external_path, slug: photo.slug };
|
||||
const isExternal = photo.source_origin === 'external' || photo.source_origin === 'reference';
|
||||
@@ -330,9 +430,20 @@ router.post('/repair-capture-dates', adminAuth, requirePermission('system.manage
|
||||
}
|
||||
}
|
||||
|
||||
captureDateProgress.isRunning = false;
|
||||
captureDateProgress.lastResult = { success: successCount, noExif: missingCount, failed: errorCount };
|
||||
if (lostClaim) {
|
||||
logger.warn(`Capture date backfill stopped: claim taken over after ${successCount} updated, ${errorCount} errors`);
|
||||
return;
|
||||
}
|
||||
await maintenanceJobs.release(JOB_CAPTURE_DATE_BACKFILL, token, { success: successCount, noExif: missingCount, failed: errorCount });
|
||||
logger.info(`Capture date backfill complete: ${successCount} updated, ${missingCount} without EXIF, ${errorCount} errors`);
|
||||
} catch (err) {
|
||||
logger.error('Capture date backfill aborted:', err);
|
||||
await maintenanceJobs
|
||||
.release(JOB_CAPTURE_DATE_BACKFILL, token, { success: successCount, noExif: missingCount, failed: errorCount, error: err.message })
|
||||
.catch(() => {});
|
||||
} finally {
|
||||
lease.stop();
|
||||
}
|
||||
});
|
||||
} catch (error) {
|
||||
logger.error('Error starting capture date backfill:', error);
|
||||
@@ -377,13 +488,15 @@ router.get('/repair-capture-dates/status', adminAuth, requirePermission('system.
|
||||
|
||||
const total = Number(counts.total);
|
||||
const withCaptureDate = Number(counts.dated);
|
||||
// Read from the shared row, so this answers the same on every replica.
|
||||
const state = await maintenanceJobs.read(JOB_CAPTURE_DATE_BACKFILL);
|
||||
|
||||
res.json({
|
||||
total,
|
||||
withCaptureDate,
|
||||
withoutCaptureDate: total - withCaptureDate,
|
||||
isRunning: captureDateProgress.isRunning,
|
||||
lastResult: captureDateProgress.lastResult
|
||||
isRunning: state.isRunning,
|
||||
lastResult: state.lastResult
|
||||
});
|
||||
} catch (error) {
|
||||
logger.error('Error fetching capture date backfill status:', error);
|
||||
|
||||
@@ -0,0 +1,178 @@
|
||||
/**
|
||||
* Shared run state for the photo maintenance sweeps (#1181).
|
||||
*
|
||||
* These jobs used to keep `{ isRunning, lastResult }` in a module-level
|
||||
* variable. That is invisible to every other replica, so on a multi-replica
|
||||
* install the status endpoint answers from whichever process the poll happens
|
||||
* to reach and a second POST can start a duplicate pass over the whole
|
||||
* library. Moving the state into the database makes both the claim and the
|
||||
* reporting shared.
|
||||
*
|
||||
* The claim is a conditional UPDATE whose affected-row count is the answer —
|
||||
* the same shape backgroundProcessor uses to hand a photo to exactly one
|
||||
* worker (backgroundProcessor.js:110-116). Two replicas issuing it
|
||||
* concurrently cannot both match: the row is locked for the duration of each
|
||||
* UPDATE, so the loser sees is_running already true and gets 0 rows back.
|
||||
*
|
||||
* A lease that can be taken over needs fencing, which is what claim_token is
|
||||
* for. Taking over a stale claim does not stop the old runner — it is a
|
||||
* process nobody can signal, quite possibly still walking the library. So
|
||||
* every write it attempts carries the token it was issued:
|
||||
*
|
||||
* - heartbeat() reports whether the renewal landed. It returns false once
|
||||
* the claim has moved on, and the run loops treat that as "stop".
|
||||
* - release() only clears the row if the token still matches, so a
|
||||
* superseded runner finishing late cannot clear the new owner's flag or
|
||||
* overwrite its result.
|
||||
*
|
||||
* Without both of those, a takeover produces two live runners and the loser
|
||||
* ends up stomping the winner's state on its way out.
|
||||
*
|
||||
* Timestamps are written as ISO strings rather than Date objects. Production
|
||||
* stores Dates fine, but inside jest the sqlite3 binding turns them into the
|
||||
* literal string "[object Object]" (see CLAUDE.md), which would silently break
|
||||
* every staleness comparison in the tests. ISO-8601 also compares correctly
|
||||
* under SQLite's lexicographic text ordering, so the `<` below means the same
|
||||
* thing on both engines.
|
||||
*/
|
||||
|
||||
const os = require('os');
|
||||
const crypto = require('crypto');
|
||||
const { db } = require('../database/db');
|
||||
const logger = require('../utils/logger');
|
||||
|
||||
const JOB_DIMENSION_REPAIR = 'photo_dimension_repair';
|
||||
const JOB_CAPTURE_DATE_BACKFILL = 'photo_capture_date_backfill';
|
||||
|
||||
// How long a run may go without renewing its lease before another replica is
|
||||
// allowed to take it over. Generous on purpose: these jobs walk the whole
|
||||
// library and a single slow original on a stalled NAS mount can block the loop
|
||||
// for a while. The cost of being too eager is a duplicate pass; the cost of
|
||||
// being too patient is a button that stays disabled after a crash.
|
||||
const DEFAULT_STALE_MS = 15 * 60 * 1000;
|
||||
|
||||
// How often a running job renews. Time-based, and comfortably inside the stale
|
||||
// window: tying renewal to a photo counter meant a job whose photos were slow
|
||||
// — a stalled mount, a handful of very large originals — could be declared
|
||||
// abandoned while it was still working.
|
||||
const HEARTBEAT_INTERVAL_MS = 60 * 1000;
|
||||
|
||||
const OWNER = `${os.hostname()}:${process.pid}`;
|
||||
|
||||
const nowIso = () => new Date().toISOString();
|
||||
const cutoffIso = (staleAfterMs) => new Date(Date.now() - staleAfterMs).toISOString();
|
||||
|
||||
/**
|
||||
* Try to become the one runner of `jobName`.
|
||||
*
|
||||
* Returns a claim token on success, or null when another replica holds it and
|
||||
* is still renewing — the caller should answer 409. The token must be passed
|
||||
* to every subsequent heartbeat/release for this run.
|
||||
*/
|
||||
async function claim(jobName, { staleAfterMs = DEFAULT_STALE_MS } = {}) {
|
||||
const stamp = nowIso();
|
||||
const cutoff = cutoffIso(staleAfterMs);
|
||||
const token = crypto.randomBytes(16).toString('hex');
|
||||
|
||||
const claimed = await db('maintenance_jobs')
|
||||
.where({ job_name: jobName })
|
||||
.where(function () {
|
||||
// Free, or held by a run that has stopped renewing. heartbeat_at is
|
||||
// always written by the claim below, so a running job cannot have a null
|
||||
// heartbeat — no third case to handle here.
|
||||
this.where('is_running', false).orWhere('heartbeat_at', '<', cutoff);
|
||||
})
|
||||
.update({
|
||||
is_running: true,
|
||||
started_at: stamp,
|
||||
heartbeat_at: stamp,
|
||||
finished_at: null,
|
||||
owner: OWNER,
|
||||
claim_token: token,
|
||||
});
|
||||
|
||||
return claimed > 0 ? token : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Renew the lease.
|
||||
*
|
||||
* Returns false when this run no longer owns the claim — it was declared stale
|
||||
* and taken over. The caller must stop working at that point: the new owner is
|
||||
* already walking the same rows, and two runners writing is exactly what the
|
||||
* lock exists to prevent.
|
||||
*/
|
||||
async function heartbeat(jobName, token) {
|
||||
try {
|
||||
const renewed = await db('maintenance_jobs')
|
||||
.where({ job_name: jobName, is_running: true, claim_token: token })
|
||||
.update({ heartbeat_at: nowIso() });
|
||||
return renewed > 0;
|
||||
} catch (err) {
|
||||
// A failed renewal query is not proof the claim is gone, and aborting a
|
||||
// long sweep over one transient database blip is the worse trade. Say the
|
||||
// claim still holds; if it really has moved on, the next renewal says so.
|
||||
logger.warn(`maintenanceJobState: heartbeat failed for ${jobName}: ${err.message}`);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Give up the claim.
|
||||
*
|
||||
* Scoped to the token, so a runner that was superseded while it was working
|
||||
* cannot clear the new owner's flag or overwrite its result on the way out.
|
||||
* Returns false when the claim had already moved on.
|
||||
*
|
||||
* `result` is stored as the job's last outcome. Pass null (the "nothing to do"
|
||||
* and error paths) to release without overwriting what the previous real run
|
||||
* reported.
|
||||
*/
|
||||
async function release(jobName, token, result = null) {
|
||||
const update = { is_running: false, finished_at: nowIso() };
|
||||
if (result !== null && result !== undefined) {
|
||||
update.last_result = JSON.stringify(result);
|
||||
}
|
||||
const released = await db('maintenance_jobs')
|
||||
.where({ job_name: jobName, claim_token: token })
|
||||
.update(update);
|
||||
return released > 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Current state, in the shape the status endpoints hand to the frontend.
|
||||
*
|
||||
* A run whose lease has gone stale is reported as not running: the owning
|
||||
* replica is gone, nothing is going to release the claim, and the operator
|
||||
* needs the button back. The next claim() takes the row over on the same
|
||||
* condition, so the two agree.
|
||||
*/
|
||||
async function read(jobName, { staleAfterMs = DEFAULT_STALE_MS } = {}) {
|
||||
const row = await db('maintenance_jobs').where({ job_name: jobName }).first();
|
||||
if (!row) return { isRunning: false, lastResult: null };
|
||||
|
||||
const alive = row.heartbeat_at && new Date(row.heartbeat_at).getTime() > Date.now() - staleAfterMs;
|
||||
|
||||
let lastResult = null;
|
||||
if (row.last_result) {
|
||||
try {
|
||||
lastResult = JSON.parse(row.last_result);
|
||||
} catch (err) {
|
||||
// Never let a malformed row take the status endpoint down with it.
|
||||
logger.warn(`maintenanceJobState: unreadable last_result for ${jobName}: ${err.message}`);
|
||||
}
|
||||
}
|
||||
|
||||
return { isRunning: Boolean(row.is_running) && Boolean(alive), lastResult };
|
||||
}
|
||||
|
||||
module.exports = {
|
||||
claim,
|
||||
heartbeat,
|
||||
release,
|
||||
read,
|
||||
JOB_DIMENSION_REPAIR,
|
||||
JOB_CAPTURE_DATE_BACKFILL,
|
||||
DEFAULT_STALE_MS,
|
||||
HEARTBEAT_INTERVAL_MS,
|
||||
};
|
||||
@@ -47,6 +47,13 @@ const EXCLUDED_TABLES = new Set([
|
||||
// they would attach to whatever ids the target's own re-scan happens to
|
||||
// mint, silently suppressing merge suggestions in unrelated galleries.
|
||||
'event_people_merge_dismissals',
|
||||
// Live lease state for the maintenance sweeps (#1181), not data. An archive
|
||||
// taken while a sweep was running would otherwise carry is_running = true and
|
||||
// a claim token belonging to a process on the source install. Restored within
|
||||
// the staleness window, the target reports the job as running and refuses new
|
||||
// POSTs, with no runner anywhere that could release it. The table is seeded
|
||||
// by its migration, so the target already has the rows it needs.
|
||||
'maintenance_jobs',
|
||||
]);
|
||||
|
||||
// Storage subdirs holding non-recalculable blobs — always included.
|
||||
|
||||
Reference in New Issue
Block a user