feat(nextcloud-status): erneute Prüfung nach Ausfall und Hinweis auf der Kachel
- Wiederholungsauftrag jede Minute für Clouds mit einem Fehlschlag älter als fünf Minuten - Neue Adresse setzt den Prüfstand zurück, der gemeldete Zustand bleibt - Kachel: Hinweis Prüfung fehlgeschlagen, Fehlercodes als lesbarer Text mit Tooltip Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -1,5 +1,6 @@
|
||||
import { Injectable, Logger, OnApplicationBootstrap } from '@nestjs/common';
|
||||
import { SchedulerRegistry } from '@nestjs/schedule';
|
||||
import { RETRY_DELAY_MS } from './nextcloud-alert-rules';
|
||||
import { NextcloudReleaseService } from './nextcloud-release.service';
|
||||
import {
|
||||
CHECK_CONCURRENCY,
|
||||
@@ -27,6 +28,10 @@ const CronJobClass: new (cronTime: string, onTick: () => void) => { start(): voi
|
||||
export const NEXTCLOUD_JOB_NAME = 'nextcloud-status-poll';
|
||||
/** Jede volle Stunde (L-08). */
|
||||
export const NEXTCLOUD_CRON = '0 * * * *';
|
||||
/** Name des Wiederholungsauftrags (quick-261002-kxc, L-03). */
|
||||
export const NEXTCLOUD_RETRY_JOB_NAME = 'nextcloud-status-retry';
|
||||
/** Jede Minute: prueft nur Clouds, deren erster Fehlschlag fuenf Minuten zurueckliegt. */
|
||||
export const NEXTCLOUD_RETRY_CRON = '* * * * *';
|
||||
|
||||
/**
|
||||
* NextcloudStatusSchedulerService — stuendliche Pruefung aller Clouds
|
||||
@@ -50,6 +55,7 @@ export const NEXTCLOUD_CRON = '0 * * * *';
|
||||
export class NextcloudStatusSchedulerService implements OnApplicationBootstrap {
|
||||
private readonly logger = new Logger(NextcloudStatusSchedulerService.name);
|
||||
private running = false;
|
||||
private retryRunning = false;
|
||||
|
||||
constructor(
|
||||
private readonly schedulerRegistry: SchedulerRegistry,
|
||||
@@ -75,6 +81,15 @@ export class NextcloudStatusSchedulerService implements OnApplicationBootstrap {
|
||||
this.schedulerRegistry.addCronJob(NEXTCLOUD_JOB_NAME, job as any);
|
||||
job.start();
|
||||
this.logger.log(`Nextcloud-Status cron job registered: ${NEXTCLOUD_CRON}`);
|
||||
const retryJob = new CronJobClass(NEXTCLOUD_RETRY_CRON, () => {
|
||||
this.retryTick().catch((err) =>
|
||||
this.logger.error(`Nextcloud retry tick failed: ${(err as Error).message}`),
|
||||
);
|
||||
});
|
||||
// biome-ignore lint/suspicious/noExplicitAny: Cast wie in ProxmoxSchedulerService
|
||||
this.schedulerRegistry.addCronJob(NEXTCLOUD_RETRY_JOB_NAME, retryJob as any);
|
||||
retryJob.start();
|
||||
this.logger.log(`Nextcloud-Status retry job registered: ${NEXTCLOUD_RETRY_CRON}`);
|
||||
void this.release.refresh().catch(() => undefined);
|
||||
} catch (err) {
|
||||
this.logger.error(`Nextcloud-Status scheduler init failed: ${(err as Error).message}`);
|
||||
@@ -89,29 +104,56 @@ export class NextcloudStatusSchedulerService implements OnApplicationBootstrap {
|
||||
}
|
||||
this.running = true;
|
||||
try {
|
||||
const rows = await this.service.loadAllInstancesForScheduler();
|
||||
// Je Mandant gruppiert, damit jede Pruefung an IHREN Mandanten gebunden bleibt.
|
||||
const byTenant = new Map<string, string[]>();
|
||||
for (const row of rows) {
|
||||
const ids = byTenant.get(row.tenantId) ?? [];
|
||||
ids.push(row.id);
|
||||
byTenant.set(row.tenantId, ids);
|
||||
}
|
||||
const work: { tenantId: string; id: string }[] = [];
|
||||
for (const [tenantId, ids] of byTenant) {
|
||||
for (const id of ids) work.push({ tenantId, id });
|
||||
}
|
||||
await runWithConcurrency(work, CHECK_CONCURRENCY, async ({ tenantId, id }) => {
|
||||
try {
|
||||
await this.service.checkInstance(tenantId, id);
|
||||
} catch (err) {
|
||||
this.logger.error(
|
||||
`Nextcloud check failed for instance ${id} (tenant ${tenantId}): ${(err as Error).message}`,
|
||||
);
|
||||
}
|
||||
});
|
||||
await this.checkRows(await this.service.loadAllInstancesForScheduler());
|
||||
} finally {
|
||||
this.running = false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Wiederholung: nur Clouds mit genau einem Fehlschlag, der mindestens
|
||||
* `RETRY_DELAY_MS` zurueckliegt. Wirft nie; ein laufender Durchlauf haelt
|
||||
* den naechsten an.
|
||||
*/
|
||||
async retryTick(now: Date = new Date()): Promise<void> {
|
||||
if (this.retryRunning) {
|
||||
this.logger.warn('Nextcloud retry tick skipped — previous run still active');
|
||||
return;
|
||||
}
|
||||
this.retryRunning = true;
|
||||
try {
|
||||
const rows = await this.service.loadAllInstancesForScheduler({
|
||||
retryDueBefore: new Date(now.getTime() - RETRY_DELAY_MS),
|
||||
});
|
||||
await this.checkRows(rows);
|
||||
} catch (err) {
|
||||
this.logger.error(`Nextcloud retry tick failed: ${(err as Error).message}`);
|
||||
} finally {
|
||||
this.retryRunning = false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Prueft jede Cloud an ihren eigenen Mandanten gebunden, hoechstens vier gleichzeitig. */
|
||||
private async checkRows(rows: { id: string; tenantId: string }[]): Promise<void> {
|
||||
// Je Mandant gruppiert, damit jede Pruefung an IHREN Mandanten gebunden bleibt.
|
||||
const byTenant = new Map<string, string[]>();
|
||||
for (const row of rows) {
|
||||
const ids = byTenant.get(row.tenantId) ?? [];
|
||||
ids.push(row.id);
|
||||
byTenant.set(row.tenantId, ids);
|
||||
}
|
||||
const work: { tenantId: string; id: string }[] = [];
|
||||
for (const [tenantId, ids] of byTenant) {
|
||||
for (const id of ids) work.push({ tenantId, id });
|
||||
}
|
||||
await runWithConcurrency(work, CHECK_CONCURRENCY, async ({ tenantId, id }) => {
|
||||
try {
|
||||
await this.service.checkInstance(tenantId, id);
|
||||
} catch (err) {
|
||||
this.logger.error(
|
||||
`Nextcloud check failed for instance ${id} (tenant ${tenantId}): ${(err as Error).message}`,
|
||||
);
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user