mirror of
https://github.com/awatertrevi/infisical.git
synced 2026-10-10 10:29:17 +00:00
Merge pull request #4773 from Infisical/revert-4760-misc/added-health-and-ready-probe-endpoints
Revert "misc: added health and ready probe endpoints"
This commit is contained in:
@@ -8,7 +8,7 @@ import path from "path";
|
||||
import { seedData1 } from "@app/db/seed-data";
|
||||
import { getDatabaseCredentials, getHsmConfig, initEnvConfig } from "@app/lib/config/env";
|
||||
import { initLogger } from "@app/lib/logger";
|
||||
import { main, markServerReady } from "@app/server/app";
|
||||
import { main } from "@app/server/app";
|
||||
import { AuthMethod, AuthTokenType } from "@app/services/auth/auth-type";
|
||||
|
||||
import { mockSmtpServer } from "./mocks/smtp";
|
||||
@@ -83,7 +83,7 @@ export default {
|
||||
|
||||
await queue.initialize();
|
||||
|
||||
const { server, completeServerInitialization } = await main({
|
||||
const server = await main({
|
||||
db,
|
||||
smtp,
|
||||
logger,
|
||||
@@ -96,10 +96,6 @@ export default {
|
||||
envConfig: envCfg
|
||||
});
|
||||
|
||||
await completeServerInitialization();
|
||||
|
||||
markServerReady();
|
||||
|
||||
await bootstrapCheck({ db });
|
||||
|
||||
// @ts-expect-error type
|
||||
|
||||
@@ -12,7 +12,6 @@ type TArgs = {
|
||||
auditLogDb?: Knex;
|
||||
applicationDb: Knex;
|
||||
logger: Logger;
|
||||
onMigrationLockAcquired?: () => void;
|
||||
};
|
||||
|
||||
const isProduction = process.env.NODE_ENV === "production";
|
||||
@@ -31,7 +30,7 @@ const migrationStatusCheckErrorHandler = (err: Error) => {
|
||||
throw err;
|
||||
};
|
||||
|
||||
export const runMigrations = async ({ applicationDb, auditLogDb, logger, onMigrationLockAcquired }: TArgs) => {
|
||||
export const runMigrations = async ({ applicationDb, auditLogDb, logger }: TArgs) => {
|
||||
try {
|
||||
// akhilmhdh(Feb 10 2025): 2 years from now remove this
|
||||
if (isProduction) {
|
||||
@@ -86,13 +85,6 @@ export const runMigrations = async ({ applicationDb, auditLogDb, logger, onMigra
|
||||
|
||||
await applicationDb.transaction(async (tx) => {
|
||||
await tx.raw("SELECT pg_advisory_xact_lock(?)", [PgSqlLock.BootUpMigration]);
|
||||
|
||||
// Signal that this container is running migrations so that it can be marked as healthy/alive
|
||||
// This is to prevent the container from being killed by the orchestrator
|
||||
if (onMigrationLockAcquired) {
|
||||
onMigrationLockAcquired();
|
||||
}
|
||||
|
||||
logger.info("Running application migrations.");
|
||||
|
||||
const didPreviousInstanceRunMigration = !(await applicationDb.migrate
|
||||
|
||||
@@ -84,7 +84,7 @@ export const isHsmActiveAndEnabled = async ({
|
||||
|
||||
rootKmsConfigEncryptionStrategy = (rootKmsConfig?.encryptionStrategy || null) as RootKeyEncryptionStrategy | null;
|
||||
if (
|
||||
(rootKmsConfigEncryptionStrategy === RootKeyEncryptionStrategy.HSM || isHsmConfigured) &&
|
||||
rootKmsConfigEncryptionStrategy === RootKeyEncryptionStrategy.HSM &&
|
||||
licenseService &&
|
||||
!licenseService.onPremFeatures.hsm
|
||||
) {
|
||||
|
||||
@@ -42,7 +42,6 @@ export const secretRotationV2QueueServiceFactory = async ({
|
||||
smtpService,
|
||||
notificationService
|
||||
}: TSecretRotationV2QueueServiceFactoryDep) => {
|
||||
const init = async () => {
|
||||
const appCfg = getConfig();
|
||||
|
||||
if (appCfg.isRotationDevelopmentMode) {
|
||||
@@ -207,7 +206,4 @@ export const secretRotationV2QueueServiceFactory = async ({
|
||||
undefined,
|
||||
{ tz: "UTC" }
|
||||
);
|
||||
};
|
||||
|
||||
return { init };
|
||||
};
|
||||
|
||||
@@ -141,61 +141,6 @@ export const secretScanningV2QueueServiceFactory = async ({
|
||||
}
|
||||
};
|
||||
|
||||
const queueResourceDiffScan = async ({
|
||||
payload,
|
||||
dataSourceId,
|
||||
dataSourceType
|
||||
}: Pick<TQueueSecretScanningResourceDiffScan, "payload" | "dataSourceId" | "dataSourceType">) => {
|
||||
const factory = SECRET_SCANNING_FACTORY_MAP[dataSourceType as SecretScanningDataSource]({
|
||||
kmsService,
|
||||
appConnectionDAL
|
||||
});
|
||||
|
||||
const resourcePayload = factory.getDiffScanResourcePayload(payload);
|
||||
|
||||
try {
|
||||
const { resourceId, scanId } = await secretScanningV2DAL.resources.transaction(async (tx) => {
|
||||
const [resource] = await secretScanningV2DAL.resources.upsert(
|
||||
[
|
||||
{
|
||||
...resourcePayload,
|
||||
dataSourceId
|
||||
}
|
||||
],
|
||||
["externalId", "dataSourceId"],
|
||||
tx
|
||||
);
|
||||
|
||||
const scan = await secretScanningV2DAL.scans.create(
|
||||
{
|
||||
resourceId: resource.id,
|
||||
type: SecretScanningScanType.DiffScan
|
||||
},
|
||||
tx
|
||||
);
|
||||
|
||||
return {
|
||||
resourceId: resource.id,
|
||||
scanId: scan.id
|
||||
};
|
||||
});
|
||||
|
||||
await queueService.queuePg(QueueJobs.SecretScanningV2DiffScan, {
|
||||
payload,
|
||||
dataSourceId,
|
||||
dataSourceType,
|
||||
scanId,
|
||||
resourceId
|
||||
});
|
||||
} catch (error) {
|
||||
logger.error(
|
||||
error,
|
||||
`secretScanningV2Queue: Failed to queue diff scan [dataSourceId=${dataSourceId}] [resourceExternalId=${resourcePayload.externalId}]`
|
||||
);
|
||||
}
|
||||
};
|
||||
|
||||
const init = async () => {
|
||||
await queueService.startPg<QueueName.SecretScanningV2>(
|
||||
QueueJobs.SecretScanningV2FullScan,
|
||||
async ([job]) => {
|
||||
@@ -392,6 +337,60 @@ export const secretScanningV2QueueServiceFactory = async ({
|
||||
}
|
||||
);
|
||||
|
||||
const queueResourceDiffScan = async ({
|
||||
payload,
|
||||
dataSourceId,
|
||||
dataSourceType
|
||||
}: Pick<TQueueSecretScanningResourceDiffScan, "payload" | "dataSourceId" | "dataSourceType">) => {
|
||||
const factory = SECRET_SCANNING_FACTORY_MAP[dataSourceType as SecretScanningDataSource]({
|
||||
kmsService,
|
||||
appConnectionDAL
|
||||
});
|
||||
|
||||
const resourcePayload = factory.getDiffScanResourcePayload(payload);
|
||||
|
||||
try {
|
||||
const { resourceId, scanId } = await secretScanningV2DAL.resources.transaction(async (tx) => {
|
||||
const [resource] = await secretScanningV2DAL.resources.upsert(
|
||||
[
|
||||
{
|
||||
...resourcePayload,
|
||||
dataSourceId
|
||||
}
|
||||
],
|
||||
["externalId", "dataSourceId"],
|
||||
tx
|
||||
);
|
||||
|
||||
const scan = await secretScanningV2DAL.scans.create(
|
||||
{
|
||||
resourceId: resource.id,
|
||||
type: SecretScanningScanType.DiffScan
|
||||
},
|
||||
tx
|
||||
);
|
||||
|
||||
return {
|
||||
resourceId: resource.id,
|
||||
scanId: scan.id
|
||||
};
|
||||
});
|
||||
|
||||
await queueService.queuePg(QueueJobs.SecretScanningV2DiffScan, {
|
||||
payload,
|
||||
dataSourceId,
|
||||
dataSourceType,
|
||||
scanId,
|
||||
resourceId
|
||||
});
|
||||
} catch (error) {
|
||||
logger.error(
|
||||
error,
|
||||
`secretScanningV2Queue: Failed to queue diff scan [dataSourceId=${dataSourceId}] [resourceExternalId=${resourcePayload.externalId}]`
|
||||
);
|
||||
}
|
||||
};
|
||||
|
||||
await queueService.startPg<QueueName.SecretScanningV2>(
|
||||
QueueJobs.SecretScanningV2DiffScan,
|
||||
async ([job]) => {
|
||||
@@ -667,11 +666,9 @@ export const secretScanningV2QueueServiceFactory = async ({
|
||||
pollingIntervalSeconds: 1
|
||||
}
|
||||
);
|
||||
};
|
||||
|
||||
return {
|
||||
queueDataSourceFullScan,
|
||||
queueResourceDiffScan,
|
||||
init
|
||||
queueResourceDiffScan
|
||||
};
|
||||
};
|
||||
|
||||
+10
-41
@@ -16,7 +16,7 @@ import { buildRedisFromConfig } from "./lib/config/redis";
|
||||
import { removeTemporaryBaseDirectory } from "./lib/files";
|
||||
import { initLogger } from "./lib/logger";
|
||||
import { queueServiceFactory } from "./queue";
|
||||
import { main, markRunningMigrations, markServerReady } from "./server/app";
|
||||
import { main } from "./server/app";
|
||||
import { bootstrapCheck } from "./server/boot-strap-check";
|
||||
import { kmsRootConfigDALFactory } from "./services/kms/kms-root-config-dal";
|
||||
import { smtpServiceFactory } from "./services/smtp/smtp-service";
|
||||
@@ -59,6 +59,8 @@ const run = async () => {
|
||||
})
|
||||
: undefined;
|
||||
|
||||
await runMigrations({ applicationDb: db, auditLogDb, logger });
|
||||
|
||||
const smtp = smtpServiceFactory(formatSmtpConfig());
|
||||
|
||||
const queue = queueServiceFactory(envConfig, {
|
||||
@@ -72,7 +74,7 @@ const run = async () => {
|
||||
const keyStore = keyStoreFactory(envConfig, keyValueStoreDAL);
|
||||
const redis = buildRedisFromConfig(envConfig);
|
||||
|
||||
const { server, completeServerInitialization } = await main({
|
||||
const server = await main({
|
||||
db,
|
||||
auditLogDb,
|
||||
superAdminDAL,
|
||||
@@ -85,6 +87,7 @@ const run = async () => {
|
||||
redis,
|
||||
envConfig
|
||||
});
|
||||
const bootstrap = await bootstrapCheck({ db });
|
||||
|
||||
// eslint-disable-next-line
|
||||
process.on("SIGINT", async () => {
|
||||
@@ -118,46 +121,12 @@ const run = async () => {
|
||||
|
||||
await server.listen({
|
||||
port: envConfig.PORT,
|
||||
host: envConfig.HOST
|
||||
});
|
||||
|
||||
logger.info(`Server listening on ${envConfig.HOST}:${envConfig.PORT}`);
|
||||
logger.info("Running migrations...");
|
||||
|
||||
// Run migrations while server is up
|
||||
// All containers start as NOT HEALTHY (waiting for migrations)
|
||||
// Container that acquires lock: becomes HEALTHY (running migrations) + NOT READY (no traffic)
|
||||
// Other containers waiting: stay NOT HEALTHY (waiting) + NOT READY (no traffic)
|
||||
await runMigrations({
|
||||
applicationDb: db,
|
||||
auditLogDb,
|
||||
logger,
|
||||
onMigrationLockAcquired: () => {
|
||||
// Called after successfully acquiring the lock
|
||||
// This container is now the migration runner
|
||||
markRunningMigrations();
|
||||
logger.info("Migration lock acquired! This container is running migrations.");
|
||||
}
|
||||
});
|
||||
|
||||
logger.info("Migrations complete. Completing server initialization...");
|
||||
|
||||
try {
|
||||
await completeServerInitialization();
|
||||
} catch (error) {
|
||||
logger.error(error, "Failed to complete server initialization");
|
||||
await server.close();
|
||||
await queue.shutdown();
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
logger.info("Server initialization complete. Marking server as READY...");
|
||||
|
||||
markServerReady();
|
||||
|
||||
logger.info("Server is ready to accept traffic");
|
||||
const bootstrap = await bootstrapCheck({ db });
|
||||
host: envConfig.HOST,
|
||||
listenTextResolver: (address) => {
|
||||
void bootstrap();
|
||||
return address;
|
||||
}
|
||||
});
|
||||
};
|
||||
|
||||
void run();
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
/* eslint-disable import/extensions */
|
||||
import path from "node:path";
|
||||
import { monitorEventLoopDelay } from "perf_hooks";
|
||||
|
||||
import type { FastifyCookieOptions } from "@fastify/cookie";
|
||||
import cookie from "@fastify/cookie";
|
||||
@@ -25,7 +24,6 @@ import { TQueueServiceFactory } from "@app/queue";
|
||||
import { TKmsRootConfigDALFactory } from "@app/services/kms/kms-root-config-dal";
|
||||
import { TSmtpService } from "@app/services/smtp/smtp-service";
|
||||
import { TSuperAdminDALFactory } from "@app/services/super-admin/super-admin-dal";
|
||||
import { getServerCfg } from "@app/services/super-admin/super-admin-service";
|
||||
|
||||
import { globalRateLimiterCfg } from "./config/rateLimiter";
|
||||
import { addErrorsToResponseSchemas } from "./plugins/add-errors-to-response-schemas";
|
||||
@@ -38,15 +36,6 @@ import { registerServeUI } from "./plugins/serve-ui";
|
||||
import { fastifySwagger } from "./plugins/swagger";
|
||||
import { registerRoutes } from "./routes";
|
||||
|
||||
const histogram = monitorEventLoopDelay({ resolution: 20 });
|
||||
histogram.enable();
|
||||
|
||||
const serverState = {
|
||||
isReady: false,
|
||||
isRunningMigrations: false,
|
||||
isWaitingForMigrations: true // Start as true - containers are unhealthy until they acquire migration lock or complete
|
||||
};
|
||||
|
||||
type TMain = {
|
||||
auditLogDb?: Knex;
|
||||
db: Knex;
|
||||
@@ -156,72 +145,7 @@ export const main = async ({
|
||||
})
|
||||
});
|
||||
|
||||
// Health check - returns 200 only if doing useful work (running migrations or ready)
|
||||
// Returns 503 if waiting for another container to finish migrations
|
||||
server.get("/api/health", async (_, reply) => {
|
||||
if (serverState.isWaitingForMigrations) {
|
||||
return reply.code(503).send({
|
||||
status: "waiting",
|
||||
message: "Waiting for migrations to complete in another container"
|
||||
});
|
||||
}
|
||||
return { status: "ok", message: "Server is alive" };
|
||||
});
|
||||
|
||||
// Global preHandler to block requests during migrations
|
||||
server.addHook("preHandler", async (request, reply) => {
|
||||
if (request.url === "/api/health" || request.url === "/api/ready") {
|
||||
return;
|
||||
}
|
||||
if (!serverState.isReady) {
|
||||
return reply.code(503).send({
|
||||
status: "unavailable",
|
||||
message: "Server is starting up, migrations in progress. Please try again in a moment."
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
// Readiness check - returns 503 until migrations are complete
|
||||
server.get("/api/ready", async (request, reply) => {
|
||||
const cfg = getConfig();
|
||||
|
||||
const meanLagMs = histogram.mean / 1e6;
|
||||
const maxLagMs = histogram.max / 1e6;
|
||||
const p99LagMs = histogram.percentile(99) / 1e6;
|
||||
|
||||
request.log.info(
|
||||
`Event loop stats - Mean: ${meanLagMs.toFixed(2)}ms, Max: ${maxLagMs.toFixed(2)}ms, p99: ${p99LagMs.toFixed(2)}ms`
|
||||
);
|
||||
|
||||
request.log.info(`Raw event loop stats: ${JSON.stringify(histogram, null, 2)}`);
|
||||
|
||||
if (!serverState.isReady) {
|
||||
return reply.code(503).send({
|
||||
date: new Date(),
|
||||
message: "Server is starting up, migrations in progress",
|
||||
emailConfigured: cfg.isSmtpConfigured,
|
||||
redisConfigured: cfg.isRedisConfigured,
|
||||
secretScanningConfigured: cfg.isSecretScanningConfigured,
|
||||
samlDefaultOrgSlug: cfg.samlDefaultOrgSlug,
|
||||
auditLogStorageDisabled: Boolean(cfg.DISABLE_AUDIT_LOG_STORAGE)
|
||||
});
|
||||
}
|
||||
|
||||
const serverCfg = await getServerCfg();
|
||||
|
||||
return {
|
||||
date: new Date(),
|
||||
message: "Ok",
|
||||
emailConfigured: cfg.isSmtpConfigured,
|
||||
inviteOnlySignup: Boolean(serverCfg.allowSignUp),
|
||||
redisConfigured: cfg.isRedisConfigured,
|
||||
secretScanningConfigured: cfg.isSecretScanningConfigured,
|
||||
samlDefaultOrgSlug: cfg.samlDefaultOrgSlug,
|
||||
auditLogStorageDisabled: Boolean(cfg.DISABLE_AUDIT_LOG_STORAGE)
|
||||
};
|
||||
});
|
||||
|
||||
const completeServerInitialization = await registerRoutes(server, {
|
||||
await server.register(registerRoutes, {
|
||||
smtp,
|
||||
queue,
|
||||
db,
|
||||
@@ -240,21 +164,10 @@ export const main = async ({
|
||||
|
||||
await server.ready();
|
||||
server.swagger();
|
||||
return { server, completeServerInitialization };
|
||||
return server;
|
||||
} catch (err) {
|
||||
server.log.error(err);
|
||||
await queue.shutdown();
|
||||
process.exit(1);
|
||||
}
|
||||
};
|
||||
|
||||
export const markServerReady = () => {
|
||||
serverState.isReady = true;
|
||||
serverState.isRunningMigrations = false;
|
||||
serverState.isWaitingForMigrations = false;
|
||||
};
|
||||
|
||||
export const markRunningMigrations = () => {
|
||||
serverState.isRunningMigrations = true;
|
||||
serverState.isWaitingForMigrations = false;
|
||||
};
|
||||
|
||||
@@ -2208,7 +2208,7 @@ export const registerRoutes = async (
|
||||
internalCaFns
|
||||
});
|
||||
|
||||
const secretRotationV2Queue = await secretRotationV2QueueServiceFactory({
|
||||
await secretRotationV2QueueServiceFactory({
|
||||
secretRotationV2Service,
|
||||
secretRotationV2DAL,
|
||||
queueService,
|
||||
@@ -2305,6 +2305,8 @@ export const registerRoutes = async (
|
||||
// If FIPS is enabled, we check to ensure that the users license includes FIPS mode.
|
||||
crypto.verifyFipsLicense(licenseService);
|
||||
|
||||
await superAdminService.initServerCfg();
|
||||
|
||||
// Start HSM service if it's configured/enabled.
|
||||
await hsmService.startService();
|
||||
|
||||
@@ -2329,9 +2331,6 @@ export const registerRoutes = async (
|
||||
}
|
||||
}
|
||||
|
||||
const completeServerInitialization = async () => {
|
||||
await superAdminService.initServerCfg();
|
||||
|
||||
await telemetryQueue.startTelemetryCheck();
|
||||
await telemetryQueue.startAggregatedEventsJob();
|
||||
await dailyResourceCleanUp.init();
|
||||
@@ -2346,44 +2345,8 @@ export const registerRoutes = async (
|
||||
await kmsService.startService(hsmStatus);
|
||||
await microsoftTeamsService.start();
|
||||
await dynamicSecretQueueService.init();
|
||||
await secretScanningV2Queue.init();
|
||||
await secretRotationV2Queue.init();
|
||||
await notificationQueue.init();
|
||||
await eventBusService.init();
|
||||
|
||||
const cronJobs: CronJob[] = [];
|
||||
if (appCfg.isProductionMode) {
|
||||
const rateLimitSyncJob = await rateLimitService.initializeBackgroundSync();
|
||||
if (rateLimitSyncJob) {
|
||||
cronJobs.push(rateLimitSyncJob);
|
||||
}
|
||||
const licenseSyncJob = await licenseService.initializeBackgroundSync();
|
||||
if (licenseSyncJob) {
|
||||
cronJobs.push(licenseSyncJob);
|
||||
}
|
||||
|
||||
const microsoftTeamsSyncJob = await microsoftTeamsService.initializeBackgroundSync();
|
||||
if (microsoftTeamsSyncJob) {
|
||||
cronJobs.push(microsoftTeamsSyncJob);
|
||||
}
|
||||
|
||||
const adminIntegrationsSyncJob = await superAdminService.initializeAdminIntegrationConfigSync();
|
||||
if (adminIntegrationsSyncJob) {
|
||||
cronJobs.push(adminIntegrationsSyncJob);
|
||||
}
|
||||
}
|
||||
|
||||
const configSyncJob = await superAdminService.initializeEnvConfigSync();
|
||||
if (configSyncJob) {
|
||||
cronJobs.push(configSyncJob);
|
||||
}
|
||||
|
||||
const oauthConfigSyncJob = await initializeOauthConfigSync();
|
||||
if (oauthConfigSyncJob) {
|
||||
cronJobs.push(oauthConfigSyncJob);
|
||||
}
|
||||
};
|
||||
|
||||
// inject all services
|
||||
server.decorate<FastifyZodProvider["services"]>("services", {
|
||||
login: loginService,
|
||||
@@ -2512,6 +2475,38 @@ export const registerRoutes = async (
|
||||
convertor: convertorService
|
||||
});
|
||||
|
||||
const cronJobs: CronJob[] = [];
|
||||
if (appCfg.isProductionMode) {
|
||||
const rateLimitSyncJob = await rateLimitService.initializeBackgroundSync();
|
||||
if (rateLimitSyncJob) {
|
||||
cronJobs.push(rateLimitSyncJob);
|
||||
}
|
||||
const licenseSyncJob = await licenseService.initializeBackgroundSync();
|
||||
if (licenseSyncJob) {
|
||||
cronJobs.push(licenseSyncJob);
|
||||
}
|
||||
|
||||
const microsoftTeamsSyncJob = await microsoftTeamsService.initializeBackgroundSync();
|
||||
if (microsoftTeamsSyncJob) {
|
||||
cronJobs.push(microsoftTeamsSyncJob);
|
||||
}
|
||||
|
||||
const adminIntegrationsSyncJob = await superAdminService.initializeAdminIntegrationConfigSync();
|
||||
if (adminIntegrationsSyncJob) {
|
||||
cronJobs.push(adminIntegrationsSyncJob);
|
||||
}
|
||||
}
|
||||
|
||||
const configSyncJob = await superAdminService.initializeEnvConfigSync();
|
||||
if (configSyncJob) {
|
||||
cronJobs.push(configSyncJob);
|
||||
}
|
||||
|
||||
const oauthConfigSyncJob = await initializeOauthConfigSync();
|
||||
if (oauthConfigSyncJob) {
|
||||
cronJobs.push(oauthConfigSyncJob);
|
||||
}
|
||||
|
||||
server.decorate<FastifyZodProvider["store"]>("store", {
|
||||
user: userDAL,
|
||||
kmipClient: kmipClientDAL
|
||||
@@ -2600,10 +2595,9 @@ export const registerRoutes = async (
|
||||
await server.register(registerV4Routes, { prefix: "/api/v4" });
|
||||
|
||||
server.addHook("onClose", async () => {
|
||||
cronJobs.forEach((job) => job.stop());
|
||||
await telemetryService.flushAll();
|
||||
await eventBusService.close();
|
||||
sseService.close();
|
||||
});
|
||||
|
||||
return completeServerInitialization;
|
||||
};
|
||||
|
||||
@@ -10,7 +10,6 @@ type TNotificationQueueServiceFactoryDep = {
|
||||
|
||||
export type TNotificationQueueServiceFactory = {
|
||||
pushUserNotifications: (data: TCreateUserNotificationDTO[]) => Promise<void>;
|
||||
init: () => Promise<void>;
|
||||
};
|
||||
|
||||
export const notificationQueueServiceFactory = async ({
|
||||
@@ -21,7 +20,6 @@ export const notificationQueueServiceFactory = async ({
|
||||
await queueService.queuePg(QueueJobs.UserNotification, { notifications: data });
|
||||
};
|
||||
|
||||
const init = async () => {
|
||||
await queueService.startPg(
|
||||
QueueJobs.UserNotification,
|
||||
async ([job]) => {
|
||||
@@ -34,10 +32,8 @@ export const notificationQueueServiceFactory = async ({
|
||||
pollingIntervalSeconds: 1
|
||||
}
|
||||
);
|
||||
};
|
||||
|
||||
return {
|
||||
pushUserNotifications,
|
||||
init
|
||||
pushUserNotifications
|
||||
};
|
||||
};
|
||||
|
||||
@@ -137,33 +137,6 @@ Configure database read replicas for high availability PostgreSQL setups:
|
||||
DB_READ_REPLICAS='[{"DB_CONNECTION_URI":"postgresql://user:pass@replica:5432/db?sslmode=require"}]'
|
||||
```
|
||||
|
||||
### Health Check Endpoints
|
||||
|
||||
Infisical provides two health check endpoints for proper container orchestration and load balancer integration:
|
||||
|
||||
#### `/api/health` - Container Health Check
|
||||
|
||||
Determines whether the application container should be kept alive or terminated.
|
||||
|
||||
- Returns `200` if the application is running and operational
|
||||
- Returns `200` even during startup tasks
|
||||
- Returns `503` only if the application has crashed or is unable to start
|
||||
|
||||
**Use for**: Docker health checks, Kubernetes liveness probes, ECS task health checks.
|
||||
|
||||
#### `/api/ready` - Traffic Readiness Check
|
||||
|
||||
Determines whether the application instance is ready to receive production traffic.
|
||||
|
||||
- Returns `200` when the application is fully ready to serve requests
|
||||
- Returns `503` during startup tasks (e.g., database migrations, initialization)
|
||||
|
||||
**Use for**: Load balancer health checks, Kubernetes readiness probes, ALB target health checks.
|
||||
|
||||
#### Why Two Endpoints?
|
||||
|
||||
Using both endpoints together enables zero-downtime deployments: containers stay alive during startup tasks (`/api/health` returns `200`) while load balancers avoid sending traffic to instances that aren't ready (`/api/ready` returns `503`). This ensures existing instances continue serving traffic until new instances complete their initialization.
|
||||
|
||||
### Operational Security
|
||||
|
||||
#### User Access Management
|
||||
@@ -234,17 +207,14 @@ docker run --memory=1g --cpus=0.5 infisical/infisical:latest
|
||||
|
||||
#### Health Monitoring
|
||||
|
||||
**Configure health checks**. Set up Docker health checks using the appropriate endpoint:
|
||||
**Configure health checks**. Set up Docker health checks:
|
||||
|
||||
```dockerfile
|
||||
# In Dockerfile or docker-compose.yml
|
||||
# Use /api/health for container health (keeps container alive during startup)
|
||||
HEALTHCHECK --interval=30s --timeout=3s --start-period=10s --retries=3 \
|
||||
CMD curl -f http://localhost:8080/api/health || exit 1
|
||||
CMD curl -f http://localhost:8080/api/status || exit 1
|
||||
```
|
||||
|
||||
**Note**: Use `/api/health` for container health checks and `/api/ready` for load balancer readiness checks. See [Health Check Endpoints](#health-check-endpoints) for detailed information.
|
||||
|
||||
#### Network Security
|
||||
|
||||
**Host firewall configuration**. Configure host-level firewall for Docker deployments:
|
||||
@@ -463,32 +433,26 @@ stringData:
|
||||
|
||||
#### Health Monitoring
|
||||
|
||||
**Set up health checks**. Configure readiness and liveness probes using the appropriate endpoints:
|
||||
**Set up health checks**. Configure readiness and liveness probes:
|
||||
|
||||
```yaml
|
||||
# Health check configuration
|
||||
containers:
|
||||
- name: infisical
|
||||
# Use /api/ready for readiness (traffic routing)
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /api/ready
|
||||
path: /api/status
|
||||
port: 8080
|
||||
initialDelaySeconds: 10
|
||||
periodSeconds: 5
|
||||
failureThreshold: 3
|
||||
# Use /api/health for liveness (container restart)
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /api/health
|
||||
path: /api/status
|
||||
port: 8080
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 10
|
||||
failureThreshold: 3
|
||||
```
|
||||
|
||||
**Important**: The `readinessProbe` uses `/api/ready` to ensure traffic is only sent to pods that are fully initialized. The `livenessProbe` uses `/api/health` to keep the container alive during startup. See [Health Check Endpoints](#health-check-endpoints) for detailed information.
|
||||
|
||||
#### Infrastructure Considerations
|
||||
|
||||
**Use managed databases (if possible)**. For production deployments, consider using managed PostgreSQL and Redis services instead of in-cluster instances when feasible, as they typically provide better security, backup, and maintenance capabilities.
|
||||
|
||||
Reference in New Issue
Block a user