From 4301277b48d38aab5ee4ea1af0c59a0167647dda Mon Sep 17 00:00:00 2001 From: keyhan Date: Tue, 2 Jun 2026 15:44:08 +0330 Subject: [PATCH] Serve preview URLs over Traefik+TLS, stabilize them, and speed up builds. MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ingress / preview URLs: - Default the app Ingress class and ACME HTTP-01 solver to Traefik (k3s default) via a new INGRESS_CLASS env, instead of hardcoding nginx — fixes 404s on clusters without ingress-nginx. - Only put public, real-TLD hosts (custom domain + preview) in the TLS block; the internal *.apps.cloudhost.local host no longer poisons the Let's Encrypt order, so certs actually issue. - Make the per-app preview number stable across redeploys so URLs stop breaking, and let PREVIEW_BASE_DOMAIN configure the base domain. Registry pulls: - Point the k3s registries.yaml mirror endpoint at the registry NodePort on loopback so node containerd never depends on cluster DNS (image pulls survive node restarts). Builds: - Pin the Kaniko image, use IfNotPresent pull policy, drop the dead build queue/processor, and retry transient Kubernetes API errors while polling build jobs. Logs & apps list: - fluent-bit reads log files from head so startup output reaches Elasticsearch. - Order joined deployments newest-first so the apps list shows the latest deployment status. Allocation: - Reserve in-flight (pending/building) capacity and stop globally degrading the cluster on a single allocation failure, so concurrent deploys don't starve or wrongly report "no healthy cluster". Co-Authored-By: Claude Opus 4.7 --- backend/.env.example | 6 + .../cloudhost-app/templates/deployment.yaml | 4 +- .../templates/fluent-bit-configmap.yaml | 4 + .../helm/cloudhost-app/templates/ingress.yaml | 21 ++- .../src/applications/applications.service.ts | 8 +- backend/src/build/build.module.ts | 5 +- backend/src/build/build.processor.ts | 42 ------ backend/src/build/build.service.ts | 57 ++++++--- backend/src/clusters/cluster-tools.service.ts | 6 +- backend/src/clusters/clusters.service.ts | 120 +++++++++++++++--- backend/src/config/configuration.ts | 13 ++ .../src/deployments/deployments.service.ts | 32 +++-- backend/src/kubernetes/kubernetes.service.ts | 35 +++-- 13 files changed, 244 insertions(+), 109 deletions(-) delete mode 100644 backend/src/build/build.processor.ts diff --git a/backend/.env.example b/backend/.env.example index b9590e7..3bd8d62 100644 --- a/backend/.env.example +++ b/backend/.env.example @@ -43,6 +43,12 @@ BUILD_SERVICE_ACCOUNT=kaniko-builder # Platform PLATFORM_DOMAIN=apps.cloudhost.local +# Base domain for per-user preview URLs (-<7-digit>.). +# Falls back to the root domain derived from FRONTEND_URL when unset. +# PREVIEW_BASE_DOMAIN=3fase.ir +# Ingress controller class for app Ingress + ACME HTTP-01 solver. +# k3s default is Traefik; use "nginx" only on clusters running ingress-nginx. +# INGRESS_CLASS=traefik UPLOAD_DIR=./uploads # PVC resize: use a dynamic StorageClass with allowVolumeExpansion (k3s: rancher.io/local-path) # k3s: use local-path and skip creating a custom class (set CREATE=false) diff --git a/backend/helm/cloudhost-app/templates/deployment.yaml b/backend/helm/cloudhost-app/templates/deployment.yaml index ce04231..20f39b9 100644 --- a/backend/helm/cloudhost-app/templates/deployment.yaml +++ b/backend/helm/cloudhost-app/templates/deployment.yaml @@ -26,7 +26,9 @@ spec: containers: - name: {{ $name }} image: {{ .Values.app.image | quote }} - imagePullPolicy: Always + # Image tags are unique per build (name:timestamp) and immutable — + # IfNotPresent avoids re-pulling on every restart/scale-up. + imagePullPolicy: IfNotPresent {{- if include "cloudhost-app.loggingWrapEnabled" . }} command: ["sh", "-c"] args: diff --git a/backend/helm/cloudhost-app/templates/fluent-bit-configmap.yaml b/backend/helm/cloudhost-app/templates/fluent-bit-configmap.yaml index d3a051a..680f867 100644 --- a/backend/helm/cloudhost-app/templates/fluent-bit-configmap.yaml +++ b/backend/helm/cloudhost-app/templates/fluent-bit-configmap.yaml @@ -26,6 +26,10 @@ data: Refresh_Interval 5 Mem_Buf_Limit 5MB Skip_Long_Lines On + # Read the whole file (incl. startup output written before fluent-bit + # attached) instead of only new lines — otherwise an idle app that + # logged only at boot would ship nothing to Elasticsearch. + Read_from_Head On [FILTER] Name record_modifier diff --git a/backend/helm/cloudhost-app/templates/ingress.yaml b/backend/helm/cloudhost-app/templates/ingress.yaml index 361bd32..7f91d00 100644 --- a/backend/helm/cloudhost-app/templates/ingress.yaml +++ b/backend/helm/cloudhost-app/templates/ingress.yaml @@ -3,6 +3,13 @@ {{- $ns := include "cloudhost-app.namespace" . -}} {{- $host := printf "%s.%s" (default $name .Values.ingress.subdomain) .Values.ingress.domain -}} {{- $previewHost := .Values.ingress.previewHost | default "" -}} +{{- $className := .Values.ingress.className | default "traefik" -}} +{{- /* Only public, real-TLD hosts (custom domain + preview) can get a managed cert. + The internal *.apps.cloudhost.local host is not a valid public suffix and would + make Let's Encrypt reject the whole order. */ -}} +{{- $tlsHosts := list -}} +{{- if .Values.ingress.customDomain }}{{- $tlsHosts = append $tlsHosts .Values.ingress.customDomain -}}{{- end -}} +{{- if $previewHost }}{{- $tlsHosts = append $tlsHosts $previewHost -}}{{- end -}} apiVersion: networking.k8s.io/v1 kind: Ingress metadata: @@ -10,10 +17,12 @@ metadata: namespace: {{ $ns }} labels: {{- include "cloudhost-app.labels" . | nindent 4 }} + {{- if gt (len $tlsHosts) 0 }} annotations: cert-manager.io/cluster-issuer: {{ .Values.ingress.clusterIssuer | quote }} + {{- end }} spec: - ingressClassName: nginx + ingressClassName: {{ $className }} rules: - host: {{ $host }} http: @@ -49,14 +58,12 @@ spec: port: number: 80 {{- end }} + {{- if gt (len $tlsHosts) 0 }} tls: - hosts: - - {{ $host }} - {{- if .Values.ingress.customDomain }} - - {{ .Values.ingress.customDomain }} - {{- end }} - {{- if $previewHost }} - - {{ $previewHost }} + {{- range $tlsHosts }} + - {{ . }} {{- end }} secretName: {{ $name }}-tls + {{- end }} {{- end }} diff --git a/backend/src/applications/applications.service.ts b/backend/src/applications/applications.service.ts index 5e7e213..60b748a 100644 --- a/backend/src/applications/applications.service.ts +++ b/backend/src/applications/applications.service.ts @@ -159,7 +159,9 @@ export class ApplicationsService { .createQueryBuilder('app') .leftJoinAndSelect('app.deployments', 'deployments') .where('app.userId = :userId', { userId }) - .orderBy('app.createdAt', 'DESC'); + .orderBy('app.createdAt', 'DESC') + // Ensure deployments[0] is the most recent so the UI shows the latest status. + .addOrderBy('deployments.createdAt', 'DESC'); if (options?.productType === 'application') { qb.andWhere( @@ -184,7 +186,9 @@ export class ApplicationsService { .createQueryBuilder('app') .leftJoinAndSelect('app.user', 'user') .leftJoinAndSelect('app.deployments', 'deployments') - .orderBy('app.createdAt', 'DESC'); + .orderBy('app.createdAt', 'DESC') + // Ensure deployments[0] is the most recent so the UI shows the latest status. + .addOrderBy('deployments.createdAt', 'DESC'); if (search && search.trim()) { const s = `%${search.trim()}%`; diff --git a/backend/src/build/build.module.ts b/backend/src/build/build.module.ts index 0abf23e..caa562f 100644 --- a/backend/src/build/build.module.ts +++ b/backend/src/build/build.module.ts @@ -1,17 +1,14 @@ import { Module, forwardRef } from '@nestjs/common'; -import { BullModule } from '@nestjs/bull'; import { BuildService } from './build.service'; -import { BuildProcessor } from './build.processor'; import { KubernetesModule } from '../kubernetes/kubernetes.module'; import { ClustersModule } from '../clusters/clusters.module'; @Module({ imports: [ - BullModule.registerQueue({ name: 'build' }), forwardRef(() => KubernetesModule), ClustersModule, ], - providers: [BuildService, BuildProcessor], + providers: [BuildService], exports: [BuildService], }) export class BuildModule {} diff --git a/backend/src/build/build.processor.ts b/backend/src/build/build.processor.ts deleted file mode 100644 index 98ed419..0000000 --- a/backend/src/build/build.processor.ts +++ /dev/null @@ -1,42 +0,0 @@ -import { Process, Processor } from '@nestjs/bull'; -import { Logger } from '@nestjs/common'; -import { Job } from 'bull'; -import { BuildService } from './build.service'; - -export interface BuildJobData { - applicationId: string; - deploymentId: string; - appName: string; - runtime: string; - gitUrl?: string; - codePath?: string; -} - -@Processor('build') -export class BuildProcessor { - private readonly logger = new Logger(BuildProcessor.name); - - constructor(private buildService: BuildService) {} - - @Process('build-image') - async handleBuild(job: Job) { - this.logger.log(`Processing build job ${job.id} for app: ${job.data.appName}`); - - try { - await job.progress(10); - - // The actual build logic is in BuildService - // This processor handles the queue job lifecycle - this.logger.log(`Build job ${job.id} started for ${job.data.appName}`); - - await job.progress(50); - await job.progress(100); - - this.logger.log(`Build job ${job.id} completed for ${job.data.appName}`); - return { status: 'completed', appName: job.data.appName }; - } catch (error: any) { - this.logger.error(`Build job ${job.id} failed: ${error.message}`); - throw error; - } - } -} diff --git a/backend/src/build/build.service.ts b/backend/src/build/build.service.ts index 176ea4d..ddaaf22 100644 --- a/backend/src/build/build.service.ts +++ b/backend/src/build/build.service.ts @@ -45,6 +45,12 @@ export class BuildService { private readonly logger = new Logger(BuildService.name); private readonly progressMap = new Map(); private readonly activeBuilds = new Map(); + /** + * Kaniko executor image. Pinned (not `:latest`) so it can be cached on the node + * with imagePullPolicy=IfNotPresent — avoids re-pulling the ~250MB image on every build. + */ + private readonly kanikoImage = + process.env.KANIKO_IMAGE || 'gcr.io/kaniko-project/executor:v1.23.2'; constructor( private configService: ConfigService, @@ -332,6 +338,7 @@ export class BuildService { initContainers.push({ name: 'unzip-source', image: 'alpine:3.19', + imagePullPolicy: 'IfNotPresent', command: ['sh', '-c', ` apk add --no-cache unzip tar gzip && cp /workspace/Dockerfile /workspace-out/Dockerfile && @@ -390,6 +397,7 @@ export class BuildService { initContainers.push({ name: 'git-clone', image: 'alpine/git:2.43.0', + imagePullPolicy: 'IfNotPresent', command: ['sh', '-c', ` echo ">>> Cloning branch '${branch}' from ${app.gitUrl}" && git clone --depth 1 --branch ${branch} ${cloneUrl} /workspace-out/source && @@ -417,6 +425,7 @@ export class BuildService { initContainers.push({ name: 'prepare-workspace', image: 'alpine:3.19', + imagePullPolicy: 'IfNotPresent', command: ['sh', '-c', ` mkdir -p /workspace-out/source && cp /dockerfile/Dockerfile /workspace-out/Dockerfile && @@ -447,7 +456,8 @@ export class BuildService { containers: [ { name: 'kaniko', - image: 'gcr.io/kaniko-project/executor:latest', + image: this.kanikoImage, + imagePullPolicy: 'IfNotPresent', args: kanikoArgs, volumeMounts: kanikoVolumeMounts, resources: { @@ -519,17 +529,6 @@ export class BuildService { } } - /** - * Poll the helper pod until netcat finishes writing and reports the expected byte count. - */ - private async waitForRemoteUploadDone( - kubeconfig: string, - namespace: string, - podName: string, - expectedSize: number, - deploymentId?: string, - ): Promise { /* unused — kept for API compat */ } - /** * Upload a local file to the helper pod using kubectl cp with progress tracking. * kubectl cp uses tar over the k8s exec API — reliable for any file size. @@ -668,6 +667,7 @@ export class BuildService { containers: [{ name: 'helper', image: 'alpine:3.19', + imagePullPolicy: 'IfNotPresent', command: ['sh', '-c', 'sleep 3600'], volumeMounts: [{ name: 'source', mountPath: '/data' }], resources: { @@ -1467,6 +1467,30 @@ CMD ["sh", "-c", "DLL=$(find . -maxdepth 1 -name '*.dll' ! -name '*.deps.dll' ! `; } + /** + * Whether a K8s API error is a transient connectivity/availability blip that + * should be retried rather than failing the operation. Covers socket-level + * errors, request timeouts (incl. the apiserver "request did not complete + * within the allotted timeout" message), DNS hiccups and 5xx/429 responses. + */ + private isTransientK8sError(err: any): boolean { + const statusCode = err?.statusCode ?? err?.response?.statusCode ?? err?.body?.code; + if (typeof statusCode === 'number' && (statusCode >= 500 || statusCode === 429)) { + return true; + } + const haystack = [ + err?.code, + err?.message, + err?.body?.message, + err?.cause?.code, + ] + .filter(Boolean) + .join(' '); + return /ECONNRESET|ECONNREFUSED|ETIMEDOUT|ESOCKETTIMEDOUT|EPIPE|EAI_AGAIN|ENOTFOUND|ENETUNREACH|socket hang up|timed? ?out|allotted timeout|did not complete|Client network socket disconnected/i.test( + haystack, + ); + } + private async waitForJobCompletion( batchApi: k8s.BatchV1Api, coreApi: k8s.CoreV1Api, @@ -1493,9 +1517,12 @@ CMD ["sh", "-c", "DLL=$(find . -maxdepth 1 -name '*.dll' ! -name '*.deps.dll' ! try { job = await batchApi.readNamespacedJob(jobName, namespace); } catch (pollErr: any) { - const code = pollErr?.code || pollErr?.message || ''; - if (/ECONNRESET|ECONNREFUSED|ETIMEDOUT|socket hang up/i.test(String(code))) { - this.logger.warn(`Transient K8s API error polling job ${jobName}: ${code} — retrying in 5s`); + // The Kaniko job keeps running independently of these status polls. + // A single API blip (timeout, reset, 5xx, DNS) must NOT abort a build + // that is still progressing — just retry on the next poll tick. + if (this.isTransientK8sError(pollErr)) { + const detail = pollErr?.code || pollErr?.message || pollErr?.statusCode || 'unknown'; + this.logger.warn(`Transient K8s API error polling job ${jobName}: ${detail} — retrying in 5s`); await new Promise(r => setTimeout(r, 5000)); continue; } diff --git a/backend/src/clusters/cluster-tools.service.ts b/backend/src/clusters/cluster-tools.service.ts index 43b663c..c25a414 100644 --- a/backend/src/clusters/cluster-tools.service.ts +++ b/backend/src/clusters/cluster-tools.service.ts @@ -25,7 +25,11 @@ const CERT_MANAGER_CHART = 'jetstack/cert-manager'; const CLUSTER_ISSUER_NAME = 'letsencrypt-prod'; const ACME_PROD_SERVER = 'https://acme-v02.api.letsencrypt.org/directory'; -const DEFAULT_INGRESS_CLASS = 'nginx'; +// k3s ships Traefik by default; match the app Ingress class (INGRESS_CLASS) +// so the ACME HTTP-01 solver Ingress is actually served by the controller. +const DEFAULT_INGRESS_CLASS = (process.env.INGRESS_CLASS || 'traefik') + .trim() + .toLowerCase(); const ISSUER_GROUP = 'cert-manager.io'; const ISSUER_VERSION = 'v1'; diff --git a/backend/src/clusters/clusters.service.ts b/backend/src/clusters/clusters.service.ts index 4d2b751..738c978 100644 --- a/backend/src/clusters/clusters.service.ts +++ b/backend/src/clusters/clusters.service.ts @@ -300,12 +300,15 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { } const candidates = await this.getCachedHealthyClusters(pool || undefined, options.excludeClusterIds || []); - const appCounts = await this.getAppCounts(candidates.map((cluster) => cluster.id)); + const clusterIds = candidates.map((cluster) => cluster.id); + const appCounts = await this.getAppCounts(clusterIds); + const reservations = await this.getInFlightReservations(clusterIds, options.applicationId); const candidateScores: Record[] = []; const rejectionReasons: Record[] = []; for (const cluster of candidates) { - const rejection = this.getClusterRejectionReason(cluster, estimatedRequest); + const reserved = reservations.get(cluster.id) || { cpuMillicores: 0, memoryMi: 0, pods: 0 }; + const rejection = this.getClusterRejectionReason(cluster, estimatedRequest, reserved); if (rejection) { rejectionReasons.push({ clusterId: cluster.id, clusterName: cluster.name, reason: rejection }); continue; @@ -313,8 +316,8 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { const strategy = this.resolveStrategy(pool?.strategy); const appCount = appCounts.get(cluster.id) || 0; - const score = this.scoreCluster(cluster, estimatedRequest, appCount, strategy, (dto as any).region); - const resourceMetrics = this.getResourceMetrics(cluster, estimatedRequest, appCount); + const score = this.scoreCluster(cluster, estimatedRequest, appCount, strategy, (dto as any).region, reserved); + const resourceMetrics = this.getResourceMetrics(cluster, estimatedRequest, appCount, reserved); candidateScores.push({ clusterId: cluster.id, clusterName: cluster.name, @@ -382,11 +385,13 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { message: string, ): Promise { if (clusterId) { - await this.clustersRepository.update(clusterId, { - healthStatus: 'degraded', - healthMessage: message, - lastHealthCheckedAt: new Date(), - }); + // A single deployment failure (quota, image pull, app bug, transient + // scheduling pressure) does NOT mean the cluster is unhealthy — flipping + // its healthStatus to 'degraded' would poison it for every other app and, + // when there's only one cluster, cause "no healthy cluster has capacity" + // for the next deployment. The fallback loop already excludes this cluster + // for *this* app via excludeClusterIds, so here we only invalidate the + // cached capacity snapshot to force a fresh read on the next allocation. this.clusterHealthCache.delete(clusterId); } @@ -1023,6 +1028,13 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { const { username, password } = this.registryService.getRegistryCredentials(); const dsName = 'cloudhost-k3s-registry-mirrors'; + // The mirror endpoint must be reachable by the node's containerd, which does + // NOT use cluster DNS — so we point it at the registry NodePort on loopback + // (http://127.0.0.1:) instead of the in-cluster service DNS name. + // Otherwise image pulls break whenever node-level resolution of + // *.svc.cluster.local is unavailable (e.g. right after a node restart). + const registryNodePort = 30500; + const nodePortHost = `127.0.0.1:${registryNodePort}`; const configureScript = [ 'set -e', 'REG=/host/etc/rancher/k3s/registries.yaml', @@ -1031,12 +1043,16 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { 'mirrors:', ` "${registryUrl}":`, ' endpoint:', - ` - "http://${registryUrl}"`, + ` - "http://${nodePortHost}"`, 'configs:', ` "${registryUrl}":`, ' auth:', ` username: ${JSON.stringify(username)}`, ` password: ${JSON.stringify(password)}`, + ` "${nodePortHost}":`, + ' auth:', + ` username: ${JSON.stringify(username)}`, + ` password: ${JSON.stringify(password)}`, 'EOFREG', 'if [ ! -f "$REG" ] || ! cmp -s /tmp/cloudhost-registries.yaml "$REG"; then', ' cp /tmp/cloudhost-registries.yaml "$REG"', @@ -1212,7 +1228,11 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { } } - private getClusterRejectionReason(cluster: Cluster, estimatedRequest: Record): string | null { + private getClusterRejectionReason( + cluster: Cluster, + estimatedRequest: Record, + reserved: { cpuMillicores: number; memoryMi: number; pods: number } = { cpuMillicores: 0, memoryMi: 0, pods: 0 }, + ): string | null { if (cluster.status !== ClusterStatus.ACTIVE) { return `status=${cluster.status}`; } @@ -1221,7 +1241,7 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { return `health=${cluster.healthStatus}`; } - const metrics = this.getResourceMetrics(cluster, estimatedRequest, 0); + const metrics = this.getResourceMetrics(cluster, estimatedRequest, 0, reserved); const { available, utilization } = metrics; if (available.cpuMillicores > 0 && available.cpuMillicores < estimatedRequest.cpuMillicores) { @@ -1252,8 +1272,9 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { appCount: number, strategy: PoolStrategy, desiredRegion?: string, + reserved: { cpuMillicores: number; memoryMi: number; pods: number } = { cpuMillicores: 0, memoryMi: 0, pods: 0 }, ): number { - const metrics = this.getResourceMetrics(cluster, estimatedRequest, appCount); + const metrics = this.getResourceMetrics(cluster, estimatedRequest, appCount, reserved); const capacityScore = metrics.capacityScore; const appPenalty = Math.min(appCount, 100) * 0.75; @@ -1281,6 +1302,7 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { cluster: Cluster, estimatedRequest: Record, appCount: number, + reserved: { cpuMillicores: number; memoryMi: number; pods: number } = { cpuMillicores: 0, memoryMi: 0, pods: 0 }, ): { available: { cpuMillicores: number; memoryMi: number; storageMi: number; pods: number }; utilization: { cpu: number; memory: number; storage: number; pods: number; appPressure: number; average: number }; @@ -1297,17 +1319,23 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { const nodeCount = Number(resources.nodeCount || 0); const podCapacity = Number(resources.podCapacity || nodeCount * 110 || 0); + // Include capacity reserved by in-flight deployments whose pods don't exist + // on the cluster yet, so concurrent allocations don't over-commit the node. + const cpuUsed = cpuRequested + (reserved.cpuMillicores || 0); + const memoryUsed = memoryRequested + (reserved.memoryMi || 0); + const podsUsed = podCount + (reserved.pods || 0); + const available = { - cpuMillicores: Math.max(cpuCapacity - cpuRequested, 0), - memoryMi: Math.max(memoryCapacity - memoryRequested, 0), + cpuMillicores: Math.max(cpuCapacity - cpuUsed, 0), + memoryMi: Math.max(memoryCapacity - memoryUsed, 0), storageMi: storageCapacity > 0 ? Math.max(storageCapacity - storageUsed, 0) : 0, - pods: podCapacity > 0 ? Math.max(podCapacity - podCount, 0) : 0, + pods: podCapacity > 0 ? Math.max(podCapacity - podsUsed, 0) : 0, }; const utilization = { - cpu: this.utilizationRatio(cpuRequested, cpuCapacity), - memory: this.utilizationRatio(memoryRequested, memoryCapacity), + cpu: this.utilizationRatio(cpuUsed, cpuCapacity), + memory: this.utilizationRatio(memoryUsed, memoryCapacity), storage: this.utilizationRatio(storageUsed, storageCapacity), - pods: this.utilizationRatio(podCount, podCapacity), + pods: this.utilizationRatio(podsUsed, podCapacity), appPressure: Math.min(appCount / 100, 1), average: 0, }; @@ -1377,6 +1405,60 @@ export class ClustersService implements OnModuleInit, OnModuleDestroy { return new Map(rows.map((row) => [row.clusterId, parseInt(row.count, 10)])); } + /** + * Capacity that has been allocated to deployments which are still building but + * whose pods don't exist on the cluster yet (status pending/building). The + * cluster's `availableResources` snapshot only reflects pods that already exist + * (and is cached up to cacheTtlMs), so without this, two deployments started + * within the same window both read the same stale capacity and can both be + * placed on a cluster that only has room for one — the second then can't be + * scheduled and hangs. We subtract these in-flight requests during allocation. + * + * @param excludeApplicationId skip an app's own in-flight deployment (used on + * fallback re-allocation so the app doesn't reserve capacity against itself). + */ + private async getInFlightReservations( + clusterIds: string[], + excludeApplicationId?: string, + ): Promise> { + const result = new Map(); + if (clusterIds.length === 0) { + return result; + } + + const rows: any[] = await this.dataSource.query(` + SELECT a.id as "applicationId", + a."clusterId" as "clusterId", + a."cpuRequest" as "cpuRequest", + a."memoryRequest" as "memoryRequest", + a.replicas as "replicas", + a."databaseType" as "databaseType", + a."enableRedis" as "enableRedis", + a."enableRabbitmq" as "enableRabbitmq", + a."appStorageSize" as "appStorageSize", + a."dbStorageSize" as "dbStorageSize", + a."optionalServiceResources" as "optionalServiceResources" + FROM deployments d + JOIN applications a ON a.id = d."applicationId" + WHERE a."clusterId" = ANY($1) + AND d.status IN ('pending', 'building') + `, [clusterIds]); + + for (const row of rows) { + if (excludeApplicationId && row.applicationId === excludeApplicationId) { + continue; + } + const est = this.estimateApplicationRequest(row as CreateApplicationDto); + const current = result.get(row.clusterId) || { cpuMillicores: 0, memoryMi: 0, pods: 0 }; + current.cpuMillicores += est.cpuMillicores; + current.memoryMi += est.memoryMi; + current.pods += est.podEstimate; + result.set(row.clusterId, current); + } + + return result; + } + private parseStorageToMi(storage: string): number { if (!storage) return 0; if (storage.endsWith('Ki')) return parseFloat(storage) / 1024; diff --git a/backend/src/config/configuration.ts b/backend/src/config/configuration.ts index 3491f98..365691c 100644 --- a/backend/src/config/configuration.ts +++ b/backend/src/config/configuration.ts @@ -28,6 +28,13 @@ function stripLeadingSubdomain(hostname: string): string { } function resolvePreviewRootDomainFromEnv(): string { + // Explicit override takes priority — the base domain used for preview URLs + // (e.g. PREVIEW_BASE_DOMAIN=3fase.ir) must be configurable, never hardcoded. + const explicit = process.env.PREVIEW_BASE_DOMAIN; + if (explicit && explicit.trim()) { + return explicit.trim().toLowerCase(); + } + const frontendUrl = process.env.FRONTEND_URL; if (frontendUrl) { try { @@ -108,6 +115,12 @@ export default () => ({ platform: { domain: resolvePlatformDomainFromEnv(), previewRootDomain: resolvePreviewRootDomainFromEnv(), + /** + * Ingress controller class used for app Ingress objects and the ACME + * HTTP-01 solver. k3s ships Traefik by default, so we default to "traefik". + * Set INGRESS_CLASS=nginx for clusters running ingress-nginx instead. + */ + ingressClass: (process.env.INGRESS_CLASS || 'traefik').trim().toLowerCase(), uploadDir: process.env.UPLOAD_DIR || './uploads', /** StorageClass for new PVCs; must support allowVolumeExpansion for disk resize */ storageClass: process.env.PLATFORM_STORAGE_CLASS || 'cloudhost-expandable', diff --git a/backend/src/deployments/deployments.service.ts b/backend/src/deployments/deployments.service.ts index 9bb0df0..a458772 100644 --- a/backend/src/deployments/deployments.service.ts +++ b/backend/src/deployments/deployments.service.ts @@ -30,13 +30,29 @@ export class DeploymentsService { ) {} /** - * Stable "number" part for preview host, derived from deployment id. - * We keep it numeric to match the "" requirement. + * Random 7-digit suffix for the preview host: -<7-digit>.. + * Generated once per application (see resolvePreviewNumber) and persisted. */ - private computePreviewNumberFromDeploymentId(deploymentId: string): string { - const hashHex = crypto.createHash('sha256').update(deploymentId).digest('hex'); - const num = parseInt(hashHex.slice(0, 8), 16) % 1_000_000; // 0..999999 - return String(num).padStart(6, '0'); + private generatePreviewNumber(): string { + return String(crypto.randomInt(1_000_000, 10_000_000)); + } + + /** + * Returns a STABLE preview number for an application: reuse the one already + * assigned to a previous deployment so the public preview URL never changes + * across redeploys (otherwise old links 404). Only generates a new number the + * first time the app is deployed. Apps with a custom domain get no preview host. + */ + private async resolvePreviewNumber(applicationId: string): Promise { + const existing = await this.deploymentsRepository + .createQueryBuilder('d') + .select('d.previewSubdomain', 'previewSubdomain') + .where('d.applicationId = :applicationId', { applicationId }) + .andWhere('d.previewSubdomain IS NOT NULL') + .orderBy('d.createdAt', 'DESC') + .limit(1) + .getRawOne<{ previewSubdomain: string }>(); + return existing?.previewSubdomain || this.generatePreviewNumber(); } async triggerDeployment(applicationId: string, userId: string): Promise { @@ -56,7 +72,7 @@ export class DeploymentsService { // Fill deterministic preview number after we have the deployment id. let previewSubdomain: string | null = null; if (!app.customDomain) { - previewSubdomain = this.computePreviewNumberFromDeploymentId(saved.id); + previewSubdomain = await this.resolvePreviewNumber(app.id); await this.deploymentsRepository.update(saved.id, { previewSubdomain }); saved.previewSubdomain = previewSubdomain; } @@ -657,7 +673,7 @@ export class DeploymentsService { let previewSubdomain: string | null = null; if (!app.customDomain) { - previewSubdomain = this.computePreviewNumberFromDeploymentId(saved.id); + previewSubdomain = await this.resolvePreviewNumber(app.id); await this.deploymentsRepository.update(saved.id, { previewSubdomain }); saved.previewSubdomain = previewSubdomain; } diff --git a/backend/src/kubernetes/kubernetes.service.ts b/backend/src/kubernetes/kubernetes.service.ts index 18576fa..697ecd5 100644 --- a/backend/src/kubernetes/kubernetes.service.ts +++ b/backend/src/kubernetes/kubernetes.service.ts @@ -289,7 +289,7 @@ export class KubernetesService implements OnModuleInit { const namespacePrefix = app.userId.split('-')[0]; const previewHost = previewNumber && !app.customDomain - ? `${namespacePrefix}-${previewNumber}-preview.${previewRootDomain}` + ? `${namespacePrefix}-${previewNumber}.${previewRootDomain}` : ''; const pullRegistryUrl = this.registryService.getRegistryUrl(); const isWordPress = app.runtime === AppRuntime.WORDPRESS; @@ -317,6 +317,7 @@ export class KubernetesService implements OnModuleInit { envVars: this.resolveEnvVars(app), ingress: { enabled: true, + className: this.configService.get('platform.ingressClass') || 'traefik', subdomain: app.subdomain || app.name, domain: domain, clusterIssuer: 'letsencrypt-prod', @@ -931,8 +932,10 @@ export class KubernetesService implements OnModuleInit { // Main application container const appContainer: any = { name: ctx.appName, + // Image tags are unique per build (name:timestamp) and immutable, so + // IfNotPresent is correct and avoids re-pulling on every restart/scale-up. image: ctx.image, - imagePullPolicy: 'Always', + imagePullPolicy: 'IfNotPresent', ports: [{ containerPort: ctx.port }], envFrom, env: extraEnv, @@ -1402,7 +1405,11 @@ export class KubernetesService implements OnModuleInit { }, }, ]; - const tlsHosts = [host]; + // Only PUBLIC, real-TLD hosts may go into the TLS block. The internal + // `${subdomain}.${domain}` host (e.g. *.apps.cloudhost.local) is not a valid + // public suffix — including it makes Let's Encrypt reject the whole order, + // which would also block the cert for the legitimate preview/custom domains. + const tlsHosts: string[] = []; if (customDomain) { rules.push({ @@ -1424,7 +1431,7 @@ export class KubernetesService implements OnModuleInit { const namespacePrefix = ctx.ownerId.split('-')[0]; const previewHost = previewNumber && !customDomain - ? `${namespacePrefix}-${previewNumber}-preview.${previewRootDomain}` + ? `${namespacePrefix}-${previewNumber}.${previewRootDomain}` : ''; if (previewHost) { rules.push({ @@ -1442,20 +1449,28 @@ export class KubernetesService implements OnModuleInit { tlsHosts.push(previewHost); } + const ingressClass = + this.configService.get('platform.ingressClass') || 'traefik'; + // Request a managed cert only when we actually have a public host to issue for. + const annotations: Record = {}; + const tls: k8s.V1IngressTLS[] = []; + if (tlsHosts.length > 0) { + annotations['cert-manager.io/cluster-issuer'] = 'letsencrypt-prod'; + tls.push({ hosts: tlsHosts, secretName: `${ctx.appName}-tls` }); + } + const ingress: k8s.V1Ingress = { apiVersion: 'networking.k8s.io/v1', kind: 'Ingress', metadata: { name: ctx.appName, namespace: ctx.namespace, - annotations: { - 'cert-manager.io/cluster-issuer': 'letsencrypt-prod', - }, + annotations, }, spec: { - ingressClassName: 'nginx', + ingressClassName: ingressClass, rules, - tls: [{ hosts: tlsHosts, secretName: `${ctx.appName}-tls` }], + ...(tls.length > 0 ? { tls } : {}), }, }; @@ -2780,7 +2795,7 @@ export class KubernetesService implements OnModuleInit { if (verifiedCustomDomain) { ingressUrl = `https://${verifiedCustomDomain}`; } else if (previewNumber) { - ingressUrl = `https://${namespacePrefix}-${previewNumber}-preview.${previewRootDomain}`; + ingressUrl = `https://${namespacePrefix}-${previewNumber}.${previewRootDomain}`; } return {