Compare commits

..
2 Commits
Author SHA1 Message Date
Celes Renata 2d40d70975 ci: remove remaining ghcr-credentials from inttest seed/minio pod overrides
ci/woodpecker/push/woodpecker Pipeline failed
Build and Push / lint-and-test (push) Has been cancelled
Build and Push / build-services (map[cmd:python -m services.adapters.broker_adapter name:broker-adapter]) (push) Has been cancelled
Build and Push / build-services (map[cmd:python -m services.aggregation.worker name:aggregation]) (push) Has been cancelled
Build and Push / build-services (map[cmd:python -m services.extractor.worker name:extractor]) (push) Has been cancelled
Build and Push / build-services (map[cmd:python -m services.ingestion.worker name:ingestion]) (push) Has been cancelled
Build and Push / build-services (map[cmd:python -m services.lake_publisher.worker name:lake-publisher]) (push) Has been cancelled
Build and Push / build-services (map[cmd:python -m services.parser.worker name:parser]) (push) Has been cancelled
Build and Push / build-services (map[cmd:python -m services.recommendation.worker name:recommendation]) (push) Has been cancelled
Build and Push / build-services (map[cmd:python -m services.scheduler.app name:scheduler]) (push) Has been cancelled
Build and Push / build-services (map[cmd:uvicorn services.api.app:app --host 0.0.0.0 --port 8000 name:query-api]) (push) Has been cancelled
Build and Push / build-services (map[cmd:uvicorn services.risk.app:app --host 0.0.0.0 --port 8000 name:risk]) (push) Has been cancelled
Build and Push / build-services (map[cmd:uvicorn services.symbol_registry.app:app --host 0.0.0.0 --port 8000 name:symbol-registry]) (push) Has been cancelled
Build and Push / build-services (map[cmd:uvicorn services.trading.app:app --host 0.0.0.0 --port 8000 name:trading-engine]) (push) Has been cancelled
Build and Push / build-dashboard (push) Has been cancelled
Build and Push / build-superset (push) Has been cancelled
Build and Push / integration-test (push) Has been cancelled
2026-04-19 06:45:46 +00:00
Celes Renata ebafe795c1 fix: bump seed pod timeout to 5m and add debug diagnostics on pipeline failures 2026-04-19 06:34:58 +00:00
5 changed files with 376 additions and 5 deletions
+44 -3
View File
@@ -88,6 +88,29 @@ stage_fail() {
log "✗ Stage: $name FAILED after ${STAGE_DURATION[$name]}s" log "✗ Stage: $name FAILED after ${STAGE_DURATION[$name]}s"
} }
debug_pod_failure() {
local pod_name="$1"
local label="${2:-}"
log "─── DEBUG: pod failure diagnostics ───"
if [ -n "$label" ]; then
# Find pod by label selector
local found_pod
found_pod=$(kubectl get pods -n "$NAMESPACE" -l "$label" -o jsonpath='{.items[0].metadata.name}' 2>/dev/null || true)
if [ -n "$found_pod" ]; then
pod_name="$found_pod"
fi
fi
log "Pod describe ($pod_name):"
kubectl describe pod "$pod_name" -n "$NAMESPACE" 2>&1 | tail -40 || true
log "Pod logs ($pod_name):"
kubectl logs "$pod_name" -n "$NAMESPACE" --tail=60 2>&1 || true
log "Pod status:"
kubectl get pod "$pod_name" -n "$NAMESPACE" -o wide 2>&1 || true
log "Recent events in namespace:"
kubectl get events -n "$NAMESPACE" --sort-by='.lastTimestamp' 2>&1 | tail -20 || true
log "─── END DEBUG ───"
}
# ── Parse CLI args ─────────────────────────────────────────────────────────── # ── Parse CLI args ───────────────────────────────────────────────────────────
while [[ $# -gt 0 ]]; do while [[ $# -gt 0 ]]; do
case $1 in case $1 in
@@ -268,6 +291,7 @@ envsubst < "$REPO_ROOT/infra/inttest/minio.yaml" | kubectl apply -n "$NAMESPACE"
log "Waiting for postgres readiness ..." log "Waiting for postgres readiness ..."
if ! kubectl wait --for=condition=ready pod -l app=postgres -n "$NAMESPACE" --timeout=120s; then if ! kubectl wait --for=condition=ready pod -l app=postgres -n "$NAMESPACE" --timeout=120s; then
log "FATAL: PostgreSQL did not become ready" log "FATAL: PostgreSQL did not become ready"
debug_pod_failure "postgres" "app=postgres"
stage_fail "infra_deploy" stage_fail "infra_deploy"
PIPELINE_EXIT_CODE=2 PIPELINE_EXIT_CODE=2
exit 2 exit 2
@@ -276,6 +300,7 @@ fi
log "Waiting for redis readiness ..." log "Waiting for redis readiness ..."
if ! kubectl wait --for=condition=ready pod -l app=redis -n "$NAMESPACE" --timeout=60s; then if ! kubectl wait --for=condition=ready pod -l app=redis -n "$NAMESPACE" --timeout=60s; then
log "FATAL: Redis did not become ready" log "FATAL: Redis did not become ready"
debug_pod_failure "redis" "app=redis"
stage_fail "infra_deploy" stage_fail "infra_deploy"
PIPELINE_EXIT_CODE=2 PIPELINE_EXIT_CODE=2
exit 2 exit 2
@@ -284,13 +309,19 @@ fi
log "Waiting for minio readiness ..." log "Waiting for minio readiness ..."
if ! kubectl wait --for=condition=ready pod -l app=minio -n "$NAMESPACE" --timeout=60s; then if ! kubectl wait --for=condition=ready pod -l app=minio -n "$NAMESPACE" --timeout=60s; then
log "FATAL: MinIO did not become ready" log "FATAL: MinIO did not become ready"
debug_pod_failure "minio" "app=minio"
stage_fail "infra_deploy" stage_fail "infra_deploy"
PIPELINE_EXIT_CODE=2 PIPELINE_EXIT_CODE=2
exit 2 exit 2
fi fi
log "Waiting for minio-bucket-init job ..." log "Waiting for minio-bucket-init job ..."
kubectl wait --for=condition=complete job/minio-bucket-init -n "$NAMESPACE" --timeout=60s || true if ! kubectl wait --for=condition=complete job/minio-bucket-init -n "$NAMESPACE" --timeout=120s; then
log "WARNING: minio-bucket-init job did not complete within 120s"
log "Bucket-init pod logs:"
kubectl logs -l app=minio-bucket-init -n "$NAMESPACE" --tail=30 2>&1 || true
kubectl describe job/minio-bucket-init -n "$NAMESPACE" 2>&1 | tail -20 || true
fi
stage_end "infra_deploy" "ok" stage_end "infra_deploy" "ok"
@@ -307,11 +338,12 @@ if ! kubectl run seed-sandbox \
--restart=Never \ --restart=Never \
--rm \ --rm \
--attach \ --attach \
--pod-running-timeout=5m \
--namespace="$NAMESPACE" \ --namespace="$NAMESPACE" \
--image-pull-policy=Always \ --image-pull-policy=Always \
--overrides='{ --overrides='{
"spec": { "spec": {
"imagePullSecrets": [{"name": "ghcr-credentials"}],
"securityContext": {"runAsNonRoot": true, "runAsUser": 1000, "runAsGroup": 1000} "securityContext": {"runAsNonRoot": true, "runAsUser": 1000, "runAsGroup": 1000}
} }
}' \ }' \
@@ -326,6 +358,7 @@ if ! kubectl run seed-sandbox \
--env="MINIO_SECRET_KEY=minioadmin" \ --env="MINIO_SECRET_KEY=minioadmin" \
--command -- python -m tests.integration.seed_sandbox; then --command -- python -m tests.integration.seed_sandbox; then
log "FATAL: Database seed failed" log "FATAL: Database seed failed"
debug_pod_failure "seed-sandbox" "run=seed-sandbox"
stage_fail "seed_data" stage_fail "seed_data"
PIPELINE_EXIT_CODE=2 PIPELINE_EXIT_CODE=2
exit 2 exit 2
@@ -337,11 +370,12 @@ if ! kubectl run seed-minio \
--restart=Never \ --restart=Never \
--rm \ --rm \
--attach \ --attach \
--pod-running-timeout=5m \
--namespace="$NAMESPACE" \ --namespace="$NAMESPACE" \
--image-pull-policy=Always \ --image-pull-policy=Always \
--overrides='{ --overrides='{
"spec": { "spec": {
"imagePullSecrets": [{"name": "ghcr-credentials"}],
"securityContext": {"runAsNonRoot": true, "runAsUser": 1000, "runAsGroup": 1000} "securityContext": {"runAsNonRoot": true, "runAsUser": 1000, "runAsGroup": 1000}
} }
}' \ }' \
@@ -351,6 +385,7 @@ if ! kubectl run seed-minio \
--env="MINIO_SECRET_KEY=minioadmin" \ --env="MINIO_SECRET_KEY=minioadmin" \
--command -- python -m tests.integration.seed_minio; then --command -- python -m tests.integration.seed_minio; then
log "FATAL: MinIO seed failed" log "FATAL: MinIO seed failed"
debug_pod_failure "seed-minio" "run=seed-minio"
stage_fail "seed_data" stage_fail "seed_data"
PIPELINE_EXIT_CODE=2 PIPELINE_EXIT_CODE=2
exit 2 exit 2
@@ -371,6 +406,11 @@ envsubst < "$REPO_ROOT/infra/inttest/services.yaml" \
log "Waiting for all API services to become ready ..." log "Waiting for all API services to become ready ..."
if ! kubectl wait --for=condition=ready pod -l tier=api -n "$NAMESPACE" --timeout=120s; then if ! kubectl wait --for=condition=ready pod -l tier=api -n "$NAMESPACE" --timeout=120s; then
log "FATAL: API services did not become ready" log "FATAL: API services did not become ready"
log "Pod statuses:"
kubectl get pods -n "$NAMESPACE" -l tier=api -o wide 2>&1 || true
for pod in $(kubectl get pods -n "$NAMESPACE" -l tier=api --no-headers -o custom-columns=':metadata.name' 2>/dev/null); do
debug_pod_failure "$pod"
done
stage_fail "service_deploy" stage_fail "service_deploy"
PIPELINE_EXIT_CODE=2 PIPELINE_EXIT_CODE=2
exit 2 exit 2
@@ -398,6 +438,7 @@ else
if kubectl wait --for=condition=failed job/inttest-runner -n "$NAMESPACE" --timeout=5s 2>/dev/null; then if kubectl wait --for=condition=failed job/inttest-runner -n "$NAMESPACE" --timeout=5s 2>/dev/null; then
log "Test runner job reported failure" log "Test runner job reported failure"
fi fi
debug_pod_failure "inttest-runner" "app=inttest-runner"
stage_fail "integration_tests" stage_fail "integration_tests"
PIPELINE_EXIT_CODE=1 PIPELINE_EXIT_CODE=1
fi fi
+86
View File
@@ -0,0 +1,86 @@
# Harbor PersistentVolumeClaims — bind to NFS PVs
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: harbor-registry-pvc
namespace: harbor-service
labels:
app: harbor
component: registry
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 100Gi
storageClassName: ""
volumeName: harbor-registry-pv
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: harbor-jobservice-pvc
namespace: harbor-service
labels:
app: harbor
component: jobservice
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 2Gi
storageClassName: ""
volumeName: harbor-jobservice-pv
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: harbor-database-pvc
namespace: harbor-service
labels:
app: harbor
component: database
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 5Gi
storageClassName: ""
volumeName: harbor-database-pv
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: harbor-redis-pvc
namespace: harbor-service
labels:
app: harbor
component: redis
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 2Gi
storageClassName: ""
volumeName: harbor-redis-pv
---
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: harbor-trivy-pvc
namespace: harbor-service
labels:
app: harbor
component: trivy
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 5Gi
storageClassName: ""
volumeName: harbor-trivy-pv
+100
View File
@@ -0,0 +1,100 @@
# Harbor Helm values — Stonks Oracle registry
# Domain: registry.celestium.life
# Ingress: Traefik with cert-manager (letsencrypt-prod)
# Storage: NFS PVs on 192.168.42.8
expose:
type: ingress
tls:
enabled: true
certSource: secret
secret:
secretName: harbor-tls
ingress:
hosts:
core: registry.celestium.life
controller: default
className: traefik
annotations:
cert-manager.io/cluster-issuer: letsencrypt-prod
traefik.ingress.kubernetes.io/router.entrypoints: websecure
ingress.kubernetes.io/ssl-redirect: "true"
ingress.kubernetes.io/proxy-body-size: "0"
externalURL: https://registry.celestium.life
# Initial admin password — change after first login
harborAdminPassword: "St0nks0racl3!"
# Use internal database and redis (bundled with Harbor)
database:
type: internal
redis:
type: internal
persistence:
enabled: true
resourcePolicy: "keep"
persistentVolumeClaim:
registry:
existingClaim: harbor-registry-pvc
size: 100Gi
jobservice:
jobLog:
existingClaim: harbor-jobservice-pvc
size: 2Gi
database:
existingClaim: harbor-database-pvc
size: 5Gi
redis:
existingClaim: harbor-redis-pvc
size: 2Gi
trivy:
existingClaim: harbor-trivy-pvc
size: 5Gi
# Trivy vulnerability scanner
trivy:
enabled: true
# Metrics for Prometheus (optional, enable if you have monitoring)
metrics:
enabled: false
# Resource limits — conservative for a 4-node cluster
core:
resources:
requests:
cpu: 100m
memory: 256Mi
limits:
cpu: 1000m
memory: 512Mi
jobservice:
resources:
requests:
cpu: 100m
memory: 256Mi
limits:
cpu: 500m
memory: 512Mi
registry:
resources:
requests:
cpu: 100m
memory: 256Mi
limits:
cpu: 1000m
memory: 1Gi
portal:
resources:
requests:
cpu: 50m
memory: 128Mi
limits:
cpu: 500m
memory: 256Mi
+87
View File
@@ -0,0 +1,87 @@
# Harbor NFS PersistentVolumes
# NFS path: nfs://192.168.42.8:/volume1/Kubernetes/harbor/data/<component>
---
apiVersion: v1
kind: PersistentVolume
metadata:
name: harbor-registry-pv
labels:
app: harbor
component: registry
spec:
capacity:
storage: 100Gi
accessModes:
- ReadWriteOnce
persistentVolumeReclaimPolicy: Retain
nfs:
server: 192.168.42.8
path: /volume1/Kubernetes/harbor/data/registry
---
apiVersion: v1
kind: PersistentVolume
metadata:
name: harbor-database-pv
labels:
app: harbor
component: database
spec:
capacity:
storage: 5Gi
accessModes:
- ReadWriteOnce
persistentVolumeReclaimPolicy: Retain
nfs:
server: 192.168.42.8
path: /volume1/Kubernetes/harbor/data/database
---
apiVersion: v1
kind: PersistentVolume
metadata:
name: harbor-redis-pv
labels:
app: harbor
component: redis
spec:
capacity:
storage: 2Gi
accessModes:
- ReadWriteOnce
persistentVolumeReclaimPolicy: Retain
nfs:
server: 192.168.42.8
path: /volume1/Kubernetes/harbor/data/redis
---
apiVersion: v1
kind: PersistentVolume
metadata:
name: harbor-jobservice-pv
labels:
app: harbor
component: jobservice
spec:
capacity:
storage: 2Gi
accessModes:
- ReadWriteOnce
persistentVolumeReclaimPolicy: Retain
nfs:
server: 192.168.42.8
path: /volume1/Kubernetes/harbor/data/jobservice
---
apiVersion: v1
kind: PersistentVolume
metadata:
name: harbor-trivy-pv
labels:
app: harbor
component: trivy
spec:
capacity:
storage: 5Gi
accessModes:
- ReadWriteOnce
persistentVolumeReclaimPolicy: Retain
nfs:
server: 192.168.42.8
path: /volume1/Kubernetes/harbor/data/trivy
+59 -2
View File
@@ -15,7 +15,7 @@ GITEA_API="http://10.1.1.12:30300/api/v1"
# 1. Create namespaces # 1. Create namespaces
# ------------------------------------------------------- # -------------------------------------------------------
echo "--- Step 1: Creating namespaces ---" echo "--- Step 1: Creating namespaces ---"
for ns in woodpecker argocd kargo stonks-beta stonks-paper; do for ns in woodpecker argocd kargo stonks-beta stonks-paper harbor-service; do
kubectl create namespace "$ns" --dry-run=client -o yaml | kubectl apply -f - kubectl create namespace "$ns" --dry-run=client -o yaml | kubectl apply -f -
echo " ✓ namespace/$ns" echo " ✓ namespace/$ns"
done done
@@ -27,7 +27,7 @@ echo ""
echo "--- Step 2: Proxy CA cert and Kyverno policies ---" echo "--- Step 2: Proxy CA cert and Kyverno policies ---"
CA_CERT_PATH="${SCRIPT_DIR}/home.crt" CA_CERT_PATH="${SCRIPT_DIR}/home.crt"
curl -sf http://192.168.42.1/home.crt -o "$CA_CERT_PATH" curl -sf http://192.168.42.1/home.crt -o "$CA_CERT_PATH"
for ns in woodpecker argocd kargo; do for ns in woodpecker argocd kargo harbor-service; do
if ! kubectl get configmap proxy-ca-cert -n "$ns" > /dev/null 2>&1; then if ! kubectl get configmap proxy-ca-cert -n "$ns" > /dev/null 2>&1; then
kubectl create configmap proxy-ca-cert --from-file=ca.crt="$CA_CERT_PATH" -n "$ns" kubectl create configmap proxy-ca-cert --from-file=ca.crt="$CA_CERT_PATH" -n "$ns"
echo " ✓ proxy-ca-cert created in $ns" echo " ✓ proxy-ca-cert created in $ns"
@@ -55,9 +55,65 @@ echo "--- Step 3: Applying NFS PersistentVolumes ---"
kubectl apply -f pvs/argocd-pv.yaml kubectl apply -f pvs/argocd-pv.yaml
kubectl apply -f pvs/kargo-pv.yaml kubectl apply -f pvs/kargo-pv.yaml
kubectl apply -f pvs/woodpecker-pv.yaml kubectl apply -f pvs/woodpecker-pv.yaml
kubectl apply -f pvs/harbor-pv.yaml
echo " ✓ PVs applied" echo " ✓ PVs applied"
echo "" echo ""
# -------------------------------------------------------
# 3b. Install Harbor container registry
# -------------------------------------------------------
echo "--- Step 3b: Installing Harbor ---"
kubectl create namespace harbor-service --dry-run=client -o yaml | kubectl apply -f -
# Remove old plain Docker Registry ingress (registry.celestium.life) if it exists
# Harbor will take over that domain
if kubectl get ingress registry-ingress -n git-server > /dev/null 2>&1; then
echo " Removing old registry ingress from git-server namespace..."
kubectl delete ingress registry-ingress -n git-server
echo " ✓ Old registry ingress removed"
fi
# Create NFS directories on the NAS (via a temporary pod)
echo " Ensuring NFS directories exist..."
ssh root@gremlin-1 "
mkdir -p /tmp/harbor-nfs-init
mount -t nfs 192.168.42.8:/volume1/Kubernetes/harbor /tmp/harbor-nfs-init 2>/dev/null || true
mkdir -p /tmp/harbor-nfs-init/data/registry
mkdir -p /tmp/harbor-nfs-init/data/database
mkdir -p /tmp/harbor-nfs-init/data/redis
mkdir -p /tmp/harbor-nfs-init/data/jobservice
mkdir -p /tmp/harbor-nfs-init/data/trivy
umount /tmp/harbor-nfs-init 2>/dev/null || true
rmdir /tmp/harbor-nfs-init 2>/dev/null || true
" 2>/dev/null || echo " ⚠ Could not create NFS dirs via SSH (non-fatal, they may already exist)"
# Apply PVCs
kubectl apply -f harbor/pvcs.yaml
echo " ✓ Harbor PVCs applied"
# Install/upgrade Harbor via Helm
helm repo add harbor https://helm.goharbor.io 2>/dev/null || true
helm repo update harbor 2>/dev/null || true
HARBOR_EXISTS=$(helm list -n harbor-service -q 2>/dev/null | grep -c harbor || true)
if [ "${HARBOR_EXISTS:-0}" -gt 0 ]; then
echo " Harbor already installed — upgrading..."
else
echo " Fresh Harbor install..."
fi
helm upgrade --install harbor harbor/harbor \
--namespace harbor-service \
--values harbor/values.yaml \
--timeout 10m \
--wait
echo " Waiting for Harbor core to be ready..."
kubectl wait --for=condition=ready pod -l app=harbor,component=core -n harbor-service --timeout=180s > /dev/null 2>&1 || true
echo " ✓ Harbor installed at https://registry.celestium.life"
echo " Default login: admin / St0nks0racl3!"
echo ""
# ------------------------------------------------------- # -------------------------------------------------------
# 4. Configure Gitea (admin user, repo, webhook config) # 4. Configure Gitea (admin user, repo, webhook config)
# ------------------------------------------------------- # -------------------------------------------------------
@@ -246,6 +302,7 @@ echo ""
echo "=== Pipeline Infrastructure Install Complete ===" echo "=== Pipeline Infrastructure Install Complete ==="
echo "" echo ""
echo "Endpoints:" echo "Endpoints:"
echo " Harbor: https://registry.celestium.life"
echo " Woodpecker CI: https://stonks-ci.celestium.life" echo " Woodpecker CI: https://stonks-ci.celestium.life"
echo " ArgoCD: https://stonks-argocd.celestium.life" echo " ArgoCD: https://stonks-argocd.celestium.life"
echo " Kargo: https://stonks-kargo.celestium.life" echo " Kargo: https://stonks-kargo.celestium.life"