`kubectl wait <kind> <name>` errors immediately with NotFound if the
resource doesn't exist yet — even with --for=condition or --for=jsonpath.
The redis test in CI run 25043350705 failed in 1 second for exactly this
reason: the redis-failover operator hadn't created the PVC by the time
the test waited for it. Previously the 3x retry on `Run E2E tests`
masked this race; with retry dropped, every such call is a flake risk.
Add a small `until kubectl get` existence backstop before each
kubectl wait, matching the pattern already established for HRs in
commit 66888c91.
33 backstops across 12 files:
bucket.bats — bucketclaims, bucketaccesses x2
clickhouse.bats — statefulset 0-0 (0-1 already covered)
harbor.bats — deploy x3, bucketclaims, bucketaccesses
kafka.bats — kafkas
mariadb.bats — statefulset, deployment
mongodb.bats — statefulset
openbao.bats — sts, pvc
postgres.bats — job.batch
qdrant.bats — sts, pvc
redis.bats — pvc, deploy, sts (the trigger)
vminstance.bats — dv, pvc, vm (with 120s for KubeVirt latency)
e2e-install-cozystack.bats — apiservices, sts/etcd, vmalert,
vmalertmanager, vlclusters, vmcluster,
clusters.postgresql.cnpg.io,
deploy/grafana-deployment, namespace
Same pattern, same 60s default timeout for the existence wait (120s for
nested-virt resources). Once the resource exists, the original wait
timeout takes over.
Run-kubernetes.sh has the same race shape on several waits (nfs, kamaji,
machinedeployment, etc.) — out of scope here; flagged for follow-up.
Signed-off-by: Myasnikov Daniil <myasnikovdaniil2001@gmail.com>
50 lines
2.3 KiB
Bash
50 lines
2.3 KiB
Bash
#!/usr/bin/env bats
|
|
|
|
@test "Create DB ClickHouse" {
|
|
name='test'
|
|
kubectl -n tenant-test delete clickhouse.apps.cozystack.io $name --ignore-not-found --timeout=2m || true
|
|
kubectl apply -f- <<EOF
|
|
apiVersion: apps.cozystack.io/v1alpha1
|
|
kind: ClickHouse
|
|
metadata:
|
|
name: $name
|
|
namespace: tenant-test
|
|
spec:
|
|
size: 10Gi
|
|
logStorageSize: 2Gi
|
|
shards: 1
|
|
replicas: 2
|
|
storageClass: ""
|
|
logTTL: 15
|
|
users:
|
|
testuser:
|
|
password: xai7Wepo
|
|
backup:
|
|
enabled: false
|
|
s3Region: us-east-1
|
|
s3Bucket: s3.example.org/clickhouse-backups
|
|
schedule: "0 2 * * *"
|
|
cleanupStrategy: "--keep-last=3 --keep-daily=3 --keep-within-weekly=1m"
|
|
s3AccessKey: oobaiRus9pah8PhohL1ThaeTa4UVa7gu
|
|
s3SecretKey: ju3eum4dekeich9ahM1te8waeGai0oog
|
|
resticPassword: ChaXoveekoh6eigh4siesheeda2quai0
|
|
clickhouseKeeper:
|
|
enabled: true
|
|
resourcesPreset: "micro"
|
|
size: "1Gi"
|
|
resources: {}
|
|
resourcesPreset: "nano"
|
|
EOF
|
|
# Wait for the operator to materialise the HelmRelease before kubectl wait
|
|
# kicks in (kubectl wait errors immediately if the object does not exist yet).
|
|
timeout 60 sh -ec "until kubectl -n tenant-test get hr clickhouse-$name >/dev/null 2>&1; do sleep 2; done"
|
|
kubectl -n tenant-test wait hr clickhouse-$name --timeout=20s --for=condition=ready
|
|
timeout 180 sh -ec "until kubectl -n tenant-test get svc chendpoint-clickhouse-$name -o jsonpath='{.spec.ports[*].port}' | grep -q '8123 9000'; do sleep 10; done"
|
|
timeout 60 sh -ec "until kubectl -n tenant-test get statefulset.apps/chi-clickhouse-$name-clickhouse-0-0 >/dev/null 2>&1; do sleep 2; done"
|
|
kubectl -n tenant-test wait statefulset.apps/chi-clickhouse-$name-clickhouse-0-0 --timeout=120s --for=jsonpath='{.status.replicas}'=1
|
|
timeout 80 sh -ec "until kubectl -n tenant-test get endpoints chi-clickhouse-$name-clickhouse-0-0 -o jsonpath='{.subsets[*].addresses[*].ip}' | grep -q '[0-9]'; do sleep 10; done"
|
|
timeout 100 sh -ec "until kubectl -n tenant-test get svc chi-clickhouse-$name-clickhouse-0-0 -o jsonpath='{.spec.ports[*].port}' | grep -q '9000 8123 9009'; do sleep 10; done"
|
|
timeout 80 sh -ec "until kubectl -n tenant-test get sts chi-clickhouse-$name-clickhouse-0-1 ; do sleep 10; done"
|
|
kubectl -n tenant-test wait statefulset.apps/chi-clickhouse-$name-clickhouse-0-1 --timeout=140s --for=jsonpath='{.status.replicas}'=1
|
|
kubectl -n tenant-test delete clickhouse $name
|
|
}
|