mirror of
https://github.com/zalando/postgres-operator.git
synced 2026-10-04 15:01:45 +02:00
Connection pooler for replica (#1127)
* Enable connection pooler for replica * Refactor code for connection pooler - Move all the relevant code to a separate file - Move all the related tests to a separate file - Avoid using cluster where not required - Simplify the logic in sync and other methods - Cleanup of duplicated or unused code * Fix labels for the replica pods * Update deleteConnectionPooler to include role * Adding test cases and other changes - Fix unit test and delete secret when required only - Make sure we use empty fresh cluster for every test case. * enhance e2e test * Disable pooler in complete manifest as this is source for e2e too an creates unnecessary pooler setups. Co-authored-by: Rafia Sabih <rafia.sabih@zalando.de> Co-authored-by: Jan Mussler <janm81@gmail.com>
This commit is contained in:
co-authored by
Rafia Sabih
Jan Mussler
parent
3fed565328
commit
49158ecb68
@@ -8,7 +8,9 @@ kubectl get statefulset -o jsonpath='{.items..metadata.annotations.zalando-postg
|
||||
echo
|
||||
echo
|
||||
echo 'Pods'
|
||||
kubectl get pods -l application=spilo -l name=postgres-operator -l application=db-connection-pooler -o wide --all-namespaces
|
||||
kubectl get pods -l application=spilo -o wide --all-namespaces
|
||||
echo
|
||||
kubectl get pods -l application=db-connection-pooler -o wide --all-namespaces
|
||||
echo
|
||||
echo 'Statefulsets'
|
||||
kubectl get statefulsets --all-namespaces
|
||||
|
||||
+20
-23
@@ -1,19 +1,14 @@
|
||||
import json
|
||||
import unittest
|
||||
import time
|
||||
import timeout_decorator
|
||||
import subprocess
|
||||
import warnings
|
||||
import os
|
||||
import yaml
|
||||
|
||||
from datetime import datetime
|
||||
from kubernetes import client, config
|
||||
from kubernetes.client.rest import ApiException
|
||||
|
||||
|
||||
def to_selector(labels):
|
||||
return ",".join(["=".join(l) for l in labels.items()])
|
||||
return ",".join(["=".join(lbl) for lbl in labels.items()])
|
||||
|
||||
|
||||
class K8sApi:
|
||||
@@ -43,8 +38,8 @@ class K8s:
|
||||
|
||||
def __init__(self, labels='x=y', namespace='default'):
|
||||
self.api = K8sApi()
|
||||
self.labels=labels
|
||||
self.namespace=namespace
|
||||
self.labels = labels
|
||||
self.namespace = namespace
|
||||
|
||||
def get_pg_nodes(self, pg_cluster_name, namespace='default'):
|
||||
master_pod_node = ''
|
||||
@@ -81,7 +76,7 @@ class K8s:
|
||||
'default', label_selector='name=postgres-operator'
|
||||
).items
|
||||
|
||||
pods = list(filter(lambda x: x.status.phase=='Running', pods))
|
||||
pods = list(filter(lambda x: x.status.phase == 'Running', pods))
|
||||
|
||||
if len(pods):
|
||||
return pods[0]
|
||||
@@ -110,7 +105,6 @@ class K8s:
|
||||
|
||||
time.sleep(self.RETRY_TIMEOUT_SEC)
|
||||
|
||||
|
||||
def get_service_type(self, svc_labels, namespace='default'):
|
||||
svc_type = ''
|
||||
svcs = self.api.core_v1.list_namespaced_service(namespace, label_selector=svc_labels, limit=1).items
|
||||
@@ -213,8 +207,8 @@ class K8s:
|
||||
self.wait_for_logical_backup_job(expected_num_of_jobs=1)
|
||||
|
||||
def delete_operator_pod(self, step="Delete operator pod"):
|
||||
# patching the pod template in the deployment restarts the operator pod
|
||||
self.api.apps_v1.patch_namespaced_deployment("postgres-operator","default", {"spec":{"template":{"metadata":{"annotations":{"step":"{}-{}".format(step, datetime.fromtimestamp(time.time()))}}}}})
|
||||
# patching the pod template in the deployment restarts the operator pod
|
||||
self.api.apps_v1.patch_namespaced_deployment("postgres-operator", "default", {"spec": {"template": {"metadata": {"annotations": {"step": "{}-{}".format(step, time.time())}}}}})
|
||||
self.wait_for_operator_pod_start()
|
||||
|
||||
def update_config(self, config_map_patch, step="Updating operator deployment"):
|
||||
@@ -237,7 +231,7 @@ class K8s:
|
||||
|
||||
def get_patroni_state(self, pod):
|
||||
r = self.exec_with_kubectl(pod, "patronictl list -f json")
|
||||
if not r.returncode == 0 or not r.stdout.decode()[0:1]=="[":
|
||||
if not r.returncode == 0 or not r.stdout.decode()[0:1] == "[":
|
||||
return []
|
||||
return json.loads(r.stdout.decode())
|
||||
|
||||
@@ -248,7 +242,7 @@ class K8s:
|
||||
pod = pod.metadata.name
|
||||
|
||||
r = self.exec_with_kubectl(pod, "curl localhost:8080/workers/all/status/")
|
||||
if not r.returncode == 0 or not r.stdout.decode()[0:1]=="{":
|
||||
if not r.returncode == 0 or not r.stdout.decode()[0:1] == "{":
|
||||
return None
|
||||
|
||||
return json.loads(r.stdout.decode())
|
||||
@@ -277,7 +271,7 @@ class K8s:
|
||||
'''
|
||||
pod = self.api.core_v1.list_namespaced_pod(
|
||||
namespace, label_selector="statefulset.kubernetes.io/pod-name=" + pod_name)
|
||||
|
||||
|
||||
if len(pod.items) == 0:
|
||||
return None
|
||||
return pod.items[0].spec.containers[0].image
|
||||
@@ -305,8 +299,8 @@ class K8sBase:
|
||||
|
||||
def __init__(self, labels='x=y', namespace='default'):
|
||||
self.api = K8sApi()
|
||||
self.labels=labels
|
||||
self.namespace=namespace
|
||||
self.labels = labels
|
||||
self.namespace = namespace
|
||||
|
||||
def get_pg_nodes(self, pg_cluster_labels='cluster-name=acid-minimal-cluster', namespace='default'):
|
||||
master_pod_node = ''
|
||||
@@ -434,10 +428,10 @@ class K8sBase:
|
||||
def count_pdbs_with_label(self, labels, namespace='default'):
|
||||
return len(self.api.policy_v1_beta1.list_namespaced_pod_disruption_budget(
|
||||
namespace, label_selector=labels).items)
|
||||
|
||||
|
||||
def count_running_pods(self, labels='application=spilo,cluster-name=acid-minimal-cluster', namespace='default'):
|
||||
pods = self.api.core_v1.list_namespaced_pod(namespace, label_selector=labels).items
|
||||
return len(list(filter(lambda x: x.status.phase=='Running', pods)))
|
||||
return len(list(filter(lambda x: x.status.phase == 'Running', pods)))
|
||||
|
||||
def wait_for_pod_failover(self, failover_targets, labels, namespace='default'):
|
||||
pod_phase = 'Failing over'
|
||||
@@ -484,14 +478,14 @@ class K8sBase:
|
||||
|
||||
def get_patroni_state(self, pod):
|
||||
r = self.exec_with_kubectl(pod, "patronictl list -f json")
|
||||
if not r.returncode == 0 or not r.stdout.decode()[0:1]=="[":
|
||||
if not r.returncode == 0 or not r.stdout.decode()[0:1] == "[":
|
||||
return []
|
||||
return json.loads(r.stdout.decode())
|
||||
|
||||
def get_patroni_running_members(self, pod):
|
||||
result = self.get_patroni_state(pod)
|
||||
return list(filter(lambda x: x["State"]=="running", result))
|
||||
|
||||
return list(filter(lambda x: x["State"] == "running", result))
|
||||
|
||||
def get_statefulset_image(self, label_selector="application=spilo,cluster-name=acid-minimal-cluster", namespace='default'):
|
||||
ssets = self.api.apps_v1.list_namespaced_stateful_set(namespace, label_selector=label_selector, limit=1)
|
||||
if len(ssets.items) == 0:
|
||||
@@ -505,7 +499,7 @@ class K8sBase:
|
||||
'''
|
||||
pod = self.api.core_v1.list_namespaced_pod(
|
||||
namespace, label_selector="statefulset.kubernetes.io/pod-name=" + pod_name)
|
||||
|
||||
|
||||
if len(pod.items) == 0:
|
||||
return None
|
||||
return pod.items[0].spec.containers[0].image
|
||||
@@ -514,10 +508,13 @@ class K8sBase:
|
||||
"""
|
||||
Inspiriational classes towards easier writing of end to end tests with one cluster per test case
|
||||
"""
|
||||
|
||||
|
||||
class K8sOperator(K8sBase):
|
||||
def __init__(self, labels="name=postgres-operator", namespace="default"):
|
||||
super().__init__(labels, namespace)
|
||||
|
||||
|
||||
class K8sPostgres(K8sBase):
|
||||
def __init__(self, labels="cluster-name=acid-minimal-cluster", namespace="default"):
|
||||
super().__init__(labels, namespace)
|
||||
|
||||
+161
-78
@@ -14,8 +14,9 @@ from kubernetes.client.rest import ApiException
|
||||
SPILO_CURRENT = "registry.opensource.zalan.do/acid/spilo-12:1.6-p5"
|
||||
SPILO_LAZY = "registry.opensource.zalan.do/acid/spilo-cdp-12:1.6-p114"
|
||||
|
||||
|
||||
def to_selector(labels):
|
||||
return ",".join(["=".join(l) for l in labels.items()])
|
||||
return ",".join(["=".join(lbl) for lbl in labels.items()])
|
||||
|
||||
|
||||
def clean_list(values):
|
||||
@@ -41,7 +42,7 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
self.assertEqual(y, x, m.format(y))
|
||||
return True
|
||||
except AssertionError:
|
||||
retries = retries -1
|
||||
retries = retries - 1
|
||||
if not retries > 0:
|
||||
raise
|
||||
time.sleep(interval)
|
||||
@@ -53,7 +54,7 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
self.assertNotEqual(y, x, m.format(y))
|
||||
return True
|
||||
except AssertionError:
|
||||
retries = retries -1
|
||||
retries = retries - 1
|
||||
if not retries > 0:
|
||||
raise
|
||||
time.sleep(interval)
|
||||
@@ -64,7 +65,7 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
self.assertTrue(f(), m)
|
||||
return True
|
||||
except AssertionError:
|
||||
retries = retries -1
|
||||
retries = retries - 1
|
||||
if not retries > 0:
|
||||
raise
|
||||
time.sleep(interval)
|
||||
@@ -104,13 +105,13 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
with open("manifests/postgres-operator.yaml", 'r+') as f:
|
||||
operator_deployment = yaml.safe_load(f)
|
||||
operator_deployment["spec"]["template"]["spec"]["containers"][0]["image"] = os.environ['OPERATOR_IMAGE']
|
||||
|
||||
|
||||
with open("manifests/postgres-operator.yaml", 'w') as f:
|
||||
yaml.dump(operator_deployment, f, Dumper=yaml.Dumper)
|
||||
|
||||
with open("manifests/configmap.yaml", 'r+') as f:
|
||||
configmap = yaml.safe_load(f)
|
||||
configmap["data"]["workers"] = "1"
|
||||
configmap = yaml.safe_load(f)
|
||||
configmap["data"]["workers"] = "1"
|
||||
|
||||
with open("manifests/configmap.yaml", 'w') as f:
|
||||
yaml.dump(configmap, f, Dumper=yaml.Dumper)
|
||||
@@ -129,8 +130,8 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
k8s.wait_for_operator_pod_start()
|
||||
|
||||
# reset taints and tolerations
|
||||
k8s.api.core_v1.patch_node("postgres-operator-e2e-tests-worker",{"spec":{"taints":[]}})
|
||||
k8s.api.core_v1.patch_node("postgres-operator-e2e-tests-worker2",{"spec":{"taints":[]}})
|
||||
k8s.api.core_v1.patch_node("postgres-operator-e2e-tests-worker", {"spec": {"taints": []}})
|
||||
k8s.api.core_v1.patch_node("postgres-operator-e2e-tests-worker2", {"spec": {"taints": []}})
|
||||
|
||||
# make sure we start a new operator on every new run,
|
||||
# this tackles the problem when kind is reused
|
||||
@@ -160,26 +161,76 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
the end turn connection pooler off to not interfere with other tests.
|
||||
'''
|
||||
k8s = self.k8s
|
||||
service_labels = {
|
||||
'cluster-name': 'acid-minimal-cluster',
|
||||
}
|
||||
pod_labels = dict({
|
||||
'connection-pooler': 'acid-minimal-cluster-pooler',
|
||||
})
|
||||
self.eventuallyEqual(lambda: k8s.get_operator_state(), {"0": "idle"}, "Operator does not get in sync")
|
||||
|
||||
# enable connection pooler
|
||||
k8s.api.custom_objects_api.patch_namespaced_custom_object(
|
||||
'acid.zalan.do', 'v1', 'default',
|
||||
'postgresqls', 'acid-minimal-cluster',
|
||||
{
|
||||
'spec': {
|
||||
'enableConnectionPooler': True,
|
||||
'enableReplicaConnectionPooler': True,
|
||||
}
|
||||
})
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_deployment_replica_count(), 2, "Deployment replicas is 2 default")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods("connection-pooler=acid-minimal-cluster-pooler"), 2, "No pooler pods found")
|
||||
self.eventuallyEqual(lambda: k8s.count_services_with_label('application=db-connection-pooler,cluster-name=acid-minimal-cluster'), 1, "No pooler service found")
|
||||
self.eventuallyEqual(lambda: k8s.get_deployment_replica_count(), 2,
|
||||
"Deployment replicas is 2 default")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(
|
||||
"connection-pooler=acid-minimal-cluster-pooler"),
|
||||
2, "No pooler pods found")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(
|
||||
"connection-pooler=acid-minimal-cluster-pooler-repl"),
|
||||
2, "No pooler replica pods found")
|
||||
self.eventuallyEqual(lambda: k8s.count_services_with_label(
|
||||
'application=db-connection-pooler,cluster-name=acid-minimal-cluster'),
|
||||
2, "No pooler service found")
|
||||
|
||||
# Turn off only master connection pooler
|
||||
k8s.api.custom_objects_api.patch_namespaced_custom_object(
|
||||
'acid.zalan.do', 'v1', 'default',
|
||||
'postgresqls', 'acid-minimal-cluster',
|
||||
{
|
||||
'spec': {
|
||||
'enableConnectionPooler': False,
|
||||
'enableReplicaConnectionPooler': True,
|
||||
}
|
||||
})
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_operator_state(), {"0": "idle"},
|
||||
"Operator does not get in sync")
|
||||
self.eventuallyEqual(lambda: k8s.get_deployment_replica_count(name="acid-minimal-cluster-pooler-repl"), 2,
|
||||
"Deployment replicas is 2 default")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(
|
||||
"connection-pooler=acid-minimal-cluster-pooler"),
|
||||
0, "Master pooler pods not deleted")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(
|
||||
"connection-pooler=acid-minimal-cluster-pooler-repl"),
|
||||
2, "Pooler replica pods not found")
|
||||
self.eventuallyEqual(lambda: k8s.count_services_with_label(
|
||||
'application=db-connection-pooler,cluster-name=acid-minimal-cluster'),
|
||||
1, "No pooler service found")
|
||||
|
||||
# Turn off only replica connection pooler
|
||||
k8s.api.custom_objects_api.patch_namespaced_custom_object(
|
||||
'acid.zalan.do', 'v1', 'default',
|
||||
'postgresqls', 'acid-minimal-cluster',
|
||||
{
|
||||
'spec': {
|
||||
'enableConnectionPooler': True,
|
||||
'enableReplicaConnectionPooler': False,
|
||||
}
|
||||
})
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_operator_state(), {"0": "idle"},
|
||||
"Operator does not get in sync")
|
||||
self.eventuallyEqual(lambda: k8s.get_deployment_replica_count(), 2,
|
||||
"Deployment replicas is 2 default")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods("connection-pooler=acid-minimal-cluster-pooler"),
|
||||
2, "Master pooler pods not found")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods("connection-pooler=acid-minimal-cluster-pooler-repl"),
|
||||
0, "Pooler replica pods not deleted")
|
||||
self.eventuallyEqual(lambda: k8s.count_services_with_label('application=db-connection-pooler,cluster-name=acid-minimal-cluster'),
|
||||
1, "No pooler service found")
|
||||
|
||||
# scale up connection pooler deployment
|
||||
k8s.api.custom_objects_api.patch_namespaced_custom_object(
|
||||
@@ -193,8 +244,10 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
}
|
||||
})
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_deployment_replica_count(), 3, "Deployment replicas is scaled to 3")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods("connection-pooler=acid-minimal-cluster-pooler"), 3, "Scale up of pooler pods does not work")
|
||||
self.eventuallyEqual(lambda: k8s.get_deployment_replica_count(), 3,
|
||||
"Deployment replicas is scaled to 3")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods("connection-pooler=acid-minimal-cluster-pooler"),
|
||||
3, "Scale up of pooler pods does not work")
|
||||
|
||||
# turn it off, keeping config should be overwritten by false
|
||||
k8s.api.custom_objects_api.patch_namespaced_custom_object(
|
||||
@@ -202,12 +255,15 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
'postgresqls', 'acid-minimal-cluster',
|
||||
{
|
||||
'spec': {
|
||||
'enableConnectionPooler': False
|
||||
'enableConnectionPooler': False,
|
||||
'enableReplicaConnectionPooler': False,
|
||||
}
|
||||
})
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods("connection-pooler=acid-minimal-cluster-pooler"), 0, "Pooler pods not scaled down")
|
||||
self.eventuallyEqual(lambda: k8s.count_services_with_label('application=db-connection-pooler,cluster-name=acid-minimal-cluster'), 0, "Pooler service not removed")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods("connection-pooler=acid-minimal-cluster-pooler"),
|
||||
0, "Pooler pods not scaled down")
|
||||
self.eventuallyEqual(lambda: k8s.count_services_with_label('application=db-connection-pooler,cluster-name=acid-minimal-cluster'),
|
||||
0, "Pooler service not removed")
|
||||
|
||||
# Verify that all the databases have pooler schema installed.
|
||||
# Do this via psql, since otherwise we need to deal with
|
||||
@@ -267,7 +323,8 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
'postgresqls', 'acid-minimal-cluster',
|
||||
{
|
||||
'spec': {
|
||||
'connectionPooler': None
|
||||
'connectionPooler': None,
|
||||
'EnableReplicaConnectionPooler': False,
|
||||
}
|
||||
})
|
||||
|
||||
@@ -281,8 +338,8 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
cluster_label = 'application=spilo,cluster-name=acid-minimal-cluster,spilo-role={}'
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_service_type(cluster_label.format("master")),
|
||||
'ClusterIP',
|
||||
"Expected ClusterIP type initially, found {}")
|
||||
'ClusterIP',
|
||||
"Expected ClusterIP type initially, found {}")
|
||||
|
||||
try:
|
||||
# enable load balancer services
|
||||
@@ -294,14 +351,14 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
}
|
||||
k8s.api.custom_objects_api.patch_namespaced_custom_object(
|
||||
"acid.zalan.do", "v1", "default", "postgresqls", "acid-minimal-cluster", pg_patch_enable_lbs)
|
||||
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_service_type(cluster_label.format("master")),
|
||||
'LoadBalancer',
|
||||
"Expected LoadBalancer service type for master, found {}")
|
||||
"Expected LoadBalancer service type for master, found {}")
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_service_type(cluster_label.format("replica")),
|
||||
'LoadBalancer',
|
||||
"Expected LoadBalancer service type for master, found {}")
|
||||
"Expected LoadBalancer service type for master, found {}")
|
||||
|
||||
# disable load balancer services again
|
||||
pg_patch_disable_lbs = {
|
||||
@@ -312,14 +369,14 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
}
|
||||
k8s.api.custom_objects_api.patch_namespaced_custom_object(
|
||||
"acid.zalan.do", "v1", "default", "postgresqls", "acid-minimal-cluster", pg_patch_disable_lbs)
|
||||
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_service_type(cluster_label.format("master")),
|
||||
'ClusterIP',
|
||||
"Expected LoadBalancer service type for master, found {}")
|
||||
"Expected LoadBalancer service type for master, found {}")
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_service_type(cluster_label.format("replica")),
|
||||
'ClusterIP',
|
||||
"Expected LoadBalancer service type for master, found {}")
|
||||
"Expected LoadBalancer service type for master, found {}")
|
||||
|
||||
except timeout_decorator.TimeoutError:
|
||||
print('Operator log: {}'.format(k8s.get_operator_log()))
|
||||
@@ -333,7 +390,8 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
k8s = self.k8s
|
||||
# update infrastructure roles description
|
||||
secret_name = "postgresql-infrastructure-roles"
|
||||
roles = "secretname: postgresql-infrastructure-roles-new, userkey: user, rolekey: memberof, passwordkey: password, defaultrolevalue: robot_zmon"
|
||||
roles = "secretname: postgresql-infrastructure-roles-new, userkey: user,"\
|
||||
"rolekey: memberof, passwordkey: password, defaultrolevalue: robot_zmon"
|
||||
patch_infrastructure_roles = {
|
||||
"data": {
|
||||
"infrastructure_roles_secret_name": secret_name,
|
||||
@@ -341,7 +399,8 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
},
|
||||
}
|
||||
k8s.update_config(patch_infrastructure_roles)
|
||||
self.eventuallyEqual(lambda: k8s.get_operator_state(), {"0": "idle"}, "Operator does not get in sync")
|
||||
self.eventuallyEqual(lambda: k8s.get_operator_state(), {"0": "idle"},
|
||||
"Operator does not get in sync")
|
||||
|
||||
try:
|
||||
# check that new roles are represented in the config by requesting the
|
||||
@@ -351,11 +410,12 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
try:
|
||||
operator_pod = k8s.get_operator_pod()
|
||||
get_config_cmd = "wget --quiet -O - localhost:8080/config"
|
||||
result = k8s.exec_with_kubectl(operator_pod.metadata.name, get_config_cmd)
|
||||
result = k8s.exec_with_kubectl(operator_pod.metadata.name,
|
||||
get_config_cmd)
|
||||
try:
|
||||
roles_dict = (json.loads(result.stdout)
|
||||
.get("controller", {})
|
||||
.get("InfrastructureRoles"))
|
||||
.get("controller", {})
|
||||
.get("InfrastructureRoles"))
|
||||
except:
|
||||
return False
|
||||
|
||||
@@ -377,7 +437,6 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
return False
|
||||
|
||||
self.eventuallyTrue(verify_role, "infrastructure role setup is not loaded")
|
||||
|
||||
|
||||
except timeout_decorator.TimeoutError:
|
||||
print('Operator log: {}'.format(k8s.get_operator_log()))
|
||||
@@ -386,13 +445,15 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
@timeout_decorator.timeout(TEST_TIMEOUT_SEC)
|
||||
def test_lazy_spilo_upgrade(self):
|
||||
'''
|
||||
Test lazy upgrade for the Spilo image: operator changes a stateful set but lets pods run with the old image
|
||||
until they are recreated for reasons other than operator's activity. That works because the operator configures
|
||||
stateful sets to use "onDelete" pod update policy.
|
||||
Test lazy upgrade for the Spilo image: operator changes a stateful set
|
||||
but lets pods run with the old image until they are recreated for
|
||||
reasons other than operator's activity. That works because the operator
|
||||
configures stateful sets to use "onDelete" pod update policy.
|
||||
|
||||
The test covers:
|
||||
1) enabling lazy upgrade in existing operator deployment
|
||||
2) forcing the normal rolling upgrade by changing the operator configmap and restarting its pod
|
||||
2) forcing the normal rolling upgrade by changing the operator
|
||||
configmap and restarting its pod
|
||||
'''
|
||||
|
||||
k8s = self.k8s
|
||||
@@ -400,8 +461,10 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
pod0 = 'acid-minimal-cluster-0'
|
||||
pod1 = 'acid-minimal-cluster-1'
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(), 2, "No 2 pods running")
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members(pod0)), 2, "Postgres status did not enter running")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(), 2,
|
||||
"No 2 pods running")
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members(pod0)),
|
||||
2, "Postgres status did not enter running")
|
||||
|
||||
patch_lazy_spilo_upgrade = {
|
||||
"data": {
|
||||
@@ -409,14 +472,20 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
"enable_lazy_spilo_upgrade": "false"
|
||||
}
|
||||
}
|
||||
k8s.update_config(patch_lazy_spilo_upgrade, step="Init baseline image version")
|
||||
k8s.update_config(patch_lazy_spilo_upgrade,
|
||||
step="Init baseline image version")
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_statefulset_image(), SPILO_CURRENT, "Stagefulset not updated initially")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(), 2, "No 2 pods running")
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members(pod0)), 2, "Postgres status did not enter running")
|
||||
self.eventuallyEqual(lambda: k8s.get_statefulset_image(), SPILO_CURRENT,
|
||||
"Statefulset not updated initially")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(), 2,
|
||||
"No 2 pods running")
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members(pod0)),
|
||||
2, "Postgres status did not enter running")
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.get_effective_pod_image(pod0), SPILO_CURRENT, "Rolling upgrade was not executed")
|
||||
self.eventuallyEqual(lambda: k8s.get_effective_pod_image(pod1), SPILO_CURRENT, "Rolling upgrade was not executed")
|
||||
self.eventuallyEqual(lambda: k8s.get_effective_pod_image(pod0),
|
||||
SPILO_CURRENT, "Rolling upgrade was not executed")
|
||||
self.eventuallyEqual(lambda: k8s.get_effective_pod_image(pod1),
|
||||
SPILO_CURRENT, "Rolling upgrade was not executed")
|
||||
|
||||
# update docker image in config and enable the lazy upgrade
|
||||
conf_image = SPILO_LAZY
|
||||
@@ -426,18 +495,25 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
"enable_lazy_spilo_upgrade": "true"
|
||||
}
|
||||
}
|
||||
k8s.update_config(patch_lazy_spilo_upgrade,step="patch image and lazy upgrade")
|
||||
self.eventuallyEqual(lambda: k8s.get_statefulset_image(), conf_image, "Statefulset not updated to next Docker image")
|
||||
k8s.update_config(patch_lazy_spilo_upgrade,
|
||||
step="patch image and lazy upgrade")
|
||||
self.eventuallyEqual(lambda: k8s.get_statefulset_image(), conf_image,
|
||||
"Statefulset not updated to next Docker image")
|
||||
|
||||
try:
|
||||
# restart the pod to get a container with the new image
|
||||
k8s.api.core_v1.delete_namespaced_pod(pod0, 'default')
|
||||
|
||||
k8s.api.core_v1.delete_namespaced_pod(pod0, 'default')
|
||||
|
||||
# verify only pod-0 which was deleted got new image from statefulset
|
||||
self.eventuallyEqual(lambda: k8s.get_effective_pod_image(pod0), conf_image, "Delete pod-0 did not get new spilo image")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(), 2, "No two pods running after lazy rolling upgrade")
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members(pod0)), 2, "Postgres status did not enter running")
|
||||
self.assertNotEqual(lambda: k8s.get_effective_pod_image(pod1), SPILO_CURRENT, "pod-1 should not have change Docker image to {}".format(SPILO_CURRENT))
|
||||
self.eventuallyEqual(lambda: k8s.get_effective_pod_image(pod0),
|
||||
conf_image, "Delete pod-0 did not get new spilo image")
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(), 2,
|
||||
"No two pods running after lazy rolling upgrade")
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members(pod0)),
|
||||
2, "Postgres status did not enter running")
|
||||
self.assertNotEqual(lambda: k8s.get_effective_pod_image(pod1),
|
||||
SPILO_CURRENT,
|
||||
"pod-1 should not have change Docker image to {}".format(SPILO_CURRENT))
|
||||
|
||||
# clean up
|
||||
unpatch_lazy_spilo_upgrade = {
|
||||
@@ -449,9 +525,14 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
|
||||
# at this point operator will complete the normal rolling upgrade
|
||||
# so we additonally test if disabling the lazy upgrade - forcing the normal rolling upgrade - works
|
||||
self.eventuallyEqual(lambda: k8s.get_effective_pod_image(pod0), conf_image, "Rolling upgrade was not executed", 50, 3)
|
||||
self.eventuallyEqual(lambda: k8s.get_effective_pod_image(pod1), conf_image, "Rolling upgrade was not executed", 50, 3)
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members(pod0)), 2, "Postgres status did not enter running")
|
||||
self.eventuallyEqual(lambda: k8s.get_effective_pod_image(pod0),
|
||||
conf_image, "Rolling upgrade was not executed",
|
||||
50, 3)
|
||||
self.eventuallyEqual(lambda: k8s.get_effective_pod_image(pod1),
|
||||
conf_image, "Rolling upgrade was not executed",
|
||||
50, 3)
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members(pod0)),
|
||||
2, "Postgres status did not enter running")
|
||||
|
||||
except timeout_decorator.TimeoutError:
|
||||
print('Operator log: {}'.format(k8s.get_operator_log()))
|
||||
@@ -505,9 +586,9 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
def get_docker_image():
|
||||
jobs = k8s.get_logical_backup_job().items
|
||||
return jobs[0].spec.job_template.spec.template.spec.containers[0].image
|
||||
|
||||
|
||||
self.eventuallyEqual(get_docker_image, image,
|
||||
"Expected job image {}, found {}".format(image, "{}"))
|
||||
"Expected job image {}, found {}".format(image, "{}"))
|
||||
|
||||
# delete the logical backup cron job
|
||||
pg_patch_disable_backup = {
|
||||
@@ -517,7 +598,7 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
}
|
||||
k8s.api.custom_objects_api.patch_namespaced_custom_object(
|
||||
"acid.zalan.do", "v1", "default", "postgresqls", "acid-minimal-cluster", pg_patch_disable_backup)
|
||||
|
||||
|
||||
self.eventuallyEqual(lambda: len(k8s.get_logical_backup_job().items), 0, "failed to create logical backup job")
|
||||
|
||||
except timeout_decorator.TimeoutError:
|
||||
@@ -563,21 +644,21 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
}
|
||||
k8s.api.custom_objects_api.patch_namespaced_custom_object(
|
||||
"acid.zalan.do", "v1", "default", "postgresqls", "acid-minimal-cluster", pg_patch_resources)
|
||||
|
||||
k8s.patch_statefulset({"metadata":{"annotations":{"zalando-postgres-operator-rolling-update-required": "False"}}})
|
||||
|
||||
k8s.patch_statefulset({"metadata": {"annotations": {"zalando-postgres-operator-rolling-update-required": "False"}}})
|
||||
k8s.update_config(patch_min_resource_limits, "Minimum resource test")
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(), 2, "No two pods running after lazy rolling upgrade")
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members()), 2, "Postgres status did not enter running")
|
||||
|
||||
|
||||
def verify_pod_limits():
|
||||
pods = k8s.api.core_v1.list_namespaced_pod('default', label_selector="cluster-name=acid-minimal-cluster,application=spilo").items
|
||||
if len(pods)<2:
|
||||
if len(pods) < 2:
|
||||
return False
|
||||
|
||||
r = pods[0].spec.containers[0].resources.limits['memory']==minMemoryLimit
|
||||
r = pods[0].spec.containers[0].resources.limits['memory'] == minMemoryLimit
|
||||
r = r and pods[0].spec.containers[0].resources.limits['cpu'] == minCPULimit
|
||||
r = r and pods[1].spec.containers[0].resources.limits['memory']==minMemoryLimit
|
||||
r = r and pods[1].spec.containers[0].resources.limits['memory'] == minMemoryLimit
|
||||
r = r and pods[1].spec.containers[0].resources.limits['cpu'] == minCPULimit
|
||||
return r
|
||||
|
||||
@@ -586,7 +667,7 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
@classmethod
|
||||
def setUp(cls):
|
||||
# cls.k8s.update_config({}, step="Setup")
|
||||
cls.k8s.patch_statefulset({"meta":{"annotations":{"zalando-postgres-operator-rolling-update-required": False}}})
|
||||
cls.k8s.patch_statefulset({"meta": {"annotations": {"zalando-postgres-operator-rolling-update-required": False}}})
|
||||
pass
|
||||
|
||||
@timeout_decorator.timeout(TEST_TIMEOUT_SEC)
|
||||
@@ -642,7 +723,7 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
}
|
||||
}
|
||||
}
|
||||
self.assertTrue(len(failover_targets)>0, "No failover targets available")
|
||||
self.assertTrue(len(failover_targets) > 0, "No failover targets available")
|
||||
for failover_target in failover_targets:
|
||||
k8s.api.core_v1.patch_node(failover_target, patch_readiness_label)
|
||||
|
||||
@@ -672,12 +753,12 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
Scale up from 2 to 3 and back to 2 pods by updating the Postgres manifest at runtime.
|
||||
'''
|
||||
k8s = self.k8s
|
||||
pod="acid-minimal-cluster-0"
|
||||
pod = "acid-minimal-cluster-0"
|
||||
|
||||
k8s.scale_cluster(3)
|
||||
k8s.scale_cluster(3)
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(), 3, "Scale up to 3 failed")
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members(pod)), 3, "Not all 3 nodes healthy")
|
||||
|
||||
|
||||
k8s.scale_cluster(2)
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(), 2, "Scale down to 2 failed")
|
||||
self.eventuallyEqual(lambda: len(k8s.get_patroni_running_members(pod)), 2, "Not all members 2 healthy")
|
||||
@@ -756,6 +837,7 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
|
||||
self.eventuallyTrue(lambda: k8s.check_statefulset_annotations(cluster_label, annotations), "Annotations missing")
|
||||
|
||||
self.eventuallyTrue(lambda: k8s.check_statefulset_annotations(cluster_label, annotations), "Annotations missing")
|
||||
|
||||
@timeout_decorator.timeout(TEST_TIMEOUT_SEC)
|
||||
@unittest.skip("Skipping this test until fixed")
|
||||
@@ -798,7 +880,7 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
"toleration": "key:postgres,operator:Exists,effect:NoExecute"
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
k8s.update_config(patch_toleration_config, step="allow tainted nodes")
|
||||
|
||||
self.eventuallyEqual(lambda: k8s.count_running_pods(), 2, "No 2 pods running")
|
||||
@@ -825,13 +907,14 @@ class EndToEndTestCase(unittest.TestCase):
|
||||
}
|
||||
}
|
||||
k8s.update_config(patch_delete_annotations)
|
||||
time.sleep(25)
|
||||
self.eventuallyEqual(lambda: k8s.get_operator_state(), {"0": "idle"}, "Operator does not get in sync")
|
||||
|
||||
try:
|
||||
# this delete attempt should be omitted because of missing annotations
|
||||
k8s.api.custom_objects_api.delete_namespaced_custom_object(
|
||||
"acid.zalan.do", "v1", "default", "postgresqls", "acid-minimal-cluster")
|
||||
time.sleep(5)
|
||||
time.sleep(15)
|
||||
self.eventuallyEqual(lambda: k8s.get_operator_state(), {"0": "idle"}, "Operator does not get in sync")
|
||||
|
||||
# check that pods and services are still there
|
||||
|
||||
Reference in New Issue
Block a user