mirror of
https://github.com/outbackdingo/patroni.git
synced 2026-08-26 15:40:21 +00:00
Compare commits
57
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5fca21849c | ||
|
|
439d292c60 | ||
|
|
7024d0a987 | ||
|
|
293c1e4cd3 | ||
|
|
7606df7196 | ||
|
|
cc076c40aa | ||
|
|
9ff6663e38 | ||
|
|
4fa666e78c | ||
|
|
bcc6a9bd93 | ||
|
|
c4032f4ce8 | ||
|
|
8fbf1b05da | ||
|
|
fac0d76081 | ||
|
|
4bdc0c7f7f | ||
|
|
fe6c864536 | ||
|
|
13cfe0af36 | ||
|
|
7319d12026 | ||
|
|
2ec9834c60 | ||
|
|
2be64e5131 | ||
|
|
93be10a655 | ||
|
|
366829e379 | ||
|
|
899cad1c0f | ||
|
|
a4ac4963d1 | ||
|
|
704d36815a | ||
|
|
4138d0b830 | ||
|
|
b7ea511511 | ||
|
|
badf1da183 | ||
|
|
6a75b1591b | ||
|
|
ba204884d8 | ||
|
|
82d2ef4878 | ||
|
|
b83f1c0f44 | ||
|
|
9209a5a133 | ||
|
|
3734ecc851 | ||
|
|
46bc1ded8e | ||
|
|
713244975c | ||
|
|
efaba9f183 | ||
|
|
d3da80196e | ||
|
|
01830ecbd1 | ||
|
|
3c1f2ff7a4 | ||
|
|
c052789c56 | ||
|
|
f24db395c6 | ||
|
|
9dd177e5c9 | ||
|
|
eb100fd586 | ||
|
|
444021e6b8 | ||
|
|
a74985f41d | ||
|
|
da9aaf6cdf | ||
|
|
17a139f890 | ||
|
|
3e96e89e8b | ||
|
|
3684a16b41 | ||
|
|
7492861238 | ||
|
|
ccb09c79f5 | ||
|
|
370808dd18 | ||
|
|
bc0f9f522e | ||
|
|
3b8610ac49 | ||
|
|
eab58ef231 | ||
|
|
6ff386de1f | ||
|
|
dca2a9ead2 | ||
|
|
39b643742a |
@@ -45,8 +45,8 @@ def install_packages(what):
|
||||
packages['exhibitor'] = packages['zookeeper']
|
||||
packages = packages.get(what, [])
|
||||
ver = versions.get(what)
|
||||
if float(ver) == 15:
|
||||
packages += ['postgresql-{0}-citus-12.0'.format(ver)]
|
||||
if float(ver) >= 15:
|
||||
packages += ['postgresql-{0}-citus-11.2'.format(ver)]
|
||||
subprocess.call(['sudo', 'apt-get', 'update', '-y'])
|
||||
return subprocess.call(['sudo', 'apt-get', 'install', '-y', 'postgresql-' + ver, 'expect-dev'] + packages)
|
||||
|
||||
|
||||
@@ -1 +1 @@
|
||||
versions = {'etcd': '9.6', 'etcd3': '16', 'consul': '13', 'exhibitor': '12', 'raft': '11', 'kubernetes': '15'}
|
||||
versions = {'etcd': '9.6', 'etcd3': '14', 'consul': '13', 'exhibitor': '12', 'raft': '11', 'kubernetes': '15'}
|
||||
|
||||
@@ -5,7 +5,6 @@ on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
- 'REL_[0-9]+_[0-9]+'
|
||||
|
||||
env:
|
||||
CODACY_PROJECT_TOKEN: ${{ secrets.CODACY_PROJECT_TOKEN }}
|
||||
@@ -174,7 +173,7 @@ jobs:
|
||||
|
||||
- uses: jakebailey/pyright-action@v1
|
||||
with:
|
||||
version: 1.1.326
|
||||
version: 1.1.320
|
||||
|
||||
docs:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
@@ -64,3 +64,6 @@ venv*/
|
||||
|
||||
# Default test data directory
|
||||
data/
|
||||
|
||||
# macOS
|
||||
**/.DS_Store
|
||||
|
||||
+4
-5
@@ -12,7 +12,7 @@ Patroni is a template for high availability (HA) PostgreSQL solutions using Pyth
|
||||
|
||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely.
|
||||
|
||||
Currently supported PostgreSQL versions: 9.3 to 16.
|
||||
Currently supported PostgreSQL versions: 9.3 to 15.
|
||||
|
||||
**Note to Citus users**: Starting from 3.0 Patroni nicely integrates with the `Citus <https://github.com/citusdata/citus>`__ database extension to Postgres. Please check the `Citus support page <https://github.com/zalando/patroni/blob/master/docs/citus.rst>`__ in the Patroni documentation for more info about how to use Patroni high availability together with a Citus distributed cluster.
|
||||
|
||||
@@ -74,9 +74,8 @@ There are a few options available:
|
||||
|
||||
::
|
||||
|
||||
sudo apt-get install python-psycopg2 # install python2 psycopg2 module on Debian/Ubuntu
|
||||
sudo apt-get install python3-psycopg2 # install python3 psycopg2 module on Debian/Ubuntu
|
||||
sudo yum install python-psycopg2 # install python2 psycopg2 on RedHat/Fedora/CentOS
|
||||
sudo apt-get install python3-psycopg2 # install psycopg2 module on Debian/Ubuntu
|
||||
sudo yum install python3-psycopg2 # install psycopg2 on RedHat/Fedora/CentOS
|
||||
|
||||
2. Install psycopg2 from the binary package
|
||||
|
||||
@@ -94,7 +93,7 @@ There are a few options available:
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg[binary]
|
||||
pip install psycopg[binary]>=3.0.0
|
||||
|
||||
**General installation for pip**
|
||||
|
||||
|
||||
Vendored
BIN
Binary file not shown.
@@ -115,7 +115,7 @@ Kubernetes
|
||||
- **PATRONI\_KUBERNETES\_ROLE\_LABEL**: (optional) name of the label containing role (master or replica or other custom value). Patroni will set this label on the pod it runs in. Default value is ``role``.
|
||||
- **PATRONI\_KUBERNETES\_LEADER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is `master`. Default value is `master`.
|
||||
- **PATRONI\_KUBERNETES\_FOLLOWER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is `replica`. Default value is `replica`.
|
||||
- **PATRONI\_KUBERNETES\_STANDBY\_LEADER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is ``standby_leader``. Default value is ``master``.
|
||||
- **PATRONI\_KUBERNETES\_STANDBY\_LEADER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is ``standby-leader``. Default value is ``standby-leader``.
|
||||
- **PATRONI\_KUBERNETES\_TMP\_ROLE\_LABEL**: (optional) name of the temporary label containing role (master or replica). Value of this label will always use the default of corresponding role. Set only when necessary.
|
||||
- **PATRONI\_KUBERNETES\_USE\_ENDPOINTS**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||
- **PATRONI\_KUBERNETES\_POD\_IP**: (optional) IP address of the pod Patroni is running in. This value is required when `PATRONI_KUBERNETES_USE_ENDPOINTS` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||
|
||||
+2
-7
@@ -46,9 +46,8 @@ There are a few options available:
|
||||
|
||||
::
|
||||
|
||||
sudo apt-get install python-psycopg2 # install python2 psycopg2 module on Debian/Ubuntu
|
||||
sudo apt-get install python3-psycopg2 # install python3 psycopg2 module on Debian/Ubuntu
|
||||
sudo yum install python-psycopg2 # install python2 psycopg2 on RedHat/Fedora/CentOS
|
||||
sudo apt-get install python3-psycopg2 # install psycopg2 module on Debian/Ubuntu
|
||||
sudo yum install python3-psycopg2 # install psycopg2 on RedHat/Fedora/CentOS
|
||||
|
||||
2. Install psycopg2 from the binary package
|
||||
|
||||
@@ -165,10 +164,6 @@ Applications Should Not Use Superusers
|
||||
|
||||
When connecting from an application, always use a non-superuser. Patroni requires access to the database to function properly. By using a superuser from an application, you can potentially use the entire connection pool, including the connections reserved for superusers, with the ``superuser_reserved_connections`` setting. If Patroni cannot access the Primary because the connection pool is full, behavior will be undesirable.
|
||||
|
||||
.. |Build Status| image:: https://travis-ci.org/zalando/patroni.svg?branch=master
|
||||
:target: https://travis-ci.org/zalando/patroni
|
||||
.. |Coverage Status| image:: https://coveralls.io/repos/zalando/patroni/badge.svg?branch=master
|
||||
:target: https://coveralls.io/r/zalando/patroni?branch=master
|
||||
|
||||
Testing Your HA Solution
|
||||
--------------------------------------
|
||||
|
||||
+1
-1
@@ -105,10 +105,10 @@ todo_include_todos = True
|
||||
# a list of builtin themes.
|
||||
#
|
||||
|
||||
html_theme = 'sphinx_rtd_theme'
|
||||
on_rtd = os.environ.get('READTHEDOCS', None) == 'True'
|
||||
if not on_rtd: # only import and set the theme if we're building docs locally
|
||||
import sphinx_rtd_theme
|
||||
html_theme = 'sphinx_rtd_theme'
|
||||
html_theme_path = [sphinx_rtd_theme.get_html_theme_path()]
|
||||
|
||||
# Theme options are theme-specific and customize the look and feel of a theme
|
||||
|
||||
+1
-1
@@ -10,7 +10,7 @@ Patroni is a template for high availability (HA) PostgreSQL solutions using Pyth
|
||||
|
||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
|
||||
|
||||
Currently supported PostgreSQL versions: 9.3 to 16.
|
||||
Currently supported PostgreSQL versions: 9.3 to 15.
|
||||
|
||||
**Note to Citus users**: Starting from 3.0 Patroni nicely integrates with the `Citus <https://github.com/citusdata/citus>`__ database extension to Postgres. Please check the :ref:`Citus support page <citus>` in the Patroni documentation for more info about how to use Patroni high availability together with a Citus distributed cluster.
|
||||
|
||||
|
||||
@@ -44,7 +44,6 @@ Some of the PostgreSQL parameters **must hold the same values on the primary and
|
||||
- **max_worker_processes**: 8
|
||||
- **max_prepared_transactions**: 0
|
||||
- **wal_level**: hot_standby
|
||||
- **wal_log_hints**: on
|
||||
- **track_commit_timestamp**: off
|
||||
|
||||
For the parameters below, PostgreSQL does not require equal values among the primary and all the replicas. However, considering the possibility of a replica to become the primary at any time, it doesn't really make sense to set them differently; therefore, **Patroni restricts setting their values to the** :ref:`dynamic configuration <dynamic_configuration>`.
|
||||
@@ -62,6 +61,7 @@ There are some other Postgres parameters controlled by Patroni:
|
||||
- **port** - is set either from ``postgresql.listen`` or from ``PATRONI_POSTGRESQL_LISTEN`` environment variable
|
||||
- **cluster_name** - is set either from ``scope`` or from ``PATRONI_SCOPE`` environment variable
|
||||
- **hot_standby: on**
|
||||
- **wal_log_hints: on** - for Postgres 9.4 and newer.
|
||||
|
||||
To be on the safe side parameters from the above lists are not written into ``postgresql.conf``, but passed as a list of arguments to the ``pg_ctl start`` which gives them the highest precedence, even above `ALTER SYSTEM <https://www.postgresql.org/docs/current/static/sql-altersystem.html>`__
|
||||
|
||||
|
||||
+1
-1
@@ -19,7 +19,7 @@ When Patroni runs in a paused mode, it does not change the state of PostgreSQL,
|
||||
|
||||
- For the Postgres primary with the leader lock Patroni updates the lock. If the node with the leader lock stops being the primary (i.e. is demoted manually), Patroni will release the lock instead of promoting the node back.
|
||||
|
||||
- Manual unscheduled restart, reinitialize and manual failover are allowed. Manual failover is only allowed if the node to failover to is specified. In the paused mode, manual failover does not require a running primary node.
|
||||
- Manual unscheduled restart, manual unscheduled failover/switchover and reinitialize are allowed. No scheduled action is allowed. Manual switchover is only allowed if the node to switchover to is specified.
|
||||
|
||||
- If 'parallel' primaries are detected by Patroni, it emits a warning, but does not demote the primary without the leader lock.
|
||||
|
||||
|
||||
@@ -3,86 +3,6 @@
|
||||
Release notes
|
||||
=============
|
||||
|
||||
Version 3.1.2
|
||||
-------------
|
||||
|
||||
**Bugfixes**
|
||||
|
||||
- Fixed bug with ``wal_keep_size`` checks (Alexander Kukushkin)
|
||||
|
||||
The ``wal_keep_size`` is a GUC that normally has a unit and Patroni was failing to cast its value to ``int``. As a result the value of ``bootstrap.dcs`` was not written to the ``/config`` key afterwards.
|
||||
|
||||
- Detect and resolve inconsistencies between ``/sync`` key and ``synchronous_standby_names`` (Alexander Kukushkin)
|
||||
|
||||
Normally, Patroni updates ``/sync`` and ``synchronous_standby_names`` in a very specific order, but in case of a bug or when someone manually reset ``synchronous_standby_names``, Patroni was getting into an inconsistent state. As a result it was possible that the failover happens to an asynchronous node.
|
||||
|
||||
- Read GUC's values when joining running Postgres (Alexander Kukushkin)
|
||||
|
||||
When restarted in ``pause``, Patroni was discarding the ``synchronous_standby_names`` GUC from the ``postgresql.conf``. To solve it and avoid similar issues, Patroni will read GUC's value if it is joining an already running Postgres.
|
||||
|
||||
- Silenced annoying warnings when checking for node uniqueness (Alexander Kukushkin)
|
||||
|
||||
``WARNING`` messages are produced by ``urllib3`` if Patroni is quickly restarted.
|
||||
|
||||
|
||||
Version 3.1.1
|
||||
-------------
|
||||
|
||||
**Bugfixes**
|
||||
|
||||
- Reset failsafe state on promote (ChenChangAo)
|
||||
|
||||
If switchover/failover happened shortly after failsafe mode had been activated, the newly promoted primary was demoting itself after failsafe becomes inactive.
|
||||
|
||||
- Silence useless warnings in ``patronictl`` (Alexander Kukushkin)
|
||||
|
||||
If ``patronictl`` uses the same patroni.yaml file as Patroni and can access ``PGDATA`` directory it might have been showing annoying warnings about incorrect values in the global configuration.
|
||||
|
||||
- Explicitly enable synchronous mode for a corner case (Alexander Kukushkin)
|
||||
|
||||
Synchronous mode effectively was never activated if there are no replicas streaming from the primary.
|
||||
|
||||
- Fixed bug with ``0`` integer values validation (Israel Barth Rubio)
|
||||
|
||||
In most cases, it didn't cause any issues, just warnings.
|
||||
|
||||
- Don't return logical slots for standby cluster (Alexander Kukushkin)
|
||||
|
||||
Patroni can't create logical replication slots in the standby cluster, thus they should be ignored if they are defined in the global configuration.
|
||||
|
||||
- Avoid showing docstring in ``patronictl --help`` output (Israel Barth Rubio)
|
||||
|
||||
The ``click`` module needs to get a special hint for that.
|
||||
|
||||
- Fixed bug with ``kubernetes.standby_leader_label_value`` (Alexander Kukushkin)
|
||||
|
||||
This feature effectively never worked.
|
||||
|
||||
- Returned cluster system identifier to the ``patronictl list`` output (Polina Bungina)
|
||||
|
||||
The problem was introduced while implementing the support for Citus, where we need to hide the identifier because it is different for coordinator and all workers.
|
||||
|
||||
- Override ``write_leader_optime`` method in Kubernetes implementation (Alexander Kukushkin)
|
||||
|
||||
The method is supposed to write shutdown LSN to the leader Endpoint/ConfigMap when there are no healthy replicas available to become the new primary.
|
||||
|
||||
- Don't start stopped postgres in pause (Alexander Kukushkin)
|
||||
|
||||
Due to a race condition, Patroni was falsely assuming that the standby should be restarted because some recovery parameters (``primary_conninfo`` or similar) were changed.
|
||||
|
||||
- Fixed bug in ``patronictl query`` command (Israel Barth Rubio)
|
||||
|
||||
It didn't work when only ``-m`` argument was provided or when none of ``-r`` or ``-m`` were provided.
|
||||
|
||||
- Properly treat integer parameters that are used in the command line to start postgres (Polina Bungina)
|
||||
|
||||
If values are supplied as strings and not casted to integer it was resulting in an incorrect calculation of ``max_prepared_transactions`` based on ``max_connections`` for Citus clusters.
|
||||
|
||||
- Don't rely on ``pg_stat_wal_receiver`` when deciding on ``pg_rewind`` (Alexander Kukushkin)
|
||||
|
||||
It could happen that ``received_tli`` reported by ``pg_stat_wal_recevier`` is ahead of the actual replayed timeline, while the timeline reported by ``DENTIFY_SYSTEM`` via replication connection is always correct.
|
||||
|
||||
|
||||
Version 3.1.0
|
||||
-------------
|
||||
|
||||
|
||||
+273
-40
@@ -92,26 +92,184 @@ Monitoring endpoint
|
||||
|
||||
The ``GET /patroni`` is used by Patroni during the leader race. It also could be used by your monitoring system. The JSON document produced by this endpoint has the same structure as the JSON produced by the health check endpoints.
|
||||
|
||||
**Example:** A healthy cluster
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/patroni | jq .
|
||||
{
|
||||
"state": "running",
|
||||
"postmaster_start_time": "2019-09-24 09:22:32.555 CEST",
|
||||
"postmaster_start_time": "2023-08-18 11:03:37.966359+00:00",
|
||||
"role": "master",
|
||||
"server_version": 110005,
|
||||
"cluster_unlocked": false,
|
||||
"server_version": 150004,
|
||||
"xlog": {
|
||||
"location": 25624640
|
||||
"location": 67395656
|
||||
},
|
||||
"timeline": 3,
|
||||
"database_system_identifier": "6739877027151648096",
|
||||
"timeline": 1,
|
||||
"replication": [
|
||||
{
|
||||
"usename": "replicator",
|
||||
"application_name": "patroni2",
|
||||
"client_addr": "10.89.0.6",
|
||||
"state": "streaming",
|
||||
"sync_state": "async",
|
||||
"sync_priority": 0
|
||||
},
|
||||
{
|
||||
"usename": "replicator",
|
||||
"application_name": "patroni3",
|
||||
"client_addr": "10.89.0.2",
|
||||
"state": "streaming",
|
||||
"sync_state": "async",
|
||||
"sync_priority": 0
|
||||
}
|
||||
],
|
||||
"dcs_last_seen": 1692356718,
|
||||
"tags": {
|
||||
"clonefrom": true
|
||||
},
|
||||
"database_system_identifier": "7268616322854375442",
|
||||
"patroni": {
|
||||
"version": "1.6.0",
|
||||
"scope": "batman"
|
||||
"version": "3.1.0",
|
||||
"scope": "demo"
|
||||
}
|
||||
}
|
||||
|
||||
**Example:** An unlocked cluster
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/patroni | jq .
|
||||
{
|
||||
"state": "running",
|
||||
"postmaster_start_time": "2023-08-18 11:09:08.615242+00:00",
|
||||
"role": "replica",
|
||||
"server_version": 150004,
|
||||
"xlog": {
|
||||
"received_location": 67419744,
|
||||
"replayed_location": 67419744,
|
||||
"replayed_timestamp": null,
|
||||
"paused": false
|
||||
},
|
||||
"timeline": 1,
|
||||
"replication": [
|
||||
{
|
||||
"usename": "replicator",
|
||||
"application_name": "patroni2",
|
||||
"client_addr": "10.89.0.6",
|
||||
"state": "streaming",
|
||||
"sync_state": "async",
|
||||
"sync_priority": 0
|
||||
},
|
||||
{
|
||||
"usename": "replicator",
|
||||
"application_name": "patroni3",
|
||||
"client_addr": "10.89.0.2",
|
||||
"state": "streaming",
|
||||
"sync_state": "async",
|
||||
"sync_priority": 0
|
||||
}
|
||||
],
|
||||
"cluster_unlocked": true,
|
||||
"dcs_last_seen": 1692356928,
|
||||
"tags": {
|
||||
"clonefrom": true
|
||||
},
|
||||
"database_system_identifier": "7268616322854375442",
|
||||
"patroni": {
|
||||
"version": "3.1.0",
|
||||
"scope": "demo"
|
||||
}
|
||||
}
|
||||
|
||||
**Example:** An unlocked cluster with :ref:`DCS failsafe mode <dcs_failsafe_mode>` enabled
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/patroni | jq .
|
||||
{
|
||||
"state": "running",
|
||||
"postmaster_start_time": "2023-08-18 11:09:08.615242+00:00",
|
||||
"role": "replica",
|
||||
"server_version": 150004,
|
||||
"xlog": {
|
||||
"location": 67420024
|
||||
},
|
||||
"timeline": 1,
|
||||
"replication": [
|
||||
{
|
||||
"usename": "replicator",
|
||||
"application_name": "patroni2",
|
||||
"client_addr": "10.89.0.6",
|
||||
"state": "streaming",
|
||||
"sync_state": "async",
|
||||
"sync_priority": 0
|
||||
},
|
||||
{
|
||||
"usename": "replicator",
|
||||
"application_name": "patroni3",
|
||||
"client_addr": "10.89.0.2",
|
||||
"state": "streaming",
|
||||
"sync_state": "async",
|
||||
"sync_priority": 0
|
||||
}
|
||||
],
|
||||
"cluster_unlocked": true,
|
||||
"failsafe_mode_is_active": true,
|
||||
"dcs_last_seen": 1692356928,
|
||||
"tags": {
|
||||
"clonefrom": true
|
||||
},
|
||||
"database_system_identifier": "7268616322854375442",
|
||||
"patroni": {
|
||||
"version": "3.1.0",
|
||||
"scope": "demo"
|
||||
}
|
||||
}
|
||||
|
||||
**Example:** A cluster with the :ref:`pause mode <pause>` enabled
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/patroni | jq .
|
||||
{
|
||||
"state": "running",
|
||||
"postmaster_start_time": "2023-08-18 11:09:08.615242+00:00",
|
||||
"role": "replica",
|
||||
"server_version": 150004,
|
||||
"xlog": {
|
||||
"location": 67420024
|
||||
},
|
||||
"timeline": 1,
|
||||
"replication": [
|
||||
{
|
||||
"usename": "replicator",
|
||||
"application_name": "patroni2",
|
||||
"client_addr": "10.89.0.6",
|
||||
"state": "streaming",
|
||||
"sync_state": "async",
|
||||
"sync_priority": 0
|
||||
},
|
||||
{
|
||||
"usename": "replicator",
|
||||
"application_name": "patroni3",
|
||||
"client_addr": "10.89.0.2",
|
||||
"state": "streaming",
|
||||
"sync_state": "async",
|
||||
"sync_priority": 0
|
||||
}
|
||||
],
|
||||
"pause": true,
|
||||
"dcs_last_seen": 1692356928,
|
||||
"tags": {
|
||||
"clonefrom": true
|
||||
},
|
||||
"database_system_identifier": "7268616322854375442",
|
||||
"patroni": {
|
||||
"version": "3.1.0",
|
||||
"scope": "demo"
|
||||
}
|
||||
}
|
||||
|
||||
Retrieve the Patroni metrics in Prometheus format through the ``GET /metrics`` endpoint.
|
||||
|
||||
@@ -131,6 +289,9 @@ Retrieve the Patroni metrics in Prometheus format through the ``GET /metrics`` e
|
||||
# HELP patroni_master Value is 1 if this node is the leader, 0 otherwise.
|
||||
# TYPE patroni_master gauge
|
||||
patroni_master{scope="batman"} 1
|
||||
# HELP patroni_primary Value is 1 if this node is the leader, 0 otherwise.
|
||||
# TYPE patroni_primary gauge
|
||||
patroni_primary{scope="batman"} 1
|
||||
# HELP patroni_xlog_location Current location of the Postgres transaction log, 0 if this node is not the leader.
|
||||
# TYPE patroni_xlog_location counter
|
||||
patroni_xlog_location{scope="batman"} 22320573386952
|
||||
@@ -169,6 +330,9 @@ Retrieve the Patroni metrics in Prometheus format through the ``GET /metrics`` e
|
||||
patroni_cluster_unlocked{scope="batman"} 0
|
||||
# HELP patroni_postgres_timeline Postgres timeline of this node (if running), 0 otherwise.
|
||||
# TYPE patroni_postgres_timeline counter
|
||||
patroni_failsafe_mode_is_active{scope="batman"} 0
|
||||
# HELP patroni_postgres_timeline Postgres timeline of this node (if running), 0 otherwise.
|
||||
# TYPE patroni_postgres_timeline counter
|
||||
patroni_postgres_timeline{scope="batman"} 24
|
||||
# HELP patroni_dcs_last_seen Epoch timestamp when DCS was last contacted successfully by Patroni.
|
||||
# TYPE patroni_dcs_last_seen gauge
|
||||
@@ -192,24 +356,24 @@ Cluster status endpoints
|
||||
{
|
||||
"members": [
|
||||
{
|
||||
"name": "postgresql0",
|
||||
"host": "127.0.0.1",
|
||||
"port": 5432,
|
||||
"name": "patroni1",
|
||||
"role": "leader",
|
||||
"state": "running",
|
||||
"api_url": "http://127.0.0.1:8008/patroni",
|
||||
"api_url": "http://10.89.0.4:8008/patroni",
|
||||
"host": "10.89.0.4",
|
||||
"port": 5432,
|
||||
"timeline": 5,
|
||||
"tags": {
|
||||
"clonefrom": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "postgresql1",
|
||||
"host": "127.0.0.1",
|
||||
"port": 5433,
|
||||
"name": "patroni2",
|
||||
"role": "replica",
|
||||
"state": "running",
|
||||
"api_url": "http://127.0.0.1:8009/patroni",
|
||||
"state": "streaming",
|
||||
"api_url": "http://10.89.0.6:8008/patroni",
|
||||
"host": "10.89.0.6",
|
||||
"port": 5433,
|
||||
"timeline": 5,
|
||||
"tags": {
|
||||
"clonefrom": true
|
||||
@@ -218,8 +382,9 @@ Cluster status endpoints
|
||||
}
|
||||
],
|
||||
"scheduled_switchover": {
|
||||
"at": "2019-09-24T10:36:00+02:00",
|
||||
"from": "postgresql0"
|
||||
"at": "2023-09-24T10:36:00+02:00",
|
||||
"from": "patroni1",
|
||||
"to": "patroni3"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -264,7 +429,7 @@ Config endpoint
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s localhost:8008/config | jq .
|
||||
$ curl -s http://localhost:8008/config | jq .
|
||||
{
|
||||
"ttl": 30,
|
||||
"loop_wait": 10,
|
||||
@@ -275,7 +440,6 @@ Config endpoint
|
||||
"use_pg_rewind": true,
|
||||
"parameters": {
|
||||
"hot_standby": "on",
|
||||
"wal_log_hints": "on",
|
||||
"wal_level": "hot_standby",
|
||||
"max_wal_senders": 5,
|
||||
"max_replication_slots": 5,
|
||||
@@ -302,7 +466,6 @@ Config endpoint
|
||||
"use_pg_rewind": true,
|
||||
"parameters": {
|
||||
"hot_standby": "on",
|
||||
"wal_log_hints": "on",
|
||||
"wal_level": "hot_standby",
|
||||
"max_wal_senders": 5,
|
||||
"max_replication_slots": 5,
|
||||
@@ -355,7 +518,6 @@ If you want to remove (reset) some setting just patch it with ``null``:
|
||||
"hot_standby": "on",
|
||||
"unix_socket_directories": ".",
|
||||
"wal_level": "hot_standby",
|
||||
"wal_log_hints": "on",
|
||||
"max_wal_senders": 5,
|
||||
"max_replication_slots": 5
|
||||
}
|
||||
@@ -369,7 +531,7 @@ The above call removes ``postgresql.parameters.max_connections`` from the dynami
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s -XPUT -d \
|
||||
'{"maximum_lag_on_failover":1048576,"retry_timeout":10,"postgresql":{"use_slots":true,"use_pg_rewind":true,"parameters":{"hot_standby":"on","wal_log_hints":"on","wal_level":"hot_standby","unix_socket_directories":".","max_wal_senders":5}},"loop_wait":3,"ttl":20}' \
|
||||
'{"maximum_lag_on_failover":1048576,"retry_timeout":10,"postgresql":{"use_slots":true,"use_pg_rewind":true,"parameters":{"hot_standby":"on","wal_level":"hot_standby","unix_socket_directories":".","max_wal_senders":5}},"loop_wait":3,"ttl":20}' \
|
||||
http://localhost:8008/config | jq .
|
||||
{
|
||||
"ttl": 20,
|
||||
@@ -381,7 +543,6 @@ The above call removes ``postgresql.parameters.max_connections`` from the dynami
|
||||
"hot_standby": "on",
|
||||
"unix_socket_directories": ".",
|
||||
"wal_level": "hot_standby",
|
||||
"wal_log_hints": "on",
|
||||
"max_wal_senders": 5
|
||||
},
|
||||
"use_pg_rewind": true
|
||||
@@ -393,39 +554,111 @@ The above call removes ``postgresql.parameters.max_connections`` from the dynami
|
||||
Switchover and failover endpoints
|
||||
---------------------------------
|
||||
|
||||
``POST /switchover`` or ``POST /failover``. These endpoints are very similar to each other. There are a couple of minor differences though:
|
||||
.. _switchover_api:
|
||||
|
||||
1. The failover endpoint allows to perform a manual failover when there are no healthy nodes, but at the same time it will not allow you to schedule a switchover.
|
||||
Switchover
|
||||
^^^^^^^^^^
|
||||
|
||||
2. The switchover endpoint is the opposite. It works only when the cluster is healthy (there is a leader) and allows to schedule a switchover at a given time.
|
||||
``/switchover`` endpoint only works when cluster is healthy (there is a leader). It allows to schedule a switchover at a given time.
|
||||
|
||||
When calling ``/switchover`` endpoint candidate can be specified but is not required, in contrast to ``/failover`` endpoint. If candidate is not provided, all the healthy nodes that are allowed to failover participate in the leader race.
|
||||
|
||||
In the JSON body of the ``POST`` request you must specify at least the ``leader`` or ``candidate`` fields and optionally the ``scheduled_at`` field if you want to schedule a switchover at a specific time.
|
||||
In the JSON body of the ``POST`` request, you must specify at least the ``leader`` field and, optionally, the ``candidate`` and ``scheduled_at`` field if you want to schedule a switchover at a specific time.
|
||||
|
||||
Depending on the situation, requests might return different HTTP status codes and bodies. Status code **200** is returned when the switchover or failover successfully completed. If the switchover was successfully scheduled, Patroni will return HTTP status code **202**. In case something went wrong, the error status code (one of **400**, **412**, or **503**) will be returned with some details in the response body.
|
||||
|
||||
Example: perform a failover to the specific node:
|
||||
``DELETE /switchover`` can be used to delete the currently scheduled switchover.
|
||||
|
||||
**Example:** perform a switchover to any healthy standby
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8009/failover -XPOST -d '{"candidate":"postgresql1"}'
|
||||
Successfully failed over to "postgresql1"
|
||||
$ curl -s http://localhost:8008/switchover -XPOST -d '{"leader":"postgresql1"}'
|
||||
Successfully switched over to "postgresql2"
|
||||
|
||||
|
||||
Example: schedule a switchover from the leader to any other healthy replica in the cluster at a specific time:
|
||||
**Example:** perform a switchover to a specific node
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/switchover -XPOST -d \
|
||||
'{"leader":"postgresql0","scheduled_at":"2019-09-24T12:00+00"}'
|
||||
Switchover scheduled
|
||||
$ curl -s http://localhost:8008/switchover -XPOST -d \
|
||||
'{"leader":"postgresql1","candidate":"postgresql2"}'
|
||||
Successfully switched over to "postgresql2"
|
||||
|
||||
|
||||
Depending on the situation the request might finish with a different HTTP status code and body. The status code **200** is returned when the switchover or failover successfully completed. If the switchover was successfully scheduled, Patroni will return HTTP status code **202**. In case something went wrong, the error status code (one of **400**, **412** or **503**) will be returned with some details in the response body. For more information please check the source code of ``patroni/api.py:do_POST_failover()`` method.
|
||||
**Example:** schedule a switchover from the leader to any other healthy standby in the cluster at a specific time.
|
||||
|
||||
- ``DELETE /switchover``: delete the scheduled switchover
|
||||
.. code-block:: bash
|
||||
|
||||
The ``POST /switchover`` and ``POST failover`` endpoints are used by ``patronictl switchover`` and ``patronictl failover``, respectively.
|
||||
The ``DELETE /switchover`` is used by ``patronictl flush <cluster-name> switchover``.
|
||||
$ curl -s http://localhost:8008/switchover -XPOST -d \
|
||||
'{"leader":"postgresql0","scheduled_at":"2019-09-24T12:00+00"}'
|
||||
Switchover scheduled
|
||||
|
||||
|
||||
Failover
|
||||
^^^^^^^^
|
||||
|
||||
``/failover`` endpoint can be used to perform a manual failover when there are no healthy nodes (e.g. to an asynchronous standby if all synchronous standbys are not healthy enough to promote). However there is no requirement for a cluster not to have leader - failover can also be run on a healthy cluster.
|
||||
|
||||
In the JSON body of the ``POST`` request you must specify ``candidate`` field. If ``leader`` field is specified, switchover is triggered.
|
||||
|
||||
**Example:**
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/failover -XPOST -d '{"candidate":"postgresql1"}'
|
||||
Successfully failed over to "postgresql1"
|
||||
|
||||
.. warning::
|
||||
:ref:`Be very careful <failover_healthcheck>` using this endpoint, as this can cause data loss in certain situations. In most cases, :ref:`the switchover endpoint <switchover_api>` satisfies the administrator's needs.
|
||||
|
||||
|
||||
``POST /switchover`` and ``POST /failover`` endpoints are used by ``patronictl switchover`` and ``patronictl failover``, respectively.
|
||||
|
||||
``DELETE /switchover`` is used by ``patronictl flush <cluster-name> switchover``.
|
||||
|
||||
.. list-table:: Failover/Switchover comparison
|
||||
:widths: 25 25 25
|
||||
:header-rows: 1
|
||||
|
||||
* -
|
||||
- Failover
|
||||
- Switchover
|
||||
* - Requires leader specified
|
||||
- no
|
||||
- yes
|
||||
* - Requires candidate specified
|
||||
- yes
|
||||
- no
|
||||
* - Can be run in pause
|
||||
- yes
|
||||
- yes (only to a specific candidate)
|
||||
* - Can be scheduled
|
||||
- no
|
||||
- yes (if not in pause)
|
||||
|
||||
.. _failover_healthcheck:
|
||||
|
||||
Healthy standby
|
||||
^^^^^^^^^^^^^^^
|
||||
|
||||
There are a couple of checks that a member of a cluster should pass to be able to participate in the leader race during a switchover or to become a leader as a failover/switchover candidate:
|
||||
|
||||
- be reachable via Patroni API,
|
||||
- not have ``nofailover`` tag set to ``true``,
|
||||
- have watchdog fully functional (if required by the configuration),
|
||||
- in case of a switchover or a failover in a healthy cluster, not exceed maximum replication lag (``maximum_lag_on_failover`` :ref:`configuration parameter <dynamic_configuration>`),
|
||||
- in case of a switchover or a failover in a healthy cluster, not have a timeline number smaller than the cluster timeline,
|
||||
- in :ref:`synchronous mode <synchronous_mode>`:
|
||||
|
||||
- In case of a switchover (both with and without a candidate): be listed in the ``/sync`` key members.
|
||||
- For a failover in both healthy and unhealthy clusters, this check is omitted.
|
||||
|
||||
.. warning::
|
||||
In case of a failover in a cluster without a leader, a candidate will be allowed to promote even if:
|
||||
- it is not in the ``/sync`` key members when synchronous mode is enabled,
|
||||
- its lag exceeds the maximum replication lag allowed,
|
||||
- it has the timeline number smaller than the cluster timeline.
|
||||
|
||||
|
||||
Restart endpoint
|
||||
|
||||
@@ -30,9 +30,15 @@ Log
|
||||
|
||||
Bootstrap configuration
|
||||
-----------------------
|
||||
|
||||
.. note::
|
||||
Once Patroni has initialized the cluster for the first time and settings have been stored in the DCS, all future
|
||||
changes to the ``bootstrap.dcs`` section of the YAML configuration will not take any effect! If you want to change
|
||||
them please use either ``patronictl edit-config`` or the Patroni :ref:`REST API <rest_api>`.
|
||||
|
||||
- **bootstrap**:
|
||||
|
||||
- **dcs**: This section will be written into `/<namespace>/<scope>/config` of the given configuration store after initializing of new cluster. The global dynamic configuration for the cluster. Under the ``bootstrap.dcs`` you can put any of the parameters described in the :ref:`Dynamic Configuration settings <dynamic_configuration>` and after Patroni initialized (bootstrapped) the new cluster, it will write this section into `/<namespace>/<scope>/config` of the configuration store. All later changes of ``bootstrap.dcs`` will not take any effect! If you want to change them please use either ``patronictl edit-config`` or Patroni :ref:`REST API <rest_api>`.
|
||||
- **dcs**: This section will be written into `/<namespace>/<scope>/config` of the given configuration store after initializing the new cluster. The global dynamic configuration for the cluster. You can put any of the parameters described in the :ref:`Dynamic Configuration settings <dynamic_configuration>` under ``bootstrap.dcs`` and after Patroni has initialized (bootstrapped) the new cluster, it will write this section into `/<namespace>/<scope>/config` of the configuration store.
|
||||
- **method**: custom script to use for bootstrapping this cluster.
|
||||
|
||||
See :ref:`custom bootstrap methods documentation <custom_bootstrap>` for details.
|
||||
@@ -165,7 +171,7 @@ Kubernetes
|
||||
- **role\_label**: (optional) name of the label containing role (master or replica or other custom value). Patroni will set this label on the pod it runs in. Default value is ``role``.
|
||||
- **leader\_label\_value**: (optional) value of the pod label when Postgres role is ``master``. Default value is ``master``.
|
||||
- **follower\_label\_value**: (optional) value of the pod label when Postgres role is ``replica``. Default value is ``replica``.
|
||||
- **standby\_leader\_label\_value**: (optional) value of the pod label when Postgres role is ``standby_leader``. Default value is ``master``.
|
||||
- **standby\_leader\_label\_value**: (optional) value of the pod label when Postgres role is ``standby-leader``. Default value is ``standby-leader``.
|
||||
- **tmp_\role\_label**: (optional) name of the temporary label containing role (master or replica). Value of this label will always use the default of corresponding role. Set only when necessary.
|
||||
- **use\_endpoints**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||
- **pod\_ip**: (optional) IP address of the pod Patroni is running in. This value is required when `use_endpoints` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||
|
||||
@@ -68,7 +68,6 @@ Scenario: check API requests for the primary-replica pair in the pause mode
|
||||
When I kill postmaster on postgres1
|
||||
And I issue a GET request to http://127.0.0.1:8009/replica
|
||||
Then I receive a response code 503
|
||||
And "members/postgres1" key in DCS has state=stopped after 10 seconds
|
||||
When I run patronictl.py restart batman postgres1 --force
|
||||
Then I receive a response returncode 0
|
||||
Then replication works from postgres0 to postgres1 after 20 seconds
|
||||
@@ -77,7 +76,7 @@ Scenario: check API requests for the primary-replica pair in the pause mode
|
||||
Then I receive a response code 200
|
||||
And I receive a response state running
|
||||
And I receive a response role replica
|
||||
When I run patronictl.py reinit batman postgres1 --force --wait
|
||||
When I run patronictl.py reinit batman postgres1 --force
|
||||
Then I receive a response returncode 0
|
||||
And I receive a response output "Success: reinitialize for member postgres1"
|
||||
And postgres1 role is the secondary after 30 seconds
|
||||
|
||||
+156
-13
@@ -1,3 +1,8 @@
|
||||
"""Patroni main entry point.
|
||||
|
||||
Implement ``patroni`` main daemon and expose its entry point.
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import signal
|
||||
@@ -16,8 +21,33 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Patroni(AbstractPatroniDaemon):
|
||||
"""Implement ``patroni`` command daemon.
|
||||
|
||||
:ivar version: Patroni version.
|
||||
:ivar dcs: DCS object.
|
||||
:ivar watchdog: watchdog handler, if configured to use watchdog.
|
||||
:ivar postgresql: managed Postgres instance.
|
||||
:ivar api: REST API server instance of this node.
|
||||
:ivar request: wrapper for performing HTTP requests.
|
||||
:ivar ha: HA handler.
|
||||
:ivar tags: cache of custom tags configured for this node.
|
||||
:ivar next_run: time when to run the next HA loop cycle.
|
||||
:ivar scheduled_restart: when a restart has been scheduled to occur, if any. In that case, should contain two keys:
|
||||
* ``schedule``: timestamp when restart should occur;
|
||||
* ``postmaster_start_time``: timestamp when Postgres was last started.
|
||||
"""
|
||||
|
||||
def __init__(self, config: 'Config') -> None:
|
||||
"""Create a :class:`Patroni` instance with the given *config*.
|
||||
|
||||
Get a connection to the DCS, configure watchdog (if required), set up Patroni interface with Postgres, configure
|
||||
the HA loop and bring the REST API up.
|
||||
|
||||
.. note::
|
||||
Expected to be instantiated and run through :func:`~patroni.daemon.abstract_main`.
|
||||
|
||||
:param config: Patroni configuration.
|
||||
"""
|
||||
from patroni.api import RestApiServer
|
||||
from patroni.dcs import get_dcs
|
||||
from patroni.ha import Ha
|
||||
@@ -46,6 +76,17 @@ class Patroni(AbstractPatroniDaemon):
|
||||
self.scheduled_restart: Dict[str, Any] = {}
|
||||
|
||||
def load_dynamic_configuration(self) -> None:
|
||||
"""Load Patroni dynamic configuration.
|
||||
|
||||
Load dynamic configuration from the DCS, if `/config` key is available in the DCS, otherwise fall back to
|
||||
``bootstrap.dcs`` section from the configuration file.
|
||||
|
||||
If the DCS connection fails returning the exception :class:`~patroni.exceptions.DCSError` an attempt will be
|
||||
remade every 5 seconds.
|
||||
|
||||
.. note::
|
||||
This method is called only once, at the time when Patroni is started.
|
||||
"""
|
||||
from patroni.exceptions import DCSError
|
||||
while True:
|
||||
try:
|
||||
@@ -65,8 +106,6 @@ class Patroni(AbstractPatroniDaemon):
|
||||
|
||||
def ensure_unique_name(self) -> None:
|
||||
"""A helper method to prevent splitbrain from operator naming error."""
|
||||
from urllib.parse import urlparse
|
||||
from urllib3.connection import HTTPConnection
|
||||
from patroni.dcs import Member
|
||||
|
||||
cluster = self.dcs.get_cluster()
|
||||
@@ -76,28 +115,54 @@ class Patroni(AbstractPatroniDaemon):
|
||||
if not isinstance(member, Member):
|
||||
return
|
||||
try:
|
||||
parts = urlparse(member.api_url)
|
||||
if isinstance(parts.hostname, str):
|
||||
connection = HTTPConnection(parts.hostname, port=parts.port or 80, timeout=3)
|
||||
connection.connect()
|
||||
logger.fatal("Can't start; there is already a node named '%s' running", self.config['name'])
|
||||
sys.exit(1)
|
||||
_ = self.request(member, endpoint="/liveness", timeout=3)
|
||||
logger.fatal("Can't start; there is already a node named '%s' running", self.config['name'])
|
||||
sys.exit(1)
|
||||
except Exception:
|
||||
return
|
||||
|
||||
def get_tags(self) -> Dict[str, Any]:
|
||||
"""Get tags configured for this node, if any.
|
||||
|
||||
Handle both predefined Patroni tags and custom defined tags.
|
||||
|
||||
.. note::
|
||||
A custom tag is any tag added to the configuration ``tags`` section that is not one of ``clonefrom``,
|
||||
``nofailover``, ``noloadbalance`` or ``nosync``.
|
||||
|
||||
For the Patroni predefined tags, the returning object will only contain them if they are enabled as they
|
||||
all are boolean values that default to disabled.
|
||||
|
||||
:returns: a dictionary of tags set for this node. The key is the tag name, and the value is the corresponding
|
||||
tag value.
|
||||
"""
|
||||
return {tag: value for tag, value in self.config.get('tags', {}).items()
|
||||
if tag not in ('clonefrom', 'nofailover', 'noloadbalance', 'nosync') or value}
|
||||
|
||||
@property
|
||||
def nofailover(self) -> bool:
|
||||
"""``True`` if ``tags.nofailover`` configuration is enabled for this node, else ``False``."""
|
||||
return bool(self.tags.get('nofailover', False))
|
||||
|
||||
@property
|
||||
def nosync(self) -> bool:
|
||||
"""``True`` if ``tags.nosync`` configuration is enabled for this node, else ``False``."""
|
||||
return bool(self.tags.get('nosync', False))
|
||||
|
||||
def reload_config(self, sighup: bool = False, local: Optional[bool] = False) -> None:
|
||||
"""Apply new configuration values for ``patroni`` daemon.
|
||||
|
||||
Reload:
|
||||
* Cached tags;
|
||||
* Request wrapper configuration;
|
||||
* REST API configuration;
|
||||
* Watchdog configuration;
|
||||
* Postgres configuration;
|
||||
* DCS configuration.
|
||||
|
||||
:param sighup: if it is related to a SIGHUP signal.
|
||||
:param local: if there has been changes to the local configuration file.
|
||||
"""
|
||||
try:
|
||||
super(Patroni, self).reload_config(sighup, local)
|
||||
if local:
|
||||
@@ -112,14 +177,21 @@ class Patroni(AbstractPatroniDaemon):
|
||||
logger.exception('Failed to reload config_file=%s', self.config.config_file)
|
||||
|
||||
@property
|
||||
def replicatefrom(self):
|
||||
def replicatefrom(self) -> Optional[str]:
|
||||
"""Value of ``tags.replicatefrom`` configuration, if any."""
|
||||
return self.tags.get('replicatefrom')
|
||||
|
||||
@property
|
||||
def noloadbalance(self):
|
||||
def noloadbalance(self) -> bool:
|
||||
"""``True`` if ``tags.noloadbalance`` configuration is enabled for this node, else ``False``."""
|
||||
return bool(self.tags.get('noloadbalance', False))
|
||||
|
||||
def schedule_next_run(self) -> None:
|
||||
"""Schedule the next run of the ``patroni`` daemon main loop.
|
||||
|
||||
Next run is scheduled based on previous run plus value of ``loop_wait`` configuration from DCS. If that has
|
||||
already been exceeded, run the next cycle immediately.
|
||||
"""
|
||||
self.next_run += self.dcs.loop_wait
|
||||
current_time = time.time()
|
||||
nap_time = self.next_run - current_time
|
||||
@@ -133,11 +205,21 @@ class Patroni(AbstractPatroniDaemon):
|
||||
self.next_run = time.time()
|
||||
|
||||
def run(self) -> None:
|
||||
"""Run ``patroni`` daemon process main loop.
|
||||
|
||||
Start the REST API and keep running HA cycles every ``loop_wait`` seconds.
|
||||
"""
|
||||
self.api.start()
|
||||
self.next_run = time.time()
|
||||
super(Patroni, self).run()
|
||||
|
||||
def _run_cycle(self) -> None:
|
||||
"""Run a cycle of the ``patroni`` daemon main loop.
|
||||
|
||||
Run an HA cycle and schedule the next cycle run. If any dynamic configuration change request is detected, apply
|
||||
the change and cache the new dynamic configuration values in ``patroni.dynamic.json`` file under Postgres data
|
||||
directory.
|
||||
"""
|
||||
logger.info(self.ha.run_cycle())
|
||||
|
||||
if self.dcs.cluster and self.dcs.cluster.config and self.dcs.cluster.config.data \
|
||||
@@ -150,6 +232,10 @@ class Patroni(AbstractPatroniDaemon):
|
||||
self.schedule_next_run()
|
||||
|
||||
def _shutdown(self) -> None:
|
||||
"""Perform shutdown of ``patroni`` daemon process.
|
||||
|
||||
Shut down the REST API and the HA handler.
|
||||
"""
|
||||
try:
|
||||
self.api.shutdown()
|
||||
except Exception:
|
||||
@@ -161,18 +247,54 @@ class Patroni(AbstractPatroniDaemon):
|
||||
|
||||
|
||||
def patroni_main(configfile: str) -> None:
|
||||
"""Configure and start ``patroni`` main daemon process.
|
||||
|
||||
:param configfile: path to Patroni configuration file.
|
||||
"""
|
||||
from multiprocessing import freeze_support
|
||||
|
||||
# Windows executables created by PyInstaller are frozen, thus we need to enable frozen support for
|
||||
# :mod:`multiprocessing` to avoid :class:`RuntimeError` exceptions.
|
||||
freeze_support()
|
||||
abstract_main(Patroni, configfile)
|
||||
|
||||
|
||||
def process_arguments() -> Namespace:
|
||||
"""Process command-line arguments.
|
||||
|
||||
Create a basic command-line parser through :func:`~patroni.daemon.get_base_arg_parser`, extend its capabilities by
|
||||
adding these flags and parse command-line arguments.:
|
||||
|
||||
* ``--validate-config`` -- used to validate the Patroni configuration file
|
||||
* ``--generate-config`` -- used to generate Patroni configuration from a running PostgreSQL instance
|
||||
* ``--generate-sample-config`` -- used to generate a sample Patroni configuration
|
||||
|
||||
.. note::
|
||||
If running with ``--generate-config``, ``--generate-sample-config`` or ``--validate-flag`` will exit
|
||||
after generating or validating configuration.
|
||||
|
||||
:returns: parsed arguments, if not running with ``--validate-config`` flag.
|
||||
"""
|
||||
from patroni.config_generator import generate_config
|
||||
|
||||
parser = get_base_arg_parser()
|
||||
parser.add_argument('--validate-config', action='store_true', help='Run config validator and exit')
|
||||
group = parser.add_mutually_exclusive_group()
|
||||
group.add_argument('--validate-config', action='store_true', help='Run config validator and exit')
|
||||
group.add_argument('--generate-sample-config', action='store_true',
|
||||
help='Generate a sample Patroni yaml configuration file')
|
||||
group.add_argument('--generate-config', action='store_true',
|
||||
help='Generate a Patroni yaml configuration file for a running instance')
|
||||
parser.add_argument('--dsn', help='Optional DSN string of the instance to be used as a source \
|
||||
for config generation. Superuser connection is required.')
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.validate_config:
|
||||
if args.generate_sample_config:
|
||||
generate_config(args.configfile, True, None)
|
||||
sys.exit(0)
|
||||
elif args.generate_config:
|
||||
generate_config(args.configfile, False, args.dsn)
|
||||
sys.exit(0)
|
||||
elif args.validate_config:
|
||||
from patroni.validator import schema
|
||||
from patroni.config import Config, ConfigParseError
|
||||
|
||||
@@ -186,6 +308,16 @@ def process_arguments() -> Namespace:
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Main entrypoint of :mod:`patroni.__main__`.
|
||||
|
||||
Process command-line arguments, ensure :mod:`psycopg2` (or :mod:`psycopg`) attendee the pre-requisites and start
|
||||
``patroni`` daemon process.
|
||||
|
||||
.. note::
|
||||
If running through a Docker container, make the main process take care of init process duties and run
|
||||
``patroni`` daemon as another process. In that case relevant signals received by the main process and forwarded
|
||||
to ``patroni`` daemon process.
|
||||
"""
|
||||
from patroni import check_psycopg
|
||||
|
||||
args = process_arguments()
|
||||
@@ -201,7 +333,13 @@ def main() -> None:
|
||||
|
||||
# Looks like we are in a docker, so we will act like init
|
||||
def sigchld_handler(signo: int, stack_frame: Optional[FrameType]) -> None:
|
||||
"""Handle ``SIGCHLD`` received by main process from ``patroni`` daemon when the daemon terminates.
|
||||
|
||||
:param signo: signal number.
|
||||
:param stack_frame: current stack frame.
|
||||
"""
|
||||
try:
|
||||
# log exit code of all children processes, and break loop when there is none left
|
||||
while True:
|
||||
ret = os.waitpid(-1, os.WNOHANG)
|
||||
if ret == (0, 0):
|
||||
@@ -211,7 +349,12 @@ def main() -> None:
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
def passtochild(signo: int, stack_frame: Optional[FrameType]):
|
||||
def passtochild(signo: int, stack_frame: Optional[FrameType]) -> None:
|
||||
"""Forward a signal *signo* from main process to child process.
|
||||
|
||||
:param signo: signal number.
|
||||
:param stack_frame: current stack frame.
|
||||
"""
|
||||
if pid:
|
||||
os.kill(pid, signo)
|
||||
|
||||
|
||||
+28
-121
@@ -12,7 +12,6 @@ import json
|
||||
import logging
|
||||
import time
|
||||
import traceback
|
||||
import dateutil.parser
|
||||
import datetime
|
||||
import os
|
||||
import socket
|
||||
@@ -28,11 +27,11 @@ from typing import Any, Callable, Dict, Iterator, List, Optional, Tuple, TYPE_CH
|
||||
|
||||
from . import psycopg
|
||||
from .__main__ import Patroni
|
||||
from .dcs import Cluster
|
||||
from .exceptions import PostgresConnectionException, PostgresException
|
||||
from .manual_failover import ManualFailover
|
||||
from .postgresql.misc import postgres_version_to_int
|
||||
from .utils import deep_compare, enable_keepalive, parse_bool, patch_config, Retry, \
|
||||
RetryFailedError, parse_int, split_host_port, tzutc, uri, cluster_as_json
|
||||
RetryFailedError, parse_int, parse_schedule, split_host_port, tzutc, uri, cluster_as_json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -770,44 +769,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
self.server.patroni.api_sigterm()
|
||||
self.write_response(202, 'shutdown scheduled')
|
||||
|
||||
@staticmethod
|
||||
def parse_schedule(schedule: str,
|
||||
action: str) -> Tuple[Union[int, None], Union[str, None], Union[datetime.datetime, None]]:
|
||||
"""Parse the given *schedule* and validate it.
|
||||
|
||||
:param schedule: a string representing a timestamp, e.g. ``2023-04-14T20:27:00+00:00``.
|
||||
:param action: the action to be scheduled (``restart``, ``switchover``, or ``failover``).
|
||||
|
||||
:returns: a tuple composed of 3 items:
|
||||
|
||||
* Suggested HTTP status code for a response:
|
||||
|
||||
* ``None``: if no issue was faced while parsing, leaving it up to the caller to decide the status; or
|
||||
* ``400``: if no timezone information could be found in *schedule*; or
|
||||
* ``422``: if *schedule* is invalid -- in the past or not parsable.
|
||||
|
||||
* An error message, if any error is faced, otherwise ``None``;
|
||||
* Parsed *schedule*, if able to parse, otherwise ``None``.
|
||||
|
||||
"""
|
||||
error = None
|
||||
scheduled_at = None
|
||||
try:
|
||||
scheduled_at = dateutil.parser.parse(schedule)
|
||||
if scheduled_at.tzinfo is None:
|
||||
error = 'Timezone information is mandatory for the scheduled {0}'.format(action)
|
||||
status_code = 400
|
||||
elif scheduled_at < datetime.datetime.now(tzutc):
|
||||
error = 'Cannot schedule {0} in the past'.format(action)
|
||||
status_code = 422
|
||||
else:
|
||||
status_code = None
|
||||
except (ValueError, TypeError):
|
||||
logger.exception('Invalid scheduled %s time: %s', action, schedule)
|
||||
error = 'Unable to parse scheduled timestamp. It should be in an unambiguous format, e.g. ISO 8601'
|
||||
status_code = 422
|
||||
return status_code, error, scheduled_at
|
||||
|
||||
@check_access
|
||||
def do_POST_restart(self) -> None:
|
||||
"""Handle a ``POST`` request to ``/restart`` path.
|
||||
@@ -863,9 +824,9 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
|
||||
for k in request:
|
||||
if k == 'schedule':
|
||||
(_, data, request[k]) = self.parse_schedule(request[k], "restart")
|
||||
if _:
|
||||
status_code = _
|
||||
parse_result, request[k] = parse_schedule(request[k])
|
||||
if parse_result:
|
||||
data, status_code = parse_result.value[0], parse_result.value[1]
|
||||
break
|
||||
elif k == 'role':
|
||||
if request[k] not in ('master', 'primary', 'replica'):
|
||||
@@ -1015,39 +976,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
logger.debug('Exception occurred during polling %s result: %s', action, e)
|
||||
return 503, action.title() + ' status unknown'
|
||||
|
||||
def is_failover_possible(self, cluster: Cluster, leader: Optional[str], candidate: Optional[str],
|
||||
action: str) -> Optional[str]:
|
||||
"""Checks whether there are nodes that could take over after demoting the primary.
|
||||
|
||||
:param cluster: the Patroni cluster.
|
||||
:param leader: name of the current Patroni leader.
|
||||
:param candidate: name of the Patroni node to be promoted.
|
||||
:param action: the action to be performed (``switchover`` or ``failover``).
|
||||
|
||||
:returns: a string with the error message or ``None`` if good nodes are found.
|
||||
"""
|
||||
is_synchronous_mode = self.server.patroni.config.get_global_config(cluster).is_synchronous_mode
|
||||
if leader and (not cluster.leader or cluster.leader.name != leader):
|
||||
return 'leader name does not match'
|
||||
if candidate:
|
||||
if action == 'switchover' and is_synchronous_mode and not cluster.sync.matches(candidate):
|
||||
return 'candidate name does not match with sync_standby'
|
||||
members = [m for m in cluster.members if m.name == candidate]
|
||||
if not members:
|
||||
return 'candidate does not exists'
|
||||
elif is_synchronous_mode:
|
||||
members = [m for m in cluster.members if cluster.sync.matches(m.name)]
|
||||
if not members:
|
||||
return action + ' is not possible: can not find sync_standby'
|
||||
else:
|
||||
members = [m for m in cluster.members if not cluster.leader or m.name != cluster.leader.name and m.api_url]
|
||||
if not members:
|
||||
return action + ' is not possible: cluster does not have members except leader'
|
||||
for st in self.server.patroni.ha.fetch_nodes_statuses(members):
|
||||
if st.failover_limitation() is None:
|
||||
return None
|
||||
return action + ' is not possible: no good candidates have been found'
|
||||
|
||||
@check_access
|
||||
def do_POST_failover(self, action: str = 'failover') -> None:
|
||||
"""Handle a ``POST`` request to ``/failover`` path.
|
||||
@@ -1075,7 +1003,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
:param action: the action to be performed (``switchover`` or ``failover``).
|
||||
"""
|
||||
request = self._read_json_content()
|
||||
(status_code, data) = (400, '')
|
||||
if not request:
|
||||
return
|
||||
|
||||
@@ -1088,26 +1015,15 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
logger.info("received %s request with leader=%s candidate=%s scheduled_at=%s",
|
||||
action, leader, candidate, scheduled_at)
|
||||
|
||||
if action == 'failover' and not candidate:
|
||||
data = 'Failover could be performed only to a specific candidate'
|
||||
elif action == 'switchover' and not leader:
|
||||
data = 'Switchover could be performed only from a specific leader'
|
||||
manual_failover = ManualFailover(action, cluster, leader, candidate, scheduled_at,
|
||||
global_config.is_paused, global_config.is_synchronous_mode,
|
||||
self.server.patroni)
|
||||
data, status_code = manual_failover.run_precheck().value
|
||||
|
||||
if not data and scheduled_at:
|
||||
if not leader:
|
||||
data = 'Scheduled {0} is possible only from a specific leader'.format(action)
|
||||
if not data and global_config.is_paused:
|
||||
data = "Can't schedule {0} in the paused state".format(action)
|
||||
if not data:
|
||||
(status_code, data, scheduled_at) = self.parse_schedule(scheduled_at, action)
|
||||
|
||||
if not data and global_config.is_paused and not candidate:
|
||||
data = action.title() + ' is possible only to a specific candidate in a paused state'
|
||||
|
||||
if not data and not scheduled_at:
|
||||
data = self.is_failover_possible(cluster, leader, candidate, action)
|
||||
if data:
|
||||
status_code = 412
|
||||
parse_result, scheduled_at = manual_failover.parse_scheduled()
|
||||
if parse_result:
|
||||
data, status_code = parse_result.value[0], parse_result.value[1]
|
||||
|
||||
if not data:
|
||||
if self.server.patroni.dcs.manual_failover(leader, candidate, scheduled_at=scheduled_at):
|
||||
@@ -1119,14 +1035,12 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
status_code, data = self.poll_failover_result(cluster.leader and cluster.leader.name,
|
||||
candidate, action)
|
||||
else:
|
||||
data = 'failed to write {0} key into DCS'.format(action)
|
||||
data = 'failed to write failover key into DCS'
|
||||
status_code = 503
|
||||
# pyright thinks ``status_code`` can be ``None`` because ``parse_schedule`` call may return ``None``. However,
|
||||
# if that's the case, ``status_code`` will be overwritten somewhere between ``parse_schedule`` and
|
||||
# ``write_response`` calls.
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
assert isinstance(status_code, int)
|
||||
self.write_response(status_code, data)
|
||||
|
||||
status_code = status_code or 400
|
||||
self.write_response(status_code, data.format(action=action, leader=leader, candidate=candidate,
|
||||
cluster_name=self.server.patroni.postgresql.scope))
|
||||
|
||||
def do_POST_switchover(self) -> None:
|
||||
"""Handle a ``POST`` request to ``/switchover`` path.
|
||||
@@ -1183,20 +1097,18 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
self.command = mname
|
||||
return ret
|
||||
|
||||
def query(self, sql: str, *params: Any, **kwargs: Any) -> List[Tuple[Any, ...]]:
|
||||
"""Execute *sql* query with *params*.
|
||||
def query(self, sql: str, *params: Any, retry: bool = False) -> List[Tuple[Any, ...]]:
|
||||
"""Execute *sql* query with *params* and optionally return results.
|
||||
|
||||
:param sql: the SQL statement to be run.
|
||||
:param params: positional arguments to call :func:`RestApiServer.query` with.
|
||||
:param kwargs: can contain the key ``retry``. If the key is present its value should be a :class:`bool` which
|
||||
indicates whether the query should be retried upon failure or given up immediately.
|
||||
:param retry: whether the query should be retried upon failure or given up immediately.
|
||||
|
||||
:returns: a list of rows that were fetched from the database.
|
||||
"""
|
||||
if not kwargs.get('retry', False):
|
||||
if not retry:
|
||||
return self.server.query(sql, *params)
|
||||
retry = Retry(delay=1, retry_exceptions=PostgresConnectionException)
|
||||
return retry(self.server.query, sql, *params)
|
||||
return Retry(delay=1, retry_exceptions=PostgresConnectionException)(self.server.query, sql, *params)
|
||||
|
||||
def get_postgresql_status(self, retry: bool = False) -> Dict[str, Any]:
|
||||
"""Builds an object representing a status of "postgres".
|
||||
@@ -1262,8 +1174,8 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
" application_name, client_addr, w.state, sync_state, sync_priority"
|
||||
" FROM pg_catalog.pg_stat_get_wal_senders() w, pg_catalog.pg_stat_get_activity(pid)) AS ri")
|
||||
|
||||
row = self.query(stmt.format(postgresql.wal_name, postgresql.lsn_name), retry=retry)[0]
|
||||
|
||||
row = self.query(stmt.format(postgresql.wal_name, postgresql.lsn_name,
|
||||
postgresql.wal_flush), retry=retry)[0]
|
||||
result = {
|
||||
'state': postgresql.state,
|
||||
'postmaster_start_time': row[0],
|
||||
@@ -1368,7 +1280,7 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
||||
self.daemon = True
|
||||
|
||||
def query(self, sql: str, *params: Any) -> List[Tuple[Any, ...]]:
|
||||
"""Execute *sql* query with *params*.
|
||||
"""Execute *sql* query with *params* and optionally return results.
|
||||
|
||||
:param sql: the SQL statement to be run.
|
||||
:param params: positional arguments to be used as parameters for *sql*.
|
||||
@@ -1379,15 +1291,10 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
||||
:class:`psycopg.Error`: if had issues while executing *sql*.
|
||||
:class:`~patroni.exceptions.PostgresConnectionException`: if had issues while connecting to the database.
|
||||
"""
|
||||
cursor = None
|
||||
try:
|
||||
with self.patroni.postgresql.connection().cursor() as cursor:
|
||||
cursor.execute(sql.encode('utf-8'), params)
|
||||
return [r for r in cursor]
|
||||
except psycopg.Error as e:
|
||||
if cursor and cursor.connection.closed == 0:
|
||||
raise e
|
||||
raise PostgresConnectionException('connection problems')
|
||||
return self.patroni.postgresql.query(sql, *params, retry=False)
|
||||
except RetryFailedError as e:
|
||||
raise PostgresConnectionException(str(e))
|
||||
|
||||
@staticmethod
|
||||
def _set_fd_cloexec(fd: socket.socket) -> None:
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
Provides a case insensitive :class:`dict` and :class:`set` object types.
|
||||
"""
|
||||
from collections import OrderedDict
|
||||
from typing import Any, Collection, Dict, Iterator, MutableMapping, MutableSet, Optional
|
||||
from typing import Any, Collection, Dict, Iterator, KeysView, MutableMapping, MutableSet, Optional
|
||||
|
||||
|
||||
class CaseInsensitiveSet(MutableSet[str]):
|
||||
@@ -187,6 +187,13 @@ class CaseInsensitiveDict(MutableMapping[str, Any]):
|
||||
"""
|
||||
return CaseInsensitiveDict({v[0]: v[1] for v in self._values.values()})
|
||||
|
||||
def keys(self) -> KeysView[str]:
|
||||
"""Return a new view of the dict's keys.
|
||||
|
||||
:returns: a set-like object providing a view on the dict's keys
|
||||
"""
|
||||
return self._values.keys()
|
||||
|
||||
def __repr__(self) -> str:
|
||||
"""Get a string representation of the dict.
|
||||
|
||||
|
||||
+347
-70
@@ -1,3 +1,4 @@
|
||||
"""Facilities related to Patroni configuration."""
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
@@ -15,7 +16,6 @@ from .dcs import ClusterConfig, Cluster
|
||||
from .exceptions import ConfigParseError
|
||||
from .file_perm import pg_perm
|
||||
from .postgresql.config import ConfigHandler
|
||||
from .validator import IntValidator
|
||||
from .utils import deep_compare, parse_bool, parse_int, patch_config
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -36,121 +36,162 @@ _AUTH_ALLOWED_PARAMETERS = (
|
||||
|
||||
|
||||
def default_validator(conf: Dict[str, Any]) -> List[str]:
|
||||
"""Ensure *conf* is not empty.
|
||||
|
||||
Designed to be used as default validator for :class:`Config` objects, if no specific validator is provided.
|
||||
|
||||
:param conf: configuration to be validated.
|
||||
|
||||
:returns: an empty list -- :class:`Config` expects the validator to return a list of 0 or more issues found while
|
||||
validating the configuration.
|
||||
|
||||
:raises:
|
||||
:class:`ConfigParseError`: if *conf* is empty.
|
||||
"""
|
||||
if not conf:
|
||||
raise ConfigParseError("Config is empty.")
|
||||
return []
|
||||
|
||||
|
||||
class GlobalConfig(object):
|
||||
"""A class that wraps global configuration and provides convenient methods to access/check values.
|
||||
|
||||
"""A class that wrapps global configuration and provides convinient methods to access/check values.
|
||||
|
||||
It is instantiated by calling :func:`Config.global_config` method which picks either a
|
||||
configuration from provided :class:`Cluster` object (the most up-to-date) or from the
|
||||
local cache if :class::`ClusterConfig` is not initialized or doesn't have a valid config.
|
||||
It is instantiated either by calling :func:`get_global_config` or :meth:`Config.get_global_config`, which picks
|
||||
either a configuration from provided :class:`Cluster` object (the most up-to-date) or from the
|
||||
local cache if :class:`ClusterConfig` is not initialized or doesn't have a valid config.
|
||||
"""
|
||||
|
||||
def __init__(self, config: Dict[str, Any]) -> None:
|
||||
"""Initialize :class:`GlobalConfig` object.
|
||||
"""Initialize :class:`GlobalConfig` object with given *config*.
|
||||
|
||||
:param config: current configuration either from
|
||||
:class:`ClusterConfig` or from :class:`Config.dynamic_configuration`
|
||||
:class:`ClusterConfig` or from :func:`Config.dynamic_configuration`.
|
||||
"""
|
||||
self.__config = config
|
||||
|
||||
def get(self, name: str) -> Any:
|
||||
"""Gets global configuration value by name.
|
||||
"""Gets global configuration value by *name*.
|
||||
|
||||
:param name: parameter name
|
||||
:returns: configuration value or `None` if it is missing
|
||||
:param name: parameter name.
|
||||
|
||||
:returns: configuration value or ``None`` if it is missing.
|
||||
"""
|
||||
return self.__config.get(name)
|
||||
|
||||
def check_mode(self, mode: str) -> bool:
|
||||
"""Checks whether the certain parameter is enabled.
|
||||
|
||||
:param mode: parameter name could be: synchronous_mode, failsafe_mode, pause, check_timeline, and so on
|
||||
:returns: `True` if *mode* is enabled in the global configuration.
|
||||
:param mode: parameter name, e.g. ``synchronous_mode``, ``failsafe_mode``, ``pause``, ``check_timeline``, and
|
||||
so on.
|
||||
|
||||
:returns: ``True`` if parameter *mode* is enabled in the global configuration.
|
||||
"""
|
||||
return bool(parse_bool(self.__config.get(mode)))
|
||||
|
||||
@property
|
||||
def is_paused(self) -> bool:
|
||||
""":returns: `True` if cluster is in maintenance mode."""
|
||||
"""``True`` if cluster is in maintenance mode."""
|
||||
return self.check_mode('pause')
|
||||
|
||||
@property
|
||||
def is_synchronous_mode(self) -> bool:
|
||||
""":returns: `True` if synchronous replication is requested."""
|
||||
"""``True`` if synchronous replication is requested."""
|
||||
return self.check_mode('synchronous_mode')
|
||||
|
||||
@property
|
||||
def is_synchronous_mode_strict(self) -> bool:
|
||||
""":returns: `True` if at least one synchronous node is required."""
|
||||
"""``True`` if at least one synchronous node is required."""
|
||||
return self.check_mode('synchronous_mode_strict')
|
||||
|
||||
def get_standby_cluster_config(self) -> Union[Dict[str, Any], Any]:
|
||||
""":returns: "standby_cluster" configuration."""
|
||||
"""Get ``standby_cluster`` configuration.
|
||||
|
||||
:returns: a copy of ``standby_cluster`` configuration.
|
||||
"""
|
||||
return deepcopy(self.get('standby_cluster'))
|
||||
|
||||
@property
|
||||
def is_standby_cluster(self) -> bool:
|
||||
""":returns: `True` if global configuration has a valid "standby_cluster" section."""
|
||||
"""``True`` if global configuration has a valid ``standby_cluster`` section."""
|
||||
config = self.get_standby_cluster_config()
|
||||
return isinstance(config, dict) and\
|
||||
bool(config.get('host') or config.get('port') or config.get('restore_command'))
|
||||
|
||||
def get_int(self, name: str, default: int = 0) -> int:
|
||||
"""Gets current value from the global configuration and trying to return it as int.
|
||||
"""Gets current value of *name* from the global configuration and try to return it as :class:`int`.
|
||||
|
||||
:param name: name of the parameter
|
||||
:param default: default value if *name* is not in the configuration or invalid
|
||||
:returns: currently configured value from the global configuration or *default* if it is not set or invalid.
|
||||
:param name: name of the parameter.
|
||||
:param default: default value if *name* is not in the configuration or invalid.
|
||||
|
||||
:returns: currently configured value of *name* from the global configuration or *default* if it is not set or
|
||||
invalid.
|
||||
"""
|
||||
ret = parse_int(self.get(name))
|
||||
return default if ret is None else ret
|
||||
|
||||
@property
|
||||
def min_synchronous_nodes(self) -> int:
|
||||
""":returns: the minimal number of synchronous nodes based on whether strict mode is requested or not."""
|
||||
"""The minimal number of synchronous nodes based on whether ``synchronous_mode_strict`` is enabled or not."""
|
||||
return 1 if self.is_synchronous_mode_strict else 0
|
||||
|
||||
@property
|
||||
def synchronous_node_count(self) -> int:
|
||||
""":returns: currently configured value from the global configuration or 1 if it is not set or invalid."""
|
||||
"""Currently configured value of ``synchronous_node_count`` from the global configuration.
|
||||
|
||||
Assume ``1`` if it is not set or invalid.
|
||||
"""
|
||||
return max(self.get_int('synchronous_node_count', 1), self.min_synchronous_nodes)
|
||||
|
||||
@property
|
||||
def maximum_lag_on_failover(self) -> int:
|
||||
""":returns: currently configured value from the global configuration or 1048576 if it is not set or invalid."""
|
||||
"""Currently configured value of ``maximum_lag_on_failover`` from the global configuration.
|
||||
|
||||
Assume ``1048576`` if it is not set or invalid.
|
||||
"""
|
||||
return self.get_int('maximum_lag_on_failover', 1048576)
|
||||
|
||||
@property
|
||||
def maximum_lag_on_syncnode(self) -> int:
|
||||
""":returns: currently configured value from the global configuration or -1 if it is not set or invalid."""
|
||||
"""Currently configured value of ``maximum_lag_on_syncnode`` from the global configuration.
|
||||
|
||||
Assume ``-1`` if it is not set or invalid.
|
||||
"""
|
||||
return self.get_int('maximum_lag_on_syncnode', -1)
|
||||
|
||||
@property
|
||||
def primary_start_timeout(self) -> int:
|
||||
""":returns: currently configured value from the global configuration or 300 if it is not set or invalid."""
|
||||
"""Currently configured value of ``primary_start_timeout`` from the global configuration.
|
||||
|
||||
Assume ``300`` if it is not set or invalid.
|
||||
|
||||
.. note::
|
||||
``master_start_timeout`` is still supported to keep backward compatibility.
|
||||
"""
|
||||
default = 300
|
||||
return self.get_int('primary_start_timeout', default)\
|
||||
if 'primary_start_timeout' in self.__config else self.get_int('master_start_timeout', default)
|
||||
|
||||
@property
|
||||
def primary_stop_timeout(self) -> int:
|
||||
""":returns: currently configured value from the global configuration or 300 if it is not set or invalid."""
|
||||
"""Currently configured value of ``primary_stop_timeout`` from the global configuration.
|
||||
|
||||
Assume ``0`` if it is not set or invalid.
|
||||
|
||||
.. note::
|
||||
``master_stop_timeout`` is still supported to keep backward compatibility.
|
||||
"""
|
||||
default = 0
|
||||
return self.get_int('primary_stop_timeout', default)\
|
||||
if 'primary_stop_timeout' in self.__config else self.get_int('master_stop_timeout', default)
|
||||
|
||||
|
||||
def get_global_config(cluster: Union[Cluster, None], default: Optional[Dict[str, Any]] = None) -> GlobalConfig:
|
||||
def get_global_config(cluster: Optional[Cluster], default: Optional[Dict[str, Any]] = None) -> GlobalConfig:
|
||||
"""Instantiates :class:`GlobalConfig` based on the input.
|
||||
|
||||
:param cluster: the currently known cluster state from DCS
|
||||
:param default: default configuration, which will be used if there is no valid *cluster.config*
|
||||
:returns: :class:`GlobalConfig` object
|
||||
:param cluster: the currently known cluster state from DCS.
|
||||
:param default: default configuration, which will be used if there is no valid *cluster.config*.
|
||||
|
||||
:returns: :class:`GlobalConfig` object.
|
||||
"""
|
||||
# Try to protect from the case when DCS was wiped out
|
||||
if cluster and cluster.config and cluster.config.modify_version:
|
||||
@@ -161,23 +202,29 @@ def get_global_config(cluster: Union[Cluster, None], default: Optional[Dict[str,
|
||||
|
||||
|
||||
class Config(object):
|
||||
"""
|
||||
"""Handle Patroni configuration.
|
||||
|
||||
This class is responsible for:
|
||||
|
||||
1) Building and giving access to `effective_configuration` from:
|
||||
* `Config.__DEFAULT_CONFIG` -- some sane default values
|
||||
* `dynamic_configuration` -- configuration stored in DCS
|
||||
* `local_configuration` -- configuration from `config.yml` or environment
|
||||
1) Building and giving access to ``effective_configuration`` from:
|
||||
|
||||
2) Saving and loading `dynamic_configuration` into 'patroni.dynamic.json' file
|
||||
* ``Config.__DEFAULT_CONFIG`` -- some sane default values;
|
||||
* ``dynamic_configuration`` -- configuration stored in DCS;
|
||||
* ``local_configuration`` -- configuration from `config.yml` or environment.
|
||||
|
||||
2) Saving and loading ``dynamic_configuration`` into 'patroni.dynamic.json' file
|
||||
located in local_configuration['postgresql']['data_dir'] directory.
|
||||
This is necessary to be able to restore `dynamic_configuration`
|
||||
if DCS was accidentally wiped
|
||||
This is necessary to be able to restore ``dynamic_configuration``
|
||||
if DCS was accidentally wiped.
|
||||
|
||||
3) Loading of configuration file in the old format and converting it into new format
|
||||
3) Loading of configuration file in the old format and converting it into new format.
|
||||
|
||||
4) Mimicking some of the `dict` interfaces to make it possible
|
||||
to work with it as with the old `config` object.
|
||||
4) Mimicking some ``dict`` interfaces to make it possible
|
||||
to work with it as with the old ``config`` object.
|
||||
|
||||
:cvar PATRONI_CONFIG_VARIABLE: name of the environment variable that can be used to load Patroni configuration from.
|
||||
:cvar __CACHE_FILENAME: name of the file used to cache dynamic configuration under Postgres data directory.
|
||||
:cvar __DEFAULT_CONFIG: default configuration values for some Patroni settings.
|
||||
"""
|
||||
|
||||
PATRONI_CONFIG_VARIABLE = PATRONI_ENV_PREFIX + 'CONFIGURATION'
|
||||
@@ -195,21 +242,38 @@ class Config(object):
|
||||
'recovery_min_apply_delay': ''
|
||||
},
|
||||
'postgresql': {
|
||||
'bin_dir': '',
|
||||
'use_slots': True,
|
||||
'parameters': CaseInsensitiveDict({p: v[0] for p, v in ConfigHandler.CMDLINE_OPTIONS.items()
|
||||
if p not in ('wal_keep_segments', 'wal_keep_size')})
|
||||
if v[0] is not None and p not in ('wal_keep_segments', 'wal_keep_size')})
|
||||
}
|
||||
}
|
||||
|
||||
def __init__(self, configfile: str,
|
||||
validator: Optional[Callable[[Dict[str, Any]], List[str]]] = default_validator) -> None:
|
||||
"""Create a new instance of :class:`Config` and validate the loaded configuration using *validator*.
|
||||
|
||||
.. note::
|
||||
Patroni will read configuration from these locations in this order:
|
||||
|
||||
* file or directory path passed as command-line argument (*configfile*), if it exists and the file or
|
||||
files found in the directory can be parsed (see :meth:`~Config._load_config_path`), otherwise
|
||||
* YAML file passed via the environment variable (see :cvar:`PATRONI_CONFIG_VARIABLE`), if the referenced
|
||||
file exists and can be parsed, otherwise
|
||||
* from configuration values defined as environment variables, see
|
||||
:meth:`~Config._build_environment_configuration`.
|
||||
|
||||
:param configfile: path to Patroni configuration file.
|
||||
:param validator: function used to validate Patroni configuration. It should receive a dictionary which
|
||||
represents Patroni configuration, and return a list of zero or more error messages based on validation.
|
||||
|
||||
:raises:
|
||||
:class:`ConfigParseError`: if any issue is reported by *validator*.
|
||||
"""
|
||||
self._modify_version = -1
|
||||
self._dynamic_configuration = {}
|
||||
|
||||
self.__environment_configuration = self._build_environment_configuration()
|
||||
|
||||
# Patroni reads the configuration from the command-line argument if it exists, otherwise from the environment
|
||||
self._config_file = configfile if configfile and os.path.exists(configfile) else None
|
||||
if self._config_file:
|
||||
self._local_configuration = self._load_config_file()
|
||||
@@ -230,17 +294,43 @@ class Config(object):
|
||||
self._cache_needs_saving = False
|
||||
|
||||
@property
|
||||
def config_file(self) -> Union[str, None]:
|
||||
def config_file(self) -> Optional[str]:
|
||||
"""Path to Patroni configuration file, if any, else ``None``."""
|
||||
return self._config_file
|
||||
|
||||
@property
|
||||
def dynamic_configuration(self) -> Dict[str, Any]:
|
||||
"""Deep copy of cached Patroni dynamic configuration."""
|
||||
return deepcopy(self._dynamic_configuration)
|
||||
|
||||
def _load_config_path(self, path: str) -> Dict[str, Any]:
|
||||
@property
|
||||
def local_configuration(self) -> Dict[str, Any]:
|
||||
"""Deep copy of cached Patroni local configuration.
|
||||
|
||||
:returns: copy of :attr:`~Config._local_configuration`
|
||||
"""
|
||||
If path is a file, loads the yml file pointed to by path.
|
||||
If path is a directory, loads all yml files in that directory in alphabetical order
|
||||
return deepcopy(dict(self._local_configuration))
|
||||
|
||||
@classmethod
|
||||
def get_default_config(cls) -> Dict[str, Any]:
|
||||
"""Deep copy default configuration.
|
||||
|
||||
:returns: copy of :attr:`~Config.__DEFAULT_CONFIG`
|
||||
"""
|
||||
return deepcopy(cls.__DEFAULT_CONFIG)
|
||||
|
||||
def _load_config_path(self, path: str) -> Dict[str, Any]:
|
||||
"""Load Patroni configuration file(s) from *path*.
|
||||
|
||||
If *path* is a file, load the yml file pointed to by *path*.
|
||||
If *path* is a directory, load all yml files in that directory in alphabetical order.
|
||||
|
||||
:param path: path to either an YAML configuration file, or to a folder containing YAML configuration files.
|
||||
|
||||
:returns: configuration after reading the configuration file(s) from *path*.
|
||||
|
||||
:raises:
|
||||
:class:`ConfigParseError`: if *path* is invalid.
|
||||
"""
|
||||
if os.path.isfile(path):
|
||||
files = [path]
|
||||
@@ -259,14 +349,18 @@ class Config(object):
|
||||
return overall_config
|
||||
|
||||
def _load_config_file(self) -> Dict[str, Any]:
|
||||
"""Loads config.yaml from filesystem and applies some values which were set via ENV"""
|
||||
"""Load configuration file(s) from filesystem and apply values which were set via environment variables.
|
||||
|
||||
:returns: final configuration after merging configuration file(s) and environment variables.
|
||||
"""
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
assert self._config_file is not None
|
||||
config = self._load_config_path(self._config_file)
|
||||
assert self.config_file is not None
|
||||
config = self._load_config_path(self.config_file)
|
||||
patch_config(config, self.__environment_configuration)
|
||||
return config
|
||||
|
||||
def _load_cache(self) -> None:
|
||||
"""Load dynamic configuration from ``patroni.dynamic.json``."""
|
||||
if os.path.isfile(self._cache_file):
|
||||
try:
|
||||
with open(self._cache_file) as f:
|
||||
@@ -275,6 +369,12 @@ class Config(object):
|
||||
logger.exception('Exception when loading file: %s', self._cache_file)
|
||||
|
||||
def save_cache(self) -> None:
|
||||
"""Save dynamic configuration to ``patroni.dynamic.json`` under Postgres data directory.
|
||||
|
||||
.. note::
|
||||
``patroni.dynamic.jsonXXXXXX`` is created as a temporary file and than renamed to ``patroni.dynamic.json``,
|
||||
where ``XXXXXX`` is a random suffix.
|
||||
"""
|
||||
if self._cache_needs_saving:
|
||||
tmpfile = fd = None
|
||||
try:
|
||||
@@ -301,9 +401,16 @@ class Config(object):
|
||||
|
||||
# configuration could be either ClusterConfig or dict
|
||||
def set_dynamic_configuration(self, configuration: Union[ClusterConfig, Dict[str, Any]]) -> bool:
|
||||
"""Set dynamic configuration values with given *configuration*.
|
||||
|
||||
:param configuration: new dynamic configuration values. Supports :class:`dict` for backward compatibility.
|
||||
|
||||
:returns: ``True`` if changes have been detected between current dynamic configuration and the new dynamic
|
||||
*configuration*, ``False`` otherwise.
|
||||
"""
|
||||
if isinstance(configuration, ClusterConfig):
|
||||
if self._modify_version == configuration.modify_version:
|
||||
return False # If the version didn't changed there is nothing to do
|
||||
return False # If the version didn't change there is nothing to do
|
||||
self._modify_version = configuration.modify_version
|
||||
configuration = configuration.data
|
||||
|
||||
@@ -319,6 +426,14 @@ class Config(object):
|
||||
return False
|
||||
|
||||
def reload_local_configuration(self) -> Optional[bool]:
|
||||
"""Reload configuration values from the configuration file(s).
|
||||
|
||||
.. note::
|
||||
Designed to be used when user applies changes to configuration file(s), so Patroni can use the new values
|
||||
with a reload instead of a restart.
|
||||
|
||||
:returns: ``True`` if changes have been detected between current local configuration
|
||||
"""
|
||||
if self.config_file:
|
||||
try:
|
||||
configuration = self._load_config_file()
|
||||
@@ -334,16 +449,46 @@ class Config(object):
|
||||
|
||||
@staticmethod
|
||||
def _process_postgresql_parameters(parameters: Dict[str, Any], is_local: bool = False) -> Dict[str, Any]:
|
||||
"""Process Postgres *parameters*.
|
||||
|
||||
.. note::
|
||||
If *is_local* configuration discard any setting from *parameters* that is listed under
|
||||
:attr:`~patroni.postgresql.config.ConfigHandler.CMDLINE_OPTIONS` as those are supposed to be set only
|
||||
through dynamic configuration.
|
||||
|
||||
When setting parameters from :attr:`~patroni.postgresql.config.ConfigHandler.CMDLINE_OPTIONS` through
|
||||
dynamic configuration their value will be validated as per the validator defined in that very same
|
||||
attribute entry. If the given value cannot be validated, a warning will be logged and the default value of
|
||||
the GUC will be used instead.
|
||||
|
||||
Some parameters from :attr:`~patroni.postgresql.config.ConfigHandler.CMDLINE_OPTIONS` cannot be set even if
|
||||
not *is_local* configuration:
|
||||
|
||||
* ``listen_addresses``: inferred from ``postgresql.listen`` local configuration or from
|
||||
``PATRONI_POSTGRESQL_LISTEN`` environment variable;
|
||||
* ``port``: inferred from ``postgresql.listen`` local configuration or from
|
||||
``PATRONI_POSTGRESQL_LISTEN`` environment variable;
|
||||
* ``cluster_name``: set through ``scope`` local configuration or through ``PATRONI_SCOPE`` environment
|
||||
variable;
|
||||
* ``hot_standby``: always enabled;
|
||||
* ``wal_log_hints``: always enabled.
|
||||
|
||||
:param parameters: Postgres parameters to be processed. Should be the parsed YAML value of
|
||||
``postgresql.parameters`` configuration, either from local or from dynamic configuration.
|
||||
|
||||
:param is_local: should be ``True`` if *parameters* refers to local configuration, or ``False`` if *parameters*
|
||||
refers to dynamic configuration.
|
||||
|
||||
:returns: new value for ``postgresql.parameters`` after processing and validating *parameters*.
|
||||
"""
|
||||
pg_params: Dict[str, Any] = {}
|
||||
|
||||
for name, value in (parameters or {}).items():
|
||||
if name not in ConfigHandler.CMDLINE_OPTIONS:
|
||||
pg_params[name] = value
|
||||
elif not is_local:
|
||||
validator = ConfigHandler.CMDLINE_OPTIONS[name][1]
|
||||
if validator(value):
|
||||
int_val = parse_int(value) if isinstance(validator, IntValidator) else None
|
||||
pg_params[name] = int_val if isinstance(int_val, int) else value
|
||||
if ConfigHandler.CMDLINE_OPTIONS[name][1](value):
|
||||
pg_params[name] = value
|
||||
else:
|
||||
logger.warning("postgresql parameter %s=%s failed validation, defaulting to %s",
|
||||
name, value, ConfigHandler.CMDLINE_OPTIONS[name][0])
|
||||
@@ -351,7 +496,33 @@ class Config(object):
|
||||
return pg_params
|
||||
|
||||
def _safe_copy_dynamic_configuration(self, dynamic_configuration: Dict[str, Any]) -> Dict[str, Any]:
|
||||
config = deepcopy(self.__DEFAULT_CONFIG)
|
||||
"""Create a copy of *dynamic_configuration*.
|
||||
|
||||
Merge *dynamic_configuration* with :attr:`__DEFAULT_CONFIG` (*dynamic_configuration* takes precedence), and
|
||||
process ``postgresql.parameters`` from *dynamic_configuration* through :func:`_process_postgresql_parameters`,
|
||||
if present.
|
||||
|
||||
.. note::
|
||||
The following settings are not allowed in ``postgresql`` section as they are intended to be local
|
||||
configuration, and are removed if present:
|
||||
|
||||
* ``connect_address``;
|
||||
* ``proxy_address``;
|
||||
* ``listen``;
|
||||
* ``config_dir``;
|
||||
* ``data_dir``;
|
||||
* ``pgpass``;
|
||||
* ``authentication``;
|
||||
|
||||
Besides that any setting present in *dynamic_configuration* but absent from :attr:`__DEFAULT_CONFIG` is
|
||||
discarded.
|
||||
|
||||
:param dynamic_configuration: Patroni dynamic configuration.
|
||||
|
||||
:returns: copy of *dynamic_configuration*, merged with default dynamic configuration and with some sanity checks
|
||||
performed over it.
|
||||
"""
|
||||
config = self.get_default_config()
|
||||
|
||||
for name, value in dynamic_configuration.items():
|
||||
if name == 'postgresql':
|
||||
@@ -371,9 +542,25 @@ class Config(object):
|
||||
|
||||
@staticmethod
|
||||
def _build_environment_configuration() -> Dict[str, Any]:
|
||||
"""Get local configuration settings that were specified through environment variables.
|
||||
|
||||
:returns: dictionary containing the found environment variables and their values, respecting the expected
|
||||
structure of Patroni configuration.
|
||||
"""
|
||||
ret: Dict[str, Any] = defaultdict(dict)
|
||||
|
||||
def _popenv(name: str) -> Union[str, None]:
|
||||
def _popenv(name: str) -> Optional[str]:
|
||||
"""Get value of environment variable *name*.
|
||||
|
||||
.. note::
|
||||
*name* is prefixed with :data:`~patroni.PATRONI_ENV_PREFIX` when searching in the environment.
|
||||
|
||||
Also, the corresponding environment variable is removed from the environment upon reading its value.
|
||||
|
||||
:param name: name of the environment variable.
|
||||
|
||||
:returns: value of *name*, if present in the environment, otherwise ``None``.
|
||||
"""
|
||||
return os.environ.pop(PATRONI_ENV_PREFIX + name.upper(), None)
|
||||
|
||||
for param in ('name', 'namespace', 'scope'):
|
||||
@@ -382,6 +569,23 @@ class Config(object):
|
||||
ret[param] = value
|
||||
|
||||
def _fix_log_env(name: str, oldname: str) -> None:
|
||||
"""Normalize a log related environment variable.
|
||||
|
||||
.. note::
|
||||
Patroni used to support different names for log related environment variables in the past. As the
|
||||
environment variables were renamed, this function takes care of mapping and normalizing the environment.
|
||||
|
||||
*name* is prefixed with :data:`~patroni.PATRONI_ENV_PREFIX` and ``LOG`` when searching in the
|
||||
environment.
|
||||
|
||||
*oldname* is prefixed with :data:`~patroni.PATRONI_ENV_PREFIX` when searching in the environment.
|
||||
|
||||
If both *name* and *oldname* are set in the environment, *name* takes precedence.
|
||||
|
||||
:param name: new name of a log related environment variable.
|
||||
:param oldname: original name of a log related environment variable.
|
||||
:type oldname: str
|
||||
"""
|
||||
value = _popenv(oldname)
|
||||
name = PATRONI_ENV_PREFIX + 'LOG_' + name.upper()
|
||||
if value and name not in os.environ:
|
||||
@@ -391,6 +595,15 @@ class Config(object):
|
||||
_fix_log_env(name, oldname)
|
||||
|
||||
def _set_section_values(section: str, params: List[str]) -> None:
|
||||
"""Get value of *params* environment variables that are related with *section*.
|
||||
|
||||
.. note::
|
||||
The values are retrieved from the environment and updated directly into the returning dictionary of
|
||||
:func:`_build_environment_configuration`.
|
||||
|
||||
:param section: configuration section the *params* belong to.
|
||||
:param params: name of the Patroni settings.
|
||||
"""
|
||||
for param in params:
|
||||
value = _popenv(section + '_' + param)
|
||||
if value:
|
||||
@@ -412,6 +625,7 @@ class Config(object):
|
||||
if value:
|
||||
ret['postgresql'].setdefault('bin_name', {})[binary] = value
|
||||
|
||||
# parse all values retrieved from the environment as Python objects, according to the expected type
|
||||
for first, second in (('restapi', 'allowlist_include_members'), ('ctl', 'insecure')):
|
||||
value = ret.get(first, {}).pop(second, None)
|
||||
if value:
|
||||
@@ -428,7 +642,13 @@ class Config(object):
|
||||
if value is not None:
|
||||
ret[first][second] = value
|
||||
|
||||
def _parse_list(value: str) -> Union[List[str], None]:
|
||||
def _parse_list(value: str) -> Optional[List[str]]:
|
||||
"""Parse an YAML list *value* as a :class:`list`.
|
||||
|
||||
:param value: YAML list as a string.
|
||||
|
||||
:returns: *value* as :class:`list`.
|
||||
"""
|
||||
if not (value.strip().startswith('-') or '[' in value):
|
||||
value = '[{0}]'.format(value)
|
||||
try:
|
||||
@@ -444,7 +664,13 @@ class Config(object):
|
||||
if value:
|
||||
ret[first][second] = value
|
||||
|
||||
def _parse_dict(value: str) -> Union[Dict[str, Any], None]:
|
||||
def _parse_dict(value: str) -> Optional[Dict[str, Any]]:
|
||||
"""Parse an YAML dictionary *value* as a :class:`dict`.
|
||||
|
||||
:param value: YAML dictionary as a string.
|
||||
|
||||
:returns: *value* as :class:`dict`.
|
||||
"""
|
||||
if not value.strip().startswith('{'):
|
||||
value = '{{{0}}}'.format(value)
|
||||
try:
|
||||
@@ -461,9 +687,16 @@ class Config(object):
|
||||
if value:
|
||||
ret[first][second] = value
|
||||
|
||||
def _get_auth(name: str, params: Optional[Collection[str]] = None) -> Dict[str, str]:
|
||||
def _get_auth(name: str, params: Collection[str] = _AUTH_ALLOWED_PARAMETERS[:2]) -> Dict[str, str]:
|
||||
"""Get authorization related environment variables *params* from section *name*.
|
||||
|
||||
:param name: name of a configuration section that may contain authorization *params*.
|
||||
:param params: the authorization settings that may be set under section *name*.
|
||||
|
||||
:returns: dictionary containing environment values for authorization *params* of section *name*.
|
||||
"""
|
||||
ret: Dict[str, str] = {}
|
||||
for param in params or _AUTH_ALLOWED_PARAMETERS[:2]:
|
||||
for param in params:
|
||||
value = _popenv(name + '_' + param)
|
||||
if value:
|
||||
ret[param] = value
|
||||
@@ -486,7 +719,7 @@ class Config(object):
|
||||
for param in list(os.environ.keys()):
|
||||
if param.startswith(PATRONI_ENV_PREFIX):
|
||||
# PATRONI_(ETCD|CONSUL|ZOOKEEPER|EXHIBITOR|...)_(HOSTS?|PORT|..)
|
||||
name, suffix = (param[8:].split('_', 1) + [''])[:2]
|
||||
name, suffix = (param[len(PATRONI_ENV_PREFIX):].split('_', 1) + [''])[:2]
|
||||
if suffix in ('HOST', 'HOSTS', 'PORT', 'USE_PROXIES', 'PROTOCOL', 'SRV', 'SRV_SUFFIX', 'URL', 'PROXY',
|
||||
'CACERT', 'CERT', 'KEY', 'VERIFY', 'TOKEN', 'CHECKS', 'DC', 'CONSISTENCY',
|
||||
'REGISTER_SERVICE', 'SERVICE_CHECK_INTERVAL', 'SERVICE_CHECK_TLS_SERVER_NAME',
|
||||
@@ -517,14 +750,14 @@ class Config(object):
|
||||
users = {}
|
||||
for param in list(os.environ.keys()):
|
||||
if param.startswith(PATRONI_ENV_PREFIX):
|
||||
name, suffix = (param[8:].rsplit('_', 1) + [''])[:2]
|
||||
name, suffix = (param[len(PATRONI_ENV_PREFIX):].rsplit('_', 1) + [''])[:2]
|
||||
# PATRONI_<username>_PASSWORD=<password>, PATRONI_<username>_OPTIONS=<option1,option2,...>
|
||||
# CREATE USER "<username>" WITH <OPTIONS> PASSWORD '<password>'
|
||||
if name and suffix == 'PASSWORD':
|
||||
password = os.environ.pop(param)
|
||||
if password:
|
||||
users[name] = {'password': password}
|
||||
options = os.environ.pop(param[:-9] + '_OPTIONS', None)
|
||||
options = os.environ.pop(param[:-9] + '_OPTIONS', None) # replace "_PASSWORD" with "_OPTIONS"
|
||||
options = options and _parse_list(options)
|
||||
if options:
|
||||
users[name]['options'] = options
|
||||
@@ -535,6 +768,16 @@ class Config(object):
|
||||
|
||||
def _build_effective_configuration(self, dynamic_configuration: Dict[str, Any],
|
||||
local_configuration: Dict[str, Union[Dict[str, Any], Any]]) -> Dict[str, Any]:
|
||||
"""Build effective configuration by merging *dynamic_configuration* and *local_configuration*.
|
||||
|
||||
.. note::
|
||||
*local_configuration* takes precedence over *dynamic_configuration* if a setting is defined in both.
|
||||
|
||||
:param dynamic_configuration: Patroni dynamic configuration.
|
||||
:param local_configuration: Patroni local configuration.
|
||||
|
||||
:returns: _description_
|
||||
"""
|
||||
config = self._safe_copy_dynamic_configuration(dynamic_configuration)
|
||||
for name, value in local_configuration.items():
|
||||
if name == 'citus': # remove invalid citus configuration
|
||||
@@ -598,23 +841,57 @@ class Config(object):
|
||||
return config
|
||||
|
||||
def get(self, key: str, default: Optional[Any] = None) -> Any:
|
||||
"""Get effective value of ``key`` setting from Patroni configuration root.
|
||||
|
||||
Designed to work the same way as :func:`dict.get`.
|
||||
|
||||
:param key: name of the setting.
|
||||
:param default: default value if *key* is not present in the effective configuration.
|
||||
|
||||
:returns: value of *key*, if present in the effective configuration, otherwise *default*.
|
||||
"""
|
||||
return self.__effective_configuration.get(key, default)
|
||||
|
||||
def __contains__(self, key: str) -> bool:
|
||||
"""Check if setting *key* is present in the effective configuration.
|
||||
|
||||
Designed to work the same way as :func:`dict.__contains__`.
|
||||
|
||||
:param key: name of the setting to be checked.
|
||||
|
||||
:returns: ``True`` if setting *key* exists in effective configuration, else ``False``.
|
||||
"""
|
||||
return key in self.__effective_configuration
|
||||
|
||||
def __getitem__(self, key: str) -> Any:
|
||||
"""Get value of setting *key* from effective configuration.
|
||||
|
||||
Designed to work the same way as :func:`dict.__getitem__`.
|
||||
|
||||
:param key: name of the setting.
|
||||
|
||||
:returns: value of setting *key*.
|
||||
|
||||
:raises:
|
||||
:class:`KeyError`: if *key* is not present in effective configuration.
|
||||
"""
|
||||
return self.__effective_configuration[key]
|
||||
|
||||
def copy(self) -> Dict[str, Any]:
|
||||
"""Get a deep copy of effective Patroni configuration.
|
||||
|
||||
:returns: a deep copy of the Patroni configuration.
|
||||
"""
|
||||
return deepcopy(self.__effective_configuration)
|
||||
|
||||
def get_global_config(self, cluster: Union[Cluster, None]) -> GlobalConfig:
|
||||
def get_global_config(self, cluster: Optional[Cluster]) -> GlobalConfig:
|
||||
"""Instantiate :class:`GlobalConfig` based on input.
|
||||
|
||||
Use the configuration from provided *cluster* (the most up-to-date) or from the
|
||||
local cache if *cluster.config* is not initialized or doesn't have a valid config.
|
||||
:param cluster: the currently known cluster state from DCS
|
||||
:returns: :class:`GlobalConfig` object
|
||||
|
||||
:param cluster: the currently known cluster state from DCS.
|
||||
|
||||
:returns: :class:`GlobalConfig` object.
|
||||
"""
|
||||
return get_global_config(cluster, self._dynamic_configuration)
|
||||
|
||||
@@ -0,0 +1,463 @@
|
||||
"""patroni ``--generate-config`` machinery."""
|
||||
import abc
|
||||
import logging
|
||||
import os
|
||||
import psutil
|
||||
import socket
|
||||
import sys
|
||||
import yaml
|
||||
|
||||
from getpass import getuser, getpass
|
||||
from contextlib import contextmanager
|
||||
from typing import Any, Dict, Iterator, List, Optional, Tuple, TYPE_CHECKING, Union
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
from psycopg import Cursor
|
||||
from psycopg2 import cursor
|
||||
|
||||
from . import psycopg
|
||||
from .config import Config
|
||||
from .exceptions import PatroniException
|
||||
from .postgresql.config import ConfigHandler, parse_dsn
|
||||
from .postgresql.misc import postgres_major_version_to_int
|
||||
from .utils import get_major_version, parse_bool, patch_config, read_stripped
|
||||
|
||||
|
||||
# Mapping between the libpq connection parameters and the environment variables.
|
||||
# This dict should be kept in sync with `patroni.utils._AUTH_ALLOWED_PARAMETERS`
|
||||
# (we use "username" in the Patroni config for some reason, other parameter names are the same).
|
||||
_AUTH_ALLOWED_PARAMETERS_MAPPING = {
|
||||
'user': 'PGUSER',
|
||||
'password': 'PGPASSWORD',
|
||||
'sslmode': 'PGSSLMODE',
|
||||
'sslcert': 'PGSSLCERT',
|
||||
'sslkey': 'PGSSLKEY',
|
||||
'sslpassword': '',
|
||||
'sslrootcert': 'PGSSLROOTCERT',
|
||||
'sslcrl': 'PGSSLCRL',
|
||||
'sslcrldir': 'PGSSLCRLDIR',
|
||||
'gssencmode': 'PGGSSENCMODE',
|
||||
'channel_binding': 'PGCHANNELBINDING'
|
||||
}
|
||||
_NO_VALUE_MSG = '#FIXME'
|
||||
|
||||
|
||||
def get_address() -> Tuple[str, str]:
|
||||
"""Try to get hostname and the ip address for it returned by :func:`~socket.gethostname`.
|
||||
|
||||
.. note::
|
||||
Can also return local ip.
|
||||
|
||||
:returns: tuple consisting of the hostname returned by :func:`~socket.gethostname`
|
||||
and the first element in the sorted list of the addresses returned by :func:`~socket.getaddrinfo`.
|
||||
Sorting guarantees it will prefer IPv4.
|
||||
If an exception occured, hostname and ip values are equal to :data:`~patroni.config_generator._NO_VALUE_MSG`.
|
||||
"""
|
||||
hostname = None
|
||||
try:
|
||||
hostname = socket.gethostname()
|
||||
return hostname, sorted(socket.getaddrinfo(hostname, 0, socket.AF_UNSPEC, socket.SOCK_STREAM, 0),
|
||||
key=lambda x: x[0])[0][4][0]
|
||||
except Exception as err:
|
||||
logging.warning('Failed to obtain address: %r', err)
|
||||
return _NO_VALUE_MSG, _NO_VALUE_MSG
|
||||
|
||||
|
||||
class AbstractConfigGenerator(abc.ABC):
|
||||
"""Object representing the generated Patroni config.
|
||||
|
||||
:ivar output_file: full path to the output file to be used.
|
||||
:ivar pg_major: integer representation of the major PostgreSQL version.
|
||||
:ivar config: dictionary used for the generated configuration storage.
|
||||
"""
|
||||
|
||||
_HOSTNAME, _IP = get_address()
|
||||
|
||||
def __init__(self, output_file: Optional[str]) -> None:
|
||||
"""Set up the output file (if passed), helper vars and the minimal config structure.
|
||||
|
||||
:param output_file: full path to the output file to be used.
|
||||
"""
|
||||
self.output_file = output_file
|
||||
self.pg_major = 0
|
||||
self.config = self.get_template_config()
|
||||
|
||||
self.generate()
|
||||
|
||||
@classmethod
|
||||
def get_template_config(cls) -> Dict[str, Any]:
|
||||
"""Generate a template config for further extension (e.g. in the inherited classes).
|
||||
|
||||
:returns: dictionary with the values gathered from Patroni env, hopefully defined hostname and ip address
|
||||
(otherwise set to :data:`~patroni.config_generator._NO_VALUE_MSG`), and some sane defaults.
|
||||
"""
|
||||
template_config: Dict[str, Any] = {
|
||||
'scope': _NO_VALUE_MSG,
|
||||
'name': cls._HOSTNAME,
|
||||
'postgresql': {
|
||||
'data_dir': _NO_VALUE_MSG,
|
||||
'connect_address': _NO_VALUE_MSG + ':5432',
|
||||
'listen': _NO_VALUE_MSG + ':5432',
|
||||
'bin_dir': '',
|
||||
'authentication': {
|
||||
'superuser': {
|
||||
'username': 'postgres',
|
||||
'password': _NO_VALUE_MSG
|
||||
},
|
||||
'replication': {
|
||||
'username': 'replicator',
|
||||
'password': _NO_VALUE_MSG
|
||||
}
|
||||
}
|
||||
},
|
||||
'restapi': {
|
||||
'connect_address': cls._IP + ':8008',
|
||||
'listen': cls._IP + ':8008'
|
||||
}
|
||||
}
|
||||
|
||||
dynamic_config = Config.get_default_config()
|
||||
# to properly dump CaseInsensitiveDict as YAML later
|
||||
dynamic_config['postgresql']['parameters'] = dict(dynamic_config['postgresql']['parameters'])
|
||||
config = Config('', None).local_configuration # Get values from env
|
||||
config.setdefault('bootstrap', {})['dcs'] = dynamic_config
|
||||
config.setdefault('postgresql', {})
|
||||
del config['bootstrap']['dcs']['standby_cluster']
|
||||
|
||||
patch_config(template_config, config)
|
||||
return template_config
|
||||
|
||||
@abc.abstractmethod
|
||||
def generate(self) -> None:
|
||||
"""Generate config and store in :attr:`~AbstractConfigGenerator.config`."""
|
||||
|
||||
def write_config(self) -> None:
|
||||
"""Write current :attr:`~AbstractConfigGenerator.config` to the output file if provided, to stdout otherwise."""
|
||||
if self.output_file:
|
||||
dir_path = os.path.dirname(self.output_file)
|
||||
if dir_path and not os.path.isdir(dir_path):
|
||||
os.makedirs(dir_path)
|
||||
with open(self.output_file, 'w', encoding='UTF-8') as output_file:
|
||||
yaml.safe_dump(self.config, output_file, default_flow_style=False, allow_unicode=True)
|
||||
else:
|
||||
yaml.safe_dump(self.config, sys.stdout, default_flow_style=False, allow_unicode=True)
|
||||
|
||||
|
||||
class SampleConfigGenerator(AbstractConfigGenerator):
|
||||
"""Object representing the generated sample Patroni config.
|
||||
|
||||
Sane defults are used based on the gathered PG version.
|
||||
"""
|
||||
|
||||
@property
|
||||
def get_auth_method(self) -> str:
|
||||
"""Return the preferred authentication method for a specific PG version if provided or the default ``md5``.
|
||||
|
||||
:returns: :class:`str` value for the preferred authentication method.
|
||||
"""
|
||||
return 'scram-sha-256' if self.pg_major and self.pg_major >= 100000 else 'md5'
|
||||
|
||||
def _get_int_major_version(self) -> int:
|
||||
"""Get major PostgreSQL version from the binary as an integer.
|
||||
|
||||
:returns: an integer PostgreSQL major version representation gathered from the PostgreSQL binary.
|
||||
See :func:`~patroni.postgresql.misc.postgres_major_version_to_int` and
|
||||
:func:`~patroni.utils.get_major_version`.
|
||||
"""
|
||||
postgres_bin = ((self.config.get('postgresql') or {}).get('bin_name') or {}).get('postgres', 'postgres')
|
||||
return postgres_major_version_to_int(get_major_version(self.config['postgresql'].get('bin_dir'), postgres_bin))
|
||||
|
||||
def generate(self) -> None:
|
||||
"""Generate sample config using some sane defaults and update :attr:`~AbstractConfigGenerator.config`."""
|
||||
self.pg_major = self._get_int_major_version()
|
||||
|
||||
self.config['postgresql']['parameters'] = {'password_encryption': self.get_auth_method}
|
||||
username = self.config["postgresql"]["authentication"]["replication"]["username"]
|
||||
self.config['postgresql']['pg_hba'] = [
|
||||
f'host all all all {self.get_auth_method}',
|
||||
f'host replication {username} all {self.get_auth_method}'
|
||||
]
|
||||
|
||||
# add version-specific configuration
|
||||
wal_keep_param = 'wal_keep_segments' if self.pg_major < 130000 else 'wal_keep_size'
|
||||
self.config['bootstrap']['dcs']['postgresql']['parameters'][wal_keep_param] = \
|
||||
ConfigHandler.CMDLINE_OPTIONS[wal_keep_param][0]
|
||||
|
||||
self.config['bootstrap']['dcs']['postgresql']['use_pg_rewind'] = True
|
||||
if self.pg_major >= 110000:
|
||||
self.config['postgresql']['authentication'].setdefault(
|
||||
'rewind', {'username': 'rewind_user'}).setdefault('password', _NO_VALUE_MSG)
|
||||
|
||||
|
||||
class RunningClusterConfigGenerator(AbstractConfigGenerator):
|
||||
"""Object representing the Patroni config generated using information gathered from the running instance.
|
||||
|
||||
:ivar dsn: DSN string for the local instance to get GUC values from (if provided).
|
||||
:ivar parsed_dsn: DSN string parsed into a dictionary (see :func:`~patroni.postgresql.config.parse_dsn`).
|
||||
"""
|
||||
|
||||
def __init__(self, output_file: Optional[str] = None, dsn: Optional[str] = None) -> None:
|
||||
"""Additionally store the passed dsn (if any) in both original and parsed version and run config generation.
|
||||
|
||||
:param output_file: full path to the output file to be used.
|
||||
:param dsn: DSN string for the local instance to get GUC values from.
|
||||
|
||||
:raises:
|
||||
:exc:`~patroni.exceptions.PatroniException`: if DSN parsing failed.
|
||||
"""
|
||||
self.dsn = dsn
|
||||
self.parsed_dsn = {}
|
||||
|
||||
super().__init__(output_file)
|
||||
|
||||
@property
|
||||
def _get_hba_conn_types(self) -> Tuple[str, ...]:
|
||||
"""Return the connection types allowed.
|
||||
|
||||
If :attr:`~RunningClusterConfigGenerator.pg_major` is defined, adds additional parameters
|
||||
for PostgreSQL version >=16.
|
||||
|
||||
:returns: tuple of the connection methods allowed.
|
||||
"""
|
||||
allowed_types = ('local', 'host', 'hostssl', 'hostnossl', 'hostgssenc', 'hostnogssenc')
|
||||
if self.pg_major and self.pg_major >= 160000:
|
||||
allowed_types += ('include', 'include_if_exists', 'include_dir')
|
||||
return allowed_types
|
||||
|
||||
@property
|
||||
def _required_pg_params(self) -> List[str]:
|
||||
"""PG configuration prameters that have to be always present in the generated config.
|
||||
|
||||
:returns: list of the parameter names.
|
||||
"""
|
||||
return ['hba_file', 'ident_file', 'config_file', 'data_directory'] + \
|
||||
list(ConfigHandler.CMDLINE_OPTIONS.keys())
|
||||
|
||||
def _get_bin_dir_from_running_instance(self) -> str:
|
||||
"""Define the directory postgres binaries reside using postmaster's pid executable.
|
||||
|
||||
:returns: path to the PostgreSQL binaries directory.
|
||||
|
||||
:raises:
|
||||
:exc:`~patroni.exceptions.PatroniException`: if:
|
||||
|
||||
* pid could not be obtained from the ``postmaster.pid`` file; or
|
||||
* :exc:`OSError` occured during ``postmaster.pid`` file handling; or
|
||||
* the obtained postmaster pid doesn't exist.
|
||||
"""
|
||||
postmaster_pid = None
|
||||
data_dir = self.config['postgresql']['data_dir']
|
||||
try:
|
||||
with open(f"{data_dir}/postmaster.pid", 'r') as pid_file:
|
||||
postmaster_pid = pid_file.readline()
|
||||
if not postmaster_pid:
|
||||
raise PatroniException('Failed to obtain postmaster pid from postmaster.pid file')
|
||||
postmaster_pid = int(postmaster_pid.strip())
|
||||
except OSError as err:
|
||||
raise PatroniException(f'Error while reading postmaster.pid file: {err}')
|
||||
try:
|
||||
return os.path.dirname(psutil.Process(postmaster_pid).exe())
|
||||
except psutil.NoSuchProcess:
|
||||
raise PatroniException("Obtained postmaster pid doesn't exist.")
|
||||
|
||||
@contextmanager
|
||||
def _get_connection_cursor(self) -> Iterator[Union['cursor', 'Cursor[Any]']]:
|
||||
"""Get cursor for the PG connection established based on the stored information.
|
||||
|
||||
:raises:
|
||||
:exc:`~patroni.exceptions.PatroniException`: if :exc:`psycopg.Error` occured.
|
||||
"""
|
||||
try:
|
||||
conn = psycopg.connect(dsn=self.dsn,
|
||||
password=self.config['postgresql']['authentication']['superuser']['password'])
|
||||
with conn.cursor() as cur:
|
||||
yield cur
|
||||
conn.close()
|
||||
except psycopg.Error as e:
|
||||
raise PatroniException(f'Failed to establish PostgreSQL connection: {e}')
|
||||
|
||||
def _set_pg_params(self, cur: Union['cursor', 'Cursor[Any]']) -> None:
|
||||
"""Extend :attr:`~RunningClusterConfigGenerator.config` with the actual PG GUCs values.
|
||||
|
||||
THe following GUC values are set:
|
||||
|
||||
* Non-internal having configuration file, postmaster command line or environment variable
|
||||
as a source.
|
||||
|
||||
* List of the always required parameters (see :meth:`~RunningClusterConfigGenerator._required_pg_params`).
|
||||
|
||||
:param cur: connection cursor to use.
|
||||
"""
|
||||
cur.execute("SELECT name, current_setting(name) FROM pg_settings "
|
||||
"WHERE context <> 'internal' "
|
||||
"AND source IN ('configuration file', 'command line', 'environment variable') "
|
||||
"AND category <> 'Write-Ahead Log / Recovery Target' "
|
||||
"AND setting <> '(disabled)' "
|
||||
"OR name = ANY(%s)", (self._required_pg_params,))
|
||||
|
||||
helper_dict = dict.fromkeys(['port', 'listen_addresses'])
|
||||
self.config['postgresql'].setdefault('parameters', {})
|
||||
for param, value in cur.fetchall():
|
||||
if param == 'data_directory':
|
||||
self.config['postgresql']['data_dir'] = value
|
||||
elif param == 'cluster_name' and value:
|
||||
self.config['scope'] = value
|
||||
elif param in ('archive_command', 'restore_command',
|
||||
'archive_cleanup_command', 'recovery_end_command',
|
||||
'ssl_passphrase_command', 'hba_file',
|
||||
'ident_file', 'config_file'):
|
||||
# write commands to the local config due to security implications
|
||||
# write hba/ident/config_file to local config to ensure they are not removed later
|
||||
self.config['postgresql']['parameters'][param] = value
|
||||
elif param in helper_dict:
|
||||
helper_dict[param] = value
|
||||
else:
|
||||
self.config['bootstrap']['dcs']['postgresql']['parameters'][param] = value
|
||||
|
||||
connect_port = self.parsed_dsn.get('port', os.getenv('PGPORT', helper_dict['port']))
|
||||
self.config['postgresql']['connect_address'] = f'{self._IP}:{connect_port}'
|
||||
self.config['postgresql']['listen'] = f'{helper_dict["listen_addresses"]}:{helper_dict["port"]}'
|
||||
|
||||
def _set_su_params(self) -> None:
|
||||
"""Extend :attr:`~RunningClusterConfigGenerator.config` with the superuser auth information.
|
||||
|
||||
Information set is based on the options used for connection.
|
||||
"""
|
||||
su_params: Dict[str, str] = {}
|
||||
for conn_param, env_var in _AUTH_ALLOWED_PARAMETERS_MAPPING.items():
|
||||
val = self.parsed_dsn.get(conn_param, os.getenv(env_var))
|
||||
if val:
|
||||
su_params[conn_param] = val
|
||||
patroni_env_su_username = ((self.config.get('authentication') or {}).get('superuser') or {}).get('username')
|
||||
patroni_env_su_pwd = ((self.config.get('authentication') or {}).get('superuser') or {}).get('password')
|
||||
# because we use "username" in the config for some reason
|
||||
su_params['username'] = su_params.pop('user', patroni_env_su_username) or getuser()
|
||||
su_params['password'] = su_params.get('password', patroni_env_su_pwd) or \
|
||||
getpass('Please enter the user password:')
|
||||
self.config['postgresql']['authentication'] = {
|
||||
'superuser': su_params,
|
||||
'replication': {'username': _NO_VALUE_MSG, 'password': _NO_VALUE_MSG}
|
||||
}
|
||||
|
||||
def _set_conf_files(self) -> None:
|
||||
"""Extend :attr:`~RunningClusterConfigGenerator.config` with ``pg_hba.conf`` and ``pg_ident.conf`` content.
|
||||
|
||||
.. note::
|
||||
This function only defines ``postgresql.pg_hba`` and ``postgresql.pg_ident`` when
|
||||
``hba_file`` and ``ident_file`` are set to the defaults. It may happen these files
|
||||
are located outside of ``PGDATA`` and Patroni doesn't have write permissions for them.
|
||||
|
||||
:raises:
|
||||
:exc:`~patroni.exceptions.PatroniException`: if :exc:`OSError` occured during the conf files handling.
|
||||
"""
|
||||
default_hba_path = os.path.join(self.config['postgresql']['data_dir'], 'pg_hba.conf')
|
||||
if self.config['postgresql']['parameters']['hba_file'] == default_hba_path:
|
||||
try:
|
||||
self.config['postgresql']['pg_hba'] = list(
|
||||
filter(lambda i: i and i.split()[0] in self._get_hba_conn_types, read_stripped(default_hba_path)))
|
||||
except OSError as err:
|
||||
raise PatroniException(f'Failed to read pg_hba.conf: {err}')
|
||||
|
||||
default_ident_path = os.path.join(self.config['postgresql']['data_dir'], 'pg_ident.conf')
|
||||
if self.config['postgresql']['parameters']['ident_file'] == default_ident_path:
|
||||
try:
|
||||
self.config['postgresql']['pg_ident'] = [i for i in read_stripped(default_ident_path)
|
||||
if i and not i.startswith('#')]
|
||||
except OSError as err:
|
||||
raise PatroniException(f'Failed to read pg_ident.conf: {err}')
|
||||
if not self.config['postgresql']['pg_ident']:
|
||||
del self.config['postgresql']['pg_ident']
|
||||
|
||||
def _enrich_config_from_running_instance(self) -> None:
|
||||
"""Extend :attr:`~RunningClusterConfigGenerator.config` with the values gathered from the running instance.
|
||||
|
||||
Retrieve the following information from the running PostgreSQL instance:
|
||||
|
||||
* superuser auth parameters (see :meth:`~RunningClusterConfigGenerator._set_su_params`);
|
||||
* some GUC values (see :meth:`~RunningClusterConfigGenerator._set_pg_params`);
|
||||
* ``postgresql.connect_address``, ``postgresql.listen``;
|
||||
* ``postgresql.pg_hba`` and ``postgresql.pg_ident`` (see :meth:`~RunningClusterConfigGenerator._set_conf_files`)
|
||||
|
||||
And redefine ``scope`` with the ``cluster_name`` GUC value if set.
|
||||
|
||||
:raises:
|
||||
:exc:`~patroni.exceptions.PatroniException`: if the provided user doesn't have superuser privileges.
|
||||
"""
|
||||
self._set_su_params()
|
||||
|
||||
with self._get_connection_cursor() as cur:
|
||||
self.pg_major = getattr(cur.connection, 'server_version', 0)
|
||||
|
||||
if not parse_bool(cur.connection.info.parameter_status('is_superuser')):
|
||||
raise PatroniException('The provided user does not have superuser privilege')
|
||||
|
||||
self._set_pg_params(cur)
|
||||
|
||||
self._set_conf_files()
|
||||
|
||||
def generate(self) -> None:
|
||||
"""Generate config using the info gathered from the specified running PG instance.
|
||||
|
||||
Result is written to :attr:`~RunningClusterConfigGenerator.config`.
|
||||
"""
|
||||
if self.dsn:
|
||||
self.parsed_dsn = parse_dsn(self.dsn) or {}
|
||||
if not self.parsed_dsn:
|
||||
raise PatroniException('Failed to parse DSN string')
|
||||
|
||||
self._enrich_config_from_running_instance()
|
||||
self.config['postgresql']['bin_dir'] = self._get_bin_dir_from_running_instance()
|
||||
|
||||
|
||||
def generate_config(output_file: str, sample: bool, dsn: Optional[str]) -> None:
|
||||
"""Generate Patroni configuration file.
|
||||
|
||||
Gather all the available non-internal GUC values having configuration file, postmaster command line or environment
|
||||
variable as a source and store them in the appropriate part of Patroni configuration (``postgresql.parameters`` or
|
||||
``bootstrap.dcs.postgresql.parameters``). Either the provided DSN (takes precedence) or PG ENV vars will be used
|
||||
for the connection. If password is not provided, it should be entered via prompt.
|
||||
|
||||
The created configuration contains:
|
||||
* ``scope``: ``cluster_name`` GUC value or ``PATRONI_SCOPE ENV`` variable value if available.
|
||||
* ``name``: ``PATRONI_NAME`` ENV variable value if set, otherwise hostname.
|
||||
|
||||
* ``bootstrap.dcs``: section with all the parameters (incl. the majority of PG GUCs) set to their default values
|
||||
defined by Patroni and adjusted by the source instances's configuration values.
|
||||
|
||||
* ``postgresql.parameters``: the source instance's ``archive_command``, ``restore_command``,
|
||||
``archive_cleanup_command``, ``recovery_end_command``, ``ssl_passphrase_command``, ``hba_file``, ``ident_file``,
|
||||
``config_file`` GUC values.
|
||||
|
||||
* ``postgresql.bin_dir``: path to Postgres binaries gathered from the running instance or, if not available,
|
||||
the value of ``PATRONI_POSTGRESQL_BIN_DIR`` ENV variable. Otherwise, an empty string.
|
||||
|
||||
* ``postgresql.datadir``: the value gathered from the corresponding PG GUC.
|
||||
* ``postgresql.listen``: source instance's ``listen_addresses`` and port GUC values.
|
||||
* ``postgresql.connect_address``: if possible, generated from the connection params.
|
||||
* ``postgresql.authentication``:
|
||||
|
||||
* superuser and replication users defined (if possible, usernames are set from the respective Patroni ENV vars,
|
||||
otherwise the default ``postgres`` and ``replicator`` values are used).
|
||||
If not a sample config, either DSN or PG ENV vars are used to define superuser authentication parameters.
|
||||
|
||||
* rewind user is defined only for sample config, if PG version can be defined and PG version is >=11
|
||||
(if possible, username is set from the respective Patroni ENV var).
|
||||
|
||||
* ``bootstrap.dcs.postgresql.use_pg_rewind`` set to ``True`` for a sample config only.
|
||||
* ``postgresql.pg_hba`` defaults or the lines gathered from the source instance's ``hba_file``.
|
||||
* ``postgresql.pg_ident`` the lines gathered from the source instance's ``ident_file``.
|
||||
|
||||
:param output_file: Full path to the configuration file to be used. If not provided, result is sent to ``stdout``.
|
||||
:param sample: Optional flag. If set, no source instance will be used - generate config with some sane defaults.
|
||||
:param dsn: Optional DSN string for the local instance to get GUC values from.
|
||||
"""
|
||||
try:
|
||||
if sample:
|
||||
config_generator = SampleConfigGenerator(output_file)
|
||||
else:
|
||||
config_generator = RunningClusterConfigGenerator(output_file, dsn)
|
||||
|
||||
config_generator.write_config()
|
||||
except PatroniException as e:
|
||||
sys.exit(str(e))
|
||||
except Exception as e:
|
||||
sys.exit(f'Unexpected exception: {e}')
|
||||
+84
-123
@@ -16,8 +16,6 @@ import click
|
||||
import codecs
|
||||
import copy
|
||||
import datetime
|
||||
import dateutil.parser
|
||||
import dateutil.tz
|
||||
import difflib
|
||||
import io
|
||||
import json
|
||||
@@ -46,10 +44,12 @@ try:
|
||||
except ImportError: # pragma: no cover
|
||||
from cdiff import markup_to_pager, PatchStream # pyright: ignore [reportMissingModuleSource]
|
||||
|
||||
from .config import Config, get_global_config
|
||||
from .dcs import get_dcs as _get_dcs, AbstractDCS, Cluster, Member
|
||||
from .exceptions import PatroniException
|
||||
from .manual_failover import ManualFailover
|
||||
from .postgresql.misc import postgres_version_to_int
|
||||
from .utils import cluster_as_json, patch_config, polling_loop
|
||||
from .utils import cluster_as_json, parse_schedule, patch_config, polling_loop
|
||||
from .request import PatroniRequest
|
||||
from .version import __version__
|
||||
|
||||
@@ -225,8 +225,6 @@ def load_config(path: str, dcs_url: Optional[str]) -> Dict[str, Any]:
|
||||
:raises:
|
||||
:class:`PatroniCtlException`: if *path* does not exist or is not readable.
|
||||
"""
|
||||
from patroni.config import Config
|
||||
|
||||
if not (os.path.exists(path) and os.access(path, os.R_OK)):
|
||||
if path != CONFIG_FILE_PATH: # bail if non-default config location specified but file not found / readable
|
||||
raise PatroniCtlException('Provided config file {0} not existing or no read rights.'
|
||||
@@ -264,9 +262,7 @@ role_choice = click.Choice(['leader', 'primary', 'standby-leader', 'replica', 's
|
||||
@click.option('-k', '--insecure', is_flag=True, help='Allow connections to SSL sites without certs')
|
||||
@click.pass_context
|
||||
def ctl(ctx: click.Context, config_file: str, dcs_url: Optional[str], insecure: bool) -> None:
|
||||
"""Command-line interface for interacting with Patroni.
|
||||
\f
|
||||
Entry point of ``patronictl`` utility.
|
||||
"""Entry point of ``patronictl`` utility.
|
||||
|
||||
Load the configuration file.
|
||||
|
||||
@@ -562,10 +558,9 @@ def get_cursor(obj: Dict[str, Any], cluster: Cluster, group: Optional[int], conn
|
||||
from . import psycopg
|
||||
conn = psycopg.connect(**params)
|
||||
cursor = conn.cursor()
|
||||
# If we want ``any`` node we are fine to return the cursor. ``None`` is similar to ``any`` at this point, as it's
|
||||
# been dealt with through :func:`get_any_member`.
|
||||
# If we want ``any`` node we are fine to return the cursor
|
||||
# If we want the Patroni leader node, :func:`get_any_member` already checks that for us
|
||||
if role in (None, 'any', 'leader'):
|
||||
if role in ('any', 'leader'):
|
||||
return cursor
|
||||
|
||||
# If we want something other than ``any`` or ``leader``, then we do not rely only on the DCS information about
|
||||
@@ -647,7 +642,8 @@ def get_members(obj: Dict[str, Any], cluster: Cluster, cluster_name: str, member
|
||||
if member_names:
|
||||
member_names = list(set(member_names) & candidates)
|
||||
if not member_names:
|
||||
raise PatroniCtlException('No {0} among provided members'.format(role))
|
||||
raise PatroniCtlException(
|
||||
'No{0} among provided members'.format('t a single cluster member' if role == 'any' else ' ' + role))
|
||||
elif action != 'reinitialize':
|
||||
member_names = list(candidates)
|
||||
|
||||
@@ -859,11 +855,9 @@ def query_member(obj: Dict[str, Any], cluster: Cluster, group: Optional[int],
|
||||
|
||||
if cursor is None:
|
||||
if member is not None:
|
||||
message = f'No connection to member {member} is available'
|
||||
elif role is not None:
|
||||
message = f'No connection to role {role} is available'
|
||||
message = 'No connection to member {0} is available'.format(member)
|
||||
else:
|
||||
message = 'No connection is available'
|
||||
message = 'No connection to role={0} is available'.format(role)
|
||||
logging.debug(message)
|
||||
return [[timestamp(0), message]], None
|
||||
|
||||
@@ -949,43 +943,6 @@ def check_response(response: urllib3.response.HTTPResponse, member_name: str,
|
||||
return True
|
||||
|
||||
|
||||
def parse_scheduled(scheduled: Optional[str]) -> Optional[datetime.datetime]:
|
||||
"""Parse a string *scheduled* timestamp as a :class:`~datetime.datetime` object.
|
||||
|
||||
:param scheduled: string representation of the timestamp. May also be ``now``.
|
||||
|
||||
:returns: the corresponding :class:`~datetime.datetime` object, if *scheduled* is not ``now``, otherwise ``None``.
|
||||
|
||||
:raises:
|
||||
:class:`PatroniCtlException`: if unable to parse *scheduled* from :class:`str` to :class:`~datetime.datetime`.
|
||||
|
||||
:Example:
|
||||
|
||||
>>> parse_scheduled(None) is None
|
||||
True
|
||||
|
||||
>>> parse_scheduled('now') is None
|
||||
True
|
||||
|
||||
>>> parse_scheduled('2023-05-29T04:32:31')
|
||||
datetime.datetime(2023, 5, 29, 4, 32, 31, tzinfo=tzlocal())
|
||||
|
||||
>>> parse_scheduled('2023-05-29T04:32:31-3')
|
||||
datetime.datetime(2023, 5, 29, 4, 32, 31, tzinfo=tzoffset(None, -10800))
|
||||
"""
|
||||
if scheduled is not None and (scheduled or 'now') != 'now':
|
||||
try:
|
||||
scheduled_at = dateutil.parser.parse(scheduled)
|
||||
if scheduled_at.tzinfo is None:
|
||||
scheduled_at = scheduled_at.replace(tzinfo=dateutil.tz.tzlocal())
|
||||
except (ValueError, TypeError):
|
||||
message = 'Unable to parse scheduled timestamp ({0}). It should be in an unambiguous format (e.g. ISO 8601)'
|
||||
raise PatroniCtlException(message.format(scheduled))
|
||||
return scheduled_at
|
||||
|
||||
return None
|
||||
|
||||
|
||||
@ctl.command('reload', help='Reload cluster member configuration')
|
||||
@click.argument('cluster_name')
|
||||
@click.argument('member_names', nargs=-1)
|
||||
@@ -1016,7 +973,6 @@ def reload(obj: Dict[str, Any], cluster_name: str, member_names: List[str],
|
||||
if r.status == 200:
|
||||
click.echo('No changes to apply on member {0}'.format(member.name))
|
||||
elif r.status == 202:
|
||||
from patroni.config import get_global_config
|
||||
config = get_global_config(cluster)
|
||||
click.echo('Reload request received for member {0} and will be processed within {1} seconds'.format(
|
||||
member.name, config.get('loop_wait') or dcs.loop_wait)
|
||||
@@ -1066,16 +1022,20 @@ def restart(obj: Dict[str, Any], cluster_name: str, group: Optional[int], member
|
||||
* *version* could not be parsed; or
|
||||
* a restart is attempted against a cluster that is in maintenance mode.
|
||||
"""
|
||||
action = 'restart'
|
||||
cluster = get_dcs(obj, cluster_name, group).get_cluster()
|
||||
|
||||
members = get_members(obj, cluster, cluster_name, member_names, role, force, 'restart', False, group=group)
|
||||
members = get_members(obj, cluster, cluster_name, member_names, role, force, action, False, group=group)
|
||||
if scheduled is None and not force:
|
||||
next_hour = (datetime.datetime.now() + datetime.timedelta(hours=1)).strftime('%Y-%m-%dT%H:%M')
|
||||
next_hour = (datetime.datetime.now() + datetime.timedelta(hours=1)).strftime('%Y-%m-%dT%H:%M+00')
|
||||
scheduled = click.prompt('When should the restart take place (e.g. ' + next_hour + ') ',
|
||||
type=str, default='now')
|
||||
scheduled = scheduled if scheduled != 'now' else None
|
||||
|
||||
scheduled_at = parse_scheduled(scheduled)
|
||||
confirm_members_action(members, force, 'restart', scheduled_at)
|
||||
parse_result, scheduled_at = parse_schedule(scheduled)
|
||||
if parse_result:
|
||||
raise PatroniCtlException(parse_result.value[0].format(action=action))
|
||||
confirm_members_action(members, force, action, scheduled_at)
|
||||
|
||||
if p_any:
|
||||
random.shuffle(members)
|
||||
@@ -1098,7 +1058,6 @@ def restart(obj: Dict[str, Any], cluster_name: str, group: Optional[int], member
|
||||
content['postgres_version'] = version
|
||||
|
||||
if scheduled_at:
|
||||
from patroni.config import get_global_config
|
||||
if get_global_config(cluster).is_paused:
|
||||
raise PatroniCtlException("Can't schedule restart in the paused state")
|
||||
content['schedule'] = scheduled_at.isoformat()
|
||||
@@ -1181,8 +1140,8 @@ def reinit(obj: Dict[str, Any], cluster_name: str, group: Optional[int],
|
||||
|
||||
|
||||
def _do_failover_or_switchover(obj: Dict[str, Any], action: str, cluster_name: str,
|
||||
group: Optional[int], leader: Optional[str], candidate: Optional[str],
|
||||
force: bool, scheduled: Optional[str] = None) -> None:
|
||||
group: Optional[int], candidate: Optional[str], force: bool,
|
||||
leader: Optional[str] = None, scheduled: Optional[str] = None) -> None:
|
||||
"""Perform a failover or a switchover operation in the cluster.
|
||||
|
||||
Informational messages are printed in the console during the operation, as well as the list of members before and
|
||||
@@ -1196,9 +1155,9 @@ def _do_failover_or_switchover(obj: Dict[str, Any], action: str, cluster_name: s
|
||||
:param cluster_name: name of the Patroni cluster.
|
||||
:param group: filter Citus group within we should perform a failover or switchover. If ``None``, user will be
|
||||
prompted for filling it -- unless *force* is ``True``, in which case an exception is raised.
|
||||
:param leader: name of the current leader member.
|
||||
:param candidate: name of a standby member to be promoted. Nodes that are tagged with ``nofailover`` cannot be used.
|
||||
:param force: perform the failover or switchover without asking for confirmations.
|
||||
:param leader: name of the leader passed to the switchover command if any.
|
||||
:param scheduled: timestamp when the switchover should be scheduled to occur. If ``now`` perform immediately.
|
||||
|
||||
:raises:
|
||||
@@ -1218,6 +1177,9 @@ def _do_failover_or_switchover(obj: Dict[str, Any], action: str, cluster_name: s
|
||||
click.echo('Current cluster topology')
|
||||
output_members(obj, cluster, cluster_name, group=group)
|
||||
|
||||
# Define everything missing via interactive input or available cluster info (if force mode)
|
||||
|
||||
# Require Citus group
|
||||
if obj.get('citus') and group is None:
|
||||
if force:
|
||||
raise PatroniCtlException('For Citus clusters the --group must me specified')
|
||||
@@ -1226,72 +1188,81 @@ def _do_failover_or_switchover(obj: Dict[str, Any], action: str, cluster_name: s
|
||||
dcs = get_dcs(obj, cluster_name, group)
|
||||
cluster = dcs.get_cluster()
|
||||
|
||||
if action == 'switchover' and (cluster.leader is None or not cluster.leader.name):
|
||||
raise PatroniCtlException('This cluster has no leader')
|
||||
global_config = get_global_config(cluster)
|
||||
|
||||
if leader is None:
|
||||
if force or action == 'failover':
|
||||
leader = cluster.leader and cluster.leader.name
|
||||
# Leader is required for switchover only
|
||||
if action == 'switchover' and leader is None:
|
||||
if cluster.leader is None or not cluster.leader.name:
|
||||
raise PatroniCtlException('This cluster has no leader')
|
||||
if force:
|
||||
leader = cluster.leader.name
|
||||
else:
|
||||
from patroni.config import get_global_config
|
||||
prompt = 'Standby Leader' if get_global_config(cluster).is_standby_cluster else 'Primary'
|
||||
leader = click.prompt(prompt, type=str, default=(cluster.leader and cluster.leader.member.name))
|
||||
|
||||
if leader is not None and cluster.leader and cluster.leader.member.name != leader:
|
||||
raise PatroniCtlException('Member {0} is not the leader of cluster {1}'.format(leader, cluster_name))
|
||||
|
||||
# excluding members with nofailover tag
|
||||
candidate_names = [str(m.name) for m in cluster.members if m.name != leader and not m.nofailover]
|
||||
# We sort the names for consistent output to the client
|
||||
candidate_names.sort()
|
||||
|
||||
if not candidate_names:
|
||||
raise PatroniCtlException('No candidates found to {0} to'.format(action))
|
||||
prompt = 'Standby Leader' if global_config.is_standby_cluster else 'Primary'
|
||||
leader = click.prompt(prompt, type=str, default=(cluster.leader and cluster.leader.name))
|
||||
|
||||
if candidate is None and not force:
|
||||
# Check if there are any candidates available at all
|
||||
candidate_names = [str(m.name) for m in cluster.members if m.name != leader and not m.nofailover]
|
||||
if not candidate_names:
|
||||
raise PatroniCtlException('No candidates found to {0} to'.format(action))
|
||||
candidate_names.sort() # we sort the names for consistent output to the client
|
||||
candidate = click.prompt('Candidate ' + str(candidate_names), type=str, default='')
|
||||
|
||||
if action == 'failover' and not candidate:
|
||||
raise PatroniCtlException('Failover could be performed only to a specific candidate')
|
||||
# We allow manual failover to an aync node in the sync mode, so we better ask for the confirmation
|
||||
if all((not force,
|
||||
action == 'failover',
|
||||
global_config.is_synchronous_mode,
|
||||
not cluster.sync.is_empty,
|
||||
not cluster.sync.matches(candidate, True))):
|
||||
if click.confirm(f'Are you sure you want to failover to the asynchronous node {candidate}'):
|
||||
raise PatroniCtlException('Aborting ' + action)
|
||||
|
||||
if candidate == leader:
|
||||
raise PatroniCtlException(action.title() + ' target and source are the same.')
|
||||
if action == 'switchover' and scheduled is None and not force:
|
||||
next_hour = (datetime.datetime.now() + datetime.timedelta(hours=1)).strftime('%Y-%m-%dT%H:%M+00')
|
||||
scheduled = click.prompt('When should the switchover take place (e.g. ' + next_hour + ') ',
|
||||
type=str, default='now')
|
||||
scheduled = scheduled if scheduled != 'now' else None
|
||||
|
||||
if candidate and candidate not in candidate_names:
|
||||
raise PatroniCtlException('Member {0} does not exist in cluster {1}'.format(candidate, cluster_name))
|
||||
# Now, when we collected all the possible info, run checks
|
||||
manual_failover = ManualFailover(action, cluster, leader, candidate, scheduled,
|
||||
global_config.is_paused, global_config.is_synchronous_mode)
|
||||
|
||||
result_text, _ = manual_failover.run_precheck().value
|
||||
if result_text:
|
||||
raise PatroniCtlException(result_text.format(action=action, leader=leader, candidate=candidate,
|
||||
cluster_name=cluster_name))
|
||||
|
||||
scheduled_at_str = None
|
||||
scheduled_at = None
|
||||
|
||||
if action == 'switchover':
|
||||
if scheduled is None and not force:
|
||||
next_hour = (datetime.datetime.now() + datetime.timedelta(hours=1)).strftime('%Y-%m-%dT%H:%M')
|
||||
scheduled = click.prompt('When should the switchover take place (e.g. ' + next_hour + ' ) ',
|
||||
type=str, default='now')
|
||||
|
||||
scheduled_at = parse_scheduled(scheduled)
|
||||
parse_result, scheduled_at = manual_failover.parse_scheduled()
|
||||
if parse_result:
|
||||
raise PatroniCtlException(parse_result.value[0].format(action=action))
|
||||
if scheduled_at:
|
||||
from patroni.config import get_global_config
|
||||
if get_global_config(cluster).is_paused:
|
||||
raise PatroniCtlException("Can't schedule switchover in the paused state")
|
||||
scheduled_at_str = scheduled_at.isoformat()
|
||||
|
||||
failover_value = {'leader': leader, 'candidate': candidate, 'scheduled_at': scheduled_at_str}
|
||||
|
||||
logging.debug(failover_value)
|
||||
|
||||
# By now we have established that the leader exists and the candidate exists
|
||||
# By now we have established that the leader exists and the candidate exists,
|
||||
# so confirm the action that is about to be run
|
||||
if not force:
|
||||
demote_msg = ', demoting current leader ' + leader if leader else ''
|
||||
demote_msg = f', demoting current leader {cluster.leader.name}' if cluster.leader else ''
|
||||
if scheduled_at_str:
|
||||
if not click.confirm('Are you sure you want to schedule {0} of cluster {1} at {2}{3}?'
|
||||
.format(action, cluster_name, scheduled_at_str, demote_msg)):
|
||||
# only switchover can be scheduled
|
||||
if not click.confirm(f'Are you sure you want to schedule switchover of cluster'
|
||||
f'{cluster_name} at {scheduled_at_str}{demote_msg}?'):
|
||||
raise PatroniCtlException('Aborting scheduled ' + action)
|
||||
else:
|
||||
if not click.confirm('Are you sure you want to {0} cluster {1}{2}?'
|
||||
.format(action, cluster_name, demote_msg)):
|
||||
if not click.confirm(f'Are you sure you want to {action} cluster {cluster_name}{demote_msg}?'):
|
||||
raise PatroniCtlException('Aborting ' + action)
|
||||
|
||||
# And finally the actual work
|
||||
failover_value = {'candidate': candidate}
|
||||
if action == 'switchover':
|
||||
failover_value['leader'] = leader
|
||||
if scheduled_at_str:
|
||||
failover_value['scheduled_at'] = scheduled_at_str
|
||||
|
||||
logging.debug(failover_value)
|
||||
|
||||
r = None
|
||||
try:
|
||||
member = cluster.leader.member if cluster.leader else candidate and cluster.get_member(candidate, False)
|
||||
@@ -1323,19 +1294,15 @@ def _do_failover_or_switchover(obj: Dict[str, Any], action: str, cluster_name: s
|
||||
@ctl.command('failover', help='Failover to a replica')
|
||||
@arg_cluster_name
|
||||
@option_citus_group
|
||||
@click.option('--leader', '--primary', '--master', 'leader', help='The name of the current leader', default=None)
|
||||
@click.option('--candidate', help='The name of the candidate', default=None)
|
||||
@option_force
|
||||
@click.pass_obj
|
||||
def failover(obj: Dict[str, Any], cluster_name: str, group: Optional[int],
|
||||
leader: Optional[str], candidate: Optional[str], force: bool) -> None:
|
||||
candidate: Optional[str], force: bool) -> None:
|
||||
"""Process ``failover`` command of ``patronictl`` utility.
|
||||
|
||||
Perform a failover operation immediately in the cluster.
|
||||
|
||||
.. note::
|
||||
If *leader* is given perform a switchover instead of a failover.
|
||||
|
||||
.. seealso::
|
||||
Refer to :func:`_do_failover_or_switchover` for details.
|
||||
|
||||
@@ -1344,12 +1311,10 @@ def failover(obj: Dict[str, Any], cluster_name: str, group: Optional[int],
|
||||
:param group: filter Citus group within we should perform a failover or switchover. If ``None``, user will be
|
||||
prompted for filling it -- unless *force* is ``True``, in which case an exception is raised by
|
||||
:func:`_do_failover_or_switchover`.
|
||||
:param leader: name of the current leader member.
|
||||
:param candidate: name of a standby member to be promoted. Nodes that are tagged with ``nofailover`` cannot be used.
|
||||
:param force: perform the failover or switchover without asking for confirmations.
|
||||
"""
|
||||
action = 'switchover' if leader else 'failover'
|
||||
_do_failover_or_switchover(obj, action, cluster_name, group, leader, candidate, force)
|
||||
_do_failover_or_switchover(obj, 'failover', cluster_name, group, candidate, force)
|
||||
|
||||
|
||||
@ctl.command('switchover', help='Switchover to a replica')
|
||||
@@ -1380,7 +1345,7 @@ def switchover(obj: Dict[str, Any], cluster_name: str, group: Optional[int],
|
||||
:param force: perform the switchover without asking for confirmations.
|
||||
:param scheduled: timestamp when the switchover should be scheduled to occur. If ``now`` perform immediately.
|
||||
"""
|
||||
_do_failover_or_switchover(obj, 'switchover', cluster_name, group, leader, candidate, force, scheduled)
|
||||
_do_failover_or_switchover(obj, 'switchover', cluster_name, group, candidate, force, leader, scheduled)
|
||||
|
||||
|
||||
def generate_topology(level: int, member: Dict[str, Any],
|
||||
@@ -1566,11 +1531,9 @@ def output_members(obj: Dict[str, Any], cluster: Cluster, name: str,
|
||||
rows.append([member.get(n.lower().replace(' ', '_'), '') for n in columns])
|
||||
|
||||
title = 'Citus cluster' if is_citus_cluster else 'Cluster'
|
||||
title_details = f' ({initialize})'
|
||||
if is_citus_cluster:
|
||||
title_details = '' if group is None else f' (group: {group}, {initialize})'
|
||||
|
||||
title = f' {title}: {name}{title_details} '
|
||||
group_title = '' if group is None else 'group: {0}, '.format(group)
|
||||
title_details = group_title and ' ({0}{1})'.format(group_title, initialize)
|
||||
title = ' {0}: {1}{2} '.format(title, name, title_details)
|
||||
print_output(columns, rows, {'Group': 'r', 'Lag in MB': 'r', 'TL': 'r'}, fmt, title)
|
||||
|
||||
if fmt not in ('pretty', 'topology'): # Omit service info when using machine-readable formats
|
||||
@@ -1721,7 +1684,6 @@ def wait_until_pause_is_applied(dcs: AbstractDCS, paused: bool, old_cluster: Clu
|
||||
:param old_cluster: original cluster information before pause or unpause has been requested. Used to report which
|
||||
nodes are still pending to have ``pause`` equal *paused* at a given point in time.
|
||||
"""
|
||||
from patroni.config import get_global_config
|
||||
config = get_global_config(old_cluster)
|
||||
|
||||
click.echo("'{0}' request sent, waiting until it is recognized by all nodes".format(paused and 'pause' or 'resume'))
|
||||
@@ -1759,7 +1721,6 @@ def toggle_pause(config: Dict[str, Any], cluster_name: str, group: Optional[int]
|
||||
* ``pause`` state is already *paused*; or
|
||||
* cluster contains no accessible members.
|
||||
"""
|
||||
from patroni.config import get_global_config
|
||||
dcs = get_dcs(config, cluster_name, group)
|
||||
cluster = dcs.get_cluster()
|
||||
if get_global_config(cluster).is_paused == paused:
|
||||
|
||||
@@ -1758,26 +1758,29 @@ class AbstractDCS(abc.ABC):
|
||||
"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def _delete_leader(self) -> bool:
|
||||
def _delete_leader(self, leader: Leader) -> bool:
|
||||
"""Remove leader key from DCS.
|
||||
|
||||
This method should remove leader key if current instance is the leader.
|
||||
|
||||
:param leader: :class:`Leader` object with information about the leader.
|
||||
|
||||
:returns: ``True`` if successfully committed to DCS.
|
||||
"""
|
||||
|
||||
def delete_leader(self, last_lsn: Optional[int] = None) -> bool:
|
||||
def delete_leader(self, leader: Optional[Leader], last_lsn: Optional[int] = None) -> bool:
|
||||
"""Update ``optime/leader`` and voluntarily remove leader key from DCS.
|
||||
|
||||
This method should remove leader key if current instance is the leader.
|
||||
|
||||
:param leader: :class:`Leader` object with information about the leader.
|
||||
:param last_lsn: latest checkpoint location in bytes.
|
||||
|
||||
:returns: boolean result of called abstract :meth:`~AbstractDCS._delete_leader`.
|
||||
"""
|
||||
if last_lsn:
|
||||
self.write_status({self._OPTIME: last_lsn})
|
||||
return self._delete_leader()
|
||||
return bool(leader) and self._delete_leader(leader)
|
||||
|
||||
@abc.abstractmethod
|
||||
def cancel_initialization(self) -> bool:
|
||||
|
||||
@@ -643,12 +643,8 @@ class Consul(AbstractDCS):
|
||||
return self._client.kv.put(self.history_path, value)
|
||||
|
||||
@catch_consul_errors
|
||||
def _delete_leader(self) -> bool:
|
||||
cluster = self.cluster
|
||||
if cluster and isinstance(cluster.leader, Leader) and\
|
||||
cluster.leader.name == self._name and isinstance(cluster.leader.version, int):
|
||||
return self._client.kv.delete(self.leader_path, cas=cluster.leader.version)
|
||||
return True
|
||||
def _delete_leader(self, leader: Leader) -> bool:
|
||||
return self._client.kv.delete(self.leader_path, cas=int(leader.version))
|
||||
|
||||
@catch_consul_errors
|
||||
def set_sync_state_value(self, value: str, version: Optional[int] = None) -> Union[int, bool]:
|
||||
|
||||
+1
-1
@@ -809,7 +809,7 @@ class Etcd(AbstractEtcd):
|
||||
return bool(self.retry(self._client.write, self.initialize_path, sysid, prevExist=(not create_new)))
|
||||
|
||||
@catch_etcd_errors
|
||||
def _delete_leader(self) -> bool:
|
||||
def _delete_leader(self, leader: Leader) -> bool:
|
||||
return bool(self._client.delete(self.leader_path, prevValue=self._name))
|
||||
|
||||
@catch_etcd_errors
|
||||
|
||||
@@ -205,7 +205,7 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
|
||||
|
||||
def __init__(self, config: Dict[str, Any], dns_resolver: DnsCachingResolver, cache_ttl: int = 300) -> None:
|
||||
self._token = None
|
||||
self._cluster_version: Tuple[int, ...] = tuple()
|
||||
self._cluster_version: Tuple[int] = tuple()
|
||||
super(Etcd3Client, self).__init__({**config, 'version_prefix': '/v3beta'}, dns_resolver, cache_ttl)
|
||||
|
||||
try:
|
||||
@@ -912,11 +912,10 @@ class Etcd3(AbstractEtcd):
|
||||
return self.retry(self._client.put, self.initialize_path, sysid, create_revision='0' if create_new else None)
|
||||
|
||||
@catch_etcd_errors
|
||||
def _delete_leader(self) -> bool:
|
||||
cluster = self.cluster
|
||||
if cluster and isinstance(cluster.leader, Leader) and cluster.leader.name == self._name:
|
||||
return self._client.deleterange(self.leader_path, mod_revision=cluster.leader.version)
|
||||
return True
|
||||
def _delete_leader(self, leader: Leader) -> bool:
|
||||
fields = build_range_request(self.leader_path)
|
||||
compare = {'key': fields['key'], 'target': 'VALUE', 'value': base64_encode(self._name)}
|
||||
return bool(self._client.txn(compare, {'request_delete_range': fields}))
|
||||
|
||||
@catch_etcd_errors
|
||||
def cancel_initialization(self) -> bool:
|
||||
|
||||
@@ -756,7 +756,7 @@ class Kubernetes(AbstractDCS):
|
||||
self._role_label = config.get('role_label', 'role')
|
||||
self._leader_label_value = config.get('leader_label_value', 'master')
|
||||
self._follower_label_value = config.get('follower_label_value', 'replica')
|
||||
self._standby_leader_label_value = config.get('standby_leader_label_value', 'master')
|
||||
self._standby_leader_label_value = config.get('standby_leader_label_value', 'standby-leader')
|
||||
self._tmp_role_label = config.get('tmp_role_label')
|
||||
self._ca_certs = os.environ.get('PATRONI_KUBERNETES_CACERT', config.get('cacert')) or SERVICE_CERT_FILENAME
|
||||
super(Kubernetes, self).__init__({**config, 'namespace': ''})
|
||||
@@ -836,7 +836,7 @@ class Kubernetes(AbstractDCS):
|
||||
self._api.configure_timeouts(self.loop_wait, self._retry.deadline, self.ttl)
|
||||
|
||||
# retriable_http_codes supposed to be either int, list of integers or comma-separated string with integers.
|
||||
retriable_http_codes: Union[str, List[Union[str, int]]] = config.get('retriable_http_codes', [])
|
||||
retriable_http_codes = config.get('retriable_http_codes', [])
|
||||
if not isinstance(retriable_http_codes, list):
|
||||
retriable_http_codes = [c.strip() for c in str(retriable_http_codes).split(',')]
|
||||
|
||||
@@ -1140,13 +1140,6 @@ class Kubernetes(AbstractDCS):
|
||||
"""Unused"""
|
||||
raise NotImplementedError # pragma: no cover
|
||||
|
||||
def write_leader_optime(self, last_lsn: int) -> None:
|
||||
"""Write value for WAL LSN to ``optime`` annotation of the leader object.
|
||||
|
||||
:param last_lsn: absolute WAL LSN in bytes.
|
||||
"""
|
||||
self.patch_or_create(self.leader_path, {self._OPTIME: str(last_lsn)}, patch=True, retry=False)
|
||||
|
||||
def _update_leader_with_retry(self, annotations: Dict[str, Any],
|
||||
resource_version: Optional[str], ips: List[str]) -> bool:
|
||||
retry = self._retry.copy()
|
||||
@@ -1276,10 +1269,13 @@ class Kubernetes(AbstractDCS):
|
||||
def touch_member(self, data: Dict[str, Any]) -> bool:
|
||||
cluster = self.cluster
|
||||
if cluster and cluster.leader and cluster.leader.name == self._name:
|
||||
role = self._standby_leader_label_value if data['role'] == 'standby_leader' else self._leader_label_value
|
||||
role = self._leader_label_value
|
||||
tmp_role = 'master'
|
||||
elif data['state'] == 'running' and data['role'] not in ('master', 'primary'):
|
||||
role = {'replica': self._follower_label_value}.get(data['role'], data['role'])
|
||||
role = {
|
||||
'replica': self._follower_label_value,
|
||||
'standby-leader': self._standby_leader_label_value,
|
||||
}.get(data['role'], data['role'])
|
||||
tmp_role = data['role']
|
||||
else:
|
||||
role = None
|
||||
@@ -1312,11 +1308,11 @@ class Kubernetes(AbstractDCS):
|
||||
if cluster and cluster.config and cluster.config.version else None
|
||||
return self.patch_or_create_config({self._INITIALIZE: sysid}, resource_version)
|
||||
|
||||
def _delete_leader(self) -> bool:
|
||||
def _delete_leader(self, leader: Leader) -> bool:
|
||||
"""Unused"""
|
||||
raise NotImplementedError # pragma: no cover
|
||||
|
||||
def delete_leader(self, last_lsn: Optional[int] = None) -> bool:
|
||||
def delete_leader(self, leader: Optional[Leader], last_lsn: Optional[int] = None) -> bool:
|
||||
ret = False
|
||||
kind = self._kinds.get(self.leader_path)
|
||||
if kind and (kind.metadata.annotations or {}).get(self._LEADER) == self._name:
|
||||
|
||||
+1
-1
@@ -446,7 +446,7 @@ class Raft(AbstractDCS):
|
||||
def initialize(self, create_new: bool = True, sysid: str = '') -> bool:
|
||||
return self._sync_obj.set(self.initialize_path, sysid, prevExist=(not create_new)) is not False
|
||||
|
||||
def _delete_leader(self) -> bool:
|
||||
def _delete_leader(self, leader: Leader) -> bool:
|
||||
return self._sync_obj.delete(self.leader_path, prevValue=self._name, timeout=1)
|
||||
|
||||
def cancel_initialization(self) -> bool:
|
||||
|
||||
+17
-24
@@ -89,7 +89,7 @@ class ZooKeeper(AbstractDCS):
|
||||
def __init__(self, config: Dict[str, Any]) -> None:
|
||||
super(ZooKeeper, self).__init__(config)
|
||||
|
||||
hosts: Union[str, List[str]] = config.get('hosts', [])
|
||||
hosts = config.get('hosts', [])
|
||||
if isinstance(hosts, list):
|
||||
hosts = ','.join(hosts)
|
||||
|
||||
@@ -393,28 +393,21 @@ class ZooKeeper(AbstractDCS):
|
||||
cluster = self.cluster
|
||||
member = cluster and cluster.get_member(self._name, fallback_to_leader=False)
|
||||
member_data = self.__last_member_data or member and member.data
|
||||
if member and member_data:
|
||||
is_leader = data.get('role') in ('master', 'primary', 'standby_leader')
|
||||
checkpoint_after_promote_changed = member_data.get('checkpoint_after_promote') \
|
||||
!= data.get('checkpoint_after_promote')
|
||||
state_running_changed = member_data.get('state') != data.get('state') \
|
||||
and 'running' in (member_data.get('state'), data.get('state'))
|
||||
tags_changed = not deep_compare(member_data.get('tags', {}), data.get('tags', {}))
|
||||
|
||||
# We want delete the member ZNode if:
|
||||
# - our session doesn't match with session id on our member key; or
|
||||
# - we want to notify leader if some important fields in the member key changed; or
|
||||
# - if we are the leader and want to notify replicas about checkpoint_after_promote;
|
||||
if self._client.client_id is not None and member.session != self._client.client_id[0] \
|
||||
or is_leader and checkpoint_after_promote_changed \
|
||||
or not is_leader and (state_running_changed or tags_changed):
|
||||
try:
|
||||
self._client.delete_async(self.member_path).get(timeout=1)
|
||||
except NoNodeError:
|
||||
pass
|
||||
except Exception:
|
||||
return False
|
||||
member = None
|
||||
# We want to notify leader if some important fields in the member key changed by removing ZNode
|
||||
if member and (self._client.client_id is not None and member.session != self._client.client_id[0]
|
||||
or not (member_data and deep_compare(member_data.get('tags', {}), data.get('tags', {}))
|
||||
and (member_data.get('state') == data.get('state')
|
||||
or 'running' not in (member_data.get('state'), data.get('state')))
|
||||
and member_data.get('version') == data.get('version')
|
||||
and member_data.get('checkpoint_after_promote')
|
||||
== data.get('checkpoint_after_promote'))):
|
||||
try:
|
||||
self._client.delete_async(self.member_path).get(timeout=1)
|
||||
except NoNodeError:
|
||||
pass
|
||||
except Exception:
|
||||
return False
|
||||
member = None
|
||||
|
||||
encoded_data = json.dumps(data, separators=(',', ':')).encode('utf-8')
|
||||
if member and member_data:
|
||||
@@ -473,7 +466,7 @@ class ZooKeeper(AbstractDCS):
|
||||
return False
|
||||
return True
|
||||
|
||||
def _delete_leader(self) -> bool:
|
||||
def _delete_leader(self, leader: Leader) -> bool:
|
||||
self._client.restart()
|
||||
return True
|
||||
|
||||
|
||||
+160
-126
@@ -147,8 +147,8 @@ class Ha(object):
|
||||
self.cluster = Cluster.empty()
|
||||
self.global_config = self.patroni.config.get_global_config(None)
|
||||
self.old_cluster = Cluster.empty()
|
||||
self._is_leader = False
|
||||
self._is_leader_lock = RLock()
|
||||
self._leader_expiry = 0
|
||||
self._leader_expiry_lock = RLock()
|
||||
self._failsafe = Failsafe(patroni.dcs)
|
||||
self._was_paused = False
|
||||
self._leader_timeline = None
|
||||
@@ -193,12 +193,38 @@ class Ha(object):
|
||||
return self.global_config.is_standby_cluster
|
||||
|
||||
def is_leader(self) -> bool:
|
||||
with self._is_leader_lock:
|
||||
return self._is_leader > time.time()
|
||||
""":returns: `True` if the current node is the leader, based on expiration set when it last held the key."""
|
||||
with self._leader_expiry_lock:
|
||||
return self._leader_expiry > time.time()
|
||||
|
||||
def set_is_leader(self, value: bool) -> None:
|
||||
with self._is_leader_lock:
|
||||
self._is_leader = time.time() + self.dcs.ttl if value else 0
|
||||
"""Update the current node's view of it's own leadership status.
|
||||
|
||||
Will update the expiry timestamp to match the dcs ttl if setting leadership to true,
|
||||
otherwise will set the expiry to the past to immediately invalidate.
|
||||
|
||||
:param value: is the current node the leader.
|
||||
"""
|
||||
with self._leader_expiry_lock:
|
||||
self._leader_expiry = time.time() + self.dcs.ttl if value else 0
|
||||
|
||||
def sync_mode_is_active(self) -> bool:
|
||||
"""Check whether synchronous replication is requested and already active.
|
||||
|
||||
:returns: ``True`` if the primary already put its name into the ``/sync`` in DCS.
|
||||
"""
|
||||
return self.is_synchronous_mode() and not self.cluster.sync.is_empty
|
||||
|
||||
def _get_failover_action_name(self) -> str:
|
||||
"""Return the currently requested manual failover action name or the default ``failover``.
|
||||
|
||||
:returns: :class:`str` representing the manually requested action (``manual failover`` if no leader
|
||||
is specified in the ``/failover`` in DCS, ``switchover`` otherwise) or ``failover`` if
|
||||
``/failover`` is empty.
|
||||
"""
|
||||
if not self.cluster.failover:
|
||||
return 'failover'
|
||||
return 'switchover' if self.cluster.failover.leader else 'manual failover'
|
||||
|
||||
def load_cluster_from_dcs(self) -> None:
|
||||
cluster = self.dcs.get_cluster()
|
||||
@@ -472,7 +498,7 @@ class Ha(object):
|
||||
if timeout == 0:
|
||||
# We are requested to prefer failing over to restarting primary. But see first if there
|
||||
# is anyone to fail over to.
|
||||
if self.is_failover_possible(self.cluster.members):
|
||||
if self.is_failover_possible():
|
||||
self.watchdog.disable()
|
||||
logger.info("Primary crashed. Failing over.")
|
||||
self.demote('immediate')
|
||||
@@ -572,7 +598,7 @@ class Ha(object):
|
||||
if refresh:
|
||||
self.load_cluster_from_dcs()
|
||||
|
||||
is_leader = self.state_handler.is_leader()
|
||||
is_leader = self.state_handler.is_primary()
|
||||
|
||||
node_to_follow = self._get_node_to_follow(self.cluster)
|
||||
|
||||
@@ -654,14 +680,6 @@ class Ha(object):
|
||||
current = CaseInsensitiveSet(sync.members)
|
||||
picked, allow_promote = self.state_handler.sync_handler.current_state(self.cluster)
|
||||
|
||||
if picked == current and current != allow_promote:
|
||||
logger.warning('Inconsistent state between synchronous_standby_names = %s and /sync = %s key '
|
||||
'detected, updating synchronous replication key...', list(allow_promote), list(current))
|
||||
sync = self.dcs.write_sync_state(self.state_handler.name, allow_promote, version=sync.version)
|
||||
if not sync:
|
||||
return logger.warning("Updating sync state failed")
|
||||
current = CaseInsensitiveSet(sync.members)
|
||||
|
||||
if picked != current:
|
||||
# update synchronous standby list in dcs temporarily to point to common nodes in current and picked
|
||||
sync_common = current & allow_promote
|
||||
@@ -743,13 +761,13 @@ class Ha(object):
|
||||
if cluster_history:
|
||||
self.dcs.set_history_value('[]')
|
||||
elif not cluster_history or cluster_history[-1][0] != primary_timeline - 1 or len(cluster_history[-1]) != 5:
|
||||
cluster_history_dict: Dict[int, List[Any]] = {line[0]: list(line) for line in cluster_history}
|
||||
cluster_history = {line[0]: line for line in cluster_history}
|
||||
history: List[List[Any]] = list(map(list, self.state_handler.get_history(primary_timeline)))
|
||||
if self.cluster.config:
|
||||
history = history[-self.cluster.config.max_timelines_history:]
|
||||
for line in history:
|
||||
# enrich current history with promotion timestamps stored in DCS
|
||||
cluster_history_line = cluster_history_dict.get(line[0], [])
|
||||
cluster_history_line = list(cluster_history.get(line[0], []))
|
||||
if len(line) == 3 and len(cluster_history_line) >= 4 and cluster_history_line[1] == line[1]:
|
||||
line.append(cluster_history_line[3])
|
||||
if len(cluster_history_line) == 5:
|
||||
@@ -768,7 +786,7 @@ class Ha(object):
|
||||
"""
|
||||
if not self.is_paused():
|
||||
if not self.watchdog.is_running and not self.watchdog.activate():
|
||||
if self.state_handler.is_leader():
|
||||
if self.state_handler.is_primary():
|
||||
self.demote('immediate')
|
||||
return 'Demoting self because watchdog could not be activated'
|
||||
else:
|
||||
@@ -784,7 +802,7 @@ class Ha(object):
|
||||
self._async_response.reset()
|
||||
return 'Promotion cancelled because the pre-promote script failed'
|
||||
|
||||
if self.state_handler.is_leader():
|
||||
if self.state_handler.is_primary():
|
||||
# Inform the state handler about its primary role.
|
||||
# It may be unaware of it if postgres is promoted manually.
|
||||
self.state_handler.set_role('master')
|
||||
@@ -834,6 +852,8 @@ class Ha(object):
|
||||
return _MemberStatus.unknown(member)
|
||||
|
||||
def fetch_nodes_statuses(self, members: List[Member]) -> List[_MemberStatus]:
|
||||
if not members:
|
||||
return []
|
||||
pool = ThreadPool(len(members))
|
||||
results = pool.map(self.fetch_node_status, members) # Run API calls on members in parallel
|
||||
pool.close()
|
||||
@@ -892,6 +912,27 @@ class Ha(object):
|
||||
lag = (self.cluster.last_lsn or 0) - wal_position
|
||||
return lag > self.global_config.maximum_lag_on_failover
|
||||
|
||||
def has_members_eligible_to_promote(self, members: List[Member], reference_lsn: int = 0,
|
||||
fast_path: bool = False) -> bool:
|
||||
ret = False
|
||||
cluster_timeline = self.cluster.timeline
|
||||
|
||||
for st in self.fetch_nodes_statuses(members):
|
||||
not_allowed_reason = st.failover_limitation()
|
||||
if not_allowed_reason:
|
||||
logger.info('Member %s is %s', st.member.name, not_allowed_reason)
|
||||
elif fast_path:
|
||||
return True
|
||||
elif reference_lsn and st.wal_position < reference_lsn or \
|
||||
not reference_lsn and self.is_lagging(st.wal_position):
|
||||
logger.info('Member %s exceeds maximum replication lag', st.member.name)
|
||||
elif self.check_timeline() and (not st.timeline or st.timeline < cluster_timeline):
|
||||
logger.info('Timeline %s of member %s is behind the cluster timeline %s',
|
||||
st.timeline, st.member.name, cluster_timeline)
|
||||
else:
|
||||
ret = True
|
||||
return ret
|
||||
|
||||
def _is_healthiest_node(self, members: Collection[Member], check_replication_lag: bool = True) -> bool:
|
||||
"""This method tries to determine whether I am healthy enough to became a new leader candidate or not."""
|
||||
|
||||
@@ -913,52 +954,38 @@ class Ha(object):
|
||||
# Prepare list of nodes to run check against
|
||||
members = [m for m in members if m.name != self.state_handler.name and not m.nofailover and m.api_url]
|
||||
|
||||
if members:
|
||||
for st in self.fetch_nodes_statuses(members):
|
||||
if st.failover_limitation() is None:
|
||||
if st.in_recovery is False:
|
||||
logger.warning('Primary (%s) is still alive', st.member.name)
|
||||
for st in self.fetch_nodes_statuses(members):
|
||||
if st.failover_limitation() is None:
|
||||
if st.in_recovery is False:
|
||||
logger.warning('Primary (%s) is still alive', st.member.name)
|
||||
return False
|
||||
if my_wal_position < st.wal_position:
|
||||
logger.info('Wal position of %s is ahead of my wal position', st.member.name)
|
||||
# In synchronous mode the former leader might be still accessible and even be ahead of us.
|
||||
# We should not disqualify himself from the leader race in such a situation.
|
||||
if not self.sync_mode_is_active() or not self.cluster.sync.leader_matches(st.member.name):
|
||||
return False
|
||||
if my_wal_position < st.wal_position:
|
||||
logger.info('Wal position of %s is ahead of my wal position', st.member.name)
|
||||
# In synchronous mode the former leader might be still accessible and even be ahead of us.
|
||||
# We should not disqualify himself from the leader race in such a situation.
|
||||
if not self.is_synchronous_mode() or self.cluster.sync.is_empty\
|
||||
or not self.cluster.sync.leader_matches(st.member.name):
|
||||
return False
|
||||
logger.info('Ignoring the former leader being ahead of us')
|
||||
logger.info('Ignoring the former leader being ahead of us')
|
||||
return True
|
||||
|
||||
def is_failover_possible(self, members: List[Member], check_synchronous: Optional[bool] = True,
|
||||
cluster_lsn: Optional[int] = 0) -> bool:
|
||||
"""Checks whether one of the members from the list can possibly win the leader race.
|
||||
def is_failover_possible(self, *, cluster_lsn: int = 0, exclude_failover_candidate: bool = False) -> bool:
|
||||
"""Checks whether any of the cluster members is allowed to promote and is healthy enough for that.
|
||||
|
||||
:param members: list of members to check
|
||||
:param check_synchronous: consider only members that are known to be listed in /sync key when sync replication.
|
||||
:param cluster_lsn: to calculate replication lag and exclude member if it is laggin
|
||||
:returns: `True` if there are members eligible to be the new leader
|
||||
:param cluster_lsn: to calculate replication lag and exclude member if it is lagging.
|
||||
:param exclude_failover_candidate: if ``True``, exclude :attr:`failover.candidate` from the members
|
||||
list against which the failover possibility checks are run.
|
||||
:returns: `True` if there are members eligible to become the new leader.
|
||||
"""
|
||||
ret = False
|
||||
cluster_timeline = self.cluster.timeline
|
||||
members = [m for m in members if m.name != self.state_handler.name and not m.nofailover and m.api_url]
|
||||
if check_synchronous and self.is_synchronous_mode() and not self.cluster.sync.is_empty:
|
||||
members = [m for m in members if self.cluster.sync.matches(m.name)]
|
||||
if members:
|
||||
for st in self.fetch_nodes_statuses(members):
|
||||
not_allowed_reason = st.failover_limitation()
|
||||
if not_allowed_reason:
|
||||
logger.info('Member %s is %s', st.member.name, not_allowed_reason)
|
||||
elif cluster_lsn and st.wal_position < cluster_lsn or\
|
||||
not cluster_lsn and self.is_lagging(st.wal_position):
|
||||
logger.info('Member %s exceeds maximum replication lag', st.member.name)
|
||||
elif self.check_timeline() and (not st.timeline or st.timeline < cluster_timeline):
|
||||
logger.info('Timeline %s of member %s is behind the cluster timeline %s',
|
||||
st.timeline, st.member.name, cluster_timeline)
|
||||
else:
|
||||
ret = True
|
||||
else:
|
||||
logger.warning('manual failover: members list is empty')
|
||||
return ret
|
||||
candidates = self.get_failover_candidates(exclude_failover_candidate)
|
||||
|
||||
action = self._get_failover_action_name()
|
||||
if self.is_synchronous_mode() and self.cluster.failover and self.cluster.failover.candidate and not candidates:
|
||||
logger.warning('%s candidate=%s does not match with sync_standbys=%s',
|
||||
action.title(), self.cluster.failover.candidate, self.cluster.sync.sync_standby)
|
||||
elif not candidates:
|
||||
logger.warning('%s: candidates list is empty', action)
|
||||
|
||||
return self.has_members_eligible_to_promote(candidates, cluster_lsn)
|
||||
|
||||
def manual_failover_process_no_leader(self) -> Optional[bool]:
|
||||
"""Handles manual failover/switchover when the old leader already stepped down.
|
||||
@@ -969,15 +996,18 @@ class Ha(object):
|
||||
failover = self.cluster.failover
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
assert failover is not None
|
||||
if failover.candidate: # manual failover to specific member
|
||||
if failover.candidate == self.state_handler.name: # manual failover to me
|
||||
|
||||
action = self._get_failover_action_name()
|
||||
|
||||
if failover.candidate: # manual failover/switchover to specific member
|
||||
if failover.candidate == self.state_handler.name: # manual failover/switchover to me
|
||||
return True
|
||||
elif self.is_paused():
|
||||
# Remove failover key if the node to failover has terminated to avoid waiting for it indefinitely
|
||||
# In order to avoid attempts to delete this key from all nodes only the primary is allowed to do it.
|
||||
if not self.cluster.get_member(failover.candidate, fallback_to_leader=False)\
|
||||
and self.state_handler.is_leader():
|
||||
logger.warning("manual failover: removing failover key because failover candidate is not running")
|
||||
and self.state_handler.is_primary():
|
||||
logger.warning("%s: removing failover key because failover candidate is not running", action)
|
||||
self.dcs.manual_failover('', '', version=failover.version)
|
||||
return None
|
||||
return False
|
||||
@@ -993,22 +1023,21 @@ class Ha(object):
|
||||
st = self.fetch_node_status(member)
|
||||
not_allowed_reason = st.failover_limitation()
|
||||
if not_allowed_reason is None: # node is healthy
|
||||
logger.info('manual failover: to %s, i am %s', st.member.name, self.state_handler.name)
|
||||
logger.info('%s: to %s, i am %s', action, st.member.name, self.state_handler.name)
|
||||
return False
|
||||
# we wanted to failover to specific member but it is not healthy
|
||||
logger.warning('manual failover: member %s is %s', st.member.name, not_allowed_reason)
|
||||
# we wanted to failover/switchover to specific member but it is not healthy
|
||||
logger.warning('%s: member %s is %s', action, st.member.name, not_allowed_reason)
|
||||
|
||||
# at this point we should consider all members as a candidates for failover
|
||||
# at this point we should consider all members as a candidates for failover/switchover
|
||||
# i.e. we assume that failover.candidate is None
|
||||
elif self.is_paused():
|
||||
return False
|
||||
|
||||
# try to pick some other members to failover and check that they are healthy
|
||||
# try to pick some other members to switchover and check that they are healthy
|
||||
if failover.leader:
|
||||
if self.state_handler.name == failover.leader: # I was the leader
|
||||
# exclude me and desired member which is unhealthy (failover.candidate can be None)
|
||||
members = [m for m in self.cluster.members if m.name not in (failover.candidate, failover.leader)]
|
||||
if self.is_failover_possible(members): # check that there are healthy members
|
||||
# exclude desired member which is unhealthy if it was specified
|
||||
if self.is_failover_possible(exclude_failover_candidate=bool(failover.candidate)):
|
||||
return False
|
||||
else: # I was the leader and it looks like currently I am the only healthy member
|
||||
return True
|
||||
@@ -1036,7 +1065,7 @@ class Ha(object):
|
||||
if ret is not None: # continue if we just deleted the stale failover key as a leader
|
||||
return ret
|
||||
|
||||
if self.state_handler.is_leader():
|
||||
if self.state_handler.is_primary():
|
||||
if self.is_paused():
|
||||
# in pause leader is the healthiest only when no initialize or sysid matches with initialize!
|
||||
return not self.cluster.initialize or self.state_handler.sysid == self.cluster.initialize
|
||||
@@ -1062,8 +1091,8 @@ class Ha(object):
|
||||
|
||||
if self.cluster.failover:
|
||||
# When doing a switchover in synchronous mode only synchronous nodes and former leader are allowed to race
|
||||
if self.is_synchronous_mode() and self.cluster.failover.leader and \
|
||||
not self.cluster.sync.is_empty and not self.cluster.sync.matches(self.state_handler.name, True):
|
||||
if self.cluster.failover.leader and self.sync_mode_is_active() \
|
||||
and not self.cluster.sync.matches(self.state_handler.name, True):
|
||||
return False
|
||||
return self.manual_failover_process_no_leader() or False
|
||||
|
||||
@@ -1084,7 +1113,7 @@ class Ha(object):
|
||||
all_known_members += self.cluster.members
|
||||
|
||||
# When in sync mode, only last known primary and sync standby are allowed to promote automatically.
|
||||
if self.is_synchronous_mode() and not self.cluster.sync.is_empty:
|
||||
if self.sync_mode_is_active():
|
||||
if not self.cluster.sync.matches(self.state_handler.name, True):
|
||||
return False
|
||||
# pick between synchronous candidates so we minimize unnecessary failovers/demotions
|
||||
@@ -1097,7 +1126,7 @@ class Ha(object):
|
||||
|
||||
def _delete_leader(self, last_lsn: Optional[int] = None) -> None:
|
||||
self.set_is_leader(False)
|
||||
self.dcs.delete_leader(last_lsn)
|
||||
self.dcs.delete_leader(self.cluster.leader, last_lsn)
|
||||
self.dcs.reset_cluster()
|
||||
|
||||
def release_leader_key_voluntarily(self, last_lsn: Optional[int] = None) -> None:
|
||||
@@ -1137,9 +1166,7 @@ class Ha(object):
|
||||
# It could happen if Postgres is still archiving the backlog of WAL files.
|
||||
# If we know that there are replicas that received the shutdown checkpoint
|
||||
# location, we can remove the leader key and allow them to start leader race.
|
||||
|
||||
# for a manual failover/switchover with a candidate, we should check the requested candidate only
|
||||
if self.is_failover_possible(self.get_failover_candidates(), cluster_lsn=checkpoint_location):
|
||||
if self.is_failover_possible(cluster_lsn=checkpoint_location):
|
||||
self.state_handler.set_role('demoted')
|
||||
with self._async_executor:
|
||||
self.release_leader_key_voluntarily(checkpoint_location)
|
||||
@@ -1229,36 +1256,35 @@ class Ha(object):
|
||||
|
||||
:returns: action message if demote was initiated, None if no action was taken"""
|
||||
failover = self.cluster.failover
|
||||
if not failover or (self.is_paused() and not self.state_handler.is_leader()):
|
||||
# if there is no failover key or
|
||||
# I am holding the lock but am not primary = I am the standby leader,
|
||||
# then do nothing
|
||||
if not failover or (self.is_paused() and not self.state_handler.is_primary()):
|
||||
return
|
||||
|
||||
action = self._get_failover_action_name()
|
||||
bare_action = action.replace('manual ', '')
|
||||
|
||||
# it is not the time for the the scheduled failover yet, do nothing
|
||||
if (failover.scheduled_at and not
|
||||
self.should_run_scheduled_action("failover", failover.scheduled_at, lambda:
|
||||
self.should_run_scheduled_action(bare_action, failover.scheduled_at, lambda:
|
||||
self.dcs.manual_failover('', '', version=failover.version))):
|
||||
return
|
||||
|
||||
if not failover.leader or failover.leader == self.state_handler.name:
|
||||
if not failover.candidate or failover.candidate != self.state_handler.name:
|
||||
if not failover.candidate and self.is_paused():
|
||||
logger.warning('Failover is possible only to a specific candidate in a paused state')
|
||||
logger.warning('%s is possible only to a specific candidate in a paused state', action.title())
|
||||
elif self.is_failover_possible():
|
||||
ret = self._async_executor.try_run_async(f'{action}: demote', self.demote, ('graceful',))
|
||||
return ret or f'{action}: demoting myself'
|
||||
else:
|
||||
if self.is_synchronous_mode():
|
||||
members = self.get_failover_candidates(check_sync=True)
|
||||
if failover.candidate and not members:
|
||||
logger.warning('Failover candidate=%s does not match with sync_standbys=%s',
|
||||
failover.candidate, self.cluster.sync.sync_standby)
|
||||
else:
|
||||
members = self.get_failover_candidates()
|
||||
if self.is_failover_possible(members, False): # check that there are healthy members
|
||||
ret = self._async_executor.try_run_async('manual failover: demote', self.demote, ('graceful',))
|
||||
return ret or 'manual failover: demoting myself'
|
||||
else:
|
||||
logger.warning('manual failover: no healthy members found, failover is not possible')
|
||||
logger.warning('%s: no healthy members found, %s is not possible',
|
||||
action, bare_action)
|
||||
else:
|
||||
logger.warning('manual failover: I am already the leader, no need to failover')
|
||||
logger.warning('%s: I am already the leader, no need to %s', action, bare_action)
|
||||
else:
|
||||
logger.warning('manual failover: leader name does not match: %s != %s',
|
||||
failover.leader, self.state_handler.name)
|
||||
logger.warning('%s: leader name does not match: %s != %s', action, failover.leader, self.state_handler.name)
|
||||
|
||||
logger.info('Cleaning up failover key')
|
||||
self.dcs.manual_failover('', '', version=failover.version)
|
||||
@@ -1307,7 +1333,7 @@ class Ha(object):
|
||||
|
||||
def process_healthy_cluster(self) -> str:
|
||||
if self.has_lock():
|
||||
if self.is_paused() and not self.state_handler.is_leader():
|
||||
if self.is_paused() and not self.state_handler.is_primary():
|
||||
if self.cluster.failover and self.cluster.failover.candidate == self.state_handler.name:
|
||||
return 'waiting to become primary after promote...'
|
||||
|
||||
@@ -1315,6 +1341,7 @@ class Ha(object):
|
||||
self._delete_leader()
|
||||
return 'removed leader lock because postgres is not running as primary'
|
||||
|
||||
# update lock to avoid split-brain
|
||||
if self.update_lock(True):
|
||||
msg = self.process_manual_failover_from_leader()
|
||||
if msg is not None:
|
||||
@@ -1339,7 +1366,7 @@ class Ha(object):
|
||||
else:
|
||||
# Either there is no connection to DCS or someone else acquired the lock
|
||||
logger.error('failed to update leader lock')
|
||||
if self.state_handler.is_leader():
|
||||
if self.state_handler.is_primary():
|
||||
if self.is_paused():
|
||||
return 'continue to run as primary after failing to update leader lock in DCS'
|
||||
self.demote('immediate-nolock')
|
||||
@@ -1512,7 +1539,7 @@ class Ha(object):
|
||||
if self.has_lock() and self.update_lock():
|
||||
if self._async_executor.scheduled_action == 'doing crash recovery in a single user mode':
|
||||
time_left = self.global_config.primary_start_timeout - (time.time() - self._crash_recovery_started)
|
||||
if time_left <= 0 and self.is_failover_possible(self.cluster.members):
|
||||
if time_left <= 0 and self.is_failover_possible():
|
||||
logger.info("Demoting self because crash recovery is taking too long")
|
||||
self.state_handler.cancellable.cancel(True)
|
||||
self.demote('immediate')
|
||||
@@ -1576,7 +1603,7 @@ class Ha(object):
|
||||
self.cancel_initialization()
|
||||
|
||||
if result is None:
|
||||
if not self.state_handler.is_leader():
|
||||
if not self.state_handler.is_primary():
|
||||
return 'waiting for end of recovery after bootstrap'
|
||||
|
||||
self.state_handler.set_role('master')
|
||||
@@ -1623,7 +1650,7 @@ class Ha(object):
|
||||
time_left = timeout - self.state_handler.time_in_state()
|
||||
|
||||
if time_left <= 0:
|
||||
if self.is_failover_possible(self.cluster.members):
|
||||
if self.is_failover_possible():
|
||||
logger.info("Demoting self because primary startup is taking too long")
|
||||
self.demote('immediate')
|
||||
return 'stopped PostgreSQL because of startup timeout'
|
||||
@@ -1760,7 +1787,7 @@ class Ha(object):
|
||||
elif self.cluster.is_unlocked() and not self.is_paused():
|
||||
# "bootstrap", but data directory is not empty
|
||||
if not self.state_handler.cb_called and self.state_handler.is_running() \
|
||||
and not self.state_handler.is_leader():
|
||||
and not self.state_handler.is_primary():
|
||||
self._join_aborted = True
|
||||
logger.error('No initialize key in DCS and PostgreSQL is running as replica, aborting start')
|
||||
logger.error('Please first start Patroni on the node running as primary')
|
||||
@@ -1803,7 +1830,7 @@ class Ha(object):
|
||||
create_slots = self._sync_replication_slots(False)
|
||||
|
||||
if not self.state_handler.cb_called:
|
||||
if not is_promoting and not self.state_handler.is_leader():
|
||||
if not is_promoting and not self.state_handler.is_primary():
|
||||
self._rewind.trigger_check_diverged_lsn()
|
||||
self.state_handler.call_nowait(CallbackAction.ON_START)
|
||||
|
||||
@@ -1828,7 +1855,7 @@ class Ha(object):
|
||||
|
||||
def _handle_dcs_error(self) -> str:
|
||||
if not self.is_paused() and self.state_handler.is_running():
|
||||
if self.state_handler.is_leader():
|
||||
if self.state_handler.is_primary():
|
||||
if self.is_failsafe_mode() and self.check_failsafe_topology():
|
||||
self.set_is_leader(True)
|
||||
self._failsafe.set_is_active(time.time())
|
||||
@@ -1904,9 +1931,8 @@ class Ha(object):
|
||||
# If we know that there are replicas that received the shutdown checkpoint
|
||||
# location, we can remove the leader key and allow them to start leader race.
|
||||
|
||||
# for a manual failover/switchover with a candidate, we should check the requested candidate only
|
||||
if self.is_failover_possible(self.get_failover_candidates(), cluster_lsn=checkpoint_location):
|
||||
self.dcs.delete_leader(checkpoint_location)
|
||||
if self.is_failover_possible(cluster_lsn=checkpoint_location):
|
||||
self.dcs.delete_leader(self.cluster.leader, checkpoint_location)
|
||||
status['deleted'] = True
|
||||
else:
|
||||
self.dcs.write_leader_optime(checkpoint_location)
|
||||
@@ -1923,7 +1949,7 @@ class Ha(object):
|
||||
if not self.state_handler.is_running():
|
||||
if self.is_leader() and not status['deleted']:
|
||||
checkpoint_location = self.state_handler.latest_checkpoint_location()
|
||||
self.dcs.delete_leader(checkpoint_location)
|
||||
self.dcs.delete_leader(self.cluster.leader, checkpoint_location)
|
||||
self.touch_member()
|
||||
else:
|
||||
# XXX: what about when Patroni is started as the wrong user that has access to the watchdog device
|
||||
@@ -1967,23 +1993,31 @@ class Ha(object):
|
||||
name = member.name if member else 'remote_member:{}'.format(uuid.uuid1())
|
||||
return RemoteMember(name, data)
|
||||
|
||||
def get_failover_candidates(self, check_sync: bool = False) -> List[Member]:
|
||||
"""Return list of candidates for either manual or automatic failover.
|
||||
def get_failover_candidates(self, exclude_failover_candidate: bool) -> List[Member]:
|
||||
"""Return a list of candidates for either manual or automatic failover.
|
||||
|
||||
Mainly used to later be passed to ``Ha.is_failover_possible()``.
|
||||
Exclude non-sync members when in synchronous mode, the current node (its checks are always performed earlier)
|
||||
and the candidate if required. If failover candidate exclusion is not requested and a candidate is specified
|
||||
in the /failover key, return the candidate only.
|
||||
The result is further evaluated in the caller :func:`Ha.is_failover_possible` to check if any member is actually
|
||||
healthy enough and is allowed to poromote.
|
||||
|
||||
:param check_sync: if ``True``, also check against the sync key members
|
||||
:param exclude_failover_candidate: if ``True``, exclude :attr:`failover.candidate` from the candidates.
|
||||
|
||||
:returns: a list of ``Member`` ojects or an empty list if there is no candidate available
|
||||
:returns: a list of :class:`Member` ojects or an empty list if there is no candidate available.
|
||||
"""
|
||||
failover = self.cluster.failover
|
||||
if check_sync:
|
||||
# TODO: allow manual failover (=no leader specified) to async node
|
||||
# every sync_standby or the candidate specified if is in sync_standbys
|
||||
return [m for m in self.cluster.members
|
||||
if self.cluster.sync.matches(m.name)
|
||||
and (not failover or not failover.candidate or m.name == failover.candidate)]
|
||||
else:
|
||||
# every member or the candidate specified
|
||||
return [m for m in self.cluster.members
|
||||
if not failover or not failover.candidate or m.name == failover.candidate]
|
||||
exclude = [self.state_handler.name] + ([failover.candidate] if failover and exclude_failover_candidate else [])
|
||||
|
||||
def is_eligible(node: Member) -> bool:
|
||||
# in synchronous mode we allow failover (not switchover!) to async node
|
||||
if self.sync_mode_is_active() and not self.cluster.sync.matches(node.name)\
|
||||
and not (failover and not failover.leader):
|
||||
return False
|
||||
# Don't spend time on "nofailover" nodes checking.
|
||||
# We also don't need nodes which we can't query with the api in the list.
|
||||
return node.name not in exclude and \
|
||||
not node.nofailover and bool(node.api_url) and \
|
||||
(not failover or not failover.candidate or node.name == failover.candidate)
|
||||
|
||||
return list(filter(is_eligible, self.cluster.members))
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
from enum import Enum
|
||||
from typing import Optional, Tuple, TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
import datetime
|
||||
|
||||
from .dcs import Cluster
|
||||
from .ha import Patroni
|
||||
from .utils import ParseScheduleErrors
|
||||
|
||||
from .utils import parse_schedule
|
||||
|
||||
|
||||
class ManualFailoverPrecheckStatus(Enum):
|
||||
FAILOVER_NO_CANDIDATE = ('Failover could be performed only to a specific candidate', 400)
|
||||
SWITCHOVER_NO_LEADER = ('Switchover could be performed only from a specific leader', 400)
|
||||
SCHEDULED_FAILOVER = ("Failover can't be scheduled", 400)
|
||||
SCHEDULED_SWITCHOVER_PAUSE = ("Can't schedule switchover in the paused state", 400)
|
||||
SWITCHOVER_PAUSE_NO_CANDIDATE = ('Switchover is possible only to a specific candidate in a paused state', 400)
|
||||
SWITCHOVER_TO_LEADER = ('Switchover target and source are the same', 400)
|
||||
|
||||
CLUSTER_NO_LEADER = ('Cluster {cluster_name} has no leader', 412)
|
||||
LEADER_NOT_MEMBER = ('Member {leader} is not the leader of cluster {cluster_name}', 412)
|
||||
CANDIDATE_NOT_SYNC_STANDBY = ('candidate name does not match with sync_standby', 412)
|
||||
NO_SYNC_CANDIDATE = ('{action} is not possible: can not find sync_standby', 412)
|
||||
ONLY_LEADER = ('{action} is not possible: cluster does not have members except leader', 412)
|
||||
CANDIDATE_NOT_MEMEBER = ('Member {candidate} does not exist in cluster {cluster_name} or is tagged as nofailover',
|
||||
412)
|
||||
NO_GOOD_CANDIDATES = ('{action} is not possible: no good candidates have been found', 412)
|
||||
|
||||
CHECK_PASSED = ('', None)
|
||||
|
||||
|
||||
class ManualFailover(object):
|
||||
|
||||
def __init__(self, action: str, cluster: 'Cluster',
|
||||
leader: Optional[str], candidate: Optional[str], scheduled: Optional[str],
|
||||
paused: bool = False, sync_mode: bool = False, patroni_obj: Optional['Patroni'] = None) -> None:
|
||||
self.action = action
|
||||
self.cluster = cluster
|
||||
self.leader = leader
|
||||
self.candidate = candidate
|
||||
self.scheduled = scheduled
|
||||
self.paused = paused
|
||||
self.sync_mode = sync_mode
|
||||
self.patroni = patroni_obj
|
||||
|
||||
def parse_scheduled(self) -> Tuple[Optional['ParseScheduleErrors'], Optional['datetime.datetime']]:
|
||||
return parse_schedule(self.scheduled)
|
||||
|
||||
def run_precheck(self) -> ManualFailoverPrecheckStatus:
|
||||
if self.action == 'failover' and not self.candidate:
|
||||
return ManualFailoverPrecheckStatus.FAILOVER_NO_CANDIDATE
|
||||
elif self.action == 'switchover' and not self.leader:
|
||||
return ManualFailoverPrecheckStatus.SWITCHOVER_NO_LEADER
|
||||
|
||||
if self.scheduled:
|
||||
if self.action == 'failover':
|
||||
return ManualFailoverPrecheckStatus.SCHEDULED_FAILOVER
|
||||
elif self.paused:
|
||||
return ManualFailoverPrecheckStatus.SCHEDULED_SWITCHOVER_PAUSE
|
||||
|
||||
if self.paused and not self.candidate:
|
||||
return ManualFailoverPrecheckStatus.SWITCHOVER_PAUSE_NO_CANDIDATE
|
||||
|
||||
if self.leader == self.candidate:
|
||||
return ManualFailoverPrecheckStatus.SWITCHOVER_TO_LEADER
|
||||
|
||||
if self.action == 'switchover':
|
||||
if self.cluster.leader is None or not self.cluster.leader.name:
|
||||
return ManualFailoverPrecheckStatus.CLUSTER_NO_LEADER
|
||||
if self.cluster.leader.name != self.leader:
|
||||
return ManualFailoverPrecheckStatus.LEADER_NOT_MEMBER
|
||||
|
||||
if self.candidate:
|
||||
if self.action == 'switchover' and self.sync_mode and not self.cluster.sync.matches(self.candidate):
|
||||
return ManualFailoverPrecheckStatus.CANDIDATE_NOT_SYNC_STANDBY
|
||||
members = [m for m in self.cluster.members if m.name == self.candidate]
|
||||
if not members:
|
||||
return ManualFailoverPrecheckStatus.CANDIDATE_NOT_MEMEBER
|
||||
elif self.sync_mode:
|
||||
members = [m for m in self.cluster.members if self.cluster.sync.matches(m.name)]
|
||||
if not members:
|
||||
return ManualFailoverPrecheckStatus.NO_SYNC_CANDIDATE
|
||||
else:
|
||||
members = [m for m in self.cluster.members if not self.cluster.leader or m.name != self.cluster.leader.name and m.api_url]
|
||||
if not members:
|
||||
return ManualFailoverPrecheckStatus.ONLY_LEADER
|
||||
|
||||
if self.patroni and not self.patroni.ha.has_members_eligible_to_promote(members, fast_path=True):
|
||||
return ManualFailoverPrecheckStatus.NO_GOOD_CANDIDATES
|
||||
|
||||
return ManualFailoverPrecheckStatus.CHECK_PASSED
|
||||
@@ -57,8 +57,8 @@ class Postgresql(object):
|
||||
TL_LSN = ("CASE WHEN pg_catalog.pg_is_in_recovery() THEN 0 "
|
||||
"ELSE ('x' || pg_catalog.substr(pg_catalog.pg_{0}file_name("
|
||||
"pg_catalog.pg_current_{0}_{1}()), 1, 8))::bit(32)::int END, " # primary timeline
|
||||
"CASE WHEN pg_catalog.pg_is_in_recovery() THEN 0 "
|
||||
"ELSE pg_catalog.pg_{0}_{1}_diff(pg_catalog.pg_current_{0}_{1}(), '0/0')::bigint END, " # write_lsn
|
||||
"CASE WHEN pg_catalog.pg_is_in_recovery() THEN 0 ELSE "
|
||||
"pg_catalog.pg_{0}_{1}_diff(pg_catalog.pg_current_{0}{2}_{1}(), '0/0')::bigint END, " # wal(_flush)?_lsn
|
||||
"pg_catalog.pg_{0}_{1}_diff(pg_catalog.pg_last_{0}_replay_{1}(), '0/0')::bigint, "
|
||||
"pg_catalog.pg_{0}_{1}_diff(COALESCE(pg_catalog.pg_last_{0}_receive_{1}(), '0/0'), '0/0')::bigint, "
|
||||
"pg_catalog.pg_is_in_recovery() AND pg_catalog.pg_is_{0}_replay_paused()")
|
||||
@@ -118,28 +118,17 @@ class Postgresql(object):
|
||||
# Last known running process
|
||||
self._postmaster_proc = None
|
||||
|
||||
if self.is_running():
|
||||
# If we found postmaster process we need to figure out whether postgres is accepting connections
|
||||
self.set_state('starting')
|
||||
self.check_startup_state_changed()
|
||||
|
||||
if self.state == 'running': # we are "joining" already running postgres
|
||||
# we know that PostgreSQL is accepting connections and can read some GUC's from pg_settings
|
||||
self.config.load_current_server_parameters()
|
||||
|
||||
self.set_role('master' if self.is_leader() else 'replica')
|
||||
|
||||
if self.is_running(): # we are "joining" already running postgres
|
||||
self.set_state('running')
|
||||
self.set_role('master' if self.is_primary() else 'replica')
|
||||
# postpone writing postgresql.conf for 12+ because recovery parameters are not yet known
|
||||
if self.major_version < 120000 or self.is_primary():
|
||||
self.config.write_postgresql_conf()
|
||||
hba_saved = self.config.replace_pg_hba()
|
||||
ident_saved = self.config.replace_pg_ident()
|
||||
|
||||
if self.major_version < 120000 or self.role in ('master', 'primary'):
|
||||
# If PostgreSQL is running as a primary or we run PostgreSQL that is older than 12 we can
|
||||
# call reload_config() once again (the first call happened in the ConfigHandler constructor),
|
||||
# so that it can figure out if config files should be updated and pg_ctl reload executed.
|
||||
self.config.reload_config(config, sighup=bool(hba_saved or ident_saved))
|
||||
elif hba_saved or ident_saved:
|
||||
if hba_saved or ident_saved:
|
||||
self.reload()
|
||||
elif not self.is_running() and self.role in ('master', 'primary'):
|
||||
elif self.role in ('master', 'primary'):
|
||||
self.set_role('demoted')
|
||||
|
||||
@property
|
||||
@@ -170,6 +159,11 @@ class Postgresql(object):
|
||||
def wal_name(self) -> str:
|
||||
return 'wal' if self._major_version >= 100000 else 'xlog'
|
||||
|
||||
@property
|
||||
def wal_flush(self) -> str:
|
||||
"""For PostgreSQL 9.6 onwards we want to use pg_current_wal_flush_lsn()/pg_current_xlog_flush_location()."""
|
||||
return '_flush' if self._major_version >= 90600 else ''
|
||||
|
||||
@property
|
||||
def lsn_name(self) -> str:
|
||||
return 'lsn' if self._major_version >= 100000 else 'location'
|
||||
@@ -224,7 +218,7 @@ class Postgresql(object):
|
||||
else:
|
||||
extra = "0, NULL, NULL, NULL, NULL, NULL, NULL" + extra
|
||||
|
||||
return ("SELECT " + self.TL_LSN + ", {2}").format(self.wal_name, self.lsn_name, extra)
|
||||
return ("SELECT " + self.TL_LSN + ", {3}").format(self.wal_name, self.lsn_name, self.wal_flush, extra)
|
||||
|
||||
@property
|
||||
def available_gucs(self) -> CaseInsensitiveSet:
|
||||
@@ -338,36 +332,50 @@ class Postgresql(object):
|
||||
self._connection.set_conn_kwargs(kwargs.copy())
|
||||
self.citus_handler.set_conn_kwargs(kwargs.copy())
|
||||
|
||||
def _query(self, sql: str, *params: Any) -> Union['Cursor[Any]', 'cursor']:
|
||||
"""We are always using the same cursor, therefore this method is not thread-safe!!!
|
||||
You can call it from different threads only if you are holding explicit `AsyncExecutor` lock,
|
||||
because the main thread is always holding this lock when running HA cycle."""
|
||||
cursor = None
|
||||
try:
|
||||
cursor = self._connection.cursor()
|
||||
cursor.execute(sql.encode('utf-8'), params or None)
|
||||
return cursor
|
||||
except psycopg.Error as e:
|
||||
if cursor and cursor.connection.closed == 0:
|
||||
# When connected via unix socket, psycopg2 can't recoginze 'connection lost'
|
||||
# and leaves `_cursor_holder.connection.closed == 0`, but psycopg2.OperationalError
|
||||
# is still raised (what is correct). It doesn't make sense to continiue with existing
|
||||
# connection and we will close it, to avoid its reuse by the `cursor` method.
|
||||
if isinstance(e, psycopg.OperationalError):
|
||||
self._connection.close()
|
||||
else:
|
||||
raise e
|
||||
if self.state == 'restarting':
|
||||
raise RetryFailedError('cluster is being restarted')
|
||||
raise PostgresConnectionException('connection problems')
|
||||
def _query(self, sql: str, *params: Any) -> List[Tuple[Any, ...]]:
|
||||
"""Execute *sql* query with *params* and optionally return results.
|
||||
|
||||
def query(self, sql: str, *args: Any, **kwargs: Any) -> Union['Cursor[Any]', 'cursor']:
|
||||
if not kwargs.get('retry', True):
|
||||
return self._query(sql, *args)
|
||||
:param sql: SQL statement to execute.
|
||||
:param params: parameters to pass.
|
||||
|
||||
:returns: a query response as a list of tuples if there is any.
|
||||
:raises:
|
||||
:exc:`~psycopg.Error` if had issues while executing *sql*.
|
||||
|
||||
:exc:`~patroni.exceptions.PostgresConnectionException`: if had issues while connecting to the database.
|
||||
|
||||
:exc:`~patroni.utils.RetryFailedError`: if it was detected that connection/query failed due to PostgreSQL
|
||||
restart.
|
||||
"""
|
||||
try:
|
||||
return self.retry(self._query, sql, *args)
|
||||
except RetryFailedError as e:
|
||||
raise PostgresConnectionException(str(e))
|
||||
return self._connection.query(sql, *params)
|
||||
except PostgresConnectionException as exc:
|
||||
if self.state == 'restarting':
|
||||
raise RetryFailedError('cluster is being restarted') from exc
|
||||
raise
|
||||
|
||||
def query(self, sql: str, *params: Any, retry: bool = True) -> List[Tuple[Any, ...]]:
|
||||
"""Execute *sql* query with *params* and optionally return results.
|
||||
|
||||
:param sql: SQL statement to execute.
|
||||
:param params: parameters to pass.
|
||||
:param retry: whether the query should be retried upon failure or given up immediately.
|
||||
|
||||
:returns: a query response as a list of tuples if there is any.
|
||||
:raises:
|
||||
:exc:`~psycopg.Error` if had issues while executing *sql*.
|
||||
|
||||
:exc:`~patroni.exceptions.PostgresConnectionException`: if had issues while connecting to the database.
|
||||
|
||||
:exc:`~patroni.utils.RetryFailedError`: if it was detected that connection/query failed due to PostgreSQL
|
||||
restart or if retry deadline was exceeded.
|
||||
"""
|
||||
if not retry:
|
||||
return self._query(sql, *params)
|
||||
try:
|
||||
return self.retry(self._query, sql, *params)
|
||||
except RetryFailedError as exc:
|
||||
raise PostgresConnectionException(str(exc)) from exc
|
||||
|
||||
def pg_control_exists(self) -> bool:
|
||||
return os.path.isfile(self._pg_control)
|
||||
@@ -445,7 +453,7 @@ class Postgresql(object):
|
||||
def _cluster_info_state_get(self, name: str) -> Optional[Any]:
|
||||
if not self._cluster_info_state:
|
||||
try:
|
||||
result = self._is_leader_retry(self._query, self.cluster_info_query).fetchone()
|
||||
result = self._is_leader_retry(self._query, self.cluster_info_query)[0]
|
||||
cluster_info_state = dict(zip(['timeline', 'wal_position', 'replayed_location',
|
||||
'received_location', 'replay_paused', 'pg_control_timeline',
|
||||
'received_tli', 'slot_name', 'conninfo', 'receiver_state',
|
||||
@@ -495,14 +503,14 @@ class Postgresql(object):
|
||||
""":returns: a result set of 'SELECT * FROM pg_stat_replication'."""
|
||||
return self._cluster_info_state_get('pg_stat_replication') or []
|
||||
|
||||
def replication_state_from_parameters(self, is_leader: bool, receiver_state: Optional[str],
|
||||
def replication_state_from_parameters(self, is_primary: bool, receiver_state: Optional[str],
|
||||
restore_command: Optional[str]) -> Optional[str]:
|
||||
"""Figure out the replication state from input parameters.
|
||||
|
||||
.. note::
|
||||
This method could be only called when Postgres is up, running and queries are successfuly executed.
|
||||
|
||||
:is_leader: `True` is postgres is not running in recovery
|
||||
:is_primary: `True` is postgres is not running in recovery
|
||||
:receiver_state: value from `pg_stat_get_wal_receiver.state` or None if Postgres is older than 9.6
|
||||
:restore_command: value of ``restore_command`` GUC for PostgreSQL 12+ or
|
||||
`postgresql.recovery_conf.restore_command` if it is set in Patroni configuration
|
||||
@@ -511,7 +519,7 @@ class Postgresql(object):
|
||||
- 'streaming' if replica is streaming according to the `pg_stat_wal_receiver` view;
|
||||
- 'in archive recovery' if replica isn't streaming and there is a `restore_command`
|
||||
"""
|
||||
if self._major_version >= 90600 and not is_leader:
|
||||
if self._major_version >= 90600 and not is_primary:
|
||||
if receiver_state == 'streaming':
|
||||
return 'streaming'
|
||||
# For Postgres older than 12 we get `restore_command` from Patroni config, otherwise we check GUC
|
||||
@@ -526,11 +534,11 @@ class Postgresql(object):
|
||||
|
||||
:returns: ``streaming``, ``in archive recovery``, or ``None``
|
||||
"""
|
||||
return self.replication_state_from_parameters(self.is_leader(),
|
||||
return self.replication_state_from_parameters(self.is_primary(),
|
||||
self._cluster_info_state_get('receiver_state'),
|
||||
self._cluster_info_state_get('restore_command'))
|
||||
|
||||
def is_leader(self) -> bool:
|
||||
def is_primary(self) -> bool:
|
||||
try:
|
||||
return bool(self._cluster_info_state_get('timeline'))
|
||||
except PostgresConnectionException:
|
||||
@@ -563,7 +571,7 @@ class Postgresql(object):
|
||||
r'lsn: ([0-9A-Fa-f]+/[0-9A-Fa-f]+), prev ([0-9A-Fa-f]+/[0-9A-Fa-f]+), '
|
||||
r'.*?desc: (.+)', out.decode('utf-8'))
|
||||
if match:
|
||||
return match.group(1), match.group(2), match.group(3), match.group(4)
|
||||
return match.groups()
|
||||
return None, None, None, None
|
||||
|
||||
def latest_checkpoint_location(self) -> Optional[int]:
|
||||
@@ -887,11 +895,10 @@ class Postgresql(object):
|
||||
|
||||
def _wait_for_connection_close(self, postmaster: PostmasterProcess) -> None:
|
||||
try:
|
||||
with self.connection().cursor() as cur:
|
||||
while postmaster.is_running(): # Need a timeout here?
|
||||
cur.execute("SELECT 1")
|
||||
time.sleep(STOP_POLLING_INTERVAL)
|
||||
except psycopg.Error:
|
||||
while postmaster.is_running(): # Need a timeout here?
|
||||
self._connection.query("SELECT 1")
|
||||
time.sleep(STOP_POLLING_INTERVAL)
|
||||
except (psycopg.Error, PostgresConnectionException):
|
||||
pass
|
||||
|
||||
def reload(self, block_callbacks: bool = False) -> bool:
|
||||
@@ -1019,7 +1026,7 @@ class Postgresql(object):
|
||||
return None, None
|
||||
|
||||
@contextmanager
|
||||
def get_replication_connection_cursor(self, host: Optional[str] = None, port: Union[int, str] = 5432,
|
||||
def get_replication_connection_cursor(self, host: Optional[str] = None, port: int = 5432,
|
||||
**kwargs: Any) -> Iterator[Union['cursor', 'Cursor[Any]']]:
|
||||
conn_kwargs = self.config.replication.copy()
|
||||
conn_kwargs.update(host=host, port=int(port) if port else None, user=conn_kwargs.pop('username'),
|
||||
@@ -1178,9 +1185,9 @@ class Postgresql(object):
|
||||
return ret
|
||||
|
||||
@staticmethod
|
||||
def _wal_position(is_leader: bool, wal_position: int,
|
||||
def _wal_position(is_primary: bool, wal_position: int,
|
||||
received_location: Optional[int], replayed_location: Optional[int]) -> int:
|
||||
return wal_position if is_leader else max(received_location or 0, replayed_location or 0)
|
||||
return wal_position if is_primary else max(received_location or 0, replayed_location or 0)
|
||||
|
||||
def timeline_wal_position(self) -> Tuple[int, int, Optional[int]]:
|
||||
# This method could be called from different threads (simultaneously with some other `_query` calls).
|
||||
@@ -1192,31 +1199,21 @@ class Postgresql(object):
|
||||
received_location = self.received_location()
|
||||
pg_control_timeline = self._cluster_info_state_get('pg_control_timeline')
|
||||
else:
|
||||
with self.connection().cursor() as cursor:
|
||||
cursor.execute(self.cluster_info_query.encode('utf-8'))
|
||||
row = cursor.fetchone()
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
assert row is not None
|
||||
(timeline, wal_position, replayed_location, received_location, _, pg_control_timeline) = row[:6]
|
||||
timeline, wal_position, replayed_location, received_location, _, pg_control_timeline = \
|
||||
self._query(self.cluster_info_query)[0][:6]
|
||||
|
||||
wal_position = self._wal_position(bool(timeline), wal_position, received_location, replayed_location)
|
||||
return (timeline, wal_position, pg_control_timeline)
|
||||
return timeline, wal_position, pg_control_timeline
|
||||
|
||||
def postmaster_start_time(self) -> Optional[str]:
|
||||
try:
|
||||
query = "SELECT " + self.POSTMASTER_START_TIME
|
||||
if current_thread().ident == self.__thread_ident:
|
||||
row = self.query(query).fetchone()
|
||||
else:
|
||||
with self.connection().cursor() as cursor:
|
||||
cursor.execute(query)
|
||||
row = cursor.fetchone()
|
||||
return row[0].isoformat(sep=' ') if row else None
|
||||
sql = "SELECT " + self.POSTMASTER_START_TIME
|
||||
return self.query(sql, retry=current_thread().ident == self.__thread_ident)[0][0].isoformat(sep=' ')
|
||||
except psycopg.Error:
|
||||
return None
|
||||
|
||||
def last_operation(self) -> int:
|
||||
return self._wal_position(self.is_leader(), self._cluster_info_state_get('wal_position') or 0,
|
||||
return self._wal_position(self.is_primary(), self._cluster_info_state_get('wal_position') or 0,
|
||||
self.received_location(), self.replayed_location())
|
||||
|
||||
def configure_server_parameters(self) -> None:
|
||||
|
||||
@@ -11,8 +11,6 @@ from ..dcs import CITUS_COORDINATOR_GROUP_ID, Cluster
|
||||
from ..psycopg import connect, quote_ident
|
||||
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
from psycopg import Cursor
|
||||
from psycopg2 import cursor
|
||||
from . import Postgresql
|
||||
|
||||
CITUS_SLOT_NAME_RE = re.compile(r'^citus_shard_(move|split)_slot(_[1-9][0-9]*){2,3}$')
|
||||
@@ -109,12 +107,10 @@ class CitusHandler(Thread):
|
||||
self._tasks[:] = []
|
||||
self._in_flight = None
|
||||
|
||||
def query(self, sql: str, *params: Any) -> Union['Cursor[Any]', 'cursor']:
|
||||
def query(self, sql: str, *params: Any) -> List[Tuple[Any, ...]]:
|
||||
try:
|
||||
logger.debug('query(%s, %s)', sql, params)
|
||||
cursor = self._connection.cursor()
|
||||
cursor.execute(sql.encode('utf-8'), params or None)
|
||||
return cursor
|
||||
return self._connection.query(sql, *params)
|
||||
except Exception as e:
|
||||
logger.error('Exception when executing query "%s", (%s): %r', sql, params, e)
|
||||
self._connection.close()
|
||||
@@ -132,13 +128,13 @@ class CitusHandler(Thread):
|
||||
self._schedule_load_pg_dist_node = False
|
||||
|
||||
try:
|
||||
cursor = self.query("SELECT nodeid, groupid, nodename, nodeport, noderole"
|
||||
" FROM pg_catalog.pg_dist_node WHERE noderole = 'primary'")
|
||||
rows = self.query("SELECT nodeid, groupid, nodename, nodeport, noderole"
|
||||
" FROM pg_catalog.pg_dist_node WHERE noderole = 'primary'")
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
with self._condition:
|
||||
self._pg_dist_node = {r[1]: PgDistNode(r[1], r[2], r[3], 'after_promote', r[0]) for r in cursor}
|
||||
self._pg_dist_node = {r[1]: PgDistNode(r[1], r[2], r[3], 'after_promote', r[0]) for r in rows}
|
||||
return True
|
||||
|
||||
def sync_pg_dist_node(self, cluster: Cluster) -> None:
|
||||
@@ -211,10 +207,8 @@ class CitusHandler(Thread):
|
||||
self.query('SELECT pg_catalog.citus_update_node(%s, %s, %s, true, %s)',
|
||||
task.nodeid, task.host, task.port, task.cooldown)
|
||||
elif task.event != 'before_demote':
|
||||
row = self.query("SELECT pg_catalog.citus_add_node(%s, %s, %s, 'primary', 'default')",
|
||||
task.host, task.port, task.group).fetchone()
|
||||
if row is not None:
|
||||
task.nodeid = row[0]
|
||||
task.nodeid = self.query("SELECT pg_catalog.citus_add_node(%s, %s, %s, 'primary', 'default')",
|
||||
task.host, task.port, task.group)[0][0]
|
||||
|
||||
def process_task(self, task: PgDistNode) -> bool:
|
||||
"""Updates a single row in `pg_dist_node` table, optionally in a transaction.
|
||||
@@ -407,14 +401,14 @@ class CitusHandler(Thread):
|
||||
parameters['shared_preload_libraries'] = ','.join(['citus'] + shared_preload_libraries)
|
||||
|
||||
# if not explicitly set Citus overrides max_prepared_transactions to max_connections*2
|
||||
if parameters['max_prepared_transactions'] == 0:
|
||||
if parameters.get('max_prepared_transactions') == 0:
|
||||
parameters['max_prepared_transactions'] = parameters['max_connections'] * 2
|
||||
|
||||
# Resharding in Citus implemented using logical replication
|
||||
parameters['wal_level'] = 'logical'
|
||||
|
||||
def ignore_replication_slot(self, slot: Dict[str, str]) -> bool:
|
||||
if isinstance(self._config, dict) and self._postgresql.is_leader() and\
|
||||
if isinstance(self._config, dict) and self._postgresql.is_primary() and\
|
||||
slot['type'] == 'logical' and slot['database'] == self._config['database']:
|
||||
m = CITUS_SLOT_NAME_RE.match(slot['name'])
|
||||
return bool(m and {'move': 'pgoutput', 'split': 'citus'}.get(m.group(1)) == slot['plugin'])
|
||||
|
||||
@@ -14,7 +14,7 @@ from typing import Any, Collection, Dict, Iterator, List, Optional, Union, Tuple
|
||||
from .validator import recovery_parameters, transform_postgresql_parameter_value, transform_recovery_parameter_value
|
||||
from ..collections import CaseInsensitiveDict, CaseInsensitiveSet
|
||||
from ..dcs import Leader, Member, RemoteMember, slot_name_from_member_name
|
||||
from ..exceptions import PatroniFatalException, PostgresConnectionException
|
||||
from ..exceptions import PatroniFatalException
|
||||
from ..file_perm import pg_perm
|
||||
from ..utils import compare_values, parse_bool, parse_int, split_host_port, uri, validate_directory, is_subpath
|
||||
from ..validator import IntValidator, EnumValidator
|
||||
@@ -326,22 +326,14 @@ class ConfigHandler(object):
|
||||
.format(self._pgpass))
|
||||
self._passfile = None
|
||||
self._passfile_mtime = None
|
||||
self._synchronous_standby_names = None
|
||||
self._postmaster_ctime = None
|
||||
self._current_recovery_params: Optional[CaseInsensitiveDict] = None
|
||||
self._config = {}
|
||||
self._recovery_params = CaseInsensitiveDict()
|
||||
self._server_parameters: CaseInsensitiveDict = CaseInsensitiveDict()
|
||||
self._server_parameters: CaseInsensitiveDict
|
||||
self.reload_config(config)
|
||||
|
||||
def load_current_server_parameters(self) -> None:
|
||||
"""Read GUC's values from ``pg_settings`` when Patroni is joining the the postgres that is already running."""
|
||||
exclude = [name.lower() for name, value in self.CMDLINE_OPTIONS.items() if value[1] == _false_validator] \
|
||||
+ [name.lower() for name in self._RECOVERY_PARAMETERS]
|
||||
self._server_parameters = CaseInsensitiveDict({r[0]: r[1] for r in self._postgresql.query(
|
||||
"SELECT name, pg_catalog.current_setting(name) FROM pg_catalog.pg_settings"
|
||||
" WHERE (source IN ('command line', 'environment variable') OR sourcefile = %s)"
|
||||
" AND pg_catalog.lower(name) != ALL(%s)", self._postgresql_conf, exclude)})
|
||||
|
||||
def setup_server_parameters(self) -> None:
|
||||
self._server_parameters = self.get_server_parameters(self._config)
|
||||
self._adjust_recovery_parameters()
|
||||
@@ -631,24 +623,7 @@ class ConfigHandler(object):
|
||||
'recovery_target_action', 'standby_mode', self._triggerfile_wrong_name})
|
||||
return CaseInsensitiveSet(self._RECOVERY_PARAMETERS - skip_params)
|
||||
|
||||
def _read_recovery_params(self) -> Tuple[Optional[CaseInsensitiveDict], bool]:
|
||||
"""Read current recovery parameters values.
|
||||
|
||||
.. note::
|
||||
We query Postgres only if we detected that Postgresql was restarted
|
||||
or when at least one of the following files was updated:
|
||||
|
||||
* ``postgresql.conf``;
|
||||
* ``postgresql.auto.conf``;
|
||||
* ``passfile`` that is used in the ``primary_conninfo``.
|
||||
|
||||
:returns: a tuple with two elements:
|
||||
|
||||
* :class:`CaseInsensitiveDict` object with current values of recovery parameters,
|
||||
or ``None`` if no configuration files were updated;
|
||||
|
||||
* ``True`` if new values of recovery parameters were queried, ``False`` otherwise.
|
||||
"""
|
||||
def _read_recovery_params(self) -> Tuple[Optional[CaseInsensitiveDict], Optional[bool]]:
|
||||
if self._postgresql.is_starting():
|
||||
return None, False
|
||||
|
||||
@@ -669,20 +644,11 @@ class ConfigHandler(object):
|
||||
self._postgresql_conf_mtime = pg_conf_mtime
|
||||
self._auto_conf_mtime = auto_conf_mtime
|
||||
self._postmaster_ctime = postmaster_ctime
|
||||
except Exception as exc:
|
||||
if all((isinstance(exc, PostgresConnectionException),
|
||||
self._postgresql_conf_mtime == pg_conf_mtime,
|
||||
self._auto_conf_mtime == auto_conf_mtime,
|
||||
self._passfile_mtime == passfile_mtime,
|
||||
self._postmaster_ctime != postmaster_ctime)):
|
||||
# We detected that the connection to postgres fails, but the process creation time of the postmaster
|
||||
# doesn't match the old value. It is an indicator that Postgres crashed and either doing crash
|
||||
# recovery or down. In this case we return values like nothing changed in the config.
|
||||
return None, False
|
||||
except Exception:
|
||||
values = None
|
||||
return values, True
|
||||
|
||||
def _read_recovery_params_pre_v12(self) -> Tuple[Optional[CaseInsensitiveDict], bool]:
|
||||
def _read_recovery_params_pre_v12(self) -> Tuple[Optional[CaseInsensitiveDict], Optional[bool]]:
|
||||
recovery_conf_mtime = mtime(self._recovery_conf)
|
||||
passfile_mtime = mtime(self._passfile) if self._passfile else False
|
||||
if recovery_conf_mtime == self._recovery_conf_mtime and passfile_mtime == self._passfile_mtime:
|
||||
@@ -930,15 +896,14 @@ class ConfigHandler(object):
|
||||
listen_addresses, port = split_host_port(config['listen'], 5432)
|
||||
parameters.update(cluster_name=self._postgresql.scope, listen_addresses=listen_addresses, port=str(port))
|
||||
if not self._postgresql.global_config or self._postgresql.global_config.is_synchronous_mode:
|
||||
synchronous_standby_names = self._server_parameters.get('synchronous_standby_names')
|
||||
if synchronous_standby_names is None:
|
||||
if self._synchronous_standby_names is None:
|
||||
if self._postgresql.global_config and self._postgresql.global_config.is_synchronous_mode_strict\
|
||||
and self._postgresql.role in ('master', 'primary', 'promoted'):
|
||||
parameters['synchronous_standby_names'] = '*'
|
||||
else:
|
||||
parameters.pop('synchronous_standby_names', None)
|
||||
else:
|
||||
parameters['synchronous_standby_names'] = synchronous_standby_names
|
||||
parameters['synchronous_standby_names'] = self._synchronous_standby_names
|
||||
|
||||
# Handle hot_standby <-> replica rename
|
||||
if parameters.get('wal_level') == ('hot_standby' if self._postgresql.major_version >= 90600 else 'replica'):
|
||||
@@ -1014,14 +979,17 @@ class ConfigHandler(object):
|
||||
self._postgresql.connection_string = uri('postgres', netloc, self._postgresql.database)
|
||||
self._postgresql.set_connection_kwargs(self.local_connect_kwargs)
|
||||
|
||||
def _get_pg_settings(self, names: Collection[str]) -> Dict[Any, Tuple[Any, ...]]:
|
||||
def _get_pg_settings(
|
||||
self, names: Collection[str]
|
||||
) -> Dict[str, Tuple[str, str, Optional[str], str, str, Optional[str]]]:
|
||||
return {r[0]: r for r in self._postgresql.query(('SELECT name, setting, unit, vartype, context, sourcefile'
|
||||
+ ' FROM pg_catalog.pg_settings '
|
||||
+ ' WHERE pg_catalog.lower(name) = ANY(%s)'),
|
||||
[n.lower() for n in names])}
|
||||
|
||||
@staticmethod
|
||||
def _handle_wal_buffers(old_values: Dict[Any, Tuple[Any, ...]], changes: CaseInsensitiveDict) -> None:
|
||||
def _handle_wal_buffers(old_values: Dict[str, Tuple[str, str, Optional[str], str, str, Optional[str]]],
|
||||
changes: CaseInsensitiveDict) -> None:
|
||||
wal_block_size = parse_int(old_values['wal_block_size'][1]) or 8192
|
||||
wal_segment_size = old_values['wal_segment_size']
|
||||
wal_segment_unit = parse_int(wal_segment_size[2], 'B') or 8192 \
|
||||
@@ -1125,10 +1093,10 @@ class ConfigHandler(object):
|
||||
if self._postgresql.major_version >= 90500:
|
||||
time.sleep(1)
|
||||
try:
|
||||
pending_restart = (self._postgresql.query(
|
||||
pending_restart = self._postgresql.query(
|
||||
'SELECT COUNT(*) FROM pg_catalog.pg_settings'
|
||||
' WHERE pg_catalog.lower(name) != ALL(%s) AND pending_restart',
|
||||
[n.lower() for n in self._RECOVERY_PARAMETERS]).fetchone() or (0,))[0] > 0
|
||||
[n.lower() for n in self._RECOVERY_PARAMETERS])[0][0] > 0
|
||||
self._postgresql.set_pending_restart(pending_restart)
|
||||
except Exception as e:
|
||||
logger.warning('Exception %r when running query', e)
|
||||
@@ -1138,11 +1106,12 @@ class ConfigHandler(object):
|
||||
def set_synchronous_standby_names(self, value: Optional[str]) -> Optional[bool]:
|
||||
"""Updates synchronous_standby_names and reloads if necessary.
|
||||
:returns: True if value was updated."""
|
||||
if value != self._server_parameters.get('synchronous_standby_names'):
|
||||
if value != self._synchronous_standby_names:
|
||||
if value is None:
|
||||
self._server_parameters.pop('synchronous_standby_names', None)
|
||||
else:
|
||||
self._server_parameters['synchronous_standby_names'] = value
|
||||
self._synchronous_standby_names = value
|
||||
if self._postgresql.state == 'running':
|
||||
self.write_postgresql_conf()
|
||||
self._postgresql.reload()
|
||||
|
||||
@@ -2,45 +2,86 @@ import logging
|
||||
|
||||
from contextlib import contextmanager
|
||||
from threading import Lock
|
||||
from typing import Any, Dict, Iterator, Union, TYPE_CHECKING
|
||||
from typing import Any, Dict, Iterator, List, Union, Tuple, TYPE_CHECKING
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
from psycopg import Connection as Connection3, Cursor
|
||||
from psycopg2 import connection, cursor
|
||||
|
||||
from .. import psycopg
|
||||
from ..exceptions import PostgresConnectionException
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Connection(object):
|
||||
class Connection:
|
||||
"""Helper class to manage connections from Patroni to PostgreSQL.
|
||||
|
||||
:ivar server_version: PostgreSQL version in integer format where we are connected to.
|
||||
"""
|
||||
|
||||
server_version: int
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._lock = Lock()
|
||||
"""Create an instance of :class:`Connection` class."""
|
||||
self._lock = Lock() # used to make sure that only one connection to postgres is established
|
||||
self._connection = None
|
||||
self._cursor_holder = None
|
||||
|
||||
def set_conn_kwargs(self, conn_kwargs: Dict[str, Any]) -> None:
|
||||
"""Set connection parameters, like user, password, host, port and so on.
|
||||
|
||||
:param conn_kwargs: connection parameters as a dictionary.
|
||||
"""
|
||||
self._conn_kwargs = conn_kwargs
|
||||
|
||||
def get(self) -> Union['connection', 'Connection3[Any]']:
|
||||
"""Get ``psycopg``/``psycopg2`` connection object.
|
||||
|
||||
.. note::
|
||||
Opens a new connection if necessary.
|
||||
|
||||
:returns: ``psycopg`` or ``psycopg2`` connection object.
|
||||
"""
|
||||
with self._lock:
|
||||
if not self._connection or self._connection.closed != 0:
|
||||
logger.info("establishing a new patroni connection to postgres")
|
||||
self._connection = psycopg.connect(**self._conn_kwargs)
|
||||
self.server_version = getattr(self._connection, 'server_version', 0)
|
||||
return self._connection
|
||||
|
||||
def cursor(self) -> Union['cursor', 'Cursor[Any]']:
|
||||
if not self._cursor_holder or self._cursor_holder.closed or self._cursor_holder.connection.closed != 0:
|
||||
logger.info("establishing a new patroni connection to the postgres cluster")
|
||||
self._cursor_holder = self.get().cursor()
|
||||
return self._cursor_holder
|
||||
def query(self, sql: str, *params: Any) -> List[Tuple[Any, ...]]:
|
||||
"""Execute a query with parameters and optionally returns a response.
|
||||
|
||||
:param sql: SQL statement to execute.
|
||||
:param params: parameters to pass.
|
||||
|
||||
:returns: a query response as a list of tuples if there is any.
|
||||
:raises:
|
||||
:exc:`~psycopg.Error` if had issues while executing *sql*.
|
||||
|
||||
:exc:`~patroni.exceptions.PostgresConnectionException`: if had issues while connecting to the database.
|
||||
"""
|
||||
cursor = None
|
||||
try:
|
||||
with self.get().cursor() as cursor:
|
||||
cursor.execute(sql.encode('utf-8'), params or None)
|
||||
return cursor.fetchall() if cursor.rowcount and cursor.rowcount > 0 else []
|
||||
except psycopg.Error as exc:
|
||||
if cursor and cursor.connection.closed == 0:
|
||||
# When connected via unix socket, psycopg2 can't recoginze 'connection lost' and leaves
|
||||
# `self._connection.closed == 0`, but the generic exception is raised. It doesn't make
|
||||
# sense to continue with existing connection and we will close it, to avoid its reuse.
|
||||
if type(exc) in (psycopg.DatabaseError, psycopg.OperationalError):
|
||||
self.close()
|
||||
else:
|
||||
raise exc
|
||||
raise PostgresConnectionException('connection problems') from exc
|
||||
|
||||
def close(self) -> None:
|
||||
"""Close the psycopg connection to postgres."""
|
||||
if self._connection and self._connection.closed == 0:
|
||||
self._connection.close()
|
||||
logger.info("closed patroni connection to the postgresql cluster")
|
||||
self._cursor_holder = self._connection = None
|
||||
logger.info("closed patroni connection to postgres")
|
||||
self._connection = None
|
||||
|
||||
|
||||
@contextmanager
|
||||
|
||||
@@ -158,7 +158,7 @@ class Rewind(object):
|
||||
def _get_local_timeline_lsn(self) -> Tuple[Optional[bool], Optional[int], Optional[int]]:
|
||||
if self._postgresql.is_running(): # if postgres is running - get timeline from replication connection
|
||||
in_recovery = True
|
||||
timeline = self._postgresql.get_replica_timeline()
|
||||
timeline = self._postgresql.received_timeline() or self._postgresql.get_replica_timeline()
|
||||
lsn = self._postgresql.replayed_location()
|
||||
else: # otherwise analyze pg_controldata output
|
||||
in_recovery, timeline, lsn = self._get_local_timeline_lsn_from_controldata()
|
||||
@@ -280,7 +280,7 @@ class Rewind(object):
|
||||
"""After promote issue a CHECKPOINT from a new thread and asynchronously check the result.
|
||||
In case if CHECKPOINT failed, just check that timeline in pg_control was updated."""
|
||||
|
||||
if self._state != REWIND_STATUS.CHECKPOINT and self._postgresql.is_leader():
|
||||
if self._state != REWIND_STATUS.CHECKPOINT and self._postgresql.is_primary():
|
||||
with self._checkpoint_task_lock:
|
||||
if self._checkpoint_task:
|
||||
with self._checkpoint_task:
|
||||
|
||||
+13
-20
@@ -186,7 +186,7 @@ class SlotsHandler:
|
||||
self.pg_replslot_dir = os.path.join(self._postgresql.data_dir, 'pg_replslot')
|
||||
self.schedule()
|
||||
|
||||
def _query(self, sql: str, *params: Any) -> Union['cursor', 'Cursor[Any]']:
|
||||
def _query(self, sql: str, *params: Any) -> List[Tuple[Any, ...]]:
|
||||
"""Helper method for :meth:`Postgresql.query`.
|
||||
|
||||
:param sql: SQL statement to execute.
|
||||
@@ -263,9 +263,8 @@ class SlotsHandler:
|
||||
extra = ", catalog_xmin, pg_catalog.pg_wal_lsn_diff(confirmed_flush_lsn, '0/0')::bigint" \
|
||||
if self._postgresql.major_version >= 100000 else ""
|
||||
skip_temp_slots = ' WHERE NOT temporary' if self._postgresql.major_version >= 100000 else ''
|
||||
cursor = self._query(f'SELECT slot_name, slot_type, plugin, database, datoid'
|
||||
f'{extra} FROM pg_catalog.pg_replication_slots{skip_temp_slots}')
|
||||
for r in cursor:
|
||||
for r in self._query('SELECT slot_name, slot_type, plugin, database, datoid'
|
||||
f'{extra} FROM pg_catalog.pg_replication_slots{skip_temp_slots}'):
|
||||
value = {'type': r[1]}
|
||||
if r[1] == 'logical':
|
||||
value.update(plugin=r[2], database=r[3], datoid=r[4])
|
||||
@@ -308,16 +307,13 @@ class SlotsHandler:
|
||||
``dropped`` is ``True`` if the slot was successfully dropped. If the slot was not found return
|
||||
``False`` for both.
|
||||
"""
|
||||
cursor = self._query(('WITH slots AS (SELECT slot_name, active'
|
||||
' FROM pg_catalog.pg_replication_slots WHERE slot_name = %s),'
|
||||
' dropped AS (SELECT pg_catalog.pg_drop_replication_slot(slot_name),'
|
||||
' true AS dropped FROM slots WHERE not active) '
|
||||
'SELECT active, COALESCE(dropped, false) FROM slots'
|
||||
' FULL OUTER JOIN dropped ON true'), name)
|
||||
row = cursor.fetchone()
|
||||
if not row:
|
||||
row = (False, False)
|
||||
return row[0], row[1]
|
||||
rows = self._query(('WITH slots AS (SELECT slot_name, active'
|
||||
' FROM pg_catalog.pg_replication_slots WHERE slot_name = %s),'
|
||||
' dropped AS (SELECT pg_catalog.pg_drop_replication_slot(slot_name),'
|
||||
' true AS dropped FROM slots WHERE not active) '
|
||||
'SELECT active, COALESCE(dropped, false) FROM slots'
|
||||
' FULL OUTER JOIN dropped ON true'), name)
|
||||
return rows[0] if rows else (False, False)
|
||||
|
||||
def _drop_incorrect_slots(self, cluster: Cluster, slots: Dict[str, Any], paused: bool) -> None:
|
||||
"""Compare required slots and configured as permanent slots with those found, dropping extraneous ones.
|
||||
@@ -514,7 +510,7 @@ class SlotsHandler:
|
||||
|
||||
self._ensure_physical_slots(slots)
|
||||
|
||||
if self._postgresql.is_leader():
|
||||
if self._postgresql.is_primary():
|
||||
self._logical_slots_processing_queue.clear()
|
||||
self._ensure_logical_slots_primary(slots)
|
||||
elif cluster.slots and slots:
|
||||
@@ -601,11 +597,8 @@ class SlotsHandler:
|
||||
|
||||
# Replica isn't streaming or the hot_standby_feedback isn't enabled
|
||||
try:
|
||||
cur = self._query("SELECT pg_catalog.current_setting('hot_standby_feedback')::boolean")
|
||||
row = cur.fetchone()
|
||||
if row and not row[0]:
|
||||
logger.error('Logical slot failover requires "hot_standby_feedback".'
|
||||
' Please check postgresql.auto.conf')
|
||||
if not self._query("SELECT pg_catalog.current_setting('hot_standby_feedback')::boolean")[0][0]:
|
||||
logger.error('Logical slot failover requires "hot_standby_feedback". Please check postgresql.auto.conf')
|
||||
except Exception as e:
|
||||
logger.error('Failed to check the hot_standby_feedback setting: %r', e)
|
||||
return False
|
||||
|
||||
@@ -182,7 +182,7 @@ class _ReplicaList(List[_Replica]):
|
||||
swapping, but only if lag on this member is exceeding a threshold (``maximum_lag_on_syncnode``).
|
||||
|
||||
:ivar max_lsn: maximum value of ``_Replica.lsn`` among all values. In case if there is just one
|
||||
element in the list we take value of ``pg_current_wal_lsn()``.
|
||||
element in the list we take value of ``pg_current_wal_flush_lsn()``.
|
||||
"""
|
||||
|
||||
def __init__(self, postgresql: 'Postgresql', cluster: Cluster) -> None:
|
||||
@@ -339,7 +339,7 @@ END;$$""")
|
||||
sync_param = next(iter(sync), None)
|
||||
|
||||
if not (self._postgresql.config.set_synchronous_standby_names(sync_param)
|
||||
and self._postgresql.state == 'running' and self._postgresql.is_leader()) or has_asterisk:
|
||||
and self._postgresql.state == 'running' and self._postgresql.is_primary()) or has_asterisk:
|
||||
return
|
||||
|
||||
time.sleep(0.1) # Usualy it takes 1ms to reload postgresql.conf, but we will give it 100ms
|
||||
|
||||
@@ -9,6 +9,8 @@
|
||||
:var DBL_RE: regular expression to match double precision numbers, signed or unsigned. Matches scientific notation too.
|
||||
:var WHITESPACE_RE: regular expression to match whitespace characters
|
||||
"""
|
||||
import datetime
|
||||
import dateutil.parser
|
||||
import errno
|
||||
import logging
|
||||
import os
|
||||
@@ -16,9 +18,11 @@ import platform
|
||||
import random
|
||||
import re
|
||||
import socket
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
from enum import Enum
|
||||
from shlex import split
|
||||
|
||||
from typing import Any, Callable, Dict, Iterator, List, Optional, Union, Tuple, Type, TYPE_CHECKING
|
||||
@@ -470,6 +474,18 @@ def _sleep(interval: Union[int, float]) -> None:
|
||||
time.sleep(interval)
|
||||
|
||||
|
||||
def read_stripped(file_path: str) -> Iterator[str]:
|
||||
"""Iterate over stripped lines in the given file.
|
||||
|
||||
:param file_path: path to the file to read from
|
||||
|
||||
:yields: each line from the given file stripped
|
||||
"""
|
||||
with open(file_path) as f:
|
||||
for line in f:
|
||||
yield line.strip()
|
||||
|
||||
|
||||
class RetryFailedError(PatroniException):
|
||||
"""Maximum number of attempts exhausted in retry operation."""
|
||||
|
||||
@@ -1015,3 +1031,56 @@ def unquote(string: str) -> str:
|
||||
except ValueError:
|
||||
ret = string
|
||||
return ret
|
||||
|
||||
|
||||
def get_major_version(bin_dir: Optional[str] = None, bin_name: str = 'postgres') -> str:
|
||||
"""Get the major version of PostgreSQL.
|
||||
|
||||
It is based on the output of ``postgres --version``.
|
||||
|
||||
:param bin_dir: path to the PostgreSQL binaries directory. If ``None`` or an empty string, it will use the first
|
||||
*bin_name* binary that is found by the subprocess in the ``PATH``.
|
||||
:param bin_name: name of the postgres binary to call (``postgres`` by default)
|
||||
|
||||
:returns: the PostgreSQL major version.
|
||||
|
||||
:raises:
|
||||
:exc:`~patroni.exceptions.PatroniException`: if the postgres binary call failed due to :exc:`OSError`.
|
||||
|
||||
:Example:
|
||||
|
||||
* Returns `9.6` for PostgreSQL 9.6.24
|
||||
* Returns `15` for PostgreSQL 15.2
|
||||
"""
|
||||
if not bin_dir:
|
||||
binary = bin_name
|
||||
else:
|
||||
binary = os.path.join(bin_dir, bin_name)
|
||||
try:
|
||||
version = subprocess.check_output([binary, '--version']).decode()
|
||||
except OSError as e:
|
||||
raise PatroniException(f'Failed to get postgres version: {e}')
|
||||
version = re.match(r'^[^\s]+ [^\s]+ (\d+)(\.(\d+))?', version)
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
assert version is not None
|
||||
return '.'.join([version.group(1), version.group(3)]) if int(version.group(1)) < 10 else version.group(1)
|
||||
|
||||
|
||||
class ParseScheduleErrors(Enum):
|
||||
NO_TIMEZONE = ('Timezone information is mandatory for the scheduled {action}', 400)
|
||||
SCHEDULED_IN_PAST = ('Cannot schedule {action} in the past', 422)
|
||||
PARSING_ERROR = ('Unable to parse scheduled timestamp. It should be in an unambiguous format, e.g. ISO 8601', 422)
|
||||
|
||||
|
||||
def parse_schedule(schedule: Optional[str]) -> Tuple[Optional[ParseScheduleErrors], Optional[datetime.datetime]]:
|
||||
scheduled_at = None
|
||||
if schedule is not None:
|
||||
try:
|
||||
scheduled_at = dateutil.parser.parse(schedule)
|
||||
if scheduled_at.tzinfo is None:
|
||||
return ParseScheduleErrors.NO_TIMEZONE, scheduled_at
|
||||
elif scheduled_at < datetime.datetime.now(tzutc):
|
||||
return ParseScheduleErrors.SCHEDULED_IN_PAST, scheduled_at
|
||||
except (ValueError, TypeError):
|
||||
return ParseScheduleErrors.PARSING_ERROR, scheduled_at
|
||||
return None, scheduled_at
|
||||
|
||||
+4
-30
@@ -6,17 +6,16 @@ This module contains facilities for validating configuration of Patroni processe
|
||||
:var schema: configuration schema of the daemon launched by ``patroni`` command.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import socket
|
||||
import subprocess
|
||||
|
||||
from typing import Any, Dict, Union, Iterator, List, Optional as OptionalType, Tuple, TYPE_CHECKING
|
||||
from typing import Any, Dict, Union, Iterator, List, Optional as OptionalType, Tuple
|
||||
|
||||
from .collections import CaseInsensitiveSet
|
||||
|
||||
from .dcs import dcs_modules
|
||||
from .exceptions import ConfigParseError
|
||||
from .utils import parse_int, split_host_port, data_directory_is_empty
|
||||
from .utils import parse_int, split_host_port, data_directory_is_empty, get_major_version
|
||||
|
||||
|
||||
def data_directory_empty(data_dir: str) -> bool:
|
||||
@@ -204,31 +203,6 @@ def get_bin_name(bin_name: str) -> str:
|
||||
return (schema.data.get('postgresql', {}).get('bin_name', {}) or {}).get(bin_name, bin_name)
|
||||
|
||||
|
||||
def get_major_version(bin_dir: OptionalType[str] = None) -> str:
|
||||
"""Get the major version of PostgreSQL.
|
||||
|
||||
It is based on the output of ``postgres --version``.
|
||||
|
||||
:param bin_dir: path to PostgreSQL binaries directory. If ``None`` it will use the first ``postgres`` binary that
|
||||
is found by subprocess in the ``PATH``.
|
||||
:returns: the PostgreSQL major version.
|
||||
|
||||
:Example:
|
||||
|
||||
* Returns `9.6` for PostgreSQL 9.6.24
|
||||
* Returns `15` for PostgreSQL 15.2
|
||||
"""
|
||||
if not bin_dir:
|
||||
binary = get_bin_name('postgres')
|
||||
else:
|
||||
binary = os.path.join(bin_dir, get_bin_name('postgres'))
|
||||
version = subprocess.check_output([binary, '--version']).decode()
|
||||
version = re.match(r'^[^\s]+ [^\s]+ (\d+)(\.(\d+))?', version)
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
assert version is not None
|
||||
return '.'.join([version.group(1), version.group(3)]) if int(version.group(1)) < 10 else version.group(1)
|
||||
|
||||
|
||||
def validate_data_dir(data_dir: str) -> bool:
|
||||
"""Validate the value of ``postgresql.data_dir`` configuration option.
|
||||
|
||||
@@ -266,7 +240,7 @@ def validate_data_dir(data_dir: str) -> bool:
|
||||
raise ConfigParseError("data dir for the cluster is not empty, but doesn't contain"
|
||||
" \"{}\" directory".format(waldir))
|
||||
bin_dir = schema.data.get("postgresql", {}).get("bin_dir", None)
|
||||
major_version = get_major_version(bin_dir)
|
||||
major_version = get_major_version(bin_dir, get_bin_name('postgres'))
|
||||
if pgversion != major_version:
|
||||
raise ConfigParseError("data_dir directory postgresql version ({}) doesn't match with "
|
||||
"'postgres --version' output ({})".format(pgversion, major_version))
|
||||
|
||||
+1
-1
@@ -2,4 +2,4 @@
|
||||
|
||||
:var __version__: the current Patroni version.
|
||||
"""
|
||||
__version__ = '3.1.2'
|
||||
__version__ = '3.1.0'
|
||||
|
||||
+15
-12
@@ -106,7 +106,7 @@ class MockCursor(object):
|
||||
elif sql.startswith('SELECT slot_name'):
|
||||
self.results = [('blabla', 'physical'), ('foobar', 'physical'), ('ls', 'logical', 'a', 'b', 5, 100, 500)]
|
||||
elif sql.startswith('WITH slots AS (SELECT slot_name, active'):
|
||||
self.results = [(False, True)] if self.rowcount == 1 else [None]
|
||||
self.results = [(False, True)] if self.rowcount == 1 else []
|
||||
elif sql.startswith('SELECT CASE WHEN pg_catalog.pg_is_in_recovery()'):
|
||||
self.results = [(1, 2, 1, 0, False, 1, 1, None, None, 'streaming', '',
|
||||
[{"slot_name": "ls", "confirmed_flush_lsn": 12345}],
|
||||
@@ -114,11 +114,8 @@ class MockCursor(object):
|
||||
elif sql.startswith('SELECT pg_catalog.pg_is_in_recovery()'):
|
||||
self.results = [(False, 2)]
|
||||
elif sql.startswith('SELECT pg_catalog.pg_postmaster_start_time'):
|
||||
replication_info = '[{"application_name":"walreceiver","client_addr":"1.2.3.4",' +\
|
||||
'"state":"streaming","sync_state":"async","sync_priority":0}]'
|
||||
now = datetime.datetime.now(tzutc)
|
||||
self.results = [(now, 0, '', 0, '', False, now, 'streaming', None, replication_info)]
|
||||
elif sql.startswith('SELECT name, pg_catalog.current_setting(name) FROM pg_catalog.pg_settings'):
|
||||
self.results = [(datetime.datetime.now(tzutc),)]
|
||||
elif sql.startswith('SELECT name, current_setting(name) FROM pg_settings'):
|
||||
self.results = [('data_directory', 'data'),
|
||||
('hba_file', os.path.join('data', 'pg_hba.conf')),
|
||||
('ident_file', os.path.join('data', 'pg_ident.conf')),
|
||||
@@ -138,16 +135,13 @@ class MockCursor(object):
|
||||
('wal_block_size', '8192', None, 'integer', 'internal'),
|
||||
('shared_buffers', '16384', '8kB', 'integer', 'postmaster'),
|
||||
('wal_buffers', '-1', '8kB', 'integer', 'postmaster'),
|
||||
('max_connections', '100', None, 'integer', 'postmaster'),
|
||||
('max_prepared_transactions', '0', None, 'integer', 'postmaster'),
|
||||
('max_worker_processes', '8', None, 'integer', 'postmaster'),
|
||||
('max_locks_per_transaction', '64', None, 'integer', 'postmaster'),
|
||||
('max_wal_senders', '5', None, 'integer', 'postmaster'),
|
||||
('search_path', 'public', None, 'string', 'user'),
|
||||
('port', '5433', None, 'integer', 'postmaster'),
|
||||
('listen_addresses', '*', None, 'string', 'postmaster'),
|
||||
('autovacuum', 'on', None, 'bool', 'sighup'),
|
||||
('unix_socket_directories', '/tmp', None, 'string', 'postmaster')]
|
||||
elif sql.startswith('SELECT COUNT(*) FROM pg_catalog.pg_settings'):
|
||||
self.results = [(1,)]
|
||||
elif sql.startswith('IDENTIFY_SYSTEM'):
|
||||
self.results = [('1', 3, '0/402EEC0', '')]
|
||||
elif sql.startswith('TIMELINE_HISTORY '):
|
||||
@@ -161,6 +155,7 @@ class MockCursor(object):
|
||||
self.results = [(1, 0, 'host1', 5432, 'primary'), (2, 1, 'host2', 5432, 'primary')]
|
||||
else:
|
||||
self.results = [(None, None, None, None, None, None, None, None, None, None)]
|
||||
self.rowcount = len(self.results)
|
||||
|
||||
def fetchone(self):
|
||||
return self.results[0]
|
||||
@@ -179,11 +174,20 @@ class MockCursor(object):
|
||||
pass
|
||||
|
||||
|
||||
class MockConnectionInfo(object):
|
||||
|
||||
def parameter_status(self, param_name):
|
||||
if param_name == 'is_superuser':
|
||||
return 'on'
|
||||
return '0'
|
||||
|
||||
|
||||
class MockConnect(object):
|
||||
|
||||
server_version = 99999
|
||||
autocommit = False
|
||||
closed = 0
|
||||
info = MockConnectionInfo()
|
||||
|
||||
def cursor(self):
|
||||
return MockCursor(self)
|
||||
@@ -242,7 +246,6 @@ class PostgresInit(unittest.TestCase):
|
||||
|
||||
class BaseTestPostgresql(PostgresInit):
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
def setUp(self):
|
||||
super(BaseTestPostgresql, self).setUp()
|
||||
|
||||
|
||||
+157
-56
@@ -3,8 +3,6 @@ import json
|
||||
import unittest
|
||||
import socket
|
||||
|
||||
import patroni.psycopg as psycopg
|
||||
|
||||
from http.server import HTTPServer
|
||||
from io import BytesIO as IO
|
||||
from mock import Mock, PropertyMock, patch
|
||||
@@ -14,10 +12,10 @@ from patroni.api import RestApiHandler, RestApiServer
|
||||
from patroni.config import GlobalConfig
|
||||
from patroni.dcs import ClusterConfig, Member
|
||||
from patroni.ha import _MemberStatus
|
||||
from patroni.utils import tzutc
|
||||
from patroni.manual_failover import ManualFailoverPrecheckStatus
|
||||
from patroni.utils import ParseScheduleErrors, RetryFailedError, tzutc
|
||||
|
||||
from . import psycopg_connect, MockCursor
|
||||
from .test_ha import get_cluster_initialized_without_leader
|
||||
from .test_ha import get_cluster_initialized_without_leader, get_cluster_initialized_with_leader
|
||||
|
||||
|
||||
future_restart_time = datetime.datetime.now(tzutc) + datetime.timedelta(days=5)
|
||||
@@ -36,14 +34,11 @@ class MockPostgresql(object):
|
||||
pending_restart = True
|
||||
wal_name = 'wal'
|
||||
lsn_name = 'lsn'
|
||||
wal_flush = '_flush'
|
||||
POSTMASTER_START_TIME = 'pg_catalog.pg_postmaster_start_time()'
|
||||
TL_LSN = 'CASE WHEN pg_catalog.pg_is_in_recovery()'
|
||||
citus_handler = Mock()
|
||||
|
||||
@staticmethod
|
||||
def connection():
|
||||
return psycopg_connect()
|
||||
|
||||
@staticmethod
|
||||
def postmaster_start_time():
|
||||
return postmaster_start_time
|
||||
@@ -60,6 +55,12 @@ class MockPostgresql(object):
|
||||
def replication_state_from_parameters(*args):
|
||||
return 'streaming'
|
||||
|
||||
@staticmethod
|
||||
def query(sql, *params, retry=False):
|
||||
return [(postmaster_start_time, 0, '', 0, '', False, postmaster_start_time, 'streaming', None,
|
||||
'[{"application_name":"walreceiver","client_addr":"1.2.3.4",'
|
||||
+ '"state":"streaming","sync_state":"async","sync_priority":0}]')]
|
||||
|
||||
|
||||
class MockWatchdog(object):
|
||||
is_healthy = False
|
||||
@@ -122,6 +123,9 @@ class MockHa(object):
|
||||
def is_paused():
|
||||
return True
|
||||
|
||||
def has_members_eligible_to_promote(*args, **kwargs):
|
||||
return True
|
||||
|
||||
|
||||
class MockLogger(object):
|
||||
|
||||
@@ -487,9 +491,7 @@ class TestRestApiHandler(unittest.TestCase):
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
def test_RestApiServer_query(self):
|
||||
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)):
|
||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /patroni'))
|
||||
with patch.object(MockPostgresql, 'connection', Mock(side_effect=psycopg.OperationalError)):
|
||||
with patch.object(MockPostgresql, 'query', Mock(side_effect=RetryFailedError('bla'))):
|
||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /patroni'))
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
@@ -500,86 +502,185 @@ class TestRestApiHandler(unittest.TestCase):
|
||||
|
||||
post = 'POST /switchover HTTP/1.0' + self._authorization + '\nContent-Length: '
|
||||
|
||||
MockRestApiServer(RestApiHandler, post + '7\n\n{"1":2}')
|
||||
# Invalid content
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, post + '7\n\n{"1":2}')
|
||||
response_mock.assert_called_with(*ManualFailoverPrecheckStatus.SWITCHOVER_NO_LEADER.value[::-1])
|
||||
|
||||
# Empty content
|
||||
request = post + '0\n\n'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
|
||||
cluster.leader.name = 'postgresql1'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# [Switchover without a candidate]
|
||||
|
||||
cluster.leader.name = 'postgresql1'
|
||||
request = post + '25\n\n{"leader": "postgresql1"}'
|
||||
|
||||
with patch.object(GlobalConfig, 'is_paused', PropertyMock(return_value=True)):
|
||||
# No candidate in pause mode
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock, \
|
||||
patch.object(GlobalConfig, 'is_paused', PropertyMock(return_value=True)):
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(*ManualFailoverPrecheckStatus.SWITCHOVER_PAUSE_NO_CANDIDATE.value[::-1])
|
||||
|
||||
for is_synchronous_mode in (True, False):
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=is_synchronous_mode)):
|
||||
# No healthy nodes to promote in both sync and async mode
|
||||
for is_synchronous_mode, response in (
|
||||
(True, ManualFailoverPrecheckStatus.NO_SYNC_CANDIDATE.value[0].format(action='switchover')),
|
||||
(False, ManualFailoverPrecheckStatus.ONLY_LEADER.value[0].format(action='switchover'))):
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=is_synchronous_mode)), \
|
||||
patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(412, response)
|
||||
|
||||
cluster.leader.name = 'postgresql2'
|
||||
request = post + '53\n\n{"leader": "postgresql1", "candidate": "postgresql2"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# [Switchover to the candidate specified]
|
||||
|
||||
# Candidate to promote is the same as the leader specified
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
request = post + '53\n\n{"leader": "postgresql2", "candidate": "postgresql2"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(*ManualFailoverPrecheckStatus.SWITCHOVER_TO_LEADER.value[::-1])
|
||||
|
||||
# Current leader is different from the one specified
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
cluster.leader.name = 'postgresql2'
|
||||
request = post + '53\n\n{"leader": "postgresql1", "candidate": "postgresql2"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(
|
||||
ManualFailoverPrecheckStatus.LEADER_NOT_MEMBER.value[1],
|
||||
ManualFailoverPrecheckStatus.LEADER_NOT_MEMBER.value[0].format(leader='postgresql1',
|
||||
cluster_name='dummy'))
|
||||
|
||||
# Candidate to promote is not a sync standby/a member of the cluster
|
||||
cluster.leader.name = 'postgresql1'
|
||||
cluster.sync.matches.return_value = False
|
||||
for is_synchronous_mode in (True, False):
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=is_synchronous_mode)):
|
||||
for is_synchronous_mode, response in (
|
||||
(True, ManualFailoverPrecheckStatus.CANDIDATE_NOT_SYNC_STANDBY.value[0]),
|
||||
(False, ManualFailoverPrecheckStatus.CANDIDATE_NOT_MEMEBER.value[0].format(candidate="postgresql2",
|
||||
cluster_name='dummy'))):
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=is_synchronous_mode)), \
|
||||
patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(412, response)
|
||||
|
||||
cluster.members = [Member(0, 'postgresql0', 30, {'api_url': 'http'}),
|
||||
Member(0, 'postgresql2', 30, {'api_url': 'http'})]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
|
||||
cluster.failover = None
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# Cluster has no leader
|
||||
cluster.leader.name = None
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
request = post + '53\n\n{"leader": "postgresql1"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(
|
||||
ManualFailoverPrecheckStatus.CLUSTER_NO_LEADER.value[1],
|
||||
ManualFailoverPrecheckStatus.CLUSTER_NO_LEADER.value[0].format(leader='leader', cluster_name='dummy'))
|
||||
|
||||
dcs.get_cluster.side_effect = [cluster]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
cluster.leader.name = 'postgresql1'
|
||||
|
||||
cluster2 = cluster.copy()
|
||||
cluster2.leader.name = 'postgresql0'
|
||||
cluster2.is_unlocked.return_value = False
|
||||
dcs.get_cluster.side_effect = [cluster, cluster2]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# Failover key is empty in DCS
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
cluster.failover = None
|
||||
request = post + '53\n\n{"leader": "postgresql1", "candidate": "postgresql2"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(503, 'Switchover failed')
|
||||
|
||||
cluster2.leader.name = 'postgresql2'
|
||||
dcs.get_cluster.side_effect = [cluster, cluster2]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# Result polling failed
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
dcs.get_cluster.side_effect = [cluster]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(503, 'Switchover status unknown')
|
||||
|
||||
# Switchover to a node different from the candidate specified
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
cluster2 = cluster.copy()
|
||||
cluster2.leader.name = 'postgresql0'
|
||||
cluster2.is_unlocked.return_value = False
|
||||
dcs.get_cluster.side_effect = [cluster, cluster2]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(200, 'Switched over to "postgresql0" instead of "postgresql2"')
|
||||
|
||||
# Successful switchover to the candidate
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
cluster2.leader.name = 'postgresql2'
|
||||
dcs.get_cluster.side_effect = [cluster, cluster2]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(200, 'Successfully switched over to "postgresql2"')
|
||||
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
dcs.manual_failover.return_value = False
|
||||
dcs.get_cluster.side_effect = None
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(503, 'failed to write failover key into DCS')
|
||||
|
||||
dcs.get_cluster.side_effect = None
|
||||
dcs.manual_failover.return_value = False
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
dcs.manual_failover.return_value = True
|
||||
|
||||
with patch.object(MockHa, 'fetch_nodes_statuses', Mock(return_value=[])):
|
||||
# Candidate is not healthy to be promoted
|
||||
with patch.object(MockHa, 'has_members_eligible_to_promote', Mock(return_value=False)), \
|
||||
patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(
|
||||
ManualFailoverPrecheckStatus.NO_GOOD_CANDIDATES.value[1],
|
||||
ManualFailoverPrecheckStatus.NO_GOOD_CANDIDATES.value[0].format(action='switchover'))
|
||||
|
||||
# [Scheduled switchover]
|
||||
|
||||
# Valid future date
|
||||
request = post + '103\n\n{"leader": "postgresql1", "member": "postgresql2",' +\
|
||||
' "scheduled_at": "6016-02-15T18:13:30.568224+01:00"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
with patch.object(GlobalConfig, 'is_paused', PropertyMock(return_value=True)), \
|
||||
patch.object(MockPatroni, 'dcs') as d:
|
||||
d.manual_failover.return_value = False
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
request = post + '103\n\n{"leader": "postgresql1", "member": "postgresql2",' + \
|
||||
' "scheduled_at": "6016-02-15T18:13:30.568224+01:00"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(202, 'Switchover scheduled')
|
||||
|
||||
# Exception: No timezone specified
|
||||
request = post + '97\n\n{"leader": "postgresql1", "member": "postgresql2",' +\
|
||||
' "scheduled_at": "6016-02-15T18:13:30.568224"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# Scheduled in pause mode
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock, \
|
||||
patch.object(GlobalConfig, 'is_paused', PropertyMock(return_value=True)):
|
||||
dcs.manual_failover.return_value = False
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(*ManualFailoverPrecheckStatus.SCHEDULED_SWITCHOVER_PAUSE.value[::-1])
|
||||
|
||||
# No timezone specified
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
request = post + '97\n\n{"leader": "postgresql1", "member": "postgresql2",' + \
|
||||
' "scheduled_at": "6016-02-15T18:13:30.568224"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(
|
||||
ParseScheduleErrors.NO_TIMEZONE.value[1],
|
||||
ParseScheduleErrors.NO_TIMEZONE.value[0].format(action='switchover'))
|
||||
|
||||
# Exception: Scheduled in the past
|
||||
request = post + '103\n\n{"leader": "postgresql1", "member": "postgresql2", "scheduled_at": "'
|
||||
MockRestApiServer(RestApiHandler, request + '1016-02-15T18:13:30.568224+01:00"}')
|
||||
|
||||
# Scheduled in the past
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, request + '1016-02-15T18:13:30.568224+01:00"}')
|
||||
response_mock.assert_called_with(
|
||||
ParseScheduleErrors.SCHEDULED_IN_PAST.value[1],
|
||||
ParseScheduleErrors.SCHEDULED_IN_PAST.value[0].format(action='switchover'))
|
||||
|
||||
# Invalid date
|
||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, request + '2010-02-29T18:13:30.568224+01:00"}'))
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, request + '2010-02-29T18:13:30.568224+01:00"}')
|
||||
response_mock.assert_called_with(*ParseScheduleErrors.PARSING_ERROR.value[::-1])
|
||||
|
||||
def test_do_POST_failover(self):
|
||||
@patch.object(MockPatroni, 'dcs')
|
||||
def test_do_POST_failover(self, mock_dcs):
|
||||
post = 'POST /failover HTTP/1.0' + self._authorization + '\nContent-Length: '
|
||||
MockRestApiServer(RestApiHandler, post + '14\n\n{"leader":"1"}')
|
||||
MockRestApiServer(RestApiHandler, post + '37\n\n{"candidate":"2","scheduled_at": "1"}')
|
||||
cluster = mock_dcs.get_cluster.return_value
|
||||
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, post + '19\n\n{"leader":"leader"}')
|
||||
response_mock.assert_called_once_with(*ManualFailoverPrecheckStatus.FAILOVER_NO_CANDIDATE.value[::-1])
|
||||
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, post + '37\n\n{"candidate":"2","scheduled_at": "1"}')
|
||||
response_mock.assert_called_once_with(*ManualFailoverPrecheckStatus.SCHEDULED_FAILOVER.value[::-1])
|
||||
|
||||
# Candidate is not healthy to be promoted
|
||||
cluster.members = [Member(0, 'postgresql0', 30, {'api_url': 'http'}),
|
||||
Member(0, 'postgresql2', 30, {'api_url': 'http'})]
|
||||
with patch.object(MockHa, 'has_members_eligible_to_promote', Mock(return_value=False)), \
|
||||
patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, post + '27\n\n{"candidate":"postgresql2"}')
|
||||
response_mock.assert_called_with(
|
||||
ManualFailoverPrecheckStatus.NO_GOOD_CANDIDATES.value[1],
|
||||
ManualFailoverPrecheckStatus.NO_GOOD_CANDIDATES.value[0].format(action='failover'))
|
||||
|
||||
@patch.object(MockHa, 'is_leader', Mock(return_value=True))
|
||||
def test_do_POST_citus(self):
|
||||
|
||||
@@ -238,8 +238,7 @@ class TestBootstrap(BaseTestPostgresql):
|
||||
self.p.reload_config({'authentication': {'superuser': {'username': 'p', 'password': 'p'},
|
||||
'replication': {'username': 'r', 'password': 'r'},
|
||||
'rewind': {'username': 'rw', 'password': 'rw'}},
|
||||
'listen': '*', 'retry_timeout': 10,
|
||||
'parameters': {'wal_level': '', 'hba_file': 'foo', 'max_prepared_transactions': 10}})
|
||||
'listen': '*', 'retry_timeout': 10, 'parameters': {'wal_level': '', 'hba_file': 'foo'}})
|
||||
with patch.object(Postgresql, 'major_version', PropertyMock(return_value=110000)), \
|
||||
patch.object(Postgresql, 'restart', Mock()) as mock_restart:
|
||||
self.b.post_bootstrap({}, task)
|
||||
|
||||
+1
-25
@@ -3,7 +3,6 @@ import sys
|
||||
import unittest
|
||||
import io
|
||||
|
||||
from copy import deepcopy
|
||||
from mock import MagicMock, Mock, patch
|
||||
from patroni.config import Config, ConfigParseError
|
||||
|
||||
@@ -23,7 +22,7 @@ class TestConfig(unittest.TestCase):
|
||||
self.assertFalse(self.config.set_dynamic_configuration({'foo': 'bar'}))
|
||||
self.assertTrue(self.config.set_dynamic_configuration({'standby_cluster': {}, 'postgresql': {
|
||||
'parameters': {'cluster_name': 1, 'hot_standby': 1, 'wal_keep_size': 1,
|
||||
'track_commit_timestamp': 1, 'wal_level': 1, 'max_connections': '100'}}}))
|
||||
'track_commit_timestamp': 1, 'wal_level': 1}}}))
|
||||
|
||||
def test_reload_local_configuration(self):
|
||||
os.environ.update({
|
||||
@@ -150,26 +149,3 @@ class TestConfig(unittest.TestCase):
|
||||
@patch('os.path.isdir', Mock(return_value=False))
|
||||
def test_invalid_path(self):
|
||||
self.assertRaises(ConfigParseError, Config, 'postgres0')
|
||||
|
||||
def test__process_postgresql_parameters(self):
|
||||
expected_params = {
|
||||
'f.oo': 'bar', # not in ConfigHandler.CMDLINE_OPTIONS
|
||||
'max_connections': 100, # IntValidator
|
||||
'wal_keep_size': '128MB', # IntValidator
|
||||
'wal_level': 'hot_standby', # EnumValidator
|
||||
}
|
||||
input_params = deepcopy(expected_params)
|
||||
|
||||
input_params['max_connections'] = '100'
|
||||
self.assertEqual(self.config._process_postgresql_parameters(input_params), expected_params)
|
||||
|
||||
expected_params['f.oo'] = input_params['f.oo'] = '100'
|
||||
self.assertEqual(self.config._process_postgresql_parameters(input_params), expected_params)
|
||||
|
||||
input_params['wal_level'] = 'cold_standby'
|
||||
expected_params.pop('wal_level')
|
||||
self.assertEqual(self.config._process_postgresql_parameters(input_params), expected_params)
|
||||
|
||||
input_params['max_connections'] = 10
|
||||
expected_params.pop('max_connections')
|
||||
self.assertEqual(self.config._process_postgresql_parameters(input_params), expected_params)
|
||||
|
||||
@@ -0,0 +1,334 @@
|
||||
import os
|
||||
import psutil
|
||||
import socket
|
||||
import unittest
|
||||
|
||||
from . import MockConnect, MockCursor, MockConnectionInfo
|
||||
from copy import deepcopy
|
||||
from mock import MagicMock, Mock, PropertyMock, mock_open, patch
|
||||
|
||||
from patroni.__main__ import main as _main
|
||||
from patroni.config import Config
|
||||
from patroni.config_generator import AbstractConfigGenerator, get_address
|
||||
|
||||
from patroni.utils import patch_config
|
||||
|
||||
from . import psycopg_connect
|
||||
|
||||
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch('socket.getaddrinfo', Mock(return_value=[(0, 0, 0, 0, ('1.9.8.4', 1984))]))
|
||||
@patch('builtins.open', MagicMock())
|
||||
@patch('subprocess.check_output', Mock(return_value=b"postgres (PostgreSQL) 16.2"))
|
||||
@patch('psutil.Process.exe', Mock(return_value='/bin/dir/from/running/postgres'))
|
||||
@patch('psutil.Process.__init__', Mock(return_value=None))
|
||||
class TestGenerateConfig(unittest.TestCase):
|
||||
|
||||
no_value_msg = '#FIXME'
|
||||
_HOSTNAME = socket.gethostname()
|
||||
_IP = sorted(socket.getaddrinfo(_HOSTNAME, 0, socket.AF_UNSPEC, socket.SOCK_STREAM, 0), key=lambda x: x[0])[0][4][0]
|
||||
|
||||
def setUp(self):
|
||||
self.maxDiff = None
|
||||
|
||||
os.environ['PATRONI_SCOPE'] = 'scope_from_env'
|
||||
os.environ['PATRONI_POSTGRESQL_BIN_DIR'] = '/bin/from/env'
|
||||
os.environ['PATRONI_SUPERUSER_USERNAME'] = 'su_user_from_env'
|
||||
os.environ['PATRONI_SUPERUSER_PASSWORD'] = 'su_pwd_from_env'
|
||||
os.environ['PATRONI_REPLICATION_USERNAME'] = 'repl_user_from_env'
|
||||
os.environ['PATRONI_REPLICATION_PASSWORD'] = 'repl_pwd_from_env'
|
||||
os.environ['PATRONI_REWIND_USERNAME'] = 'rewind_user_from_env'
|
||||
os.environ['PGUSER'] = 'pguser_from_env'
|
||||
os.environ['PGPASSWORD'] = 'pguser_pwd_from_env'
|
||||
os.environ['PATRONI_RESTAPI_CONNECT_ADDRESS'] = 'localhost:8080'
|
||||
os.environ['PATRONI_RESTAPI_LISTEN'] = 'localhost:8080'
|
||||
os.environ['PATRONI_POSTGRESQL_BIN_POSTGRES'] = 'custom_postgres_bin_from_env'
|
||||
|
||||
self.environ = deepcopy(os.environ)
|
||||
|
||||
dynamic_config = Config.get_default_config()
|
||||
dynamic_config['postgresql']['parameters'] = dict(dynamic_config['postgresql']['parameters'])
|
||||
del dynamic_config['standby_cluster']
|
||||
dynamic_config['postgresql']['parameters']['wal_keep_segments'] = 8
|
||||
dynamic_config['postgresql']['use_pg_rewind'] = True
|
||||
|
||||
self.config = {
|
||||
'scope': self.environ['PATRONI_SCOPE'],
|
||||
'name': self._HOSTNAME,
|
||||
'bootstrap': {
|
||||
'dcs': dynamic_config
|
||||
},
|
||||
'postgresql': {
|
||||
'connect_address': self.no_value_msg + ':5432',
|
||||
'data_dir': self.no_value_msg,
|
||||
'listen': self.no_value_msg + ':5432',
|
||||
'pg_hba': ['host all all all md5',
|
||||
f'host replication {self.environ["PATRONI_REPLICATION_USERNAME"]} all md5'],
|
||||
'authentication': {'superuser': {'username': self.environ['PATRONI_SUPERUSER_USERNAME'],
|
||||
'password': self.environ['PATRONI_SUPERUSER_PASSWORD']},
|
||||
'replication': {'username': self.environ['PATRONI_REPLICATION_USERNAME'],
|
||||
'password': self.environ['PATRONI_REPLICATION_PASSWORD']},
|
||||
'rewind': {'username': self.environ['PATRONI_REWIND_USERNAME']}},
|
||||
'bin_dir': self.environ['PATRONI_POSTGRESQL_BIN_DIR'],
|
||||
'bin_name': {'postgres': self.environ['PATRONI_POSTGRESQL_BIN_POSTGRES']},
|
||||
'parameters': {'password_encryption': 'md5'}
|
||||
},
|
||||
'restapi': {
|
||||
'connect_address': self.environ['PATRONI_RESTAPI_CONNECT_ADDRESS'],
|
||||
'listen': self.environ['PATRONI_RESTAPI_LISTEN']
|
||||
}
|
||||
}
|
||||
|
||||
def _set_running_instance_config_vals(self):
|
||||
# values are taken from tests/__init__.py
|
||||
conf = {
|
||||
'scope': 'my_cluster',
|
||||
'bootstrap': {
|
||||
'dcs': {
|
||||
'postgresql': {
|
||||
'parameters': {
|
||||
'max_connections': 42,
|
||||
'max_locks_per_transaction': 73,
|
||||
'max_replication_slots': 21,
|
||||
'max_wal_senders': 37,
|
||||
'wal_level': 'replica',
|
||||
'wal_keep_segments': None
|
||||
},
|
||||
'use_pg_rewind': None
|
||||
}
|
||||
}
|
||||
},
|
||||
'postgresql': {
|
||||
'connect_address': f'{self._IP}:bar',
|
||||
'listen': '6.6.6.6:1984',
|
||||
'data_dir': 'data',
|
||||
'bin_dir': '/bin/dir/from/running',
|
||||
'parameters': {
|
||||
'archive_command': 'my archive command',
|
||||
'hba_file': os.path.join('data', 'pg_hba.conf'),
|
||||
'ident_file': os.path.join('data', 'pg_ident.conf'),
|
||||
'password_encryption': None
|
||||
},
|
||||
'authentication': {
|
||||
'superuser': {
|
||||
'username': 'foobar',
|
||||
'password': 'qwerty',
|
||||
'channel_binding': 'prefer',
|
||||
'gssencmode': 'prefer',
|
||||
'sslmode': 'prefer'
|
||||
},
|
||||
'replication': {
|
||||
'username': self.no_value_msg,
|
||||
'password': self.no_value_msg
|
||||
},
|
||||
'rewind': None
|
||||
},
|
||||
}
|
||||
}
|
||||
patch_config(self.config, conf)
|
||||
|
||||
def _get_running_instance_open_res(self):
|
||||
hba_content = '\n'.join(self.config['postgresql']['pg_hba'] + ['#host all all all md5',
|
||||
' host all all all md5',
|
||||
'',
|
||||
'hostall all all md5'])
|
||||
ident_content = '\n'.join(['# something very interesting', ' '])
|
||||
|
||||
self.config['postgresql']['pg_hba'] += ['host all all all md5']
|
||||
return [
|
||||
mock_open(read_data=hba_content)(),
|
||||
mock_open(read_data=ident_content)(),
|
||||
mock_open(read_data='1984')(),
|
||||
mock_open()()
|
||||
]
|
||||
|
||||
@patch('os.makedirs')
|
||||
@patch('yaml.safe_dump')
|
||||
def test_generate_sample_config_pre_13_dir_creation(self, mock_config_dump, mock_makedir):
|
||||
with patch('sys.argv', ['patroni.py', '--generate-sample-config', '/foo/bar.yml']), \
|
||||
patch('subprocess.check_output', Mock(return_value=b"postgres (PostgreSQL) 9.4.3")) as pg_bin_mock, \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertEqual(e.exception.code, 0)
|
||||
self.assertEqual(self.config, mock_config_dump.call_args[0][0])
|
||||
mock_makedir.assert_called_once()
|
||||
pg_bin_mock.assert_called_once_with([os.path.join(self.environ['PATRONI_POSTGRESQL_BIN_DIR'],
|
||||
self.environ['PATRONI_POSTGRESQL_BIN_POSTGRES']),
|
||||
'--version'])
|
||||
|
||||
@patch('os.makedirs', Mock())
|
||||
@patch('yaml.safe_dump')
|
||||
def test_generate_sample_config_16(self, mock_config_dump):
|
||||
conf = {
|
||||
'bootstrap': {
|
||||
'dcs': {
|
||||
'postgresql': {
|
||||
'parameters': {
|
||||
'wal_keep_size': '128MB',
|
||||
'wal_keep_segments': None
|
||||
},
|
||||
}
|
||||
}
|
||||
},
|
||||
'postgresql': {
|
||||
'parameters': {
|
||||
'password_encryption': 'scram-sha-256'
|
||||
},
|
||||
'pg_hba': ['host all all all scram-sha-256',
|
||||
f'host replication {self.environ["PATRONI_REPLICATION_USERNAME"]} all scram-sha-256'],
|
||||
'authentication': {
|
||||
'rewind': {
|
||||
'username': self.environ['PATRONI_REWIND_USERNAME'],
|
||||
'password': self.no_value_msg}
|
||||
},
|
||||
}
|
||||
}
|
||||
patch_config(self.config, conf)
|
||||
|
||||
with patch('sys.argv', ['patroni.py', '--generate-sample-config', '/foo/bar.yml']), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertEqual(e.exception.code, 0)
|
||||
self.assertEqual(self.config, mock_config_dump.call_args[0][0])
|
||||
|
||||
@patch('os.makedirs', Mock())
|
||||
@patch('yaml.safe_dump')
|
||||
def test_generate_config_running_instance_16(self, mock_config_dump):
|
||||
self._set_running_instance_config_vals()
|
||||
|
||||
with patch('builtins.open', Mock(side_effect=self._get_running_instance_open_res())), \
|
||||
patch('sys.argv', ['patroni.py', '--generate-config',
|
||||
'--dsn', 'host=foo port=bar user=foobar password=qwerty']), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertEqual(e.exception.code, 0)
|
||||
self.assertEqual(self.config, mock_config_dump.call_args[0][0])
|
||||
|
||||
@patch('os.makedirs', Mock())
|
||||
@patch('yaml.safe_dump')
|
||||
def test_generate_config_running_instance_16_connect_from_env(self, mock_config_dump):
|
||||
self._set_running_instance_config_vals()
|
||||
# su auth params and connect host from env
|
||||
os.environ['PGCHANNELBINDING'] = \
|
||||
self.config['postgresql']['authentication']['superuser']['channel_binding'] = 'disable'
|
||||
|
||||
conf = {
|
||||
'scope': 'my_cluster',
|
||||
'bootstrap': {
|
||||
'dcs': {
|
||||
'postgresql': {
|
||||
'parameters': {
|
||||
'max_connections': 42,
|
||||
'max_locks_per_transaction': 73,
|
||||
'max_replication_slots': 21,
|
||||
'max_wal_senders': 37,
|
||||
'wal_level': 'replica',
|
||||
'wal_keep_segments': None
|
||||
},
|
||||
'use_pg_rewind': None
|
||||
}
|
||||
}
|
||||
},
|
||||
'postgresql': {
|
||||
'connect_address': f'{self._IP}:1984',
|
||||
'authentication': {
|
||||
'superuser': {
|
||||
'username': self.environ['PGUSER'],
|
||||
'password': self.environ['PGPASSWORD'],
|
||||
'gssencmode': None,
|
||||
'sslmode': None
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
patch_config(self.config, conf)
|
||||
|
||||
with patch('builtins.open', Mock(side_effect=self._get_running_instance_open_res())), \
|
||||
patch('sys.argv', ['patroni.py', '--generate-config']), \
|
||||
patch.object(MockConnect, 'server_version', PropertyMock(return_value=160000)), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertEqual(e.exception.code, 0)
|
||||
self.assertEqual(self.config, mock_config_dump.call_args[0][0])
|
||||
|
||||
def test_generate_config_running_instance_errors(self):
|
||||
# 1. Wrong DSN format
|
||||
with patch('sys.argv', ['patroni.py', '--generate-config', '--dsn', 'host:foo port:bar user:foobar']), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertIn('Failed to parse DSN string', e.exception.code)
|
||||
|
||||
# 2. User is not a superuser
|
||||
with patch('sys.argv', ['patroni.py',
|
||||
'--generate-config', '--dsn', 'host=foo port=bar user=foobar password=pwd_from_dsn']), \
|
||||
patch.object(MockCursor, 'rowcount', PropertyMock(return_value=0), create=True), \
|
||||
patch.object(MockConnectionInfo, 'parameter_status', Mock(return_value='off')), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertIn('The provided user does not have superuser privilege', e.exception.code)
|
||||
|
||||
# 3. Error while calling postgres --version
|
||||
with patch('subprocess.check_output', Mock(side_effect=OSError)), \
|
||||
patch('sys.argv', ['patroni.py', '--generate-sample-config']), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertIn('Failed to get postgres version:', e.exception.code)
|
||||
|
||||
with patch('sys.argv', ['patroni.py', '--generate-config']):
|
||||
|
||||
# 4. empty postmaster.pid
|
||||
with patch('builtins.open', Mock(side_effect=[mock_open(read_data='hba_content')(),
|
||||
mock_open(read_data='ident_content')(),
|
||||
mock_open(read_data='')()])), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertIn('Failed to obtain postmaster pid from postmaster.pid file', e.exception.code)
|
||||
|
||||
# 5. Failed to open postmaster.pid
|
||||
with patch('builtins.open', Mock(side_effect=[mock_open(read_data='hba_content')(),
|
||||
mock_open(read_data='ident_content')(),
|
||||
OSError])), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertIn('Error while reading postmaster.pid file', e.exception.code)
|
||||
|
||||
# 6. Invalid postmaster pid
|
||||
with patch('builtins.open', Mock(side_effect=[mock_open(read_data='hba_content')(),
|
||||
mock_open(read_data='ident_content')(),
|
||||
mock_open(read_data='1984')()])), \
|
||||
patch('psutil.Process.__init__', Mock(return_value=None)), \
|
||||
patch('psutil.Process.exe', Mock(side_effect=psutil.NoSuchProcess(1984))), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertIn("Obtained postmaster pid doesn't exist", e.exception.code)
|
||||
|
||||
# 7. Failed to open pg_hba
|
||||
with patch('builtins.open', Mock(side_effect=OSError)), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertIn('Failed to read pg_hba.conf', e.exception.code)
|
||||
|
||||
# 8. Failed to open pg_ident
|
||||
with patch('builtins.open', Mock(side_effect=[mock_open(read_data='hba_content')(), OSError])), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertIn('Failed to read pg_ident.conf', e.exception.code)
|
||||
|
||||
# 9. Failed PG connecttion
|
||||
from . import psycopg
|
||||
with patch('patroni.psycopg.connect', side_effect=psycopg.Error), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertIn('Failed to establish PostgreSQL connection', e.exception.code)
|
||||
|
||||
# 10. An unexpected error
|
||||
with patch.object(AbstractConfigGenerator, '__init__', side_effect=psycopg.Error), \
|
||||
self.assertRaises(SystemExit) as e:
|
||||
_main()
|
||||
self.assertIn('Unexpected exception', e.exception.code)
|
||||
|
||||
def test_get_address(self):
|
||||
with patch('socket.getaddrinfo', Mock(side_effect=Exception)), \
|
||||
patch('logging.warning') as mock_warning:
|
||||
self.assertEqual(get_address(), (self.no_value_msg, self.no_value_msg))
|
||||
self.assertIn('Failed to obtain address: %r', mock_warning.call_args_list[0][0])
|
||||
@@ -197,9 +197,10 @@ class TestConsul(unittest.TestCase):
|
||||
|
||||
@patch.object(consul.Consul.KV, 'delete', Mock(return_value=True))
|
||||
def test_delete_leader(self):
|
||||
self.c.delete_leader()
|
||||
leader = self.c.get_cluster().leader
|
||||
self.c.delete_leader(leader)
|
||||
self.c._name = 'other'
|
||||
self.c.delete_leader()
|
||||
self.c.delete_leader(leader)
|
||||
|
||||
@patch.object(consul.Consul.KV, 'put', Mock(return_value=True))
|
||||
def test_initialize(self):
|
||||
|
||||
+235
-148
@@ -6,12 +6,14 @@ import unittest
|
||||
from click.testing import CliRunner
|
||||
from datetime import datetime, timedelta
|
||||
from mock import patch, Mock, PropertyMock
|
||||
from patroni.config import GlobalConfig
|
||||
from patroni.ctl import ctl, load_config, output_members, get_dcs, parse_dcs, \
|
||||
get_all_members, get_any_member, get_cursor, query_member, PatroniCtlException, apply_config_changes, \
|
||||
format_config_for_editing, show_diff, invoke_editor, format_pg_version, CONFIG_FILE_PATH, PatronictlPrettyTable
|
||||
from patroni.dcs.etcd import AbstractEtcdClientWithFailover, Cluster, Failover
|
||||
from patroni.manual_failover import ManualFailoverPrecheckStatus
|
||||
from patroni.psycopg import OperationalError
|
||||
from patroni.utils import tzutc
|
||||
from patroni.utils import ParseScheduleErrors, tzutc
|
||||
from prettytable import PrettyTable, ALL
|
||||
from urllib3 import PoolManager
|
||||
|
||||
@@ -28,6 +30,10 @@ from .test_ha import get_cluster_initialized_without_leader, get_cluster_initial
|
||||
class TestCtl(unittest.TestCase):
|
||||
TEST_ROLES = ('master', 'primary', 'leader')
|
||||
|
||||
SCHEDULED_TS = '2055-01-01T12:00:00+01:00'
|
||||
SCHEDULED_TS_NO_TZ = '2055-01-01T12:00:00'
|
||||
SCHEDULED_TS_INVALID = '2055-02-30T12:00:00'
|
||||
|
||||
@patch('socket.getaddrinfo', socket_getaddrinfo)
|
||||
@patch.object(AbstractEtcdClientWithFailover, '_get_machines_list', Mock(return_value=['http://remotehost:2379']))
|
||||
def setUp(self):
|
||||
@@ -68,21 +74,6 @@ class TestCtl(unittest.TestCase):
|
||||
|
||||
self.assertIsNotNone(get_cursor({}, get_cluster_initialized_with_leader(), None, {'dbname': 'foo'}, role='any'))
|
||||
|
||||
# Mutually exclusive options
|
||||
with self.assertRaises(PatroniCtlException) as e:
|
||||
get_cursor({}, get_cluster_initialized_with_leader(), None, {'dbname': 'foo'}, member_name='other',
|
||||
role='replica')
|
||||
|
||||
self.assertEqual(str(e.exception), '--role and --member are mutually exclusive options')
|
||||
|
||||
# Invalid member provided
|
||||
self.assertIsNone(get_cursor({}, get_cluster_initialized_with_leader(), None, {'dbname': 'foo'},
|
||||
member_name='invalid'))
|
||||
|
||||
# Valid member provided
|
||||
self.assertIsNotNone(get_cursor({}, get_cluster_initialized_with_leader(), None, {'dbname': 'foo'},
|
||||
member_name='other'))
|
||||
|
||||
def test_parse_dcs(self):
|
||||
assert parse_dcs(None) is None
|
||||
assert parse_dcs('localhost') == {'etcd': {'host': 'localhost:2379'}}
|
||||
@@ -111,91 +102,185 @@ class TestCtl(unittest.TestCase):
|
||||
mock_get_dcs.return_value = self.e
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
mock_get_dcs.return_value.set_failover_value = Mock()
|
||||
|
||||
# Confirm
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
assert 'leader' in result.output
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'],
|
||||
input='leader\nother\n2300-01-01T12:23:00\ny')
|
||||
assert result.exit_code == 0
|
||||
|
||||
with patch('patroni.config.GlobalConfig.is_paused', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', '2015-01-01T12:00:00'])
|
||||
assert result.exit_code == 1
|
||||
|
||||
# Aborting switchover, as we answer NO to the confirmation
|
||||
# Abort
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\nN')
|
||||
assert result.exit_code == 1
|
||||
|
||||
# Aborting scheduled switchover, as we answer NO to the confirmation
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--scheduled', '2015-01-01T12:00:00+01:00'], input='leader\nother\n\nN')
|
||||
assert result.exit_code == 1
|
||||
|
||||
# Target and source are equal
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nleader\n\ny')
|
||||
assert result.exit_code == 1
|
||||
|
||||
# Reality is not part of this cluster
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nReality\n\ny')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
|
||||
# Without a candidate with --force option
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0', '--force'])
|
||||
assert 'Member' in result.output
|
||||
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', '2015-01-01T12:00:00+01:00'])
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Invalid timestamp
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0', '--force', '--scheduled', 'invalid'])
|
||||
assert result.exit_code != 0
|
||||
|
||||
# Invalid timestamp
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', '2115-02-30T12:00:00+01:00'])
|
||||
assert result.exit_code != 0
|
||||
|
||||
# Specifying wrong leader
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='dummy')
|
||||
assert result.exit_code == 1
|
||||
|
||||
with patch.object(PoolManager, 'request', Mock(side_effect=Exception)):
|
||||
# Non-responding patroni
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'],
|
||||
input='leader\nother\n2300-01-01T12:23:00\ny')
|
||||
assert 'falling back to DCS' in result.output
|
||||
|
||||
with patch.object(PoolManager, 'request') as mocked:
|
||||
mocked.return_value.status = 500
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
assert 'Switchover failed' in result.output
|
||||
|
||||
mocked.return_value.status = 501
|
||||
mocked.return_value.data = b'Server does not support this operation'
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
assert 'Switchover failed' in result.output
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# No members available
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_only_leader
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('No candidates found to switchover to', result.output)
|
||||
|
||||
# No leader available
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_without_leader
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('This cluster has no leader', result.output)
|
||||
|
||||
# Citus cluster, no group number specified
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--force'], input='\n')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('For Citus clusters the --group must me specified', result.output)
|
||||
|
||||
# [Scheduled]
|
||||
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
|
||||
# Scheduled (confirm)
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'],
|
||||
input=f'leader\nother\n{self.SCHEDULED_TS}\ny')
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
self.assertIn(f'Are you sure you want to schedule switchover of cluster dummy '
|
||||
f'at {self.SCHEDULED_TS}, demoting current leader', result.output)
|
||||
|
||||
# Scheduled (abort)
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--scheduled', self.SCHEDULED_TS], input='leader\nother\n\nN')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
|
||||
# Scheduled with --force option
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', self.SCHEDULED_TS])
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# Scheduled in pause mode
|
||||
with patch('patroni.config.GlobalConfig.is_paused', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', self.SCHEDULED_TS])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.SCHEDULED_SWITCHOVER_PAUSE.value[0], result.output)
|
||||
|
||||
# Invalid timestamp with force
|
||||
result = self.runner.invoke(ctl,['switchover', 'dummy', '--group', '0', '--force', '--scheduled',
|
||||
self.SCHEDULED_TS_INVALID])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('Unable to parse scheduled timestamp', result.output)
|
||||
|
||||
# Invalid timestamp
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', self.SCHEDULED_TS_INVALID])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('Unable to parse scheduled timestamp', result.output)
|
||||
|
||||
# Invalid timestamp - no timezone
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', self.SCHEDULED_TS_NO_TZ])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ParseScheduleErrors.NO_TIMEZONE.value[0].format(action='switchover'), result.output)
|
||||
|
||||
# [Other erroneous combinations]
|
||||
|
||||
# No candidate in pause mode
|
||||
with patch('patroni.config.GlobalConfig.is_paused', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\n\n\ny')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.SWITCHOVER_PAUSE_NO_CANDIDATE.value[0], result.output)
|
||||
|
||||
# Target and source are equal
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nleader\n\ny')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.SWITCHOVER_TO_LEADER.value[0], result.output)
|
||||
|
||||
# Candidate is not a member of the cluster
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nReality\n\ny')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.CANDIDATE_NOT_MEMEBER.value[0].format(candidate='Reality',
|
||||
cluster_name='dummy'),
|
||||
result.output)
|
||||
|
||||
# Specifying wrong leader
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='dummy')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(
|
||||
ManualFailoverPrecheckStatus.LEADER_NOT_MEMBER.value[0].format(leader='dummy',
|
||||
cluster_name='dummy'),
|
||||
result.output)
|
||||
|
||||
mock_get_dcs.return_value.get_cluster = Mock(
|
||||
return_value=get_cluster_initialized_with_leader(sync=('leader', 'other')))
|
||||
|
||||
# Candidate is not a sync standby
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\notherMember\n\ny')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.CANDIDATE_NOT_SYNC_STANDBY.value[0], result.output)
|
||||
|
||||
# No healthy nodes to promote in sync mode
|
||||
mock_get_dcs.return_value.get_cluster = Mock(return_value=get_cluster_initialized_with_leader(sync=('leader')))
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0', '--force'])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.NO_SYNC_CANDIDATE.value[0].format(action='switchover'),
|
||||
result.output)
|
||||
|
||||
# No healthy nodes to promote in async mode
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_only_leader
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=False)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0', '--force'])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.ONLY_LEADER.value[0].format(action='switchover'),
|
||||
result.output)
|
||||
|
||||
# Cluster has no leader
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_without_leader
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0', '--leader', 'leader', '--force'])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(
|
||||
ManualFailoverPrecheckStatus.CLUSTER_NO_LEADER.value[0].format(leader='leader', cluster_name='dummy'),
|
||||
result.output)
|
||||
|
||||
# [Errors while sending Patroni REST API request]
|
||||
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
|
||||
with patch.object(PoolManager, 'request', Mock(side_effect=Exception)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'],
|
||||
input=f'leader\nother\n{self.SCHEDULED_TS}\ny')
|
||||
self.assertIn('falling back to DCS', result.output)
|
||||
|
||||
with patch.object(PoolManager, 'request') as mock_api_request:
|
||||
mock_api_request.return_value.status = 500
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
self.assertIn('Switchover failed', result.output)
|
||||
|
||||
mock_api_request.return_value.status = 501
|
||||
mock_api_request.return_value.data = b'Server does not support this operation'
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
self.assertIn('Switchover failed', result.output)
|
||||
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
@patch.object(PoolManager, 'request', Mock(return_value=MockResponse()))
|
||||
@patch('patroni.ctl.request_patroni', Mock(return_value=MockResponse()))
|
||||
def test_failover(self, mock_get_dcs):
|
||||
mock_get_dcs.return_value = self.e
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
mock_get_dcs.return_value.set_failover_value = Mock()
|
||||
result = self.runner.invoke(ctl, ['failover', 'dummy', '--force'], input='\n')
|
||||
assert 'For Citus clusters the --group must me specified' in result.output
|
||||
|
||||
# No candidate specified
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
result = self.runner.invoke(ctl, ['failover', 'dummy'], input='0\n')
|
||||
assert 'Failover could be performed only to a specific candidate' in result.output
|
||||
self.assertIn(ManualFailoverPrecheckStatus.FAILOVER_NO_CANDIDATE.value[0], result.output)
|
||||
|
||||
# Failover to an async member in sync mode (confirm)
|
||||
cluster = get_cluster_initialized_with_leader(sync=('leader', 'other'))
|
||||
cluster.members.append(Member(0, 'async', 28, {'api_url': 'http://127.0.0.1:8012/patroni'}))
|
||||
cluster.config.data['synchronous_mode'] = True
|
||||
mock_get_dcs.return_value.get_cluster = Mock(return_value=cluster)
|
||||
result = self.runner.invoke(ctl, ['failover', 'dummy', '--group', '0', '--candidate', 'async'], input='y\ny')
|
||||
self.assertIn('Are you sure you want to failover to the asynchronous node async', result.output)
|
||||
|
||||
# Failover to an async member in sync mode (abort)
|
||||
mock_get_dcs.return_value.get_cluster = Mock(return_value=cluster)
|
||||
result = self.runner.invoke(ctl, ['failover', 'dummy', '--group', '0', '--candidate', 'async'], input='N')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
|
||||
@patch('patroni.dcs.dcs_modules', Mock(return_value=['patroni.dcs.dummy', 'patroni.dcs.etcd']))
|
||||
def test_get_dcs(self):
|
||||
@@ -246,17 +331,11 @@ class TestCtl(unittest.TestCase):
|
||||
rows = query_member({}, None, None, None, None, 'replica', 'SELECT pg_catalog.pg_is_in_recovery()', {})
|
||||
|
||||
with patch('patroni.ctl.get_cursor', Mock(return_value=None)):
|
||||
# No role nor member given -- generic message
|
||||
rows = query_member({}, None, None, None, None, None, 'SELECT pg_catalog.pg_is_in_recovery()', {})
|
||||
self.assertTrue('No connection is available' in str(rows))
|
||||
self.assertTrue('No connection to' in str(rows))
|
||||
|
||||
# Member given -- message pointing to member
|
||||
rows = query_member({}, None, None, None, 'foo', None, 'SELECT pg_catalog.pg_is_in_recovery()', {})
|
||||
self.assertTrue('No connection to member foo' in str(rows))
|
||||
|
||||
# Role given -- message pointing to role
|
||||
rows = query_member({}, None, None, None, None, 'replica', 'SELECT pg_catalog.pg_is_in_recovery()', {})
|
||||
self.assertTrue('No connection to role replica' in str(rows))
|
||||
rows = query_member({}, None, None, None, 'foo', 'replica', 'SELECT pg_catalog.pg_is_in_recovery()', {})
|
||||
self.assertTrue('No connection to' in str(rows))
|
||||
|
||||
with patch('patroni.ctl.get_cursor', Mock(side_effect=OperationalError('bla'))):
|
||||
rows = query_member({}, None, None, None, None, 'replica', 'SELECT pg_catalog.pg_is_in_recovery()', {})
|
||||
@@ -294,12 +373,9 @@ class TestCtl(unittest.TestCase):
|
||||
|
||||
@patch.object(PoolManager, 'request')
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
def test_restart_reinit(self, mock_get_dcs, mock_post):
|
||||
def test_reinit(self, mock_get_dcs, mock_post):
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
mock_post.return_value.status = 503
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha'], input='now\ny\n')
|
||||
assert 'Failed: restart for' in result.output
|
||||
assert result.exit_code == 0
|
||||
|
||||
result = self.runner.invoke(ctl, ['reinit', 'alpha'], input='y')
|
||||
assert result.exit_code == 1
|
||||
@@ -308,67 +384,88 @@ class TestCtl(unittest.TestCase):
|
||||
result = self.runner.invoke(ctl, ['reinit', 'alpha', 'other'], input='y\ny')
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Aborted restart
|
||||
@patch.object(PoolManager, 'request')
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
def test_restart(self, mock_get_dcs, mock_post):
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
mock_post.return_value.status = 200
|
||||
|
||||
# Successful restart
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha'], input='now\ny\n')
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# Aborted
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha'], input='now\nN')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
|
||||
# With pending the flag
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--pending', '--force'])
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Aborted scheduled restart
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--scheduled', '2019-10-01T14:30'], input='N')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# Not a member
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', 'dummy', '--any'], input='now\ny')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('Not a single cluster member among provided members', result.output)
|
||||
|
||||
# Not a member with the specified role
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', 'other', '--role', 'primary'], input='now\ny')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('No primary among provided members', result.output)
|
||||
|
||||
# Wrong pg version
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--any', '--pg-version', '9.1'], input='now\ny')
|
||||
assert 'Error: Invalid PostgreSQL version format' in result.output
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('Error: Invalid PostgreSQL version format', result.output)
|
||||
|
||||
# Restart with timeout
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--pending', '--force', '--timeout', '10min'])
|
||||
assert result.exit_code == 0
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# normal restart, the schedule is actually parsed, but not validated in patronictl
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', 'other', '--force', '--scheduled', '2300-10-01T14:30'])
|
||||
assert 'Failed: flush scheduled restart' in result.output
|
||||
# Scheduled restart
|
||||
|
||||
# Aborted scheduled restart
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--scheduled', self.SCHEDULED_TS], input='N')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
|
||||
# Error parsing scheduled flag value (no tz)
|
||||
result = self.runner.invoke(ctl,
|
||||
['restart', 'alpha', 'other', '--force', '--scheduled', self.SCHEDULED_TS_NO_TZ])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ParseScheduleErrors.NO_TIMEZONE.value[0].format(action='restart'), result.output)
|
||||
|
||||
# Error parsing scheduled flag value (invalid date)
|
||||
result = self.runner.invoke(ctl,
|
||||
['restart', 'alpha', 'other', '--force', '--scheduled', self.SCHEDULED_TS_INVALID])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('Unable to parse scheduled timestamp', result.output)
|
||||
|
||||
# Successfully scheduled restart
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--scheduled', self.SCHEDULED_TS], input='Y')
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
self.assertIn('Success: restart on member other', result.output)
|
||||
|
||||
# Not possible to schedule in pause mode
|
||||
with patch('patroni.config.GlobalConfig.is_paused', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl,
|
||||
['restart', 'alpha', 'other', '--force', '--scheduled', '2300-10-01T14:30'])
|
||||
assert result.exit_code == 1
|
||||
['restart', 'alpha', 'other', '--force', '--scheduled', self.SCHEDULED_TS])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn("Can't schedule restart in the paused state", result.output)
|
||||
|
||||
# force restart with restart already present
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', 'other', '--force', '--scheduled', '2300-10-01T14:30'])
|
||||
assert result.exit_code == 0
|
||||
|
||||
ctl_args = ['restart', 'alpha', '--pg-version', '99.0', '--scheduled', '2300-10-01T14:30']
|
||||
# normal restart, the schedule is actually parsed, but not validated in patronictl
|
||||
mock_post.return_value.status = 200
|
||||
result = self.runner.invoke(ctl, ctl_args, input='y')
|
||||
assert result.exit_code == 0
|
||||
# Force restart with restart already scheduled
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', 'other', '--force', '--scheduled', self.SCHEDULED_TS])
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# get restart with the non-200 return code
|
||||
# normal restart, the schedule is actually parsed, but not validated in patronictl
|
||||
mock_post.return_value.status = 204
|
||||
result = self.runner.invoke(ctl, ctl_args, input='y')
|
||||
assert result.exit_code == 0
|
||||
|
||||
# get restart with the non-200 return code
|
||||
# normal restart, the schedule is actually parsed, but not validated in patronictl
|
||||
mock_post.return_value.status = 202
|
||||
result = self.runner.invoke(ctl, ctl_args, input='y')
|
||||
assert 'Success: restart scheduled' in result.output
|
||||
assert result.exit_code == 0
|
||||
|
||||
# get restart with the non-200 return code
|
||||
# normal restart, the schedule is actually parsed, but not validated in patronictl
|
||||
mock_post.return_value.status = 409
|
||||
result = self.runner.invoke(ctl, ctl_args, input='y')
|
||||
assert 'Failed: another restart is already' in result.output
|
||||
assert result.exit_code == 0
|
||||
ctl_args = ['restart', 'alpha', '--pg-version', '99.0', '--scheduled', self.SCHEDULED_TS]
|
||||
for code, output in [
|
||||
(204, 'Failed: restart for member other, status code=204'),
|
||||
(202, 'Success: restart scheduled'),
|
||||
(409, 'Failed: another restart is already')
|
||||
]:
|
||||
mock_post.return_value.status = code
|
||||
result = self.runner.invoke(ctl, ctl_args, input='y')
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
self.assertIn(output, result.output)
|
||||
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
def test_remove(self, mock_get_dcs):
|
||||
@@ -423,19 +520,9 @@ class TestCtl(unittest.TestCase):
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
def test_members(self, mock_get_dcs):
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
|
||||
result = self.runner.invoke(ctl, ['list'])
|
||||
assert '127.0.0.1' in result.output
|
||||
assert result.exit_code == 0
|
||||
assert 'Citus cluster: alpha -' in result.output
|
||||
|
||||
result = self.runner.invoke(ctl, ['list', '--group', '0'])
|
||||
assert 'Citus cluster: alpha (group: 0, 12345678901) -' in result.output
|
||||
|
||||
with patch('patroni.ctl.load_config', Mock(return_value={'scope': 'alpha'})):
|
||||
result = self.runner.invoke(ctl, ['list'])
|
||||
assert 'Cluster: alpha (12345678901) -' in result.output
|
||||
|
||||
with patch('patroni.ctl.load_config', Mock(return_value={})):
|
||||
self.runner.invoke(ctl, ['list'])
|
||||
|
||||
|
||||
+1
-1
@@ -313,7 +313,7 @@ class TestEtcd(unittest.TestCase):
|
||||
self.assertFalse(self.etcd.cancel_initialization())
|
||||
|
||||
def test_delete_leader(self):
|
||||
self.assertFalse(self.etcd.delete_leader())
|
||||
self.assertFalse(self.etcd.delete_leader(self.etcd.get_cluster().leader))
|
||||
|
||||
def test_delete_cluster(self):
|
||||
self.assertFalse(self.etcd.delete_cluster())
|
||||
|
||||
+4
-3
@@ -298,9 +298,10 @@ class TestEtcd3(BaseTestEtcd3):
|
||||
self.etcd3.cancel_initialization()
|
||||
|
||||
def test_delete_leader(self):
|
||||
self.etcd3.delete_leader()
|
||||
leader = self.etcd3.get_cluster().leader
|
||||
self.etcd3.delete_leader(leader)
|
||||
self.etcd3._name = 'other'
|
||||
self.etcd3.delete_leader()
|
||||
self.etcd3.delete_leader(leader)
|
||||
|
||||
def test_delete_cluster(self):
|
||||
self.etcd3.delete_cluster()
|
||||
@@ -312,7 +313,7 @@ class TestEtcd3(BaseTestEtcd3):
|
||||
self.etcd3.set_sync_state_value('', 1)
|
||||
|
||||
def test_delete_sync_state(self):
|
||||
self.etcd3.delete_sync_state()
|
||||
self.etcd3.delete_sync_state('1')
|
||||
|
||||
def test_watch(self):
|
||||
self.etcd3.set_ttl(10)
|
||||
|
||||
+312
-159
@@ -162,7 +162,7 @@ def run_async(self, func, args=()):
|
||||
|
||||
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=MockPostmaster()))
|
||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'is_primary', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'timeline_wal_position', Mock(return_value=(1, 10, 1)))
|
||||
@patch.object(Postgresql, '_cluster_info_state_get', Mock(return_value=10))
|
||||
@patch.object(Postgresql, 'data_directory_empty', Mock(return_value=False))
|
||||
@@ -224,7 +224,7 @@ class TestHa(PostgresInit):
|
||||
@patch.object(Postgresql, 'received_timeline', Mock(return_value=None))
|
||||
def test_touch_member(self):
|
||||
self.p._major_version = 110000
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.timeline_wal_position = Mock(return_value=(0, 1, 0))
|
||||
self.p.replica_cached_timeline = Mock(side_effect=Exception)
|
||||
with patch.object(Postgresql, '_cluster_info_state_get', Mock(return_value='streaming')):
|
||||
@@ -320,7 +320,7 @@ class TestHa(PostgresInit):
|
||||
@patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True))
|
||||
@patch.object(Rewind, 'can_rewind', PropertyMock(return_value=True))
|
||||
def test_crash_recovery_before_rewind(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.is_running = false
|
||||
self.p.controldata = lambda: {'Database cluster state': 'in archive recovery',
|
||||
'Database system identifier': SYSID}
|
||||
@@ -365,7 +365,7 @@ class TestHa(PostgresInit):
|
||||
|
||||
@patch.object(Cluster, 'is_unlocked', Mock(return_value=False))
|
||||
def test_start_as_readonly(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.is_healthy = true
|
||||
self.ha.has_lock = true
|
||||
self.p.controldata = lambda: {'Database cluster state': 'in production', 'Database system identifier': SYSID}
|
||||
@@ -383,11 +383,11 @@ class TestHa(PostgresInit):
|
||||
|
||||
def test_promoted_by_acquiring_lock(self):
|
||||
self.ha.is_healthiest_node = true
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
|
||||
def test_promotion_cancelled_after_pre_promote_failed(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p._pre_promote = false
|
||||
self.ha._is_healthiest_node = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
@@ -402,7 +402,7 @@ class TestHa(PostgresInit):
|
||||
@patch.object(Cluster, 'is_unlocked', Mock(return_value=False))
|
||||
def test_long_promote(self):
|
||||
self.ha.has_lock = true
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.set_role('primary')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
|
||||
@@ -413,7 +413,7 @@ class TestHa(PostgresInit):
|
||||
def test_follow_new_leader_after_failing_to_obtain_lock(self):
|
||||
self.ha.is_healthiest_node = true
|
||||
self.ha.acquire_lock = false
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'following new leader after trying and failing to obtain lock')
|
||||
|
||||
def test_demote_because_not_healthiest(self):
|
||||
@@ -422,21 +422,21 @@ class TestHa(PostgresInit):
|
||||
|
||||
def test_follow_new_leader_because_not_healthiest(self):
|
||||
self.ha.is_healthiest_node = false
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
@patch.object(Cluster, 'is_unlocked', Mock(return_value=False))
|
||||
def test_promote_because_have_lock(self):
|
||||
self.ha.has_lock = true
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader because I had the session lock')
|
||||
|
||||
def test_promote_without_watchdog(self):
|
||||
self.ha.has_lock = true
|
||||
self.p.is_leader = true
|
||||
self.p.is_primary = true
|
||||
with patch.object(Watchdog, 'activate', Mock(return_value=False)):
|
||||
self.assertEqual(self.ha.run_cycle(), 'Demoting self because watchdog could not be activated')
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'Not promoting self because watchdog could not be activated')
|
||||
|
||||
def test_leader_with_lock(self):
|
||||
@@ -462,12 +462,12 @@ class TestHa(PostgresInit):
|
||||
self.assertEqual(self.ha.run_cycle(), 'demoted self because failed to update leader lock in DCS')
|
||||
with patch.object(Ha, '_get_node_to_follow', Mock(side_effect=DCSError('foo'))):
|
||||
self.assertEqual(self.ha.run_cycle(), 'demoted self because failed to update leader lock in DCS')
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'not promoting because failed to update leader lock in DCS')
|
||||
|
||||
@patch.object(Cluster, 'is_unlocked', Mock(return_value=False))
|
||||
def test_follow(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), a secondary, and following a leader ()')
|
||||
self.ha.patroni.replicatefrom = "foo"
|
||||
self.p.config.check_recovery_conf = Mock(return_value=(True, False))
|
||||
@@ -484,13 +484,13 @@ class TestHa(PostgresInit):
|
||||
def test_follow_in_pause(self):
|
||||
self.ha.is_paused = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: continue to run as primary without lock')
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: no action. I am (postgresql0)')
|
||||
|
||||
@patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True))
|
||||
@patch.object(Rewind, 'can_rewind', PropertyMock(return_value=True))
|
||||
def test_follow_triggers_rewind(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.ha._rewind.trigger_check_diverged_lsn()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.assertEqual(self.ha.run_cycle(), 'running pg_rewind from leader')
|
||||
@@ -544,7 +544,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.global_config = self.ha.patroni.config.get_global_config(self.ha.cluster)
|
||||
self.ha.update_failsafe({'name': 'leader', 'api_url': 'http://127.0.0.1:8008/patroni',
|
||||
'conn_url': 'postgres://127.0.0.1:5432/postgres', 'slots': {'foo': 1000}})
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'DCS is not accessible')
|
||||
|
||||
def test_no_dcs_connection_replica_failsafe_not_enabled_but_active(self):
|
||||
@@ -552,7 +552,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.ha.update_failsafe({'name': 'leader', 'api_url': 'http://127.0.0.1:8008/patroni',
|
||||
'conn_url': 'postgres://127.0.0.1:5432/postgres', 'slots': {'foo': 1000}})
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'DCS is not accessible')
|
||||
|
||||
def test_update_failsafe(self):
|
||||
@@ -591,9 +591,9 @@ class TestHa(PostgresInit):
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.e.initialize = true
|
||||
self.assertEqual(self.ha.bootstrap(), 'trying to bootstrap a new cluster')
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'waiting for end of recovery after bootstrap')
|
||||
self.p.is_leader = true
|
||||
self.p.is_primary = true
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'running post_bootstrap')
|
||||
self.assertEqual(self.ha.run_cycle(), 'initialized a new cluster')
|
||||
@@ -613,7 +613,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.e.initialize = true
|
||||
self.ha.bootstrap()
|
||||
self.p.is_leader = true
|
||||
self.p.is_primary = true
|
||||
with patch.object(Watchdog, 'activate', Mock(return_value=False)), \
|
||||
patch('patroni.ha.logger.error') as mock_logger:
|
||||
self.assertEqual(self.ha.post_bootstrap(), 'running post_bootstrap')
|
||||
@@ -687,112 +687,289 @@ class TestHa(PostgresInit):
|
||||
|
||||
@patch('patroni.postgresql.citus.CitusHandler.is_coordinator', Mock(return_value=False))
|
||||
def test_manual_failover_from_leader(self):
|
||||
self.ha.has_lock = true # I am the leader
|
||||
|
||||
# to me
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', self.p.name, None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
mock_warning.assert_called_with('%s: I am already the leader, no need to %s', 'manual failover', 'failover')
|
||||
|
||||
# to a non-existent candidate
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'blabla', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
mock_warning.assert_called_with(
|
||||
'%s: no healthy members found, %s is not possible', 'manual failover', 'failover')
|
||||
|
||||
# to an existent candidate
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.ha.has_lock = true
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', self.p.name, None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'blabla', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
f = Failover(0, self.p.name, '', None)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(f)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'b', None))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||
|
||||
# to a candidate on an older timeline
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(timeline=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0],
|
||||
('Timeline %s of member %s is behind the cluster timeline %s', 1, 'b', 2))
|
||||
|
||||
# to a lagging candidate
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||
self.ha.cluster.config.data.update({'maximum_lag_on_failover': 5})
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0],
|
||||
('Member %s exceeds maximum replication lag', 'b'))
|
||||
self.ha.cluster.members.pop()
|
||||
|
||||
@patch('patroni.postgresql.citus.CitusHandler.is_coordinator', Mock(return_value=False))
|
||||
def test_manual_switchover_from_leader(self):
|
||||
self.ha.has_lock = true # I am the leader
|
||||
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
|
||||
# different leader specified in failover key, no candidate
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
mock_warning.assert_called_with(
|
||||
'%s: leader name does not match: %s != %s', 'switchover', 'blabla', 'postgresql0')
|
||||
|
||||
# no candidate
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'switchover: demoting myself')
|
||||
|
||||
self.ha._rewind.rewind_or_reinitialize_needed_and_possible = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(watchdog_failed=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(timeline=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||
self.ha.cluster.config.data.update({'maximum_lag_on_failover': 5})
|
||||
self.ha.global_config = self.ha.patroni.config.get_global_config(self.ha.cluster)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
# manual failover from the previous leader to us won't happen if we hold the nofailover flag
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'switchover: demoting myself')
|
||||
|
||||
# Failover scheduled time must include timezone
|
||||
scheduled = datetime.datetime.now()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.ha.run_cycle()
|
||||
# other members with failover_limitation_s
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0], ('Member %s is %s', 'leader', 'not allowed to promote'))
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(watchdog_failed=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0], ('Member %s is %s', 'leader', 'not watchdog capable'))
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(timeline=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0],
|
||||
('Timeline %s of member %s is behind the cluster timeline %s', 1, 'leader', 2))
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||
self.ha.cluster.config.data.update({'maximum_lag_on_failover': 5})
|
||||
self.ha.global_config = self.ha.patroni.config.get_global_config(self.ha.cluster)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0], ('Member %s exceeds maximum replication lag', 'leader'))
|
||||
|
||||
@patch('patroni.postgresql.citus.CitusHandler.is_coordinator', Mock(return_value=False))
|
||||
def test_scheduled_switchover_from_leader(self):
|
||||
self.ha.has_lock = true # I am the leader
|
||||
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
|
||||
# switchover scheduled time must include timezone
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
scheduled = datetime.datetime.now()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'blabla', scheduled))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertIn('Incorrect value of scheduled_at: %s', mock_warning.call_args_list[0][0])
|
||||
|
||||
# scheduled now
|
||||
scheduled = datetime.datetime.utcnow().replace(tzinfo=tzutc)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'b', scheduled))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual('switchover: demoting myself', self.ha.run_cycle())
|
||||
|
||||
scheduled = scheduled + datetime.timedelta(seconds=30)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
# scheduled in the future
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
scheduled = scheduled + datetime.timedelta(seconds=30)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'blabla', scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.assertIn('Awaiting %s at %s (in %.0f seconds)', mock_info.call_args_list[0][0])
|
||||
|
||||
scheduled = scheduled + datetime.timedelta(seconds=-600)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
# stale value
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
scheduled = scheduled + datetime.timedelta(seconds=-600)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'b', scheduled))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.assertIn('Found a stale %s value, cleaning up: %s', mock_warning.call_args_list[0][0])
|
||||
|
||||
scheduled = None
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
def test_manual_switchover_from_leader_in_pause(self):
|
||||
self.ha.has_lock = true # I am the leader
|
||||
self.ha.is_paused = true
|
||||
|
||||
# no candidate
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, '', None))
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.assertEqual('PAUSE: no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
mock_warning.assert_called_with(
|
||||
'%s is possible only to a specific candidate in a paused state', 'Switchover')
|
||||
|
||||
def test_manual_failover_from_leader_in_pause(self):
|
||||
self.ha.has_lock = true
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.ha.is_paused = true
|
||||
scheduled = datetime.datetime.now()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('PAUSE: no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, '', None))
|
||||
self.assertEqual('PAUSE: no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
|
||||
# failover from me, candidate is healthy
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, None, 'b', None))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual('PAUSE: manual failover: demoting myself', self.ha.run_cycle())
|
||||
self.ha.cluster.members.pop()
|
||||
|
||||
def test_manual_failover_from_leader_in_synchronous_mode(self):
|
||||
self.p.is_leader = true
|
||||
self.ha.has_lock = true
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.ha.is_failover_possible = false
|
||||
self.ha.process_sync_replication = Mock()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, None))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, 'a'))
|
||||
self.ha.is_failover_possible = true
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
|
||||
# I am the leader
|
||||
self.p.is_primary = true
|
||||
self.ha.has_lock = true
|
||||
|
||||
# the candidate is not in sync members but we allow failover to an async candidate
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, None, 'b', None), sync=(self.p.name, 'a'))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual('manual failover: demoting myself', self.ha.run_cycle())
|
||||
self.ha.cluster.members.pop()
|
||||
|
||||
def test_manual_switchover_from_leader_in_synchronous_mode(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.ha.process_sync_replication = Mock()
|
||||
|
||||
# I am the leader
|
||||
self.p.is_primary = true
|
||||
self.ha.has_lock = true
|
||||
|
||||
# candidate specified is not in sync members
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None),
|
||||
sync=(self.p.name, 'blabla'))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual(mock_warning.call_args_list[0][0],
|
||||
('%s candidate=%s does not match with sync_standbys=%s', 'Switchover', 'a', 'blabla'))
|
||||
|
||||
# the candidate is in sync members and is healthy
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=305419896)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None),
|
||||
sync=(self.p.name, 'a'))
|
||||
self.ha.cluster.members.append(Member(0, 'a', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual('switchover: demoting myself', self.ha.run_cycle())
|
||||
|
||||
# the candidate is in sync members but is not healthy
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=true)
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual(mock_info.call_args_list[0][0], ('Member %s is %s', 'a', 'not allowed to promote'))
|
||||
|
||||
def test_manual_failover_process_no_leader(self):
|
||||
self.p.is_leader = false
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', self.p.name, None))
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'leader', None))
|
||||
self.p.is_primary = false
|
||||
self.p.set_role('replica')
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
self.ha.fetch_node_status = get_node_status() # accessible, in_recovery
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, self.p.name, '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
self.ha.fetch_node_status = get_node_status(reachable=False) # inaccessible, in_recovery
|
||||
|
||||
# failover to another member, fetch_node_status for candidate fails
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'leader', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
self.assertEqual(mock_warning.call_args_list[1][0],
|
||||
('%s: member %s is %s', 'manual failover', 'leader', 'not reachable'))
|
||||
|
||||
# failover to another member, candidate is accessible, in_recovery
|
||||
self.p.set_role('replica')
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
# set failover flag to True for all members of the cluster
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# set nofailover flag to True for all members of the cluster
|
||||
# this should elect the current member, as we are not going to call the API for it.
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None))
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True) # accessible, in_recovery
|
||||
self.p.set_role('replica')
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
# same as previous, but set the current member to nofailover. In no case it should be elected as a leader
|
||||
|
||||
# failover to me but I am set to nofailover. In no case I should be elected as a leader
|
||||
self.p.set_role('replica')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None))
|
||||
self.ha.patroni.nofailover = True
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
||||
|
||||
self.ha.patroni.nofailover = False
|
||||
|
||||
# failover to another member that is on an older timeline (only failover_limitation() is checked)
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'b', None))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.ha.fetch_node_status = get_node_status(timeline=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
mock_info.assert_called_with('%s: to %s, i am %s', 'manual failover', 'b', 'postgresql0')
|
||||
|
||||
# failover to another member lagging behind the cluster_lsn (only failover_limitation() is checked)
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.cluster.config.data.update({'maximum_lag_on_failover': 5})
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
mock_info.assert_called_with('%s: to %s, i am %s', 'manual failover', 'b', 'postgresql0')
|
||||
|
||||
def test_manual_switchover_process_no_leader(self):
|
||||
self.p.is_primary = false
|
||||
self.p.set_role('replica')
|
||||
|
||||
# I was the leader, other members are healthy
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, self.p.name, '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# I was the leader, I am the only healthy member
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(reachable=False) # inaccessible, in_recovery
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0], ('Member %s is %s', 'leader', 'not reachable'))
|
||||
self.assertEqual(mock_info.call_args_list[1][0], ('Member %s is %s', 'other', 'not reachable'))
|
||||
|
||||
def test_manual_failover_process_no_leader_in_synchronous_mode(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True) # other nodes are not healthy
|
||||
|
||||
# switchover to a specific node, which name doesn't match our name (postgresql0)
|
||||
# manual failover when our name (postgresql0) isn't in the /sync key and the candidate node is not available
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
|
||||
sync=('leader1', 'blabla'))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# manual failover when the candidate node isn't available but our name is in the /sync key
|
||||
# while other sync node is nofailover
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
|
||||
sync=('leader1', 'postgresql0'))
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet(), CaseInsensitiveSet()))
|
||||
self.ha.dcs.write_sync_state = Mock(return_value=SyncState.empty())
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
self.assertEqual(mock_warning.call_args_list[0][0],
|
||||
('%s: member %s is %s', 'manual failover', 'other', 'not allowed to promote'))
|
||||
|
||||
# manual failover to our node (postgresql0),
|
||||
# which name is not in sync nodes list (some sync nodes are available)
|
||||
self.p.set_role('replica')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None),
|
||||
sync=('leader1', 'other'))
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet(['leader1']),
|
||||
CaseInsensitiveSet(['leader1'])))
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
|
||||
def test_manual_switchover_process_no_leader_in_synchronous_mode(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.p.is_primary = false
|
||||
|
||||
# to a specific node, which name doesn't match our name (postgresql0)
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'other', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# switchover to our node (postgresql0), which name is not in sync nodes list
|
||||
# to our node (postgresql0), which name is not in sync nodes list
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'postgresql0', None),
|
||||
sync=('leader1', 'blabla'))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# switchover from a specific leader, but our name (postgresql0) is not in the sync nodes list
|
||||
# without candidate, our name (postgresql0) is not in the sync nodes list
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', '', None),
|
||||
sync=('leader', 'blabla'))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
@@ -802,53 +979,39 @@ class TestHa(PostgresInit):
|
||||
sync=('postgresql0'))
|
||||
self.ha.patroni.nofailover = True
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
||||
self.ha.patroni.nofailover = False
|
||||
|
||||
# manual failover when our name (postgresql0) isn't in the /sync key and the `other` node is not available
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True) # accessible, in_recovery
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
|
||||
sync=('leader1', 'blabla'))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# manual failover when the `other` node isn't available but our name is in the /sync key
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
|
||||
sync=('leader1', 'postgresql0'))
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet(), CaseInsensitiveSet()))
|
||||
self.ha.dcs.write_sync_state = Mock(return_value=SyncState.empty())
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
|
||||
# manual failover to our node (postgresql0),
|
||||
# which name is not in sync nodes list (the leader and all sync nodes are not available)
|
||||
self.p.set_role('replica')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None),
|
||||
sync=('leader1', 'other'))
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
|
||||
# manual failover to our node (postgresql0),
|
||||
# which name is not in sync nodes list (some sync nodes are available)
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None),
|
||||
sync=('leader1', 'other'))
|
||||
self.p.set_role('replica')
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet(['leader1']),
|
||||
CaseInsensitiveSet(['leader1'])))
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
|
||||
def test_manual_failover_process_no_leader_in_pause(self):
|
||||
self.ha.is_paused = true
|
||||
|
||||
# I am running as primary, cluster is unlocked, the candidate is allowed to promote
|
||||
# but we are in pause
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: continue to run as primary without lock')
|
||||
|
||||
def test_manual_switchover_process_no_leader_in_pause(self):
|
||||
self.ha.is_paused = true
|
||||
|
||||
# I am running as primary, cluster is unlocked, no candidate specified
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: continue to run as primary without lock')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'blabla', None))
|
||||
self.assertEqual('PAUSE: acquired session lock as a leader', self.ha.run_cycle())
|
||||
self.p.is_leader = false
|
||||
|
||||
# the candidate is not running
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'blabla', None))
|
||||
self.assertEqual('PAUSE: acquired session lock as a leader', self.ha.run_cycle())
|
||||
self.assertEqual(
|
||||
mock_warning.call_args_list[0][0],
|
||||
('%s: removing failover key because failover candidate is not running', 'switchover'))
|
||||
|
||||
# switchover to me, I am not leader
|
||||
self.p.is_primary = false
|
||||
self.p.set_role('replica')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', self.p.name, None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: promoted self to leader by acquiring session lock')
|
||||
|
||||
def test_is_healthiest_node(self):
|
||||
self.ha.is_failsafe_mode = true
|
||||
self.ha.state_handler.is_leader = false
|
||||
self.ha.state_handler.is_primary = false
|
||||
self.ha.patroni.nofailover = False
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.ha.dcs._last_failsafe = {'foo': ''}
|
||||
@@ -862,7 +1025,7 @@ class TestHa(PostgresInit):
|
||||
self.assertFalse(self.ha.is_healthiest_node())
|
||||
|
||||
def test__is_healthiest_node(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(sync=('postgresql1', self.p.name))
|
||||
self.ha.global_config = self.ha.patroni.config.get_global_config(self.ha.cluster)
|
||||
self.assertTrue(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
||||
@@ -961,7 +1124,7 @@ class TestHa(PostgresInit):
|
||||
self.assertTrue(self.ha.restart_matches("replica", "9.5.2", False))
|
||||
|
||||
def test_process_healthy_cluster_in_pause(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.ha.is_paused = true
|
||||
self.p.name = 'leader'
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
@@ -972,7 +1135,7 @@ class TestHa(PostgresInit):
|
||||
@patch('patroni.postgresql.mtime', Mock(return_value=1588316884))
|
||||
@patch('builtins.open', mock_open(read_data='1\t0/40159C0\tno recovery target specified\n'))
|
||||
def test_process_healthy_standby_cluster_as_standby_leader(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.name = 'leader'
|
||||
self.ha.cluster = get_standby_cluster_initialized_with_only_leader()
|
||||
self.p.config.check_recovery_conf = Mock(return_value=(False, False))
|
||||
@@ -984,7 +1147,7 @@ class TestHa(PostgresInit):
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to a standby leader because i had the session lock')
|
||||
|
||||
def test_process_healthy_standby_cluster_as_cascade_replica(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.name = 'replica'
|
||||
self.ha.cluster = get_standby_cluster_initialized_with_only_leader()
|
||||
self.assertEqual(self.ha.run_cycle(),
|
||||
@@ -994,7 +1157,7 @@ class TestHa(PostgresInit):
|
||||
|
||||
@patch.object(Cluster, 'is_unlocked', Mock(return_value=True))
|
||||
def test_process_unhealthy_standby_cluster_as_standby_leader(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.name = 'leader'
|
||||
self.ha.cluster = get_standby_cluster_initialized_with_only_leader()
|
||||
self.ha.sysid_valid = true
|
||||
@@ -1004,13 +1167,13 @@ class TestHa(PostgresInit):
|
||||
@patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True))
|
||||
@patch.object(Rewind, 'can_rewind', PropertyMock(return_value=True))
|
||||
def test_process_unhealthy_standby_cluster_as_cascade_replica(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.name = 'replica'
|
||||
self.ha.cluster = get_standby_cluster_initialized_with_only_leader()
|
||||
self.assertTrue(self.ha.run_cycle().startswith('running pg_rewind from remote_member:'))
|
||||
|
||||
def test_recover_unhealthy_leader_in_standby_cluster(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.name = 'leader'
|
||||
self.p.is_running = false
|
||||
self.p.follow = false
|
||||
@@ -1019,7 +1182,7 @@ class TestHa(PostgresInit):
|
||||
|
||||
@patch.object(Cluster, 'is_unlocked', Mock(return_value=True))
|
||||
def test_recover_unhealthy_unlocked_standby_cluster(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.name = 'leader'
|
||||
self.p.is_running = false
|
||||
self.p.follow = false
|
||||
@@ -1079,7 +1242,7 @@ class TestHa(PostgresInit):
|
||||
check_calls([(update_lock, True), (demote, True)])
|
||||
|
||||
self.ha.has_lock = false
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertEqual(self.ha.run_cycle(),
|
||||
'no action. I am (postgresql0), a secondary, and following a leader (leader)')
|
||||
check_calls([(update_lock, False), (demote, False)])
|
||||
@@ -1090,7 +1253,7 @@ class TestHa(PostgresInit):
|
||||
f = Failover(0, self.p.name, '', None)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(f)
|
||||
self.ha.fetch_node_status = get_node_status() # accessible, in_recovery
|
||||
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||
self.assertEqual(self.ha.run_cycle(), 'switchover: demoting myself')
|
||||
|
||||
@patch('patroni.ha.Ha.demote')
|
||||
def test_failover_immediately_on_zero_primary_start_timeout(self, demote):
|
||||
@@ -1213,7 +1376,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.is_synchronous_mode = true
|
||||
|
||||
mock_set_sync = self.p.sync_handler.set_synchronous_standby_names = Mock()
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.set_role('replica')
|
||||
self.ha.has_lock = true
|
||||
mock_write_sync = self.ha.dcs.write_sync_state = Mock(return_value=SyncState.empty())
|
||||
@@ -1236,7 +1399,7 @@ class TestHa(PostgresInit):
|
||||
def test_unhealthy_sync_mode(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.set_role('replica')
|
||||
self.p.name = 'other'
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(sync=('leader', 'other2'))
|
||||
@@ -1267,7 +1430,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.is_synchronous_mode = true
|
||||
|
||||
self.p.name = 'other'
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.p.set_role('replica')
|
||||
mock_restart = self.p.restart = Mock(return_value=True)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(sync=('leader', 'other'))
|
||||
@@ -1298,34 +1461,14 @@ class TestHa(PostgresInit):
|
||||
mock_restart.assert_called_once()
|
||||
self.ha.dcs.get_cluster.assert_not_called()
|
||||
|
||||
@patch.object(Cluster, 'is_unlocked', Mock(return_value=False))
|
||||
def test_enable_synchronous_mode(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.ha.has_lock = true
|
||||
self.p.name = 'leader'
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet(), CaseInsensitiveSet()))
|
||||
self.ha.dcs.write_sync_state = Mock(return_value=SyncState.empty())
|
||||
with patch('patroni.ha.logger.info') as mock_logger:
|
||||
self.ha.run_cycle()
|
||||
self.assertEqual(mock_logger.call_args_list[0][0][0], 'Enabled synchronous replication')
|
||||
self.ha.dcs.write_sync_state = Mock(return_value=None)
|
||||
with patch('patroni.ha.logger.warning') as mock_logger:
|
||||
self.ha.run_cycle()
|
||||
self.assertEqual(mock_logger.call_args[0][0], 'Updating sync state failed')
|
||||
|
||||
@patch.object(Cluster, 'is_unlocked', Mock(return_value=False))
|
||||
def test_inconsistent_synchronous_state(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.ha.has_lock = true
|
||||
self.p.name = 'leader'
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(sync=('leader', 'a'))
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet('a'), CaseInsensitiveSet()))
|
||||
self.ha.dcs.write_sync_state = Mock(return_value=SyncState.empty())
|
||||
mock_set_sync = self.p.sync_handler.set_synchronous_standby_names = Mock()
|
||||
with patch('patroni.ha.logger.warning') as mock_logger:
|
||||
self.ha.run_cycle()
|
||||
mock_set_sync.assert_called_once()
|
||||
self.assertTrue(mock_logger.call_args_list[0][0][0].startswith('Inconsistent state between '))
|
||||
self.assertEqual(mock_logger.call_args[0][0], 'Enabled synchronous replication')
|
||||
self.ha.dcs.write_sync_state = Mock(return_value=None)
|
||||
with patch('patroni.ha.logger.warning') as mock_logger:
|
||||
self.ha.run_cycle()
|
||||
@@ -1403,7 +1546,7 @@ class TestHa(PostgresInit):
|
||||
@patch('sys.exit', return_value=1)
|
||||
def test_abort_join(self, exit_mock):
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.ha.run_cycle()
|
||||
exit_mock.assert_called_once_with(1)
|
||||
|
||||
@@ -1463,7 +1606,7 @@ class TestHa(PostgresInit):
|
||||
@patch.object(SlotsHandler, 'sync_replication_slots', Mock(return_value=['ls']))
|
||||
def test_follow_copy(self):
|
||||
self.ha.cluster.config.data['slots'] = {'ls': {'database': 'a', 'plugin': 'b'}}
|
||||
self.p.is_leader = false
|
||||
self.p.is_primary = false
|
||||
self.assertTrue(self.ha.run_cycle().startswith('Copying logical slots'))
|
||||
|
||||
def test_acquire_lock(self):
|
||||
@@ -1483,3 +1626,13 @@ class TestHa(PostgresInit):
|
||||
self.assertEqual(self.ha.patroni.request.call_args[1]['timeout'], 2)
|
||||
mock_logger.assert_called()
|
||||
self.assertTrue(mock_logger.call_args[0][0].startswith('Request to Citus coordinator'))
|
||||
|
||||
def test_has_members_eligible_to_promote(self):
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
members = [
|
||||
Member(0, 'test', 1, {'api_url': 'http://127.0.0.1:8011/patroni', 'conn_url': 'postgres://127.0.0.1:5432/postgres'}),
|
||||
Member(0, 'test2', 1, {'api_url': 'http://127.0.0.1:8011/patroni', 'conn_url': 'postgres://127.0.0.1:5432/postgres'}),
|
||||
]
|
||||
with patch('patroni.ha.logger.info') as mock_logger:
|
||||
self.assertTrue(self.ha.has_members_eligible_to_promote(members, fast_path=True))
|
||||
mock_logger.assert_not_called()
|
||||
|
||||
@@ -308,15 +308,13 @@ class TestKubernetesConfigMaps(BaseTestKubernetes):
|
||||
mock_patch_namespaced_pod.assert_called()
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['isMaster'], 'false')
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['tmp_role'], 'replica')
|
||||
mock_patch_namespaced_pod.rest_mock()
|
||||
|
||||
self.k._name = 'p-0'
|
||||
self.k.touch_member({'role': 'standby_leader'})
|
||||
self.k.touch_member({'state': 'running', 'role': 'standby-leader'})
|
||||
mock_patch_namespaced_pod.assert_called()
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['isMaster'], 'false')
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['tmp_role'], 'master')
|
||||
mock_patch_namespaced_pod.rest_mock()
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['tmp_role'], 'standby-leader')
|
||||
|
||||
self.k._name = 'p-0'
|
||||
self.k.touch_member({'role': 'primary'})
|
||||
mock_patch_namespaced_pod.assert_called()
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['isMaster'], 'true')
|
||||
@@ -326,7 +324,7 @@ class TestKubernetesConfigMaps(BaseTestKubernetes):
|
||||
self.k.initialize()
|
||||
|
||||
def test_delete_leader(self):
|
||||
self.k.delete_leader(1)
|
||||
self.k.delete_leader(self.k.get_cluster().leader, 1)
|
||||
|
||||
def test_cancel_initialization(self):
|
||||
self.k.cancel_initialization()
|
||||
@@ -436,10 +434,6 @@ class TestKubernetesEndpoints(BaseTestKubernetes):
|
||||
mock_logger_exception.assert_called_once()
|
||||
self.assertEqual(('create_config_service failed',), mock_logger_exception.call_args[0])
|
||||
|
||||
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_endpoints', mock_namespaced_kind, create=True)
|
||||
def test_write_leader_optime(self):
|
||||
self.k.write_leader_optime(12345)
|
||||
|
||||
|
||||
def mock_watch(*args):
|
||||
return urllib3.HTTPResponse()
|
||||
|
||||
@@ -40,7 +40,7 @@ class MockFrozenImporter(object):
|
||||
@patch('time.sleep', Mock())
|
||||
@patch('subprocess.call', Mock(return_value=0))
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch('urllib3.connection.HTTPConnection.connect', Mock(side_effect=Exception))
|
||||
@patch('urllib3.PoolManager.request', Mock(side_effect=Exception))
|
||||
@patch.object(ConfigHandler, 'append_pg_hba', Mock())
|
||||
@patch.object(ConfigHandler, 'write_postgresql_conf', Mock())
|
||||
@patch.object(ConfigHandler, 'write_recovery_conf', Mock())
|
||||
@@ -64,7 +64,7 @@ class TestPatroni(unittest.TestCase):
|
||||
self.assertRaises(SystemExit, _main)
|
||||
|
||||
@patch('pkgutil.iter_importers', Mock(return_value=[MockFrozenImporter()]))
|
||||
@patch('urllib3.connection.HTTPConnection.connect', Mock(side_effect=Exception))
|
||||
@patch('urllib3.PoolManager.request', Mock(side_effect=Exception))
|
||||
@patch('sys.frozen', Mock(return_value=True), create=True)
|
||||
@patch.object(HTTPServer, '__init__', Mock())
|
||||
@patch.object(etcd.Client, 'read', etcd_read)
|
||||
@@ -108,7 +108,6 @@ class TestPatroni(unittest.TestCase):
|
||||
@patch('os.getpid')
|
||||
@patch('multiprocessing.Process')
|
||||
@patch('patroni.__main__.patroni_main', Mock())
|
||||
@patch('sys.argv', ['patroni.py', 'postgres0.yml'])
|
||||
def test_patroni_main(self, mock_process, mock_getpid):
|
||||
mock_getpid.return_value = 2
|
||||
_main()
|
||||
@@ -234,8 +233,8 @@ class TestPatroni(unittest.TestCase):
|
||||
)
|
||||
with patch('patroni.dcs.AbstractDCS.get_cluster', Mock(return_value=bad_cluster)):
|
||||
# If the api of the running node cannot be reached, this implies unique name
|
||||
with patch('urllib3.connection.HTTPConnection.connect', Mock(side_effect=ConnectionError)):
|
||||
with patch.object(self.p, 'request', Mock(side_effect=ConnectionError)):
|
||||
self.assertIsNone(self.p.ensure_unique_name())
|
||||
# Only if the api of the running node is reachable do we throw an error
|
||||
with patch('urllib3.connection.HTTPConnection.connect', Mock()):
|
||||
with patch.object(self.p, 'request', Mock()):
|
||||
self.assertRaises(SystemExit, self.p.ensure_unique_name)
|
||||
|
||||
+21
-24
@@ -310,17 +310,6 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
self.p.config.write_postgresql_conf()
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (False, False))
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (False, False))
|
||||
|
||||
# Config files changed, but can't connect to postgres
|
||||
mock_get_pg_settings.side_effect = PostgresConnectionException('')
|
||||
with patch('patroni.postgresql.config.mtime', mock_mtime):
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
||||
|
||||
# Config files didn't change, but postgres crashed or in crash recovery
|
||||
with patch.object(MockPostmaster, 'create_time', Mock(return_value=1234568), create=True):
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (False, False))
|
||||
|
||||
# Any other exception raised when executing the query
|
||||
mock_get_pg_settings.side_effect = Exception
|
||||
with patch('patroni.postgresql.config.mtime', mock_mtime):
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
||||
@@ -374,11 +363,11 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
self.assertRaises(psycopg.ProgrammingError, self.p.query, 'blabla')
|
||||
|
||||
@patch.object(Postgresql, 'pg_isready', Mock(return_value=STATE_REJECT))
|
||||
def test_is_leader(self):
|
||||
self.assertTrue(self.p.is_leader())
|
||||
def test_is_primary(self):
|
||||
self.assertTrue(self.p.is_primary())
|
||||
self.p.reset_cluster_info_state(None)
|
||||
with patch.object(Postgresql, '_query', Mock(side_effect=RetryFailedError(''))):
|
||||
self.assertFalse(self.p.is_leader())
|
||||
self.assertFalse(self.p.is_primary())
|
||||
|
||||
@patch.object(Postgresql, 'controldata', Mock(return_value={'Database cluster state': 'shut down',
|
||||
'Latest checkpoint location': '0/1ADBC18',
|
||||
@@ -472,7 +461,7 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
self.assertIsNone(self.p.call_nowait(CallbackAction.ON_START))
|
||||
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=MockPostmaster()))
|
||||
def test_is_leader_exception(self):
|
||||
def test_is_primary_exception(self):
|
||||
self.p.start()
|
||||
self.p.query = Mock(side_effect=psycopg.OperationalError("not supported"))
|
||||
self.assertTrue(self.p.stop())
|
||||
@@ -556,9 +545,7 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||
@patch.object(MockCursor, 'fetchone')
|
||||
def test_reload_config(self, mock_fetchone):
|
||||
mock_fetchone.return_value = (1,)
|
||||
def test_reload_config(self):
|
||||
parameters = self._PARAMETERS.copy()
|
||||
parameters.pop('f.oo')
|
||||
parameters['wal_buffers'] = '512'
|
||||
@@ -566,9 +553,14 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
'authentication': {},
|
||||
'retry_timeout': 10, 'listen': '*', 'krbsrvname': 'postgres', 'parameters': parameters}
|
||||
self.p.reload_config(config)
|
||||
mock_fetchone.side_effect = Exception
|
||||
parameters['b.ar'] = 'bar'
|
||||
self.p.reload_config(config)
|
||||
with patch.object(MockCursor, 'fetchall',
|
||||
Mock(side_effect=[[('wal_block_size', '8191', None, 'integer', 'internal'),
|
||||
('wal_segment_size', '2048', '8kB', 'integer', 'internal'),
|
||||
('shared_buffers', '16384', '8kB', 'integer', 'postmaster'),
|
||||
('wal_buffers', '-1', '8kB', 'integer', 'postmaster'),
|
||||
('port', '5433', None, 'integer', 'postmaster')], Exception])):
|
||||
self.p.reload_config(config)
|
||||
parameters['autovacuum'] = 'on'
|
||||
self.p.reload_config(config)
|
||||
parameters['autovacuum'] = 'off'
|
||||
@@ -596,7 +588,7 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
|
||||
def test_postmaster_start_time(self):
|
||||
now = datetime.datetime.now()
|
||||
with patch.object(MockCursor, "fetchone", Mock(return_value=(now, True, '', '', '', '', False))):
|
||||
with patch.object(MockCursor, "fetchall", Mock(return_value=[(now, True, '', '', '', '', False)])):
|
||||
self.assertEqual(self.p.postmaster_start_time(), now.isoformat(sep=' '))
|
||||
t = Thread(target=self.p.postmaster_start_time)
|
||||
t.start()
|
||||
@@ -682,7 +674,7 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
self.assertIsNone(self.p.wait_for_startup())
|
||||
|
||||
def test_get_server_parameters(self):
|
||||
config = {'parameters': {'wal_level': 'hot_standby', 'max_prepared_transactions': 100}, 'listen': '0'}
|
||||
config = {'parameters': {'wal_level': 'hot_standby'}, 'listen': '0'}
|
||||
self.p._global_config = GlobalConfig({'synchronous_mode': True})
|
||||
self.p.config.get_server_parameters(config)
|
||||
self.p._global_config = GlobalConfig({'synchronous_mode': True, 'synchronous_mode_strict': True})
|
||||
@@ -715,7 +707,6 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
self.assertEqual(self.p.get_primary_timeline(), 1)
|
||||
|
||||
@patch.object(Postgresql, 'get_postgres_role_from_data_directory', Mock(return_value='replica'))
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=False))
|
||||
@patch.object(Bootstrap, 'running_custom_bootstrap', PropertyMock(return_value=True))
|
||||
@patch.object(Postgresql, 'controldata', Mock(return_value={'max_connections setting': '200',
|
||||
'max_worker_processes setting': '20',
|
||||
@@ -959,7 +950,6 @@ class TestPostgresql2(BaseTestPostgresql):
|
||||
@patch('patroni.postgresql.CallbackExecutor', Mock())
|
||||
@patch.object(Postgresql, 'get_major_version', Mock(return_value=140000))
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||
def setUp(self):
|
||||
super(TestPostgresql2, self).setUp()
|
||||
|
||||
@@ -968,3 +958,10 @@ class TestPostgresql2(BaseTestPostgresql):
|
||||
gucs = self.p.available_gucs
|
||||
self.assertIsInstance(gucs, CaseInsensitiveSet)
|
||||
self.assertEqual(gucs, mock_available_gucs.return_value)
|
||||
|
||||
def test_cluster_info_query(self):
|
||||
self.assertIn('diff(pg_catalog.pg_current_wal_flush_lsn(', self.p.cluster_info_query)
|
||||
self.p._major_version = 90600
|
||||
self.assertIn('diff(pg_catalog.pg_current_xlog_flush_location(', self.p.cluster_info_query)
|
||||
self.p._major_version = 90500
|
||||
self.assertIn('diff(pg_catalog.pg_current_xlog_location(', self.p.cluster_info_query)
|
||||
|
||||
+4
-4
@@ -142,25 +142,25 @@ class TestRaft(unittest.TestCase):
|
||||
raft._citus_group = '1'
|
||||
self.assertTrue(raft.manual_failover('foo', 'bar'))
|
||||
raft._citus_group = '0'
|
||||
self.assertTrue(raft.take_leader())
|
||||
cluster = raft.get_cluster()
|
||||
self.assertIsInstance(cluster, Cluster)
|
||||
self.assertIsInstance(cluster.workers[1], Cluster)
|
||||
leader = cluster.leader
|
||||
self.assertTrue(raft.delete_leader(leader))
|
||||
self.assertTrue(raft._sync_obj.set(raft.status_path, '{"optime":1234567,"slots":{"ls":12345}}'))
|
||||
leader = raft.get_cluster().leader
|
||||
raft.get_cluster()
|
||||
self.assertTrue(raft.update_leader(leader, '1', failsafe={'foo': 'bat'}))
|
||||
self.assertTrue(raft._sync_obj.set(raft.failsafe_path, '{"foo"}'))
|
||||
self.assertTrue(raft._sync_obj.set(raft.status_path, '{'))
|
||||
raft.get_citus_coordinator()
|
||||
self.assertTrue(raft.delete_sync_state())
|
||||
self.assertTrue(raft.delete_leader())
|
||||
self.assertTrue(raft.set_history_value(''))
|
||||
self.assertTrue(raft.delete_cluster())
|
||||
raft._citus_group = '1'
|
||||
self.assertTrue(raft.delete_cluster())
|
||||
raft._citus_group = None
|
||||
raft.get_cluster()
|
||||
self.assertTrue(raft.take_leader())
|
||||
raft.get_cluster()
|
||||
raft.watch(None, 0.001)
|
||||
raft._sync_obj.destroy()
|
||||
|
||||
|
||||
@@ -92,8 +92,9 @@ class TestRewind(BaseTestPostgresql):
|
||||
self.r.rewind_or_reinitialize_needed_and_possible(self.leader)
|
||||
|
||||
with patch.object(Postgresql, 'is_running', Mock(return_value=True)), \
|
||||
patch.object(MockCursor, 'fetchone',
|
||||
Mock(side_effect=[Exception, (0, 0, 1, 1, 0, 0, 0, 0, 0, None, None, None)])):
|
||||
patch.object(MockCursor, 'fetchone', Mock(side_effect=Exception)), \
|
||||
patch.object(MockCursor, 'fetchall',
|
||||
Mock(return_value=[(0, 0, 1, 1, 0, 0, 0, 0, 0, None, None, None)])):
|
||||
self.r.rewind_or_reinitialize_needed_and_possible(self.leader)
|
||||
|
||||
@patch.object(CancellableSubprocess, 'call', mock_cancellable_call)
|
||||
|
||||
+11
-11
@@ -51,7 +51,7 @@ class TestSlotsHandler(BaseTestPostgresql):
|
||||
self.s.sync_replication_slots(cluster, False)
|
||||
mock_debug.assert_called_once()
|
||||
self.p.set_role('replica')
|
||||
with patch.object(Postgresql, 'is_leader', Mock(return_value=False)), \
|
||||
with patch.object(Postgresql, 'is_primary', Mock(return_value=False)), \
|
||||
patch.object(SlotsHandler, 'drop_replication_slot') as mock_drop:
|
||||
self.s.sync_replication_slots(cluster, False, paused=True)
|
||||
mock_drop.assert_not_called()
|
||||
@@ -82,7 +82,7 @@ class TestSlotsHandler(BaseTestPostgresql):
|
||||
None, SyncState.empty(), None, {'ls': 10}, None)
|
||||
self.p.set_role('replica')
|
||||
with patch.object(Postgresql, '_query') as mock_query, \
|
||||
patch.object(Postgresql, 'is_leader', Mock(return_value=False)):
|
||||
patch.object(Postgresql, 'is_primary', Mock(return_value=False)):
|
||||
mock_query.return_value = [('ls', 'logical', 'b', 'a', 5, 12345, 105)]
|
||||
ret = self.s.sync_replication_slots(cluster, False)
|
||||
self.assertEqual(ret, [])
|
||||
@@ -96,20 +96,20 @@ class TestSlotsHandler(BaseTestPostgresql):
|
||||
self.s.sync_replication_slots(cluster, False)
|
||||
with patch.object(Postgresql, '_query') as mock_query:
|
||||
self.p.reset_cluster_info_state(None)
|
||||
mock_query.return_value.fetchone.return_value = (
|
||||
mock_query.return_value = [(
|
||||
1, 0, 0, 0, 0, 0, 0, 0, 0, None, None,
|
||||
[{"slot_name": "ls", "type": "logical", "datoid": 5, "plugin": "b",
|
||||
"confirmed_flush_lsn": 12345, "catalog_xmin": 105}])
|
||||
"confirmed_flush_lsn": 12345, "catalog_xmin": 105}])]
|
||||
self.assertEqual(self.p.slots(), {'ls': 12345})
|
||||
|
||||
self.p.reset_cluster_info_state(None)
|
||||
mock_query.return_value.fetchone.return_value = (
|
||||
mock_query.return_value = [(
|
||||
1, 0, 0, 0, 0, 0, 0, 0, 0, None, None,
|
||||
[{"slot_name": "ls", "type": "logical", "datoid": 6, "plugin": "b",
|
||||
"confirmed_flush_lsn": 12345, "catalog_xmin": 105}])
|
||||
"confirmed_flush_lsn": 12345, "catalog_xmin": 105}])]
|
||||
self.assertEqual(self.p.slots(), {})
|
||||
|
||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||
@patch.object(Postgresql, 'is_primary', Mock(return_value=False))
|
||||
def test__ensure_logical_slots_replica(self):
|
||||
self.p.set_role('replica')
|
||||
self.cluster.slots['ls'] = 12346
|
||||
@@ -136,21 +136,21 @@ class TestSlotsHandler(BaseTestPostgresql):
|
||||
|
||||
@patch.object(Postgresql, 'stop', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||
@patch.object(Postgresql, 'is_primary', Mock(return_value=False))
|
||||
def test_check_logical_slots_readiness(self):
|
||||
self.s.copy_logical_slots(self.cluster, ['ls'])
|
||||
with patch.object(MockCursor, '__iter__', Mock(return_value=iter([('postgresql0', None)]))), \
|
||||
patch.object(MockCursor, 'fetchone', Mock(side_effect=Exception)):
|
||||
patch.object(MockCursor, 'fetchall', Mock(side_effect=Exception)):
|
||||
self.assertFalse(self.s.check_logical_slots_readiness(self.cluster, None))
|
||||
with patch.object(MockCursor, '__iter__', Mock(return_value=iter([('postgresql0', None)]))), \
|
||||
patch.object(MockCursor, 'fetchone', Mock(return_value=(False,))):
|
||||
patch.object(MockCursor, 'fetchall', Mock(return_value=[(False,)])):
|
||||
self.assertFalse(self.s.check_logical_slots_readiness(self.cluster, None))
|
||||
with patch.object(MockCursor, '__iter__', Mock(return_value=iter([('ls', 100)]))):
|
||||
self.s.check_logical_slots_readiness(self.cluster, None)
|
||||
|
||||
@patch.object(Postgresql, 'stop', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||
@patch.object(Postgresql, 'is_primary', Mock(return_value=False))
|
||||
def test_on_promote(self):
|
||||
self.s.schedule_advance_slots({'foo': {'bar': 100}})
|
||||
self.s.copy_logical_slots(self.cluster, ['ls'])
|
||||
|
||||
@@ -202,7 +202,7 @@ class TestZooKeeper(unittest.TestCase):
|
||||
mock_logger.assert_called_once()
|
||||
|
||||
def test_delete_leader(self):
|
||||
self.assertTrue(self.zk.delete_leader())
|
||||
self.assertTrue(self.zk.delete_leader(self.zk.get_cluster().leader))
|
||||
|
||||
def test_set_failover_value(self):
|
||||
self.zk.set_failover_value('')
|
||||
@@ -276,7 +276,6 @@ class TestZooKeeper(unittest.TestCase):
|
||||
self.assertTrue(self.zk.delete_cluster())
|
||||
|
||||
def test_watch(self):
|
||||
self.zk.event.wait = Mock()
|
||||
self.zk.watch(None, 0)
|
||||
self.zk.event.is_set = Mock(return_value=True)
|
||||
self.zk._fetch_status = False
|
||||
|
||||
@@ -6,7 +6,6 @@ postgres_matrix =
|
||||
pg13: PG_MAJOR = 13
|
||||
pg14: PG_MAJOR = 14
|
||||
pg15: PG_MAJOR = 15
|
||||
pg16: PG_MAJOR = 16
|
||||
psycopg_deps =
|
||||
py{37,38,39,310,311}-{lin,win}: psycopg[binary]
|
||||
mac: psycopg2-binary
|
||||
@@ -107,7 +106,7 @@ description = Reformat code with black
|
||||
deps = black
|
||||
commands = black {posargs:patroni tests}
|
||||
|
||||
[testenv:pg{12,13,14,15,16}-docker-build]
|
||||
[testenv:pg{12,13,14,15}-docker-build]
|
||||
description = Build docker containers needed for testing
|
||||
labels =
|
||||
behave
|
||||
@@ -125,7 +124,7 @@ commands =
|
||||
--file features/Dockerfile
|
||||
allowlist_externals = docker
|
||||
|
||||
[testenv:pg{12,13,14,15,16}-docker-behave-{etcd}-{lin,mac}]
|
||||
[testenv:pg{12,13,14,15}-docker-behave-{etcd}-{lin,mac}]
|
||||
description = Run behaviour tests in patroni-dev docker container
|
||||
setenv =
|
||||
etcd: DCS=etcd
|
||||
@@ -134,7 +133,7 @@ setenv =
|
||||
labels =
|
||||
behave
|
||||
depends =
|
||||
pg{11,12,13,14,15,16}-docker-build
|
||||
pg{11,12,13,14,15}-docker-build
|
||||
|
||||
# There's a bug which affects calling multiple envs on the command line
|
||||
# This should be a valid command: tox -e 'py{36,37,38,39,310,311}-behave-{env:DCS}-lin'
|
||||
|
||||
Reference in New Issue
Block a user