mirror of
https://github.com/outbackdingo/patroni.git
synced 2026-08-29 00:49:32 +00:00
Compare commits
20
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e013c1a6ee | ||
|
|
03d9226633 | ||
|
|
5b4291bfab | ||
|
|
b31a4d55c9 | ||
|
|
3c24c33e59 | ||
|
|
83a060fc15 | ||
|
|
a2ceff1517 | ||
|
|
19f20ec2eb | ||
|
|
30f0f132e8 | ||
|
|
941e883dde | ||
|
|
89a162e000 | ||
|
|
0ab5b49757 | ||
|
|
80a03a4892 | ||
|
|
d2603402ea | ||
|
|
6b7f914da7 | ||
|
|
03107e6d8b | ||
|
|
77dba39585 | ||
|
|
89d794facc | ||
|
|
3333e78500 | ||
|
|
0ab4bc9d27 |
+3
-1
@@ -33,7 +33,7 @@ nosetests.xml
|
||||
coverage.xml
|
||||
htmlcov
|
||||
junit.xml
|
||||
features/output
|
||||
features/output*
|
||||
dummy
|
||||
|
||||
# Translations
|
||||
@@ -48,10 +48,12 @@ pgpass
|
||||
scm-source.json
|
||||
|
||||
# Sphinx-generated documentation
|
||||
docs/_build/
|
||||
docs/build/
|
||||
docs/source/_static/
|
||||
docs/source/_templates/
|
||||
docs/modules/
|
||||
docs/pdf/
|
||||
|
||||
# Pycharm IDE
|
||||
.idea/
|
||||
|
||||
@@ -115,7 +115,7 @@ Kubernetes
|
||||
- **PATRONI\_KUBERNETES\_ROLE\_LABEL**: (optional) name of the label containing role (master or replica or other custom value). Patroni will set this label on the pod it runs in. Default value is ``role``.
|
||||
- **PATRONI\_KUBERNETES\_LEADER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is `master`. Default value is `master`.
|
||||
- **PATRONI\_KUBERNETES\_FOLLOWER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is `replica`. Default value is `replica`.
|
||||
- **PATRONI\_KUBERNETES\_STANDBY\_LEADER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is ``standby-leader``. Default value is ``standby-leader``.
|
||||
- **PATRONI\_KUBERNETES\_STANDBY\_LEADER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is ``standby_leader``. Default value is ``master``.
|
||||
- **PATRONI\_KUBERNETES\_TMP\_ROLE\_LABEL**: (optional) name of the temporary label containing role (master or replica). Value of this label will always use the default of corresponding role. Set only when necessary.
|
||||
- **PATRONI\_KUBERNETES\_USE\_ENDPOINTS**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||
- **PATRONI\_KUBERNETES\_POD\_IP**: (optional) IP address of the pod Patroni is running in. This value is required when `PATRONI_KUBERNETES_USE_ENDPOINTS` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||
|
||||
+1
-76
@@ -25,82 +25,7 @@ We report new releases information :ref:`here <releases>`.
|
||||
Technical Requirements/Installation
|
||||
-----------------------------------
|
||||
|
||||
**Pre-requirements for Mac OS**
|
||||
|
||||
To install requirements on a Mac, run the following:
|
||||
|
||||
::
|
||||
|
||||
brew install postgresql etcd haproxy libyaml python
|
||||
|
||||
.. _psycopg2_install_options:
|
||||
|
||||
**Psycopg**
|
||||
|
||||
Starting from `psycopg2-2.8 <http://initd.org/psycopg/articles/2019/04/04/psycopg-28-released/>`__ the binary version of psycopg2 will no longer be installed by default. Installing it from the source code requires C compiler and postgres+python dev packages.
|
||||
Since in the python world it is not possible to specify dependency as ``psycopg2 OR psycopg2-binary`` you will have to decide how to install it.
|
||||
|
||||
There are a few options available:
|
||||
|
||||
1. Use the package manager from your distro
|
||||
|
||||
::
|
||||
|
||||
sudo apt-get install python3-psycopg2 # install psycopg2 module on Debian/Ubuntu
|
||||
sudo yum install python3-psycopg2 # install psycopg2 on RedHat/Fedora/CentOS
|
||||
|
||||
2. Install psycopg2 from the binary package
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg2-binary
|
||||
|
||||
3. Install psycopg2 from source
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg2>=2.5.4
|
||||
|
||||
4. Use psycopg 3.0 instead of psycopg2
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg[binary]>=3.0.0
|
||||
|
||||
**General installation for pip**
|
||||
|
||||
Patroni can be installed with pip:
|
||||
|
||||
::
|
||||
|
||||
pip install patroni[dependencies]
|
||||
|
||||
where dependencies can be either empty, or consist of one or more of the following:
|
||||
|
||||
etcd or etcd3
|
||||
`python-etcd` module in order to use Etcd as Distributed Configuration Store (DCS)
|
||||
consul
|
||||
`python-consul` module in order to use Consul as DCS
|
||||
zookeeper
|
||||
`kazoo` module in order to use Zookeeper as DCS
|
||||
exhibitor
|
||||
`kazoo` module in order to use Exhibitor as DCS (same dependencies as for Zookeeper)
|
||||
kubernetes
|
||||
`kubernetes` module in order to use Kubernetes as DCS in Patroni
|
||||
raft
|
||||
`pysyncobj` module in order to use python Raft implementation as DCS
|
||||
aws
|
||||
`boto3` in order to use AWS callbacks
|
||||
|
||||
For example, the command in order to install Patroni together with dependencies for Etcd as a DCS and AWS callbacks is:
|
||||
|
||||
::
|
||||
|
||||
pip install patroni[etcd,aws]
|
||||
|
||||
Note that external tools to call in the replica creation or custom bootstrap scripts (i.e. WAL-E) should be installed
|
||||
independently of Patroni.
|
||||
|
||||
Go :ref:`here <installation>` for guidance on installing and upgrading Patroni on various platforms.
|
||||
|
||||
.. _running_configuring:
|
||||
|
||||
|
||||
+1
-1
@@ -140,7 +140,7 @@ An example of ``patronictl switchover`` on the worker cluster::
|
||||
| work2-1 | 172.27.0.5 | Sync Standby | running | 1 | 0 |
|
||||
| work2-2 | 172.27.0.7 | Leader | running | 1 | |
|
||||
+---------+------------+--------------+---------+----+-----------+
|
||||
Are you sure you want to switchover cluster demo, demoting current primary work2-2? [y/N]: y
|
||||
Are you sure you want to perform a switchover in the cluster demo, demoting current primary work2-2? [y/N]: y
|
||||
2022-12-22 07:02:40.33003 Successfully switched over to "work2-1"
|
||||
+ Citus cluster: demo (group: 2, 7179854924063375386) ------+
|
||||
| Member | Host | Role | State | TL | Lag in MB |
|
||||
|
||||
@@ -54,6 +54,13 @@ apidoc_output_dir = 'modules'
|
||||
apidoc_excluded_paths = excludes
|
||||
apidoc_separate_modules = True
|
||||
|
||||
# Include autodoc for all members, including private ones and the ones that are missing a docstring.
|
||||
autodoc_default_options = {
|
||||
"members": True,
|
||||
"undoc-members": True,
|
||||
"private-members": True,
|
||||
}
|
||||
|
||||
# Add any paths that contain templates here, relative to this directory.
|
||||
templates_path = ['_templates']
|
||||
|
||||
@@ -280,6 +287,14 @@ def doctree_read(app, doctree):
|
||||
toc_tree_node['entries'].remove(e)
|
||||
|
||||
|
||||
def autodoc_skip(app, what, name, obj, would_skip, options):
|
||||
"""Include autodoc of ``__init__`` methods, which are skipped by default."""
|
||||
if name == "__init__":
|
||||
return False
|
||||
return would_skip
|
||||
|
||||
|
||||
|
||||
# A possibility to have an own stylesheet, to add new rules or override existing ones
|
||||
# For the latter case, the CSS specificity of the rules should be higher than the default ones
|
||||
def setup(app):
|
||||
@@ -292,3 +307,4 @@ def setup(app):
|
||||
app.connect('builder-inited', builder_inited)
|
||||
app.connect('env-get-outdated', env_get_outdated)
|
||||
app.connect('doctree-read', doctree_read)
|
||||
app.connect("autodoc-skip-member", autodoc_skip)
|
||||
|
||||
@@ -22,6 +22,7 @@ Currently supported PostgreSQL versions: 9.3 to 15.
|
||||
:caption: Contents:
|
||||
|
||||
README
|
||||
installation
|
||||
patroni_configuration
|
||||
rest_api
|
||||
replica_bootstrap
|
||||
|
||||
@@ -0,0 +1,201 @@
|
||||
.. _installation:
|
||||
|
||||
Installation
|
||||
============
|
||||
|
||||
Pre-requirements for Mac OS
|
||||
---------------------------
|
||||
|
||||
To install requirements on a Mac, run the following:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
brew install postgresql etcd haproxy libyaml python
|
||||
|
||||
.. _psycopg2_install_options:
|
||||
|
||||
Psycopg
|
||||
-------
|
||||
|
||||
Starting from `psycopg2-2.8`_ the binary version of psycopg2 will no longer be installed by default. Installing it from
|
||||
the source code requires C compiler and postgres+python dev packages. Since in the python world it is not possible to
|
||||
specify dependency as ``psycopg2 OR psycopg2-binary`` you will have to decide how to install it.
|
||||
|
||||
There are a few options available:
|
||||
|
||||
1. Use the package manager from your distro
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
sudo apt-get install python3-psycopg2 # install psycopg2 module on Debian/Ubuntu
|
||||
sudo yum install python3-psycopg2 # install psycopg2 on RedHat/Fedora/CentOS
|
||||
|
||||
2. Install psycopg2 from the binary package
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
pip install psycopg2-binary
|
||||
|
||||
3. Install psycopg2 from source
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
pip install psycopg2>=2.5.4
|
||||
|
||||
4. Use psycopg 3.0 instead of psycopg2
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
pip install psycopg[binary]>=3.0.0
|
||||
|
||||
General installation for pip
|
||||
----------------------------
|
||||
|
||||
Patroni can be installed with pip:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
pip install patroni[dependencies]
|
||||
|
||||
where ``dependencies`` can be either empty, or consist of one or more of the following:
|
||||
|
||||
etcd or etcd3
|
||||
`python-etcd` module in order to use Etcd as Distributed Configuration Store (DCS)
|
||||
consul
|
||||
`python-consul` module in order to use Consul as DCS
|
||||
zookeeper
|
||||
`kazoo` module in order to use Zookeeper as DCS
|
||||
exhibitor
|
||||
`kazoo` module in order to use Exhibitor as DCS (same dependencies as for Zookeeper)
|
||||
kubernetes
|
||||
`kubernetes` module in order to use Kubernetes as DCS in Patroni
|
||||
raft
|
||||
`pysyncobj` module in order to use python Raft implementation as DCS
|
||||
aws
|
||||
`boto3` in order to use AWS callbacks
|
||||
|
||||
For example, the command in order to install Patroni together with dependencies for Etcd as a DCS and AWS callbacks is:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
pip install patroni[etcd,aws]
|
||||
|
||||
Note that external tools to call in the replica creation or custom bootstrap scripts (i.e. WAL-E) should be installed
|
||||
independently of Patroni.
|
||||
|
||||
.. _package_installation:
|
||||
|
||||
Package installation on Linux
|
||||
-----------------------------
|
||||
|
||||
Patroni packages may be available for your operating system, produced by the Postgres community for:
|
||||
|
||||
* RHEL, RockyLinux, AlmaLinux;
|
||||
* Debian and Ubuntu;
|
||||
* SUSE Enterprise Linux.
|
||||
|
||||
You can also find packages for direct dependencies of Patroni, like python modules that might not be available in
|
||||
the official operating system repositories.
|
||||
|
||||
For more information see the `PGDG repository`_ documentation.
|
||||
|
||||
If you are on a RedHat Enterprise Linux derivative operating system you may also require packages from EPEL, see
|
||||
`EPEL repository`_ documentation.
|
||||
|
||||
Once you have installed the PGDG repository for your OS you can install patroni.
|
||||
|
||||
.. note::
|
||||
|
||||
Patroni packages are not maintained by the Patroni developers, but rather by the Postgres community. If you
|
||||
require support please first try connecting on `Postgres slack`_.
|
||||
|
||||
Installing on Debian derivatives
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
With PGDG repo installed, see :ref:`above <package_installation>`, install Patroni via apt run:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
apt-get install patroni
|
||||
|
||||
Installing on RedHat derivatives
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
With PGDG repo installed, see :ref:`above <package_installation>`, install patroni with an etcd DCS via dnf on RHEL 9
|
||||
(and derivatives) run:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
dnf install patroni patroni-etcd
|
||||
|
||||
You can install etcd from PGDG if your RedHat derivative distribution does not provide packages. On the nodes that will
|
||||
host the DCS run:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
dnf install 'dnf-command(config-manager)'
|
||||
dnf config-manager --enable pgdg-rhel9-extras
|
||||
dnf install etcd
|
||||
|
||||
You can replace the version of RHEL with `8` in the repo to make `pgdg-rhel8-extras` if needed. The repo name is still
|
||||
`pgdg-rhelN-extras` on RockyLinux, AlmaLinux, Oracle Linux, etc...
|
||||
|
||||
Installing on SUSE Enterprise Linux
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
You might need to enable the SUSE PackageHub repositories for some dependencies. see `SUSE PackageHub`_ documentation.
|
||||
|
||||
For SLES 15 with PGDG repo installed, see :ref:`above <package_installation>`, you can install patroni using:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
zypper install patroni patroni-etcd
|
||||
|
||||
With the SUSE PackageHub repo enabled you can also install etcd:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
SUSEConnect -p PackageHub/15.5/x86_64
|
||||
zypper install etcd
|
||||
|
||||
Upgrading
|
||||
---------
|
||||
|
||||
Upgrading patroni is a very simple process, just update the software installation and restart the Patroni daemon on
|
||||
each node in the cluster.
|
||||
|
||||
However, restarting the Patroni daemon will result in a Postgres database restart. In some situations this may cause
|
||||
a failover of the primary node in your cluster, therefore it is recommended to put the cluster into maintenance mode
|
||||
until the Patroni daemon restart has been completed.
|
||||
|
||||
To put the cluster in maintenance mode, run the following command on one of the patroni nodes:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
patronictl pause --wait
|
||||
|
||||
Then on each node in the cluster, perform the package upgrade required for your OS:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
apt-get update && apt-get install patroni patroni-etcd
|
||||
|
||||
Restart the patroni daemon process on each node:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
systemctl restart patroni
|
||||
|
||||
Then finally resume monitoring of Postgres with patroni to take it out of maintenance mode:
|
||||
|
||||
.. code-block:: shell
|
||||
|
||||
patronictl resume --wait
|
||||
|
||||
The cluster will now be full operational with the new version of Patroni.
|
||||
|
||||
.. _psycopg2-2.8: http://initd.org/psycopg/articles/2019/04/04/psycopg-28-released/
|
||||
.. _PGDG repository: https://www.postgresql.org/download/linux/
|
||||
.. _EPEL repository: https://docs.fedoraproject.org/en-US/epel/
|
||||
.. _SUSE PackageHub: https://packagehub.suse.com/how-to-use/
|
||||
.. _Postgres slack: http://pgtreats.info/slack-invite
|
||||
@@ -90,6 +90,42 @@ The parameters would be applied in the following order (run-time are given the h
|
||||
This allows configuration for all the nodes (2), configuration for a specific node using ``ALTER SYSTEM`` (3) and ensures that parameters essential to the running of Patroni are enforced (4), as well as leaves room for configuration tools that manage `postgresql.conf` directly without involving Patroni (1).
|
||||
|
||||
|
||||
PostgreSQL parameters that touch shared memory
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
PostgreSQL has some parameters that determine the size of the shared memory used by them:
|
||||
|
||||
- **max_connections**
|
||||
- **max_prepared_transactions**
|
||||
- **max_locks_per_transaction**
|
||||
- **max_wal_senders**
|
||||
- **max_worker_processes**
|
||||
|
||||
Changing these parameters require a PostgreSQL restart to take effect, and their shared memory structures cannot be smaller on the standby nodes than on the primary node.
|
||||
|
||||
As explained before, Patroni restrict changing their values through :ref:`dynamic configuration <dynamic_configuration>`, which usually consists of:
|
||||
|
||||
1. Applying changes through ``patronictl edit-config`` (or via REST API ``/config`` endpoint)
|
||||
2. Restarting nodes through ``patronictl restart`` (or via REST API ``/restart`` endpoint)
|
||||
|
||||
**Note:** please keep in mind that you should perform a restart of the PostgreSQL nodes through ``patronictl restart`` command, or via REST API ``/restart`` endpoint. An attempt to restart PostgreSQL by restarting the Patroni daemon, e.g. by executing ``systemctl restart patroni``, can cause a failover to occur in the cluster, if you are restarting the primary node.
|
||||
|
||||
However, as those settings manage shared memory, some extra care should be taken when restarting the nodes:
|
||||
|
||||
* If you want to **increase** the value of any of those settings:
|
||||
|
||||
1. Restart all standbys first
|
||||
2. Restart the primary after that
|
||||
|
||||
* If you want to **decrease** the value of any of those settings:
|
||||
|
||||
1. Restart the primary first
|
||||
2. Restart all standbys after that
|
||||
|
||||
**Note:** if you attempt to restart all nodes in one go after **decreasing** the value of any of those settings, Patroni will ignore the change and restart the standby with the original setting value, thus requiring that you restart the standbys again later. Patroni does that to prevent the standby to enter in an infinite crash loop, because PostgreSQL quits with a `FATAL` message if you attempt to set any of those parameters to a value lower than what is visible in ``pg_controldata`` on the Standby node. In other words, we can only decrease the setting on the standby once its ``pg_controldata`` is up-to-date with the primary in regards to these changes on the primary.
|
||||
|
||||
More information about that can be found at `PostgreSQL Administrator's Overview <https://www.postgresql.org/docs/current/hot-standby.html#HOT-STANDBY-ADMIN>`__.
|
||||
|
||||
Patroni configuration parameters
|
||||
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
|
||||
+1
-1
@@ -19,7 +19,7 @@ When Patroni runs in a paused mode, it does not change the state of PostgreSQL,
|
||||
|
||||
- For the Postgres primary with the leader lock Patroni updates the lock. If the node with the leader lock stops being the primary (i.e. is demoted manually), Patroni will release the lock instead of promoting the node back.
|
||||
|
||||
- Manual unscheduled restart, reinitialize and manual failover are allowed. Manual failover is only allowed if the node to failover to is specified. In the paused mode, manual failover does not require a running primary node.
|
||||
- Manual unscheduled restart, manual unscheduled failover/switchover and reinitialize are allowed. No scheduled action is allowed. Manual switchover is only allowed if the node to switch over to is specified.
|
||||
|
||||
- If 'parallel' primaries are detected by Patroni, it emits a warning, but does not demote the primary without the leader lock.
|
||||
|
||||
|
||||
+120
-42
@@ -131,7 +131,8 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
||||
"database_system_identifier": "7268616322854375442",
|
||||
"patroni": {
|
||||
"version": "3.1.0",
|
||||
"scope": "demo"
|
||||
"scope": "demo",
|
||||
"name": "patroni1"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -178,7 +179,8 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
||||
"database_system_identifier": "7268616322854375442",
|
||||
"patroni": {
|
||||
"version": "3.1.0",
|
||||
"scope": "demo"
|
||||
"scope": "demo",
|
||||
"name": "patroni1"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -223,7 +225,8 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
||||
"database_system_identifier": "7268616322854375442",
|
||||
"patroni": {
|
||||
"version": "3.1.0",
|
||||
"scope": "demo"
|
||||
"scope": "demo",
|
||||
"name": "patroni1"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -267,7 +270,8 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
||||
"database_system_identifier": "7268616322854375442",
|
||||
"patroni": {
|
||||
"version": "3.1.0",
|
||||
"scope": "demo"
|
||||
"scope": "demo",
|
||||
"name": "patroni1"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -279,70 +283,70 @@ Retrieve the Patroni metrics in Prometheus format through the ``GET /metrics`` e
|
||||
|
||||
# HELP patroni_version Patroni semver without periods. \
|
||||
# TYPE patroni_version gauge
|
||||
patroni_version{scope="batman"} 020103
|
||||
patroni_version{scope="batman",name="patroni1"} 020103
|
||||
# HELP patroni_postgres_running Value is 1 if Postgres is running, 0 otherwise.
|
||||
# TYPE patroni_postgres_running gauge
|
||||
patroni_postgres_running{scope="batman"} 1
|
||||
patroni_postgres_running{scope="batman",name="patroni1"} 1
|
||||
# HELP patroni_postmaster_start_time Epoch seconds since Postgres started.
|
||||
# TYPE patroni_postmaster_start_time gauge
|
||||
patroni_postmaster_start_time{scope="batman"} 1657656955.179243
|
||||
patroni_postmaster_start_time{scope="batman",name="patroni1"} 1657656955.179243
|
||||
# HELP patroni_master Value is 1 if this node is the leader, 0 otherwise.
|
||||
# TYPE patroni_master gauge
|
||||
patroni_master{scope="batman"} 1
|
||||
patroni_master{scope="batman",name="patroni1"} 1
|
||||
# HELP patroni_primary Value is 1 if this node is the leader, 0 otherwise.
|
||||
# TYPE patroni_primary gauge
|
||||
patroni_primary{scope="batman"} 1
|
||||
patroni_primary{scope="batman",name="patroni1"} 1
|
||||
# HELP patroni_xlog_location Current location of the Postgres transaction log, 0 if this node is not the leader.
|
||||
# TYPE patroni_xlog_location counter
|
||||
patroni_xlog_location{scope="batman"} 22320573386952
|
||||
patroni_xlog_location{scope="batman",name="patroni1"} 22320573386952
|
||||
# HELP patroni_standby_leader Value is 1 if this node is the standby_leader, 0 otherwise.
|
||||
# TYPE patroni_standby_leader gauge
|
||||
patroni_standby_leader{scope="batman"} 0
|
||||
patroni_standby_leader{scope="batman",name="patroni1"} 0
|
||||
# HELP patroni_replica Value is 1 if this node is a replica, 0 otherwise.
|
||||
# TYPE patroni_replica gauge
|
||||
patroni_replica{scope="batman"} 0
|
||||
patroni_replica{scope="batman",name="patroni1"} 0
|
||||
# HELP patroni_sync_standby Value is 1 if this node is a sync standby replica, 0 otherwise.
|
||||
# TYPE patroni_sync_standby gauge
|
||||
patroni_sync_standby{scope="batman"} 0
|
||||
patroni_sync_standby{scope="batman",name="patroni1"} 0
|
||||
# HELP patroni_xlog_received_location Current location of the received Postgres transaction log, 0 if this node is not a replica.
|
||||
# TYPE patroni_xlog_received_location counter
|
||||
patroni_xlog_received_location{scope="batman"} 0
|
||||
patroni_xlog_received_location{scope="batman",name="patroni1"} 0
|
||||
# HELP patroni_xlog_replayed_location Current location of the replayed Postgres transaction log, 0 if this node is not a replica.
|
||||
# TYPE patroni_xlog_replayed_location counter
|
||||
patroni_xlog_replayed_location{scope="batman"} 0
|
||||
patroni_xlog_replayed_location{scope="batman",name="patroni1"} 0
|
||||
# HELP patroni_xlog_replayed_timestamp Current timestamp of the replayed Postgres transaction log, 0 if null.
|
||||
# TYPE patroni_xlog_replayed_timestamp gauge
|
||||
patroni_xlog_replayed_timestamp{scope="batman"} 0
|
||||
patroni_xlog_replayed_timestamp{scope="batman",name="patroni1"} 0
|
||||
# HELP patroni_xlog_paused Value is 1 if the Postgres xlog is paused, 0 otherwise.
|
||||
# TYPE patroni_xlog_paused gauge
|
||||
patroni_xlog_paused{scope="batman"} 0
|
||||
patroni_xlog_paused{scope="batman",name="patroni1"} 0
|
||||
# HELP patroni_postgres_streaming Value is 1 if Postgres is streaming, 0 otherwise.
|
||||
# TYPE patroni_postgres_streaming gauge
|
||||
patroni_postgres_streaming{scope="batman"} 1
|
||||
patroni_postgres_streaming{scope="batman",name="patroni1"} 1
|
||||
# HELP patroni_postgres_in_archive_recovery Value is 1 if Postgres is replicating from archive, 0 otherwise.
|
||||
# TYPE patroni_postgres_in_archive_recovery gauge
|
||||
patroni_postgres_in_archive_recovery{scope="batman"} 0
|
||||
patroni_postgres_in_archive_recovery{scope="batman",name="patroni1"} 0
|
||||
# HELP patroni_postgres_server_version Version of Postgres (if running), 0 otherwise.
|
||||
# TYPE patroni_postgres_server_version gauge
|
||||
patroni_postgres_server_version {scope="batman"} 140004
|
||||
patroni_postgres_server_version{scope="batman",name="patroni1"} 140004
|
||||
# HELP patroni_cluster_unlocked Value is 1 if the cluster is unlocked, 0 if locked.
|
||||
# TYPE patroni_cluster_unlocked gauge
|
||||
patroni_cluster_unlocked{scope="batman"} 0
|
||||
patroni_cluster_unlocked{scope="batman",name="patroni1"} 0
|
||||
# HELP patroni_postgres_timeline Postgres timeline of this node (if running), 0 otherwise.
|
||||
# TYPE patroni_postgres_timeline counter
|
||||
patroni_failsafe_mode_is_active{scope="batman"} 0
|
||||
patroni_failsafe_mode_is_active{scope="batman",name="patroni1"} 0
|
||||
# HELP patroni_postgres_timeline Postgres timeline of this node (if running), 0 otherwise.
|
||||
# TYPE patroni_postgres_timeline counter
|
||||
patroni_postgres_timeline{scope="batman"} 24
|
||||
patroni_postgres_timeline{scope="batman",name="patroni1"} 24
|
||||
# HELP patroni_dcs_last_seen Epoch timestamp when DCS was last contacted successfully by Patroni.
|
||||
# TYPE patroni_dcs_last_seen gauge
|
||||
patroni_dcs_last_seen{scope="batman"} 1677658321
|
||||
patroni_dcs_last_seen{scope="batman",name="patroni1"} 1677658321
|
||||
# HELP patroni_pending_restart Value is 1 if the node needs a restart, 0 otherwise.
|
||||
# TYPE patroni_pending_restart gauge
|
||||
patroni_pending_restart{scope="batman"} 1
|
||||
patroni_pending_restart{scope="batman",name="patroni1"} 1
|
||||
# HELP patroni_is_paused Value is 1 if auto failover is disabled, 0 otherwise.
|
||||
# TYPE patroni_is_paused gauge
|
||||
patroni_is_paused{scope="batman"} 1
|
||||
patroni_is_paused{scope="batman",name="patroni1"} 1
|
||||
|
||||
|
||||
Cluster status endpoints
|
||||
@@ -381,6 +385,7 @@ Cluster status endpoints
|
||||
"lag": 0
|
||||
}
|
||||
],
|
||||
"scope": "demo",
|
||||
"scheduled_switchover": {
|
||||
"at": "2023-09-24T10:36:00+02:00",
|
||||
"from": "patroni1",
|
||||
@@ -489,8 +494,9 @@ Let's check that the node processed this configuration. First of all it should s
|
||||
"location": 2197818976
|
||||
},
|
||||
"patroni": {
|
||||
"version": "1.0",
|
||||
"scope": "batman",
|
||||
"version": "1.0"
|
||||
"name": "patroni1"
|
||||
},
|
||||
"state": "running",
|
||||
"role": "master",
|
||||
@@ -554,39 +560,111 @@ The above call removes ``postgresql.parameters.max_connections`` from the dynami
|
||||
Switchover and failover endpoints
|
||||
---------------------------------
|
||||
|
||||
``POST /switchover`` or ``POST /failover``. These endpoints are very similar to each other. There are a couple of minor differences though:
|
||||
.. _switchover_api:
|
||||
|
||||
1. The failover endpoint allows to perform a manual failover when there are no healthy nodes, but at the same time it will not allow you to schedule a switchover.
|
||||
Switchover
|
||||
^^^^^^^^^^
|
||||
|
||||
2. The switchover endpoint is the opposite. It works only when the cluster is healthy (there is a leader) and allows to schedule a switchover at a given time.
|
||||
``/switchover`` endpoint only works when the cluster is healthy (there is a leader). It also allows to schedule a switchover at a given time.
|
||||
|
||||
When calling ``/switchover`` endpoint a candidate can be specified but is not required, in contrast to ``/failover`` endpoint. If a candidate is not provided, all the eligible nodes of the cluster will participate in the leader race after the leader stepped down.
|
||||
|
||||
In the JSON body of the ``POST`` request you must specify at least the ``leader`` or ``candidate`` fields and optionally the ``scheduled_at`` field if you want to schedule a switchover at a specific time.
|
||||
In the JSON body of the ``POST`` request you must specify the ``leader`` field. The ``candidate`` and the ``scheduled_at`` fields are optional and can be used to schedule a switchover at a specific time.
|
||||
|
||||
Depending on the situation, requests might return different HTTP status codes and bodies. Status code **200** is returned when the switchover or failover successfully completed. If the switchover was successfully scheduled, Patroni will return HTTP status code **202**. In case something went wrong, the error status code (one of **400**, **412**, or **503**) will be returned with some details in the response body.
|
||||
|
||||
Example: perform a failover to the specific node:
|
||||
``DELETE /switchover`` can be used to delete the currently scheduled switchover.
|
||||
|
||||
**Example:** perform a switchover to any healthy standby
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8009/failover -XPOST -d '{"candidate":"postgresql1"}'
|
||||
Successfully failed over to "postgresql1"
|
||||
$ curl -s http://localhost:8008/switchover -XPOST -d '{"leader":"postgresql1"}'
|
||||
Successfully switched over to "postgresql2"
|
||||
|
||||
|
||||
Example: schedule a switchover from the leader to any other healthy replica in the cluster at a specific time:
|
||||
**Example:** perform a switchover to a specific node
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/switchover -XPOST -d \
|
||||
'{"leader":"postgresql0","scheduled_at":"2019-09-24T12:00+00"}'
|
||||
Switchover scheduled
|
||||
$ curl -s http://localhost:8008/switchover -XPOST -d \
|
||||
'{"leader":"postgresql1","candidate":"postgresql2"}'
|
||||
Successfully switched over to "postgresql2"
|
||||
|
||||
|
||||
Depending on the situation the request might finish with a different HTTP status code and body. The status code **200** is returned when the switchover or failover successfully completed. If the switchover was successfully scheduled, Patroni will return HTTP status code **202**. In case something went wrong, the error status code (one of **400**, **412** or **503**) will be returned with some details in the response body. For more information please check the source code of ``patroni/api.py:do_POST_failover()`` method.
|
||||
**Example:** schedule a switchover from the leader to any other healthy standby in the cluster at a specific time.
|
||||
|
||||
- ``DELETE /switchover``: delete the scheduled switchover
|
||||
.. code-block:: bash
|
||||
|
||||
The ``POST /switchover`` and ``POST failover`` endpoints are used by ``patronictl switchover`` and ``patronictl failover``, respectively.
|
||||
The ``DELETE /switchover`` is used by ``patronictl flush <cluster-name> switchover``.
|
||||
$ curl -s http://localhost:8008/switchover -XPOST -d \
|
||||
'{"leader":"postgresql0","scheduled_at":"2019-09-24T12:00+00"}'
|
||||
Switchover scheduled
|
||||
|
||||
|
||||
Failover
|
||||
^^^^^^^^
|
||||
|
||||
``/failover`` endpoint can be used to perform a manual failover when there are no healthy nodes (e.g. to an asynchronous standby if all synchronous standbys are not healthy enough to promote). However there is no requirement for a cluster not to have leader - failover can also be run on a healthy cluster.
|
||||
|
||||
In the JSON body of the ``POST`` request you must specify the ``candidate`` field. If the ``leader`` field is specified, a switchover is triggered instead.
|
||||
|
||||
**Example:**
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/failover -XPOST -d '{"candidate":"postgresql1"}'
|
||||
Successfully failed over to "postgresql1"
|
||||
|
||||
.. warning::
|
||||
:ref:`Be very careful <failover_healthcheck>` when using this endpoint, as this can cause data loss in certain situations. In most cases, :ref:`the switchover endpoint <switchover_api>` satisfies the administrator's needs.
|
||||
|
||||
|
||||
``POST /switchover`` and ``POST /failover`` endpoints are used by ``patronictl switchover`` and ``patronictl failover``, respectively.
|
||||
|
||||
``DELETE /switchover`` is used by ``patronictl flush <cluster-name> switchover``.
|
||||
|
||||
.. list-table:: Failover/Switchover comparison
|
||||
:widths: 25 25 25
|
||||
:header-rows: 1
|
||||
|
||||
* -
|
||||
- Failover
|
||||
- Switchover
|
||||
* - Requires leader specified
|
||||
- no
|
||||
- yes
|
||||
* - Requires candidate specified
|
||||
- yes
|
||||
- no
|
||||
* - Can be run in pause
|
||||
- yes
|
||||
- yes (only to a specific candidate)
|
||||
* - Can be scheduled
|
||||
- no
|
||||
- yes (if not in pause)
|
||||
|
||||
.. _failover_healthcheck:
|
||||
|
||||
Healthy standby
|
||||
^^^^^^^^^^^^^^^
|
||||
|
||||
There are a couple of checks that a member of a cluster should pass to be able to participate in the leader race during a switchover or to become a leader as a failover/switchover candidate:
|
||||
|
||||
- be reachable via Patroni API;
|
||||
- not have ``nofailover`` tag set to ``true``;
|
||||
- have watchdog fully functional (if required by the configuration);
|
||||
- in case of a switchover in a healthy cluster or an automatic failover, not exceed maximum replication lag (``maximum_lag_on_failover`` :ref:`configuration parameter <dynamic_configuration>`);
|
||||
- in case of a switchover in a healthy cluster or an automatic failover, not have a timeline number smaller than the cluster timeline if ``check_timeline`` :ref:`configuration parameter <dynamic_configuration>` is set to ``true``;
|
||||
- in :ref:`synchronous mode <synchronous_mode>`:
|
||||
|
||||
- In case of a switchover (both with and without a candidate): be listed in the ``/sync`` key members;
|
||||
- For a failover in both healthy and unhealthy clusters, this check is omitted.
|
||||
|
||||
.. warning::
|
||||
In case of a manual failover in a cluster without a leader, a candidate will be allowed to promote even if:
|
||||
- it is not in the ``/sync`` key members when synchronous mode is enabled;
|
||||
- its lag exceeds the maximum replication lag allowed;
|
||||
- it has the timeline number smaller than the last known cluster timeline.
|
||||
|
||||
|
||||
Restart endpoint
|
||||
|
||||
@@ -171,7 +171,7 @@ Kubernetes
|
||||
- **role\_label**: (optional) name of the label containing role (master or replica or other custom value). Patroni will set this label on the pod it runs in. Default value is ``role``.
|
||||
- **leader\_label\_value**: (optional) value of the pod label when Postgres role is ``master``. Default value is ``master``.
|
||||
- **follower\_label\_value**: (optional) value of the pod label when Postgres role is ``replica``. Default value is ``replica``.
|
||||
- **standby\_leader\_label\_value**: (optional) value of the pod label when Postgres role is ``standby-leader``. Default value is ``standby-leader``.
|
||||
- **standby\_leader\_label\_value**: (optional) value of the pod label when Postgres role is ``standby_leader``. Default value is ``master``.
|
||||
- **tmp_\role\_label**: (optional) name of the temporary label containing role (master or replica). Value of this label will always use the default of corresponding role. Set only when necessary.
|
||||
- **use\_endpoints**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||
- **pod\_ip**: (optional) IP address of the pod Patroni is running in. This value is required when `use_endpoints` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||
|
||||
+5
-2
@@ -27,8 +27,8 @@ RUN set -ex \
|
||||
&& apt-get update \
|
||||
&& apt-get reinstall init-system-helpers \
|
||||
&& apt-get install -y \
|
||||
python3-pip \
|
||||
python3-dev \
|
||||
python3-venv \
|
||||
rsync \
|
||||
curl \
|
||||
gcc \
|
||||
@@ -40,7 +40,9 @@ RUN set -ex \
|
||||
net-tools \
|
||||
iputils-ping \
|
||||
&& rm -rf /var/cache/apt \
|
||||
&& python3 -m pip install --no-cache-dir tox \
|
||||
\
|
||||
&& python3 -m venv /tox \
|
||||
&& /tox/bin/pip install --no-cache-dir tox>=4 \
|
||||
\
|
||||
&& mkdir -p "$PGHOME" \
|
||||
&& sed -i "s|/var/lib/postgresql.*|$PGHOME:/bin/bash|" /etc/passwd \
|
||||
@@ -50,6 +52,7 @@ RUN set -ex \
|
||||
&& curl -sL "$ETCDURL/etcd-v$ETCDVERSION-linux-$(dpkg --print-architecture).tar.gz" \
|
||||
| tar xz -C /usr/local/bin --strip=1 --wildcards --no-anchored etcd etcdctl
|
||||
|
||||
ENV PATH="/tox/bin:$PATH"
|
||||
|
||||
# This Dockerfile syntax only works with docker buildx and the syntax
|
||||
# line at the top of this file.
|
||||
|
||||
+10
-36
@@ -13,6 +13,7 @@ from argparse import Namespace
|
||||
from typing import Any, Dict, Optional, TYPE_CHECKING
|
||||
|
||||
from patroni.daemon import AbstractPatroniDaemon, abstract_main, get_base_arg_parser
|
||||
from patroni.tags import Tags
|
||||
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
from .config import Config
|
||||
@@ -20,7 +21,7 @@ if TYPE_CHECKING: # pragma: no cover
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Patroni(AbstractPatroniDaemon):
|
||||
class Patroni(AbstractPatroniDaemon, Tags):
|
||||
"""Implement ``patroni`` command daemon.
|
||||
|
||||
:ivar version: Patroni version.
|
||||
@@ -30,7 +31,6 @@ class Patroni(AbstractPatroniDaemon):
|
||||
:ivar api: REST API server instance of this node.
|
||||
:ivar request: wrapper for performing HTTP requests.
|
||||
:ivar ha: HA handler.
|
||||
:ivar tags: cache of custom tags configured for this node.
|
||||
:ivar next_run: time when to run the next HA loop cycle.
|
||||
:ivar scheduled_restart: when a restart has been scheduled to occur, if any. In that case, should contain two keys:
|
||||
* ``schedule``: timestamp when restart should occur;
|
||||
@@ -71,7 +71,7 @@ class Patroni(AbstractPatroniDaemon):
|
||||
self.api = RestApiServer(self, self.config['restapi'])
|
||||
self.ha = Ha(self)
|
||||
|
||||
self.tags = self.get_tags()
|
||||
self._tags = self._get_tags()
|
||||
self.next_run = time.time()
|
||||
self.scheduled_restart: Dict[str, Any] = {}
|
||||
|
||||
@@ -121,33 +121,12 @@ class Patroni(AbstractPatroniDaemon):
|
||||
except Exception:
|
||||
return
|
||||
|
||||
def get_tags(self) -> Dict[str, Any]:
|
||||
def _get_tags(self) -> Dict[str, Any]:
|
||||
"""Get tags configured for this node, if any.
|
||||
|
||||
Handle both predefined Patroni tags and custom defined tags.
|
||||
|
||||
.. note::
|
||||
A custom tag is any tag added to the configuration ``tags`` section that is not one of ``clonefrom``,
|
||||
``nofailover``, ``noloadbalance`` or ``nosync``.
|
||||
|
||||
For the Patroni predefined tags, the returning object will only contain them if they are enabled as they
|
||||
all are boolean values that default to disabled.
|
||||
|
||||
:returns: a dictionary of tags set for this node. The key is the tag name, and the value is the corresponding
|
||||
tag value.
|
||||
:returns: a dictionary of tags set for this node.
|
||||
"""
|
||||
return {tag: value for tag, value in self.config.get('tags', {}).items()
|
||||
if tag not in ('clonefrom', 'nofailover', 'noloadbalance', 'nosync') or value}
|
||||
|
||||
@property
|
||||
def nofailover(self) -> bool:
|
||||
"""``True`` if ``tags.nofailover`` configuration is enabled for this node, else ``False``."""
|
||||
return bool(self.tags.get('nofailover', False))
|
||||
|
||||
@property
|
||||
def nosync(self) -> bool:
|
||||
"""``True`` if ``tags.nosync`` configuration is enabled for this node, else ``False``."""
|
||||
return bool(self.tags.get('nosync', False))
|
||||
return self._filter_tags(self.config.get('tags', {}))
|
||||
|
||||
def reload_config(self, sighup: bool = False, local: Optional[bool] = False) -> None:
|
||||
"""Apply new configuration values for ``patroni`` daemon.
|
||||
@@ -166,7 +145,7 @@ class Patroni(AbstractPatroniDaemon):
|
||||
try:
|
||||
super(Patroni, self).reload_config(sighup, local)
|
||||
if local:
|
||||
self.tags = self.get_tags()
|
||||
self._tags = self._get_tags()
|
||||
self.request.reload_config(self.config)
|
||||
if local or sighup and self.api.reload_local_certificate():
|
||||
self.api.reload_config(self.config['restapi'])
|
||||
@@ -177,14 +156,9 @@ class Patroni(AbstractPatroniDaemon):
|
||||
logger.exception('Failed to reload config_file=%s', self.config.config_file)
|
||||
|
||||
@property
|
||||
def replicatefrom(self) -> Optional[str]:
|
||||
"""Value of ``tags.replicatefrom`` configuration, if any."""
|
||||
return self.tags.get('replicatefrom')
|
||||
|
||||
@property
|
||||
def noloadbalance(self) -> bool:
|
||||
"""``True`` if ``tags.noloadbalance`` configuration is enabled for this node, else ``False``."""
|
||||
return bool(self.tags.get('noloadbalance', False))
|
||||
def tags(self) -> Dict[str, Any]:
|
||||
"""Tags configured for this node, if any."""
|
||||
return self._tags
|
||||
|
||||
def schedule_next_run(self) -> None:
|
||||
"""Schedule the next run of the ``patroni`` daemon main loop.
|
||||
|
||||
+69
-133
@@ -12,7 +12,6 @@ import json
|
||||
import logging
|
||||
import time
|
||||
import traceback
|
||||
import dateutil.parser
|
||||
import datetime
|
||||
import os
|
||||
import socket
|
||||
@@ -28,11 +27,11 @@ from typing import Any, Callable, Dict, Iterator, List, Optional, Tuple, TYPE_CH
|
||||
|
||||
from . import psycopg
|
||||
from .__main__ import Patroni
|
||||
from .dcs import Cluster
|
||||
from .exceptions import PostgresConnectionException, PostgresException
|
||||
from .manual_failover import ManualFailover
|
||||
from .postgresql.misc import postgres_version_to_int
|
||||
from .utils import deep_compare, enable_keepalive, parse_bool, patch_config, Retry, \
|
||||
RetryFailedError, parse_int, split_host_port, tzutc, uri, cluster_as_json
|
||||
RetryFailedError, parse_int, parse_schedule, split_host_port, tzutc, uri, cluster_as_json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -198,7 +197,11 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
response['database_system_identifier'] = patroni.postgresql.sysid
|
||||
if patroni.postgresql.pending_restart:
|
||||
response['pending_restart'] = True
|
||||
response['patroni'] = {'version': patroni.version, 'scope': patroni.postgresql.scope}
|
||||
response['patroni'] = {
|
||||
'version': patroni.version,
|
||||
'scope': patroni.postgresql.scope,
|
||||
'name': patroni.postgresql.name
|
||||
}
|
||||
if patroni.scheduled_restart:
|
||||
response['scheduled_restart'] = patroni.scheduled_restart.copy()
|
||||
del response['scheduled_restart']['postmaster_start_time']
|
||||
@@ -449,7 +452,10 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
"""
|
||||
cluster = self.server.patroni.dcs.get_cluster(True)
|
||||
global_config = self.server.patroni.config.get_global_config(cluster)
|
||||
self._write_json_response(200, cluster_as_json(cluster, global_config))
|
||||
|
||||
response = cluster_as_json(cluster, global_config)
|
||||
response['scope'] = self.server.patroni.postgresql.scope
|
||||
self._write_json_response(200, response)
|
||||
|
||||
def do_GET_history(self) -> None:
|
||||
"""Handle a ``GET`` request to ``/history`` path.
|
||||
@@ -526,113 +532,113 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
|
||||
metrics: List[str] = []
|
||||
|
||||
scope_label = '{{scope="{0}"}}'.format(patroni.postgresql.scope)
|
||||
labels = f'{{scope="{patroni.postgresql.scope}",name="{patroni.postgresql.name}"}}'
|
||||
metrics.append("# HELP patroni_version Patroni semver without periods.")
|
||||
metrics.append("# TYPE patroni_version gauge")
|
||||
padded_semver = ''.join([x.zfill(2) for x in patroni.version.split('.')]) # 2.0.2 => 020002
|
||||
metrics.append("patroni_version{0} {1}".format(scope_label, padded_semver))
|
||||
metrics.append("patroni_version{0} {1}".format(labels, padded_semver))
|
||||
|
||||
metrics.append("# HELP patroni_postgres_running Value is 1 if Postgres is running, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_postgres_running gauge")
|
||||
metrics.append("patroni_postgres_running{0} {1}".format(scope_label, int(postgres['state'] == 'running')))
|
||||
metrics.append("patroni_postgres_running{0} {1}".format(labels, int(postgres['state'] == 'running')))
|
||||
|
||||
metrics.append("# HELP patroni_postmaster_start_time Epoch seconds since Postgres started.")
|
||||
metrics.append("# TYPE patroni_postmaster_start_time gauge")
|
||||
postmaster_start_time = postgres.get('postmaster_start_time')
|
||||
postmaster_start_time = (postmaster_start_time - epoch).total_seconds() if postmaster_start_time else 0
|
||||
metrics.append("patroni_postmaster_start_time{0} {1}".format(scope_label, postmaster_start_time))
|
||||
metrics.append("patroni_postmaster_start_time{0} {1}".format(labels, postmaster_start_time))
|
||||
|
||||
metrics.append("# HELP patroni_master Value is 1 if this node is the leader, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_master gauge")
|
||||
metrics.append("patroni_master{0} {1}".format(scope_label, int(postgres['role'] in ('master', 'primary'))))
|
||||
metrics.append("patroni_master{0} {1}".format(labels, int(postgres['role'] in ('master', 'primary'))))
|
||||
|
||||
metrics.append("# HELP patroni_primary Value is 1 if this node is the leader, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_primary gauge")
|
||||
metrics.append("patroni_primary{0} {1}".format(scope_label, int(postgres['role'] in ('master', 'primary'))))
|
||||
metrics.append("patroni_primary{0} {1}".format(labels, int(postgres['role'] in ('master', 'primary'))))
|
||||
|
||||
metrics.append("# HELP patroni_xlog_location Current location of the Postgres"
|
||||
" transaction log, 0 if this node is not the leader.")
|
||||
metrics.append("# TYPE patroni_xlog_location counter")
|
||||
metrics.append("patroni_xlog_location{0} {1}".format(scope_label, postgres.get('xlog', {}).get('location', 0)))
|
||||
metrics.append("patroni_xlog_location{0} {1}".format(labels, postgres.get('xlog', {}).get('location', 0)))
|
||||
|
||||
metrics.append("# HELP patroni_standby_leader Value is 1 if this node is the standby_leader, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_standby_leader gauge")
|
||||
metrics.append("patroni_standby_leader{0} {1}".format(scope_label, int(postgres['role'] == 'standby_leader')))
|
||||
metrics.append("patroni_standby_leader{0} {1}".format(labels, int(postgres['role'] == 'standby_leader')))
|
||||
|
||||
metrics.append("# HELP patroni_replica Value is 1 if this node is a replica, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_replica gauge")
|
||||
metrics.append("patroni_replica{0} {1}".format(scope_label, int(postgres['role'] == 'replica')))
|
||||
metrics.append("patroni_replica{0} {1}".format(labels, int(postgres['role'] == 'replica')))
|
||||
|
||||
metrics.append("# HELP patroni_sync_standby Value is 1 if this node is a sync standby replica, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_sync_standby gauge")
|
||||
metrics.append("patroni_sync_standby{0} {1}".format(scope_label, int(postgres.get('sync_standby', False))))
|
||||
metrics.append("patroni_sync_standby{0} {1}".format(labels, int(postgres.get('sync_standby', False))))
|
||||
|
||||
metrics.append("# HELP patroni_xlog_received_location Current location of the received"
|
||||
" Postgres transaction log, 0 if this node is not a replica.")
|
||||
metrics.append("# TYPE patroni_xlog_received_location counter")
|
||||
metrics.append("patroni_xlog_received_location{0} {1}"
|
||||
.format(scope_label, postgres.get('xlog', {}).get('received_location', 0)))
|
||||
.format(labels, postgres.get('xlog', {}).get('received_location', 0)))
|
||||
|
||||
metrics.append("# HELP patroni_xlog_replayed_location Current location of the replayed"
|
||||
" Postgres transaction log, 0 if this node is not a replica.")
|
||||
metrics.append("# TYPE patroni_xlog_replayed_location counter")
|
||||
metrics.append("patroni_xlog_replayed_location{0} {1}"
|
||||
.format(scope_label, postgres.get('xlog', {}).get('replayed_location', 0)))
|
||||
.format(labels, postgres.get('xlog', {}).get('replayed_location', 0)))
|
||||
|
||||
metrics.append("# HELP patroni_xlog_replayed_timestamp Current timestamp of the replayed"
|
||||
" Postgres transaction log, 0 if null.")
|
||||
metrics.append("# TYPE patroni_xlog_replayed_timestamp gauge")
|
||||
replayed_timestamp = postgres.get('xlog', {}).get('replayed_timestamp')
|
||||
replayed_timestamp = (replayed_timestamp - epoch).total_seconds() if replayed_timestamp else 0
|
||||
metrics.append("patroni_xlog_replayed_timestamp{0} {1}".format(scope_label, replayed_timestamp))
|
||||
metrics.append("patroni_xlog_replayed_timestamp{0} {1}".format(labels, replayed_timestamp))
|
||||
|
||||
metrics.append("# HELP patroni_xlog_paused Value is 1 if the Postgres xlog is paused, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_xlog_paused gauge")
|
||||
metrics.append("patroni_xlog_paused{0} {1}"
|
||||
.format(scope_label, int(postgres.get('xlog', {}).get('paused', False) is True)))
|
||||
.format(labels, int(postgres.get('xlog', {}).get('paused', False) is True)))
|
||||
|
||||
if postgres.get('server_version', 0) >= 90600:
|
||||
metrics.append("# HELP patroni_postgres_streaming Value is 1 if Postgres is streaming, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_postgres_streaming gauge")
|
||||
metrics.append("patroni_postgres_streaming{0} {1}"
|
||||
.format(scope_label, int(postgres.get('replication_state') == 'streaming')))
|
||||
.format(labels, int(postgres.get('replication_state') == 'streaming')))
|
||||
|
||||
metrics.append("# HELP patroni_postgres_in_archive_recovery Value is 1"
|
||||
" if Postgres is replicating from archive, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_postgres_in_archive_recovery gauge")
|
||||
metrics.append("patroni_postgres_in_archive_recovery{0} {1}"
|
||||
.format(scope_label, int(postgres.get('replication_state') == 'in archive recovery')))
|
||||
.format(labels, int(postgres.get('replication_state') == 'in archive recovery')))
|
||||
|
||||
metrics.append("# HELP patroni_postgres_server_version Version of Postgres (if running), 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_postgres_server_version gauge")
|
||||
metrics.append("patroni_postgres_server_version {0} {1}".format(scope_label, postgres.get('server_version', 0)))
|
||||
metrics.append("patroni_postgres_server_version {0} {1}".format(labels, postgres.get('server_version', 0)))
|
||||
|
||||
metrics.append("# HELP patroni_cluster_unlocked Value is 1 if the cluster is unlocked, 0 if locked.")
|
||||
metrics.append("# TYPE patroni_cluster_unlocked gauge")
|
||||
metrics.append("patroni_cluster_unlocked{0} {1}".format(scope_label, int(postgres.get('cluster_unlocked', 0))))
|
||||
metrics.append("patroni_cluster_unlocked{0} {1}".format(labels, int(postgres.get('cluster_unlocked', 0))))
|
||||
|
||||
metrics.append("# HELP patroni_failsafe_mode_is_active Value is 1 if failsafe mode is active, 0 if inactive.")
|
||||
metrics.append("# TYPE patroni_failsafe_mode_is_active gauge")
|
||||
metrics.append("patroni_failsafe_mode_is_active{0} {1}"
|
||||
.format(scope_label, int(postgres.get('failsafe_mode_is_active', 0))))
|
||||
.format(labels, int(postgres.get('failsafe_mode_is_active', 0))))
|
||||
|
||||
metrics.append("# HELP patroni_postgres_timeline Postgres timeline of this node (if running), 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_postgres_timeline counter")
|
||||
metrics.append("patroni_postgres_timeline{0} {1}".format(scope_label, postgres.get('timeline', 0)))
|
||||
metrics.append("patroni_postgres_timeline{0} {1}".format(labels, postgres.get('timeline', 0)))
|
||||
|
||||
metrics.append("# HELP patroni_dcs_last_seen Epoch timestamp when DCS was last contacted successfully"
|
||||
" by Patroni.")
|
||||
metrics.append("# TYPE patroni_dcs_last_seen gauge")
|
||||
metrics.append("patroni_dcs_last_seen{0} {1}".format(scope_label, postgres.get('dcs_last_seen', 0)))
|
||||
metrics.append("patroni_dcs_last_seen{0} {1}".format(labels, postgres.get('dcs_last_seen', 0)))
|
||||
|
||||
metrics.append("# HELP patroni_pending_restart Value is 1 if the node needs a restart, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_pending_restart gauge")
|
||||
metrics.append("patroni_pending_restart{0} {1}"
|
||||
.format(scope_label, int(patroni.postgresql.pending_restart)))
|
||||
.format(labels, int(patroni.postgresql.pending_restart)))
|
||||
|
||||
metrics.append("# HELP patroni_is_paused Value is 1 if auto failover is disabled, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_is_paused gauge")
|
||||
metrics.append("patroni_is_paused{0} {1}".format(scope_label, int(postgres.get('pause', 0))))
|
||||
metrics.append("patroni_is_paused{0} {1}".format(labels, int(postgres.get('pause', 0))))
|
||||
|
||||
self.write_response(200, '\n'.join(metrics) + '\n', content_type='text/plain')
|
||||
|
||||
@@ -770,44 +776,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
self.server.patroni.api_sigterm()
|
||||
self.write_response(202, 'shutdown scheduled')
|
||||
|
||||
@staticmethod
|
||||
def parse_schedule(schedule: str,
|
||||
action: str) -> Tuple[Union[int, None], Union[str, None], Union[datetime.datetime, None]]:
|
||||
"""Parse the given *schedule* and validate it.
|
||||
|
||||
:param schedule: a string representing a timestamp, e.g. ``2023-04-14T20:27:00+00:00``.
|
||||
:param action: the action to be scheduled (``restart``, ``switchover``, or ``failover``).
|
||||
|
||||
:returns: a tuple composed of 3 items:
|
||||
|
||||
* Suggested HTTP status code for a response:
|
||||
|
||||
* ``None``: if no issue was faced while parsing, leaving it up to the caller to decide the status; or
|
||||
* ``400``: if no timezone information could be found in *schedule*; or
|
||||
* ``422``: if *schedule* is invalid -- in the past or not parsable.
|
||||
|
||||
* An error message, if any error is faced, otherwise ``None``;
|
||||
* Parsed *schedule*, if able to parse, otherwise ``None``.
|
||||
|
||||
"""
|
||||
error = None
|
||||
scheduled_at = None
|
||||
try:
|
||||
scheduled_at = dateutil.parser.parse(schedule)
|
||||
if scheduled_at.tzinfo is None:
|
||||
error = 'Timezone information is mandatory for the scheduled {0}'.format(action)
|
||||
status_code = 400
|
||||
elif scheduled_at < datetime.datetime.now(tzutc):
|
||||
error = 'Cannot schedule {0} in the past'.format(action)
|
||||
status_code = 422
|
||||
else:
|
||||
status_code = None
|
||||
except (ValueError, TypeError):
|
||||
logger.exception('Invalid scheduled %s time: %s', action, schedule)
|
||||
error = 'Unable to parse scheduled timestamp. It should be in an unambiguous format, e.g. ISO 8601'
|
||||
status_code = 422
|
||||
return status_code, error, scheduled_at
|
||||
|
||||
@check_access
|
||||
def do_POST_restart(self) -> None:
|
||||
"""Handle a ``POST`` request to ``/restart`` path.
|
||||
@@ -863,9 +831,9 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
|
||||
for k in request:
|
||||
if k == 'schedule':
|
||||
(_, data, request[k]) = self.parse_schedule(request[k], "restart")
|
||||
if _:
|
||||
status_code = _
|
||||
parse_result, request[k] = parse_schedule(request[k])
|
||||
if parse_result:
|
||||
data, status_code = parse_result.value[0], parse_result.value[1]
|
||||
break
|
||||
elif k == 'role':
|
||||
if request[k] not in ('master', 'primary', 'replica'):
|
||||
@@ -1015,39 +983,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
logger.debug('Exception occurred during polling %s result: %s', action, e)
|
||||
return 503, action.title() + ' status unknown'
|
||||
|
||||
def is_failover_possible(self, cluster: Cluster, leader: Optional[str], candidate: Optional[str],
|
||||
action: str) -> Optional[str]:
|
||||
"""Checks whether there are nodes that could take over after demoting the primary.
|
||||
|
||||
:param cluster: the Patroni cluster.
|
||||
:param leader: name of the current Patroni leader.
|
||||
:param candidate: name of the Patroni node to be promoted.
|
||||
:param action: the action to be performed (``switchover`` or ``failover``).
|
||||
|
||||
:returns: a string with the error message or ``None`` if good nodes are found.
|
||||
"""
|
||||
is_synchronous_mode = self.server.patroni.config.get_global_config(cluster).is_synchronous_mode
|
||||
if leader and (not cluster.leader or cluster.leader.name != leader):
|
||||
return 'leader name does not match'
|
||||
if candidate:
|
||||
if action == 'switchover' and is_synchronous_mode and not cluster.sync.matches(candidate):
|
||||
return 'candidate name does not match with sync_standby'
|
||||
members = [m for m in cluster.members if m.name == candidate]
|
||||
if not members:
|
||||
return 'candidate does not exists'
|
||||
elif is_synchronous_mode:
|
||||
members = [m for m in cluster.members if cluster.sync.matches(m.name)]
|
||||
if not members:
|
||||
return action + ' is not possible: can not find sync_standby'
|
||||
else:
|
||||
members = [m for m in cluster.members if not cluster.leader or m.name != cluster.leader.name and m.api_url]
|
||||
if not members:
|
||||
return action + ' is not possible: cluster does not have members except leader'
|
||||
for st in self.server.patroni.ha.fetch_nodes_statuses(members):
|
||||
if st.failover_limitation() is None:
|
||||
return None
|
||||
return action + ' is not possible: no good candidates have been found'
|
||||
|
||||
@check_access
|
||||
def do_POST_failover(self, action: str = 'failover') -> None:
|
||||
"""Handle a ``POST`` request to ``/failover`` path.
|
||||
@@ -1067,7 +1002,8 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
* ``412``: if operation is not possible;
|
||||
* ``503``: if unable to register the operation to the DCS;
|
||||
* HTTP status returned by :func:`parse_schedule`, if any error was observed while parsing the schedule;
|
||||
* HTTP status returned by :func:`poll_failover_result` if the operation has been processed immediately.
|
||||
* HTTP status returned by :func:`poll_failover_result` if the operation has been processed immediately;
|
||||
* ``400``: if none of the above applies.
|
||||
|
||||
.. note::
|
||||
If unable to parse the request body, then the request is silently discarded.
|
||||
@@ -1075,7 +1011,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
:param action: the action to be performed (``switchover`` or ``failover``).
|
||||
"""
|
||||
request = self._read_json_content()
|
||||
(status_code, data) = (400, '')
|
||||
if not request:
|
||||
return
|
||||
|
||||
@@ -1088,26 +1023,15 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
logger.info("received %s request with leader=%s candidate=%s scheduled_at=%s",
|
||||
action, leader, candidate, scheduled_at)
|
||||
|
||||
if action == 'failover' and not candidate:
|
||||
data = 'Failover could be performed only to a specific candidate'
|
||||
elif action == 'switchover' and not leader:
|
||||
data = 'Switchover could be performed only from a specific leader'
|
||||
manual_failover = ManualFailover(action, cluster, leader, candidate, scheduled_at,
|
||||
global_config.is_paused, global_config.is_synchronous_mode,
|
||||
self.server.patroni)
|
||||
data, status_code = manual_failover.run_precheck().value
|
||||
|
||||
if not data and scheduled_at:
|
||||
if not leader:
|
||||
data = 'Scheduled {0} is possible only from a specific leader'.format(action)
|
||||
if not data and global_config.is_paused:
|
||||
data = "Can't schedule {0} in the paused state".format(action)
|
||||
if not data:
|
||||
(status_code, data, scheduled_at) = self.parse_schedule(scheduled_at, action)
|
||||
|
||||
if not data and global_config.is_paused and not candidate:
|
||||
data = action.title() + ' is possible only to a specific candidate in a paused state'
|
||||
|
||||
if not data and not scheduled_at:
|
||||
data = self.is_failover_possible(cluster, leader, candidate, action)
|
||||
if data:
|
||||
status_code = 412
|
||||
parse_result, scheduled_at = manual_failover.parse_scheduled()
|
||||
if parse_result:
|
||||
data, status_code = parse_result.value[0], parse_result.value[1]
|
||||
|
||||
if not data:
|
||||
if self.server.patroni.dcs.manual_failover(leader, candidate, scheduled_at=scheduled_at):
|
||||
@@ -1119,14 +1043,12 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
status_code, data = self.poll_failover_result(cluster.leader and cluster.leader.name,
|
||||
candidate, action)
|
||||
else:
|
||||
data = 'failed to write {0} key into DCS'.format(action)
|
||||
data = 'failed to write failover key into DCS'
|
||||
status_code = 503
|
||||
# pyright thinks ``status_code`` can be ``None`` because ``parse_schedule`` call may return ``None``. However,
|
||||
# if that's the case, ``status_code`` will be overwritten somewhere between ``parse_schedule`` and
|
||||
# ``write_response`` calls.
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
assert isinstance(status_code, int)
|
||||
self.write_response(status_code, data)
|
||||
|
||||
status_code = status_code or 400
|
||||
self.write_response(status_code, data.format(action=action, leader=leader, candidate=candidate,
|
||||
cluster_name=self.server.patroni.postgresql.scope))
|
||||
|
||||
def do_POST_switchover(self) -> None:
|
||||
"""Handle a ``POST`` request to ``/switchover`` path.
|
||||
@@ -1368,6 +1290,9 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
||||
def query(self, sql: str, *params: Any) -> List[Tuple[Any, ...]]:
|
||||
"""Execute *sql* query with *params* and optionally return results.
|
||||
|
||||
.. note::
|
||||
Prefer to use own connection to postgres and fallback to ``heartbeat`` when own isn't available.
|
||||
|
||||
:param sql: the SQL statement to be run.
|
||||
:param params: positional arguments to be used as parameters for *sql*.
|
||||
|
||||
@@ -1377,10 +1302,21 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
||||
:class:`psycopg.Error`: if had issues while executing *sql*.
|
||||
:class:`~patroni.exceptions.PostgresConnectionException`: if had issues while connecting to the database.
|
||||
"""
|
||||
# We first try to get a heartbeat connection because it is always required for the main thread.
|
||||
try:
|
||||
return self.patroni.postgresql.query(sql, *params, retry=False)
|
||||
except RetryFailedError as e:
|
||||
raise PostgresConnectionException(str(e))
|
||||
heartbeat_connection = self.patroni.postgresql.connection_pool.get('heartbeat')
|
||||
heartbeat_connection.get() # try to open psycopg connection to postgres
|
||||
except psycopg.Error as exc:
|
||||
raise PostgresConnectionException('connection problems') from exc
|
||||
|
||||
try:
|
||||
connection = self.patroni.postgresql.connection_pool.get('restapi')
|
||||
connection.get() # try to open psycopg connection to postgres
|
||||
except psycopg.Error:
|
||||
logger.debug('restapi connection to postgres is not available')
|
||||
connection = heartbeat_connection
|
||||
|
||||
return connection.query(sql, *params)
|
||||
|
||||
@staticmethod
|
||||
def _set_fd_cloexec(fd: socket.socket) -> None:
|
||||
@@ -1546,7 +1482,7 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
||||
:param listen: IP and port to bind REST API to. It should be a string in the format ``host:port``, where
|
||||
``host`` can be a hostname or IP address. It is the value of ``restapi.listen`` setting.
|
||||
:param ssl_options: dictionary that may contain the following keys, depending on what has been configured in
|
||||
``restapi` section:
|
||||
``restapi`` section:
|
||||
|
||||
* ``certfile``: path to PEM certificate. If given, will start in HTTPS mode;
|
||||
* ``keyfile``: path to key of ``certfile``;
|
||||
|
||||
+1
-1
@@ -257,7 +257,7 @@ class Config(object):
|
||||
|
||||
* file or directory path passed as command-line argument (*configfile*), if it exists and the file or
|
||||
files found in the directory can be parsed (see :meth:`~Config._load_config_path`), otherwise
|
||||
* YAML file passed via the environment variable (see :cvar:`PATRONI_CONFIG_VARIABLE`), if the referenced
|
||||
* YAML file passed via the environment variable (see :attr:`PATRONI_CONFIG_VARIABLE`), if the referenced
|
||||
file exists and can be parsed, otherwise
|
||||
* from configuration values defined as environment variables, see
|
||||
:meth:`~Config._build_environment_configuration`.
|
||||
|
||||
+87
-101
@@ -16,8 +16,6 @@ import click
|
||||
import codecs
|
||||
import copy
|
||||
import datetime
|
||||
import dateutil.parser
|
||||
import dateutil.tz
|
||||
import difflib
|
||||
import io
|
||||
import json
|
||||
@@ -46,10 +44,12 @@ try:
|
||||
except ImportError: # pragma: no cover
|
||||
from cdiff import markup_to_pager, PatchStream # pyright: ignore [reportMissingModuleSource]
|
||||
|
||||
from .config import Config, get_global_config
|
||||
from .dcs import get_dcs as _get_dcs, AbstractDCS, Cluster, Member
|
||||
from .exceptions import PatroniException
|
||||
from .manual_failover import ManualFailover
|
||||
from .postgresql.misc import postgres_version_to_int
|
||||
from .utils import cluster_as_json, patch_config, polling_loop
|
||||
from .utils import cluster_as_json, parse_schedule, patch_config, polling_loop
|
||||
from .request import PatroniRequest
|
||||
from .version import __version__
|
||||
|
||||
@@ -225,8 +225,6 @@ def load_config(path: str, dcs_url: Optional[str]) -> Dict[str, Any]:
|
||||
:raises:
|
||||
:class:`PatroniCtlException`: if *path* does not exist or is not readable.
|
||||
"""
|
||||
from patroni.config import Config
|
||||
|
||||
if not (os.path.exists(path) and os.access(path, os.R_OK)):
|
||||
if path != CONFIG_FILE_PATH: # bail if non-default config location specified but file not found / readable
|
||||
raise PatroniCtlException('Provided config file {0} not existing or no read rights.'
|
||||
@@ -264,7 +262,9 @@ role_choice = click.Choice(['leader', 'primary', 'standby-leader', 'replica', 's
|
||||
@click.option('-k', '--insecure', is_flag=True, help='Allow connections to SSL sites without certs')
|
||||
@click.pass_context
|
||||
def ctl(ctx: click.Context, config_file: str, dcs_url: Optional[str], insecure: bool) -> None:
|
||||
"""Entry point of ``patronictl`` utility.
|
||||
"""Command-line interface for interacting with Patroni.
|
||||
\f
|
||||
Entry point of ``patronictl`` utility.
|
||||
|
||||
Load the configuration file.
|
||||
|
||||
@@ -644,7 +644,8 @@ def get_members(obj: Dict[str, Any], cluster: Cluster, cluster_name: str, member
|
||||
if member_names:
|
||||
member_names = list(set(member_names) & candidates)
|
||||
if not member_names:
|
||||
raise PatroniCtlException('No {0} among provided members'.format(role))
|
||||
raise PatroniCtlException(
|
||||
'No{0} among provided members'.format('t a single cluster member' if role == 'any' else ' ' + role))
|
||||
elif action != 'reinitialize':
|
||||
member_names = list(candidates)
|
||||
|
||||
@@ -944,43 +945,6 @@ def check_response(response: urllib3.response.HTTPResponse, member_name: str,
|
||||
return True
|
||||
|
||||
|
||||
def parse_scheduled(scheduled: Optional[str]) -> Optional[datetime.datetime]:
|
||||
"""Parse a string *scheduled* timestamp as a :class:`~datetime.datetime` object.
|
||||
|
||||
:param scheduled: string representation of the timestamp. May also be ``now``.
|
||||
|
||||
:returns: the corresponding :class:`~datetime.datetime` object, if *scheduled* is not ``now``, otherwise ``None``.
|
||||
|
||||
:raises:
|
||||
:class:`PatroniCtlException`: if unable to parse *scheduled* from :class:`str` to :class:`~datetime.datetime`.
|
||||
|
||||
:Example:
|
||||
|
||||
>>> parse_scheduled(None) is None
|
||||
True
|
||||
|
||||
>>> parse_scheduled('now') is None
|
||||
True
|
||||
|
||||
>>> parse_scheduled('2023-05-29T04:32:31')
|
||||
datetime.datetime(2023, 5, 29, 4, 32, 31, tzinfo=tzlocal())
|
||||
|
||||
>>> parse_scheduled('2023-05-29T04:32:31-3')
|
||||
datetime.datetime(2023, 5, 29, 4, 32, 31, tzinfo=tzoffset(None, -10800))
|
||||
"""
|
||||
if scheduled is not None and (scheduled or 'now') != 'now':
|
||||
try:
|
||||
scheduled_at = dateutil.parser.parse(scheduled)
|
||||
if scheduled_at.tzinfo is None:
|
||||
scheduled_at = scheduled_at.replace(tzinfo=dateutil.tz.tzlocal())
|
||||
except (ValueError, TypeError):
|
||||
message = 'Unable to parse scheduled timestamp ({0}). It should be in an unambiguous format (e.g. ISO 8601)'
|
||||
raise PatroniCtlException(message.format(scheduled))
|
||||
return scheduled_at
|
||||
|
||||
return None
|
||||
|
||||
|
||||
@ctl.command('reload', help='Reload cluster member configuration')
|
||||
@click.argument('cluster_name')
|
||||
@click.argument('member_names', nargs=-1)
|
||||
@@ -1011,7 +975,6 @@ def reload(obj: Dict[str, Any], cluster_name: str, member_names: List[str],
|
||||
if r.status == 200:
|
||||
click.echo('No changes to apply on member {0}'.format(member.name))
|
||||
elif r.status == 202:
|
||||
from patroni.config import get_global_config
|
||||
config = get_global_config(cluster)
|
||||
click.echo('Reload request received for member {0} and will be processed within {1} seconds'.format(
|
||||
member.name, config.get('loop_wait') or dcs.loop_wait)
|
||||
@@ -1061,16 +1024,20 @@ def restart(obj: Dict[str, Any], cluster_name: str, group: Optional[int], member
|
||||
* *version* could not be parsed; or
|
||||
* a restart is attempted against a cluster that is in maintenance mode.
|
||||
"""
|
||||
action = 'restart'
|
||||
cluster = get_dcs(obj, cluster_name, group).get_cluster()
|
||||
|
||||
members = get_members(obj, cluster, cluster_name, member_names, role, force, 'restart', False, group=group)
|
||||
members = get_members(obj, cluster, cluster_name, member_names, role, force, action, False, group=group)
|
||||
if scheduled is None and not force:
|
||||
next_hour = (datetime.datetime.now() + datetime.timedelta(hours=1)).strftime('%Y-%m-%dT%H:%M')
|
||||
next_hour = (datetime.datetime.now() + datetime.timedelta(hours=1)).strftime('%Y-%m-%dT%H:%M+00')
|
||||
scheduled = click.prompt('When should the restart take place (e.g. ' + next_hour + ') ',
|
||||
type=str, default='now')
|
||||
scheduled = scheduled if scheduled != 'now' else None
|
||||
|
||||
scheduled_at = parse_scheduled(scheduled)
|
||||
confirm_members_action(members, force, 'restart', scheduled_at)
|
||||
parse_result, scheduled_at = parse_schedule(scheduled)
|
||||
if parse_result:
|
||||
raise PatroniCtlException(parse_result.value[0].format(action=action))
|
||||
confirm_members_action(members, force, action, scheduled_at)
|
||||
|
||||
if p_any:
|
||||
random.shuffle(members)
|
||||
@@ -1093,7 +1060,6 @@ def restart(obj: Dict[str, Any], cluster_name: str, group: Optional[int], member
|
||||
content['postgres_version'] = version
|
||||
|
||||
if scheduled_at:
|
||||
from patroni.config import get_global_config
|
||||
if get_global_config(cluster).is_paused:
|
||||
raise PatroniCtlException("Can't schedule restart in the paused state")
|
||||
content['schedule'] = scheduled_at.isoformat()
|
||||
@@ -1213,6 +1179,9 @@ def _do_failover_or_switchover(obj: Dict[str, Any], action: str, cluster_name: s
|
||||
click.echo('Current cluster topology')
|
||||
output_members(obj, cluster, cluster_name, group=group)
|
||||
|
||||
# Define everything missing via interactive input or available cluster info (if force mode)
|
||||
|
||||
# Require Citus group
|
||||
if obj.get('citus') and group is None:
|
||||
if force:
|
||||
raise PatroniCtlException('For Citus clusters the --group must me specified')
|
||||
@@ -1221,72 +1190,82 @@ def _do_failover_or_switchover(obj: Dict[str, Any], action: str, cluster_name: s
|
||||
dcs = get_dcs(obj, cluster_name, group)
|
||||
cluster = dcs.get_cluster()
|
||||
|
||||
if action == 'switchover' and (cluster.leader is None or not cluster.leader.name):
|
||||
raise PatroniCtlException('This cluster has no leader')
|
||||
global_config = get_global_config(cluster)
|
||||
|
||||
if leader is None:
|
||||
if force or action == 'failover':
|
||||
leader = cluster.leader and cluster.leader.name
|
||||
# Leader is required for switchover only
|
||||
if action == 'switchover' and leader is None:
|
||||
if cluster.leader is None or not cluster.leader.name:
|
||||
raise PatroniCtlException('This cluster has no leader')
|
||||
if force:
|
||||
leader = cluster.leader.name
|
||||
else:
|
||||
from patroni.config import get_global_config
|
||||
prompt = 'Standby Leader' if get_global_config(cluster).is_standby_cluster else 'Primary'
|
||||
leader = click.prompt(prompt, type=str, default=(cluster.leader and cluster.leader.member.name))
|
||||
|
||||
if leader is not None and cluster.leader and cluster.leader.member.name != leader:
|
||||
raise PatroniCtlException('Member {0} is not the leader of cluster {1}'.format(leader, cluster_name))
|
||||
|
||||
# excluding members with nofailover tag
|
||||
candidate_names = [str(m.name) for m in cluster.members if m.name != leader and not m.nofailover]
|
||||
# We sort the names for consistent output to the client
|
||||
candidate_names.sort()
|
||||
|
||||
if not candidate_names:
|
||||
raise PatroniCtlException('No candidates found to {0} to'.format(action))
|
||||
prompt = 'Standby Leader' if global_config.is_standby_cluster else 'Primary'
|
||||
leader = click.prompt(prompt, type=str, default=(cluster.leader and cluster.leader.name))
|
||||
|
||||
if candidate is None and not force:
|
||||
# Check if there are any candidates available at all
|
||||
candidate_names = [str(m.name) for m in cluster.members if m.name != leader and not m.nofailover]
|
||||
if not candidate_names:
|
||||
raise PatroniCtlException('No candidates found to {0} to'.format(action))
|
||||
candidate_names.sort() # we sort the names for consistent output to the client
|
||||
candidate = click.prompt('Candidate ' + str(candidate_names), type=str, default='')
|
||||
|
||||
if action == 'failover' and not candidate:
|
||||
raise PatroniCtlException('Failover could be performed only to a specific candidate')
|
||||
# We allow manual failover to an aync node in the sync mode, so we better ask for the confirmation
|
||||
if all((not force,
|
||||
action == 'failover',
|
||||
global_config.is_synchronous_mode,
|
||||
not cluster.sync.is_empty,
|
||||
not cluster.sync.matches(candidate, True))):
|
||||
if click.confirm(f'Are you sure you want to failover to the asynchronous node {candidate}'):
|
||||
raise PatroniCtlException('Aborting ' + action)
|
||||
|
||||
if candidate == leader:
|
||||
raise PatroniCtlException(action.title() + ' target and source are the same.')
|
||||
if action == 'switchover' and scheduled is None and not force:
|
||||
next_hour = (datetime.datetime.now() + datetime.timedelta(hours=1)).strftime('%Y-%m-%dT%H:%M+00')
|
||||
scheduled = click.prompt('When should the switchover take place (e.g. ' + next_hour + ') ',
|
||||
type=str, default='now')
|
||||
scheduled = scheduled if scheduled != 'now' else None
|
||||
|
||||
if candidate and candidate not in candidate_names:
|
||||
raise PatroniCtlException('Member {0} does not exist in cluster {1}'.format(candidate, cluster_name))
|
||||
# Now, when we collected all the possible info, run checks
|
||||
manual_failover = ManualFailover(action, cluster, leader, candidate, scheduled,
|
||||
global_config.is_paused, global_config.is_synchronous_mode)
|
||||
|
||||
result_text, _ = manual_failover.run_precheck().value
|
||||
if result_text:
|
||||
raise PatroniCtlException(result_text.format(action=action, leader=leader, candidate=candidate,
|
||||
cluster_name=cluster_name))
|
||||
|
||||
scheduled_at_str = None
|
||||
scheduled_at = None
|
||||
|
||||
if action == 'switchover':
|
||||
if scheduled is None and not force:
|
||||
next_hour = (datetime.datetime.now() + datetime.timedelta(hours=1)).strftime('%Y-%m-%dT%H:%M')
|
||||
scheduled = click.prompt('When should the switchover take place (e.g. ' + next_hour + ' ) ',
|
||||
type=str, default='now')
|
||||
|
||||
scheduled_at = parse_scheduled(scheduled)
|
||||
parse_result, scheduled_at = manual_failover.parse_scheduled()
|
||||
if parse_result:
|
||||
raise PatroniCtlException(parse_result.value[0].format(action=action))
|
||||
if scheduled_at:
|
||||
from patroni.config import get_global_config
|
||||
if get_global_config(cluster).is_paused:
|
||||
raise PatroniCtlException("Can't schedule switchover in the paused state")
|
||||
scheduled_at_str = scheduled_at.isoformat()
|
||||
|
||||
failover_value = {'leader': leader, 'candidate': candidate, 'scheduled_at': scheduled_at_str}
|
||||
|
||||
logging.debug(failover_value)
|
||||
|
||||
# By now we have established that the leader exists and the candidate exists
|
||||
# By now we have established that the leader exists and the candidate exists,
|
||||
# so confirm the action that is about to be run
|
||||
if not force:
|
||||
demote_msg = ', demoting current leader ' + leader if leader else ''
|
||||
demote_msg = f', demoting current leader {cluster.leader.name}' if cluster.leader else ''
|
||||
if scheduled_at_str:
|
||||
if not click.confirm('Are you sure you want to schedule {0} of cluster {1} at {2}{3}?'
|
||||
.format(action, cluster_name, scheduled_at_str, demote_msg)):
|
||||
# only switchover can be scheduled
|
||||
if not click.confirm(f'Are you sure you want to schedule a switchover in the cluster '
|
||||
f'{cluster_name} at {scheduled_at_str}{demote_msg}?'):
|
||||
# action as a var to catch a regression in the tests
|
||||
raise PatroniCtlException('Aborting scheduled ' + action)
|
||||
else:
|
||||
if not click.confirm('Are you sure you want to {0} cluster {1}{2}?'
|
||||
.format(action, cluster_name, demote_msg)):
|
||||
if not click.confirm(f'Are you sure you want to perform a {action} in the cluster {cluster_name}{demote_msg}?'):
|
||||
raise PatroniCtlException('Aborting ' + action)
|
||||
|
||||
# And finally the actual work
|
||||
failover_value = {'candidate': candidate}
|
||||
if action == 'switchover':
|
||||
failover_value['leader'] = leader
|
||||
if scheduled_at_str:
|
||||
failover_value['scheduled_at'] = scheduled_at_str
|
||||
|
||||
logging.debug(failover_value)
|
||||
|
||||
r = None
|
||||
try:
|
||||
member = cluster.leader.member if cluster.leader else candidate and cluster.get_member(candidate, False)
|
||||
@@ -1330,6 +1309,8 @@ def failover(obj: Dict[str, Any], cluster_name: str, group: Optional[int],
|
||||
|
||||
.. note::
|
||||
If *leader* is given perform a switchover instead of a failover.
|
||||
This behavior is deprecated. ``--leader`` option support will be
|
||||
removed in the next major release.
|
||||
|
||||
.. seealso::
|
||||
Refer to :func:`_do_failover_or_switchover` for details.
|
||||
@@ -1343,7 +1324,12 @@ def failover(obj: Dict[str, Any], cluster_name: str, group: Optional[int],
|
||||
:param candidate: name of a standby member to be promoted. Nodes that are tagged with ``nofailover`` cannot be used.
|
||||
:param force: perform the failover or switchover without asking for confirmations.
|
||||
"""
|
||||
action = 'switchover' if leader else 'failover'
|
||||
action = 'failover'
|
||||
if leader:
|
||||
action = 'switchover'
|
||||
click.echo(click.style(
|
||||
'Supplying a leader name using this command is deprecated and will be removed in a future version of'
|
||||
' Patroni, change your scripts to use `switchover` instead.\nExecuting switchover!', fg='red'))
|
||||
_do_failover_or_switchover(obj, action, cluster_name, group, leader, candidate, force)
|
||||
|
||||
|
||||
@@ -1561,9 +1547,11 @@ def output_members(obj: Dict[str, Any], cluster: Cluster, name: str,
|
||||
rows.append([member.get(n.lower().replace(' ', '_'), '') for n in columns])
|
||||
|
||||
title = 'Citus cluster' if is_citus_cluster else 'Cluster'
|
||||
group_title = '' if group is None else 'group: {0}, '.format(group)
|
||||
title_details = group_title and ' ({0}{1})'.format(group_title, initialize)
|
||||
title = ' {0}: {1}{2} '.format(title, name, title_details)
|
||||
title_details = f' ({initialize})'
|
||||
if is_citus_cluster:
|
||||
title_details = '' if group is None else f' (group: {group}, {initialize})'
|
||||
|
||||
title = f' {title}: {name}{title_details} '
|
||||
print_output(columns, rows, {'Group': 'r', 'Lag in MB': 'r', 'TL': 'r'}, fmt, title)
|
||||
|
||||
if fmt not in ('pretty', 'topology'): # Omit service info when using machine-readable formats
|
||||
@@ -1714,7 +1702,6 @@ def wait_until_pause_is_applied(dcs: AbstractDCS, paused: bool, old_cluster: Clu
|
||||
:param old_cluster: original cluster information before pause or unpause has been requested. Used to report which
|
||||
nodes are still pending to have ``pause`` equal *paused* at a given point in time.
|
||||
"""
|
||||
from patroni.config import get_global_config
|
||||
config = get_global_config(old_cluster)
|
||||
|
||||
click.echo("'{0}' request sent, waiting until it is recognized by all nodes".format(paused and 'pause' or 'resume'))
|
||||
@@ -1752,7 +1739,6 @@ def toggle_pause(config: Dict[str, Any], cluster_name: str, group: Optional[int]
|
||||
* ``pause`` state is already *paused*; or
|
||||
* cluster contains no accessible members.
|
||||
"""
|
||||
from patroni.config import get_global_config
|
||||
dcs = get_dcs(config, cluster_name, group)
|
||||
cluster = dcs.get_cluster()
|
||||
if get_global_config(cluster).is_paused == paused:
|
||||
|
||||
+53
-47
@@ -23,6 +23,7 @@ import dateutil.parser
|
||||
|
||||
from ..exceptions import PatroniFatalException
|
||||
from ..utils import deep_compare, uri
|
||||
from ..tags import Tags
|
||||
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
from ..config import Config
|
||||
@@ -191,11 +192,11 @@ _Version = Union[int, str]
|
||||
_Session = Union[int, float, str, None]
|
||||
|
||||
|
||||
class Member(NamedTuple('Member',
|
||||
[('version', _Version),
|
||||
('name', str),
|
||||
('session', _Session),
|
||||
('data', Dict[str, Any])])):
|
||||
class Member(Tags, NamedTuple('Member',
|
||||
[('version', _Version),
|
||||
('name', str),
|
||||
('session', _Session),
|
||||
('data', Dict[str, Any])])):
|
||||
"""Immutable object (namedtuple) which represents single member of PostgreSQL cluster.
|
||||
|
||||
.. note::
|
||||
@@ -316,20 +317,10 @@ class Member(NamedTuple('Member',
|
||||
"""The ``tags`` value from :attr:`~Member.data` if defined, otherwise an empty dictionary."""
|
||||
return self.data.get('tags', {})
|
||||
|
||||
@property
|
||||
def nofailover(self) -> bool:
|
||||
"""The value for ``nofailover`` in :attr:`Member`.tags`` if defined, otherwise ``False``."""
|
||||
return self.tags.get('nofailover', False)
|
||||
|
||||
@property
|
||||
def replicatefrom(self) -> Optional[str]:
|
||||
"""The value for ``replicatefrom`` in :attr:`Member`.tags`` if defined."""
|
||||
return self.tags.get('replicatefrom')
|
||||
|
||||
@property
|
||||
def clonefrom(self) -> bool:
|
||||
"""``True`` if both ``clonefrom`` tag is ``True`` and a connection URL is defined."""
|
||||
return self.tags.get('clonefrom', False) and bool(self.conn_url)
|
||||
return super().clonefrom and bool(self.conn_url)
|
||||
|
||||
@property
|
||||
def state(self) -> str:
|
||||
@@ -460,7 +451,7 @@ class Leader(NamedTuple):
|
||||
|
||||
|
||||
class Failover(NamedTuple):
|
||||
"""Immutable object (namedtuple) representing configuration information required for failover/switchover capability.
|
||||
"""Immutable object (namedtuple) which represents failover key.
|
||||
|
||||
:ivar version: version of the object.
|
||||
:ivar leader: name of the leader. If value isn't empty we treat it as a switchover from the specified node.
|
||||
@@ -556,6 +547,13 @@ class Failover(NamedTuple):
|
||||
"""
|
||||
return int(bool(self.leader)) + int(bool(self.candidate))
|
||||
|
||||
@property
|
||||
def is_switchover(self) -> bool:
|
||||
return bool(self.leader)
|
||||
|
||||
@property
|
||||
def is_failover(self) -> bool:
|
||||
return not self.is_switchover
|
||||
|
||||
class ClusterConfig(NamedTuple):
|
||||
"""Immutable object (namedtuple) which represents cluster configuration.
|
||||
@@ -912,8 +910,16 @@ class Cluster(NamedTuple('Cluster',
|
||||
|
||||
@property
|
||||
def __permanent_slots(self) -> Dict[str, Union[Dict[str, Any], Any]]:
|
||||
"""Dictionary of permanent replication slots."""
|
||||
return self.config and self.config.permanent_slots or {}
|
||||
"""Dictionary of permanent replication slots with their known LSN."""
|
||||
ret = deepcopy(self.config.permanent_slots if self.config else {})
|
||||
# If primary reported flush LSN for permanent slots we want to enrich our structure with it
|
||||
for name, lsn in (self.slots or {}).items():
|
||||
if name in ret:
|
||||
if not ret[name]:
|
||||
ret[name] = {}
|
||||
if isinstance(ret[name], dict):
|
||||
ret[name]['lsn'] = lsn
|
||||
return ret
|
||||
|
||||
@property
|
||||
def __permanent_physical_slots(self) -> Dict[str, Any]:
|
||||
@@ -938,7 +944,6 @@ class Cluster(NamedTuple('Cluster',
|
||||
|
||||
Will log an error if:
|
||||
|
||||
* Conflicting slot names between members are found
|
||||
* Any logical slots are disabled, due to version compatibility, and *show_error* is ``True``.
|
||||
|
||||
:param my_name: name of this node.
|
||||
@@ -951,21 +956,9 @@ class Cluster(NamedTuple('Cluster',
|
||||
|
||||
:returns: final dictionary of slot names, after merging with permanent slots and performing sanity checks.
|
||||
"""
|
||||
slot_members: List[str] = self._get_slot_members(my_name, role)
|
||||
|
||||
slots: Dict[str, Dict[str, str]] = {slot_name_from_member_name(name): {'type': 'physical'}
|
||||
for name in slot_members}
|
||||
|
||||
if len(slots) < len(slot_members):
|
||||
# Find which names are conflicting for a nicer error message
|
||||
slot_conflicts: Dict[str, List[str]] = defaultdict(list)
|
||||
for name in slot_members:
|
||||
slot_conflicts[slot_name_from_member_name(name)].append(name)
|
||||
logger.error("Following cluster members share a replication slot name: %s",
|
||||
"; ".join(f"{', '.join(v)} map to {k}"
|
||||
for k, v in slot_conflicts.items() if len(v) > 1))
|
||||
|
||||
slots: Dict[str, Dict[str, str]] = self._get_members_slots(my_name, role)
|
||||
permanent_slots: Dict[str, Any] = self._get_permanent_slots(is_standby_cluster, role, nofailover)
|
||||
|
||||
disabled_permanent_logical_slots: List[str] = self._merge_permanent_slots(
|
||||
slots, permanent_slots, my_name, major_version)
|
||||
|
||||
@@ -1025,7 +1018,7 @@ class Cluster(NamedTuple('Cluster',
|
||||
return disabled_permanent_logical_slots
|
||||
|
||||
def _get_permanent_slots(self, is_standby_cluster: bool, role: str, nofailover: bool) -> Dict[str, Any]:
|
||||
"""Get configured permanent slot names.
|
||||
"""Get configured permanent replication slots.
|
||||
|
||||
.. note::
|
||||
Permanent replication slots are only considered if ``use_slots`` configuration is enabled.
|
||||
@@ -1051,35 +1044,48 @@ class Cluster(NamedTuple('Cluster',
|
||||
|
||||
return self.__permanent_slots if role in ('master', 'primary') else self.__permanent_logical_slots
|
||||
|
||||
def _get_slot_members(self, my_name: str, role: str) -> List[str]:
|
||||
"""Get a list of member names that have replication slots sourcing from this node.
|
||||
def _get_members_slots(self, my_name: str, role: str) -> Dict[str, Dict[str, str]]:
|
||||
"""Get physical replication slots configuration for members that sourcing from this node.
|
||||
|
||||
If the ``replicatefrom`` tag is set on the member - we should not create the replication slot for it on
|
||||
the current primary, because that member would replicate from elsewhere. We still create the slot if
|
||||
the ``replicatefrom`` destination member is currently not a member of the cluster (fallback to the
|
||||
primary), or if ``replicatefrom`` destination member happens to be the current primary.
|
||||
|
||||
Will log an error if:
|
||||
|
||||
* Conflicting slot names between members are found
|
||||
|
||||
:param my_name: name of this node.
|
||||
:param role: role of this node, if this is a ``primary`` or ``standby_leader`` return list of members
|
||||
replicating from this node. If not then return a list of members replicating as cascaded
|
||||
replicas from this node.
|
||||
|
||||
:returns: list of member names.
|
||||
:returns: dictionary of physical replication slots that should exist on a given node.
|
||||
"""
|
||||
if not self.use_slots:
|
||||
return []
|
||||
return {}
|
||||
|
||||
# we always want to exclude the member with our name from the list
|
||||
members = filter(lambda m: m.name != my_name, self.members)
|
||||
|
||||
if role in ('master', 'primary', 'standby_leader'):
|
||||
slot_members = [m.name for m in self.members
|
||||
if m.name != my_name
|
||||
and (m.replicatefrom is None
|
||||
or m.replicatefrom == my_name
|
||||
or not self.has_member(m.replicatefrom))]
|
||||
members = [m for m in members if m.replicatefrom is None
|
||||
or m.replicatefrom == my_name or not self.has_member(m.replicatefrom)]
|
||||
else:
|
||||
# only manage slots for replicas that replicate from this one, except for the leader among them
|
||||
slot_members = [m.name for m in self.members
|
||||
if m.replicatefrom == my_name and m.name != self.leader_name]
|
||||
return slot_members
|
||||
members = [m for m in members if m.replicatefrom == my_name and m.name != self.leader_name]
|
||||
|
||||
slots = {slot_name_from_member_name(m.name): {'type': 'physical'} for m in members}
|
||||
if len(slots) < len(members):
|
||||
# Find which names are conflicting for a nicer error message
|
||||
slot_conflicts: Dict[str, List[str]] = defaultdict(list)
|
||||
for member in members:
|
||||
slot_conflicts[slot_name_from_member_name(member.name)].append(member.name)
|
||||
logger.error("Following cluster members share a replication slot name: %s",
|
||||
"; ".join(f"{', '.join(v)} map to {k}"
|
||||
for k, v in slot_conflicts.items() if len(v) > 1))
|
||||
return slots
|
||||
|
||||
def has_permanent_logical_slots(self, my_name: str, nofailover: bool, major_version: int = 110000) -> bool:
|
||||
"""Check if the given member node has permanent ``logical`` replication slots configured.
|
||||
|
||||
@@ -141,6 +141,36 @@ class HTTPClient(object):
|
||||
class ConsulClient(base.Consul):
|
||||
|
||||
def __init__(self, *args: Any, **kwargs: Any) -> None:
|
||||
"""
|
||||
Consul client with Patroni customisations.
|
||||
|
||||
.. note::
|
||||
|
||||
Parameters, *token*, *cert* and *ca_cert* are not passed to the parent class :class:`consul.base.Consul`.
|
||||
|
||||
Original class documentation,
|
||||
|
||||
*token* is an optional ``ACL token``. If supplied it will be used by
|
||||
default for all requests made with this client session. It's still
|
||||
possible to override this token by passing a token explicitly for a
|
||||
request.
|
||||
|
||||
*consistency* sets the consistency mode to use by default for all reads
|
||||
that support the consistency option. It's still possible to override
|
||||
this by passing explicitly for a given request. *consistency* can be
|
||||
either 'default', 'consistent' or 'stale'.
|
||||
|
||||
*dc* is the datacenter that this agent will communicate with.
|
||||
By default, the datacenter of the host is used.
|
||||
|
||||
*verify* is whether to verify the SSL certificate for HTTPS requests
|
||||
|
||||
*cert* client side certificates for HTTPS requests
|
||||
|
||||
:param args: positional arguments to pass to :class:`consul.base.Consul`
|
||||
:param kwargs: keyword arguments, with *cert*, *ca_cert* and *token* removed, passed to
|
||||
:class:`consul.base.Consul`
|
||||
"""
|
||||
self._cert = kwargs.pop('cert', None)
|
||||
self._ca_cert = kwargs.pop('ca_cert', None)
|
||||
self.token = kwargs.get('token')
|
||||
|
||||
@@ -756,7 +756,7 @@ class Kubernetes(AbstractDCS):
|
||||
self._role_label = config.get('role_label', 'role')
|
||||
self._leader_label_value = config.get('leader_label_value', 'master')
|
||||
self._follower_label_value = config.get('follower_label_value', 'replica')
|
||||
self._standby_leader_label_value = config.get('standby_leader_label_value', 'standby-leader')
|
||||
self._standby_leader_label_value = config.get('standby_leader_label_value', 'master')
|
||||
self._tmp_role_label = config.get('tmp_role_label')
|
||||
self._ca_certs = os.environ.get('PATRONI_KUBERNETES_CACERT', config.get('cacert')) or SERVICE_CERT_FILENAME
|
||||
super(Kubernetes, self).__init__({**config, 'namespace': ''})
|
||||
@@ -1140,6 +1140,13 @@ class Kubernetes(AbstractDCS):
|
||||
"""Unused"""
|
||||
raise NotImplementedError # pragma: no cover
|
||||
|
||||
def write_leader_optime(self, last_lsn: int) -> None:
|
||||
"""Write value for WAL LSN to ``optime`` annotation of the leader object.
|
||||
|
||||
:param last_lsn: absolute WAL LSN in bytes.
|
||||
"""
|
||||
self.patch_or_create(self.leader_path, {self._OPTIME: str(last_lsn)}, patch=True, retry=False)
|
||||
|
||||
def _update_leader_with_retry(self, annotations: Dict[str, Any],
|
||||
resource_version: Optional[str], ips: List[str]) -> bool:
|
||||
retry = self._retry.copy()
|
||||
@@ -1269,13 +1276,10 @@ class Kubernetes(AbstractDCS):
|
||||
def touch_member(self, data: Dict[str, Any]) -> bool:
|
||||
cluster = self.cluster
|
||||
if cluster and cluster.leader and cluster.leader.name == self._name:
|
||||
role = self._leader_label_value
|
||||
role = self._standby_leader_label_value if data['role'] == 'standby_leader' else self._leader_label_value
|
||||
tmp_role = 'master'
|
||||
elif data['state'] == 'running' and data['role'] not in ('master', 'primary'):
|
||||
role = {
|
||||
'replica': self._follower_label_value,
|
||||
'standby-leader': self._standby_leader_label_value,
|
||||
}.get(data['role'], data['role'])
|
||||
role = {'replica': self._follower_label_value}.get(data['role'], data['role'])
|
||||
tmp_role = data['role']
|
||||
else:
|
||||
role = None
|
||||
|
||||
+104
-63
@@ -20,31 +20,28 @@ from .postgresql.callback_executor import CallbackAction
|
||||
from .postgresql.misc import postgres_version_to_int
|
||||
from .postgresql.postmaster import PostmasterProcess
|
||||
from .postgresql.rewind import Rewind
|
||||
from .tags import Tags
|
||||
from .utils import polling_loop, tzutc
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class _MemberStatus(NamedTuple):
|
||||
"""Node status distilled from API response:
|
||||
class _MemberStatus(Tags, NamedTuple('_MemberStatus',
|
||||
[('member', Member),
|
||||
('reachable', bool),
|
||||
('in_recovery', Optional[bool]),
|
||||
('wal_position', int),
|
||||
('data', Dict[str, Any])])):
|
||||
"""Node status distilled from API response.
|
||||
|
||||
member - dcs.Member object of the node
|
||||
reachable - `!False` if the node is not reachable or is not responding with correct JSON
|
||||
in_recovery - `!True` if pg_is_in_recovery() == true
|
||||
dcs_last_seen - timestamp from JSON of last succesful communication with DCS
|
||||
timeline - timeline value from JSON
|
||||
wal_position - maximum value of `replayed_location` or `received_location` from JSON
|
||||
tags - dictionary with values of different tags (i.e. nofailover)
|
||||
watchdog_failed - indicates that watchdog is required by configuration but not available or failed
|
||||
Consists of the following fields:
|
||||
|
||||
:ivar member: :class:`~patroni.dcs.Member` object of the node.
|
||||
:ivar reachable: ``False`` if the node is not reachable or is not responding with correct JSON.
|
||||
:ivar in_recovery: ``False`` if the node is running as a primary (`if pg_is_in_recovery() == true`).
|
||||
:ivar wal_position: maximum value of ``replayed_location`` or ``received_location`` from JSON.
|
||||
:ivar data: the whole JSON response for future usage.
|
||||
"""
|
||||
member: Member
|
||||
reachable: bool
|
||||
in_recovery: Optional[bool]
|
||||
dcs_last_seen: int
|
||||
timeline: int
|
||||
wal_position: int
|
||||
tags: Dict[str, Any]
|
||||
watchdog_failed: bool
|
||||
|
||||
@classmethod
|
||||
def from_api_response(cls, member: Member, json: Dict[str, Any]) -> '_MemberStatus':
|
||||
@@ -57,21 +54,34 @@ class _MemberStatus(NamedTuple):
|
||||
wal: Dict[str, Any] = json.get('wal') or json['xlog']
|
||||
# abuse difference in primary/replica response format
|
||||
in_recovery = not (bool(wal.get('location')) or json.get('role') in ('master', 'primary'))
|
||||
timeline = json.get('timeline', 0)
|
||||
dcs_last_seen = json.get('dcs_last_seen', 0)
|
||||
lsn = int(in_recovery and max(wal.get('received_location', 0), wal.get('replayed_location', 0)))
|
||||
return cls(member, True, in_recovery, dcs_last_seen, timeline, lsn,
|
||||
json.get('tags', {}), json.get('watchdog_failed', False))
|
||||
return cls(member, True, in_recovery, lsn, json)
|
||||
|
||||
@property
|
||||
def tags(self) -> Dict[str, Any]:
|
||||
"""Dictionary with values of different tags (i.e. nofailover)."""
|
||||
return self.data.get('tags', {})
|
||||
|
||||
@property
|
||||
def timeline(self) -> int:
|
||||
"""Timeline value from JSON."""
|
||||
return self.data.get('timeline', 0)
|
||||
|
||||
@property
|
||||
def watchdog_failed(self) -> bool:
|
||||
"""Indicates that watchdog is required by configuration but not available or failed."""
|
||||
return self.data.get('watchdog_failed', False)
|
||||
|
||||
@classmethod
|
||||
def unknown(cls, member: Member) -> '_MemberStatus':
|
||||
return cls(member, False, None, 0, 0, 0, {}, False)
|
||||
"""Create a new class instance with empty or null values."""
|
||||
return cls(member, False, None, 0, {})
|
||||
|
||||
def failover_limitation(self) -> Optional[str]:
|
||||
"""Returns reason why this node can't promote or None if everything is ok."""
|
||||
if not self.reachable:
|
||||
return 'not reachable'
|
||||
if self.tags.get('nofailover', False):
|
||||
if self.nofailover:
|
||||
return 'not allowed to promote'
|
||||
if self.watchdog_failed:
|
||||
return 'not watchdog capable'
|
||||
@@ -215,6 +225,17 @@ class Ha(object):
|
||||
"""
|
||||
return self.is_synchronous_mode() and not self.cluster.sync.is_empty
|
||||
|
||||
def _get_failover_action_name(self) -> str:
|
||||
"""Return the currently requested manual failover action name or the default ``failover``.
|
||||
|
||||
:returns: :class:`str` representing the manually requested action (``manual failover`` if no leader
|
||||
is specified in the ``/failover`` in DCS, ``switchover`` otherwise) or ``failover`` if
|
||||
``/failover`` is empty.
|
||||
"""
|
||||
if not self.cluster.failover:
|
||||
return 'failover'
|
||||
return 'switchover' if self.cluster.failover.is_switchover else 'manual failover'
|
||||
|
||||
def load_cluster_from_dcs(self) -> None:
|
||||
cluster = self.dcs.get_cluster()
|
||||
|
||||
@@ -901,6 +922,27 @@ class Ha(object):
|
||||
lag = (self.cluster.last_lsn or 0) - wal_position
|
||||
return lag > self.global_config.maximum_lag_on_failover
|
||||
|
||||
def has_members_eligible_to_promote(self, members: List[Member], reference_lsn: int = 0,
|
||||
fast_path: bool = False) -> bool:
|
||||
ret = False
|
||||
cluster_timeline = self.cluster.timeline
|
||||
|
||||
for st in self.fetch_nodes_statuses(members):
|
||||
not_allowed_reason = st.failover_limitation()
|
||||
if not_allowed_reason:
|
||||
logger.info('Member %s is %s', st.member.name, not_allowed_reason)
|
||||
elif fast_path:
|
||||
return True
|
||||
elif reference_lsn and st.wal_position < reference_lsn or \
|
||||
not reference_lsn and self.is_lagging(st.wal_position):
|
||||
logger.info('Member %s exceeds maximum replication lag', st.member.name)
|
||||
elif self.check_timeline() and (not st.timeline or st.timeline < cluster_timeline):
|
||||
logger.info('Timeline %s of member %s is behind the cluster timeline %s',
|
||||
st.timeline, st.member.name, cluster_timeline)
|
||||
else:
|
||||
ret = True
|
||||
return ret
|
||||
|
||||
def _is_healthiest_node(self, members: Collection[Member], check_replication_lag: bool = True) -> bool:
|
||||
"""This method tries to determine whether I am healthy enough to became a new leader candidate or not."""
|
||||
|
||||
@@ -946,27 +988,14 @@ class Ha(object):
|
||||
"""
|
||||
candidates = self.get_failover_candidates(exclude_failover_candidate)
|
||||
|
||||
action = self._get_failover_action_name()
|
||||
if self.is_synchronous_mode() and self.cluster.failover and self.cluster.failover.candidate and not candidates:
|
||||
logger.warning('Failover candidate=%s does not match with sync_standbys=%s',
|
||||
self.cluster.failover.candidate, self.cluster.sync.sync_standby)
|
||||
logger.warning('%s candidate=%s does not match with sync_standbys=%s',
|
||||
action.title(), self.cluster.failover.candidate, self.cluster.sync.sync_standby)
|
||||
elif not candidates:
|
||||
logger.warning('manual failover: candidates list is empty')
|
||||
logger.warning('%s: candidates list is empty', action)
|
||||
|
||||
ret = False
|
||||
cluster_timeline = self.cluster.timeline
|
||||
for st in self.fetch_nodes_statuses(candidates):
|
||||
not_allowed_reason = st.failover_limitation()
|
||||
if not_allowed_reason:
|
||||
logger.info('Member %s is %s', st.member.name, not_allowed_reason)
|
||||
elif cluster_lsn and st.wal_position < cluster_lsn or \
|
||||
not cluster_lsn and self.is_lagging(st.wal_position):
|
||||
logger.info('Member %s exceeds maximum replication lag', st.member.name)
|
||||
elif self.check_timeline() and (not st.timeline or st.timeline < cluster_timeline):
|
||||
logger.info('Timeline %s of member %s is behind the cluster timeline %s',
|
||||
st.timeline, st.member.name, cluster_timeline)
|
||||
else:
|
||||
ret = True
|
||||
return ret
|
||||
return self.has_members_eligible_to_promote(candidates, cluster_lsn)
|
||||
|
||||
def manual_failover_process_no_leader(self) -> Optional[bool]:
|
||||
"""Handles manual failover/switchover when the old leader already stepped down.
|
||||
@@ -977,15 +1006,18 @@ class Ha(object):
|
||||
failover = self.cluster.failover
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
assert failover is not None
|
||||
if failover.candidate: # manual failover to specific member
|
||||
if failover.candidate == self.state_handler.name: # manual failover to me
|
||||
|
||||
action = self._get_failover_action_name()
|
||||
|
||||
if failover.candidate: # manual failover/switchover to specific member
|
||||
if failover.candidate == self.state_handler.name: # manual failover/switchover to me
|
||||
return True
|
||||
elif self.is_paused():
|
||||
# Remove failover key if the node to failover has terminated to avoid waiting for it indefinitely
|
||||
# In order to avoid attempts to delete this key from all nodes only the primary is allowed to do it.
|
||||
if not self.cluster.get_member(failover.candidate, fallback_to_leader=False)\
|
||||
and self.state_handler.is_primary():
|
||||
logger.warning("manual failover: removing failover key because failover candidate is not running")
|
||||
logger.warning("%s: removing failover key because failover candidate is not running", action)
|
||||
self.dcs.manual_failover('', '', version=failover.version)
|
||||
return None
|
||||
return False
|
||||
@@ -1001,18 +1033,18 @@ class Ha(object):
|
||||
st = self.fetch_node_status(member)
|
||||
not_allowed_reason = st.failover_limitation()
|
||||
if not_allowed_reason is None: # node is healthy
|
||||
logger.info('manual failover: to %s, i am %s', st.member.name, self.state_handler.name)
|
||||
logger.info('%s: to %s, i am %s', action, st.member.name, self.state_handler.name)
|
||||
return False
|
||||
# we wanted to failover to specific member but it is not healthy
|
||||
logger.warning('manual failover: member %s is %s', st.member.name, not_allowed_reason)
|
||||
# we wanted to failover/switchover to specific member but it is not healthy
|
||||
logger.warning('%s: member %s is %s', action, st.member.name, not_allowed_reason)
|
||||
|
||||
# at this point we should consider all members as a candidates for failover
|
||||
# at this point we should consider all members as a candidates for failover/switchover
|
||||
# i.e. we assume that failover.candidate is None
|
||||
elif self.is_paused():
|
||||
return False
|
||||
|
||||
# try to pick some other members to failover and check that they are healthy
|
||||
if failover.leader:
|
||||
# try to pick some other members for switchover and check that they are healthy
|
||||
if failover.is_switchover:
|
||||
if self.state_handler.name == failover.leader: # I was the leader
|
||||
# exclude desired member which is unhealthy if it was specified
|
||||
if self.is_failover_possible(exclude_failover_candidate=bool(failover.candidate)):
|
||||
@@ -1069,8 +1101,8 @@ class Ha(object):
|
||||
|
||||
if self.cluster.failover:
|
||||
# When doing a switchover in synchronous mode only synchronous nodes and former leader are allowed to race
|
||||
if self.sync_mode_is_active() and not self.cluster.sync.matches(self.state_handler.name, True) and \
|
||||
self.cluster.failover.leader:
|
||||
if self.cluster.failover.is_switchover and self.sync_mode_is_active() \
|
||||
and not self.cluster.sync.matches(self.state_handler.name, True):
|
||||
return False
|
||||
return self.manual_failover_process_no_leader() or False
|
||||
|
||||
@@ -1234,28 +1266,35 @@ class Ha(object):
|
||||
|
||||
:returns: action message if demote was initiated, None if no action was taken"""
|
||||
failover = self.cluster.failover
|
||||
# if there is no failover key or
|
||||
# I am holding the lock but am not primary = I am the standby leader,
|
||||
# then do nothing
|
||||
if not failover or (self.is_paused() and not self.state_handler.is_primary()):
|
||||
return
|
||||
|
||||
action = self._get_failover_action_name()
|
||||
bare_action = action.replace('manual ', '')
|
||||
|
||||
# it is not the time for the scheduled switchover yet, do nothing
|
||||
if (failover.scheduled_at and not
|
||||
self.should_run_scheduled_action("failover", failover.scheduled_at, lambda:
|
||||
self.should_run_scheduled_action(bare_action, failover.scheduled_at, lambda:
|
||||
self.dcs.manual_failover('', '', version=failover.version))):
|
||||
return
|
||||
|
||||
if not failover.leader or failover.leader == self.state_handler.name:
|
||||
if not failover.candidate or failover.candidate != self.state_handler.name:
|
||||
if not failover.candidate and self.is_paused():
|
||||
logger.warning('Failover is possible only to a specific candidate in a paused state')
|
||||
logger.warning('%s is possible only to a specific candidate in a paused state', action.title())
|
||||
elif self.is_failover_possible():
|
||||
ret = self._async_executor.try_run_async('manual failover: demote', self.demote, ('graceful',))
|
||||
return ret or 'manual failover: demoting myself'
|
||||
ret = self._async_executor.try_run_async(f'{action}: demote', self.demote, ('graceful',))
|
||||
return ret or f'{action}: demoting myself'
|
||||
else:
|
||||
logger.warning('manual failover: no healthy members found, failover is not possible')
|
||||
logger.warning('%s: no healthy members found, %s is not possible',
|
||||
action, bare_action)
|
||||
else:
|
||||
logger.warning('manual failover: I am already the leader, no need to failover')
|
||||
logger.warning('%s: I am already the leader, no need to %s', action, bare_action)
|
||||
else:
|
||||
logger.warning('manual failover: leader name does not match: %s != %s',
|
||||
failover.leader, self.state_handler.name)
|
||||
logger.warning('%s: leader name does not match: %s != %s', action, failover.leader, self.state_handler.name)
|
||||
|
||||
logger.info('Cleaning up failover key')
|
||||
self.dcs.manual_failover('', '', version=failover.version)
|
||||
@@ -1312,6 +1351,7 @@ class Ha(object):
|
||||
self._delete_leader()
|
||||
return 'removed leader lock because postgres is not running as primary'
|
||||
|
||||
# update lock to avoid split-brain
|
||||
if self.update_lock(True):
|
||||
msg = self.process_manual_failover_from_leader()
|
||||
if msg is not None:
|
||||
@@ -1980,8 +2020,9 @@ class Ha(object):
|
||||
exclude = [self.state_handler.name] + ([failover.candidate] if failover and exclude_failover_candidate else [])
|
||||
|
||||
def is_eligible(node: Member) -> bool:
|
||||
# TODO: allow manual failover (=no leader specified) to async node
|
||||
if self.sync_mode_is_active() and not self.cluster.sync.matches(node.name):
|
||||
# in synchronous mode we allow failover (not switchover!) to async node
|
||||
if self.sync_mode_is_active() and not self.cluster.sync.matches(node.name)\
|
||||
and not (failover and failover.is_failover):
|
||||
return False
|
||||
# Don't spend time on "nofailover" nodes checking.
|
||||
# We also don't need nodes which we can't query with the api in the list.
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
from enum import Enum
|
||||
from typing import Optional, Tuple, TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
import datetime
|
||||
|
||||
from .dcs import Cluster
|
||||
from .ha import Patroni
|
||||
from .utils import ParseScheduleErrors
|
||||
|
||||
from .utils import parse_schedule
|
||||
|
||||
|
||||
class ManualFailoverPrecheckStatus(Enum):
|
||||
FAILOVER_NO_CANDIDATE = ('Failover could be performed only to a specific candidate', 400)
|
||||
SWITCHOVER_NO_LEADER = ('Switchover could be performed only from a specific leader', 400)
|
||||
SCHEDULED_FAILOVER = ("Failover can't be scheduled", 400)
|
||||
SCHEDULED_SWITCHOVER_PAUSE = ("Can't schedule switchover in the paused state", 400)
|
||||
SWITCHOVER_PAUSE_NO_CANDIDATE = ('Switchover is possible only to a specific candidate in a paused state', 400)
|
||||
SWITCHOVER_TO_LEADER = ('Switchover target and source are the same', 400)
|
||||
|
||||
CLUSTER_NO_LEADER = ('Cluster {cluster_name} has no leader', 412)
|
||||
LEADER_NOT_MEMBER = ('Member {leader} is not the leader of cluster {cluster_name}', 412)
|
||||
CANDIDATE_NOT_SYNC_STANDBY = ('candidate name does not match with sync_standby', 412)
|
||||
NO_SYNC_CANDIDATE = ('{action} is not possible: can not find sync_standby', 412)
|
||||
ONLY_LEADER = ('{action} is not possible: cluster does not have members except leader', 412)
|
||||
CANDIDATE_NOT_MEMEBER = ('Member {candidate} does not exist in cluster {cluster_name} or is tagged as nofailover',
|
||||
412)
|
||||
NO_GOOD_CANDIDATES = ('{action} is not possible: no good candidates have been found', 412)
|
||||
|
||||
CHECK_PASSED = ('', None)
|
||||
|
||||
|
||||
class ManualFailover(object):
|
||||
|
||||
def __init__(self, action: str, cluster: 'Cluster',
|
||||
leader: Optional[str], candidate: Optional[str], scheduled: Optional[str],
|
||||
paused: bool = False, sync_mode: bool = False, patroni_obj: Optional['Patroni'] = None) -> None:
|
||||
self.action = action
|
||||
self.cluster = cluster
|
||||
self.leader = leader
|
||||
self.candidate = candidate
|
||||
self.scheduled = scheduled
|
||||
self.paused = paused
|
||||
self.sync_mode = sync_mode
|
||||
self.patroni = patroni_obj
|
||||
|
||||
def parse_scheduled(self) -> Tuple[Optional['ParseScheduleErrors'], Optional['datetime.datetime']]:
|
||||
return parse_schedule(self.scheduled)
|
||||
|
||||
def run_precheck(self) -> ManualFailoverPrecheckStatus:
|
||||
if self.action == 'failover' and not self.candidate:
|
||||
return ManualFailoverPrecheckStatus.FAILOVER_NO_CANDIDATE
|
||||
elif self.action == 'switchover' and not self.leader:
|
||||
return ManualFailoverPrecheckStatus.SWITCHOVER_NO_LEADER
|
||||
|
||||
if self.scheduled:
|
||||
if self.action == 'failover':
|
||||
return ManualFailoverPrecheckStatus.SCHEDULED_FAILOVER
|
||||
elif self.paused:
|
||||
return ManualFailoverPrecheckStatus.SCHEDULED_SWITCHOVER_PAUSE
|
||||
|
||||
if self.paused and not self.candidate:
|
||||
return ManualFailoverPrecheckStatus.SWITCHOVER_PAUSE_NO_CANDIDATE
|
||||
|
||||
if self.leader == self.candidate:
|
||||
return ManualFailoverPrecheckStatus.SWITCHOVER_TO_LEADER
|
||||
|
||||
if self.action == 'switchover':
|
||||
if self.cluster.leader is None or not self.cluster.leader.name:
|
||||
return ManualFailoverPrecheckStatus.CLUSTER_NO_LEADER
|
||||
if self.cluster.leader.name != self.leader:
|
||||
return ManualFailoverPrecheckStatus.LEADER_NOT_MEMBER
|
||||
|
||||
if self.candidate:
|
||||
if self.action == 'switchover' and self.sync_mode and not self.cluster.sync.matches(self.candidate):
|
||||
return ManualFailoverPrecheckStatus.CANDIDATE_NOT_SYNC_STANDBY
|
||||
members = [m for m in self.cluster.members if m.name == self.candidate]
|
||||
if not members:
|
||||
return ManualFailoverPrecheckStatus.CANDIDATE_NOT_MEMEBER
|
||||
elif self.sync_mode:
|
||||
members = [m for m in self.cluster.members if self.cluster.sync.matches(m.name)]
|
||||
if not members:
|
||||
return ManualFailoverPrecheckStatus.NO_SYNC_CANDIDATE
|
||||
else:
|
||||
members = [m for m in self.cluster.members if not self.cluster.leader or m.name != self.cluster.leader.name and m.api_url]
|
||||
if not members:
|
||||
return ManualFailoverPrecheckStatus.ONLY_LEADER
|
||||
|
||||
if self.patroni and not self.patroni.ha.has_members_eligible_to_promote(members, fast_path=True):
|
||||
return ManualFailoverPrecheckStatus.NO_GOOD_CANDIDATES
|
||||
|
||||
return ManualFailoverPrecheckStatus.CHECK_PASSED
|
||||
@@ -18,7 +18,7 @@ from .bootstrap import Bootstrap
|
||||
from .callback_executor import CallbackAction, CallbackExecutor
|
||||
from .cancellable import CancellableSubprocess
|
||||
from .config import ConfigHandler, mtime
|
||||
from .connection import Connection, get_connection_cursor
|
||||
from .connection import ConnectionPool, get_connection_cursor
|
||||
from .citus import CitusHandler
|
||||
from .misc import parse_history, parse_lsn, postgres_major_version_to_int
|
||||
from .postmaster import PostmasterProcess
|
||||
@@ -79,7 +79,8 @@ class Postgresql(object):
|
||||
self.set_state('stopped')
|
||||
|
||||
self._pending_restart = False
|
||||
self._connection = Connection()
|
||||
self.connection_pool = ConnectionPool()
|
||||
self._connection = self.connection_pool.get('heartbeat')
|
||||
self.citus_handler = CitusHandler(self, config.get('citus'))
|
||||
self.config = ConfigHandler(self, config)
|
||||
self.config.check_directories()
|
||||
@@ -277,7 +278,7 @@ class Postgresql(object):
|
||||
|
||||
:returns: 'ok' if PostgreSQL is up, 'reject' if starting up, 'no_resopnse' if not up."""
|
||||
|
||||
r = self.config.local_connect_kwargs
|
||||
r = self.connection_pool.conn_kwargs
|
||||
cmd = [self.pgcommand('pg_isready'), '-p', r['port'], '-d', self._database]
|
||||
|
||||
# Host is not set if we are connecting via default unix socket
|
||||
@@ -328,10 +329,6 @@ class Postgresql(object):
|
||||
def connection(self) -> Union['connection3', 'Connection3[Any]']:
|
||||
return self._connection.get()
|
||||
|
||||
def set_connection_kwargs(self, kwargs: Dict[str, Any]) -> None:
|
||||
self._connection.set_conn_kwargs(kwargs.copy())
|
||||
self.citus_handler.set_conn_kwargs(kwargs.copy())
|
||||
|
||||
def _query(self, sql: str, *params: Any) -> List[Tuple[Any, ...]]:
|
||||
"""Execute *sql* query with *params* and optionally return results.
|
||||
|
||||
@@ -694,7 +691,7 @@ class Postgresql(object):
|
||||
# the former node, otherwise, we might get a stalled one
|
||||
# after kill -9, which would report incorrect data to
|
||||
# patroni.
|
||||
self._connection.close()
|
||||
self.connection_pool.close()
|
||||
|
||||
if self.is_running():
|
||||
logger.error('Cannot start PostgreSQL because one is already running.')
|
||||
@@ -765,7 +762,7 @@ class Postgresql(object):
|
||||
def checkpoint(self, connect_kwargs: Optional[Dict[str, Any]] = None,
|
||||
timeout: Optional[float] = None) -> Optional[str]:
|
||||
check_not_is_in_recovery = connect_kwargs is not None
|
||||
connect_kwargs = connect_kwargs or self.config.local_connect_kwargs
|
||||
connect_kwargs = connect_kwargs or self.connection_pool.conn_kwargs
|
||||
for p in ['connect_timeout', 'options']:
|
||||
connect_kwargs.pop(p, None)
|
||||
if timeout:
|
||||
|
||||
@@ -176,7 +176,7 @@ class Bootstrap(object):
|
||||
"""
|
||||
cmd = config.get('post_bootstrap') or config.get('post_init')
|
||||
if cmd:
|
||||
r = self._postgresql.config.local_connect_kwargs
|
||||
r = self._postgresql.connection_pool.conn_kwargs
|
||||
connstring = self._postgresql.config.format_dsn(r, True)
|
||||
if 'host' not in r:
|
||||
# https://www.postgresql.org/docs/current/static/libpq-pgpass.html
|
||||
|
||||
@@ -6,7 +6,6 @@ from threading import Condition, Event, Thread
|
||||
from urllib.parse import urlparse
|
||||
from typing import Any, Dict, List, Optional, Union, Tuple, TYPE_CHECKING
|
||||
|
||||
from .connection import Connection
|
||||
from ..dcs import CITUS_COORDINATOR_GROUP_ID, Cluster
|
||||
from ..psycopg import connect, quote_ident
|
||||
|
||||
@@ -71,7 +70,10 @@ class CitusHandler(Thread):
|
||||
self.daemon = True
|
||||
self._postgresql = postgresql
|
||||
self._config = config
|
||||
self._connection = Connection()
|
||||
if config:
|
||||
self._connection = postgresql.connection_pool.get(
|
||||
'citus', {'dbname': config['database'],
|
||||
'options': '-c statement_timeout=0 -c idle_in_transaction_session_timeout=0'})
|
||||
self._pg_dist_node: Dict[int, PgDistNode] = {} # Cache of pg_dist_node: {groupid: PgDistNode()}
|
||||
self._tasks: List[PgDistNode] = [] # Requests to change pg_dist_node, every task is a `PgDistNode`
|
||||
self._in_flight: Optional[PgDistNode] = None # Reference to the `PgDistNode` being changed in a transaction
|
||||
@@ -91,12 +93,6 @@ class CitusHandler(Thread):
|
||||
def is_worker(self) -> bool:
|
||||
return self.is_enabled() and not self.is_coordinator()
|
||||
|
||||
def set_conn_kwargs(self, kwargs: Dict[str, Any]) -> None:
|
||||
if isinstance(self._config, dict): # self.is_enabled():
|
||||
kwargs.update({'dbname': self._config['database'],
|
||||
'options': '-c statement_timeout=0 -c idle_in_transaction_session_timeout=0'})
|
||||
self._connection.set_conn_kwargs(kwargs)
|
||||
|
||||
def schedule_cache_rebuild(self) -> None:
|
||||
with self._condition:
|
||||
self._schedule_load_pg_dist_node = True
|
||||
@@ -359,8 +355,8 @@ class CitusHandler(Thread):
|
||||
if not isinstance(self._config, dict): # self.is_enabled()
|
||||
return
|
||||
|
||||
conn_kwargs = self._postgresql.config.local_connect_kwargs
|
||||
conn_kwargs['options'] = '-c synchronous_commit=local -c statement_timeout=0'
|
||||
conn_kwargs = {**self._postgresql.connection_pool.conn_kwargs,
|
||||
'options': '-c synchronous_commit=local -c statement_timeout=0'}
|
||||
if self._config['database'] != self._postgresql.database:
|
||||
conn = connect(**conn_kwargs)
|
||||
try:
|
||||
|
||||
@@ -14,7 +14,7 @@ from typing import Any, Collection, Dict, Iterator, List, Optional, Union, Tuple
|
||||
from .validator import recovery_parameters, transform_postgresql_parameter_value, transform_recovery_parameter_value
|
||||
from ..collections import CaseInsensitiveDict, CaseInsensitiveSet
|
||||
from ..dcs import Leader, Member, RemoteMember, slot_name_from_member_name
|
||||
from ..exceptions import PatroniFatalException
|
||||
from ..exceptions import PatroniFatalException, PostgresConnectionException
|
||||
from ..file_perm import pg_perm
|
||||
from ..utils import compare_values, parse_bool, parse_int, split_host_port, uri, validate_directory, is_subpath
|
||||
from ..validator import IntValidator, EnumValidator
|
||||
@@ -623,7 +623,24 @@ class ConfigHandler(object):
|
||||
'recovery_target_action', 'standby_mode', self._triggerfile_wrong_name})
|
||||
return CaseInsensitiveSet(self._RECOVERY_PARAMETERS - skip_params)
|
||||
|
||||
def _read_recovery_params(self) -> Tuple[Optional[CaseInsensitiveDict], Optional[bool]]:
|
||||
def _read_recovery_params(self) -> Tuple[Optional[CaseInsensitiveDict], bool]:
|
||||
"""Read current recovery parameters values.
|
||||
|
||||
.. note::
|
||||
We query Postgres only if we detected that Postgresql was restarted
|
||||
or when at least one of the following files was updated:
|
||||
|
||||
* ``postgresql.conf``;
|
||||
* ``postgresql.auto.conf``;
|
||||
* ``passfile`` that is used in the ``primary_conninfo``.
|
||||
|
||||
:returns: a tuple with two elements:
|
||||
|
||||
* :class:`CaseInsensitiveDict` object with current values of recovery parameters,
|
||||
or ``None`` if no configuration files were updated;
|
||||
|
||||
* ``True`` if new values of recovery parameters were queried, ``False`` otherwise.
|
||||
"""
|
||||
if self._postgresql.is_starting():
|
||||
return None, False
|
||||
|
||||
@@ -644,11 +661,20 @@ class ConfigHandler(object):
|
||||
self._postgresql_conf_mtime = pg_conf_mtime
|
||||
self._auto_conf_mtime = auto_conf_mtime
|
||||
self._postmaster_ctime = postmaster_ctime
|
||||
except Exception:
|
||||
except Exception as exc:
|
||||
if all((isinstance(exc, PostgresConnectionException),
|
||||
self._postgresql_conf_mtime == pg_conf_mtime,
|
||||
self._auto_conf_mtime == auto_conf_mtime,
|
||||
self._passfile_mtime == passfile_mtime,
|
||||
self._postmaster_ctime != postmaster_ctime)):
|
||||
# We detected that the connection to postgres fails, but the process creation time of the postmaster
|
||||
# doesn't match the old value. It is an indicator that Postgres crashed and either doing crash
|
||||
# recovery or down. In this case we return values like nothing changed in the config.
|
||||
return None, False
|
||||
values = None
|
||||
return values, True
|
||||
|
||||
def _read_recovery_params_pre_v12(self) -> Tuple[Optional[CaseInsensitiveDict], Optional[bool]]:
|
||||
def _read_recovery_params_pre_v12(self) -> Tuple[Optional[CaseInsensitiveDict], bool]:
|
||||
recovery_conf_mtime = mtime(self._recovery_conf)
|
||||
passfile_mtime = mtime(self._passfile) if self._passfile else False
|
||||
if recovery_conf_mtime == self._recovery_conf_mtime and passfile_mtime == self._passfile_mtime:
|
||||
@@ -942,24 +968,32 @@ class ConfigHandler(object):
|
||||
return 'localhost' # connection via localhost is preferred
|
||||
return listen_addresses[0].strip() # can't use localhost, take first address from listen_addresses
|
||||
|
||||
@property
|
||||
def local_connect_kwargs(self) -> Dict[str, Any]:
|
||||
ret = self._local_address.copy()
|
||||
# add all of the other connection settings that are available
|
||||
ret.update(self._superuser)
|
||||
# if the "username" parameter is present, it actually needs to be "user"
|
||||
# for connecting to PostgreSQL
|
||||
if 'username' in self._superuser:
|
||||
ret['user'] = self._superuser['username']
|
||||
del ret['username']
|
||||
# ensure certain Patroni configurations are available
|
||||
ret.update({'dbname': self._postgresql.database,
|
||||
'fallback_application_name': 'Patroni',
|
||||
'connect_timeout': 3,
|
||||
'options': '-c statement_timeout=2000'})
|
||||
return ret
|
||||
|
||||
def resolve_connection_addresses(self) -> None:
|
||||
"""Calculates and sets local and remote connection urls and options.
|
||||
|
||||
This method sets:
|
||||
* :attr:`Postgresql.connection_string <patroni.postgresql.Postgresql.connection_string>` attribute, which
|
||||
is later written to the member key in DCS as ``conn_url``.
|
||||
* :attr:`ConfigHandler.local_replication_address` attribute, which is used for replication connections to
|
||||
local postgres.
|
||||
* :attr:`ConnectionPool.conn_kwargs <patroni.postgresql.connection.ConnectionPool.conn_kwargs>` attribute,
|
||||
which is used for superuser connections to local postgres.
|
||||
|
||||
.. note::
|
||||
If there is a valid directory in ``postgresql.parameters.unix_socket_directories`` in the Patroni
|
||||
configuration and ``postgresql.use_unix_socket`` and/or ``postgresql.use_unix_socket_repl``
|
||||
are set to ``True``, we respectively use unix sockets for superuser and replication connections
|
||||
to local postgres.
|
||||
|
||||
If there is a requirement to use unix sockets, but nothing is set in the
|
||||
``postgresql.parameters.unix_socket_directories``, we omit a ``host`` in connection parameters relying
|
||||
on the ability of ``libpq`` to connect via some default unix socket directory.
|
||||
|
||||
If unix sockets are not requested we "switch" to TCP, prefering to use ``localhost`` if it is possible
|
||||
to deduce that Postgres is listening on a local interface address.
|
||||
|
||||
Otherwise we just used the first address specified in the ``listen_addresses`` GUC.
|
||||
"""
|
||||
port = self._server_parameters['port']
|
||||
tcp_local_address = self._get_tcp_local_address()
|
||||
netloc = self._config.get('connect_address') or tcp_local_address + ':' + port
|
||||
@@ -972,12 +1006,25 @@ class ConfigHandler(object):
|
||||
|
||||
tcp_local_address = {'host': tcp_local_address, 'port': port}
|
||||
|
||||
self._local_address = unix_local_address if self._config.get('use_unix_socket') else tcp_local_address
|
||||
self.local_replication_address = unix_local_address\
|
||||
if self._config.get('use_unix_socket_repl') else tcp_local_address
|
||||
|
||||
self._postgresql.connection_string = uri('postgres', netloc, self._postgresql.database)
|
||||
self._postgresql.set_connection_kwargs(self.local_connect_kwargs)
|
||||
|
||||
local_address = unix_local_address if self._config.get('use_unix_socket') else tcp_local_address
|
||||
local_conn_kwargs = {
|
||||
**local_address,
|
||||
**self._superuser,
|
||||
'dbname': self._postgresql.database,
|
||||
'fallback_application_name': 'Patroni',
|
||||
'connect_timeout': 3,
|
||||
'options': '-c statement_timeout=2000'
|
||||
}
|
||||
# if the "username" parameter is present, it actually needs to be "user" for connecting to PostgreSQL
|
||||
if 'username' in local_conn_kwargs:
|
||||
local_conn_kwargs['user'] = local_conn_kwargs.pop('username')
|
||||
# "notify" connection_pool about the "new" local connection address
|
||||
self._postgresql.connection_pool.conn_kwargs = local_conn_kwargs
|
||||
|
||||
def _get_pg_settings(
|
||||
self, names: Collection[str]
|
||||
|
||||
@@ -2,9 +2,9 @@ import logging
|
||||
|
||||
from contextlib import contextmanager
|
||||
from threading import Lock
|
||||
from typing import Any, Dict, Iterator, List, Union, Tuple, TYPE_CHECKING
|
||||
from typing import Any, Dict, Iterator, List, Optional, Union, Tuple, TYPE_CHECKING
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
from psycopg import Connection as Connection3, Cursor
|
||||
from psycopg import Connection, Cursor
|
||||
from psycopg2 import connection, cursor
|
||||
|
||||
from .. import psycopg
|
||||
@@ -13,27 +13,34 @@ from ..exceptions import PostgresConnectionException
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Connection:
|
||||
"""Helper class to manage connections from Patroni to PostgreSQL.
|
||||
class NamedConnection:
|
||||
"""Helper class to manage ``psycopg`` connections from Patroni to PostgreSQL.
|
||||
|
||||
:ivar server_version: PostgreSQL version in integer format where we are connected to.
|
||||
"""
|
||||
|
||||
server_version: int
|
||||
|
||||
def __init__(self) -> None:
|
||||
"""Create an instance of :class:`Connection` class."""
|
||||
def __init__(self, pool: 'ConnectionPool', name: str, kwargs_override: Optional[Dict[str, Any]]) -> None:
|
||||
"""Create an instance of :class:`NamedConnection` class.
|
||||
|
||||
:param pool: reference to a :class:`ConnectionPool` object.
|
||||
:param name: name of the connection.
|
||||
:param kwargs_override: :class:`dict` object with connection parameters that should be
|
||||
different from default values provided by connection *pool*.
|
||||
"""
|
||||
self._pool = pool
|
||||
self._name = name
|
||||
self._kwargs_override = kwargs_override or {}
|
||||
self._lock = Lock() # used to make sure that only one connection to postgres is established
|
||||
self._connection = None
|
||||
|
||||
def set_conn_kwargs(self, conn_kwargs: Dict[str, Any]) -> None:
|
||||
"""Set connection parameters, like user, password, host, port and so on.
|
||||
@property
|
||||
def _conn_kwargs(self) -> Dict[str, Any]:
|
||||
"""Connection parameters for this :class:`NamedConnection`."""
|
||||
return {**self._pool.conn_kwargs, **self._kwargs_override, 'application_name': f'Patroni {self._name}'}
|
||||
|
||||
:param conn_kwargs: connection parameters as a dictionary.
|
||||
"""
|
||||
self._conn_kwargs = conn_kwargs
|
||||
|
||||
def get(self) -> Union['connection', 'Connection3[Any]']:
|
||||
def get(self) -> Union['connection', 'Connection[Any]']:
|
||||
"""Get ``psycopg``/``psycopg2`` connection object.
|
||||
|
||||
.. note::
|
||||
@@ -43,7 +50,7 @@ class Connection:
|
||||
"""
|
||||
with self._lock:
|
||||
if not self._connection or self._connection.closed != 0:
|
||||
logger.info("establishing a new patroni connection to postgres")
|
||||
logger.info("establishing a new patroni %s connection to postgres", self._name)
|
||||
self._connection = psycopg.connect(**self._conn_kwargs)
|
||||
self.server_version = getattr(self._connection, 'server_version', 0)
|
||||
return self._connection
|
||||
@@ -76,12 +83,72 @@ class Connection:
|
||||
raise exc
|
||||
raise PostgresConnectionException('connection problems') from exc
|
||||
|
||||
def close(self) -> None:
|
||||
"""Close the psycopg connection to postgres."""
|
||||
def close(self, silent: bool = False) -> bool:
|
||||
"""Close the psycopg connection to postgres.
|
||||
|
||||
:param silent: whether the method should not write logs.
|
||||
|
||||
:returns: ``True`` if ``psycopg`` connection was closed, ``False`` otherwise.``
|
||||
"""
|
||||
ret = False
|
||||
if self._connection and self._connection.closed == 0:
|
||||
self._connection.close()
|
||||
logger.info("closed patroni connection to postgres")
|
||||
if not silent:
|
||||
logger.info("closed patroni %s connection to postgres", self._name)
|
||||
ret = True
|
||||
self._connection = None
|
||||
return ret
|
||||
|
||||
|
||||
class ConnectionPool:
|
||||
"""Helper class to manage named connections from Patroni to PostgreSQL.
|
||||
|
||||
The instance keeps named :class:`NamedConnection` objects and parameters that must be used for new connections.
|
||||
"""
|
||||
|
||||
def __init__(self) -> None:
|
||||
"""Create an instance of :class:`ConnectionPool` class."""
|
||||
self._lock = Lock()
|
||||
self._connections: Dict[str, NamedConnection] = {}
|
||||
self._conn_kwargs: Dict[str, Any] = {}
|
||||
|
||||
@property
|
||||
def conn_kwargs(self) -> Dict[str, Any]:
|
||||
"""Connection parameters that must be used for new ``psycopg`` connections."""
|
||||
with self._lock:
|
||||
return self._conn_kwargs.copy()
|
||||
|
||||
@conn_kwargs.setter
|
||||
def conn_kwargs(self, value: Dict[str, Any]) -> None:
|
||||
"""Set new connection parameters.
|
||||
|
||||
:param value: :class:`dict` object with connection parameters.
|
||||
"""
|
||||
with self._lock:
|
||||
self._conn_kwargs = value
|
||||
|
||||
def get(self, name: str, kwargs_override: Optional[Dict[str, Any]] = None) -> NamedConnection:
|
||||
"""Get a new named :class:`NamedConnection` object from the pool.
|
||||
|
||||
.. note::
|
||||
Creates a new :class:`NamedConnection` object if it doesn't yet exist in the pool.
|
||||
|
||||
:param name: name of the connection.
|
||||
:param kwargs_override: :class:`dict` object with connection parameters that should be
|
||||
different from default values provided by :attr:`conn_kwargs`.
|
||||
|
||||
:returns: :class:`NamedConnection` object.
|
||||
"""
|
||||
with self._lock:
|
||||
if name not in self._connections:
|
||||
self._connections[name] = NamedConnection(self, name, kwargs_override)
|
||||
return self._connections[name]
|
||||
|
||||
def close(self) -> None:
|
||||
"""Close all named connections from Patroni to PostgreSQL registered in the pool."""
|
||||
with self._lock:
|
||||
if any(conn.close(True) for conn in self._connections.values()):
|
||||
logger.info("closed patroni connections to postgres")
|
||||
|
||||
|
||||
@contextmanager
|
||||
|
||||
+11
-13
@@ -384,8 +384,7 @@ class SlotsHandler:
|
||||
|
||||
:yields: connection cursor object, note implementation varies depending on version of :mod:`psycopg`.
|
||||
"""
|
||||
conn_kwargs = self._postgresql.config.local_connect_kwargs
|
||||
conn_kwargs.update(kwargs)
|
||||
conn_kwargs = {**self._postgresql.connection_pool.conn_kwargs, **kwargs}
|
||||
with get_connection_cursor(**conn_kwargs) as cur:
|
||||
yield cur
|
||||
|
||||
@@ -435,7 +434,7 @@ class SlotsHandler:
|
||||
self._advance = SlotsAdvanceThread(self)
|
||||
return self._advance.schedule(slots)
|
||||
|
||||
def _ensure_logical_slots_replica(self, cluster: Cluster, slots: Dict[str, Any]) -> List[str]:
|
||||
def _ensure_logical_slots_replica(self, slots: Dict[str, Any]) -> List[str]:
|
||||
"""Update logical *slots* on replicas.
|
||||
|
||||
If the logical slot already exists, copy state information into the replication slots structure stored in the
|
||||
@@ -445,7 +444,6 @@ class SlotsHandler:
|
||||
As logical slots can only be created when the primary is available, pass the list of slots that need to be
|
||||
copied back to the caller. They will be created on replicas with :meth:`SlotsHandler.copy_logical_slots`.
|
||||
|
||||
:param cluster: object containing stateful information for the cluster.
|
||||
:param slots: A dictionary mapping slot name to slot attributes. This method only considers a slot
|
||||
if the value is a dictionary with the key ``type`` and a value of ``logical``.
|
||||
|
||||
@@ -460,15 +458,16 @@ class SlotsHandler:
|
||||
continue
|
||||
|
||||
# If the logical already exists, copy some information about it into the original structure
|
||||
if self._replication_slots.get(name, {}).get('datoid'):
|
||||
if name in self._replication_slots and compare_slots(value, self._replication_slots[name]):
|
||||
self._copy_items(self._replication_slots[name], value)
|
||||
if cluster.slots and name in cluster.slots:
|
||||
if 'lsn' in value: # The slot has feedback in DCS
|
||||
try: # Skip slots that don't need to be advanced
|
||||
if value['confirmed_flush_lsn'] < int(cluster.slots[name]):
|
||||
advance_slots[value['database']][name] = int(cluster.slots[name])
|
||||
if value['confirmed_flush_lsn'] < int(value['lsn']):
|
||||
advance_slots[value['database']][name] = int(value['lsn'])
|
||||
except Exception as e:
|
||||
logger.error('Failed to parse "%s": %r', cluster.slots[name], e)
|
||||
elif cluster.slots and name in cluster.slots: # We want to copy only slots with feedback in a DCS
|
||||
logger.error('Failed to parse "%s": %r', value['lsn'], e)
|
||||
elif name not in self._replication_slots and 'lsn' in value:
|
||||
# We want to copy only slots with feedback in a DCS
|
||||
create_slots.append(name)
|
||||
|
||||
# Slots to be copied from the primary should be removed from the *slots* structure,
|
||||
@@ -513,10 +512,9 @@ class SlotsHandler:
|
||||
if self._postgresql.is_primary():
|
||||
self._logical_slots_processing_queue.clear()
|
||||
self._ensure_logical_slots_primary(slots)
|
||||
elif cluster.slots and slots:
|
||||
else:
|
||||
self.check_logical_slots_readiness(cluster, replicatefrom)
|
||||
|
||||
ret = self._ensure_logical_slots_replica(cluster, slots)
|
||||
ret = self._ensure_logical_slots_replica(slots)
|
||||
|
||||
self._replication_slots = slots
|
||||
except Exception:
|
||||
|
||||
@@ -209,7 +209,7 @@ class _ReplicaList(List[_Replica]):
|
||||
# 2. can be mapped to a ``Member`` of the ``Cluster``:
|
||||
# a. ``Member`` doesn't have ``nosync`` tag set;
|
||||
# b. PostgreSQL on the member is known to be running and accepting client connections.
|
||||
if member and row[sort_col] is not None and member.is_running and not member.tags.get('nosync', False):
|
||||
if member and row[sort_col] is not None and member.is_running and not member.nosync:
|
||||
self.append(_Replica(row['pid'], row['application_name'],
|
||||
row['sync_state'], row[sort_col], bool(member.nofailover)))
|
||||
|
||||
|
||||
@@ -290,7 +290,7 @@ def _load_postgres_gucs_validators() -> None:
|
||||
Any problem faced while reading or parsing files will be logged as a ``WARNING`` by the child function, and the
|
||||
corresponding file or validator will be ignored.
|
||||
|
||||
By default Patroni only ships the file ``0_postgres.yml``, which contains Community Postgres GUCs validators, but
|
||||
By default, Patroni only ships the file ``0_postgres.yml``, which contains Community Postgres GUCs validators, but
|
||||
that behavior can be extended. For example: if a vendor wants to add GUC validators to Patroni for covering a custom
|
||||
Postgres build, then they can create their custom YAML files under ``available_parameters`` directory.
|
||||
|
||||
@@ -300,8 +300,10 @@ def _load_postgres_gucs_validators() -> None:
|
||||
writes them to ``postgresql.conf`` if running PG 12 and above).
|
||||
|
||||
Then, each of these sections, if specified, may contain one or more attributes with the following structure:
|
||||
|
||||
* key: the name of a GUC;
|
||||
* value: a list of validators. Each item in the list must contain a ``type`` attribute, which must be one among:
|
||||
|
||||
* ``Bool``; or
|
||||
* ``Integer``; or
|
||||
* ``Real``; or
|
||||
@@ -313,6 +315,7 @@ def _load_postgres_gucs_validators() -> None:
|
||||
class in this module.
|
||||
|
||||
.. seealso::
|
||||
|
||||
* :class:`Bool`;
|
||||
* :class:`Integer`;
|
||||
* :class:`Real`;
|
||||
@@ -325,61 +328,62 @@ def _load_postgres_gucs_validators() -> None:
|
||||
This is a sample content for an YAML file based on Postgres GUCs, showing each of the supported types and
|
||||
sections:
|
||||
|
||||
```yaml
|
||||
parameters:
|
||||
archive_command:
|
||||
- type: String
|
||||
version_from: 90300
|
||||
version_till: null
|
||||
archive_mode:
|
||||
- type: Bool
|
||||
version_from: 90300
|
||||
version_till: 90500
|
||||
- type: EnumBool
|
||||
version_from: 90500
|
||||
version_till: null
|
||||
possible_values:
|
||||
- always
|
||||
archive_timeout:
|
||||
- type: Integer
|
||||
version_from: 90300
|
||||
version_till: null
|
||||
min_val: 0
|
||||
max_val: 1073741823
|
||||
unit: s
|
||||
autovacuum_vacuum_cost_delay:
|
||||
- type: Integer
|
||||
version_from: 90300
|
||||
version_till: 120000
|
||||
min_val: -1
|
||||
max_val: 100
|
||||
unit: ms
|
||||
- type: Real
|
||||
version_from: 120000
|
||||
version_till: null
|
||||
min_val: -1
|
||||
max_val: 100
|
||||
unit: ms
|
||||
client_min_messages:
|
||||
- type: Enum
|
||||
version_from: 90300
|
||||
version_till: null
|
||||
possible_values:
|
||||
- debug5
|
||||
- debug4
|
||||
- debug3
|
||||
- debug2
|
||||
- debug1
|
||||
- log
|
||||
- notice
|
||||
- warning
|
||||
- error
|
||||
recovery_parameters:
|
||||
archive_cleanup_command:
|
||||
- type: String
|
||||
version_from: 90300
|
||||
version_till: null
|
||||
```
|
||||
.. code-block:: yaml
|
||||
|
||||
parameters:
|
||||
archive_command:
|
||||
- type: String
|
||||
version_from: 90300
|
||||
version_till: null
|
||||
archive_mode:
|
||||
- type: Bool
|
||||
version_from: 90300
|
||||
version_till: 90500
|
||||
- type: EnumBool
|
||||
version_from: 90500
|
||||
version_till: null
|
||||
possible_values:
|
||||
- always
|
||||
archive_timeout:
|
||||
- type: Integer
|
||||
version_from: 90300
|
||||
version_till: null
|
||||
min_val: 0
|
||||
max_val: 1073741823
|
||||
unit: s
|
||||
autovacuum_vacuum_cost_delay:
|
||||
- type: Integer
|
||||
version_from: 90300
|
||||
version_till: 120000
|
||||
min_val: -1
|
||||
max_val: 100
|
||||
unit: ms
|
||||
- type: Real
|
||||
version_from: 120000
|
||||
version_till: null
|
||||
min_val: -1
|
||||
max_val: 100
|
||||
unit: ms
|
||||
client_min_messages:
|
||||
- type: Enum
|
||||
version_from: 90300
|
||||
version_till: null
|
||||
possible_values:
|
||||
- debug5
|
||||
- debug4
|
||||
- debug3
|
||||
- debug2
|
||||
- debug1
|
||||
- log
|
||||
- notice
|
||||
- warning
|
||||
- error
|
||||
recovery_parameters:
|
||||
archive_cleanup_command:
|
||||
- type: String
|
||||
version_from: 90300
|
||||
version_till: null
|
||||
|
||||
"""
|
||||
conf_dir = os.path.join(
|
||||
os.path.dirname(os.path.abspath(__file__)),
|
||||
@@ -434,13 +438,15 @@ def _transform_parameter_value(validators: MutableMapping[str, Tuple[_Transforma
|
||||
:param value: value of the Postgres GUC.
|
||||
:param available_gucs: a set of all GUCs available in Postgres *version*. Each item is the name of a Postgres
|
||||
GUC. Used for a couple purposes:
|
||||
|
||||
* Disallow writing GUCs to ``postgresql.conf`` (or ``recovery.conf``) that does not exist in Postgres *version*;
|
||||
* Avoid ignoring GUC *name* if it does not have a validator in *validators*, but is a valid GUC in Postgres
|
||||
*version*.
|
||||
*version*.
|
||||
|
||||
:returns: the return value may be one among:
|
||||
* *value* transformed to the expected format for GUC *name* in Postgres *version*, if *name* is present in
|
||||
*available_gucs* and has a validator in *validators* for the corresponding Postgres *version*; or
|
||||
|
||||
* *value* transformed to the expected format for GUC *name* in Postgres *version*, if *name* is present
|
||||
in *available_gucs* and has a validator in *validators* for the corresponding Postgres *version*; or
|
||||
* The own *value* if *name* is present in *available_gucs* but not in *validators*; or
|
||||
* ``None`` if *name* is not present in *available_gucs*.
|
||||
"""
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
"""Tags handling."""
|
||||
import abc
|
||||
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
|
||||
class Tags(abc.ABC):
|
||||
"""An abstract class that encapsulates all the ``tags`` logic.
|
||||
|
||||
Child classes that want to use provided facilities must implement ``tags`` abstract property.
|
||||
"""
|
||||
|
||||
@staticmethod
|
||||
def _filter_tags(tags: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""Get tags configured for this node, if any.
|
||||
|
||||
Handle both predefined Patroni tags and custom defined tags.
|
||||
|
||||
.. note::
|
||||
A custom tag is any tag added to the configuration ``tags`` section that is not one of ``clonefrom``,
|
||||
``nofailover``, ``noloadbalance`` or ``nosync``.
|
||||
|
||||
For the Patroni predefined tags, the returning object will only contain them if they are enabled as they
|
||||
all are boolean values that default to disabled.
|
||||
|
||||
:returns: a dictionary of tags set for this node. The key is the tag name, and the value is the corresponding
|
||||
tag value.
|
||||
"""
|
||||
return {tag: value for tag, value in tags.items()
|
||||
if tag not in ('clonefrom', 'nofailover', 'noloadbalance', 'nosync') or value}
|
||||
|
||||
@property
|
||||
@abc.abstractmethod
|
||||
def tags(self) -> Dict[str, Any]:
|
||||
"""Configured tags.
|
||||
|
||||
Must be implemented in a child class.
|
||||
"""
|
||||
raise NotImplementedError # pragma: no cover
|
||||
|
||||
@property
|
||||
def clonefrom(self) -> bool:
|
||||
"""``True`` if ``clonefrom`` tag is ``True``, else ``False``."""
|
||||
return self.tags.get('clonefrom', False)
|
||||
|
||||
@property
|
||||
def nofailover(self) -> bool:
|
||||
"""``True`` if ``nofailover`` is ``True``, else ``False``."""
|
||||
return bool(self.tags.get('nofailover', False))
|
||||
|
||||
@property
|
||||
def noloadbalance(self) -> bool:
|
||||
"""``True`` if ``noloadbalance`` is ``True``, else ``False``."""
|
||||
return bool(self.tags.get('noloadbalance', False))
|
||||
|
||||
@property
|
||||
def nosync(self) -> bool:
|
||||
"""``True`` if ``nosync`` is ``True``, else ``False``."""
|
||||
return bool(self.tags.get('nosync', False))
|
||||
|
||||
@property
|
||||
def replicatefrom(self) -> Optional[str]:
|
||||
"""Value of ``replicatefrom`` tag, if any."""
|
||||
return self.tags.get('replicatefrom')
|
||||
+26
-2
@@ -9,6 +9,8 @@
|
||||
:var DBL_RE: regular expression to match double precision numbers, signed or unsigned. Matches scientific notation too.
|
||||
:var WHITESPACE_RE: regular expression to match whitespace characters
|
||||
"""
|
||||
import datetime
|
||||
import dateutil.parser
|
||||
import errno
|
||||
import logging
|
||||
import os
|
||||
@@ -20,6 +22,7 @@ import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
from enum import Enum
|
||||
from shlex import split
|
||||
|
||||
from typing import Any, Callable, Dict, Iterator, List, Optional, Union, Tuple, Type, TYPE_CHECKING
|
||||
@@ -836,8 +839,9 @@ def cluster_as_json(cluster: 'Cluster', global_config: Optional['GlobalConfig']
|
||||
ret['pause'] = True
|
||||
if cluster.failover and cluster.failover.scheduled_at:
|
||||
ret['scheduled_switchover'] = {'at': cluster.failover.scheduled_at.isoformat()}
|
||||
if cluster.failover.leader:
|
||||
ret['scheduled_switchover']['from'] = cluster.failover.leader
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
assert cluster.failover.leader
|
||||
ret['scheduled_switchover']['from'] = cluster.failover.leader
|
||||
if cluster.failover.candidate:
|
||||
ret['scheduled_switchover']['to'] = cluster.failover.candidate
|
||||
return ret
|
||||
@@ -1061,3 +1065,23 @@ def get_major_version(bin_dir: Optional[str] = None, bin_name: str = 'postgres')
|
||||
if TYPE_CHECKING: # pragma: no cover
|
||||
assert version is not None
|
||||
return '.'.join([version.group(1), version.group(3)]) if int(version.group(1)) < 10 else version.group(1)
|
||||
|
||||
|
||||
class ParseScheduleErrors(Enum):
|
||||
NO_TIMEZONE = ('Timezone information is mandatory for the scheduled {action}', 400)
|
||||
SCHEDULED_IN_PAST = ('Cannot schedule {action} in the past', 422)
|
||||
PARSING_ERROR = ('Unable to parse scheduled timestamp. It should be in an unambiguous format, e.g. ISO 8601', 422)
|
||||
|
||||
|
||||
def parse_schedule(schedule: Optional[str]) -> Tuple[Optional[ParseScheduleErrors], Optional[datetime.datetime]]:
|
||||
scheduled_at = None
|
||||
if schedule is not None:
|
||||
try:
|
||||
scheduled_at = dateutil.parser.parse(schedule)
|
||||
if scheduled_at.tzinfo is None:
|
||||
return ParseScheduleErrors.NO_TIMEZONE, scheduled_at
|
||||
elif scheduled_at < datetime.datetime.now(tzutc):
|
||||
return ParseScheduleErrors.SCHEDULED_IN_PAST, scheduled_at
|
||||
except (ValueError, TypeError):
|
||||
return ParseScheduleErrors.PARSING_ERROR, scheduled_at
|
||||
return None, scheduled_at
|
||||
|
||||
+43
-37
@@ -333,15 +333,17 @@ class Case(object):
|
||||
"""Create a :class:`Case` object.
|
||||
|
||||
:param schema: the schema for validating a set of attributes that may be available in the configuration.
|
||||
Each key is the configuration that is available in a given scope and that should be validated, and the
|
||||
related value is the validation function or expected type.
|
||||
Each key is the configuration that is available in a given scope and that should be validated,
|
||||
and the related value is the validation function or expected type.
|
||||
|
||||
:Example:
|
||||
|
||||
Case({
|
||||
"host": validate_host_port,
|
||||
"url": str,
|
||||
})
|
||||
.. code-block:: python
|
||||
|
||||
Case({
|
||||
"host": validate_host_port,
|
||||
"url": str,
|
||||
})
|
||||
|
||||
That will check that ``host`` configuration, if given, is valid based on :func:`validate_host_port`, and will
|
||||
also check that ``url`` configuration, if given, is a ``str`` instance.
|
||||
@@ -363,14 +365,16 @@ class Or(object):
|
||||
|
||||
:Example:
|
||||
|
||||
Or("host", "hosts"): Case({
|
||||
"host": validate_host_port,
|
||||
"hosts": Or(comma_separated_host_port, [validate_host_port]),
|
||||
})
|
||||
.. code-block:: python
|
||||
|
||||
The outer :class:`Or` is used to define that ``host`` and ``hosts`` are possible options in this scope.
|
||||
The inner :class`Or` in the ``hosts`` key value is used to define that ``hosts`` option is valid if either of
|
||||
:func:`comma_separated_host_port` or :func:`validate_host_port` succeed to validate it.
|
||||
Or("host", "hosts"): Case({
|
||||
"host": validate_host_port,
|
||||
"hosts": Or(comma_separated_host_port, [validate_host_port]),
|
||||
})
|
||||
|
||||
The outer :class:`Or` is used to define that ``host`` and ``hosts`` are possible options in this scope.
|
||||
The inner :class`Or` in the ``hosts`` key value is used to define that ``hosts`` option is valid if either
|
||||
of :func:`comma_separated_host_port` or :func:`validate_host_port` succeed to validate it.
|
||||
"""
|
||||
self.args = args
|
||||
|
||||
@@ -535,32 +539,34 @@ class Schema(object):
|
||||
|
||||
:Example:
|
||||
|
||||
Schema({
|
||||
"application_name": str,
|
||||
"bind": {
|
||||
"host": validate_host,
|
||||
"port": int,
|
||||
},
|
||||
"aliases": [str],
|
||||
Optional("data_directory"): "/var/lib/myapp",
|
||||
Or("log_to_file", "log_to_db"): Case({
|
||||
"log_to_file": bool,
|
||||
"log_to_db": bool,
|
||||
}),
|
||||
"version": Or(int, float),
|
||||
})
|
||||
.. code-block:: python
|
||||
|
||||
This sample schema defines that your YAML configuration follows these rules:
|
||||
Schema({
|
||||
"application_name": str,
|
||||
"bind": {
|
||||
"host": validate_host,
|
||||
"port": int,
|
||||
},
|
||||
"aliases": [str],
|
||||
Optional("data_directory"): "/var/lib/myapp",
|
||||
Or("log_to_file", "log_to_db"): Case({
|
||||
"log_to_file": bool,
|
||||
"log_to_db": bool,
|
||||
}),
|
||||
"version": Or(int, float),
|
||||
})
|
||||
|
||||
* It must contain an ``application_name`` entry which value should be a :class:`str` instance;
|
||||
* It must contain a ``bind.host`` entry which value should be valid as per function ``validate_host``;
|
||||
* It must contain a ``bind.port`` entry which value should be an :class:`int` instance;
|
||||
* It must contain a ``aliases`` entry which value should be a :class:`list` of :class:`str` instances;
|
||||
* It may optionally contain a ``data_directory`` entry, with a value which should be a string;
|
||||
* It must contain at least one of ``log_to_file`` or ``log_to_db``, with a value which should be a
|
||||
:class:`bool` instance;
|
||||
* It must contain a ``version`` entry which value should be either an :class:`int` or a :class:`float`
|
||||
instance.
|
||||
This sample schema defines that your YAML configuration follows these rules:
|
||||
|
||||
* It must contain an ``application_name`` entry which value should be a :class:`str` instance;
|
||||
* It must contain a ``bind.host`` entry which value should be valid as per function ``validate_host``;
|
||||
* It must contain a ``bind.port`` entry which value should be an :class:`int` instance;
|
||||
* It must contain a ``aliases`` entry which value should be a :class:`list` of :class:`str` instances;
|
||||
* It may optionally contain a ``data_directory`` entry, with a value which should be a string;
|
||||
* It must contain at least one of ``log_to_file`` or ``log_to_db``, with a value which should be a
|
||||
:class:`bool` instance;
|
||||
* It must contain a ``version`` entry which value should be either an :class:`int` or a :class:`float`
|
||||
instance.
|
||||
"""
|
||||
self.validator = validator
|
||||
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
sphinx>=4
|
||||
sphinx_rtd_theme>1
|
||||
sphinxcontrib-apidoc
|
||||
sphinx-github-style
|
||||
pyyaml
|
||||
sphinx-github-style<1.0.3
|
||||
|
||||
+1
-1
@@ -104,7 +104,7 @@ class MockCursor(object):
|
||||
elif sql.startswith('SELECT slot_name, slot_type, datname, plugin, catalog_xmin'):
|
||||
self.results = [('ls', 'logical', 'a', 'b', 100, 500, b'123456')]
|
||||
elif sql.startswith('SELECT slot_name'):
|
||||
self.results = [('blabla', 'physical'), ('foobar', 'physical'), ('ls', 'logical', 'a', 'b', 5, 100, 500)]
|
||||
self.results = [('blabla', 'physical'), ('foobar', 'physical'), ('ls', 'logical', 'b', 'a', 5, 100, 500)]
|
||||
elif sql.startswith('WITH slots AS (SELECT slot_name, active'):
|
||||
self.results = [(False, True)] if self.rowcount == 1 else []
|
||||
elif sql.startswith('SELECT CASE WHEN pg_catalog.pg_is_in_recovery()'):
|
||||
|
||||
+184
-55
@@ -11,18 +11,43 @@ from socketserver import ThreadingMixIn
|
||||
from patroni.api import RestApiHandler, RestApiServer
|
||||
from patroni.config import GlobalConfig
|
||||
from patroni.dcs import ClusterConfig, Member
|
||||
from patroni.exceptions import PostgresConnectionException
|
||||
from patroni.ha import _MemberStatus
|
||||
from patroni.utils import RetryFailedError, tzutc
|
||||
from patroni.manual_failover import ManualFailoverPrecheckStatus
|
||||
from patroni.psycopg import OperationalError
|
||||
from patroni.utils import ParseScheduleErrors, RetryFailedError, tzutc
|
||||
|
||||
from .test_ha import get_cluster_initialized_without_leader
|
||||
from . import MockConnect, psycopg_connect
|
||||
from .test_ha import get_cluster_initialized_without_leader, get_cluster_initialized_with_leader
|
||||
|
||||
|
||||
future_restart_time = datetime.datetime.now(tzutc) + datetime.timedelta(days=5)
|
||||
postmaster_start_time = datetime.datetime.now(tzutc)
|
||||
|
||||
|
||||
class MockPostgresql(object):
|
||||
class MockConnection:
|
||||
|
||||
@staticmethod
|
||||
def get(*args):
|
||||
return psycopg_connect()
|
||||
|
||||
@staticmethod
|
||||
def query(sql, *params):
|
||||
return [(postmaster_start_time, 0, '', 0, '', False, postmaster_start_time, 'streaming', None,
|
||||
'[{"application_name":"walreceiver","client_addr":"1.2.3.4",'
|
||||
+ '"state":"streaming","sync_state":"async","sync_priority":0}]')]
|
||||
|
||||
|
||||
class MockConnectionPool:
|
||||
|
||||
@staticmethod
|
||||
def get(*args):
|
||||
return MockConnection()
|
||||
|
||||
|
||||
class MockPostgresql:
|
||||
|
||||
connection_pool = MockConnectionPool()
|
||||
name = 'test'
|
||||
state = 'running'
|
||||
role = 'primary'
|
||||
@@ -54,12 +79,6 @@ class MockPostgresql(object):
|
||||
def replication_state_from_parameters(*args):
|
||||
return 'streaming'
|
||||
|
||||
@staticmethod
|
||||
def query(sql, *params, retry=False):
|
||||
return [(postmaster_start_time, 0, '', 0, '', False, postmaster_start_time, 'streaming', None,
|
||||
'[{"application_name":"walreceiver","client_addr":"1.2.3.4",'
|
||||
+ '"state":"streaming","sync_state":"async","sync_priority":0}]')]
|
||||
|
||||
|
||||
class MockWatchdog(object):
|
||||
is_healthy = False
|
||||
@@ -100,7 +119,7 @@ class MockHa(object):
|
||||
|
||||
@staticmethod
|
||||
def fetch_nodes_statuses(members):
|
||||
return [_MemberStatus(None, True, None, 0, 0, None, {}, False)]
|
||||
return [_MemberStatus(None, True, None, 0, {})]
|
||||
|
||||
@staticmethod
|
||||
def schedule_future_restart(data):
|
||||
@@ -122,6 +141,9 @@ class MockHa(object):
|
||||
def is_paused():
|
||||
return True
|
||||
|
||||
def has_members_eligible_to_promote(*args, **kwargs):
|
||||
return True
|
||||
|
||||
|
||||
class MockLogger(object):
|
||||
|
||||
@@ -487,7 +509,7 @@ class TestRestApiHandler(unittest.TestCase):
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
def test_RestApiServer_query(self):
|
||||
with patch.object(MockPostgresql, 'query', Mock(side_effect=RetryFailedError('bla'))):
|
||||
with patch.object(MockConnection, 'query', Mock(side_effect=RetryFailedError('bla'))):
|
||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /patroni'))
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
@@ -498,86 +520,185 @@ class TestRestApiHandler(unittest.TestCase):
|
||||
|
||||
post = 'POST /switchover HTTP/1.0' + self._authorization + '\nContent-Length: '
|
||||
|
||||
MockRestApiServer(RestApiHandler, post + '7\n\n{"1":2}')
|
||||
# Invalid content
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, post + '7\n\n{"1":2}')
|
||||
response_mock.assert_called_with(*ManualFailoverPrecheckStatus.SWITCHOVER_NO_LEADER.value[::-1])
|
||||
|
||||
# Empty content
|
||||
request = post + '0\n\n'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
|
||||
cluster.leader.name = 'postgresql1'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# [Switchover without a candidate]
|
||||
|
||||
cluster.leader.name = 'postgresql1'
|
||||
request = post + '25\n\n{"leader": "postgresql1"}'
|
||||
|
||||
with patch.object(GlobalConfig, 'is_paused', PropertyMock(return_value=True)):
|
||||
# No candidate in pause mode
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock, \
|
||||
patch.object(GlobalConfig, 'is_paused', PropertyMock(return_value=True)):
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(*ManualFailoverPrecheckStatus.SWITCHOVER_PAUSE_NO_CANDIDATE.value[::-1])
|
||||
|
||||
for is_synchronous_mode in (True, False):
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=is_synchronous_mode)):
|
||||
# No healthy nodes to promote in both sync and async mode
|
||||
for is_synchronous_mode, response in (
|
||||
(True, ManualFailoverPrecheckStatus.NO_SYNC_CANDIDATE.value[0].format(action='switchover')),
|
||||
(False, ManualFailoverPrecheckStatus.ONLY_LEADER.value[0].format(action='switchover'))):
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=is_synchronous_mode)), \
|
||||
patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(412, response)
|
||||
|
||||
cluster.leader.name = 'postgresql2'
|
||||
request = post + '53\n\n{"leader": "postgresql1", "candidate": "postgresql2"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# [Switchover to the candidate specified]
|
||||
|
||||
# Candidate to promote is the same as the leader specified
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
request = post + '53\n\n{"leader": "postgresql2", "candidate": "postgresql2"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(*ManualFailoverPrecheckStatus.SWITCHOVER_TO_LEADER.value[::-1])
|
||||
|
||||
# Current leader is different from the one specified
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
cluster.leader.name = 'postgresql2'
|
||||
request = post + '53\n\n{"leader": "postgresql1", "candidate": "postgresql2"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(
|
||||
ManualFailoverPrecheckStatus.LEADER_NOT_MEMBER.value[1],
|
||||
ManualFailoverPrecheckStatus.LEADER_NOT_MEMBER.value[0].format(leader='postgresql1',
|
||||
cluster_name='dummy'))
|
||||
|
||||
# Candidate to promote is not a sync standby/a member of the cluster
|
||||
cluster.leader.name = 'postgresql1'
|
||||
cluster.sync.matches.return_value = False
|
||||
for is_synchronous_mode in (True, False):
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=is_synchronous_mode)):
|
||||
for is_synchronous_mode, response in (
|
||||
(True, ManualFailoverPrecheckStatus.CANDIDATE_NOT_SYNC_STANDBY.value[0]),
|
||||
(False, ManualFailoverPrecheckStatus.CANDIDATE_NOT_MEMEBER.value[0].format(candidate="postgresql2",
|
||||
cluster_name='dummy'))):
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=is_synchronous_mode)), \
|
||||
patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(412, response)
|
||||
|
||||
cluster.members = [Member(0, 'postgresql0', 30, {'api_url': 'http'}),
|
||||
Member(0, 'postgresql2', 30, {'api_url': 'http'})]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
|
||||
cluster.failover = None
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# Cluster has no leader
|
||||
cluster.leader.name = None
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
request = post + '53\n\n{"leader": "postgresql1"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(
|
||||
ManualFailoverPrecheckStatus.CLUSTER_NO_LEADER.value[1],
|
||||
ManualFailoverPrecheckStatus.CLUSTER_NO_LEADER.value[0].format(leader='leader', cluster_name='dummy'))
|
||||
|
||||
dcs.get_cluster.side_effect = [cluster]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
cluster.leader.name = 'postgresql1'
|
||||
|
||||
cluster2 = cluster.copy()
|
||||
cluster2.leader.name = 'postgresql0'
|
||||
cluster2.is_unlocked.return_value = False
|
||||
dcs.get_cluster.side_effect = [cluster, cluster2]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# Failover key is empty in DCS
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
cluster.failover = None
|
||||
request = post + '53\n\n{"leader": "postgresql1", "candidate": "postgresql2"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(503, 'Switchover failed')
|
||||
|
||||
cluster2.leader.name = 'postgresql2'
|
||||
dcs.get_cluster.side_effect = [cluster, cluster2]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# Result polling failed
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
dcs.get_cluster.side_effect = [cluster]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(503, 'Switchover status unknown')
|
||||
|
||||
# Switchover to a node different from the candidate specified
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
cluster2 = cluster.copy()
|
||||
cluster2.leader.name = 'postgresql0'
|
||||
cluster2.is_unlocked.return_value = False
|
||||
dcs.get_cluster.side_effect = [cluster, cluster2]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(200, 'Switched over to "postgresql0" instead of "postgresql2"')
|
||||
|
||||
# Successful switchover to the candidate
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
cluster2.leader.name = 'postgresql2'
|
||||
dcs.get_cluster.side_effect = [cluster, cluster2]
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(200, 'Successfully switched over to "postgresql2"')
|
||||
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
dcs.manual_failover.return_value = False
|
||||
dcs.get_cluster.side_effect = None
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(503, 'failed to write failover key into DCS')
|
||||
|
||||
dcs.get_cluster.side_effect = None
|
||||
dcs.manual_failover.return_value = False
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
dcs.manual_failover.return_value = True
|
||||
|
||||
with patch.object(MockHa, 'fetch_nodes_statuses', Mock(return_value=[])):
|
||||
# Candidate is not healthy to be promoted
|
||||
with patch.object(MockHa, 'has_members_eligible_to_promote', Mock(return_value=False)), \
|
||||
patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(
|
||||
ManualFailoverPrecheckStatus.NO_GOOD_CANDIDATES.value[1],
|
||||
ManualFailoverPrecheckStatus.NO_GOOD_CANDIDATES.value[0].format(action='switchover'))
|
||||
|
||||
# [Scheduled switchover]
|
||||
|
||||
# Valid future date
|
||||
request = post + '103\n\n{"leader": "postgresql1", "member": "postgresql2",' +\
|
||||
' "scheduled_at": "6016-02-15T18:13:30.568224+01:00"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
with patch.object(GlobalConfig, 'is_paused', PropertyMock(return_value=True)), \
|
||||
patch.object(MockPatroni, 'dcs') as d:
|
||||
d.manual_failover.return_value = False
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
request = post + '103\n\n{"leader": "postgresql1", "member": "postgresql2",' + \
|
||||
' "scheduled_at": "6016-02-15T18:13:30.568224+01:00"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(202, 'Switchover scheduled')
|
||||
|
||||
# Exception: No timezone specified
|
||||
request = post + '97\n\n{"leader": "postgresql1", "member": "postgresql2",' +\
|
||||
' "scheduled_at": "6016-02-15T18:13:30.568224"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
# Scheduled in pause mode
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock, \
|
||||
patch.object(GlobalConfig, 'is_paused', PropertyMock(return_value=True)):
|
||||
dcs.manual_failover.return_value = False
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(*ManualFailoverPrecheckStatus.SCHEDULED_SWITCHOVER_PAUSE.value[::-1])
|
||||
|
||||
# No timezone specified
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
request = post + '97\n\n{"leader": "postgresql1", "member": "postgresql2",' + \
|
||||
' "scheduled_at": "6016-02-15T18:13:30.568224"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
response_mock.assert_called_with(
|
||||
ParseScheduleErrors.NO_TIMEZONE.value[1],
|
||||
ParseScheduleErrors.NO_TIMEZONE.value[0].format(action='switchover'))
|
||||
|
||||
# Exception: Scheduled in the past
|
||||
request = post + '103\n\n{"leader": "postgresql1", "member": "postgresql2", "scheduled_at": "'
|
||||
MockRestApiServer(RestApiHandler, request + '1016-02-15T18:13:30.568224+01:00"}')
|
||||
|
||||
# Scheduled in the past
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, request + '1016-02-15T18:13:30.568224+01:00"}')
|
||||
response_mock.assert_called_with(
|
||||
ParseScheduleErrors.SCHEDULED_IN_PAST.value[1],
|
||||
ParseScheduleErrors.SCHEDULED_IN_PAST.value[0].format(action='switchover'))
|
||||
|
||||
# Invalid date
|
||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, request + '2010-02-29T18:13:30.568224+01:00"}'))
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, request + '2010-02-29T18:13:30.568224+01:00"}')
|
||||
response_mock.assert_called_with(*ParseScheduleErrors.PARSING_ERROR.value[::-1])
|
||||
|
||||
def test_do_POST_failover(self):
|
||||
@patch.object(MockPatroni, 'dcs')
|
||||
def test_do_POST_failover(self, mock_dcs):
|
||||
post = 'POST /failover HTTP/1.0' + self._authorization + '\nContent-Length: '
|
||||
MockRestApiServer(RestApiHandler, post + '14\n\n{"leader":"1"}')
|
||||
MockRestApiServer(RestApiHandler, post + '37\n\n{"candidate":"2","scheduled_at": "1"}')
|
||||
cluster = mock_dcs.get_cluster.return_value
|
||||
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, post + '19\n\n{"leader":"leader"}')
|
||||
response_mock.assert_called_once_with(*ManualFailoverPrecheckStatus.FAILOVER_NO_CANDIDATE.value[::-1])
|
||||
|
||||
with patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, post + '37\n\n{"candidate":"2","scheduled_at": "1"}')
|
||||
response_mock.assert_called_once_with(*ManualFailoverPrecheckStatus.SCHEDULED_FAILOVER.value[::-1])
|
||||
|
||||
# Candidate is not healthy to be promoted
|
||||
cluster.members = [Member(0, 'postgresql0', 30, {'api_url': 'http'}),
|
||||
Member(0, 'postgresql2', 30, {'api_url': 'http'})]
|
||||
with patch.object(MockHa, 'has_members_eligible_to_promote', Mock(return_value=False)), \
|
||||
patch.object(RestApiHandler, 'write_response') as response_mock:
|
||||
MockRestApiServer(RestApiHandler, post + '27\n\n{"candidate":"postgresql2"}')
|
||||
response_mock.assert_called_with(
|
||||
ManualFailoverPrecheckStatus.NO_GOOD_CANDIDATES.value[1],
|
||||
ManualFailoverPrecheckStatus.NO_GOOD_CANDIDATES.value[0].format(action='failover'))
|
||||
|
||||
@patch.object(MockHa, 'is_leader', Mock(return_value=True))
|
||||
def test_do_POST_citus(self):
|
||||
@@ -659,3 +780,11 @@ class TestRestApiServer(unittest.TestCase):
|
||||
|
||||
def test_get_certificate_serial_number(self):
|
||||
self.assertIsNone(self.srv.get_certificate_serial_number())
|
||||
|
||||
def test_query(self):
|
||||
with patch.object(MockConnection, 'get', Mock(side_effect=OperationalError)):
|
||||
self.assertRaises(PostgresConnectionException, self.srv.query, 'SELECT 1')
|
||||
with patch.object(MockConnection, 'get', Mock(side_effect=[MockConnect(), OperationalError])), \
|
||||
patch.object(MockConnection, 'query') as mock_query:
|
||||
self.srv.query('SELECT 1')
|
||||
mock_query.assert_called_once_with('SELECT 1')
|
||||
|
||||
@@ -250,7 +250,7 @@ class TestBootstrap(BaseTestPostgresql):
|
||||
self.assertFalse(self.b.call_post_bootstrap({'post_init': '/bin/false'}))
|
||||
|
||||
mock_cancellable_subprocess_call.return_value = 0
|
||||
self.p.config.superuser.pop('username')
|
||||
self.p.connection_pool._conn_kwargs.pop('user')
|
||||
self.assertTrue(self.b.call_post_bootstrap({'post_init': '/bin/false'}))
|
||||
mock_cancellable_subprocess_call.assert_called()
|
||||
args, kwargs = mock_cancellable_subprocess_call.call_args
|
||||
@@ -258,7 +258,7 @@ class TestBootstrap(BaseTestPostgresql):
|
||||
self.assertEqual(args[0], ['/bin/false', 'dbname=postgres host=127.0.0.2 port=5432'])
|
||||
|
||||
mock_cancellable_subprocess_call.reset_mock()
|
||||
self.p.config._local_address.pop('host')
|
||||
self.p.connection_pool._conn_kwargs.pop('host')
|
||||
self.assertTrue(self.b.call_post_bootstrap({'post_init': '/bin/false'}))
|
||||
mock_cancellable_subprocess_call.assert_called()
|
||||
self.assertEqual(mock_cancellable_subprocess_call.call_args[0][0], ['/bin/false', 'dbname=postgres port=5432'])
|
||||
|
||||
+1
-1
@@ -13,7 +13,7 @@ class TestCitus(BaseTestPostgresql):
|
||||
def setUp(self):
|
||||
super(TestCitus, self).setUp()
|
||||
self.c = self.p.citus_handler
|
||||
self.c.set_conn_kwargs({'host': 'localhost', 'dbname': 'postgres'})
|
||||
self.p.connection_pool.conn_kwargs = {'host': 'localhost', 'dbname': 'postgres'}
|
||||
self.cluster = get_cluster_initialized_with_leader()
|
||||
self.cluster.workers[1] = self.cluster
|
||||
|
||||
|
||||
+262
-118
@@ -6,12 +6,14 @@ import unittest
|
||||
from click.testing import CliRunner
|
||||
from datetime import datetime, timedelta
|
||||
from mock import patch, Mock, PropertyMock
|
||||
from patroni.config import GlobalConfig
|
||||
from patroni.ctl import ctl, load_config, output_members, get_dcs, parse_dcs, \
|
||||
get_all_members, get_any_member, get_cursor, query_member, PatroniCtlException, apply_config_changes, \
|
||||
format_config_for_editing, show_diff, invoke_editor, format_pg_version, CONFIG_FILE_PATH, PatronictlPrettyTable
|
||||
from patroni.dcs.etcd import AbstractEtcdClientWithFailover, Cluster, Failover
|
||||
from patroni.manual_failover import ManualFailoverPrecheckStatus
|
||||
from patroni.psycopg import OperationalError
|
||||
from patroni.utils import tzutc
|
||||
from patroni.utils import ParseScheduleErrors, tzutc
|
||||
from prettytable import PrettyTable, ALL
|
||||
from urllib3 import PoolManager
|
||||
|
||||
@@ -21,13 +23,24 @@ from .test_ha import get_cluster_initialized_without_leader, get_cluster_initial
|
||||
get_cluster_initialized_with_only_leader, get_cluster_not_initialized_without_leader, get_cluster, Member
|
||||
|
||||
|
||||
@patch('patroni.ctl.load_config', Mock(return_value={
|
||||
'scope': 'alpha', 'restapi': {'listen': '::', 'certfile': 'a'}, 'ctl': {'certfile': 'a'},
|
||||
'etcd': {'host': 'localhost:2379'}, 'citus': {'database': 'citus', 'group': 0},
|
||||
'postgresql': {'data_dir': '.', 'pgpass': './pgpass', 'parameters': {}, 'retry_timeout': 5}}))
|
||||
DEFAULT_CONFIG = {
|
||||
'scope': 'alpha',
|
||||
'restapi': {'listen': '::', 'certfile': 'a'},
|
||||
'ctl': {'certfile': 'a'},
|
||||
'etcd': {'host': 'localhost:2379'},
|
||||
'citus': {'database': 'citus', 'group': 0},
|
||||
'postgresql': {'data_dir': '.', 'pgpass': './pgpass', 'parameters': {}, 'retry_timeout': 5}
|
||||
}
|
||||
|
||||
|
||||
@patch('patroni.ctl.load_config', Mock(return_value=DEFAULT_CONFIG))
|
||||
class TestCtl(unittest.TestCase):
|
||||
TEST_ROLES = ('master', 'primary', 'leader')
|
||||
|
||||
SCHEDULED_TS = '2055-01-01T12:00:00+01:00'
|
||||
SCHEDULED_TS_NO_TZ = '2055-01-01T12:00:00'
|
||||
SCHEDULED_TS_INVALID = '2055-02-30T12:00:00'
|
||||
|
||||
@patch('socket.getaddrinfo', socket_getaddrinfo)
|
||||
@patch.object(AbstractEtcdClientWithFailover, '_get_machines_list', Mock(return_value=['http://remotehost:2379']))
|
||||
def setUp(self):
|
||||
@@ -96,91 +109,194 @@ class TestCtl(unittest.TestCase):
|
||||
mock_get_dcs.return_value = self.e
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
mock_get_dcs.return_value.set_failover_value = Mock()
|
||||
|
||||
# Confirm
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
assert 'leader' in result.output
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'],
|
||||
input='leader\nother\n2300-01-01T12:23:00\ny')
|
||||
assert result.exit_code == 0
|
||||
|
||||
with patch('patroni.config.GlobalConfig.is_paused', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', '2015-01-01T12:00:00'])
|
||||
assert result.exit_code == 1
|
||||
|
||||
# Aborting switchover, as we answer NO to the confirmation
|
||||
# Abort
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\nN')
|
||||
assert result.exit_code == 1
|
||||
|
||||
# Aborting scheduled switchover, as we answer NO to the confirmation
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--scheduled', '2015-01-01T12:00:00+01:00'], input='leader\nother\n\nN')
|
||||
assert result.exit_code == 1
|
||||
|
||||
# Target and source are equal
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nleader\n\ny')
|
||||
assert result.exit_code == 1
|
||||
|
||||
# Reality is not part of this cluster
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nReality\n\ny')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
|
||||
# Without a candidate with --force option
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0', '--force'])
|
||||
assert 'Member' in result.output
|
||||
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', '2015-01-01T12:00:00+01:00'])
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Invalid timestamp
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0', '--force', '--scheduled', 'invalid'])
|
||||
assert result.exit_code != 0
|
||||
|
||||
# Invalid timestamp
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', '2115-02-30T12:00:00+01:00'])
|
||||
assert result.exit_code != 0
|
||||
|
||||
# Specifying wrong leader
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='dummy')
|
||||
assert result.exit_code == 1
|
||||
|
||||
with patch.object(PoolManager, 'request', Mock(side_effect=Exception)):
|
||||
# Non-responding patroni
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'],
|
||||
input='leader\nother\n2300-01-01T12:23:00\ny')
|
||||
assert 'falling back to DCS' in result.output
|
||||
|
||||
with patch.object(PoolManager, 'request') as mocked:
|
||||
mocked.return_value.status = 500
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
assert 'Switchover failed' in result.output
|
||||
|
||||
mocked.return_value.status = 501
|
||||
mocked.return_value.data = b'Server does not support this operation'
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
assert 'Switchover failed' in result.output
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# No members available
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_only_leader
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('No candidates found to switchover to', result.output)
|
||||
|
||||
# No leader available
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_without_leader
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('This cluster has no leader', result.output)
|
||||
|
||||
# Citus cluster, no group number specified
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--force'], input='\n')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('For Citus clusters the --group must me specified', result.output)
|
||||
|
||||
# [Scheduled]
|
||||
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
|
||||
# Scheduled (confirm)
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'],
|
||||
input=f'leader\nother\n{self.SCHEDULED_TS}\ny')
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
self.assertIn(f'Are you sure you want to schedule a switchover in the cluster dummy '
|
||||
f'at {self.SCHEDULED_TS}, demoting current leader', result.output)
|
||||
|
||||
# Scheduled (abort)
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--scheduled', self.SCHEDULED_TS], input='leader\nother\n\nN')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
|
||||
# Scheduled with --force option
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', self.SCHEDULED_TS])
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# Scheduled in pause mode
|
||||
with patch('patroni.config.GlobalConfig.is_paused', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', self.SCHEDULED_TS])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.SCHEDULED_SWITCHOVER_PAUSE.value[0], result.output)
|
||||
|
||||
# Invalid timestamp with force
|
||||
result = self.runner.invoke(ctl,['switchover', 'dummy', '--group', '0', '--force', '--scheduled',
|
||||
self.SCHEDULED_TS_INVALID])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('Unable to parse scheduled timestamp', result.output)
|
||||
|
||||
# Invalid timestamp
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', self.SCHEDULED_TS_INVALID])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('Unable to parse scheduled timestamp', result.output)
|
||||
|
||||
# Invalid timestamp - no timezone
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0',
|
||||
'--force', '--scheduled', self.SCHEDULED_TS_NO_TZ])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ParseScheduleErrors.NO_TIMEZONE.value[0].format(action='switchover'), result.output)
|
||||
|
||||
# [Other erroneous combinations]
|
||||
|
||||
# No candidate in pause mode
|
||||
with patch('patroni.config.GlobalConfig.is_paused', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\n\n\ny')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.SWITCHOVER_PAUSE_NO_CANDIDATE.value[0], result.output)
|
||||
|
||||
# Target and source are equal
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nleader\n\ny')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.SWITCHOVER_TO_LEADER.value[0], result.output)
|
||||
|
||||
# Candidate is not a member of the cluster
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nReality\n\ny')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.CANDIDATE_NOT_MEMEBER.value[0].format(candidate='Reality',
|
||||
cluster_name='dummy'),
|
||||
result.output)
|
||||
|
||||
# Specifying wrong leader
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='dummy')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(
|
||||
ManualFailoverPrecheckStatus.LEADER_NOT_MEMBER.value[0].format(leader='dummy',
|
||||
cluster_name='dummy'),
|
||||
result.output)
|
||||
|
||||
mock_get_dcs.return_value.get_cluster = Mock(
|
||||
return_value=get_cluster_initialized_with_leader(sync=('leader', 'other')))
|
||||
|
||||
# Candidate is not a sync standby
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\notherMember\n\ny')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.CANDIDATE_NOT_SYNC_STANDBY.value[0], result.output)
|
||||
|
||||
# No healthy nodes to promote in sync mode
|
||||
mock_get_dcs.return_value.get_cluster = Mock(return_value=get_cluster_initialized_with_leader(sync=('leader')))
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0', '--force'])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.NO_SYNC_CANDIDATE.value[0].format(action='switchover'),
|
||||
result.output)
|
||||
|
||||
# No healthy nodes to promote in async mode
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_only_leader
|
||||
with patch.object(GlobalConfig, 'is_synchronous_mode', PropertyMock(return_value=False)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0', '--force'])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ManualFailoverPrecheckStatus.ONLY_LEADER.value[0].format(action='switchover'),
|
||||
result.output)
|
||||
|
||||
# Cluster has no leader
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_without_leader
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0', '--leader', 'leader', '--force'])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(
|
||||
ManualFailoverPrecheckStatus.CLUSTER_NO_LEADER.value[0].format(leader='leader', cluster_name='dummy'),
|
||||
result.output)
|
||||
|
||||
# [Errors while sending Patroni REST API request]
|
||||
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
|
||||
with patch.object(PoolManager, 'request', Mock(side_effect=Exception)):
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'],
|
||||
input=f'leader\nother\n{self.SCHEDULED_TS}\ny')
|
||||
self.assertIn('falling back to DCS', result.output)
|
||||
|
||||
with patch.object(PoolManager, 'request') as mock_api_request:
|
||||
mock_api_request.return_value.status = 500
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
self.assertIn('Switchover failed', result.output)
|
||||
|
||||
mock_api_request.return_value.status = 501
|
||||
mock_api_request.return_value.data = b'Server does not support this operation'
|
||||
result = self.runner.invoke(ctl, ['switchover', 'dummy', '--group', '0'], input='leader\nother\n\ny')
|
||||
self.assertIn('Switchover failed', result.output)
|
||||
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
@patch.object(PoolManager, 'request', Mock(return_value=MockResponse()))
|
||||
@patch('patroni.ctl.request_patroni', Mock(return_value=MockResponse()))
|
||||
def test_failover(self, mock_get_dcs):
|
||||
mock_get_dcs.return_value = self.e
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
mock_get_dcs.return_value.set_failover_value = Mock()
|
||||
result = self.runner.invoke(ctl, ['failover', 'dummy', '--force'], input='\n')
|
||||
assert 'For Citus clusters the --group must me specified' in result.output
|
||||
|
||||
# No candidate specified
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
result = self.runner.invoke(ctl, ['failover', 'dummy'], input='0\n')
|
||||
assert 'Failover could be performed only to a specific candidate' in result.output
|
||||
self.assertIn(ManualFailoverPrecheckStatus.FAILOVER_NO_CANDIDATE.value[0], result.output)
|
||||
|
||||
# Failover to an async member in sync mode (confirm)
|
||||
cluster = get_cluster_initialized_with_leader(sync=('leader', 'other'))
|
||||
|
||||
# Temp test to check a fallback to switchover if leader is specified
|
||||
with patch('patroni.ctl._do_failover_or_switchover') as failover_func_mock:
|
||||
result = self.runner.invoke(ctl, ['failover', '--leader', 'leader', 'dummy'], input='0\n')
|
||||
self.assertIn('Supplying a leader name using this command is deprecated', result.output)
|
||||
failover_func_mock.assert_called_once_with(
|
||||
DEFAULT_CONFIG, 'switchover', 'dummy', None, 'leader', None, False)
|
||||
|
||||
# Failover to an async member in sync mode (confirm)
|
||||
cluster.members.append(Member(0, 'async', 28, {'api_url': 'http://127.0.0.1:8012/patroni'}))
|
||||
cluster.config.data['synchronous_mode'] = True
|
||||
mock_get_dcs.return_value.get_cluster = Mock(return_value=cluster)
|
||||
result = self.runner.invoke(ctl, ['failover', 'dummy', '--group', '0', '--candidate', 'async'], input='y\ny')
|
||||
self.assertIn('Are you sure you want to failover to the asynchronous node async', result.output)
|
||||
|
||||
# Failover to an async member in sync mode (abort)
|
||||
mock_get_dcs.return_value.get_cluster = Mock(return_value=cluster)
|
||||
result = self.runner.invoke(ctl, ['failover', 'dummy', '--group', '0', '--candidate', 'async'], input='N')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
|
||||
@patch('patroni.dcs.dcs_modules', Mock(return_value=['patroni.dcs.dummy', 'patroni.dcs.etcd']))
|
||||
def test_get_dcs(self):
|
||||
@@ -273,12 +389,9 @@ class TestCtl(unittest.TestCase):
|
||||
|
||||
@patch.object(PoolManager, 'request')
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
def test_restart_reinit(self, mock_get_dcs, mock_post):
|
||||
def test_reinit(self, mock_get_dcs, mock_post):
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
mock_post.return_value.status = 503
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha'], input='now\ny\n')
|
||||
assert 'Failed: restart for' in result.output
|
||||
assert result.exit_code == 0
|
||||
|
||||
result = self.runner.invoke(ctl, ['reinit', 'alpha'], input='y')
|
||||
assert result.exit_code == 1
|
||||
@@ -287,67 +400,88 @@ class TestCtl(unittest.TestCase):
|
||||
result = self.runner.invoke(ctl, ['reinit', 'alpha', 'other'], input='y\ny')
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Aborted restart
|
||||
@patch.object(PoolManager, 'request')
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
def test_restart(self, mock_get_dcs, mock_post):
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
mock_post.return_value.status = 200
|
||||
|
||||
# Successful restart
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha'], input='now\ny\n')
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# Aborted
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha'], input='now\nN')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
|
||||
# With pending the flag
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--pending', '--force'])
|
||||
assert result.exit_code == 0
|
||||
|
||||
# Aborted scheduled restart
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--scheduled', '2019-10-01T14:30'], input='N')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# Not a member
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', 'dummy', '--any'], input='now\ny')
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('Not a single cluster member among provided members', result.output)
|
||||
|
||||
# Not a member with the specified role
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', 'other', '--role', 'primary'], input='now\ny')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('No primary among provided members', result.output)
|
||||
|
||||
# Wrong pg version
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--any', '--pg-version', '9.1'], input='now\ny')
|
||||
assert 'Error: Invalid PostgreSQL version format' in result.output
|
||||
assert result.exit_code == 1
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('Error: Invalid PostgreSQL version format', result.output)
|
||||
|
||||
# Restart with timeout
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--pending', '--force', '--timeout', '10min'])
|
||||
assert result.exit_code == 0
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# normal restart, the schedule is actually parsed, but not validated in patronictl
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', 'other', '--force', '--scheduled', '2300-10-01T14:30'])
|
||||
assert 'Failed: flush scheduled restart' in result.output
|
||||
# Scheduled restart
|
||||
|
||||
# Aborted scheduled restart
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--scheduled', self.SCHEDULED_TS], input='N')
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
|
||||
# Error parsing scheduled flag value (no tz)
|
||||
result = self.runner.invoke(ctl,
|
||||
['restart', 'alpha', 'other', '--force', '--scheduled', self.SCHEDULED_TS_NO_TZ])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn(ParseScheduleErrors.NO_TIMEZONE.value[0].format(action='restart'), result.output)
|
||||
|
||||
# Error parsing scheduled flag value (invalid date)
|
||||
result = self.runner.invoke(ctl,
|
||||
['restart', 'alpha', 'other', '--force', '--scheduled', self.SCHEDULED_TS_INVALID])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn('Unable to parse scheduled timestamp', result.output)
|
||||
|
||||
# Successfully scheduled restart
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', '--scheduled', self.SCHEDULED_TS], input='Y')
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
self.assertIn('Success: restart on member other', result.output)
|
||||
|
||||
# Not possible to schedule in pause mode
|
||||
with patch('patroni.config.GlobalConfig.is_paused', PropertyMock(return_value=True)):
|
||||
result = self.runner.invoke(ctl,
|
||||
['restart', 'alpha', 'other', '--force', '--scheduled', '2300-10-01T14:30'])
|
||||
assert result.exit_code == 1
|
||||
['restart', 'alpha', 'other', '--force', '--scheduled', self.SCHEDULED_TS])
|
||||
self.assertEqual(result.exit_code, 1)
|
||||
self.assertIn("Can't schedule restart in the paused state", result.output)
|
||||
|
||||
# force restart with restart already present
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', 'other', '--force', '--scheduled', '2300-10-01T14:30'])
|
||||
assert result.exit_code == 0
|
||||
|
||||
ctl_args = ['restart', 'alpha', '--pg-version', '99.0', '--scheduled', '2300-10-01T14:30']
|
||||
# normal restart, the schedule is actually parsed, but not validated in patronictl
|
||||
mock_post.return_value.status = 200
|
||||
result = self.runner.invoke(ctl, ctl_args, input='y')
|
||||
assert result.exit_code == 0
|
||||
# Force restart with restart already scheduled
|
||||
result = self.runner.invoke(ctl, ['restart', 'alpha', 'other', '--force', '--scheduled', self.SCHEDULED_TS])
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
|
||||
# get restart with the non-200 return code
|
||||
# normal restart, the schedule is actually parsed, but not validated in patronictl
|
||||
mock_post.return_value.status = 204
|
||||
result = self.runner.invoke(ctl, ctl_args, input='y')
|
||||
assert result.exit_code == 0
|
||||
|
||||
# get restart with the non-200 return code
|
||||
# normal restart, the schedule is actually parsed, but not validated in patronictl
|
||||
mock_post.return_value.status = 202
|
||||
result = self.runner.invoke(ctl, ctl_args, input='y')
|
||||
assert 'Success: restart scheduled' in result.output
|
||||
assert result.exit_code == 0
|
||||
|
||||
# get restart with the non-200 return code
|
||||
# normal restart, the schedule is actually parsed, but not validated in patronictl
|
||||
mock_post.return_value.status = 409
|
||||
result = self.runner.invoke(ctl, ctl_args, input='y')
|
||||
assert 'Failed: another restart is already' in result.output
|
||||
assert result.exit_code == 0
|
||||
ctl_args = ['restart', 'alpha', '--pg-version', '99.0', '--scheduled', self.SCHEDULED_TS]
|
||||
for code, output in [
|
||||
(204, 'Failed: restart for member other, status code=204'),
|
||||
(202, 'Success: restart scheduled'),
|
||||
(409, 'Failed: another restart is already')
|
||||
]:
|
||||
mock_post.return_value.status = code
|
||||
result = self.runner.invoke(ctl, ctl_args, input='y')
|
||||
self.assertEqual(result.exit_code, 0)
|
||||
self.assertIn(output, result.output)
|
||||
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
def test_remove(self, mock_get_dcs):
|
||||
@@ -402,9 +536,19 @@ class TestCtl(unittest.TestCase):
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
def test_members(self, mock_get_dcs):
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
|
||||
result = self.runner.invoke(ctl, ['list'])
|
||||
assert '127.0.0.1' in result.output
|
||||
assert result.exit_code == 0
|
||||
assert 'Citus cluster: alpha -' in result.output
|
||||
|
||||
result = self.runner.invoke(ctl, ['list', '--group', '0'])
|
||||
assert 'Citus cluster: alpha (group: 0, 12345678901) -' in result.output
|
||||
|
||||
with patch('patroni.ctl.load_config', Mock(return_value={'scope': 'alpha'})):
|
||||
result = self.runner.invoke(ctl, ['list'])
|
||||
assert 'Cluster: alpha (12345678901) -' in result.output
|
||||
|
||||
with patch('patroni.ctl.load_config', Mock(return_value={})):
|
||||
self.runner.invoke(ctl, ['list'])
|
||||
|
||||
|
||||
+281
-100
@@ -99,7 +99,9 @@ def get_node_status(reachable=True, in_recovery=True, dcs_last_seen=0,
|
||||
tags = {}
|
||||
if nofailover:
|
||||
tags['nofailover'] = True
|
||||
return _MemberStatus(e, reachable, in_recovery, dcs_last_seen, timeline, wal_position, tags, watchdog_failed)
|
||||
return _MemberStatus(e, reachable, in_recovery, wal_position,
|
||||
{'tags': tags, 'watchdog_failed': watchdog_failed,
|
||||
'dcs_last_seen': dcs_last_seen, 'timeline': timeline})
|
||||
return fetch_node_status
|
||||
|
||||
|
||||
@@ -433,6 +435,7 @@ class TestHa(PostgresInit):
|
||||
|
||||
def test_promote_without_watchdog(self):
|
||||
self.ha.has_lock = true
|
||||
self.p.is_primary = true
|
||||
with patch.object(Watchdog, 'activate', Mock(return_value=False)):
|
||||
self.assertEqual(self.ha.run_cycle(), 'Demoting self because watchdog could not be activated')
|
||||
self.p.is_primary = false
|
||||
@@ -612,6 +615,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.e.initialize = true
|
||||
self.ha.bootstrap()
|
||||
self.p.is_primary = true
|
||||
with patch.object(Watchdog, 'activate', Mock(return_value=False)), \
|
||||
patch('patroni.ha.logger.error') as mock_logger:
|
||||
self.assertEqual(self.ha.post_bootstrap(), 'running post_bootstrap')
|
||||
@@ -685,110 +689,289 @@ class TestHa(PostgresInit):
|
||||
|
||||
@patch('patroni.postgresql.citus.CitusHandler.is_coordinator', Mock(return_value=False))
|
||||
def test_manual_failover_from_leader(self):
|
||||
self.ha.has_lock = true # I am the leader
|
||||
|
||||
# to me
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', self.p.name, None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
mock_warning.assert_called_with('%s: I am already the leader, no need to %s', 'manual failover', 'failover')
|
||||
|
||||
# to a non-existent candidate
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'blabla', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
mock_warning.assert_called_with(
|
||||
'%s: no healthy members found, %s is not possible', 'manual failover', 'failover')
|
||||
|
||||
# to an existent candidate
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.ha.has_lock = true
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', self.p.name, None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'blabla', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
f = Failover(0, self.p.name, '', None)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(f)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'b', None))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||
|
||||
# to a candidate on an older timeline
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(timeline=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0],
|
||||
('Timeline %s of member %s is behind the cluster timeline %s', 1, 'b', 2))
|
||||
|
||||
# to a lagging candidate
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||
self.ha.cluster.config.data.update({'maximum_lag_on_failover': 5})
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0],
|
||||
('Member %s exceeds maximum replication lag', 'b'))
|
||||
self.ha.cluster.members.pop()
|
||||
|
||||
@patch('patroni.postgresql.citus.CitusHandler.is_coordinator', Mock(return_value=False))
|
||||
def test_manual_switchover_from_leader(self):
|
||||
self.ha.has_lock = true # I am the leader
|
||||
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
|
||||
# different leader specified in failover key, no candidate
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
mock_warning.assert_called_with(
|
||||
'%s: leader name does not match: %s != %s', 'switchover', 'blabla', 'postgresql0')
|
||||
|
||||
# no candidate
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'switchover: demoting myself')
|
||||
|
||||
self.ha._rewind.rewind_or_reinitialize_needed_and_possible = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(watchdog_failed=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(timeline=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||
self.ha.cluster.config.data.update({'maximum_lag_on_failover': 5})
|
||||
self.ha.global_config = self.ha.patroni.config.get_global_config(self.ha.cluster)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
# manual failover from the previous leader to us won't happen if we hold the nofailover flag
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'switchover: demoting myself')
|
||||
|
||||
# Failover scheduled time must include timezone
|
||||
scheduled = datetime.datetime.now()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.ha.run_cycle()
|
||||
# other members with failover_limitation_s
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0], ('Member %s is %s', 'leader', 'not allowed to promote'))
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(watchdog_failed=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0], ('Member %s is %s', 'leader', 'not watchdog capable'))
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(timeline=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0],
|
||||
('Timeline %s of member %s is behind the cluster timeline %s', 1, 'leader', 2))
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||
self.ha.cluster.config.data.update({'maximum_lag_on_failover': 5})
|
||||
self.ha.global_config = self.ha.patroni.config.get_global_config(self.ha.cluster)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0], ('Member %s exceeds maximum replication lag', 'leader'))
|
||||
|
||||
@patch('patroni.postgresql.citus.CitusHandler.is_coordinator', Mock(return_value=False))
|
||||
def test_scheduled_switchover_from_leader(self):
|
||||
self.ha.has_lock = true # I am the leader
|
||||
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
|
||||
# switchover scheduled time must include timezone
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
scheduled = datetime.datetime.now()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'blabla', scheduled))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.assertIn('Incorrect value of scheduled_at: %s', mock_warning.call_args_list[0][0])
|
||||
|
||||
# scheduled now
|
||||
scheduled = datetime.datetime.utcnow().replace(tzinfo=tzutc)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'b', scheduled))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual('switchover: demoting myself', self.ha.run_cycle())
|
||||
|
||||
scheduled = scheduled + datetime.timedelta(seconds=30)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
# scheduled in the future
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
scheduled = scheduled + datetime.timedelta(seconds=30)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'blabla', scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.assertIn('Awaiting %s at %s (in %.0f seconds)', mock_info.call_args_list[0][0])
|
||||
|
||||
scheduled = scheduled + datetime.timedelta(seconds=-600)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
# stale value
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
scheduled = scheduled + datetime.timedelta(seconds=-600)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'b', scheduled))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.assertIn('Found a stale %s value, cleaning up: %s', mock_warning.call_args_list[0][0])
|
||||
|
||||
scheduled = None
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
def test_manual_switchover_from_leader_in_pause(self):
|
||||
self.ha.has_lock = true # I am the leader
|
||||
self.ha.is_paused = true
|
||||
|
||||
# no candidate
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, '', None))
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.assertEqual('PAUSE: no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
mock_warning.assert_called_with(
|
||||
'%s is possible only to a specific candidate in a paused state', 'Switchover')
|
||||
|
||||
def test_manual_failover_from_leader_in_pause(self):
|
||||
self.ha.has_lock = true
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.ha.is_paused = true
|
||||
scheduled = datetime.datetime.now()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('PAUSE: no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, '', None))
|
||||
self.assertEqual('PAUSE: no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
|
||||
# failover from me, candidate is healthy
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, None, 'b', None))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual('PAUSE: manual failover: demoting myself', self.ha.run_cycle())
|
||||
self.ha.cluster.members.pop()
|
||||
|
||||
def test_manual_failover_from_leader_in_synchronous_mode(self):
|
||||
self.ha.has_lock = true
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.ha.process_sync_replication = Mock()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, None))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, 'a'))
|
||||
self.ha.is_failover_possible = true
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
|
||||
# I am the leader
|
||||
self.p.is_primary = true
|
||||
self.ha.has_lock = true
|
||||
|
||||
# the candidate is not in sync members but we allow failover to an async candidate
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, None, 'b', None), sync=(self.p.name, 'a'))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual('manual failover: demoting myself', self.ha.run_cycle())
|
||||
self.ha.cluster.members.pop()
|
||||
|
||||
def test_manual_switchover_from_leader_in_synchronous_mode(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.ha.process_sync_replication = Mock()
|
||||
|
||||
# I am the leader
|
||||
self.p.is_primary = true
|
||||
self.ha.has_lock = true
|
||||
|
||||
# candidate specified is not in sync members
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None),
|
||||
sync=(self.p.name, 'blabla'))
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual(mock_warning.call_args_list[0][0],
|
||||
('%s candidate=%s does not match with sync_standbys=%s', 'Switchover', 'a', 'blabla'))
|
||||
|
||||
# the candidate is in sync members and is healthy
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=305419896)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None),
|
||||
sync=(self.p.name, 'a'))
|
||||
self.ha.cluster.members.append(Member(0, 'a', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.assertEqual('switchover: demoting myself', self.ha.run_cycle())
|
||||
|
||||
# the candidate is in sync members but is not healthy
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=true)
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual(mock_info.call_args_list[0][0], ('Member %s is %s', 'a', 'not allowed to promote'))
|
||||
|
||||
def test_manual_failover_process_no_leader(self):
|
||||
self.p.is_primary = false
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', self.p.name, None))
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'leader', None))
|
||||
self.p.set_role('replica')
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
self.ha.fetch_node_status = get_node_status() # accessible, in_recovery
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, self.p.name, '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
self.ha.fetch_node_status = get_node_status(reachable=False) # inaccessible, in_recovery
|
||||
|
||||
# failover to another member, fetch_node_status for candidate fails
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'leader', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
self.assertEqual(mock_warning.call_args_list[1][0],
|
||||
('%s: member %s is %s', 'manual failover', 'leader', 'not reachable'))
|
||||
|
||||
# failover to another member, candidate is accessible, in_recovery
|
||||
self.p.set_role('replica')
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
# set failover flag to True for all members of the cluster
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# set nofailover flag to True for all members of the cluster
|
||||
# this should elect the current member, as we are not going to call the API for it.
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None))
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True) # accessible, in_recovery
|
||||
self.p.set_role('replica')
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
# same as previous, but set the current member to nofailover. In no case it should be elected as a leader
|
||||
|
||||
# failover to me but I am set to nofailover. In no case I should be elected as a leader
|
||||
self.p.set_role('replica')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None))
|
||||
self.ha.patroni.nofailover = True
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
||||
|
||||
self.ha.patroni.nofailover = False
|
||||
|
||||
# failover to another member that is on an older timeline (only failover_limitation() is checked)
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'b', None))
|
||||
self.ha.cluster.members.append(Member(0, 'b', 28, {'api_url': 'http://127.0.0.1:8011/patroni'}))
|
||||
self.ha.fetch_node_status = get_node_status(timeline=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
mock_info.assert_called_with('%s: to %s, i am %s', 'manual failover', 'b', 'postgresql0')
|
||||
|
||||
# failover to another member lagging behind the cluster_lsn (only failover_limitation() is checked)
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.cluster.config.data.update({'maximum_lag_on_failover': 5})
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
mock_info.assert_called_with('%s: to %s, i am %s', 'manual failover', 'b', 'postgresql0')
|
||||
|
||||
def test_manual_switchover_process_no_leader(self):
|
||||
self.p.is_primary = false
|
||||
self.p.set_role('replica')
|
||||
|
||||
# I was the leader, other members are healthy
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, self.p.name, '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# I was the leader, I am the only healthy member
|
||||
with patch('patroni.ha.logger.info') as mock_info:
|
||||
self.ha.fetch_node_status = get_node_status(reachable=False) # inaccessible, in_recovery
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
self.assertEqual(mock_info.call_args_list[0][0], ('Member %s is %s', 'leader', 'not reachable'))
|
||||
self.assertEqual(mock_info.call_args_list[1][0], ('Member %s is %s', 'other', 'not reachable'))
|
||||
|
||||
def test_manual_failover_process_no_leader_in_synchronous_mode(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.p.is_primary = false
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True) # other nodes are not healthy
|
||||
|
||||
# switchover to a specific node, which name doesn't match our name (postgresql0)
|
||||
# manual failover when our name (postgresql0) isn't in the /sync key and the candidate node is not available
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
|
||||
sync=('leader1', 'blabla'))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# manual failover when the candidate node isn't available but our name is in the /sync key
|
||||
# while other sync node is nofailover
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
|
||||
sync=('leader1', 'postgresql0'))
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet(), CaseInsensitiveSet()))
|
||||
self.ha.dcs.write_sync_state = Mock(return_value=SyncState.empty())
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
self.assertEqual(mock_warning.call_args_list[0][0],
|
||||
('%s: member %s is %s', 'manual failover', 'other', 'not allowed to promote'))
|
||||
|
||||
# manual failover to our node (postgresql0),
|
||||
# which name is not in sync nodes list (some sync nodes are available)
|
||||
self.p.set_role('replica')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None),
|
||||
sync=('leader1', 'other'))
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet(['leader1']),
|
||||
CaseInsensitiveSet(['leader1'])))
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
|
||||
def test_manual_switchover_process_no_leader_in_synchronous_mode(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.p.is_primary = false
|
||||
|
||||
# to a specific node, which name doesn't match our name (postgresql0)
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'other', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# switchover to our node (postgresql0), which name is not in sync nodes list
|
||||
# to our node (postgresql0), which name is not in sync nodes list
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'postgresql0', None),
|
||||
sync=('leader1', 'blabla'))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# switchover from a specific leader, but our name (postgresql0) is not in the sync nodes list
|
||||
# without candidate, our name (postgresql0) is not in the sync nodes list
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', '', None),
|
||||
sync=('leader', 'blabla'))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
@@ -798,45 +981,31 @@ class TestHa(PostgresInit):
|
||||
sync=('postgresql0'))
|
||||
self.ha.patroni.nofailover = True
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
||||
self.ha.patroni.nofailover = False
|
||||
|
||||
# manual failover when our name (postgresql0) isn't in the /sync key and the `other` node is not available
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True) # accessible, in_recovery
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
|
||||
sync=('leader1', 'blabla'))
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
# manual failover when the `other` node isn't available but our name is in the /sync key
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
|
||||
sync=('leader1', 'postgresql0'))
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet(), CaseInsensitiveSet()))
|
||||
self.ha.dcs.write_sync_state = Mock(return_value=SyncState.empty())
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
|
||||
# manual failover to our node (postgresql0),
|
||||
# which name is not in sync nodes list (the leader and all sync nodes are not available)
|
||||
self.p.set_role('replica')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None),
|
||||
sync=('leader1', 'other'))
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
|
||||
# manual failover to our node (postgresql0),
|
||||
# which name is not in sync nodes list (some sync nodes are available)
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None),
|
||||
sync=('leader1', 'other'))
|
||||
self.p.set_role('replica')
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet(['leader1']),
|
||||
CaseInsensitiveSet(['leader1'])))
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||
|
||||
def test_manual_failover_process_no_leader_in_pause(self):
|
||||
self.ha.is_paused = true
|
||||
|
||||
# I am running as primary, cluster is unlocked, the candidate is allowed to promote
|
||||
# but we are in pause
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: continue to run as primary without lock')
|
||||
|
||||
def test_manual_switchover_process_no_leader_in_pause(self):
|
||||
self.ha.is_paused = true
|
||||
|
||||
# I am running as primary, cluster is unlocked, no candidate specified
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: continue to run as primary without lock')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'blabla', None))
|
||||
self.assertEqual('PAUSE: acquired session lock as a leader', self.ha.run_cycle())
|
||||
|
||||
# the candidate is not running
|
||||
with patch('patroni.ha.logger.warning') as mock_warning:
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'blabla', None))
|
||||
self.assertEqual('PAUSE: acquired session lock as a leader', self.ha.run_cycle())
|
||||
self.assertEqual(
|
||||
mock_warning.call_args_list[0][0],
|
||||
('%s: removing failover key because failover candidate is not running', 'switchover'))
|
||||
|
||||
# switchover to me, I am not leader
|
||||
self.p.is_primary = false
|
||||
self.p.set_role('replica')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', self.p.name, None))
|
||||
@@ -844,7 +1013,7 @@ class TestHa(PostgresInit):
|
||||
|
||||
def test_is_healthiest_node(self):
|
||||
self.ha.is_failsafe_mode = true
|
||||
self.p.is_primary = false
|
||||
self.ha.state_handler.is_primary = false
|
||||
self.ha.patroni.nofailover = False
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.ha.dcs._last_failsafe = {'foo': ''}
|
||||
@@ -1086,7 +1255,7 @@ class TestHa(PostgresInit):
|
||||
f = Failover(0, self.p.name, '', None)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(f)
|
||||
self.ha.fetch_node_status = get_node_status() # accessible, in_recovery
|
||||
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||
self.assertEqual(self.ha.run_cycle(), 'switchover: demoting myself')
|
||||
|
||||
@patch('patroni.ha.Ha.demote')
|
||||
def test_failover_immediately_on_zero_primary_start_timeout(self, demote):
|
||||
@@ -1294,14 +1463,16 @@ class TestHa(PostgresInit):
|
||||
mock_restart.assert_called_once()
|
||||
self.ha.dcs.get_cluster.assert_not_called()
|
||||
|
||||
@patch.object(Cluster, 'is_unlocked', Mock(return_value=False))
|
||||
def test_enable_synchronous_mode(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.ha.has_lock = true
|
||||
self.p.name = 'leader'
|
||||
self.p.sync_handler.current_state = Mock(return_value=(CaseInsensitiveSet(), CaseInsensitiveSet()))
|
||||
self.ha.dcs.write_sync_state = Mock(return_value=SyncState.empty())
|
||||
with patch('patroni.ha.logger.info') as mock_logger:
|
||||
self.ha.run_cycle()
|
||||
self.assertEqual(mock_logger.call_args[0][0], 'Enabled synchronous replication')
|
||||
self.assertEqual(mock_logger.call_args_list[0][0][0], 'Enabled synchronous replication')
|
||||
self.ha.dcs.write_sync_state = Mock(return_value=None)
|
||||
with patch('patroni.ha.logger.warning') as mock_logger:
|
||||
self.ha.run_cycle()
|
||||
@@ -1459,3 +1630,13 @@ class TestHa(PostgresInit):
|
||||
self.assertEqual(self.ha.patroni.request.call_args[1]['timeout'], 2)
|
||||
mock_logger.assert_called()
|
||||
self.assertTrue(mock_logger.call_args[0][0].startswith('Request to Citus coordinator'))
|
||||
|
||||
def test_has_members_eligible_to_promote(self):
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
members = [
|
||||
Member(0, 'test', 1, {'api_url': 'http://127.0.0.1:8011/patroni', 'conn_url': 'postgres://127.0.0.1:5432/postgres'}),
|
||||
Member(0, 'test2', 1, {'api_url': 'http://127.0.0.1:8011/patroni', 'conn_url': 'postgres://127.0.0.1:5432/postgres'}),
|
||||
]
|
||||
with patch('patroni.ha.logger.info') as mock_logger:
|
||||
self.assertTrue(self.ha.has_members_eligible_to_promote(members, fast_path=True))
|
||||
mock_logger.assert_not_called()
|
||||
|
||||
@@ -308,13 +308,15 @@ class TestKubernetesConfigMaps(BaseTestKubernetes):
|
||||
mock_patch_namespaced_pod.assert_called()
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['isMaster'], 'false')
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['tmp_role'], 'replica')
|
||||
|
||||
self.k.touch_member({'state': 'running', 'role': 'standby-leader'})
|
||||
mock_patch_namespaced_pod.assert_called()
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['isMaster'], 'false')
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['tmp_role'], 'standby-leader')
|
||||
mock_patch_namespaced_pod.rest_mock()
|
||||
|
||||
self.k._name = 'p-0'
|
||||
self.k.touch_member({'role': 'standby_leader'})
|
||||
mock_patch_namespaced_pod.assert_called()
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['isMaster'], 'false')
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['tmp_role'], 'master')
|
||||
mock_patch_namespaced_pod.rest_mock()
|
||||
|
||||
self.k.touch_member({'role': 'primary'})
|
||||
mock_patch_namespaced_pod.assert_called()
|
||||
self.assertEqual(mock_patch_namespaced_pod.call_args[0][2].metadata.labels['isMaster'], 'true')
|
||||
@@ -434,6 +436,10 @@ class TestKubernetesEndpoints(BaseTestKubernetes):
|
||||
mock_logger_exception.assert_called_once()
|
||||
self.assertEqual(('create_config_service failed',), mock_logger_exception.call_args[0])
|
||||
|
||||
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_endpoints', mock_namespaced_kind, create=True)
|
||||
def test_write_leader_optime(self):
|
||||
self.k.write_leader_optime(12345)
|
||||
|
||||
|
||||
def mock_watch(*args):
|
||||
return urllib3.HTTPResponse()
|
||||
|
||||
@@ -185,7 +185,7 @@ class TestPatroni(unittest.TestCase):
|
||||
|
||||
def test_reload_config(self):
|
||||
self.p.reload_config()
|
||||
self.p.get_tags = Mock(side_effect=Exception)
|
||||
self.p._get_tags = Mock(side_effect=Exception)
|
||||
self.p.reload_config(local=True)
|
||||
|
||||
def test_nosync(self):
|
||||
|
||||
@@ -310,6 +310,17 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
self.p.config.write_postgresql_conf()
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (False, False))
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (False, False))
|
||||
|
||||
# Config files changed, but can't connect to postgres
|
||||
mock_get_pg_settings.side_effect = PostgresConnectionException('')
|
||||
with patch('patroni.postgresql.config.mtime', mock_mtime):
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
||||
|
||||
# Config files didn't change, but postgres crashed or in crash recovery
|
||||
with patch.object(MockPostmaster, 'create_time', Mock(return_value=1234568), create=True):
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (False, False))
|
||||
|
||||
# Any other exception raised when executing the query
|
||||
mock_get_pg_settings.side_effect = Exception
|
||||
with patch('patroni.postgresql.config.mtime', mock_mtime):
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
||||
@@ -577,7 +588,10 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
self.assertEqual(self.p.config.local_replication_address, {'host': '/tmp', 'port': '5432'})
|
||||
self.p.config._server_parameters.pop('unix_socket_directories')
|
||||
self.p.config.resolve_connection_addresses()
|
||||
self.assertEqual(self.p.config._local_address, {'port': '5432'})
|
||||
self.assertEqual(self.p.connection_pool.conn_kwargs, {'connect_timeout': 3, 'dbname': 'postgres',
|
||||
'fallback_application_name': 'Patroni',
|
||||
'options': '-c statement_timeout=2000',
|
||||
'password': 'test', 'port': '5432', 'user': 'foo'})
|
||||
|
||||
@patch.object(Postgresql, '_version_file_exists', Mock(return_value=True))
|
||||
def test_get_major_version(self):
|
||||
|
||||
+3
-2
@@ -32,9 +32,9 @@ class TestSlotsHandler(BaseTestPostgresql):
|
||||
self.p._global_config = GlobalConfig({})
|
||||
self.s = self.p.slots_handler
|
||||
self.p.start()
|
||||
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}}}, 1)
|
||||
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}, 'ls2': None}}, 1)
|
||||
self.cluster = Cluster(True, config, self.leader, 0, [self.me, self.other, self.leadermem],
|
||||
None, SyncState.empty(), None, {'ls': 12345}, None)
|
||||
None, SyncState.empty(), None, {'ls': 12345, 'ls2': 12345}, None)
|
||||
|
||||
def test_sync_replication_slots(self):
|
||||
config = ClusterConfig(1, {'slots': {'test_3': {'database': 'a', 'plugin': 'b'},
|
||||
@@ -123,6 +123,7 @@ class TestSlotsHandler(BaseTestPostgresql):
|
||||
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), ['ls'])
|
||||
self.cluster.slots['ls'] = 'a'
|
||||
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), [])
|
||||
self.cluster.config.data['slots']['ls']['database'] = 'b'
|
||||
with patch.object(MockCursor, 'rowcount', PropertyMock(return_value=1), create=True):
|
||||
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), ['ls'])
|
||||
|
||||
|
||||
Reference in New Issue
Block a user