mirror of
https://github.com/outbackdingo/patroni.git
synced 2026-08-26 07:30:14 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
02698acd69 |
@@ -65,7 +65,6 @@ install:
|
||||
echo -e 'HTTP/1.0 200 OK\nContent-Type: application/json\n\n{"servers":["127.0.0.1"],"port":2181}' \
|
||||
| nc -l 8181 &> /dev/null
|
||||
done&
|
||||
ZK_PID=$!
|
||||
}
|
||||
|
||||
attempt_num=1
|
||||
@@ -111,8 +110,5 @@ script:
|
||||
|
||||
set +e
|
||||
after_success:
|
||||
# before_cache is executed earlier than after_success, so we need to restore one of virtualenv directories
|
||||
- fpv=$(basename $(readlink $HOME/virtualenv/python3.5)) && mv $HOME/mycache/${fpv} $HOME/virtualenv/${fpv}
|
||||
- coveralls
|
||||
- if [[ $TEST_SUITE != "behave" ]]; then python-codacy-coverage -r coverage.xml; fi
|
||||
- if [[ $DCS == "exhibitor" ]]; then ~/mycache/zookeeper-${ZKVERSION}/bin/zkServer.sh stop; kill -9 $ZK_PID; fi
|
||||
|
||||
+6
-10
@@ -3,17 +3,14 @@
|
||||
FROM postgres:9.6
|
||||
MAINTAINER Alexander Kukushkin <[email protected]>
|
||||
|
||||
RUN export DEBIAN_FRONTEND=noninteractive \
|
||||
&& echo 'APT::Install-Recommends "0";\nAPT::Install-Suggests "0";' > /etc/apt/apt.conf.d/01norecommend \
|
||||
RUN echo 'APT::Install-Recommends "0";\nAPT::Install-Suggests "0";' > /etc/apt/apt.conf.d/01norecommend \
|
||||
&& apt-get update -y \
|
||||
&& apt-get upgrade -y \
|
||||
&& apt-get install -y curl jq haproxy python-psycopg2 python-yaml python-requests python-six python-pysocks \
|
||||
python-dateutil python-pip python-setuptools python-prettytable python-wheel python-psutil python locales \
|
||||
&& apt-get install -y curl jq haproxy python-psycopg2 python-yaml python-requests \
|
||||
python-six python-dateutil python-urllib3 python-dnspython \
|
||||
python-pip python-setuptools python-kazoo python-prettytable python-wheel python \
|
||||
|
||||
## Make sure we have a en_US.UTF-8 locale available
|
||||
&& localedef -i en_US -c -f UTF-8 -A /usr/share/locale/locale.alias en_US.UTF-8 \
|
||||
|
||||
&& pip install 'python-etcd>=0.4.3,<0.5' click tzlocal cdiff \
|
||||
&& pip install python-etcd==0.4.3 python-consul==0.7.0 click tzlocal --upgrade \
|
||||
|
||||
&& mkdir -p /home/postgres \
|
||||
&& chown postgres:postgres /home/postgres \
|
||||
@@ -24,7 +21,7 @@ RUN export DEBIAN_FRONTEND=noninteractive \
|
||||
&& apt-get clean -y \
|
||||
&& rm -rf /var/lib/apt/lists/* /root/.cache
|
||||
|
||||
ENV ETCDVERSION 3.2.3
|
||||
ENV ETCDVERSION 3.1.2
|
||||
RUN curl -L https://github.com/coreos/etcd/releases/download/v${ETCDVERSION}/etcd-v${ETCDVERSION}-linux-amd64.tar.gz \
|
||||
| tar xz -C /usr/local/bin --strip=1 --wildcards --no-anchored etcd etcdctl
|
||||
|
||||
@@ -43,6 +40,5 @@ RUN mkdir /data/ && touch /pgpass /patroni.yml \
|
||||
|
||||
EXPOSE 2379 5432 8008
|
||||
|
||||
ENV LC_ALL=en_US.UTF-8 LANG=en_US.UTF-8
|
||||
ENTRYPOINT ["/bin/bash", "/entrypoint.sh"]
|
||||
USER postgres
|
||||
|
||||
@@ -22,7 +22,7 @@ __EOF__
|
||||
|
||||
DOCKER_IP=$(hostname --ip-address)
|
||||
PATRONI_SCOPE=${PATRONI_SCOPE:-batman}
|
||||
ETCD_ARGS="--data-dir /tmp/etcd.data -advertise-client-urls=http://${DOCKER_IP}:2379 -listen-client-urls=http://0.0.0.0:2379"
|
||||
ETCD_ARGS="--data-dir /tmp/etcd.data -advertise-client-urls=http://${DOCKER_IP}:2379 -listen-client-urls=http://0.0.0.0:2379 -listen-peer-urls=http://0.0.0.0:2380"
|
||||
|
||||
optspec=":vh-:"
|
||||
while getopts "$optspec" optchar; do
|
||||
|
||||
@@ -25,16 +25,6 @@ Example: defining ``PATRONI_admin_PASSWORD=strongpasswd`` and ``PATRONI_admin_OP
|
||||
Consul
|
||||
------
|
||||
- **PATRONI\_CONSUL\_HOST**: the host:port for the Consul endpoint.
|
||||
- **PATRONI\_CONSUL\_URL**: url for the Consul, in format: http(s)://host:port
|
||||
- **PATRONI\_CONSUL\_PORT**: (optional) Consul port
|
||||
- **PATRONI\_CONSUL\_SCHEME**: (optional) **http** or **https**, defaults to **http**
|
||||
- **PATRONI\_CONSUL\_TOKEN**: (optional) ACL token
|
||||
- **PATRONI\_CONSUL\_VERIFY**: (optional) whether to verify the SSL certificate for HTTPS requests
|
||||
- **PATRONI\_CONSUL\_CACERT**: (optional) The ca certificate. If pressent it will enable validation.
|
||||
- **PATRONI\_CONSUL\_CERT**: (optional) File with the client certificate
|
||||
- **PATRONI\_CONSUL\_KEY**: (optional) File with the client key. Can be empty if the key is part of certificate.
|
||||
- **PATRONI\_CONSUL\_DC**: (optional) Datacenter to communicate with. By default the datacenter of the host is used.
|
||||
- **PATRONI\_CONSUL\_CHECKS**: (optional) list of Consul health checks used for the session. If not specified Consul will use "serfHealth" in additional to the TTL based check created by Patroni. Additional checks, in particular the "serfHealth", may cause the leader lock to expire faster than in `ttl` seconds when the leader instance becomes unavailable.
|
||||
|
||||
Etcd
|
||||
----
|
||||
@@ -56,7 +46,6 @@ PostgreSQL
|
||||
- **PATRONI\_POSTGRESQL\_LISTEN**: IP address + port that Postgres listens to. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node.
|
||||
- **PATRONI\_POSTGRESQL\_CONNECT\_ADDRESS**: IP address + port through which Postgres is accessible from other nodes and applications.
|
||||
- **PATRONI\_POSTGRESQL\_DATA\_DIR**: The location of the Postgres data directory, either existing or to be initialized by Patroni.
|
||||
- **PATRONI\_POSTGRESQL\_CONFIG\_DIR**: The location of the Postgres configuration directory, defaults to the data directory. Must be writable by Patroni.
|
||||
- **PATRONI\_POSTGRESQL\_BIN_DIR**: Path to PostgreSQL binaries. (pg_ctl, pg_rewind, pg_basebackup, postgres) The default value is an empty string meaning that PATH environment variable will be used to find the executables.
|
||||
- **PATRONI\_POSTGRESQL\_PGPASS**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup and under some other circumstances. The location must be writable by Patroni.
|
||||
- **PATRONI\_REPLICATION\_USERNAME**: replication username; the user will be created during initialization. Replicas will use this user to access master via streaming replication
|
||||
|
||||
+4
-37
@@ -15,7 +15,6 @@ Bootstrap configuration
|
||||
- **dcs**: This section will be written into `/<namespace>/<scope>/config` of a given configuration store after initializing of new cluster. This is the global configuration for the cluster. If you want to change some parameters for all cluster nodes - just do it in DCS (or via Patroni API) and all nodes will apply this configuration.
|
||||
- **loop\_wait**: the number of seconds the loop will sleep. Default value: 10
|
||||
- **ttl**: the TTL to acquire the leader lock. Think of it as the length of time before initiation of the automatic failover process. Default value: 30
|
||||
- **retry\_timeout**: timeout for DCS and PostgreSQL operation retries. DCS or network issues shorter than this will not cause Patroni to demote the leader. Default value: 10
|
||||
- **maximum\_lag\_on\_failover**: the maximum bytes a follower may lag to be able to participate in leader election.
|
||||
- **master\_start\_timeout**: the amount of time a master is allowed to recover from failures before failover is triggered. Default is 300 seconds. When set to 0 failover is done immediately after a crash is detected if possible. When using asynchronous replication a failover can cause lost transactions. Best worst case failover time for master failure is: loop\_wait + master\_start\_timeout + loop\_wait, unless master\_start\_timeout is zero, in which case it's just loop\_wait. Set the value according to your durability/availability tradeoff.
|
||||
- **synchronous\_mode**: turns on synchronous replication mode. In this mode a replica will be chosen as synchronous and only the latest leader and synchronous replica are able to participate in leader election. Synchronous mode makes sure that succesfully committed transactions will not be lost at failover, at the cost of losing availability for writes when Patroni cannot ensure transaction durability. See `replication modes documentation <https://github.com/zalando/patroni/blob/master/docs/replication_modes.rst>`__ for details.
|
||||
@@ -24,10 +23,6 @@ Bootstrap configuration
|
||||
- **use\_slots**: whether or not to use replication_slots. Must be False for PostgreSQL 9.3. You should comment out max_replication_slots before it becomes ineligible for leader status.
|
||||
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower.
|
||||
- **parameters**: list of configuration settings for Postgres. Many of these are required for replication to work.
|
||||
- **method**: custom script to use for bootstrpapping this cluster.
|
||||
See :ref:`custom bootstrap methods documentation <custom_bootstrap>` for details.
|
||||
When ``initdb`` is specified revert to the default ``initdb`` command. ``initdb`` is also triggered when no ``method``
|
||||
parameter is present in the configuration file.
|
||||
- **initdb**: List options to be passed on to initdb.
|
||||
- **- data-checksums**: Must be enabled when pg_rewind is needed on 9.3.
|
||||
- **- encoding: UTF8**: default encoding for new databases.
|
||||
@@ -41,30 +36,15 @@ Bootstrap configuration
|
||||
- **options**: list of options for CREATE USER statement
|
||||
- **- createrole**
|
||||
- **- createdb**
|
||||
- **post\_bootstrap** or **post\_init**: An additional script that will be executed after initializing the cluster. The script receives a connection string URL (with the cluster superuser as a user name). The PGPASSFILE variable is set to the location of pgpass file.
|
||||
|
||||
.. _consul_settings:
|
||||
- **post_init**: An additional script that will be executed after initializing the cluster. The script receives a connection string URL (with the cluster superuser as a user name). The PGPASSFILE variable is set to the location of pgpass file.
|
||||
|
||||
Consul
|
||||
------
|
||||
Most of the parameters are optional, but you have to specify one of the **host** or **url**
|
||||
|
||||
- **host**: the host:port for the Consul endpoint, in format: http(s)://host:port
|
||||
- **url**: url for the Consul endpoint
|
||||
- **port**: (optional) Consul port
|
||||
- **scheme**: (optional) **http** or **https**, defaults to **http**
|
||||
- **token**: (optional) ACL token
|
||||
- **verify** (optional) whether to verify the SSL certificate for HTTPS requests
|
||||
- **cacert**: (optional) The ca certificate. If pressent it will enable validation.
|
||||
- **cert**: (optional) file with the client certificate
|
||||
- **key**: (optional) file with the client key. Can be empty if the key is part of **cert**.
|
||||
- **dc**: (optional) Datacenter to communicate with. By default the datacenter of the host is used.
|
||||
- **checks**: (optional) list of Consul health checks used for the session. If not specified Consul will use "serfHealth" in additional to the TTL based check created by Patroni. Additional checks, in particular the "serfHealth", may cause the leader lock to expire faster than in `ttl` seconds when the leader instance becomes unavailable
|
||||
- **host**: the host:port for the Consul endpoint.
|
||||
|
||||
Etcd
|
||||
----
|
||||
Most of the parameters are optional, but you have to specify one of the **host**, **url**, **proxy** or **srv**
|
||||
|
||||
- **host**: the host:port for the etcd endpoint.
|
||||
- **url**: url for the etcd
|
||||
- **proxy**: proxy url for the etcd. If you are connecting to the etcd using proxy, use this parameter instead of **url**
|
||||
@@ -100,21 +80,14 @@ PostgreSQL
|
||||
- **on\_start**: run this script when the cluster starts.
|
||||
- **on\_stop**: run this script when the cluster stops.
|
||||
- **connect\_address**: IP address + port through which Postgres is accessible from other nodes and applications.
|
||||
- **create\_replica\_method**: an ordered list of the create methods for turning a Patroni node into a new replica.
|
||||
"basebackup" is the default method; other methods are assumed to refer to scripts, each of which is configured as its
|
||||
own config item. See :ref:`custom replica creation methods documentation <custom_replica_creation>` for further explanation.
|
||||
- **create\_replica\_methods**: an ordered list of the create methods for turning a Patroni node into a new replica. "basebackup" is the default method; other methods are assumed to refer to scripts, each of which is configured as its own config item.
|
||||
- **data\_dir**: The location of the Postgres data directory, either existing or to be initialized by Patroni.
|
||||
- **config\_dir**: The location of the Postgres configuration directory, defaults to the data directory. Must be writable by Patroni.
|
||||
- **bin\_dir**: Path to PostgreSQL binaries. (pg_ctl, pg_rewind, pg_basebackup, postgres) The default value is an empty string meaning that PATH environment variable will be used to find the executables.
|
||||
- **listen**: IP address + port that Postgres listens to; must be accessible from other nodes in the cluster, if you're using streaming replication. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node.
|
||||
- **use\_unix\_socket**: specifies that Patroni should prefer to use unix sockets to connect to the cluster. Default value is ``false``. If ``unix_socket_directories`` is definded, Patroni will use first suitable value from it to connect to the cluster and fallback to tcp if nothing is suitable. If ``unix_socket_directories`` is not specified in ``postgresql.parameters``, Patroni will assume that default value should be used and omit ``host`` from connection parameters.
|
||||
- **pgpass**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup, the post_init script and under some other circumstances. The location must be writable by Patroni.
|
||||
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower.
|
||||
- **custom\_conf** : path to an optional custom ``postgresql.conf`` file, that will be used in place of ``postgresql.base.conf``. The file must exist on all cluster nodes, be readable by PostgreSQL and will be included from its location on the real ``postgresql.conf``. Note that Patroni will not monitor this file for changes, nor backup it. However, its settings can still be overriden by Patroni's own configuration facilities - see `dynamic configuration <https://github.com/zalando/patroni/blob/master/docs/dynamic_configuration.rst>`__ for details.
|
||||
- **custom_conf** : path to an optional custom ``postgresql.conf`` file, that will be used in place of ``postgresql.base.conf``. The file must exist on all cluster nodes, be readable by PostgreSQL and will be included from its location on the real ``postgresql.conf``. Note that Patroni will not monitor this file for changes, nor backup it. However, its settings can still be overriden by Patroni's own configuration facilities - see `dynamic configuration <https://github.com/zalando/patroni/blob/master/docs/dynamic_configuration.rst>`__ for details.
|
||||
- **parameters**: list of configuration settings for Postgres. Many of these are required for replication to work.
|
||||
- **pg\_hba**: list of lines that Patroni will use to generate ``pg_hba.conf``. This parameter has higher priority than ``bootstrap.pg_hba``. Together with :ref:`dynamic configuration <dynamic_configuration>` it simplifies management of ``pg_hba.conf``.
|
||||
- **- host all all 0.0.0.0/0 md5**.
|
||||
- **- host replication replicator 127.0.0.1/32 md5**: A line like this is required for replication.
|
||||
- **pg\_ctl\_timeout**: How long should pg_ctl wait when doing ``start``, ``stop`` or ``restart``. Default value is 60 seconds.
|
||||
- **use\_pg\_rewind**: try to use pg\_rewind on the former leader when it joins cluster as a replica.
|
||||
- **remove\_data\_directory\_on\_rewind\_failure**: If this option is enabled, Patroni will remove postgres data directory and recreate replica. Otherwise it will try to follow the new leader. Default value is **false**.
|
||||
@@ -135,9 +108,3 @@ REST API
|
||||
ZooKeeper
|
||||
----------
|
||||
- **hosts**: list of ZooKeeper cluster members in format: ['host1:port1', 'host2:port2', 'etc...'].
|
||||
|
||||
Watchdog
|
||||
--------
|
||||
- **mode**: ``off``, ``automatic`` or ``required``. When ``off`` watchdog is disabled. When ``automatic`` watchdog will be used if available, but ignored if it is not. When ``required`` the node will not become a leader unless watchdog can be succesfully enabled.
|
||||
- **device**: Path to watchdog device. Defaults to ``/dev/watchdog``.
|
||||
- **safety_margin**: Number of seconds of safety margin between watchdog triggering and leader key expiration.
|
||||
|
||||
@@ -21,7 +21,6 @@ We call Patroni a "template" because it is far from being a one-size-fits-all or
|
||||
dynamic_configuration
|
||||
ENVIRONMENT
|
||||
SETTINGS
|
||||
replica_bootstrap
|
||||
replication_modes
|
||||
pause
|
||||
releases
|
||||
|
||||
+1
-265
@@ -3,270 +3,6 @@
|
||||
Release notes
|
||||
=============
|
||||
|
||||
Version 1.3.6
|
||||
-------------
|
||||
|
||||
**Stability improvements**
|
||||
|
||||
- Verify process start time when checking if postgres is running. (Ants Aasma)
|
||||
|
||||
After a crash that doesn't clean up postmaster.pid there could be a new process with the same pid, resulting in a false positive for is_running(), which will lead to all kinds of bad behavior.
|
||||
|
||||
- Shutdown postgresql before bootstrap when we lost data directory (ainlolcat)
|
||||
|
||||
When data directory on the master is forcefully removed, postgres process can still stay alive for some time and prevent the replica created in place of that former master from starting or replicating.
|
||||
The fix makes Patroni cache the postmaster pid and its start time and let it terminate the old postmaster in case it is still running after the corresponding data directory has been removed.
|
||||
|
||||
- Perform crash recovery in a single user mode if postgres master dies (Alexander Kukushkin)
|
||||
|
||||
It is unsafe to start immediately as a standby and not possible to run ``pg_rewind`` if postgres hasn't been shut down cleanly.
|
||||
The single user crash recovery only kicks in if ``pg_rewind`` is enabled or there is no master at the moment.
|
||||
|
||||
**Consul improvements**
|
||||
|
||||
- Make it possible to provide datacenter configuration for Consul (DeathBorn, Alexander)
|
||||
|
||||
Before that Patroni was always communicating with datacenter of the host it runs on.
|
||||
|
||||
- Always send a token in X-Consul-Token http header (Alexander)
|
||||
|
||||
If ``consul.token`` is defined in Patroni configuration, we will always send it in the 'X-Consul-Token' http header.
|
||||
python-consul module tries to be "consistent" with Consul REST API, which doesn't accept token as a query parameter for `session API <https://www.consul.io/api/session.html>`__, but it still works with 'X-Consul-Token' header.
|
||||
|
||||
- Adjust session TTL if supplied value is smaller than the minimum possible (Stas Fomin, Alexander)
|
||||
|
||||
It could happen that the TTL provided in the Patroni configuration is smaller than the minimum one supported by Consul. In that case, Consul agent fails to create a new session.
|
||||
Without a session Patroni cannot create member and leader keys in the Consul KV store, resulting in an unhealthy cluster.
|
||||
|
||||
**Other improvements**
|
||||
|
||||
- Define custom log format via environment variable ``PATRONI_LOGFORMAT`` (Stas)
|
||||
|
||||
Allow disabling timestamps and other similar fields in Patroni logs if they are already added by the system logger (usually when Patroni runs as a service).
|
||||
|
||||
Version 1.3.5
|
||||
-------------
|
||||
|
||||
**Bugfix**
|
||||
|
||||
- Set role to 'uninitialized' if data directory was removed (Alexander Kukushkin)
|
||||
|
||||
If the node was running as a master it was preventing from failover.
|
||||
|
||||
**Stability improvement**
|
||||
|
||||
- Try to run postmaster in a single-user mode if we tried and failed to start postgres (Alexander)
|
||||
|
||||
Usually such problem happens when node running as a master was terminated and timelines were diverged.
|
||||
If ``recovery.conf`` has ``restore_command`` defined, there are really high chances that postgres will abort startup and leave controldata unchanged.
|
||||
It makes impossible to use ``pg_rewind``, which requires a clean shutdown.
|
||||
|
||||
**Consul improvements**
|
||||
|
||||
- Make it possible to specify health checks when creating session (Alexander)
|
||||
|
||||
If not specified, Consul will use "serfHealth". From one side it allows fast detection of isolated master, but from another side it makes it impossible for Patroni to tolerate short network lags.
|
||||
|
||||
**Bugfix**
|
||||
|
||||
- Fix watchdog on Python 3 (Ants Aasma)
|
||||
|
||||
A misunderstanding of the ioctl() call interface. If mutable=False then fcntl.ioctl() actually returns the arg buffer back.
|
||||
This accidentally worked on Python2 because int and str comparison did not return an error.
|
||||
Error reporting is actually done by raising IOError on Python2 and OSError on Python3.
|
||||
|
||||
Version 1.3.4
|
||||
-------------
|
||||
|
||||
**Different Consul improvements**
|
||||
|
||||
- Pass the consul token as a header (Andrew Colin Kissa)
|
||||
|
||||
Headers are now the prefered way to pass the token to the consul `API <https://www.consul.io/api/index.html#authentication>`__.
|
||||
|
||||
|
||||
- Advanced configuration for Consul (Alexander Kukushkin)
|
||||
|
||||
possibility to specify ``scheme``, ``token``, client and ca certificates :ref:`details <consul_settings>`.
|
||||
|
||||
- compatibility with python-consul-0.7.1 and above (Alexander)
|
||||
|
||||
new python-consul module has changed signature of some methods
|
||||
|
||||
- "Could not take out TTL lock" message was never logged (Alexander)
|
||||
|
||||
Not a critical bug, but lack of proper logging complicates investigation in case of problems.
|
||||
|
||||
|
||||
**Quote synchronous_standby_names using quote_ident**
|
||||
|
||||
- When writing ``synchronous_standby_names`` into the ``postgresql.conf`` its value must be quoted (Alexander)
|
||||
|
||||
If it is not quoted properly, PostgreSQL will effectively disable synchronous replication and continue to work.
|
||||
|
||||
|
||||
**Different bugfixes around pause state, mostly related to watchdog** (Alexander)
|
||||
|
||||
- Do not send keepalives if watchdog is not active
|
||||
- Avoid activating watchdog in a pause mode
|
||||
- Set correct postgres state in pause mode
|
||||
- Do not try to run queries from API if postgres is stopped
|
||||
|
||||
|
||||
Version 1.3.3
|
||||
-------------
|
||||
|
||||
**Bugfixes**
|
||||
|
||||
- synchronous replication was disabled shortly after promotion even when synchronous_mode_strict was turned on (Alexander Kukushkin)
|
||||
- create empty ``pg_ident.conf`` file if it is missing after restoring from the backup (Alexander)
|
||||
- open access in ``pg_hba.conf`` to all databases, not only postgres (Franco Bellagamba)
|
||||
|
||||
|
||||
Version 1.3.2
|
||||
-------------
|
||||
|
||||
**Bugfix**
|
||||
|
||||
- patronictl edit-config didn't work with ZooKeeper (Alexander Kukushkin)
|
||||
|
||||
|
||||
Version 1.3.1
|
||||
-------------
|
||||
|
||||
**Bugfix**
|
||||
|
||||
- failover via API was broken due to change in ``_MemberStatus`` (Alexander Kukushkin)
|
||||
|
||||
|
||||
Version 1.3
|
||||
-----------
|
||||
|
||||
Version 1.3 adds custom bootstrap possibility, significantly improves support for pg_rewind, enhances the
|
||||
synchronous mode support, adds configuration editing to patronictl and implements watchdog support on Linux.
|
||||
In addition, this is the first version to work correctly with PostgreSQL 10.
|
||||
|
||||
**Upgrade notice**
|
||||
|
||||
There are no known compatibility issues with the new version of Patroni. Configuration from version 1.2 should work
|
||||
without any changes. It is possible to upgrade by installing new packages and either restarting Patroni (will cause
|
||||
PostgreSQL restart), or by putting Patroni into a :ref:`pause mode <pause>` first and then restarting Patroni on all
|
||||
nodes in the cluster (Patroni in a pause mode will not attempt to stop/start PostgreSQL), resuming from the pause mode
|
||||
at the end.
|
||||
|
||||
**Custom bootstrap**
|
||||
|
||||
- Make the process of bootstrapping the cluster configurable (Alexander Kukushkin)
|
||||
|
||||
Allow custom bootstrap scripts instead of ``initdb`` when initializing the very first node in the cluster.
|
||||
The bootstrap command receives the name of the cluster and the path to the data directory. The resulting cluster can
|
||||
be configured to perform recovery, making it possible to bootstrap from a backup and do point in time recovery. Refer
|
||||
to the :ref:`documentaton page <custom_bootstrap>` for more detailed description of this feature.
|
||||
|
||||
**Smarter pg_rewind support**
|
||||
|
||||
- Decide on whether to run pg_rewind by looking at the timeline differences from the current master (Alexander)
|
||||
|
||||
Previously, Patroni had a fixed set of conditions to trigger pg_rewind, namely when starting a former master, when
|
||||
doing a switchover to the designated node for every other node in the cluster or when there is a replica with the
|
||||
nofailover tag. All those cases have in common a chance that some replica may be ahead of the new master. In some cases,
|
||||
pg_rewind did nothing, in some other ones it was not running when necessary. Instead of relying on this limited list
|
||||
of rules make Patroni compare the master and the replica WAL positions (using the streaming replication protocol)
|
||||
in order to reliably decide if rewind is necessary for the replica.
|
||||
|
||||
**Synchronous replication mode strict**
|
||||
|
||||
- Enhance synchronous replication support by adding the strict mode (James Sewell, Alexander)
|
||||
|
||||
Normally, when ``synchronous_mode`` is enabled and there are no replicas attached to the master, Patroni will disable
|
||||
synchronous replication in order to keep the master available for writes. The ``synchronous_mode_strict`` option
|
||||
changes that, when it is set Patroni will not disable the synchronous replication in a lack of replicas, effectively
|
||||
blocking all clients writing data to the master. In addition to the synchronous mode guarantee of preventing any data
|
||||
loss due to automatic failover, the strict mode ensures that each write is either durably stored on two nodes or not
|
||||
happening altogether if there is only one node in the cluster.
|
||||
|
||||
**Configuration editing with patronictl**
|
||||
|
||||
- Add configuration editing to patronictl (Ants Aasma, Alexander)
|
||||
|
||||
Add the ability to patronictl of editing dynamic cluster configuration stored in DCS. Support either specifying the
|
||||
parameter/values from the command-line, invoking the $EDITOR, or applying configuration from the yaml file.
|
||||
|
||||
**Linux watchdog support**
|
||||
|
||||
- Implement watchdog support for Linux (Ants)
|
||||
|
||||
Support Linux software watchdog in order to reboot the node where Patroni is not running or not responding (e.g because
|
||||
of the high load) The Linux software watchdog reboots the non-responsive node. It is possible to configure the watchdog
|
||||
device to use (`/dev/watchdog` by default) and the mode (on, automatic, off) from the watchdog section of the Patroni
|
||||
configuration. You can get more information from the :ref:`watchdog documentation <watchdog>`.
|
||||
|
||||
**Add support for PostgreSQL 10**
|
||||
|
||||
- Patroni is compatible with all beta versions of PostgreSQL 10 released so far and we expect it to be compatible with
|
||||
the PostgreSQL 10 when it will be released.
|
||||
|
||||
**PostgreSQL-related minor improvements**
|
||||
|
||||
- Define pg_hba.conf via the Patroni configuration file or the dynamic configuration in DCS (Alexander)
|
||||
|
||||
Allow to define the contents of ``pg_hba.conf`` in the ``pg_hba`` sub-section of the ``postgresql`` section of the
|
||||
configuration. This simplifies managing ``pg_hba.conf`` on multiple nodes, as one needs to define it only ones in DCS
|
||||
instead of logging to every node, changing it manually and reload the configuration.
|
||||
|
||||
When defined, the contents of this section will replace the current ``pg_hba.conf`` completely. Patroni ignores it
|
||||
if ``hba_file`` PostgreSQL parameter is set.
|
||||
|
||||
- Support connecting via a UNIX socket to the local PostgreSQL cluster (Alexander)
|
||||
|
||||
Add the ``use_unix_socket`` option to the ``postgresql`` section of Patroni configuration. When set to true and the
|
||||
PostgreSQL ``unix_socket_directories`` option is not empty, enables Patroni to use the first value from it to connect
|
||||
to the local PostgreSQL cluster. If ``unix_socket_directories`` is not defined, Patroni will assume its default value
|
||||
and omit the ``host`` parameter in the PostgreSQL connection string altogether.
|
||||
|
||||
- Support change of superuser and replication credentials on reload (Alexander)
|
||||
|
||||
- Support storing of configuration files outside of PostgreSQL data directory (@jouir)
|
||||
|
||||
Add the new configuration ``postgresql`` configuration directive ``config_dir``.
|
||||
It defaults to the data directory and must be writable by Patroni.
|
||||
|
||||
**Bug fixes and stability improvements**
|
||||
|
||||
- Handle EtcdEventIndexCleared and EtcdWatcherCleared exceptions (Alexander)
|
||||
|
||||
Faster recovery when the watch operation is ended by Etcd by avoiding useless retries.
|
||||
|
||||
- Remove error spinning on Etcd failure and reduce log spam (Ants)
|
||||
|
||||
Avoid immediate retrying and emitting stack traces in the log on the second and subsequent Etcd connection failures.
|
||||
|
||||
- Export locale variables when forking PostgreSQL processes (Oleksii Kliukin)
|
||||
|
||||
Avoid the `postmaster became multithreaded during startup` fatal error on non-English locales for PostgreSQL built with NLS.
|
||||
|
||||
- Extra checks when dropping the replication slot (Alexander)
|
||||
|
||||
In some cases Patroni is prevented from dropping the replication slot by the WAL sender.
|
||||
|
||||
- Truncate the replication slot name to 63 (NAMEDATALEN - 1) characters to comply with PostgreSQL naming rules (Nick Scott)
|
||||
|
||||
- Fix a race condition resulting in extra connections being opened to the PostgreSQL cluster from Patroni (Alexander)
|
||||
|
||||
- Release the leader key when the node restarts with an empty data directory (Alex Kerney)
|
||||
|
||||
- Set asynchronous executor busy when running bootstrap without a leader (Alexander)
|
||||
|
||||
Failure to do so could have resulted in errors stating the node belonged to a different cluster, as Patroni proceeded with
|
||||
the normal business while being bootstrapped by a bootstrap method that doesn't require a leader to be present in the
|
||||
cluster.
|
||||
|
||||
- Improve WAL-E replica creation method (Joar Wandborg, Alexander).
|
||||
|
||||
- Use csv.DictReader when parsing WAL-E base backup, accepting ISO dates with space-delimited date and time.
|
||||
- Support fetching current WAL position from the replica to estimate the amount of WAL to restore. Previously, the code used to call system information functions that were available only on the master node.
|
||||
|
||||
|
||||
Version 1.2
|
||||
-----------
|
||||
|
||||
@@ -655,4 +391,4 @@ This release adds support for *cascading replication* and simplifies Patroni man
|
||||
|
||||
The tests can be launched manually using the *behave* command. They are also launched automatically for pull requests and after commits.
|
||||
|
||||
Release notes for some older versions can be found on `project's github page <https://github.com/zalando/patroni/releases>`__.
|
||||
Releases notes for some older versions can be found on `project's github page <https://github.com/zalando/patroni/releases>`__.
|
||||
@@ -1,98 +0,0 @@
|
||||
Replica imaging and bootstrap
|
||||
=============================
|
||||
|
||||
Patroni allows customizing creation of a new replica. It also supports defining what happens when the new empty cluster
|
||||
is being bootstrapped. The distinction between two is well defined: Patroni creates replicas only if the ``initialize``
|
||||
key is present in Etcd for the cluster. If there is no ``initialize`` key - Patroni calls bootstrap exclusively on the
|
||||
first node that takes the initialize key lock.
|
||||
|
||||
.. _custom_bootstrap:
|
||||
|
||||
Bootstrap
|
||||
---------
|
||||
|
||||
PostgreSQL provides ``initdb`` command to initialize a new cluster and Patroni calls it by default. In certain cases,
|
||||
particularly when creating a new cluster as a copy of an existing one, it is necessary to replace a built-in method with
|
||||
custom actions. Patroni supports executing user-defined scripts to bootstrap new clusters, supplying some required
|
||||
arguments to them, i.e. the name of the cluster and the path to the data directory. This is configured in the
|
||||
``bootstrap`` section of the Patroni configuration. For example:
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
bootstrap:
|
||||
method: <custom_bootstrap_method_name>
|
||||
<custom_bootstrap_method_name>:
|
||||
command: <path_to_custom_bootstrap_script> [param1 [, ...]]
|
||||
recovery_conf:
|
||||
recovery_target_action: promote
|
||||
recovery_target_timeline: latest
|
||||
restore_command: <method_specific_restore_command>
|
||||
|
||||
|
||||
Each bootstrap method must define at least a ``name`` and a ``command``. A special ``initdb`` method is available to trigger
|
||||
the default behavior, in which case ``method`` parameter can be omitted altogether. The ``command`` can be specified using either
|
||||
an absolute path, or the one relative to the ``patroni`` command location. In addition to the fixed parameters defined
|
||||
in the configuration files, Patroni supplies two cluster-specific ones:
|
||||
|
||||
--scope
|
||||
Name of the cluster to be bootstrapped
|
||||
--datadir
|
||||
Path to the data directory of the cluster instance to be bootstrapped
|
||||
|
||||
If the bootstrap script returns 0, Patroni tries to configure and start the PostgreSQL instance produced by it. If any
|
||||
of the intermediate steps fail, or the script returns a non-zero value, Patroni assumes that the bootstrap has failed,
|
||||
cleans up after itself and releases the initialize lock to give another node the opportunity to bootstrap.
|
||||
|
||||
If a ``recovery_conf`` block is defined in the same section as the custom bootstrap method, Patroni will generate a
|
||||
``recovery.conf`` before starting the newly bootstrapped instance. Typically, such recovery.conf should contain at least
|
||||
one of the ``recovery_target_*`` parameters, together with the ``recovery_target_timeline`` set to ``promote``.
|
||||
|
||||
.. note:: Bootstrap methods are neither chained, nor fallen-back to the default one in case the primary one fails
|
||||
|
||||
|
||||
.. _custom_replica_creation:
|
||||
|
||||
Building replicas
|
||||
-----------------
|
||||
|
||||
Patroni uses tried and proven ``pg_basebackup`` in order to create new replicas. One downside of it is that it requires
|
||||
a running master node. Another one is the lack of 'on-the-fly' compression for the backup data and no built-in cleanup
|
||||
for outdated backup files. Some people prefer other backup solutions, such as ``WAL-E``, ``pgBackRest``, ``Barman`` and
|
||||
others, or simply roll their own scripts. In order to accommodate all those use-cases Patroni supports running custom
|
||||
scripts to clone a new replica. Those are configured in the ``postgresql`` configuration block:
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
postgresql:
|
||||
create_replica_method:
|
||||
- wal_e
|
||||
- basebackup
|
||||
wal_e:
|
||||
command: patroni_wale_restore
|
||||
no_master: 1
|
||||
envdir: {{WALE_ENV_DIR}}
|
||||
use_iam: 1
|
||||
|
||||
|
||||
The ``create_replica_method`` defines available replica creation methods and the order of executing them. Patroni will
|
||||
stop on the first one that returns 0. The basebackup is the built-in method and doesn't require any configuration. The
|
||||
rest of the methods should define a separate section in the configuration file, listing the command to execute and any
|
||||
custom parameters that should be passed to that command. All parameters will be passed in a ``--name=value`` format.
|
||||
Besides user-defined parameters, Patroni supplies a couple of cluster-specific ones:
|
||||
|
||||
--scope
|
||||
Which cluster this replica belongs to
|
||||
--datadir
|
||||
Path to the data directory of the replica
|
||||
--role
|
||||
Always 'replica'
|
||||
--connstring
|
||||
Connection string to connect to the cluster member to clone from (master or other replica). The user in the
|
||||
connection string can execute SQL and replication protocol commands.
|
||||
|
||||
A special ``no_master`` parameter, if defined, allows Patroni to call the replica creation method even if there is no
|
||||
running master or replicas. In that case, an empty string will be passed in a connection string. This is useful for
|
||||
restoring the formerly running cluster from the binary backup.
|
||||
|
||||
If all replica creation methods fail, Patroni will try again all methods in order during the next event loop cycle.
|
||||
|
||||
@@ -44,7 +44,7 @@ When ``synchronous_mode`` is on and a standby crashes, commits will block until
|
||||
|
||||
You can ensure that a standby never becomes the synchronous standby by setting ``nosync`` tag to true. This is recommended to set for standbys that are behind slow network connections and would cause performance degradation when becoming a synchronous standby.
|
||||
|
||||
Synchronous mode can be switched on and off via Patroni REST interface. See :ref:`dynamic configuration <dynamic_configuration>` for instructions.
|
||||
Synchronous mode can be switched on and off via Patroni REST interface. See `dynamic configuration <https://github.com/zalando/patroni/blob/master/docs/dynamic_configuration.rst>`__ for instructions.
|
||||
|
||||
|
||||
Synchronous mode implementation
|
||||
|
||||
@@ -1,40 +0,0 @@
|
||||
.. _watchdog:
|
||||
|
||||
================
|
||||
Watchdog support
|
||||
================
|
||||
|
||||
Having multiple PostgreSQL servers running as master can result in transactions lost due to diverging timelines. This situation is also called a split-brain problem. To avoid split-brain Patroni needs to ensure PostgreSQL will not accept any transaction commits after leader key expires in the DCS. Under normal circumstances Patroni will try to achieve this by stopping PostgreSQL when leader lock update fails for any reason. However, this may fail to happen due to various reasons:
|
||||
|
||||
- Patroni has crashed due to a bug, out-of-memory condition or by being accidentally killed by a system administrator.
|
||||
|
||||
- Shutting down PostgreSQL is too slow.
|
||||
|
||||
- Patroni does not get to run due to high load on the system, th VM being paused by the hypervisor, or other infrastructure issues.
|
||||
|
||||
To guarantee correct behavior under these conditions Patroni supports watchdog devices. Watchdog devices are software or hardware mechanisms that will reset the whole system when they do not get a keepalive heartbeat within a specified timeframe. This adds an additional layer of fail safe in case usual Patroni split-brain protection mechanisms fail.
|
||||
|
||||
Patroni will try to activate the watchdog before promoting PostgreSQL to master. If watchdog activation fails and watchdog mode is ``required`` then the node will refuse to become master. When deciding to participate in leader election Patroni will also check that watchdog configuration will allow it to become leader at all. After demoting PostgreSQL (for example due to a manual failover) Patroni will disable the watchdog again. Watchdog will also be disabled while Patroni is in paused state.
|
||||
|
||||
By default Patroni will set up the watchdog to expire 5 seconds before TTL expires. With the default setup of ``loop_wait=10`` and ``ttl=30`` this gives HA loop at least 15 seconds (``ttl`` - ``safety_margin`` - ``loop_wait``) to complete before the system gets forcefully reset. By default accessing DCS is configured to time out after 10 seconds. This means that when DCS is unavailable, for example due to network issues, Patroni and PostgreSQL will have at least 5 seconds (``ttl`` - ``safety_margin`` - ``loop_wait`` - ``retry_timeout``) to come to a state where all client connections are terminated.
|
||||
|
||||
Safety margin is the amount of time that Patroni reserves for time between leader key update and watchdog keepalive. Patroni will try to send a keepalive immediately after confirmation of leader key update. If Patroni process is suspended for extended amount of time at exactly the right moment the keepalive may be delayed for more than the safety margin without triggering the watchdog. This results in a window of time where watchdog will not trigger before leader key expiration, invalidating the guarantee. To be absolutely sure that watchdog will trigger under all circumstances set up the watchdog to expire after half of TTL by setting ``safety_margin`` to -1 to set watchdog timeout to ``ttl // 2``. If you need this guarantee you probably should increase ``ttl`` and/or reduce ``loop_wait`` and ``retry_timeout``.
|
||||
|
||||
Currently watchdogs are only supported using Linux watchdog device interface.
|
||||
|
||||
Setting up software watchdog on Linux
|
||||
-------------------------------------
|
||||
|
||||
Default Patroni configuration will try to use ``/dev/watchdog`` on Linux if it is accessible to Patroni. For most use cases using software watchdog built into the Linux kernel is secure enough.
|
||||
|
||||
To enable software watchdog issue the following commands as root before starting Patroni:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
modprobe softdog
|
||||
# Replace postgres with the user you will be running patroni under
|
||||
chown postgres /dev/watchdog
|
||||
|
||||
For testing it may be helpful to disable rebooting by adding ``soft_noboot=1`` to the modprobe command line. In this case the watchdog will just log a line in kernel ring buffer, visible via `dmesg`.
|
||||
|
||||
Patroni will log information about the watchdog when it is successfully enabled.
|
||||
@@ -11,12 +11,3 @@ Upstart job for Ubuntu 12.04 or 14.04. Requires Upstart > 1.4. Intended for sys
|
||||
|
||||
### patroni.service
|
||||
Systemd service file, to be copied to /etc/systemd/system/patroni.service, tested on Centos 7.1 with Patroni installed from pip.
|
||||
|
||||
### patroni
|
||||
Init.d service file for Debian-like distributions. Copy it to /etc/init.d/, make executable:
|
||||
```chmod 755 /etc/init.d/patroni``` and run with ```service patroni start```, or make it starting on boot with ```update-rc.d patroni defaults```. Also you might edit some configuration variables in it:
|
||||
PATRONI for patroni.py location
|
||||
CONF for configuration file
|
||||
LOGFILE for log (script creates it if does not exist)
|
||||
|
||||
Note. If you have several versions of Postgres installed, please add to POSTGRES_VERSION the release number which you wish to run. Script uses this value to append PATH environment with correct path to Postgres bin.
|
||||
|
||||
@@ -1,144 +0,0 @@
|
||||
#!/bin/sh
|
||||
#
|
||||
### BEGIN INIT INFO
|
||||
# Provides: patroni
|
||||
# Required-Start: $remote_fs $syslog
|
||||
# Required-Stop: $remote_fs $syslog
|
||||
# Default-Start: 2 3 4 5
|
||||
# Default-Stop: 0 1 6
|
||||
# Short-Description: Patroni init script
|
||||
# Description: Runners to orchestrate a high-availability PostgreSQL
|
||||
### END INIT INFO
|
||||
|
||||
### BEGIN USER CONFIGURATION
|
||||
|
||||
CONF="/etc/patroni/postgres.yml"
|
||||
LOGFILE="/var/log/patroni.log"
|
||||
USER="postgres"
|
||||
GROUP="postgres"
|
||||
|
||||
NAME=patroni
|
||||
PATRONI="/opt/patroni/$NAME.py"
|
||||
PIDFILE="/var/run/$NAME.pid"
|
||||
|
||||
# Set this parameter, if you have several Postgres versions installed
|
||||
# POSTGRES_VERSION="9.4"
|
||||
POSTGRES_VERSION=""
|
||||
|
||||
### END USER CONFIGURATION
|
||||
|
||||
. /lib/lsb/init-functions
|
||||
|
||||
# Loading this library for get_versions() function
|
||||
if test ! -e /usr/share/postgresql-common/init.d-functions; then
|
||||
log_failure_msg "Probably postgresql-common does not installed."
|
||||
exit 1
|
||||
else
|
||||
. /usr/share/postgresql-common/init.d-functions
|
||||
fi
|
||||
|
||||
# Is there Patroni executable?
|
||||
if test ! -e $PATRONI; then
|
||||
log_failure_msg "Patroni executable $PATRONI does not exist."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Is there Patroni configuration file?
|
||||
if test ! -e $CONF; then
|
||||
log_failure_msg "Patroni configuration file $CONF does not exist."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Create logfile if doesn't exist
|
||||
if test ! -e $LOGFILE; then
|
||||
log_action_msg "Creating logfile for Patroni..."
|
||||
touch $LOGFILE
|
||||
chown $USER:$GROUP $LOGFILE
|
||||
fi
|
||||
|
||||
prepare_pgpath() {
|
||||
if [ "$POSTGRES_VERSION" != "" ]; then
|
||||
if [ -x /usr/lib/postgresql/$POSTGRES_VERSION/bin/pg_ctl ]; then
|
||||
PGPATH="/usr/lib/postgresql/$POSTGRES_VERSION/bin"
|
||||
else
|
||||
log_failure_msg "Postgres version incorrect, check POSTGRES_VERSION variable."
|
||||
exit 0
|
||||
fi
|
||||
else
|
||||
get_versions
|
||||
if echo $versions | grep -q -e "\s"; then
|
||||
log_warning_msg "You have several Postgres versions installed. Please, use POSTGRES_VERSION to define correct environment."
|
||||
else
|
||||
versions=`echo $versions | sed -e 's/^[ \t]*//'`
|
||||
PGPATH="/usr/lib/postgresql/$versions/bin"
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
get_pid() {
|
||||
if test -e $PIDFILE; then
|
||||
PID=`cat $PIDFILE`
|
||||
CHILDPID=`ps --ppid $PID -o %p --no-headers`
|
||||
else
|
||||
log_failure_msg "Could not find PID file. Patroni probably down."
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
|
||||
case "$1" in
|
||||
start)
|
||||
prepare_pgpath
|
||||
PGPATH=$PATH:$PGPATH
|
||||
log_success_msg "Starting Patroni\n"
|
||||
exec start-stop-daemon --start --quiet \
|
||||
--background \
|
||||
--pidfile $PIDFILE --make-pidfile \
|
||||
--chuid $USER:$GROUP \
|
||||
--chdir `eval echo ~$USER` \
|
||||
--exec $PATRONI \
|
||||
--startas /bin/sh -- \
|
||||
-c "/usr/bin/env PATH=$PGPATH /usr/bin/python $PATRONI $CONF >> $LOGFILE 2>&1"
|
||||
;;
|
||||
|
||||
stop)
|
||||
log_success_msg "Stopping Patroni"
|
||||
get_pid
|
||||
start-stop-daemon --stop --pid $CHILDPID
|
||||
start-stop-daemon --stop --pidfile $PIDFILE --remove-pidfile --quiet
|
||||
;;
|
||||
|
||||
reload)
|
||||
log_success_msg "Reloading Patroni configuration"
|
||||
get_pid
|
||||
kill -HUP $CHILDPID
|
||||
;;
|
||||
|
||||
status)
|
||||
get_pid
|
||||
if start-stop-daemon -T --pid $CHILDPID; then
|
||||
log_success_msg "Patroni is running\n"
|
||||
exit 0
|
||||
else
|
||||
log_warning_msg "Patroni in not running\n"
|
||||
fi
|
||||
;;
|
||||
|
||||
restart)
|
||||
$0 stop
|
||||
$0 start
|
||||
;;
|
||||
|
||||
*)
|
||||
echo "Usage: /etc/init.d/$NAME {start|stop|restart|reload|status}"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
if [ $? -eq 0 ]; then
|
||||
echo .
|
||||
exit 0
|
||||
else
|
||||
echo " failed"
|
||||
exit 1
|
||||
fi
|
||||
@@ -21,7 +21,7 @@ ExecStart=/bin/patroni /etc/patroni.yml
|
||||
KillMode=process
|
||||
|
||||
# Give a reasonable amount of time for the server to start up/shut down
|
||||
TimeoutSec=30
|
||||
TimeoutSec=10
|
||||
|
||||
# Do not restart the service if it crashes, we want to manually inspect database on failure
|
||||
Restart=no
|
||||
|
||||
@@ -1,22 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
while getopts ":-:" optchar; do
|
||||
[[ "${optchar}" == "-" ]] || continue
|
||||
case "${OPTARG}" in
|
||||
datadir=* )
|
||||
PGDATA=${OPTARG#*=}
|
||||
;;
|
||||
dbname=* )
|
||||
DBNAME=${OPTARG#*=}
|
||||
;;
|
||||
walmethod=* )
|
||||
WALMETHOD=${OPTARG#*=}
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
[[ -z $PGDATA || -z $DBNAME || -z $WALMETHOD ]] && exit 1
|
||||
|
||||
[[ $WALMETHOD != "none" ]] && WALMETHOD="-X $WALMETHOD" || WALMETHOD=""
|
||||
|
||||
exec pg_basebackup -D $PGDATA $WALMETHOD -c fast -d $DBNAME
|
||||
@@ -1,21 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -x
|
||||
|
||||
while getopts ":-:" optchar; do
|
||||
[[ "${optchar}" == "-" ]] || continue
|
||||
case "${OPTARG}" in
|
||||
datadir=* )
|
||||
PGDATA=${OPTARG#*=}
|
||||
;;
|
||||
sourcedir=* )
|
||||
SOURCE=${OPTARG#*=}
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
[[ -z $PGDATA || -z $SOURCE ]] && exit 1
|
||||
|
||||
mkdir -p $(dirname $PGDATA)
|
||||
|
||||
exec cp -af $SOURCE $PGDATA
|
||||
@@ -34,8 +34,7 @@ Feature: basic replication
|
||||
And postgres1 role is the primary after 10 seconds
|
||||
|
||||
Scenario: check rejoin of the former master with pg_rewind
|
||||
Given I add the table splitbrain to postgres0
|
||||
And I start postgres0
|
||||
Given I start postgres0
|
||||
Then postgres0 role is the secondary after 20 seconds
|
||||
When I add the table buz to postgres1
|
||||
Then table buz is present on postgres0 after 20 seconds
|
||||
|
||||
@@ -1,17 +0,0 @@
|
||||
Feature: custom bootstrap
|
||||
We should check that patroni can bootstrap a new cluster from a backup
|
||||
|
||||
Scenario: clone existing cluster using pg_basebackup
|
||||
Given I start postgres0
|
||||
Then postgres0 is a leader after 10 seconds
|
||||
When I add the table foo to postgres0
|
||||
And I start postgres1 in a cluster batman1 as a clone of postgres0
|
||||
Then postgres1 is a leader of batman1 after 10 seconds
|
||||
Then table foo is present on postgres1 after 10 seconds
|
||||
|
||||
Scenario: make a backup and do a restore into a new cluster
|
||||
Given I add the table bar to postgres1
|
||||
And I do a backup of postgres1
|
||||
When I start postgres2 in a cluster batman2 from backup
|
||||
Then postgres2 is a leader of batman2 after 10 seconds
|
||||
And table bar is present on postgres2 after 10 seconds
|
||||
+42
-337
@@ -1,18 +1,14 @@
|
||||
import abc
|
||||
import consul
|
||||
import datetime
|
||||
import etcd
|
||||
import kazoo.client
|
||||
import kazoo.exceptions
|
||||
import os
|
||||
import psutil
|
||||
import psycopg2
|
||||
import shutil
|
||||
import signal
|
||||
import six
|
||||
import subprocess
|
||||
import tempfile
|
||||
import threading
|
||||
import time
|
||||
import yaml
|
||||
|
||||
@@ -78,28 +74,18 @@ class AbstractController(object):
|
||||
if self._log:
|
||||
self._log.close()
|
||||
|
||||
def cancel_background(self):
|
||||
pass
|
||||
|
||||
|
||||
class PatroniController(AbstractController):
|
||||
__PORT = 5440
|
||||
PATRONI_CONFIG = '{}.yml'
|
||||
""" starts and stops individual patronis"""
|
||||
|
||||
def __init__(self, context, name, work_directory, output_dir, custom_config=None):
|
||||
def __init__(self, context, name, work_directory, output_dir, tags=None):
|
||||
super(PatroniController, self).__init__(context, 'patroni_' + name, work_directory, output_dir)
|
||||
PatroniController.__PORT += 1
|
||||
self._data_dir = os.path.join(work_directory, 'data', name)
|
||||
self._connstring = None
|
||||
if custom_config and 'watchdog' in custom_config:
|
||||
self.watchdog = WatchdogMonitor(name, work_directory, output_dir)
|
||||
custom_config['watchdog'] = {'driver': 'testing', 'device': self.watchdog.fifo_path, 'mode': 'required'}
|
||||
else:
|
||||
self.watchdog = None
|
||||
|
||||
self._config = self._make_patroni_test_config(name, custom_config)
|
||||
self._closables = []
|
||||
self._config = self._make_patroni_test_config(name, tags)
|
||||
|
||||
self._conn = None
|
||||
self._curs = None
|
||||
@@ -123,8 +109,6 @@ class PatroniController(AbstractController):
|
||||
yaml.safe_dump(config, w, default_flow_style=False)
|
||||
|
||||
def _start(self):
|
||||
if self.watchdog:
|
||||
self.watchdog.start()
|
||||
return subprocess.Popen(['coverage', 'run', '--source=patroni', '-p', 'patroni.py', self._config],
|
||||
stdout=self._log, stderr=subprocess.STDOUT, cwd=self._work_directory)
|
||||
|
||||
@@ -132,16 +116,11 @@ class PatroniController(AbstractController):
|
||||
if postgres:
|
||||
return subprocess.call(['pg_ctl', '-D', self._data_dir, 'stop', '-mi', '-w'])
|
||||
super(PatroniController, self).stop(kill, timeout)
|
||||
if self.watchdog:
|
||||
self.watchdog.stop()
|
||||
|
||||
def _is_accessible(self):
|
||||
cursor = self.query("SELECT 1", fail_ok=True)
|
||||
if cursor is not None:
|
||||
cursor.execute("SET synchronous_commit TO 'local'")
|
||||
return True
|
||||
return self.query("SELECT 1", fail_ok=True) is not None
|
||||
|
||||
def _make_patroni_test_config(self, name, custom_config):
|
||||
def _make_patroni_test_config(self, name, tags):
|
||||
patroni_config_name = self.PATRONI_CONFIG.format(name)
|
||||
patroni_config_path = os.path.join(self._output_dir, patroni_config_name)
|
||||
|
||||
@@ -153,37 +132,24 @@ class PatroniController(AbstractController):
|
||||
|
||||
config['postgresql']['listen'] = config['postgresql']['connect_address'] = '{0}:{1}'.format(host, self.__PORT)
|
||||
|
||||
config['name'] = name
|
||||
config['postgresql']['data_dir'] = self._data_dir
|
||||
config['postgresql']['use_unix_socket'] = True
|
||||
config['postgresql']['parameters'].update({
|
||||
'logging_collector': 'on', 'log_destination': 'csvlog', 'log_directory': self._output_dir,
|
||||
'log_filename': name + '.log', 'log_statement': 'all', 'log_min_messages': 'debug1',
|
||||
'unix_socket_directories': self._data_dir})
|
||||
|
||||
if 'bootstrap' in config:
|
||||
config['bootstrap']['post_bootstrap'] = 'psql -w -c "SELECT 1"'
|
||||
if 'initdb' in config['bootstrap']:
|
||||
config['bootstrap']['initdb'].extend([{'auth': 'md5'}, {'auth-host': 'md5'}])
|
||||
|
||||
if custom_config is not None:
|
||||
def recursive_update(dst, src):
|
||||
for k, v in src.items():
|
||||
if k in dst and isinstance(dst[k], dict):
|
||||
recursive_update(dst[k], v)
|
||||
else:
|
||||
dst[k] = v
|
||||
recursive_update(config, custom_config)
|
||||
|
||||
with open(patroni_config_path, 'w') as f:
|
||||
yaml.safe_dump(config, f, default_flow_style=False)
|
||||
|
||||
user = config['postgresql'].get('authentication', config['postgresql']).get('superuser', {})
|
||||
self._connkwargs = {k: user[n] for n, k in [('username', 'user'), ('password', 'password')] if n in user}
|
||||
self._connkwargs.update({'host': host, 'port': self.__PORT, 'database': 'postgres'})
|
||||
|
||||
self._replication = config['postgresql'].get('authentication', config['postgresql']).get('replication', {})
|
||||
self._replication.update({'host': host, 'port': self.__PORT, 'database': 'postgres'})
|
||||
config['name'] = name
|
||||
config['postgresql']['data_dir'] = self._data_dir
|
||||
config['postgresql']['parameters'].update({
|
||||
'logging_collector': 'on', 'log_destination': 'csvlog', 'log_directory': self._output_dir,
|
||||
'log_filename': name + '.log', 'log_statement': 'all', 'log_min_messages': 'debug1'})
|
||||
|
||||
if 'bootstrap' in config and 'initdb' in config['bootstrap']:
|
||||
config['bootstrap']['initdb'].extend([{'auth': 'md5'}, {'auth-host': 'md5'}])
|
||||
|
||||
if tags:
|
||||
config['tags'] = tags
|
||||
|
||||
with open(patroni_config_path, 'w') as f:
|
||||
yaml.safe_dump(config, f, default_flow_style=False)
|
||||
|
||||
return patroni_config_path
|
||||
|
||||
@@ -219,101 +185,10 @@ class PatroniController(AbstractController):
|
||||
time.sleep(1)
|
||||
return False
|
||||
|
||||
def get_watchdog(self):
|
||||
return self.watchdog
|
||||
|
||||
def _get_pid(self):
|
||||
try:
|
||||
pidfile = os.path.join(self._data_dir, 'postmaster.pid')
|
||||
if not os.path.exists(pidfile):
|
||||
return None
|
||||
return int(open(pidfile).readline().strip())
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
def database_is_running(self):
|
||||
pid = self._get_pid()
|
||||
if not pid:
|
||||
return False
|
||||
try:
|
||||
os.kill(pid, 0)
|
||||
except OSError:
|
||||
return False
|
||||
return True
|
||||
|
||||
def patroni_hang(self, timeout):
|
||||
hang = ProcessHang(self._handle.pid, timeout)
|
||||
self._closables.append(hang)
|
||||
hang.start()
|
||||
|
||||
def checkpoint_hang(self, timeout):
|
||||
pid = self._get_pid()
|
||||
if not pid:
|
||||
return False
|
||||
proc = psutil.Process(pid)
|
||||
for child in proc.children():
|
||||
if 'checkpoint' in child.cmdline()[0]:
|
||||
checkpointer = child
|
||||
break
|
||||
else:
|
||||
return False
|
||||
hang = ProcessHang(checkpointer.pid, timeout)
|
||||
self._closables.append(hang)
|
||||
hang.start()
|
||||
return True
|
||||
|
||||
def cancel_background(self):
|
||||
for obj in self._closables:
|
||||
obj.close()
|
||||
self._closables = []
|
||||
|
||||
def terminate_backends(self):
|
||||
pid = self._get_pid()
|
||||
if not pid:
|
||||
return False
|
||||
proc = psutil.Process(pid)
|
||||
for p in proc.children():
|
||||
if 'process' not in p.cmdline()[0]:
|
||||
p.terminate()
|
||||
|
||||
@property
|
||||
def backup_source(self):
|
||||
return 'postgres://{username}:{password}@{host}:{port}/{database}'.format(**self._replication)
|
||||
|
||||
def backup(self, dest='basebackup'):
|
||||
subprocess.call([PatroniPoolController.BACKUP_SCRIPT, '--walmethod=none',
|
||||
'--datadir=' + os.path.join(self._output_dir, dest),
|
||||
'--dbname=' + self.backup_source])
|
||||
|
||||
|
||||
class ProcessHang(object):
|
||||
|
||||
"""A background thread implementing a cancelable process hang via SIGSTOP."""
|
||||
|
||||
def __init__(self, pid, timeout):
|
||||
self._cancelled = threading.Event()
|
||||
self._thread = threading.Thread(target=self.run)
|
||||
self.pid = pid
|
||||
self.timeout = timeout
|
||||
|
||||
def start(self):
|
||||
self._thread.start()
|
||||
|
||||
def run(self):
|
||||
os.kill(self.pid, signal.SIGSTOP)
|
||||
try:
|
||||
self._cancelled.wait(self.timeout)
|
||||
finally:
|
||||
os.kill(self.pid, signal.SIGCONT)
|
||||
|
||||
def close(self):
|
||||
self._cancelled.set()
|
||||
self._thread.join()
|
||||
|
||||
|
||||
class AbstractDcsController(AbstractController):
|
||||
|
||||
_CLUSTER_NODE = '/service/{0}'
|
||||
_CLUSTER_NODE = '/service/batman'
|
||||
|
||||
def __init__(self, context, mktemp=True):
|
||||
work_directory = mktemp and tempfile.mkdtemp() or None
|
||||
@@ -328,11 +203,11 @@ class AbstractDcsController(AbstractController):
|
||||
if self._work_directory:
|
||||
shutil.rmtree(self._work_directory)
|
||||
|
||||
def path(self, key=None, scope='batman'):
|
||||
return self._CLUSTER_NODE.format(scope) + (key and '/' + key or '')
|
||||
def path(self, key=None):
|
||||
return self._CLUSTER_NODE + (key and '/' + key or '')
|
||||
|
||||
@abc.abstractmethod
|
||||
def query(self, key, scope='batman'):
|
||||
def query(self, key):
|
||||
""" query for a value of a given key """
|
||||
|
||||
@abc.abstractmethod
|
||||
@@ -361,19 +236,18 @@ class ConsulController(AbstractDcsController):
|
||||
super(ConsulController, self).__init__(context)
|
||||
os.environ['PATRONI_CONSUL_HOST'] = 'localhost:8500'
|
||||
self._client = consul.Consul()
|
||||
self._config_file = None
|
||||
|
||||
def _start(self):
|
||||
self._config_file = self._work_directory + '.json'
|
||||
with open(self._config_file, 'wb') as f:
|
||||
config_file = self._work_directory + '.json'
|
||||
with open(config_file, 'wb') as f:
|
||||
f.write(b'{"session_ttl_min":"5s","server":true,"bootstrap":true,"advertise_addr":"127.0.0.1"}')
|
||||
return subprocess.Popen(['consul', 'agent', '-config-file', self._config_file, '-data-dir',
|
||||
self._work_directory], stdout=self._log, stderr=subprocess.STDOUT)
|
||||
return subprocess.Popen(['consul', 'agent', '-config-file', config_file, '-data-dir', self._work_directory],
|
||||
stdout=self._log, stderr=subprocess.STDOUT)
|
||||
|
||||
def stop(self, kill=False, timeout=15):
|
||||
super(ConsulController, self).stop(kill=kill, timeout=timeout)
|
||||
if self._config_file:
|
||||
os.unlink(self._config_file)
|
||||
if self._work_directory:
|
||||
os.unlink(self._work_directory + '.json')
|
||||
|
||||
def _is_running(self):
|
||||
try:
|
||||
@@ -381,18 +255,18 @@ class ConsulController(AbstractDcsController):
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def path(self, key=None, scope='batman'):
|
||||
return super(ConsulController, self).path(key, scope)[1:]
|
||||
def path(self, key=None):
|
||||
return super(ConsulController, self).path(key)[1:]
|
||||
|
||||
def query(self, key, scope='batman'):
|
||||
_, value = self._client.kv.get(self.path(key, scope))
|
||||
def query(self, key):
|
||||
_, value = self._client.kv.get(self.path(key))
|
||||
return value and value['Value'].decode('utf-8')
|
||||
|
||||
def set(self, key, value):
|
||||
self._client.kv.put(self.path(key), value)
|
||||
|
||||
def cleanup_service_tree(self):
|
||||
self._client.kv.delete(self.path(scope=''), recurse=True)
|
||||
self._client.kv.delete(self.path(), recurse=True)
|
||||
|
||||
def start(self, max_wait_limit=15):
|
||||
super(ConsulController, self).start(max_wait_limit)
|
||||
@@ -411,9 +285,9 @@ class EtcdController(AbstractDcsController):
|
||||
return subprocess.Popen(["etcd", "--debug", "--data-dir", self._work_directory],
|
||||
stdout=self._log, stderr=subprocess.STDOUT)
|
||||
|
||||
def query(self, key, scope='batman'):
|
||||
def query(self, key):
|
||||
try:
|
||||
return self._client.get(self.path(key, scope)).value
|
||||
return self._client.get(self.path(key)).value
|
||||
except etcd.EtcdKeyNotFound:
|
||||
return None
|
||||
|
||||
@@ -422,7 +296,7 @@ class EtcdController(AbstractDcsController):
|
||||
|
||||
def cleanup_service_tree(self):
|
||||
try:
|
||||
self._client.delete(self.path(scope=''), recursive=True)
|
||||
self._client.delete(self.path(), recursive=True)
|
||||
except (etcd.EtcdKeyNotFound, etcd.EtcdConnectionFailed):
|
||||
return
|
||||
except Exception as e:
|
||||
@@ -449,9 +323,9 @@ class ZooKeeperController(AbstractDcsController):
|
||||
def _start(self):
|
||||
pass # TODO: implement later
|
||||
|
||||
def query(self, key, scope='batman'):
|
||||
def query(self, key):
|
||||
try:
|
||||
return self._client.get(self.path(key, scope))[0].decode('utf-8')
|
||||
return self._client.get(self.path(key))[0].decode('utf-8')
|
||||
except kazoo.exceptions.NoNodeError:
|
||||
return None
|
||||
|
||||
@@ -460,7 +334,7 @@ class ZooKeeperController(AbstractDcsController):
|
||||
|
||||
def cleanup_service_tree(self):
|
||||
try:
|
||||
self._client.delete(self.path(scope=''), recursive=True)
|
||||
self._client.delete(self.path(), recursive=True)
|
||||
except (kazoo.exceptions.NoNodeError):
|
||||
return
|
||||
except Exception as e:
|
||||
@@ -485,8 +359,6 @@ class ExhibitorController(ZooKeeperController):
|
||||
|
||||
class PatroniPoolController(object):
|
||||
|
||||
BACKUP_SCRIPT = 'features/backup_create.sh'
|
||||
|
||||
def __init__(self, context):
|
||||
self._context = context
|
||||
self._dcs = None
|
||||
@@ -511,16 +383,13 @@ class PatroniPoolController(object):
|
||||
def output_dir(self):
|
||||
return self._output_dir
|
||||
|
||||
def start(self, name, max_wait_limit=20, custom_config=None):
|
||||
def start(self, name, max_wait_limit=20, tags=None):
|
||||
if name not in self._processes:
|
||||
self._processes[name] = PatroniController(self._context, name, self.patroni_path,
|
||||
self._output_dir, custom_config)
|
||||
self._processes[name] = PatroniController(self._context, name, self.patroni_path, self._output_dir, tags)
|
||||
self._processes[name].start(max_wait_limit)
|
||||
|
||||
def __getattr__(self, func):
|
||||
if func not in ['stop', 'query', 'write_label', 'read_label', 'check_role_has_changed_to', 'add_tag_to_config',
|
||||
'get_watchdog', 'database_is_running', 'checkpoint_hang', 'patroni_hang',
|
||||
'terminate_backends', 'backup']:
|
||||
if func not in ['stop', 'query', 'write_label', 'read_label', 'check_role_has_changed_to', 'add_tag_to_config']:
|
||||
raise AttributeError("PatroniPoolController instance has no attribute '{0}'".format(func))
|
||||
|
||||
def wrapper(name, *args, **kwargs):
|
||||
@@ -529,7 +398,6 @@ class PatroniPoolController(object):
|
||||
|
||||
def stop_all(self):
|
||||
for ctl in self._processes.values():
|
||||
ctl.cancel_background()
|
||||
ctl.stop()
|
||||
self._processes.clear()
|
||||
|
||||
@@ -540,53 +408,6 @@ class PatroniPoolController(object):
|
||||
os.makedirs(feature_dir)
|
||||
self._output_dir = feature_dir
|
||||
|
||||
def clone(self, from_name, cluster_name, to_name):
|
||||
f = self._processes[from_name]
|
||||
custom_config = {
|
||||
'scope': cluster_name,
|
||||
'bootstrap': {
|
||||
'method': 'pg_basebackup',
|
||||
'pg_basebackup': {
|
||||
'command': self.BACKUP_SCRIPT + ' --walmethod=stream --dbname=' + f.backup_source
|
||||
}
|
||||
},
|
||||
'postgresql': {
|
||||
'parameters': {
|
||||
'archive_mode': 'on',
|
||||
'archive_command': 'mkdir -p {0} && test ! -f {0}/%f && cp %p {0}/%f'.format(
|
||||
os.path.join(self._output_dir, 'wal_archive'))
|
||||
},
|
||||
'authentication': {
|
||||
'superuser': {'password': 'zalando1'},
|
||||
'replication': {'password': 'rep-pass1'}
|
||||
}
|
||||
}
|
||||
}
|
||||
self.start(to_name, custom_config=custom_config)
|
||||
|
||||
def bootstrap_from_backup(self, name, cluster_name):
|
||||
custom_config = {
|
||||
'scope': cluster_name,
|
||||
'bootstrap': {
|
||||
'method': 'backup_restore',
|
||||
'backup_restore': {
|
||||
'command': 'features/backup_restore.sh --sourcedir=' + os.path.join(self._output_dir, 'basebackup'),
|
||||
'recovery_conf': {
|
||||
'recovery_target_action': 'promote',
|
||||
'recovery_target_timeline': 'latest',
|
||||
'restore_command': 'cp {0}/wal_archive/%f %p'.format(self._output_dir)
|
||||
}
|
||||
}
|
||||
},
|
||||
'postgresql': {
|
||||
'authentication': {
|
||||
'superuser': {'password': 'zalando2'},
|
||||
'replication': {'password': 'rep-pass2'}
|
||||
}
|
||||
}
|
||||
}
|
||||
self.start(name, custom_config=custom_config)
|
||||
|
||||
@property
|
||||
def dcs(self):
|
||||
if self._dcs is None:
|
||||
@@ -595,122 +416,6 @@ class PatroniPoolController(object):
|
||||
return self._dcs
|
||||
|
||||
|
||||
class WatchdogMonitor(object):
|
||||
"""Testing harness for emulating a watchdog device as a named pipe. Because we can't easily emulate ioctl's we
|
||||
require a custom driver on Patroni side. The device takes no action, only notes if it was pinged and/or triggered.
|
||||
"""
|
||||
def __init__(self, name, work_directory, output_dir):
|
||||
self.fifo_path = os.path.join(work_directory, 'data', 'watchdog.{0}.fifo'.format(name))
|
||||
self.fifo_file = None
|
||||
self._stop_requested = False # Relying on bool setting being atomic
|
||||
self._thread = None
|
||||
self.last_ping = None
|
||||
self.was_pinged = False
|
||||
self.was_closed = False
|
||||
self._was_triggered = False
|
||||
self.timeout = 60
|
||||
self._log_file = open(os.path.join(output_dir, 'watchdog.{0}.log'.format(name)), 'w')
|
||||
self._log("watchdog {0} initialized".format(name))
|
||||
|
||||
def _log(self, msg):
|
||||
tstamp = datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S,%f")
|
||||
self._log_file.write("{0}: {1}\n".format(tstamp, msg))
|
||||
|
||||
def start(self):
|
||||
assert self._thread is None
|
||||
self._stop_requested = False
|
||||
self._log("starting fifo {0}".format(self.fifo_path))
|
||||
fifo_dir = os.path.dirname(self.fifo_path)
|
||||
if os.path.exists(self.fifo_path):
|
||||
os.unlink(self.fifo_path)
|
||||
elif not os.path.exists(fifo_dir):
|
||||
os.mkdir(fifo_dir)
|
||||
os.mkfifo(self.fifo_path)
|
||||
self.last_ping = time.time()
|
||||
|
||||
self._thread = threading.Thread(target=self.run)
|
||||
self._thread.start()
|
||||
|
||||
def run(self):
|
||||
try:
|
||||
while not self._stop_requested:
|
||||
self._log("opening")
|
||||
self.fifo_file = os.open(self.fifo_path, os.O_RDONLY)
|
||||
try:
|
||||
self._log("Fifo {0} connected".format(self.fifo_path))
|
||||
self.was_closed = False
|
||||
while not self._stop_requested:
|
||||
c = os.read(self.fifo_file, 1)
|
||||
|
||||
if c == b'X':
|
||||
self._log("Stop requested")
|
||||
return
|
||||
elif c == b'':
|
||||
self._log("Pipe closed")
|
||||
break
|
||||
elif c == b'C':
|
||||
command = b''
|
||||
c = os.read(self.fifo_file, 1)
|
||||
while c != b'\n' and c != b'':
|
||||
command += c
|
||||
c = os.read(self.fifo_file, 1)
|
||||
command = command.decode('utf8')
|
||||
|
||||
if command.startswith('timeout='):
|
||||
self.timeout = int(command.split('=')[1])
|
||||
self._log("timeout={0}".format(self.timeout))
|
||||
elif c in [b'V', b'1']:
|
||||
cur_time = time.time()
|
||||
if cur_time - self.last_ping > self.timeout:
|
||||
self._log("Triggered")
|
||||
self._was_triggered = True
|
||||
if c == b'V':
|
||||
self._log("magic close")
|
||||
self.was_closed = True
|
||||
elif c == b'1':
|
||||
self.was_pinged = True
|
||||
self._log("ping after {0} seconds".format(cur_time - (self.last_ping or cur_time)))
|
||||
self.last_ping = cur_time
|
||||
else:
|
||||
self._log('Unknown command {0} received from fifo'.format(c))
|
||||
finally:
|
||||
self.was_closed = True
|
||||
self._log("closing")
|
||||
os.close(self.fifo_file)
|
||||
except Exception as e:
|
||||
self._log("Error {0}".format(e))
|
||||
finally:
|
||||
self._log("stopping")
|
||||
self._log_file.flush()
|
||||
if os.path.exists(self.fifo_path):
|
||||
os.unlink(self.fifo_path)
|
||||
|
||||
def stop(self):
|
||||
self._log("Monitor stop")
|
||||
self._stop_requested = True
|
||||
try:
|
||||
if os.path.exists(self.fifo_path):
|
||||
fd = os.open(self.fifo_path, os.O_WRONLY)
|
||||
os.write(fd, b'X')
|
||||
os.close(fd)
|
||||
except Exception as e:
|
||||
self._log("err while closing: {0}".format(str(e)))
|
||||
if self._thread:
|
||||
self._thread.join()
|
||||
self._thread = None
|
||||
|
||||
def reset(self):
|
||||
self._log("reset")
|
||||
self.was_pinged = self.was_closed = self._was_triggered = False
|
||||
|
||||
@property
|
||||
def was_triggered(self):
|
||||
delta = time.time() - self.last_ping
|
||||
triggered = self._was_triggered or not self.was_closed and delta > self.timeout
|
||||
self._log("triggered={0}, {1}s left".format(triggered, self.timeout - delta))
|
||||
return triggered
|
||||
|
||||
|
||||
# actions to execute on start/stop of the tests and before running invidual features
|
||||
def before_all(context):
|
||||
context.ci = 'TRAVIS_BUILD_NUMBER' in os.environ or 'BUILD_NUMBER' in os.environ
|
||||
|
||||
@@ -34,9 +34,9 @@ Scenario: check local configuration reload
|
||||
Then I receive a response code 202
|
||||
|
||||
Scenario: check dynamic configuration change via DCS
|
||||
Given I run patronictl.py edit-config -s 'ttl=10' -s 'loop_wait=2' -p 'max_connections=101' --force batman
|
||||
Then I receive a response returncode 0
|
||||
And I receive a response output "+loop_wait: 2"
|
||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 10, "loop_wait": 2, "postgresql": {"parameters": {"max_connections": 101}}}
|
||||
Then I receive a response code 200
|
||||
And I receive a response loop_wait 2
|
||||
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 11 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8008/config
|
||||
Then I receive a response code 200
|
||||
@@ -65,8 +65,8 @@ Scenario: check API requests for the primary-replica pair in the pause mode
|
||||
Then postgres1 role is the secondary after 15 seconds
|
||||
|
||||
Scenario: check the failover via the API in the pause mode
|
||||
Given I issue a POST request to http://127.0.0.1:8008/failover with {"leader": "postgres0", "candidate": "postgres1"}
|
||||
Then I receive a response code 200
|
||||
Given I run patronictl.py failover batman --master postgres0 --candidate postgres1 --force
|
||||
Then I receive a response returncode 0
|
||||
And postgres1 is a leader after 5 seconds
|
||||
And postgres1 role is the primary after 10 seconds
|
||||
And postgres0 role is the secondary after 10 seconds
|
||||
|
||||
@@ -11,7 +11,7 @@ def start_patroni(context, name):
|
||||
|
||||
@step('I shut down {name:w}')
|
||||
def stop_patroni(context, name):
|
||||
return context.pctl.stop(name, timeout=60)
|
||||
return context.pctl.stop(name)
|
||||
|
||||
|
||||
@step('I kill {name:w}')
|
||||
|
||||
@@ -6,7 +6,7 @@ from behave import step, then
|
||||
|
||||
@step('I configure and start {name:w} with a tag {tag_name:w} {tag_value:w}')
|
||||
def start_patroni_with_a_name_value_tag(context, name, tag_name, tag_value):
|
||||
return context.pctl.start(name, custom_config={'tags': {tag_name: tag_value}})
|
||||
return context.pctl.start(name, tags={tag_name: tag_value})
|
||||
|
||||
|
||||
@then('There is a label with "{content:w}" in {name:w} data directory')
|
||||
|
||||
@@ -1,27 +0,0 @@
|
||||
import time
|
||||
|
||||
from behave import step, then
|
||||
|
||||
|
||||
@step('I start {name:w} in a cluster {cluster_name:w} as a clone of {name2:w}')
|
||||
def start_cluster_clone(context, name, cluster_name, name2):
|
||||
context.pctl.clone(name2, cluster_name, name)
|
||||
|
||||
|
||||
@step('I start {name:w} in a cluster {cluster_name:w} from backup')
|
||||
def start_cluster_from_backup(context, name, cluster_name):
|
||||
context.pctl.bootstrap_from_backup(name, cluster_name)
|
||||
|
||||
|
||||
@then('{name:w} is a leader of {cluster_name:w} after {time_limit:d} seconds')
|
||||
def is_a_leader(context, name, cluster_name, time_limit):
|
||||
time_limit *= context.timeout_multiplier
|
||||
max_time = time.time() + int(time_limit)
|
||||
while (context.dcs_ctl.query("leader", scope=cluster_name) != name):
|
||||
time.sleep(1)
|
||||
assert time.time() < max_time, "{0} is not a leader in dcs after {1} seconds".format(name, time_limit)
|
||||
|
||||
|
||||
@step('I do a backup of {name:w}')
|
||||
def do_backup(context, name):
|
||||
context.pctl.backup(name)
|
||||
@@ -111,8 +111,7 @@ def check_response(context, component, data):
|
||||
assert context.status_code == int(data),\
|
||||
"status code {0} != {1}, response: {2}".format(context.status_code, data, context.response)
|
||||
elif component == 'returncode':
|
||||
assert context.status_code == int(data), "return code {0} != {1}, {2}".format(context.status_code,
|
||||
data, context.response)
|
||||
assert context.status_code == int(data), "return code {0} != {1}".format(context.status_code, data)
|
||||
elif component == 'text':
|
||||
assert context.response == data.strip('"'), "response {0} does not contain {1}".format(context.response, data)
|
||||
elif component == 'output':
|
||||
|
||||
@@ -1,74 +0,0 @@
|
||||
from behave import step, then
|
||||
import time
|
||||
|
||||
|
||||
def polling_loop(timeout, interval=1):
|
||||
"""Returns an iterator that returns values until timeout has passed. Timeout is measured from start of iteration."""
|
||||
start_time = time.time()
|
||||
iteration = 0
|
||||
end_time = start_time + timeout
|
||||
while time.time() < end_time:
|
||||
yield iteration
|
||||
iteration += 1
|
||||
time.sleep(interval)
|
||||
|
||||
|
||||
@step('I start {name:w} with watchdog')
|
||||
def start_patroni_with_watchdog(context, name):
|
||||
return context.pctl.start(name, custom_config={'watchdog': True})
|
||||
|
||||
|
||||
@step('{name:w} watchdog has been pinged after {timeout:d} seconds')
|
||||
def watchdog_was_pinged(context, name, timeout):
|
||||
for _ in polling_loop(timeout):
|
||||
if context.pctl.get_watchdog(name).was_pinged:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
@then('{name:w} watchdog has been closed')
|
||||
def watchdog_was_closed(context, name):
|
||||
assert context.pctl.get_watchdog(name).was_closed
|
||||
|
||||
|
||||
@step('I reset {name:w} watchdog state')
|
||||
def watchdog_reset_pinged(context, name):
|
||||
context.pctl.get_watchdog(name).reset()
|
||||
|
||||
|
||||
@then('{name:w} watchdog is triggered after {timeout:d} seconds')
|
||||
def watchdog_was_triggered(context, name, timeout):
|
||||
for _ in polling_loop(timeout):
|
||||
if context.pctl.get_watchdog(name).was_triggered:
|
||||
return True
|
||||
assert False
|
||||
|
||||
|
||||
@then('{name:w} watchdog was not triggered')
|
||||
def watchdog_was_not_triggered(context, name):
|
||||
assert not context.pctl.get_watchdog(name).was_triggered
|
||||
|
||||
|
||||
@step('{name:w} checkpoint takes {timeout:d} seconds')
|
||||
def checkpoint_hang(context, name, timeout):
|
||||
assert context.pctl.checkpoint_hang(name, timeout)
|
||||
|
||||
|
||||
@step('{name:w} hangs for {timeout:d} seconds')
|
||||
def patroni_hang(context, name, timeout):
|
||||
return context.pctl.patroni_hang(name, timeout)
|
||||
|
||||
|
||||
@step('I terminate {name:w} user processes')
|
||||
def terminate_backends(context, name):
|
||||
return context.pctl.terminate_backends(name)
|
||||
|
||||
|
||||
@step('Sleep for {timeout:d} seconds')
|
||||
def dcs_connection_lost(context, timeout):
|
||||
time.sleep(timeout)
|
||||
|
||||
|
||||
@then('{name:w} database is running')
|
||||
def database_is_running(context, name):
|
||||
assert context.pctl.database_is_running(name)
|
||||
@@ -1,31 +0,0 @@
|
||||
Feature: watchdog
|
||||
Verify that watchdog gets pinged and triggered under appropriate circumstances.
|
||||
|
||||
Scenario: watchdog is opened and pinged
|
||||
Given I start postgres0 with watchdog
|
||||
Then postgres0 is a leader after 10 seconds
|
||||
And postgres0 role is the primary after 10 seconds
|
||||
And postgres0 watchdog has been pinged after 10 seconds
|
||||
|
||||
Scenario: watchdog is disabled during pause
|
||||
Given I run patronictl.py pause batman
|
||||
Then I receive a response returncode 0
|
||||
When I sleep for 2 seconds
|
||||
Then postgres0 watchdog has been closed
|
||||
|
||||
Scenario: watchdog is opened and pinged after resume
|
||||
Given I reset postgres0 watchdog state
|
||||
And I run patronictl.py resume batman
|
||||
Then I receive a response returncode 0
|
||||
And postgres0 watchdog has been pinged after 10 seconds
|
||||
|
||||
Scenario: watchdog is disabled when shutting down
|
||||
Given I shut down postgres0
|
||||
Then postgres0 watchdog has been closed
|
||||
|
||||
Scenario: watchdog is triggered if patroni stops responding
|
||||
Given I reset postgres0 watchdog state
|
||||
And I start postgres0 with watchdog
|
||||
Then postgres0 role is the primary after 10 seconds
|
||||
When postgres0 hangs for 30 seconds
|
||||
Then postgres0 watchdog is triggered after 30 seconds
|
||||
+14
-18
@@ -1,25 +1,21 @@
|
||||
global
|
||||
maxconn 100
|
||||
maxconn 100
|
||||
|
||||
defaults
|
||||
log global
|
||||
mode tcp
|
||||
retries 2
|
||||
timeout client 30m
|
||||
timeout connect 4s
|
||||
timeout server 30m
|
||||
timeout check 5s
|
||||
log global
|
||||
mode tcp
|
||||
retries 2
|
||||
timeout client 30m
|
||||
timeout connect 4s
|
||||
timeout server 30m
|
||||
timeout check 5s
|
||||
|
||||
listen stats
|
||||
mode http
|
||||
bind *:7000
|
||||
stats enable
|
||||
stats uri /
|
||||
frontend ft_postgresql
|
||||
bind *:5000
|
||||
default_backend bk_db
|
||||
|
||||
backend bk_db
|
||||
option httpchk
|
||||
|
||||
listen batman
|
||||
bind *:5000
|
||||
option httpchk
|
||||
http-check expect status 200
|
||||
default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions
|
||||
server postgresql_127.0.0.1_5432 127.0.0.1:5432 maxconn 100 check port 8008
|
||||
server postgresql_127.0.0.1_5433 127.0.0.1:5433 maxconn 100 check port 8009
|
||||
|
||||
+8
-12
@@ -16,14 +16,12 @@ class Patroni(object):
|
||||
from patroni.ha import Ha
|
||||
from patroni.postgresql import Postgresql
|
||||
from patroni.version import __version__
|
||||
from patroni.watchdog import Watchdog
|
||||
|
||||
self.setup_signal_handlers()
|
||||
|
||||
self.version = __version__
|
||||
self.config = Config()
|
||||
self.dcs = get_dcs(self.config)
|
||||
self.watchdog = Watchdog(self.config)
|
||||
self.load_dynamic_configuration()
|
||||
|
||||
self.postgresql = Postgresql(self.config['postgresql'])
|
||||
@@ -42,7 +40,6 @@ class Patroni(object):
|
||||
if cluster and cluster.config:
|
||||
if self.config.set_dynamic_configuration(cluster.config):
|
||||
self.dcs.reload_config(self.config)
|
||||
self.watchdog.reload_config(self.config)
|
||||
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
||||
if self.config.set_dynamic_configuration(self.config['bootstrap']['dcs']):
|
||||
self.dcs.reload_config(self.config)
|
||||
@@ -66,7 +63,6 @@ class Patroni(object):
|
||||
try:
|
||||
self.tags = self.get_tags()
|
||||
self.dcs.reload_config(self.config)
|
||||
self.watchdog.reload_config(self.config)
|
||||
self.api.reload_config(self.config['restapi'])
|
||||
self.postgresql.reload_config(self.config['postgresql'])
|
||||
except Exception:
|
||||
@@ -117,7 +113,7 @@ class Patroni(object):
|
||||
if cluster and cluster.config and self.config.set_dynamic_configuration(cluster.config):
|
||||
self.reload_config()
|
||||
|
||||
if self.postgresql.role != 'uninitialized':
|
||||
if not self.postgresql.data_directory_empty():
|
||||
self.config.save_cache()
|
||||
|
||||
self.schedule_next_run()
|
||||
@@ -128,14 +124,9 @@ class Patroni(object):
|
||||
signal.signal(signal.SIGHUP, self.sighup_handler)
|
||||
signal.signal(signal.SIGTERM, self.sigterm_handler)
|
||||
|
||||
def shutdown(self):
|
||||
self.api.shutdown()
|
||||
self.ha.shutdown()
|
||||
|
||||
|
||||
def patroni_main():
|
||||
logformat = os.environ.get('PATRONI_LOGFORMAT', '%(asctime)s %(levelname)s: %(message)s')
|
||||
logging.basicConfig(format=logformat, level=logging.INFO)
|
||||
logging.basicConfig(format='%(asctime)s %(levelname)s: %(message)s', level=logging.INFO)
|
||||
logging.getLogger('requests').setLevel(logging.WARNING)
|
||||
|
||||
patroni = Patroni()
|
||||
@@ -144,7 +135,12 @@ def patroni_main():
|
||||
except KeyboardInterrupt:
|
||||
pass
|
||||
finally:
|
||||
patroni.shutdown()
|
||||
patroni.api.shutdown()
|
||||
if patroni.ha.is_paused():
|
||||
logger.info('Leader key is not deleted and Postgresql is not stopped due paused state')
|
||||
else:
|
||||
patroni.ha.while_not_sync_standby(lambda: patroni.postgresql.stop(checkpoint=False))
|
||||
patroni.dcs.delete_leader()
|
||||
|
||||
|
||||
def pg_ctl_start(args):
|
||||
|
||||
+8
-20
@@ -68,8 +68,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
response['scheduled_restart'] = patroni.scheduled_restart.copy()
|
||||
del response['scheduled_restart']['postmaster_start_time']
|
||||
response['scheduled_restart']['schedule'] = (response['scheduled_restart']['schedule']).isoformat()
|
||||
if not patroni.ha.watchdog.is_healthy:
|
||||
response['watchdog_failed'] = True
|
||||
self._write_json_response(status_code, response)
|
||||
|
||||
def do_GET(self, write_status_code_only=False):
|
||||
@@ -293,21 +291,15 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
if leader and (not cluster.leader or cluster.leader.name != leader):
|
||||
return 'leader name does not match'
|
||||
if candidate:
|
||||
if cluster.is_synchronous_mode() and cluster.sync.sync_standby != candidate:
|
||||
return 'candidate name does not match with sync_standby'
|
||||
members = [m for m in cluster.members if m.name == candidate]
|
||||
if not members:
|
||||
return 'candidate does not exists'
|
||||
elif cluster.is_synchronous_mode():
|
||||
members = [m for m in cluster.members if m.name == cluster.sync.sync_standby]
|
||||
if not members:
|
||||
return 'failover is not possible: can not find sync_standby'
|
||||
else:
|
||||
members = [m for m in cluster.members if m.name != cluster.leader.name and m.api_url]
|
||||
if not members:
|
||||
return 'failover is not possible: cluster does not have members except leader'
|
||||
for st in self.server.patroni.ha.fetch_nodes_statuses(members):
|
||||
if st.failover_limitation() is None:
|
||||
for _, reachable, _, _, tags in self.server.patroni.ha.fetch_nodes_statuses(members):
|
||||
if reachable and not tags.get('nofailover', False):
|
||||
return None
|
||||
return 'failover is not possible: no good candidates have been found'
|
||||
|
||||
@@ -382,8 +374,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
|
||||
def get_postgresql_status(self, retry=False):
|
||||
try:
|
||||
if self.server.patroni.postgresql.state not in ('running', 'restarting', 'starting'):
|
||||
raise RetryFailedError('')
|
||||
row = self.query("""WITH replication_info AS (
|
||||
SELECT usename, application_name, client_addr, state, sync_state, sync_priority
|
||||
FROM pg_stat_replication
|
||||
@@ -392,16 +382,14 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
pg_is_in_recovery(),
|
||||
CASE WHEN pg_is_in_recovery()
|
||||
THEN 0
|
||||
ELSE pg_{0}_{1}_diff(pg_current_{0}_{1}(), '0/0')::bigint
|
||||
ELSE pg_xlog_location_diff(pg_current_xlog_location(), '0/0')::bigint
|
||||
END,
|
||||
pg_{0}_{1}_diff(COALESCE(pg_last_{0}_receive_{1}(),
|
||||
pg_last_{0}_replay_{1}()), '0/0')::bigint,
|
||||
pg_{0}_{1}_diff(pg_last_{0}_replay_{1}(), '0/0')::bigint,
|
||||
pg_xlog_location_diff(COALESCE(pg_last_xlog_receive_location(),
|
||||
pg_last_xlog_replay_location()), '0/0')::bigint,
|
||||
pg_xlog_location_diff(pg_last_xlog_replay_location(), '0/0')::bigint,
|
||||
to_char(pg_last_xact_replay_timestamp(), 'YYYY-MM-DD HH24:MI:SS.MS TZ'),
|
||||
pg_is_in_recovery() AND pg_is_{0}_replay_paused(),
|
||||
(SELECT array_to_json(array_agg(row_to_json(ri)))
|
||||
FROM replication_info ri)""".format(self.server.patroni.postgresql.wal_name,
|
||||
self.server.patroni.postgresql.lsn_name),
|
||||
pg_is_in_recovery() AND pg_is_xlog_replay_paused(),
|
||||
(SELECT array_to_json(array_agg(row_to_json(ri))) FROM replication_info ri)""",
|
||||
retry=retry)[0]
|
||||
|
||||
result = {
|
||||
|
||||
@@ -1,55 +1,9 @@
|
||||
import logging
|
||||
from threading import Lock, RLock, Thread
|
||||
from threading import RLock, Thread
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class CriticalTask(object):
|
||||
"""Represents a critical task in a background process that we either need to cancel or get the result of.
|
||||
|
||||
Fields of this object may be accessed only when holding a lock on it. To perform the critical task the background
|
||||
thread must, while holding lock on this object, check `is_cancelled` flag, run the task and mark the task as
|
||||
complete using `complete()`.
|
||||
|
||||
The main thread must hold async lock to prevent the task from completing, hold lock on critical task object,
|
||||
call cancel. If the task has completed `cancel()` will return False and `result` field will contain the result of
|
||||
the task. When cancel returns True it is guaranteed that the background task will notice the `is_cancelled` flag.
|
||||
"""
|
||||
def __init__(self):
|
||||
self._lock = Lock()
|
||||
self.is_cancelled = False
|
||||
self.result = None
|
||||
|
||||
def reset(self):
|
||||
"""Must be called every time the background task is finished.
|
||||
|
||||
Must be called from async thread. Caller must hold lock on async executor when calling."""
|
||||
self.is_cancelled = False
|
||||
self.result = None
|
||||
|
||||
def cancel(self):
|
||||
"""Tries to cancel the task, returns True if the task has already run.
|
||||
|
||||
Caller must hold lock on async executor and the task when calling."""
|
||||
if self.result is not None:
|
||||
return False
|
||||
self.is_cancelled = True
|
||||
return True
|
||||
|
||||
def complete(self, result):
|
||||
"""Mark task as completed along with a result.
|
||||
|
||||
Must be called from async thread. Caller must hold lock on task when calling."""
|
||||
self.result = result
|
||||
|
||||
def __enter__(self):
|
||||
self._lock.acquire()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_val, exc_tb):
|
||||
self._lock.release()
|
||||
|
||||
|
||||
class AsyncExecutor(object):
|
||||
|
||||
def __init__(self, ha_wakeup):
|
||||
@@ -57,7 +11,6 @@ class AsyncExecutor(object):
|
||||
self._thread_lock = RLock()
|
||||
self._scheduled_action = None
|
||||
self._scheduled_action_lock = RLock()
|
||||
self.critical_task = CriticalTask()
|
||||
|
||||
@property
|
||||
def busy(self):
|
||||
@@ -85,13 +38,11 @@ class AsyncExecutor(object):
|
||||
# if the func returned something (not None) - wake up main HA loop
|
||||
wakeup = func(*args) if args else func()
|
||||
return wakeup
|
||||
except Exception:
|
||||
except:
|
||||
logger.exception('Exception during execution of long running task %s', self.scheduled_action)
|
||||
finally:
|
||||
with self:
|
||||
self.reset_scheduled_action()
|
||||
with self.critical_task:
|
||||
self.critical_task.reset()
|
||||
if wakeup is not None:
|
||||
self._ha_wakeup()
|
||||
|
||||
|
||||
+5
-9
@@ -43,14 +43,10 @@ class Config(object):
|
||||
'maximum_lag_on_failover': 1048576,
|
||||
'master_start_timeout': 300,
|
||||
'synchronous_mode': False,
|
||||
'synchronous_mode_strict': False,
|
||||
'postgresql': {
|
||||
'bin_dir': '',
|
||||
'use_slots': True,
|
||||
'parameters': {p: v[0] for p, v in Postgresql.CMDLINE_OPTIONS.items()}
|
||||
},
|
||||
'watchdog': {
|
||||
'mode': 'automatic',
|
||||
}
|
||||
}
|
||||
|
||||
@@ -178,7 +174,7 @@ class Config(object):
|
||||
elif name not in ('connect_address', 'listen', 'data_dir', 'pgpass', 'authentication'):
|
||||
config['postgresql'][name] = deepcopy(value)
|
||||
elif name in config: # only variables present in __DEFAULT_CONFIG allowed to be overriden from DCS
|
||||
if name in ('synchronous_mode', 'synchronous_mode_strict'):
|
||||
if name == 'synchronous_mode':
|
||||
config[name] = value
|
||||
else:
|
||||
config[name] = int(value)
|
||||
@@ -242,12 +238,12 @@ class Config(object):
|
||||
name, suffix = (param[8:].rsplit('_', 1) + [''])[:2]
|
||||
if name and suffix:
|
||||
# PATRONI_(ETCD|CONSUL|ZOOKEEPER|EXHIBITOR|...)_(HOSTS?|PORT|..)
|
||||
if suffix in ('HOST', 'HOSTS', 'PORT', 'SRV', 'URL', 'PROXY', 'CACERT', 'CERT', 'KEY',
|
||||
'VERIFY', 'TOKEN', 'CHECKS', 'DC') and '_' not in name:
|
||||
if suffix in ('HOST', 'HOSTS', 'PORT', 'SRV', 'URL', 'PROXY', 'CACERT', 'CERT', 'KEY') \
|
||||
and '_' not in name:
|
||||
value = os.environ.pop(param)
|
||||
if suffix == 'PORT':
|
||||
value = value and parse_int(value)
|
||||
elif suffix in ('HOSTS', 'CHECKS'):
|
||||
elif suffix == 'HOSTS':
|
||||
value = value and _parse_list(value)
|
||||
if value:
|
||||
ret[name.lower()][suffix.lower()] = value
|
||||
@@ -275,7 +271,7 @@ class Config(object):
|
||||
config['postgresql'][name].update(self._process_postgresql_parameters(value, True))
|
||||
elif name != 'use_slots': # replication slots must be enabled/disabled globally
|
||||
config['postgresql'][name] = deepcopy(value)
|
||||
elif name not in config or name in ['watchdog']:
|
||||
elif name not in config:
|
||||
config[name] = deepcopy(value) if value else {}
|
||||
|
||||
# restapi server expects to get restapi.auth = 'username:password'
|
||||
|
||||
+1
-211
@@ -4,36 +4,27 @@ Patroni Control
|
||||
|
||||
import base64
|
||||
import click
|
||||
import codecs
|
||||
import datetime
|
||||
import dateutil.parser
|
||||
import cdiff
|
||||
import copy
|
||||
import difflib
|
||||
import io
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import psycopg2
|
||||
import random
|
||||
import requests
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
import tzlocal
|
||||
import yaml
|
||||
|
||||
from click import ClickException
|
||||
from contextlib import contextmanager
|
||||
from patroni.config import Config
|
||||
from patroni.dcs import get_dcs as _get_dcs
|
||||
from patroni.exceptions import PatroniException
|
||||
from patroni.postgresql import Postgresql
|
||||
from patroni.utils import is_valid_pg_version, patch_config
|
||||
from patroni.utils import is_valid_pg_version
|
||||
from prettytable import PrettyTable
|
||||
from six.moves.urllib_parse import urlparse
|
||||
from six import text_type
|
||||
|
||||
CONFIG_DIR_PATH = click.get_app_dir('patroni')
|
||||
CONFIG_FILE_PATH = os.path.join(CONFIG_DIR_PATH, 'patronictl.yaml')
|
||||
@@ -828,204 +819,3 @@ def pause(obj, cluster_name):
|
||||
@click.pass_obj
|
||||
def resume(obj, cluster_name):
|
||||
return toggle_pause(obj, cluster_name, False)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def temporary_file(contents, suffix='', prefix='tmp'):
|
||||
"""Creates a temporary file with specified contents that persists for the context.
|
||||
|
||||
:param contents: binary string that will be written to the file.
|
||||
:param prefix: will be prefixed to the filename.
|
||||
:param suffix: will be appended to the filename.
|
||||
:returns path of the created file.
|
||||
"""
|
||||
tmp = tempfile.NamedTemporaryFile(suffix=suffix, prefix=prefix, delete=False)
|
||||
with tmp:
|
||||
tmp.write(contents)
|
||||
|
||||
try:
|
||||
yield tmp.name
|
||||
finally:
|
||||
os.unlink(tmp.name)
|
||||
|
||||
|
||||
def show_diff(before_editing, after_editing):
|
||||
"""Shows a diff between two strings.
|
||||
|
||||
If the output is to a tty the diff will be colored. Inputs are expected to be unicode strings.
|
||||
"""
|
||||
def listify(string):
|
||||
return [l+'\n' for l in string.rstrip('\n').split('\n')]
|
||||
|
||||
unified_diff = difflib.unified_diff(listify(before_editing), listify(after_editing))
|
||||
|
||||
if sys.stdout.isatty():
|
||||
buf = io.StringIO()
|
||||
for line in unified_diff:
|
||||
# Force cast to unicode as difflib on Python 2.7 returns a mix of unicode and str.
|
||||
buf.write(text_type(line))
|
||||
buf.seek(0)
|
||||
|
||||
class opts:
|
||||
side_by_side = False
|
||||
width = 80
|
||||
tab_width = 8
|
||||
cdiff.markup_to_pager(cdiff.PatchStream(buf), opts)
|
||||
else:
|
||||
for line in unified_diff:
|
||||
click.echo(line.rstrip('\n'))
|
||||
|
||||
|
||||
def format_config_for_editing(data):
|
||||
"""Formats configuration as YAML for human consumption.
|
||||
|
||||
:param data: configuration as nested dictionaries
|
||||
:returns unicode YAML of the configuration"""
|
||||
return yaml.safe_dump(data, default_flow_style=False, encoding=None, allow_unicode=True)
|
||||
|
||||
|
||||
def apply_config_changes(before_editing, data, kvpairs):
|
||||
"""Applies config changes specified as a list of key-value pairs.
|
||||
|
||||
Keys are interpreted as dotted paths into the configuration data structure. Except for paths beginning with
|
||||
`postgresql.parameters` where rest of the path is used directly to allow for PostgreSQL GUCs containing dots.
|
||||
Values are interpreted as YAML values.
|
||||
|
||||
:param before_editing: human representation before editing
|
||||
:param data: configuration datastructure
|
||||
:param kvpairs: list of strings containing key value pairs separated by =
|
||||
:returns tuple of human readable and parsed datastructure after changes
|
||||
"""
|
||||
changed_data = copy.deepcopy(data)
|
||||
|
||||
def set_path_value(config, path, value, prefix=()):
|
||||
# Postgresql GUCs can't be nested, but can contain dots so we re-flatten the structure for this case
|
||||
if prefix == ('postgresql', 'parameters'):
|
||||
path = ['.'.join(path)]
|
||||
|
||||
key = path[0]
|
||||
if len(path) == 1:
|
||||
if value is None:
|
||||
config.pop(key, None)
|
||||
else:
|
||||
config[key] = value
|
||||
else:
|
||||
if not isinstance(config.get(key), dict):
|
||||
config[key] = {}
|
||||
set_path_value(config[key], path[1:], value, prefix + (key,))
|
||||
if config[key] == {}:
|
||||
del config[key]
|
||||
|
||||
for pair in kvpairs:
|
||||
if not pair or "=" not in pair:
|
||||
raise PatroniCtlException("Invalid parameter setting {0}".format(pair))
|
||||
key_path, value = pair.split("=", 1)
|
||||
set_path_value(changed_data, key_path.strip().split("."), yaml.safe_load(value))
|
||||
|
||||
return format_config_for_editing(changed_data), changed_data
|
||||
|
||||
|
||||
def apply_yaml_file(data, filename):
|
||||
"""Applies changes from a YAML file to configuration
|
||||
|
||||
:param data: configuration datastructure
|
||||
:param filename: name of the YAML file, - is taken to mean standard input
|
||||
:returns tuple of human readable and parsed datastructure after changes
|
||||
"""
|
||||
changed_data = copy.deepcopy(data)
|
||||
|
||||
if filename == '-':
|
||||
new_options = yaml.safe_load(sys.stdin)
|
||||
else:
|
||||
with open(filename) as fd:
|
||||
new_options = yaml.safe_load(fd)
|
||||
|
||||
patch_config(changed_data, new_options)
|
||||
|
||||
return format_config_for_editing(changed_data), changed_data
|
||||
|
||||
|
||||
def invoke_editor(before_editing, cluster_name):
|
||||
"""Starts editor command to edit configuration in human readable format
|
||||
|
||||
:param before_editing: human representation before editing
|
||||
:returns tuple of human readable and parsed datastructure after changes
|
||||
"""
|
||||
editor_cmd = os.environ.get('EDITOR')
|
||||
if not editor_cmd:
|
||||
raise PatroniCtlException('EDITOR environment variable is not set')
|
||||
|
||||
with temporary_file(contents=before_editing.encode('utf-8'),
|
||||
suffix='.yaml',
|
||||
prefix='{0}-config-'.format(cluster_name)) as tmpfile:
|
||||
ret = subprocess.call([editor_cmd, tmpfile])
|
||||
if ret:
|
||||
raise PatroniCtlException("Editor exited with return code {0}".format(ret))
|
||||
|
||||
with codecs.open(tmpfile, encoding='utf-8') as fd:
|
||||
after_editing = fd.read()
|
||||
|
||||
return after_editing, yaml.safe_load(after_editing)
|
||||
|
||||
|
||||
@ctl.command('edit-config', help="Edit cluster configuration")
|
||||
@click.argument('cluster_name')
|
||||
@click.option('--quiet', '-q', is_flag=True, help='Do not show changes')
|
||||
@click.option('--set', '-s', 'kvpairs', multiple=True,
|
||||
help='Set specific configuration value. Can be specified multiple times')
|
||||
@click.option('--pg', '-p', 'pgkvpairs', multiple=True,
|
||||
help='Set specific PostgreSQL parameter value. Shorthand for -s postgresql.parameters. '
|
||||
'Can be specified multiple times')
|
||||
@click.option('--apply', 'apply_filename', help='Apply configuration from file. Use - for stdin.')
|
||||
@click.option('--replace', 'replace_filename', help='Apply configuration from file, replacing existing configuration.'
|
||||
' Use - for stdin.')
|
||||
@option_force
|
||||
@click.pass_obj
|
||||
def edit_config(obj, cluster_name, force, quiet, kvpairs, pgkvpairs, apply_filename, replace_filename):
|
||||
dcs = get_dcs(obj, cluster_name)
|
||||
cluster = dcs.get_cluster()
|
||||
|
||||
before_editing = format_config_for_editing(cluster.config.data)
|
||||
|
||||
after_editing = None # Serves as a flag if any changes were requested
|
||||
changed_data = cluster.config.data
|
||||
|
||||
if replace_filename:
|
||||
after_editing, changed_data = apply_yaml_file({}, replace_filename)
|
||||
|
||||
if apply_filename:
|
||||
after_editing, changed_data = apply_yaml_file(changed_data, apply_filename)
|
||||
|
||||
if kvpairs or pgkvpairs:
|
||||
all_pairs = list(kvpairs) + ['postgresql.parameters.'+v.lstrip() for v in pgkvpairs]
|
||||
after_editing, changed_data = apply_config_changes(before_editing, changed_data, all_pairs)
|
||||
|
||||
# If no changes were specified on the command line invoke editor
|
||||
if after_editing is None:
|
||||
after_editing, changed_data = invoke_editor(before_editing, cluster_name)
|
||||
|
||||
if cluster.config.data == changed_data:
|
||||
if not quiet:
|
||||
click.echo("Not changed")
|
||||
return
|
||||
|
||||
if not quiet:
|
||||
show_diff(before_editing, after_editing)
|
||||
|
||||
if (apply_filename == '-' or replace_filename == '-') and not force:
|
||||
click.echo("Use --force option to apply changes")
|
||||
return
|
||||
|
||||
if force or click.confirm('Apply these changes?'):
|
||||
if not dcs.set_config_value(json.dumps(changed_data), cluster.config.index):
|
||||
raise PatroniCtlException("Config modification aborted due to concurrent changes")
|
||||
click.echo("Configuration changed")
|
||||
|
||||
|
||||
@ctl.command('show-config', help="Show cluster configuration")
|
||||
@click.argument('cluster_name')
|
||||
@click.pass_obj
|
||||
def show_config(obj, cluster_name):
|
||||
cluster = get_dcs(obj, cluster_name).get_cluster()
|
||||
|
||||
click.echo(format_config_for_editing(cluster.config.data))
|
||||
|
||||
+13
-26
@@ -3,7 +3,6 @@ import dateutil
|
||||
import importlib
|
||||
import inspect
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import pkgutil
|
||||
import six
|
||||
@@ -15,8 +14,6 @@ from random import randint
|
||||
from six.moves.urllib_parse import urlparse, urlunparse, parse_qsl
|
||||
from threading import Event, Lock
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def parse_connection_string(value):
|
||||
"""Original Governor stores connection strings for each cluster members if a following format:
|
||||
@@ -54,22 +51,18 @@ def dcs_modules():
|
||||
def get_dcs(config):
|
||||
available_implementations = set()
|
||||
for module_name in dcs_modules():
|
||||
try:
|
||||
module = importlib.import_module(module_name)
|
||||
for name in filter(lambda name: not name.startswith('__'), dir(module)): # iterate through module content
|
||||
item = getattr(module, name)
|
||||
name = name.lower()
|
||||
# try to find implementation of AbstractDCS interface, class name must match with module_name
|
||||
if inspect.isclass(item) and issubclass(item, AbstractDCS) and __package__ + '.' + name == module_name:
|
||||
available_implementations.add(name)
|
||||
if name in config: # which has configuration section in the config file
|
||||
# propagate some parameters
|
||||
config[name].update({p: config[p] for p in ('namespace', 'name', 'scope', 'loop_wait',
|
||||
'patronictl', 'ttl', 'retry_timeout') if p in config})
|
||||
return item(config[name])
|
||||
except ImportError:
|
||||
if not config.get('patronictl'):
|
||||
logger.info('Failed to import %s', module_name)
|
||||
module = importlib.import_module(module_name)
|
||||
for name in filter(lambda name: not name.startswith('__'), dir(module)): # iterate through module content
|
||||
value = getattr(module, name)
|
||||
name = name.lower()
|
||||
# try to find implementation of AbstractDCS interface, class name must match with module_name
|
||||
if inspect.isclass(value) and issubclass(value, AbstractDCS) and __package__ + '.' + name == module_name:
|
||||
available_implementations.add(name)
|
||||
if name in config: # which has configuration section in the config file
|
||||
# propagate some parameters
|
||||
config[name].update({p: config[p] for p in ('namespace', 'name', 'scope', 'loop_wait',
|
||||
'patronictl', 'ttl', 'retry_timeout') if p in config})
|
||||
return value(config[name])
|
||||
raise PatroniException("""Can not find suitable configuration of distributed configuration store
|
||||
Available implementations: """ + ', '.join(available_implementations))
|
||||
|
||||
@@ -320,12 +313,6 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_leader_operat
|
||||
def is_paused(self):
|
||||
return self.config and self.config.data.get('pause', False) or False
|
||||
|
||||
def is_synchronous_mode(self):
|
||||
return bool(self.config and self.config.data.get('synchronous_mode'))
|
||||
|
||||
def is_synchronous_mode_strict(self):
|
||||
return bool(self.config and self.config.data.get('synchronous_mode_strict'))
|
||||
|
||||
|
||||
@six.add_metaclass(abc.ABCMeta)
|
||||
class AbstractDCS(object):
|
||||
@@ -424,7 +411,7 @@ class AbstractDCS(object):
|
||||
with self._cluster_thread_lock:
|
||||
try:
|
||||
self._load_cluster()
|
||||
except Exception:
|
||||
except:
|
||||
self._cluster = None
|
||||
raise
|
||||
return self._cluster
|
||||
|
||||
+24
-105
@@ -2,16 +2,15 @@ from __future__ import absolute_import
|
||||
import logging
|
||||
import os
|
||||
import socket
|
||||
import ssl
|
||||
import time
|
||||
import urllib3
|
||||
|
||||
from consul import ConsulException, NotFound, base
|
||||
from patroni.dcs import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState
|
||||
from patroni.exceptions import DCSError
|
||||
from patroni.utils import parse_bool, Retry, RetryFailedError
|
||||
from patroni.utils import Retry, RetryFailedError
|
||||
from urllib3.exceptions import HTTPError
|
||||
from six.moves.urllib.parse import urlencode, urlparse
|
||||
from six.moves.urllib.parse import urlencode
|
||||
from six.moves.http_client import HTTPException
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -25,39 +24,21 @@ class ConsulInternalError(ConsulException):
|
||||
"""An internal Consul server error occurred"""
|
||||
|
||||
|
||||
class InvalidSessionTTL(ConsulInternalError):
|
||||
"""Session TTL is too small or too big"""
|
||||
|
||||
|
||||
class HTTPClient(object):
|
||||
|
||||
def __init__(self, host='127.0.0.1', port=8500, token=None, scheme='http', verify=True, cert=None, ca_cert=None):
|
||||
self.token = token
|
||||
self._read_timeout = 10
|
||||
self.base_uri = '{0}://{1}:{2}'.format(scheme, host, port)
|
||||
kwargs = {}
|
||||
if cert:
|
||||
if isinstance(cert, tuple):
|
||||
# Key and cert are separate
|
||||
kwargs['cert_file'] = cert[0]
|
||||
kwargs['key_file'] = cert[1]
|
||||
else:
|
||||
# combined certificate
|
||||
kwargs['cert_file'] = cert
|
||||
if ca_cert:
|
||||
kwargs['ca_certs'] = ca_cert
|
||||
if verify or ca_cert:
|
||||
kwargs['cert_reqs'] = ssl.CERT_REQUIRED
|
||||
self.http = urllib3.PoolManager(num_pools=10, **kwargs)
|
||||
def __init__(self, host='127.0.0.1', port=8500, scheme='http', verify=True, timeout=10):
|
||||
self.host = host
|
||||
self.port = port
|
||||
self.scheme = scheme
|
||||
self.verify = verify
|
||||
self.set_read_timeout(timeout)
|
||||
self.base_uri = '{0}://{1}:{2}'.format(self.scheme, self.host, self.port)
|
||||
self.http = urllib3.PoolManager(num_pools=10)
|
||||
self._ttl = None
|
||||
|
||||
def set_read_timeout(self, timeout):
|
||||
self._read_timeout = timeout/3.0
|
||||
|
||||
@property
|
||||
def ttl(self):
|
||||
return self._ttl
|
||||
|
||||
def set_ttl(self, ttl):
|
||||
ret = self._ttl != ttl
|
||||
self._ttl = ttl
|
||||
@@ -67,11 +48,7 @@ class HTTPClient(object):
|
||||
def response(response):
|
||||
data = response.data.decode('utf-8')
|
||||
if response.status == 500:
|
||||
msg = '{0} {1}'.format(response.status, data)
|
||||
if data.startswith('Invalid Session TTL'):
|
||||
raise InvalidSessionTTL(msg)
|
||||
else:
|
||||
raise ConsulInternalError(msg)
|
||||
raise ConsulInternalError('{0} {1}'.format(response.status, data))
|
||||
return base.Response(response.status, response.headers, data)
|
||||
|
||||
def uri(self, path, params=None):
|
||||
@@ -95,30 +72,15 @@ class HTTPClient(object):
|
||||
kwargs['timeout'] = (float(params['wait'][:-1]) if 'wait' in params else 300) + 1
|
||||
else:
|
||||
kwargs['timeout'] = self._read_timeout
|
||||
token = params.pop('token', self.token) if isinstance(params, dict) else self.token
|
||||
if token:
|
||||
kwargs['headers'] = {'X-Consul-Token': token}
|
||||
return callback(self.response(self.http.request(method.upper(), self.uri(path, params), **kwargs)))
|
||||
return wrapper
|
||||
|
||||
|
||||
class ConsulClient(base.Consul):
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
self._cert = kwargs.pop('cert', None)
|
||||
self._ca_cert = kwargs.pop('ca_cert', None)
|
||||
self._token = kwargs.get('token')
|
||||
super(ConsulClient, self).__init__(*args, **kwargs)
|
||||
|
||||
def connect(self, *args, **kwargs):
|
||||
kwargs.update(dict(zip(['host', 'port', 'scheme', 'verify'], args)))
|
||||
if self._cert:
|
||||
kwargs['cert'] = self._cert
|
||||
if self._ca_cert:
|
||||
kwargs['ca_cert'] = self._ca_cert
|
||||
if self._token:
|
||||
kwargs['token'] = self._token
|
||||
return HTTPClient(**kwargs)
|
||||
@staticmethod
|
||||
def connect(host, port, scheme, verify=True):
|
||||
return HTTPClient(host, port, scheme, verify)
|
||||
|
||||
|
||||
def catch_consul_errors(func):
|
||||
@@ -142,36 +104,11 @@ class Consul(AbstractDCS):
|
||||
HTTPError, socket.error, socket.timeout))
|
||||
|
||||
self._my_member_data = None
|
||||
kwargs = {}
|
||||
if 'url' in config:
|
||||
r = urlparse(config['url'])
|
||||
config.update({'scheme': r.scheme, 'host': r.hostname, 'port': r.port or 8500})
|
||||
elif 'host' in config:
|
||||
host, port = (config.get('host', '127.0.0.1:8500') + ':8500').split(':')[:2]
|
||||
config['host'] = host
|
||||
if 'port' not in config:
|
||||
config['port'] = int(port)
|
||||
|
||||
if config.get('cacert'):
|
||||
config['ca_cert'] = config.pop('cacert')
|
||||
|
||||
if config.get('key') and config.get('cert'):
|
||||
config['cert'] = (config['cert'], config['key'])
|
||||
|
||||
config_keys = ('host', 'port', 'token', 'scheme', 'cert', 'ca_cert', 'dc')
|
||||
kwargs = {p: config.get(p) for p in config_keys if config.get(p)}
|
||||
|
||||
verify = config.get('verify')
|
||||
if not isinstance(verify, bool):
|
||||
verify = parse_bool(verify)
|
||||
if isinstance(verify, bool):
|
||||
kwargs['verify'] = verify
|
||||
|
||||
self._client = ConsulClient(**kwargs)
|
||||
host, port = config.get('host', '127.0.0.1:8500').split(':')
|
||||
self._client = ConsulClient(host=host, port=port)
|
||||
self.set_retry_timeout(config['retry_timeout'])
|
||||
self.set_ttl(config.get('ttl') or 30)
|
||||
self._last_session_refresh = 0
|
||||
self.__session_checks = config.get('checks')
|
||||
if not self._ctl:
|
||||
self.create_session()
|
||||
|
||||
@@ -195,15 +132,6 @@ class Consul(AbstractDCS):
|
||||
self._retry.deadline = retry_timeout
|
||||
self._client.http.set_read_timeout(retry_timeout)
|
||||
|
||||
def adjust_ttl(self):
|
||||
try:
|
||||
settings = self._client.agent.self()
|
||||
min_ttl = (settings['Config']['SessionTTLMin'] or 10000000000)/1000000000.0
|
||||
logger.warning('Changing Session TTL from %s to %s', self._client.http.ttl, min_ttl)
|
||||
self._client.http.set_ttl(min_ttl)
|
||||
except Exception:
|
||||
logger.exception('adjust_ttl')
|
||||
|
||||
def _do_refresh_session(self):
|
||||
""":returns: `!True` if it had to create new session"""
|
||||
if self._session and self._last_session_refresh + self._loop_wait > time.time():
|
||||
@@ -216,15 +144,8 @@ class Consul(AbstractDCS):
|
||||
self._session = None
|
||||
ret = not self._session
|
||||
if ret:
|
||||
try:
|
||||
self._session = self._client.session.create(name=self._scope + '-' + self._name,
|
||||
checks=self.__session_checks,
|
||||
lock_delay=0.001, behavior='delete')
|
||||
except InvalidSessionTTL:
|
||||
logger.exception('session.create')
|
||||
self.adjust_ttl()
|
||||
raise
|
||||
|
||||
self._session = self._client.session.create(name=self._scope + '-' + self._name,
|
||||
lock_delay=0.001, behavior='delete')
|
||||
self._last_session_refresh = time.time()
|
||||
return ret
|
||||
|
||||
@@ -295,14 +216,14 @@ class Consul(AbstractDCS):
|
||||
self._cluster = Cluster(initialize, config, leader, last_leader_operation, members, failover, sync)
|
||||
except NotFound:
|
||||
self._cluster = Cluster(None, None, None, None, [], None, None)
|
||||
except Exception:
|
||||
except:
|
||||
logger.exception('get_cluster')
|
||||
raise ConsulError('Consul is not responding properly')
|
||||
|
||||
def touch_member(self, data, ttl=None, permanent=False):
|
||||
def touch_member(self, data, **kwargs):
|
||||
cluster = self.cluster
|
||||
member = cluster and cluster.get_member(self._name, fallback_to_leader=False)
|
||||
create_member = not permanent and self.refresh_session()
|
||||
create_member = self.refresh_session()
|
||||
|
||||
if member and (create_member or member.session != self._session):
|
||||
try:
|
||||
@@ -315,7 +236,7 @@ class Consul(AbstractDCS):
|
||||
return True
|
||||
|
||||
try:
|
||||
args = {} if permanent else {'acquire': self._session}
|
||||
args = {} if kwargs.get('permanent', False) else {'acquire': self._session}
|
||||
self._client.kv.put(self.member_path, data, **args)
|
||||
self._my_member_data = data
|
||||
return True
|
||||
@@ -324,14 +245,12 @@ class Consul(AbstractDCS):
|
||||
return False
|
||||
|
||||
@catch_consul_errors
|
||||
def _do_attempt_to_acquire_leader(self, kwargs):
|
||||
return self.retry(self._client.kv.put, self.leader_path, self._name, **kwargs)
|
||||
|
||||
def attempt_to_acquire_leader(self, permanent=False):
|
||||
if not self._session and not permanent:
|
||||
self.refresh_session()
|
||||
|
||||
ret = self._do_attempt_to_acquire_leader({} if permanent else {'acquire': self._session})
|
||||
args = {} if permanent else {'acquire': self._session}
|
||||
ret = self.retry(self._client.kv.put, self.leader_path, self._name, **args)
|
||||
if not ret:
|
||||
logger.info('Could not take out TTL lock')
|
||||
return ret
|
||||
|
||||
+12
-15
@@ -1,7 +1,6 @@
|
||||
import logging
|
||||
import time
|
||||
|
||||
from kazoo.client import KazooClient, KazooState, KazooRetry
|
||||
from kazoo.client import KazooClient, KazooState
|
||||
from kazoo.exceptions import NoNodeError, NodeExistsError
|
||||
from kazoo.handlers.threading import SequentialThreadingHandler
|
||||
from patroni.dcs import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState
|
||||
@@ -52,9 +51,8 @@ class ZooKeeper(AbstractDCS):
|
||||
hosts = ','.join(hosts)
|
||||
|
||||
self._client = KazooClient(hosts, handler=PatroniSequentialThreadingHandler(config['retry_timeout']),
|
||||
timeout=config['ttl'], connection_retry=KazooRetry(max_delay=1, max_tries=-1,
|
||||
sleep_func=time.sleep), command_retry=KazooRetry(deadline=config['retry_timeout'],
|
||||
max_delay=1, max_tries=-1, sleep_func=time.sleep))
|
||||
timeout=config['ttl'], connection_retry={'max_delay': 1, 'max_tries': -1},
|
||||
command_retry={'deadline': config['retry_timeout'], 'max_delay': 1, 'max_tries': -1})
|
||||
self._client.add_listener(self.session_listener)
|
||||
|
||||
self._my_member_data = None
|
||||
@@ -116,8 +114,7 @@ class ZooKeeper(AbstractDCS):
|
||||
return True
|
||||
|
||||
def set_retry_timeout(self, retry_timeout):
|
||||
retry = self._client.retry if isinstance(self._client.retry, KazooRetry) else self._client._retry
|
||||
retry.deadline = retry_timeout
|
||||
self._client._retry.deadline = retry_timeout
|
||||
|
||||
def get_node(self, key, watch=None):
|
||||
try:
|
||||
@@ -206,7 +203,7 @@ class ZooKeeper(AbstractDCS):
|
||||
try:
|
||||
self._client.retry(self._client.create, path, value.encode('utf-8'), **kwargs)
|
||||
return True
|
||||
except Exception:
|
||||
except:
|
||||
return False
|
||||
|
||||
def attempt_to_acquire_leader(self, permanent=False):
|
||||
@@ -221,7 +218,7 @@ class ZooKeeper(AbstractDCS):
|
||||
return True
|
||||
except NoNodeError:
|
||||
return value == '' or (index is None and self._create(self.failover_path, value))
|
||||
except Exception:
|
||||
except:
|
||||
logging.exception('set_failover_value')
|
||||
return False
|
||||
|
||||
@@ -248,7 +245,7 @@ class ZooKeeper(AbstractDCS):
|
||||
self._client.delete_async(self.member_path).get(timeout=1)
|
||||
except NoNodeError:
|
||||
pass
|
||||
except Exception:
|
||||
except:
|
||||
return False
|
||||
member = None
|
||||
|
||||
@@ -268,7 +265,7 @@ class ZooKeeper(AbstractDCS):
|
||||
self._client.set_async(self.member_path, data).get(timeout=1)
|
||||
self._my_member_data = data
|
||||
return True
|
||||
except Exception:
|
||||
except:
|
||||
logger.exception('touch_member')
|
||||
|
||||
return False
|
||||
@@ -285,9 +282,9 @@ class ZooKeeper(AbstractDCS):
|
||||
try:
|
||||
self._client.create_async(self.leader_optime_path, last_operation, makepath=True).get(timeout=1)
|
||||
return True
|
||||
except Exception:
|
||||
except:
|
||||
logger.exception('Failed to create %s', self.leader_optime_path)
|
||||
except Exception:
|
||||
except:
|
||||
logger.exception('Failed to update %s', self.leader_optime_path)
|
||||
return False
|
||||
|
||||
@@ -307,7 +304,7 @@ class ZooKeeper(AbstractDCS):
|
||||
def cancel_initialization(self):
|
||||
try:
|
||||
self._client.retry(self._cancel_initialization)
|
||||
except Exception:
|
||||
except:
|
||||
logger.exception("Unable to delete initialize key")
|
||||
|
||||
def delete_cluster(self):
|
||||
@@ -322,7 +319,7 @@ class ZooKeeper(AbstractDCS):
|
||||
return True
|
||||
except NoNodeError:
|
||||
return value == '' or (index is None and self._create(self.sync_path, value))
|
||||
except Exception:
|
||||
except:
|
||||
logging.exception('set_sync_state_value')
|
||||
return False
|
||||
|
||||
|
||||
@@ -23,7 +23,3 @@ class DCSError(PatroniException):
|
||||
|
||||
class PostgresConnectionException(PostgresException):
|
||||
pass
|
||||
|
||||
|
||||
class WatchdogError(PatroniException):
|
||||
pass
|
||||
|
||||
+103
-269
@@ -9,8 +9,8 @@ import time
|
||||
|
||||
from collections import namedtuple
|
||||
from multiprocessing.pool import ThreadPool
|
||||
from patroni.async_executor import AsyncExecutor, CriticalTask
|
||||
from patroni.exceptions import DCSError, PostgresConnectionException, PatroniException
|
||||
from patroni.async_executor import AsyncExecutor
|
||||
from patroni.exceptions import DCSError, PostgresConnectionException
|
||||
from patroni.postgresql import ACTION_ON_START
|
||||
from patroni.utils import polling_loop, tzutc
|
||||
from threading import RLock
|
||||
@@ -18,25 +18,24 @@ from threading import RLock
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class _MemberStatus(namedtuple('_MemberStatus', 'member,reachable,in_recovery,wal_position,tags,watchdog_failed')):
|
||||
class _MemberStatus(namedtuple('_MemberStatus', 'member,reachable,in_recovery,xlog_location,tags')):
|
||||
"""Node status distilled from API response:
|
||||
|
||||
member - dcs.Member object of the node
|
||||
reachable - `!False` if the node is not reachable or is not responding with correct JSON
|
||||
in_recovery - `!True` if pg_is_in_recovery() == true
|
||||
wal_position - value of `replayed_location` or `location` from JSON, dependin on its role.
|
||||
xlog_location - value of `replayed_location` or `location` from JSON, dependin on its role.
|
||||
tags - dictionary with values of different tags (i.e. nofailover)
|
||||
watchdog_failed - indicates that watchdog is required by configuration but not available or failed
|
||||
"""
|
||||
@classmethod
|
||||
def from_api_response(cls, member, json):
|
||||
is_master = json['role'] == 'master'
|
||||
wal = not is_master and max(json['xlog'].get('received_location', 0), json['xlog'].get('replayed_location', 0))
|
||||
return cls(member, True, not is_master, wal, json.get('tags', {}), json.get('watchdog_failed', False))
|
||||
xlog = not is_master and max(json['xlog'].get('received_location', 0), json['xlog'].get('replayed_location', 0))
|
||||
return cls(member, True, not is_master, xlog, json.get('tags', {}))
|
||||
|
||||
@classmethod
|
||||
def unknown(cls, member):
|
||||
return cls(member, False, None, 0, {}, False)
|
||||
return cls(member, False, None, 0, {})
|
||||
|
||||
def failover_limitation(self):
|
||||
"""Returns reason why this node can't promote or None if everything is ok."""
|
||||
@@ -44,8 +43,6 @@ class _MemberStatus(namedtuple('_MemberStatus', 'member,reachable,in_recovery,wa
|
||||
return 'not reachable'
|
||||
if self.tags.get('nofailover', False):
|
||||
return 'not allowed to promote'
|
||||
if self.watchdog_failed:
|
||||
return 'not watchdog capable'
|
||||
return None
|
||||
|
||||
|
||||
@@ -58,11 +55,8 @@ class Ha(object):
|
||||
self.cluster = None
|
||||
self.old_cluster = None
|
||||
self.recovering = False
|
||||
self._post_bootstrap_task = None
|
||||
self._crash_recovery_executed = False
|
||||
self._start_timeout = None
|
||||
self._async_executor = AsyncExecutor(self.wakeup)
|
||||
self.watchdog = patroni.watchdog
|
||||
|
||||
# Each member publishes various pieces of information to the DCS using touch_member. This lock protects
|
||||
# the state and publishing procedure to have consistent ordering and avoid publishing stale values.
|
||||
@@ -87,13 +81,11 @@ class Ha(object):
|
||||
|
||||
def update_lock(self, write_leader_optime=False):
|
||||
ret = self.dcs.update_leader()
|
||||
if ret:
|
||||
self.watchdog.keepalive()
|
||||
if write_leader_optime:
|
||||
try:
|
||||
self.dcs.write_leader_optime(self.state_handler.last_operation())
|
||||
except Exception:
|
||||
pass
|
||||
if ret and write_leader_optime:
|
||||
try:
|
||||
self.dcs.write_leader_optime(self.state_handler.last_operation())
|
||||
except:
|
||||
pass
|
||||
return ret
|
||||
|
||||
def has_lock(self):
|
||||
@@ -124,8 +116,8 @@ class Ha(object):
|
||||
data['pending_restart'] = True
|
||||
if not self._async_executor.busy and data['state'] in ['running', 'restarting', 'starting']:
|
||||
try:
|
||||
data['xlog_location'] = self.state_handler.wal_position(retry=False)
|
||||
except Exception:
|
||||
data['xlog_location'] = self.state_handler.xlog_position(retry=False)
|
||||
except:
|
||||
pass
|
||||
if self.patroni.scheduled_restart:
|
||||
scheduled_restart_data = self.patroni.scheduled_restart.copy()
|
||||
@@ -139,7 +131,7 @@ class Ha(object):
|
||||
logger.info('bootstrapped %s', msg)
|
||||
cluster = self.dcs.get_cluster()
|
||||
node_to_follow = self._get_node_to_follow(cluster)
|
||||
return self.state_handler.follow(node_to_follow)
|
||||
return self.state_handler.follow(node_to_follow, cluster.leader, True)
|
||||
else:
|
||||
logger.error('failed to bootstrap %s', msg)
|
||||
self.state_handler.remove_data_directory()
|
||||
@@ -155,82 +147,44 @@ class Ha(object):
|
||||
# no initialize key and node is allowed to be master and has 'bootstrap' section in a configuration file
|
||||
elif self.cluster.initialize is None and not self.patroni.nofailover and 'bootstrap' in self.patroni.config:
|
||||
if self.dcs.initialize(create_new=True): # race for initialization
|
||||
self.state_handler.bootstrapping = True
|
||||
self._post_bootstrap_task = CriticalTask()
|
||||
self._async_executor.schedule('bootstrap')
|
||||
self._async_executor.run_async(self.state_handler.bootstrap, args=(self.patroni.config['bootstrap'],))
|
||||
return 'trying to bootstrap a new cluster'
|
||||
try:
|
||||
self.state_handler.bootstrap(self.patroni.config['bootstrap'])
|
||||
self.dcs.initialize(create_new=False, sysid=self.state_handler.sysid)
|
||||
except: # initdb or start failed
|
||||
# remove initialization key and give a chance to other members
|
||||
logger.info("removing initialize key after failed attempt to initialize the cluster")
|
||||
self.dcs.cancel_initialization()
|
||||
self.state_handler.stop('immediate')
|
||||
self.state_handler.move_data_directory()
|
||||
raise
|
||||
self.dcs.set_config_value(json.dumps(self.patroni.config.dynamic_configuration, separators=(',', ':')))
|
||||
self.dcs.take_leader()
|
||||
self.load_cluster_from_dcs()
|
||||
return 'initialized a new cluster'
|
||||
else:
|
||||
return 'failed to acquire initialize lock'
|
||||
else:
|
||||
if self.state_handler.can_create_replica_without_replication_connection():
|
||||
msg = 'bootstrap (without leader)'
|
||||
self._async_executor.schedule(msg)
|
||||
self._async_executor.run_async(self.clone)
|
||||
return 'trying to ' + msg
|
||||
return "trying to bootstrap (without leader)"
|
||||
return 'waiting for leader to bootstrap'
|
||||
|
||||
def _handle_rewind(self):
|
||||
if self.state_handler.rewind_needed_and_possible(self.cluster.leader):
|
||||
self._async_executor.schedule('running pg_rewind from ' + self.cluster.leader.name)
|
||||
self._async_executor.run_async(self.state_handler.rewind, (self.cluster.leader,))
|
||||
return True
|
||||
|
||||
def _start_crash_recovery(self, msg):
|
||||
self._async_executor.schedule(msg)
|
||||
self._async_executor.run_async(self.state_handler.fix_cluster_state)
|
||||
return msg
|
||||
|
||||
def recover(self):
|
||||
# Postgres is not running and we will restart in standby mode. Watchdog is not needed until we promote.
|
||||
self.watchdog.disable()
|
||||
|
||||
if self.has_lock() and self.update_lock():
|
||||
timeout = self.patroni.config['master_start_timeout']
|
||||
if timeout == 0:
|
||||
# We are requested to prefer failing over to restarting master. But see first if there
|
||||
# is anyone to fail over to.
|
||||
members = self.cluster.members
|
||||
if self.is_synchronous_mode():
|
||||
members = [m for m in members if self.cluster.sync.matches(m.name)]
|
||||
if self.is_failover_possible(members):
|
||||
if self.is_failover_possible(self.cluster.members):
|
||||
logger.info("Master crashed. Failing over.")
|
||||
self.demote('immediate')
|
||||
return 'stopped PostgreSQL to fail over after a crash'
|
||||
else:
|
||||
timeout = None
|
||||
|
||||
data = self.state_handler.controldata()
|
||||
if data.get('Database cluster state') == 'in production' and not self._crash_recovery_executed and \
|
||||
(self.cluster.is_unlocked() or self.state_handler.can_rewind):
|
||||
self._crash_recovery_executed = True
|
||||
return self._start_crash_recovery('doing crash recovery in a single user mode')
|
||||
|
||||
self.load_cluster_from_dcs()
|
||||
|
||||
if self.has_lock():
|
||||
msg = "starting as readonly because i had the session lock"
|
||||
node_to_follow = None
|
||||
else:
|
||||
if not self.state_handler.rewind_executed:
|
||||
self.state_handler.trigger_check_diverged_lsn()
|
||||
if self._handle_rewind():
|
||||
return self._async_executor.scheduled_action
|
||||
msg = "starting as a secondary"
|
||||
node_to_follow = self._get_node_to_follow(self.cluster)
|
||||
|
||||
# once we already tried to start postgres but failed, single user mode is a rescue in this case
|
||||
if self.recovering and not self.state_handler.rewind_executed \
|
||||
and not self._crash_recovery_executed and self.state_handler.can_rewind \
|
||||
and data.get('Database cluster state') not in ('shut down', 'shut down in recovery'):
|
||||
self.recovering = False
|
||||
return self._start_crash_recovery('fixing cluster state in a single user mode')
|
||||
|
||||
self.recovering = True
|
||||
|
||||
self._async_executor.schedule('restarting after failure')
|
||||
self._async_executor.run_async(self.state_handler.follow, (node_to_follow, timeout))
|
||||
return msg
|
||||
return self.follow("starting as readonly because i had the session lock",
|
||||
"starting as a secondary", True, True, None, timeout)
|
||||
|
||||
def _get_node_to_follow(self, cluster):
|
||||
# determine the node to follow. If replicatefrom tag is set,
|
||||
@@ -242,39 +196,32 @@ class Ha(object):
|
||||
|
||||
return node_to_follow if node_to_follow and node_to_follow.name != self.state_handler.name else None
|
||||
|
||||
def follow(self, demote_reason, follow_reason, refresh=True):
|
||||
def follow(self, demote_reason, follow_reason, refresh=True, recovery=False, need_rewind=None, timeout=None):
|
||||
if refresh:
|
||||
self.load_cluster_from_dcs()
|
||||
|
||||
is_leader = self.state_handler.is_leader()
|
||||
if recovery:
|
||||
ret = demote_reason if self.has_lock() else follow_reason
|
||||
else:
|
||||
is_leader = self.state_handler.is_leader()
|
||||
ret = demote_reason if is_leader else follow_reason
|
||||
|
||||
node_to_follow = self._get_node_to_follow(self.cluster)
|
||||
|
||||
if self.is_paused():
|
||||
if not (self.state_handler.need_rewind and self.state_handler.can_rewind) or self.cluster.is_unlocked():
|
||||
self.state_handler.set_role('master' if is_leader else 'replica')
|
||||
if is_leader:
|
||||
return 'continue to run as master without lock'
|
||||
elif not node_to_follow:
|
||||
return 'no action'
|
||||
elif is_leader:
|
||||
self.demote('immediate-nolock')
|
||||
return demote_reason
|
||||
if self.is_paused() and not (self.state_handler.need_rewind and self.state_handler.can_rewind):
|
||||
self.state_handler.set_role('master' if is_leader else 'replica')
|
||||
if is_leader:
|
||||
return 'continue to run as master without lock'
|
||||
elif not node_to_follow:
|
||||
return 'no action'
|
||||
|
||||
if self._handle_rewind():
|
||||
return self._async_executor.scheduled_action
|
||||
self.state_handler.follow(node_to_follow, self.cluster.leader, recovery,
|
||||
self._async_executor, need_rewind, timeout)
|
||||
|
||||
if not self.state_handler.check_recovery_conf(node_to_follow):
|
||||
self._async_executor.schedule('changing primary_conninfo and restarting')
|
||||
self._async_executor.run_async(self.state_handler.follow, (node_to_follow,))
|
||||
|
||||
return follow_reason
|
||||
return ret
|
||||
|
||||
def is_synchronous_mode(self):
|
||||
return bool(self.cluster and self.cluster.is_synchronous_mode())
|
||||
|
||||
def is_synchronous_mode_strict(self):
|
||||
return bool(self.cluster and self.cluster.is_synchronous_mode_strict())
|
||||
return bool(self.cluster and self.cluster.config and self.cluster.config.data.get('synchronous_mode'))
|
||||
|
||||
def process_sync_replication(self):
|
||||
"""Process synchronous standby beahvior.
|
||||
@@ -295,15 +242,10 @@ class Ha(object):
|
||||
if not self.dcs.write_sync_state(self.state_handler.name, None, index=self.cluster.sync.index):
|
||||
logger.info('Synchronous replication key updated by someone else.')
|
||||
return
|
||||
|
||||
if self.is_synchronous_mode_strict() and picked is None:
|
||||
picked = '*'
|
||||
logger.warning("No standbys available!")
|
||||
|
||||
logger.info("Assigning synchronous standby status to %s", picked)
|
||||
self.state_handler.set_synchronous_standby(picked)
|
||||
|
||||
if picked and picked != '*' and not allow_promote:
|
||||
if picked and not allow_promote:
|
||||
# Wait for PostgreSQL to enable synchronous mode and see if we can immediately set sync_standby
|
||||
time.sleep(2)
|
||||
picked, allow_promote = self.state_handler.pick_synchronous_standby(self.cluster)
|
||||
@@ -363,14 +305,6 @@ class Ha(object):
|
||||
self._disable_sync -= 1
|
||||
|
||||
def enforce_master_role(self, message, promote_message):
|
||||
if not self.is_paused() and not self.watchdog.is_running and not self.watchdog.activate():
|
||||
if self.state_handler.is_leader():
|
||||
self.demote('immediate')
|
||||
return 'Demoting self because watchdog could not be activated'
|
||||
else:
|
||||
self.release_leader_key_voluntarily()
|
||||
return 'Not promoting self because watchdog could not be activated'
|
||||
|
||||
if self.state_handler.is_leader() or self.state_handler.role == 'master':
|
||||
# Inform the state handler about its master role.
|
||||
# It may be unaware of it if postgres is promoted manually.
|
||||
@@ -385,7 +319,7 @@ class Ha(object):
|
||||
# Somebody else updated sync state, it may be due to us losing the lock. To be safe, postpone
|
||||
# promotion until next cycle. TODO: trigger immediate retry of run_cycle
|
||||
return 'Postponing promotion because synchronous replication state was updated by somebody else'
|
||||
self.state_handler.set_synchronous_standby('*' if self.is_synchronous_mode_strict() else None)
|
||||
self.state_handler.set_synchronous_standby(None)
|
||||
self.state_handler.promote()
|
||||
return promote_message
|
||||
|
||||
@@ -410,21 +344,21 @@ class Ha(object):
|
||||
pool.join()
|
||||
return results
|
||||
|
||||
def is_lagging(self, wal_position):
|
||||
"""Returns if instance with an wal should consider itself unhealthy to be promoted due to replication lag.
|
||||
def is_lagging(self, xlog_location):
|
||||
"""Returns if instance with an xlog should consider itself unhealthy to be promoted due to replication lag.
|
||||
|
||||
:param wal_position: Current wal position.
|
||||
:param xlog_location: Current xlog location.
|
||||
:returns True when node is lagging
|
||||
"""
|
||||
lag = (self.cluster.last_leader_operation or 0) - wal_position
|
||||
lag = (self.cluster.last_leader_operation or 0) - xlog_location
|
||||
return lag > self.state_handler.config.get('maximum_lag_on_failover', 0)
|
||||
|
||||
def _is_healthiest_node(self, members, check_replication_lag=True):
|
||||
"""This method tries to determine whether I am healthy enough to became a new leader candidate or not."""
|
||||
|
||||
my_wal_position = self.state_handler.wal_position()
|
||||
if check_replication_lag and self.is_lagging(my_wal_position):
|
||||
return False # Too far behind last reported wal position on master
|
||||
my_xlog_location = self.state_handler.xlog_position()
|
||||
if check_replication_lag and self.is_lagging(my_xlog_location):
|
||||
return False # Too far behind last reported xlog location on master
|
||||
|
||||
# Prepare list of nodes to run check against
|
||||
members = [m for m in members if m.name != self.state_handler.name and not m.nofailover and m.api_url]
|
||||
@@ -435,7 +369,7 @@ class Ha(object):
|
||||
if not st.in_recovery:
|
||||
logger.warning('Master (%s) is still alive', st.member.name)
|
||||
return False
|
||||
if my_wal_position < st.wal_position:
|
||||
if my_xlog_location < st.xlog_location:
|
||||
return False
|
||||
return True
|
||||
|
||||
@@ -447,7 +381,7 @@ class Ha(object):
|
||||
not_allowed_reason = st.failover_limitation()
|
||||
if not_allowed_reason:
|
||||
logger.info('Member %s is %s', st.member.name, not_allowed_reason)
|
||||
elif self.is_lagging(st.wal_position):
|
||||
elif self.is_lagging(st.xlog_location):
|
||||
logger.info('Member %s exceeds maximum replication lag', st.member.name)
|
||||
else:
|
||||
ret = True
|
||||
@@ -524,9 +458,6 @@ class Ha(object):
|
||||
if self.cluster.failover:
|
||||
return self.manual_failover_process_no_leader()
|
||||
|
||||
if not self.watchdog.is_healthy:
|
||||
return False
|
||||
|
||||
# When in sync mode, only last known master and sync standby are allowed to promote automatically.
|
||||
all_known_members = self.cluster.members + self.old_cluster.members
|
||||
if self.is_synchronous_mode() and self.cluster.sync.leader:
|
||||
@@ -554,42 +485,31 @@ class Ha(object):
|
||||
graceful is used when failing over to another node due to user request. May only be called running async.
|
||||
immediate is used when we determine that we are not suitable for master and want to failover quickly
|
||||
without regard for data durability. May only be called synchronously.
|
||||
immediate-nolock is used when find out that we have lost the lock to be master. Need to bring down
|
||||
PostgreSQL as quickly as possible without regard for data durability. May only be called synchronously.
|
||||
"""
|
||||
mode_control = {
|
||||
'offline': dict(stop='fast', checkpoint=False, release=False, offline=True, async=False),
|
||||
'graceful': dict(stop='fast', checkpoint=True, release=True, offline=False, async=False),
|
||||
'immediate': dict(stop='immediate', checkpoint=False, release=True, offline=False, async=True),
|
||||
'immediate-nolock': dict(stop='immediate', checkpoint=False, release=False, offline=False, async=True),
|
||||
}[mode]
|
||||
|
||||
self.state_handler.trigger_check_diverged_lsn()
|
||||
self.state_handler.stop(mode_control['stop'], checkpoint=mode_control['checkpoint'],
|
||||
on_safepoint=self.watchdog.disable if self.watchdog.is_running else None)
|
||||
self.state_handler.set_role('demoted')
|
||||
|
||||
if mode_control['release']:
|
||||
assert mode in ['offline', 'graceful', 'immediate']
|
||||
if mode != 'offline':
|
||||
if mode == 'immediate':
|
||||
self.state_handler.stop('immediate', checkpoint=False)
|
||||
else:
|
||||
self.state_handler.stop()
|
||||
self.state_handler.set_role('demoted')
|
||||
self.release_leader_key_voluntarily()
|
||||
time.sleep(2) # Give a time to somebody to take the leader lock
|
||||
if mode_control['offline']:
|
||||
node_to_follow, leader = None, None
|
||||
else:
|
||||
cluster = self.dcs.get_cluster()
|
||||
node_to_follow, leader = self._get_node_to_follow(cluster), cluster.leader
|
||||
|
||||
# FIXME: with mode offline called from DCS exception handler and handle_long_action_in_progress
|
||||
# there could be an async action already running, calling follow from here will lead
|
||||
# to racy state handler state updates.
|
||||
if mode_control['async']:
|
||||
self._async_executor.schedule('starting after demotion')
|
||||
self._async_executor.run_async(self.state_handler.follow, (node_to_follow,))
|
||||
node_to_follow = self._get_node_to_follow(cluster)
|
||||
if mode == 'immediate':
|
||||
# We will try to start up as a standby now. If no one takes the leader lock before we finish
|
||||
# recovery we will try to promote ourselves.
|
||||
self._async_executor.schedule('waiting for failover to complete')
|
||||
self._async_executor.run_async(self.state_handler.follow,
|
||||
(node_to_follow, cluster.leader, True, None, True))
|
||||
else:
|
||||
return self.state_handler.follow(node_to_follow, cluster.leader, recovery=True, need_rewind=True)
|
||||
else:
|
||||
if self.is_synchronous_mode():
|
||||
self.state_handler.set_synchronous_standby(None)
|
||||
if self.state_handler.rewind_needed_and_possible(leader):
|
||||
return False # do not start postgres, but run pg_rewind on the next iteration
|
||||
self.state_handler.follow(node_to_follow)
|
||||
# Need to become unavailable as soon as possible, so initiate a stop here. However as we can't release
|
||||
# the leader key we don't care about confirming the shutdown quickly and can use a regular stop.
|
||||
self.state_handler.stop(checkpoint=False)
|
||||
self.state_handler.follow(None, None, recovery=True)
|
||||
|
||||
def should_run_scheduled_action(self, action_name, scheduled_at, cleanup_fn):
|
||||
if scheduled_at and not self.is_paused():
|
||||
@@ -645,16 +565,8 @@ class Ha(object):
|
||||
if not failover.candidate and self.is_paused():
|
||||
logger.warning('Failover is possible only to a specific candidate in a paused state')
|
||||
else:
|
||||
if self.is_synchronous_mode():
|
||||
if failover.candidate and not self.cluster.sync.matches(failover.candidate):
|
||||
logger.warning('Failover candidate=%s does not match with sync_standby=%s',
|
||||
failover.candidate, self.cluster.sync.sync_standby)
|
||||
members = []
|
||||
else:
|
||||
members = [m for m in self.cluster.members if self.cluster.sync.matches(m.name)]
|
||||
else:
|
||||
members = [m for m in self.cluster.members
|
||||
if not failover.candidate or m.name == failover.candidate]
|
||||
members = [m for m in self.cluster.members
|
||||
if not failover.candidate or m.name == failover.candidate]
|
||||
if self.is_failover_possible(members): # check that there are healthy members
|
||||
self._async_executor.schedule('manual failover: demote')
|
||||
self._async_executor.run_async(self.demote, ('graceful',))
|
||||
@@ -692,15 +604,17 @@ class Ha(object):
|
||||
else:
|
||||
# when we are doing manual failover there is no guaranty that new leader is ahead of any other node
|
||||
# node tagged as nofailover can be ahead of the new leader either, but it is always excluded from elections
|
||||
if bool(self.cluster.failover) or self.patroni.nofailover:
|
||||
self.state_handler.trigger_check_diverged_lsn()
|
||||
need_rewind = bool(self.cluster.failover) or self.patroni.nofailover
|
||||
if need_rewind:
|
||||
time.sleep(2) # Give a time to somebody to take the leader lock
|
||||
|
||||
if self.patroni.nofailover:
|
||||
return self.follow('demoting self because I am not allowed to become master',
|
||||
'following a different leader because I am not allowed to promote')
|
||||
'following a different leader because I am not allowed to promote',
|
||||
need_rewind=need_rewind)
|
||||
return self.follow('demoting self because i am not the healthiest node',
|
||||
'following a different leader because i am not the healthiest node')
|
||||
'following a different leader because i am not the healthiest node',
|
||||
need_rewind=need_rewind)
|
||||
|
||||
def process_healthy_cluster(self):
|
||||
if self.has_lock():
|
||||
@@ -723,14 +637,14 @@ class Ha(object):
|
||||
# Either there is no connection to DCS or someone else acquired the lock
|
||||
logger.error('failed to update leader lock')
|
||||
if self.state_handler.is_leader():
|
||||
self.demote('immediate-nolock')
|
||||
self.demote('offline')
|
||||
return 'demoted self because failed to update leader lock in DCS'
|
||||
else:
|
||||
return 'not promoting because failed to update leader lock in DCS'
|
||||
else:
|
||||
logger.info('does not have lock')
|
||||
return self.follow('demoting self because i do not have the lock and i was a leader',
|
||||
'no action. i am a secondary and i am following a leader', refresh=False)
|
||||
'no action. i am a secondary and i am following a leader', False)
|
||||
|
||||
def evaluate_scheduled_restart(self):
|
||||
if self._async_executor.busy: # Restart already in progress
|
||||
@@ -822,13 +736,13 @@ class Ha(object):
|
||||
# leader key (if it belong to us) rather than trying to start postgres once again.
|
||||
self.recovering = True
|
||||
|
||||
# Now that restart is scheduled we can set timeout for startup, it will get reset
|
||||
# No that restart is scheduled we can set timeout for startup, it will get reset
|
||||
# once async executor runs and main loop notices PostgreSQL as up.
|
||||
timeout = restart_data.get('timeout', self.patroni.config['master_start_timeout'])
|
||||
self.set_start_timeout(timeout)
|
||||
|
||||
# For non async cases we want to wait for restart to complete or timeout before returning.
|
||||
do_restart = functools.partial(self.state_handler.restart, timeout, self._async_executor.critical_task)
|
||||
do_restart = functools.partial(self.state_handler.restart, timeout)
|
||||
if self.is_synchronous_mode() and not self.has_lock():
|
||||
do_restart = functools.partial(self.while_not_sync_standby, do_restart)
|
||||
|
||||
@@ -869,25 +783,15 @@ class Ha(object):
|
||||
self._async_executor.run_async(self._do_reinitialize, args=(self.cluster, ))
|
||||
|
||||
def handle_long_action_in_progress(self):
|
||||
if self.has_lock() and self.update_lock():
|
||||
return 'updated leader lock during ' + self._async_executor.scheduled_action
|
||||
elif not self.state_handler.bootstrapping:
|
||||
# Don't have lock, make sure we are not starting up a master in the background
|
||||
if self.state_handler.role == 'master':
|
||||
logger.info("Demoting master during " + self._async_executor.scheduled_action)
|
||||
if self._async_executor.scheduled_action == 'restart':
|
||||
# Restart needs a special interlocking cancel because postmaster may be just started in a
|
||||
# background thread and has not even written a pid file yet.
|
||||
with self._async_executor.critical_task as task:
|
||||
if not task.cancel():
|
||||
self.state_handler.terminate_starting_postmaster(pid=task.result)
|
||||
self.demote('immediate-nolock')
|
||||
return 'lost leader lock during ' + self._async_executor.scheduled_action
|
||||
|
||||
if self.cluster.is_unlocked():
|
||||
logger.info('not healthy enough for leader race')
|
||||
|
||||
return self._async_executor.scheduled_action + ' in progress'
|
||||
if self.has_lock():
|
||||
if self.update_lock():
|
||||
return 'updated leader lock during ' + self._async_executor.scheduled_action
|
||||
else:
|
||||
return 'failed to update leader lock during ' + self._async_executor.scheduled_action
|
||||
elif self.cluster.is_unlocked():
|
||||
return 'not healthy enough for leader race'
|
||||
else:
|
||||
return self._async_executor.scheduled_action + ' in progress'
|
||||
|
||||
@staticmethod
|
||||
def sysid_valid(sysid):
|
||||
@@ -898,48 +802,14 @@ class Ha(object):
|
||||
|
||||
def post_recover(self):
|
||||
if not self.state_handler.is_running():
|
||||
self.watchdog.disable()
|
||||
if self.has_lock():
|
||||
self.state_handler.set_role('demoted')
|
||||
self.dcs.delete_leader()
|
||||
self.dcs.reset_cluster()
|
||||
return 'removed leader key after trying and failing to start postgres'
|
||||
return 'failed to start postgres'
|
||||
self._crash_recovery_executed = False
|
||||
return None
|
||||
|
||||
def cancel_initialization(self):
|
||||
logger.info('removing initialize key after failed attempt to bootstrap the cluster')
|
||||
self.dcs.cancel_initialization()
|
||||
self.state_handler.stop('immediate')
|
||||
self.state_handler.move_data_directory()
|
||||
raise PatroniException('Failed to bootstrap cluster')
|
||||
|
||||
def post_bootstrap(self):
|
||||
# bootstrap has failed if postgres is not running
|
||||
if not self.state_handler.is_running() or self._post_bootstrap_task.result is False:
|
||||
self.cancel_initialization()
|
||||
|
||||
if self._post_bootstrap_task.result is None:
|
||||
if not self.state_handler.is_leader():
|
||||
return 'waiting for end of recovery after bootstrap'
|
||||
|
||||
self.state_handler.set_role('master')
|
||||
self._async_executor.schedule('post_bootstrap')
|
||||
self._async_executor.run_async(self.state_handler.post_bootstrap,
|
||||
args=(self.patroni.config['bootstrap'], self._post_bootstrap_task))
|
||||
return 'running post_bootstrap'
|
||||
|
||||
self.state_handler.bootstrapping = False
|
||||
self.dcs.set_config_value(json.dumps(self.patroni.config.dynamic_configuration, separators=(',', ':')))
|
||||
if not self.watchdog.activate():
|
||||
logger.error('Cancelling bootstrap because watchdog activation failed')
|
||||
self.cancel_initialization()
|
||||
self.dcs.take_leader()
|
||||
self.state_handler.call_nowait(ACTION_ON_START)
|
||||
self.load_cluster_from_dcs()
|
||||
return 'initialized a new cluster'
|
||||
|
||||
def handle_starting_instance(self):
|
||||
"""Starting up PostgreSQL may take a long time. In case we are the leader we may want to
|
||||
fail over to."""
|
||||
@@ -947,15 +817,13 @@ class Ha(object):
|
||||
# Check if we are in startup, when paused defer to main loop for manual failovers.
|
||||
if not self.state_handler.check_for_startup() or self.is_paused():
|
||||
self.set_start_timeout(None)
|
||||
if self.is_paused():
|
||||
self.state_handler.set_state(self.state_handler.is_running() and 'running' or 'stopped')
|
||||
return None
|
||||
|
||||
# state_handler.state == 'starting' here
|
||||
if self.has_lock():
|
||||
if not self.update_lock():
|
||||
logger.info("Lost lock while starting up. Demoting self.")
|
||||
self.demote('immediate-nolock')
|
||||
self.demote('immediate')
|
||||
return 'stopped PostgreSQL while starting up because leader key was lost'
|
||||
|
||||
timeout = self._start_timeout or self.patroni.config['master_start_timeout']
|
||||
@@ -990,9 +858,6 @@ class Ha(object):
|
||||
try:
|
||||
self.load_cluster_from_dcs()
|
||||
|
||||
if self.is_paused():
|
||||
self.watchdog.disable()
|
||||
|
||||
if not self.cluster.has_member(self.state_handler.name):
|
||||
self.touch_member()
|
||||
|
||||
@@ -1012,9 +877,6 @@ class Ha(object):
|
||||
return msg
|
||||
|
||||
# we've got here, so any async action has finished.
|
||||
if self.state_handler.bootstrapping:
|
||||
return self.post_bootstrap()
|
||||
|
||||
if self.recovering and not self.state_handler.need_rewind:
|
||||
self.recovering = False
|
||||
# Check if we tried to recover and failed
|
||||
@@ -1024,11 +886,6 @@ class Ha(object):
|
||||
|
||||
# is data directory empty?
|
||||
if self.state_handler.data_directory_empty():
|
||||
self.state_handler.set_role('uninitialized')
|
||||
self.state_handler.stop()
|
||||
# In case datadir went away while we were master. TODO: check for this and try to stop postgresql.
|
||||
self.watchdog.disable()
|
||||
|
||||
# is this instance the leader?
|
||||
if self.has_lock():
|
||||
self.release_leader_key_voluntarily()
|
||||
@@ -1051,8 +908,7 @@ class Ha(object):
|
||||
self.dcs.delete_leader()
|
||||
self.dcs.reset_cluster()
|
||||
return 'removed leader lock because postgres is not running'
|
||||
elif not (self.state_handler.rewind_executed or
|
||||
self.state_handler.need_rewind and self.state_handler.can_rewind):
|
||||
elif not (self.state_handler.need_rewind and self.state_handler.can_rewind):
|
||||
return 'postgres is not running'
|
||||
|
||||
# try to start dead postgres
|
||||
@@ -1070,8 +926,6 @@ class Ha(object):
|
||||
# asynchronous processes are running (should be always the case for the master)
|
||||
if not self._async_executor.busy and not self.state_handler.is_starting():
|
||||
if not self.state_handler.cb_called:
|
||||
if not self.state_handler.is_leader():
|
||||
self.state_handler.trigger_check_diverged_lsn()
|
||||
self.state_handler.call_nowait(ACTION_ON_START)
|
||||
self.state_handler.sync_replication_slots(self.cluster)
|
||||
except DCSError:
|
||||
@@ -1092,26 +946,6 @@ class Ha(object):
|
||||
info = self._run_cycle()
|
||||
return (self.is_paused() and 'PAUSE: ' or '') + info
|
||||
|
||||
def shutdown(self):
|
||||
if self.is_paused():
|
||||
logger.info('Leader key is not deleted and Postgresql is not stopped due paused state')
|
||||
self.watchdog.disable()
|
||||
else:
|
||||
# FIXME: If stop doesn't reach safepoint quickly enough keepalive is triggered. If shutdown checkpoint
|
||||
# takes longer than ttl, then leader key is lost and replication might not have sent out all xlog.
|
||||
# This might not be the desired behavior of users, as a graceful shutdown of the host can mean lost data.
|
||||
# We probably need to something smarter here.
|
||||
disable_wd = self.watchdog.disable if self.watchdog.is_running else None
|
||||
self.while_not_sync_standby(lambda: self.state_handler.stop(checkpoint=False, on_safepoint=disable_wd))
|
||||
if not self.state_handler.is_running():
|
||||
self.dcs.delete_leader()
|
||||
else:
|
||||
# XXX: what about when Patroni is started as the wrong user that has access to the watchdog device
|
||||
# but cannot shut down PostgreSQL. Root would be the obvious example. Would be nice to not kill the
|
||||
# system due to a bad config.
|
||||
logger.error("PostgreSQL shutdown failed, leader key not removed." +
|
||||
(" Leaving watchdog running." if self.watchdog.is_running else ""))
|
||||
|
||||
def watch(self, timeout):
|
||||
cluster = self.cluster
|
||||
# watch on leader key changes if the postgres is running and leader is known and current node is not lock owner
|
||||
|
||||
+279
-787
File diff suppressed because it is too large
Load Diff
@@ -10,27 +10,27 @@ from patroni.utils import Retry, RetryFailedError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
retry_timeout = 15
|
||||
|
||||
|
||||
class AWSConnection(object):
|
||||
|
||||
def __init__(self, cluster_name):
|
||||
self.available = False
|
||||
self.cluster_name = cluster_name if cluster_name is not None else 'unknown'
|
||||
self._retry = Retry(deadline=300, max_delay=30, max_tries=-1, retry_exceptions=(boto.exception.StandardError,))
|
||||
self._retry = Retry(deadline=retry_timeout, max_delay=5, max_tries=-1, retry_exceptions=(boto.exception,))
|
||||
try:
|
||||
# get the instance id
|
||||
r = requests.get('http://169.254.169.254/latest/dynamic/instance-identity/document', timeout=2.1)
|
||||
r = requests.get('http://169.254.169.254/latest/dynamic/instance-identity/document', timeout=0.1)
|
||||
except RequestException:
|
||||
logger.error('cannot query AWS meta-data')
|
||||
logger.info("cannot query AWS meta-data")
|
||||
return
|
||||
|
||||
if r.ok:
|
||||
try:
|
||||
content = r.json()
|
||||
self.instance_id = content['instanceId']
|
||||
self.region = content['region']
|
||||
except Exception:
|
||||
logger.exception('unable to fetch instance id and region from AWS meta-data')
|
||||
except Exception as e:
|
||||
logger.info('unable to fetch instance id and region from AWS meta-data: {}'.format(e))
|
||||
return
|
||||
self.available = True
|
||||
|
||||
|
||||
+56
-193
@@ -23,29 +23,19 @@
|
||||
# currently also requires that you configure the restore_command to use wal_e, example:
|
||||
# recovery_conf:
|
||||
# restore_command: envdir /etc/wal-e.d/env wal-e wal-fetch "%f" "%p" -p 1
|
||||
import argparse
|
||||
import csv
|
||||
|
||||
from collections import namedtuple
|
||||
import logging
|
||||
import os
|
||||
import psycopg2
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
from collections import namedtuple
|
||||
import argparse
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
RETRY_SLEEP_INTERVAL = 1
|
||||
si_prefixes = ['K', 'M', 'G', 'T', 'P', 'E', 'Z', 'Y']
|
||||
|
||||
|
||||
# Meaningful names to the exit codes used by WALERestore
|
||||
ExitCode = type('Enum', (), {
|
||||
'SUCCESS': 0, #: Succeeded
|
||||
'RETRY_LATER': 1, #: External issue, retry later
|
||||
'FAIL': 2 #: Don't try again unless configuration changes
|
||||
})
|
||||
|
||||
|
||||
# We need to know the current PG version in order to figure out the correct WAL directory name
|
||||
@@ -60,141 +50,66 @@ def get_major_version(data_dir):
|
||||
return 0.0
|
||||
|
||||
|
||||
def repr_size(n_bytes):
|
||||
"""
|
||||
>>> repr_size(1000)
|
||||
'1000 Bytes'
|
||||
>>> repr_size(8257332324597)
|
||||
'7.5 TiB'
|
||||
"""
|
||||
if n_bytes < 1024:
|
||||
return '{0} Bytes'.format(n_bytes)
|
||||
i = -1
|
||||
while n_bytes > 1023:
|
||||
n_bytes /= 1024.0
|
||||
i += 1
|
||||
return '{0} {1}iB'.format(round(n_bytes, 1), si_prefixes[i])
|
||||
|
||||
|
||||
def size_as_bytes(size_, prefix):
|
||||
"""
|
||||
>>> size_as_bytes(7.5, 'T')
|
||||
8246337208320
|
||||
"""
|
||||
prefix = prefix.upper()
|
||||
|
||||
assert prefix in si_prefixes
|
||||
|
||||
exponent = si_prefixes.index(prefix) + 1
|
||||
|
||||
return int(size_ * (1024.0 ** exponent))
|
||||
|
||||
|
||||
WALEConfig = namedtuple(
|
||||
'WALEConfig',
|
||||
[
|
||||
'env_dir',
|
||||
'threshold_mb',
|
||||
'threshold_pct',
|
||||
'cmd',
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
class WALERestore(object):
|
||||
def __init__(self, scope, datadir, connstring, env_dir, threshold_mb,
|
||||
threshold_pct, use_iam, no_master, retries):
|
||||
def __init__(self, scope, datadir, connstring, env_dir, threshold_mb, threshold_pct, use_iam, no_master, retries):
|
||||
self.scope = scope
|
||||
self.master_connection = connstring
|
||||
self.data_dir = datadir
|
||||
self.wal_e = namedtuple('wale', 'dir,threshold_mb,threshold_pct,iam_string,cmd')
|
||||
self.wal_e.dir = env_dir
|
||||
self.wal_e.threshold_mb = threshold_mb
|
||||
self.wal_e.threshold_pct = threshold_pct
|
||||
self.wal_e.iam_string = ' --aws-instance-profile ' if use_iam == 1 else ''
|
||||
self.no_master = no_master
|
||||
|
||||
wale_cmd = [
|
||||
'envdir',
|
||||
env_dir,
|
||||
'wal-e',
|
||||
]
|
||||
|
||||
if use_iam == 1:
|
||||
wale_cmd += ['--aws-instance-profile']
|
||||
|
||||
self.wal_e = WALEConfig(
|
||||
env_dir=env_dir,
|
||||
threshold_mb=threshold_mb,
|
||||
threshold_pct=threshold_pct,
|
||||
cmd=wale_cmd,
|
||||
)
|
||||
|
||||
self.init_error = (not os.path.exists(self.wal_e.env_dir))
|
||||
self.wal_e.cmd = 'envdir {0} wal-e {1} '.format(self.wal_e.dir, self.wal_e.iam_string)
|
||||
self.init_error = (not os.path.exists(self.wal_e.dir))
|
||||
self.retries = retries
|
||||
|
||||
def run(self):
|
||||
"""
|
||||
Creates a new replica using WAL-E
|
||||
|
||||
Returns
|
||||
-------
|
||||
ExitCode
|
||||
0 = Success
|
||||
1 = Error, try again
|
||||
2 = Error, don't try again
|
||||
|
||||
"""
|
||||
if self.init_error:
|
||||
logger.error('init error: %r did not exist at initialization time',
|
||||
self.wal_e.env_dir)
|
||||
return ExitCode.FAIL
|
||||
|
||||
try:
|
||||
should_use_s3 = self.should_use_s3_to_create_replica()
|
||||
if should_use_s3 is None: # Need to retry
|
||||
return ExitCode.RETRY_LATER
|
||||
elif should_use_s3:
|
||||
return self.create_replica_with_s3()
|
||||
elif not should_use_s3:
|
||||
return ExitCode.FAIL
|
||||
except Exception:
|
||||
logger.exception("Unhandled exception when running WAL-E restore")
|
||||
return ExitCode.FAIL
|
||||
""" creates a new replica using WAL-E """
|
||||
if not self.init_error:
|
||||
try:
|
||||
ret = self.should_use_s3_to_create_replica()
|
||||
if ret:
|
||||
return self.create_replica_with_s3()
|
||||
elif ret is None: # caught an exception, need to retry
|
||||
return 1
|
||||
except Exception:
|
||||
logger.exception("Exception when running WAL-E restore")
|
||||
return 2
|
||||
|
||||
def should_use_s3_to_create_replica(self):
|
||||
""" determine whether it makes sense to use S3 and not pg_basebackup """
|
||||
|
||||
threshold_megabytes = self.wal_e.threshold_mb
|
||||
threshold_percent = self.wal_e.threshold_pct
|
||||
threshold_backup_size_percentage = self.wal_e.threshold_pct
|
||||
|
||||
try:
|
||||
cmd = self.wal_e.cmd + ['backup-list', '--detail', 'LATEST']
|
||||
|
||||
logger.debug('calling %r', cmd)
|
||||
wale_output = subprocess.check_output(cmd)
|
||||
|
||||
reader = csv.DictReader(wale_output.decode('utf-8').splitlines(),
|
||||
dialect='excel-tab')
|
||||
rows = list(reader)
|
||||
if not len(rows):
|
||||
logger.warning('wal-e did not find any backups')
|
||||
latest_backup = subprocess.check_output(self.wal_e.cmd.split() + ['backup-list', '--detail', 'LATEST'])
|
||||
# name last_modified expanded_size_bytes wal_segment_backup_start wal_segment_offset_backup_start
|
||||
# wal_segment_backup_stop wal_segment_offset_backup_stop
|
||||
# base_00000001000000000000007F_00000040 2015-05-18T10:13:25.000Z
|
||||
# 20310671 00000001000000000000007F 00000040
|
||||
# 00000001000000000000007F 00000240
|
||||
backup_strings = latest_backup.decode('utf-8').splitlines() if latest_backup else ()
|
||||
if len(backup_strings) != 2:
|
||||
return False
|
||||
|
||||
# This check might not add much, it was performed in the previous
|
||||
# version of this code. since the old version rolled CSV parsing the
|
||||
# check may have been part of the CSV parsing.
|
||||
if len(rows) > 1:
|
||||
logger.warning(
|
||||
'wal-e returned more than one row of backups: %r',
|
||||
rows)
|
||||
names = backup_strings[0].split()
|
||||
vals = backup_strings[1].split()
|
||||
if (len(names) != len(vals)) or (len(names) != 7):
|
||||
return False
|
||||
|
||||
backup_info = rows[0]
|
||||
backup_info = dict(zip(names, vals))
|
||||
except subprocess.CalledProcessError:
|
||||
logger.exception("could not query wal-e latest backup")
|
||||
return None
|
||||
|
||||
try:
|
||||
backup_size = int(backup_info['expanded_size_bytes'])
|
||||
backup_size = backup_info['expanded_size_bytes']
|
||||
backup_start_segment = backup_info['wal_segment_backup_start']
|
||||
backup_start_offset = backup_info['wal_segment_offset_backup_start']
|
||||
except KeyError:
|
||||
except Exception:
|
||||
logger.exception("unable to get some of WALE backup parameters")
|
||||
return None
|
||||
|
||||
@@ -209,29 +124,24 @@ class WALERestore(object):
|
||||
# construct the LSN from the segment and offset
|
||||
backup_start_lsn = '{0}/{1}'.format(lsn_segment, lsn_offset)
|
||||
|
||||
diff_in_bytes = backup_size
|
||||
diff_in_bytes = int(backup_size)
|
||||
attempts_no = 0
|
||||
while True:
|
||||
if self.master_connection:
|
||||
try:
|
||||
# get the difference in bytes between the current WAL location and the backup start offset
|
||||
with psycopg2.connect(self.master_connection) as con:
|
||||
if con.server_version >= 100000:
|
||||
wal_name = 'wal'
|
||||
lsn_name = 'lsn'
|
||||
else:
|
||||
wal_name = 'xlog'
|
||||
lsn_name = 'location'
|
||||
con.autocommit = True
|
||||
with con.cursor() as cur:
|
||||
cur.execute("""SELECT CASE WHEN pg_is_in_recovery()
|
||||
THEN GREATEST(
|
||||
pg_{0}_{1}_diff(COALESCE(
|
||||
pg_last_{0}_receive_{1}(), '0/0'), %s)::bigint,
|
||||
pg_{0}_{1}_diff(pg_last_{0}_replay_{1}(), %s)::bigint)
|
||||
ELSE pg_{0}_{1}_diff(pg_current_{0}_{1}(), %s)::bigint
|
||||
END""".format(wal_name, lsn_name),
|
||||
(backup_start_lsn, backup_start_lsn, backup_start_lsn))
|
||||
pg_xlog_location_diff(COALESCE(
|
||||
pg_last_xlog_receive_location(), '0/0'), %s)::bigint,
|
||||
pg_xlog_location_diff(
|
||||
pg_last_xlog_replay_location(), %s)::bigint)
|
||||
ELSE pg_xlog_location_diff(
|
||||
pg_current_xlog_location(), %s)::bigint
|
||||
END""", (backup_start_lsn, backup_start_lsn, backup_start_lsn))
|
||||
|
||||
diff_in_bytes = int(cur.fetchone()[0])
|
||||
except psycopg2.Error:
|
||||
@@ -253,47 +163,8 @@ class WALERestore(object):
|
||||
|
||||
# if the size of the accumulated WAL segments is more than a certan percentage of the backup size
|
||||
# or exceeds the pre-determined size - pg_basebackup is chosen instead.
|
||||
is_size_thresh_ok = diff_in_bytes < int(threshold_megabytes) * 1048576
|
||||
threshold_pct_bytes = backup_size * threshold_percent / 100.0
|
||||
is_percentage_thresh_ok = float(diff_in_bytes) < int(threshold_pct_bytes)
|
||||
are_thresholds_ok = is_size_thresh_ok and is_percentage_thresh_ok
|
||||
|
||||
class Size(object):
|
||||
def __init__(self, n_bytes, prefix=None):
|
||||
self.n_bytes = n_bytes
|
||||
self.prefix = prefix
|
||||
|
||||
def __repr__(self):
|
||||
if self.prefix is not None:
|
||||
n_bytes = size_as_bytes(self.n_bytes, self.prefix)
|
||||
else:
|
||||
n_bytes = self.n_bytes
|
||||
return repr_size(n_bytes)
|
||||
|
||||
class HumanContext(object):
|
||||
def __init__(self, items):
|
||||
self.items = items
|
||||
|
||||
def __repr__(self):
|
||||
return ', '.join('{}={!r}'.format(key, value)
|
||||
for key, value in self.items)
|
||||
|
||||
human_context = repr(HumanContext([
|
||||
('threshold_size', Size(threshold_megabytes, 'M')),
|
||||
('threshold_percent', threshold_percent),
|
||||
('threshold_percent_size', Size(threshold_pct_bytes)),
|
||||
('backup_size', Size(backup_size)),
|
||||
('backup_diff', Size(diff_in_bytes)),
|
||||
('is_size_thresh_ok', is_size_thresh_ok),
|
||||
('is_percentage_thresh_ok', is_percentage_thresh_ok),
|
||||
]))
|
||||
|
||||
if not are_thresholds_ok:
|
||||
logger.info('wal-e backup size diff is over threshold, falling back '
|
||||
'to other means of restore: %s', human_context)
|
||||
else:
|
||||
logger.info('Thresholds are OK, using wal-e basebackup: %s', human_context)
|
||||
return are_thresholds_ok
|
||||
return (diff_in_bytes < int(threshold_megabytes) * 1048576) and\
|
||||
(diff_in_bytes < int(backup_size) * float(threshold_backup_size_percentage) / 100)
|
||||
|
||||
def fix_subdirectory_path_if_broken(self, dirname):
|
||||
# in case it is a symlink pointing to a non-existing location, remove it and create the actual directory
|
||||
@@ -316,19 +187,15 @@ class WALERestore(object):
|
||||
def create_replica_with_s3(self):
|
||||
# if we're set up, restore the replica using fetch latest
|
||||
try:
|
||||
cmd = self.wal_e.cmd + ['backup-fetch',
|
||||
'{}'.format(self.data_dir),
|
||||
'LATEST']
|
||||
logger.debug('calling: %r', cmd)
|
||||
exit_code = subprocess.call(cmd)
|
||||
ret = subprocess.call(self.wal_e.cmd.split() + ['backup-fetch', '{}'.format(self.data_dir), 'LATEST'])
|
||||
except Exception as e:
|
||||
logger.error('Error when fetching backup with WAL-E: {0}'.format(e))
|
||||
return ExitCode.RETRY_LATER
|
||||
return 1
|
||||
|
||||
if (exit_code == 0 and not
|
||||
self.fix_subdirectory_path_if_broken('pg_xlog' if get_major_version(self.data_dir) < 10 else 'pg_wal')):
|
||||
return ExitCode.FAIL
|
||||
return exit_code
|
||||
if (ret == 0 and not
|
||||
self.fix_subdirectory_path_if_broken('pg_xlog' if get_major_version(self.data_dir) < 10.0 else 'pg_wal')):
|
||||
return 2
|
||||
return ret
|
||||
|
||||
|
||||
def main():
|
||||
@@ -346,9 +213,6 @@ def main():
|
||||
parser.add_argument('--no_master', type=int, default=0)
|
||||
args = parser.parse_args()
|
||||
|
||||
exit_code = None
|
||||
assert args.retries >= 0
|
||||
|
||||
# Retry cloning in a loop. We do separate retries for the master
|
||||
# connection attempt inside should_use_s3_to_create_replica,
|
||||
# because we need to differentiate between the last attempt and
|
||||
@@ -359,13 +223,12 @@ def main():
|
||||
env_dir=args.envdir, threshold_mb=args.threshold_megabytes,
|
||||
threshold_pct=args.threshold_backup_size_percentage, use_iam=args.use_iam,
|
||||
no_master=args.no_master, retries=args.retries)
|
||||
exit_code = restore.run()
|
||||
if not exit_code == ExitCode.RETRY_LATER: # only WAL-E failures lead to the retry
|
||||
logger.debug('exit_code is %r, not retrying', exit_code)
|
||||
ret = restore.run()
|
||||
if ret != 1: # only WAL-E failures lead to the retry
|
||||
break
|
||||
time.sleep(RETRY_SLEEP_INTERVAL)
|
||||
|
||||
return exit_code
|
||||
return ret
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
+5
-13
@@ -95,17 +95,17 @@ def strtol(value, strict=True):
|
||||
True
|
||||
"""
|
||||
value = str(value).strip()
|
||||
ln = len(value)
|
||||
l = len(value)
|
||||
i = 0
|
||||
# skip sign:
|
||||
if i < ln and value[i] in ('-', '+'):
|
||||
if i < l and value[i] in ('-', '+'):
|
||||
i += 1
|
||||
|
||||
# we always expect to get digit in the beginning
|
||||
if i < ln and value[i].isdigit():
|
||||
if i < l and value[i].isdigit():
|
||||
if value[i] == '0':
|
||||
i += 1
|
||||
if i < ln and value[i] in ('x', 'X'): # '0' followed by 'x': HEX
|
||||
if i < l and value[i] in ('x', 'X'): # '0' followed by 'x': HEX
|
||||
base = 16
|
||||
i += 1
|
||||
else: # just starts with '0': OCT
|
||||
@@ -114,7 +114,7 @@ def strtol(value, strict=True):
|
||||
base = 10
|
||||
|
||||
ret = None
|
||||
while i <= ln:
|
||||
while i <= l:
|
||||
try: # try to find maximally long number
|
||||
i += 1 # by giving to `int` longer and longer strings
|
||||
ret = int(value[:i], base)
|
||||
@@ -280,11 +280,3 @@ def polling_loop(timeout, interval=1):
|
||||
yield iteration
|
||||
iteration += 1
|
||||
time.sleep(interval)
|
||||
|
||||
|
||||
def int_or_none(val):
|
||||
"""Returns integer value of the parameter if convertible to int, None otherwise."""
|
||||
try:
|
||||
return int(val)
|
||||
except (ValueError, TypeError):
|
||||
return None
|
||||
|
||||
+1
-1
@@ -1 +1 @@
|
||||
__version__ = '1.3.6'
|
||||
__version__ = '1.2.5'
|
||||
|
||||
@@ -1,2 +0,0 @@
|
||||
from patroni.watchdog.base import WatchdogError, Watchdog
|
||||
__all__ = ['WatchdogError', 'Watchdog']
|
||||
@@ -1,313 +0,0 @@
|
||||
import abc
|
||||
import logging
|
||||
import platform
|
||||
import six
|
||||
import sys
|
||||
from threading import RLock
|
||||
|
||||
from patroni.exceptions import WatchdogError
|
||||
|
||||
__all__ = ['WatchdogError', 'Watchdog']
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
MODE_REQUIRED = 'required' # Will not run if a watchdog is not available
|
||||
MODE_AUTOMATIC = 'automatic' # Will use a watchdog if one is available
|
||||
MODE_OFF = 'off' # Will not try to use a watchdog
|
||||
|
||||
|
||||
def parse_mode(mode):
|
||||
if mode is False:
|
||||
return MODE_OFF
|
||||
mode = mode.lower()
|
||||
if mode in ['require', 'required']:
|
||||
return MODE_REQUIRED
|
||||
elif mode in ['auto', 'automatic']:
|
||||
return MODE_AUTOMATIC
|
||||
else:
|
||||
if mode not in ['off', 'disable', 'disabled']:
|
||||
logger.warning("Watchdog mode {0} not recognized, disabling watchdog".format(mode))
|
||||
return MODE_OFF
|
||||
|
||||
|
||||
def synchronized(func):
|
||||
def wrapped(self, *args, **kwargs):
|
||||
with self._lock:
|
||||
return func(self, *args, **kwargs)
|
||||
return wrapped
|
||||
|
||||
|
||||
class WatchdogConfig(object):
|
||||
"""Helper to contain a snapshot of configuration"""
|
||||
def __init__(self, config):
|
||||
self.mode = parse_mode(config['watchdog'].get('mode', 'automatic'))
|
||||
self.ttl = config['ttl']
|
||||
self.loop_wait = config['loop_wait']
|
||||
self.safety_margin = config['watchdog'].get('safety_margin', 5)
|
||||
self.driver = config['watchdog'].get('driver', 'default')
|
||||
self.driver_config = dict((k, v) for k, v in config['watchdog'].items()
|
||||
if k not in ['mode', 'safety_margin', 'driver'])
|
||||
|
||||
def __eq__(self, other):
|
||||
return isinstance(other, WatchdogConfig) and \
|
||||
all(getattr(self, attr) == getattr(other, attr) for attr in
|
||||
['mode', 'ttl', 'loop_wait', 'safety_margin', 'driver', 'driver_config'])
|
||||
|
||||
def __ne__(self, other):
|
||||
return not self == other
|
||||
|
||||
def get_impl(self):
|
||||
if self.driver == 'testing':
|
||||
from patroni.watchdog.linux import TestingWatchdogDevice
|
||||
return TestingWatchdogDevice.from_config(self.driver_config)
|
||||
elif platform.system() == 'Linux' and self.driver == 'default':
|
||||
from patroni.watchdog.linux import LinuxWatchdogDevice
|
||||
return LinuxWatchdogDevice.from_config(self.driver_config)
|
||||
else:
|
||||
return NullWatchdog()
|
||||
|
||||
@property
|
||||
def timeout(self):
|
||||
if self.safety_margin == -1:
|
||||
return int(self.ttl // 2)
|
||||
else:
|
||||
return self.ttl - self.safety_margin
|
||||
|
||||
@property
|
||||
def timing_slack(self):
|
||||
return self.timeout - self.loop_wait
|
||||
|
||||
|
||||
class Watchdog(object):
|
||||
"""Facade to dynamically manage watchdog implementations and handle config changes.
|
||||
|
||||
When activation fails underlying implementation will be switched to a Null implementation. To avoid log spam
|
||||
activation will only be retried when watchdog configuration is changed."""
|
||||
def __init__(self, config):
|
||||
self.active_config = self.config = WatchdogConfig(config)
|
||||
self._lock = RLock()
|
||||
self.active = False
|
||||
|
||||
if self.config.mode == MODE_OFF:
|
||||
self.impl = NullWatchdog()
|
||||
else:
|
||||
self.impl = self.config.get_impl()
|
||||
if self.config.mode == MODE_REQUIRED and self.impl.is_null:
|
||||
logger.error("Configuration requires a watchdog, but watchdog is not supported on this platform.")
|
||||
sys.exit(1)
|
||||
|
||||
@synchronized
|
||||
def reload_config(self, config):
|
||||
self.config = WatchdogConfig(config)
|
||||
# Turning a watchdog off can always be done immediately
|
||||
if self.config.mode == MODE_OFF:
|
||||
if self.active:
|
||||
self._disable()
|
||||
self.active_config = self.config
|
||||
self.impl = NullWatchdog()
|
||||
# If watchdog is not active we can apply config immediately to show any warnings early. Otherwise we need to
|
||||
# delay until next time a keepalive is sent so timeout matches up with leader key update.
|
||||
if not self.active:
|
||||
if self.config.driver != self.active_config.driver or \
|
||||
self.config.driver_config != self.active_config.driver_config:
|
||||
self.impl = self.config.get_impl()
|
||||
self.active_config = self.config
|
||||
|
||||
@synchronized
|
||||
def activate(self):
|
||||
"""Activates the watchdog device with suitable timeouts. While watchdog is active keepalive needs
|
||||
to be called every time loop_wait expires.
|
||||
|
||||
:returns False if a safe watchdog could not be configured, but is required.
|
||||
"""
|
||||
self.active = True
|
||||
return self._activate()
|
||||
|
||||
def _activate(self):
|
||||
self.active_config = self.config
|
||||
|
||||
if self.config.timing_slack < 0:
|
||||
logger.warning('Watchdog not supported because leader TTL {0} is less than 2x loop_wait {1}'
|
||||
.format(self.config.ttl, self.config.loop_wait))
|
||||
self.impl = NullWatchdog()
|
||||
|
||||
try:
|
||||
self.impl.open()
|
||||
actual_timeout = self._set_timeout()
|
||||
except WatchdogError as e:
|
||||
logger.warning("Could not activate %s: %s", self.impl.describe(), e)
|
||||
self.impl = NullWatchdog()
|
||||
|
||||
if self.impl.is_running and not self.impl.can_be_disabled:
|
||||
logger.warning("Watchdog implementation can't be disabled."
|
||||
" Watchdog will trigger after Patroni loses leader key.")
|
||||
|
||||
if not self.impl.is_running or actual_timeout > self.config.timeout:
|
||||
if self.config.mode == MODE_REQUIRED:
|
||||
if self.impl.is_null:
|
||||
logger.error("Configuration requires watchdog, but watchdog could not be configured.")
|
||||
else:
|
||||
logger.error("Configuration requires watchdog, but a safe watchdog timeout {0} could"
|
||||
" not be configured. Watchdog timeout is {1}.".format(
|
||||
self.config.timeout, actual_timeout))
|
||||
return False
|
||||
else:
|
||||
if not self.impl.is_null:
|
||||
logger.warning("Watchdog timeout {0} seconds does not ensure safe termination within {1} seconds"
|
||||
.format(actual_timeout, self.config.timeout))
|
||||
|
||||
if self.is_running:
|
||||
logger.info("{0} activated with {1} second timeout, timing slack {2} seconds"
|
||||
.format(self.impl.describe(), actual_timeout, self.config.timing_slack))
|
||||
else:
|
||||
if self.config.mode == MODE_REQUIRED:
|
||||
logger.error("Configuration requires watchdog, but watchdog could not be activated")
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
def _set_timeout(self):
|
||||
if self.impl.has_set_timeout():
|
||||
self.impl.set_timeout(self.config.timeout)
|
||||
|
||||
# Safety checks for watchdog implementations that don't support configurable timeouts
|
||||
actual_timeout = self.impl.get_timeout()
|
||||
if self.impl.is_running and actual_timeout < self.config.loop_wait:
|
||||
logger.error('loop_wait of {0} seconds is too long for watchdog {1} second timeout'
|
||||
.format(self.config.loop_wait, actual_timeout))
|
||||
if self.impl.can_be_disabled:
|
||||
logger.info('Disabling watchdog due to unsafe timeout.')
|
||||
self.impl.close()
|
||||
self.impl = NullWatchdog()
|
||||
return None
|
||||
return actual_timeout
|
||||
|
||||
@synchronized
|
||||
def disable(self):
|
||||
self._disable()
|
||||
self.active = False
|
||||
|
||||
def _disable(self):
|
||||
try:
|
||||
if self.impl.is_running and not self.impl.can_be_disabled:
|
||||
# Give sysadmin some extra time to clean stuff up.
|
||||
self.impl.keepalive()
|
||||
logger.warning("Watchdog implementation can't be disabled. System will reboot after "
|
||||
"{0} seconds when watchdog times out.".format(self.impl.get_timeout()))
|
||||
self.impl.close()
|
||||
except WatchdogError as e:
|
||||
logger.error("Error while disabling watchdog: %s", e)
|
||||
|
||||
@synchronized
|
||||
def keepalive(self):
|
||||
try:
|
||||
if self.active:
|
||||
self.impl.keepalive()
|
||||
# In case there are any pending configuration changes apply them now.
|
||||
if self.active and self.config != self.active_config:
|
||||
if self.config.mode != MODE_OFF and self.active_config.mode == MODE_OFF:
|
||||
self.impl = self.config.get_impl()
|
||||
self._activate()
|
||||
if self.config.driver != self.active_config.driver \
|
||||
or self.config.driver_config != self.active_config.driver_config:
|
||||
self._disable()
|
||||
self.impl = self.config.get_impl()
|
||||
self._activate()
|
||||
if self.config.timeout != self.active_config.timeout:
|
||||
self.impl.set_timeout(self.config.timeout)
|
||||
except WatchdogError as e:
|
||||
logger.error("Error while sending keepalive: %s", e)
|
||||
|
||||
@property
|
||||
@synchronized
|
||||
def is_running(self):
|
||||
return self.impl.is_running
|
||||
|
||||
@property
|
||||
@synchronized
|
||||
def is_healthy(self):
|
||||
if self.config.mode != MODE_REQUIRED:
|
||||
return True
|
||||
return self.config.timing_slack >= 0 and self.impl.is_healthy
|
||||
|
||||
|
||||
@six.add_metaclass(abc.ABCMeta)
|
||||
class WatchdogBase(object):
|
||||
"""A watchdog object when opened requires periodic calls to keepalive.
|
||||
When keepalive is not called within a timeout the system will be terminated."""
|
||||
is_null = False
|
||||
|
||||
@property
|
||||
def is_running(self):
|
||||
"""Returns True when watchdog is activated and capable of performing it's task."""
|
||||
return False
|
||||
|
||||
@property
|
||||
def is_healthy(self):
|
||||
"""Returns False when calling open() is known to fail."""
|
||||
return False
|
||||
|
||||
@property
|
||||
def can_be_disabled(self):
|
||||
"""Returns True when watchdog will be disabled by calling close(). Some watchdog devices
|
||||
will keep running no matter what once activated. May raise WatchdogError if called without
|
||||
calling open() first."""
|
||||
return True
|
||||
|
||||
@abc.abstractmethod
|
||||
def open(self):
|
||||
"""Open watchdog device.
|
||||
|
||||
When watchdog is opened keepalive must be called. Returns nothing on success
|
||||
or raises WatchdogError if the device could not be opened."""
|
||||
|
||||
@abc.abstractmethod
|
||||
def close(self):
|
||||
"""Gracefully close watchdog device."""
|
||||
|
||||
@abc.abstractmethod
|
||||
def keepalive(self):
|
||||
"""Resets the watchdog timer.
|
||||
|
||||
Watchdog must be open when keepalive is called."""
|
||||
|
||||
@abc.abstractmethod
|
||||
def get_timeout(self):
|
||||
"""Returns the current keepalive timeout in effect."""
|
||||
|
||||
@staticmethod
|
||||
def has_set_timeout():
|
||||
"""Returns True if setting a timeout is supported."""
|
||||
return False
|
||||
|
||||
def set_timeout(self, timeout):
|
||||
"""Set the watchdog timer timeout.
|
||||
|
||||
:param timeout: watchdog timeout in seconds"""
|
||||
raise WatchdogError("Setting timeout is not supported on {0}".format(self.describe()))
|
||||
|
||||
def describe(self):
|
||||
"""Human readable name for this device"""
|
||||
return self.__class__.__name__
|
||||
|
||||
@classmethod
|
||||
def from_config(cls, config):
|
||||
return cls()
|
||||
|
||||
|
||||
class NullWatchdog(WatchdogBase):
|
||||
"""Null implementation when watchdog is not supported."""
|
||||
is_null = True
|
||||
|
||||
def open(self):
|
||||
return
|
||||
|
||||
def close(self):
|
||||
return
|
||||
|
||||
def keepalive(self):
|
||||
return
|
||||
|
||||
def get_timeout(self):
|
||||
# A big enough number to not matter
|
||||
return 1000000000
|
||||
@@ -1,234 +0,0 @@
|
||||
import collections
|
||||
import ctypes
|
||||
import fcntl
|
||||
import os
|
||||
import platform
|
||||
from patroni.watchdog.base import WatchdogBase, WatchdogError
|
||||
|
||||
# Pythonification of linux/ioctl.h
|
||||
IOC_NONE = 0
|
||||
IOC_WRITE = 1
|
||||
IOC_READ = 2
|
||||
|
||||
IOC_NRBITS = 8
|
||||
IOC_TYPEBITS = 8
|
||||
IOC_SIZEBITS = 14
|
||||
IOC_DIRBITS = 2
|
||||
|
||||
# Non-generic platform special cases
|
||||
machine = platform.machine()
|
||||
if machine in ['mips', 'sparc', 'powerpc', 'ppc64']:
|
||||
IOC_SIZEBITS = 13
|
||||
IOC_DIRBITS = 3
|
||||
IOC_NONE, IOC_WRITE, IOC_READ = 1, 2, 4
|
||||
elif machine == 'parisc':
|
||||
IOC_WRITE, IOC_READ = 2, 1
|
||||
|
||||
IOC_NRSHIFT = 0
|
||||
IOC_TYPESHIFT = IOC_NRSHIFT + IOC_NRBITS
|
||||
IOC_SIZESHIFT = IOC_TYPESHIFT + IOC_TYPEBITS
|
||||
IOC_DIRSHIFT = IOC_SIZESHIFT + IOC_SIZEBITS
|
||||
|
||||
|
||||
def IOW(type_, nr, size):
|
||||
return IOC(IOC_WRITE, type_, nr, size)
|
||||
|
||||
|
||||
def IOR(type_, nr, size):
|
||||
return IOC(IOC_READ, type_, nr, size)
|
||||
|
||||
|
||||
def IOWR(type_, nr, size):
|
||||
return IOC(IOC_READ | IOC_WRITE, type_, nr, size)
|
||||
|
||||
|
||||
def IOC(dir_, type_, nr, size):
|
||||
return (dir_ << IOC_DIRSHIFT) \
|
||||
| (ord(type_) << IOC_TYPESHIFT) \
|
||||
| (nr << IOC_NRSHIFT) \
|
||||
| (size << IOC_SIZESHIFT)
|
||||
|
||||
|
||||
# Pythonification of linux/watchdog.h
|
||||
|
||||
WATCHDOG_IOCTL_BASE = 'W'
|
||||
|
||||
|
||||
class watchdog_info(ctypes.Structure):
|
||||
_fields_ = [
|
||||
('options', ctypes.c_uint32), # Options the card/driver supports
|
||||
('firmware_version', ctypes.c_uint32), # Firmware version of the card
|
||||
('identity', ctypes.c_uint8 * 32), # Identity of the board
|
||||
]
|
||||
|
||||
|
||||
struct_watchdog_info_size = ctypes.sizeof(watchdog_info)
|
||||
int_size = ctypes.sizeof(ctypes.c_int)
|
||||
|
||||
WDIOC_GETSUPPORT = IOR(WATCHDOG_IOCTL_BASE, 0, struct_watchdog_info_size)
|
||||
WDIOC_GETSTATUS = IOR(WATCHDOG_IOCTL_BASE, 1, int_size)
|
||||
WDIOC_GETBOOTSTATUS = IOR(WATCHDOG_IOCTL_BASE, 2, int_size)
|
||||
WDIOC_GETTEMP = IOR(WATCHDOG_IOCTL_BASE, 3, int_size)
|
||||
WDIOC_SETOPTIONS = IOR(WATCHDOG_IOCTL_BASE, 4, int_size)
|
||||
WDIOC_KEEPALIVE = IOR(WATCHDOG_IOCTL_BASE, 5, int_size)
|
||||
WDIOC_SETTIMEOUT = IOWR(WATCHDOG_IOCTL_BASE, 6, int_size)
|
||||
WDIOC_GETTIMEOUT = IOR(WATCHDOG_IOCTL_BASE, 7, int_size)
|
||||
WDIOC_SETPRETIMEOUT = IOWR(WATCHDOG_IOCTL_BASE, 8, int_size)
|
||||
WDIOC_GETPRETIMEOUT = IOR(WATCHDOG_IOCTL_BASE, 9, int_size)
|
||||
WDIOC_GETTIMELEFT = IOR(WATCHDOG_IOCTL_BASE, 10, int_size)
|
||||
|
||||
|
||||
WDIOF_UNKNOWN = -1 # Unknown flag error
|
||||
WDIOS_UNKNOWN = -1 # Unknown status error
|
||||
|
||||
WDIOF = {
|
||||
"OVERHEAT": 0x0001, # Reset due to CPU overheat
|
||||
"FANFAULT": 0x0002, # Fan failed
|
||||
"EXTERN1": 0x0004, # External relay 1
|
||||
"EXTERN2": 0x0008, # External relay 2
|
||||
"POWERUNDER": 0x0010, # Power bad/power fault
|
||||
"CARDRESET": 0x0020, # Card previously reset the CPU
|
||||
"POWEROVER": 0x0040, # Power over voltage
|
||||
"SETTIMEOUT": 0x0080, # Set timeout (in seconds)
|
||||
"MAGICCLOSE": 0x0100, # Supports magic close char
|
||||
"PRETIMEOUT": 0x0200, # Pretimeout (in seconds), get/set
|
||||
"ALARMONLY": 0x0400, # Watchdog triggers a management or other external alarm not a reboot
|
||||
"KEEPALIVEPING": 0x8000, # Keep alive ping reply
|
||||
}
|
||||
|
||||
WDIOS = {
|
||||
"DISABLECARD": 0x0001, # Turn off the watchdog timer
|
||||
"ENABLECARD": 0x0002, # Turn on the watchdog timer
|
||||
"TEMPPANIC": 0x0004, # Kernel panic on temperature trip
|
||||
}
|
||||
|
||||
# Implementation
|
||||
|
||||
|
||||
class WatchdogInfo(collections.namedtuple('WatchdogInfo', 'options,version,identity')):
|
||||
"""Watchdog descriptor from the kernel"""
|
||||
def __getattr__(self, name):
|
||||
"""Convenience has_XYZ attributes for checking WDIOF bits in options"""
|
||||
if name.startswith('has_') and name[4:] in WDIOF:
|
||||
return bool(self.options & WDIOF[name[4:]])
|
||||
|
||||
raise AttributeError("WatchdogInfo instance has no attribute '{0}'".format(name))
|
||||
|
||||
|
||||
class LinuxWatchdogDevice(WatchdogBase):
|
||||
DEFAULT_DEVICE = '/dev/watchdog'
|
||||
|
||||
def __init__(self, device):
|
||||
self.device = device
|
||||
self._support_cache = None
|
||||
self._fd = None
|
||||
|
||||
@classmethod
|
||||
def from_config(cls, config):
|
||||
device = config.get('device', cls.DEFAULT_DEVICE)
|
||||
return cls(device)
|
||||
|
||||
@property
|
||||
def is_running(self):
|
||||
return self._fd is not None
|
||||
|
||||
@property
|
||||
def is_healthy(self):
|
||||
return os.path.exists(self.device) and os.access(self.device, os.W_OK)
|
||||
|
||||
def open(self):
|
||||
try:
|
||||
self._fd = os.open(self.device, os.O_WRONLY)
|
||||
except OSError as e:
|
||||
raise WatchdogError("Can't open watchdog device: {0}".format(e))
|
||||
|
||||
def close(self):
|
||||
if self.is_running:
|
||||
try:
|
||||
os.write(self._fd, b'V')
|
||||
os.close(self._fd)
|
||||
self._fd = None
|
||||
except OSError as e:
|
||||
raise WatchdogError("Error while closing {0}: {1}".format(self.describe(), e))
|
||||
|
||||
@property
|
||||
def can_be_disabled(self):
|
||||
return self.get_support().has_MAGICCLOSE
|
||||
|
||||
def _ioctl(self, func, arg):
|
||||
"""Runs the specified ioctl on the underlying fd.
|
||||
|
||||
Raises WatchdogError if the device is closed.
|
||||
Raises OSError or IOError (Python 2) when the ioctl fails."""
|
||||
if self._fd is None:
|
||||
raise WatchdogError("Watchdog device is closed")
|
||||
fcntl.ioctl(self._fd, func, arg, True)
|
||||
|
||||
def get_support(self):
|
||||
if self._support_cache is None:
|
||||
info = watchdog_info()
|
||||
try:
|
||||
self._ioctl(WDIOC_GETSUPPORT, info)
|
||||
except (WatchdogError, OSError, IOError) as e:
|
||||
raise WatchdogError("Could not get information about watchdog device: {}".format(e))
|
||||
self._support_cache = WatchdogInfo(info.options,
|
||||
info.firmware_version,
|
||||
bytearray(info.identity).decode(errors='ignore').rstrip('\x00'))
|
||||
return self._support_cache
|
||||
|
||||
def describe(self):
|
||||
dev_str = " at {0}".format(self.device) if self.device != self.DEFAULT_DEVICE else ""
|
||||
ver_str = ""
|
||||
identity = "Linux watchdog device"
|
||||
if self._fd:
|
||||
try:
|
||||
_, version, identity = self.get_support()
|
||||
ver_str = " (firmware {0})".format(version) if version else ""
|
||||
except WatchdogError:
|
||||
pass
|
||||
|
||||
return identity + ver_str + dev_str
|
||||
|
||||
def keepalive(self):
|
||||
try:
|
||||
os.write(self._fd, b'1')
|
||||
except OSError as e:
|
||||
raise WatchdogError("Could not send watchdog keepalive: {0}".format(e))
|
||||
|
||||
def has_set_timeout(self):
|
||||
"""Returns True if setting a timeout is supported."""
|
||||
return self.get_support().has_SETTIMEOUT
|
||||
|
||||
def set_timeout(self, timeout):
|
||||
timeout = int(timeout)
|
||||
if not 0 < timeout < 0xFFFF:
|
||||
raise WatchdogError("Invalid timeout {0}. Supported values are between 1 and 65535".format(timeout))
|
||||
try:
|
||||
self._ioctl(WDIOC_SETTIMEOUT, ctypes.c_int(timeout))
|
||||
except (WatchdogError, OSError, IOError) as e:
|
||||
raise WatchdogError("Could not set timeout on watchdog device: {}".format(e))
|
||||
|
||||
def get_timeout(self):
|
||||
timeout = ctypes.c_int()
|
||||
try:
|
||||
self._ioctl(WDIOC_GETTIMEOUT, timeout)
|
||||
except (WatchdogError, OSError, IOError) as e:
|
||||
raise WatchdogError("Could not get timeout on watchdog device: {}".format(e))
|
||||
return timeout.value
|
||||
|
||||
|
||||
class TestingWatchdogDevice(LinuxWatchdogDevice):
|
||||
"""Converts timeout ioctls to regular writes that can be intercepted from a named pipe."""
|
||||
timeout = 60
|
||||
|
||||
def get_support(self):
|
||||
return WatchdogInfo(WDIOF['MAGICCLOSE'] | WDIOF['SETTIMEOUT'], 0, "Watchdog test harness")
|
||||
|
||||
def set_timeout(self, timeout):
|
||||
buf = "Ctimeout={0}\n".format(timeout).encode('utf8')
|
||||
while len(buf):
|
||||
buf = buf[os.write(self._fd, buf):]
|
||||
self.timeout = timeout
|
||||
|
||||
def get_timeout(self):
|
||||
return self.timeout
|
||||
@@ -66,7 +66,6 @@ postgresql:
|
||||
connect_address: 127.0.0.1:5432
|
||||
data_dir: data/postgresql0
|
||||
# bin_dir:
|
||||
# config_dir:
|
||||
pgpass: /tmp/pgpass0
|
||||
authentication:
|
||||
replication:
|
||||
@@ -77,12 +76,6 @@ postgresql:
|
||||
password: zalando
|
||||
parameters:
|
||||
unix_socket_directories: '.'
|
||||
|
||||
#watchdog:
|
||||
# mode: automatic # Allowed values: off, automatic, required
|
||||
# device: /dev/watchdog
|
||||
# safety_margin: 5
|
||||
|
||||
tags:
|
||||
nofailover: false
|
||||
noloadbalance: false
|
||||
|
||||
@@ -64,7 +64,6 @@ postgresql:
|
||||
connect_address: 127.0.0.1:5433
|
||||
data_dir: data/postgresql1
|
||||
# bin_dir:
|
||||
# config_dir:
|
||||
pgpass: /tmp/pgpass1
|
||||
authentication:
|
||||
replication:
|
||||
|
||||
@@ -61,7 +61,6 @@ postgresql:
|
||||
connect_address: 127.0.0.1:5434
|
||||
data_dir: data/postgresql2
|
||||
# bin_dir:
|
||||
# config_dir:
|
||||
pgpass: /tmp/pgpass2
|
||||
authentication:
|
||||
replication:
|
||||
|
||||
+2
-4
@@ -5,11 +5,9 @@ PyYAML
|
||||
requests
|
||||
six >= 1.7
|
||||
kazoo==2.2.1
|
||||
python-etcd>=0.4.3,<0.5
|
||||
python-consul>=0.7.0
|
||||
python-etcd==0.4.3
|
||||
python-consul==0.7.0
|
||||
click>=4.1
|
||||
prettytable>=0.7
|
||||
tzlocal
|
||||
python-dateutil
|
||||
psutil
|
||||
cdiff
|
||||
|
||||
@@ -87,7 +87,7 @@ class PyTest(TestCommand):
|
||||
def run_tests(self):
|
||||
try:
|
||||
import pytest
|
||||
except Exception:
|
||||
except:
|
||||
raise RuntimeError('py.test is not installed, run: pip install pytest')
|
||||
params = {'args': self.test_args}
|
||||
if self.cov:
|
||||
|
||||
+5
-18
@@ -3,10 +3,9 @@ import json
|
||||
import psycopg2
|
||||
import unittest
|
||||
|
||||
from mock import Mock, PropertyMock, patch
|
||||
from mock import Mock, patch
|
||||
from patroni.api import RestApiHandler, RestApiServer
|
||||
from patroni.dcs import ClusterConfig, Member
|
||||
from patroni.ha import _MemberStatus
|
||||
from patroni.utils import tzutc
|
||||
from six import BytesIO as IO
|
||||
from six.moves import BaseHTTPServer
|
||||
@@ -26,8 +25,6 @@ class MockPostgresql(object):
|
||||
sysid = 'dummysysid'
|
||||
scope = 'dummy'
|
||||
pending_restart = True
|
||||
wal_name = 'wal'
|
||||
lsn_name = 'lsn'
|
||||
|
||||
@staticmethod
|
||||
def connection():
|
||||
@@ -38,14 +35,9 @@ class MockPostgresql(object):
|
||||
return str(postmaster_start_time)
|
||||
|
||||
|
||||
class MockWatchdog(object):
|
||||
is_healthy = False
|
||||
|
||||
|
||||
class MockHa(object):
|
||||
|
||||
state_handler = MockPostgresql()
|
||||
watchdog = MockWatchdog()
|
||||
|
||||
@staticmethod
|
||||
def reinitialize():
|
||||
@@ -65,14 +57,14 @@ class MockHa(object):
|
||||
|
||||
@staticmethod
|
||||
def fetch_nodes_statuses(members):
|
||||
return [_MemberStatus(None, True, None, None, {}, False)]
|
||||
return [[None, True, None, None, {}]]
|
||||
|
||||
@staticmethod
|
||||
def schedule_future_restart(data):
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def is_lagging(wal):
|
||||
def is_lagging(xlog):
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
@@ -112,7 +104,6 @@ class MockRequest(object):
|
||||
def sendall(self, *args, **kwargs):
|
||||
pass
|
||||
|
||||
|
||||
class MockRestApiServer(RestApiServer):
|
||||
|
||||
def __init__(self, Handler, request):
|
||||
@@ -152,7 +143,6 @@ class TestRestApiHandler(unittest.TestCase):
|
||||
def test_do_OPTIONS(self):
|
||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'OPTIONS / HTTP/1.0'))
|
||||
|
||||
@patch.object(MockPostgresql, 'state', PropertyMock(return_value='stopped'))
|
||||
def test_do_GET_patroni(self):
|
||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /patroni'))
|
||||
|
||||
@@ -281,7 +271,6 @@ class TestRestApiHandler(unittest.TestCase):
|
||||
def test_do_POST_failover(self, dcs):
|
||||
dcs.loop_wait = 10
|
||||
cluster = dcs.get_cluster.return_value
|
||||
cluster.is_synchronous_mode.return_value = False
|
||||
|
||||
post = 'POST /failover HTTP/1.0' + self._authorization + '\nContent-Length: '
|
||||
|
||||
@@ -293,16 +282,14 @@ class TestRestApiHandler(unittest.TestCase):
|
||||
cluster.leader.name = 'postgresql1'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
|
||||
for cluster.is_synchronous_mode.return_value in (True, False):
|
||||
MockRestApiServer(RestApiHandler, post + '25\n\n{"leader": "postgresql1"}')
|
||||
MockRestApiServer(RestApiHandler, post + '25\n\n{"leader": "postgresql1"}')
|
||||
|
||||
cluster.leader.name = 'postgresql2'
|
||||
request = post + '53\n\n{"leader": "postgresql1", "candidate": "postgresql2"}'
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
|
||||
cluster.leader.name = 'postgresql1'
|
||||
for cluster.is_synchronous_mode.return_value in (True, False):
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
MockRestApiServer(RestApiHandler, request)
|
||||
|
||||
cluster.members = [Member(0, 'postgresql0', 30, {'api_url': 'http'}),
|
||||
Member(0, 'postgresql2', 30, {'api_url': 'http'})]
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import unittest
|
||||
|
||||
from mock import Mock, patch
|
||||
from patroni.async_executor import AsyncExecutor, CriticalTask
|
||||
from patroni.async_executor import AsyncExecutor
|
||||
from threading import Thread
|
||||
|
||||
|
||||
@@ -16,11 +16,3 @@ class TestAsyncExecutor(unittest.TestCase):
|
||||
|
||||
def test_run(self):
|
||||
self.a.run(Mock(side_effect=Exception()))
|
||||
|
||||
|
||||
class TestCriticalTask(unittest.TestCase):
|
||||
|
||||
def test_completed_task(self):
|
||||
ct = CriticalTask()
|
||||
ct.complete(1)
|
||||
self.assertFalse(ct.cancel())
|
||||
|
||||
+37
-23
@@ -1,69 +1,83 @@
|
||||
import boto.ec2
|
||||
import requests
|
||||
import sys
|
||||
import unittest
|
||||
|
||||
from mock import Mock, patch
|
||||
from collections import namedtuple
|
||||
from patroni.scripts.aws import AWSConnection, main as _main
|
||||
from patroni.utils import RetryFailedError
|
||||
from requests.exceptions import RequestException
|
||||
|
||||
|
||||
class MockEc2Connection(object):
|
||||
|
||||
@staticmethod
|
||||
def get_all_volumes(*args, **kwargs):
|
||||
def __init__(self, error=False):
|
||||
self.error = error
|
||||
|
||||
def get_all_volumes(self, filters):
|
||||
if self.error:
|
||||
raise boto.exception("get_all_volumes")
|
||||
oid = namedtuple('Volume', 'id')
|
||||
return [oid(id='a'), oid(id='b')]
|
||||
|
||||
@staticmethod
|
||||
def create_tags(objects, *args, **kwargs):
|
||||
if len(objects) == 0:
|
||||
raise boto.exception.BotoServerError(503, 'Service Unavailable', 'Request limit exceeded')
|
||||
def create_tags(self, objects, tags):
|
||||
if self.error or len(objects) == 0:
|
||||
raise boto.exception("create_tags")
|
||||
return True
|
||||
|
||||
|
||||
class MockResponse(object):
|
||||
ok = True
|
||||
|
||||
def __init__(self, content):
|
||||
self.content = content
|
||||
self.ok = True
|
||||
|
||||
def json(self):
|
||||
return self.content
|
||||
|
||||
|
||||
def requests_get(url, **kwargs):
|
||||
if url.split('/')[-1] == 'document':
|
||||
result = {"instanceId": "012345", "region": "eu-west-1"}
|
||||
else:
|
||||
result = 'foo'
|
||||
return MockResponse(result)
|
||||
|
||||
|
||||
@patch('boto.ec2.connect_to_region', Mock(return_value=MockEc2Connection()))
|
||||
class TestAWSConnection(unittest.TestCase):
|
||||
|
||||
@patch('requests.get', requests_get)
|
||||
def boto_ec2_connect_to_region(self, region):
|
||||
return MockEc2Connection(self.error)
|
||||
|
||||
def requests_get(self, url, **kwargs):
|
||||
if self.error:
|
||||
raise RequestException("foo")
|
||||
result = namedtuple('Request', 'ok content')
|
||||
result.ok = True
|
||||
if url.split('/')[-1] == 'document' and not self.json_error:
|
||||
result = {"instanceId": "012345", "region": "eu-west-1"}
|
||||
else:
|
||||
result = 'foo'
|
||||
return MockResponse(result)
|
||||
|
||||
def setUp(self):
|
||||
self.error = False
|
||||
self.json_error = False
|
||||
requests.get = self.requests_get
|
||||
boto.ec2.connect_to_region = self.boto_ec2_connect_to_region
|
||||
self.conn = AWSConnection('test')
|
||||
|
||||
def test_aws_available(self):
|
||||
self.assertTrue(self.conn.aws_available())
|
||||
|
||||
def test_on_role_change(self):
|
||||
self.assertTrue(self.conn.on_role_change('master'))
|
||||
with patch.object(MockEc2Connection, 'get_all_volumes', Mock(return_value=[])):
|
||||
self.conn._retry.max_tries = 1
|
||||
self.assertFalse(self.conn.on_role_change('master'))
|
||||
self.conn.retry = Mock(side_effect=RetryFailedError("retry failed"))
|
||||
self.assertFalse(self.conn.on_role_change('master'))
|
||||
|
||||
@patch('requests.get', Mock(side_effect=RequestException('foo')))
|
||||
def test_non_aws(self):
|
||||
self.error = True
|
||||
conn = AWSConnection('test')
|
||||
self.assertFalse(conn.on_role_change("master"))
|
||||
|
||||
@patch('requests.get', Mock(return_value=MockResponse('foo')))
|
||||
def test_aws_bizare_response(self):
|
||||
self.json_error = True
|
||||
conn = AWSConnection('test')
|
||||
self.assertFalse(conn.aws_available())
|
||||
|
||||
@patch('requests.get', requests_get)
|
||||
@patch('sys.exit', Mock())
|
||||
def test_main(self):
|
||||
self.assertIsNone(_main())
|
||||
|
||||
@@ -39,7 +39,6 @@ class TestConfig(unittest.TestCase):
|
||||
'PATRONI_POSTGRESQL_LISTEN': '0.0.0.0:5432',
|
||||
'PATRONI_POSTGRESQL_CONNECT_ADDRESS': '127.0.0.1:5432',
|
||||
'PATRONI_POSTGRESQL_DATA_DIR': 'data/postgres0',
|
||||
'PATRONI_POSTGRESQL_CONFIG_DIR': 'data/postgres0',
|
||||
'PATRONI_POSTGRESQL_PGPASS': '/tmp/pgpass0',
|
||||
'PATRONI_ETCD_HOST': '127.0.0.1:2379',
|
||||
'PATRONI_ETCD_URL': 'https://127.0.0.1:2379',
|
||||
|
||||
+3
-13
@@ -3,8 +3,7 @@ import unittest
|
||||
|
||||
from consul import ConsulException, NotFound
|
||||
from mock import Mock, patch
|
||||
from patroni.dcs.consul import AbstractDCS, Cluster, Consul, ConsulInternalError, \
|
||||
ConsulError, HTTPClient, InvalidSessionTTL
|
||||
from patroni.dcs.consul import AbstractDCS, Cluster, Consul, ConsulInternalError, ConsulError, HTTPClient
|
||||
from test_etcd import SleepException
|
||||
|
||||
|
||||
@@ -46,12 +45,9 @@ class TestHTTPClient(unittest.TestCase):
|
||||
|
||||
def test_get(self):
|
||||
self.client.get(Mock(), '')
|
||||
self.client.get(Mock(), '', {'wait': '1s', 'index': 1, 'token': 'foo'})
|
||||
self.client.get(Mock(), '', {'wait': '1s', 'index': 1})
|
||||
self.client.http.request.return_value.status = 500
|
||||
self.client.http.request.return_value.data = b'Foo'
|
||||
self.assertRaises(ConsulInternalError, self.client.get, Mock(), '')
|
||||
self.client.http.request.return_value.data = b"Invalid Session TTL '3000000000', must be between [10s=24h0m0s]"
|
||||
self.assertRaises(InvalidSessionTTL, self.client.get, Mock(), '')
|
||||
|
||||
def test_unknown_method(self):
|
||||
try:
|
||||
@@ -73,10 +69,6 @@ class TestConsul(unittest.TestCase):
|
||||
@patch.object(consul.Consul.KV, 'get', kv_get)
|
||||
@patch.object(consul.Consul.KV, 'delete', Mock())
|
||||
def setUp(self):
|
||||
Consul({'ttl': 30, 'scope': 't', 'name': 'p', 'url': 'https://l:1', 'retry_timeout': 10,
|
||||
'verify': 'on', 'key': 'foo', 'cert': 'bar', 'cacert': 'buz', 'token': 'asd', 'dc': 'dc1'})
|
||||
Consul({'ttl': 30, 'scope': 't', 'name': 'p', 'url': 'https://l:1', 'retry_timeout': 10,
|
||||
'verify': 'on', 'cert': 'bar', 'cacert': 'buz'})
|
||||
self.c = Consul({'ttl': 30, 'scope': 'test', 'name': 'postgresql1', 'host': 'localhost:1', 'retry_timeout': 10})
|
||||
self.c._base_path = '/service/good'
|
||||
self.c._load_cluster()
|
||||
@@ -88,9 +80,7 @@ class TestConsul(unittest.TestCase):
|
||||
self.assertRaises(SleepException, self.c.create_session)
|
||||
|
||||
@patch.object(consul.Consul.Session, 'renew', Mock(side_effect=NotFound))
|
||||
@patch.object(consul.Consul.Session, 'create', Mock(side_effect=[InvalidSessionTTL, ConsulException]))
|
||||
@patch.object(consul.Consul.Agent, 'self', Mock(return_value={'Config': {'SessionTTLMin': 0}}))
|
||||
@patch.object(HTTPClient, 'set_ttl', Mock(side_effect=ValueError))
|
||||
@patch.object(consul.Consul.Session, 'create', Mock(side_effect=ConsulException))
|
||||
def test_referesh_session(self):
|
||||
self.c._session = '1'
|
||||
self.assertFalse(self.c.refresh_session())
|
||||
|
||||
+1
-69
@@ -7,8 +7,7 @@ import unittest
|
||||
from click.testing import CliRunner
|
||||
from mock import patch, Mock
|
||||
from patroni.ctl import ctl, members, store_config, load_config, output_members, request_patroni, get_dcs, parse_dcs, \
|
||||
get_all_members, get_any_member, get_cursor, query_member, configure, PatroniCtlException, apply_config_changes, \
|
||||
format_config_for_editing, show_diff, invoke_editor
|
||||
get_all_members, get_any_member, get_cursor, query_member, configure, PatroniCtlException
|
||||
from patroni.dcs.etcd import Client
|
||||
from psycopg2 import OperationalError
|
||||
from test_etcd import etcd_read, requests_get, socket_getaddrinfo, MockResponse
|
||||
@@ -439,70 +438,3 @@ class TestCtl(unittest.TestCase):
|
||||
with patch('requests.patch', Mock(side_effect=Exception)):
|
||||
result = self.runner.invoke(ctl, ['resume', 'dummy'])
|
||||
assert 'Can not find accessible cluster member' in result.output
|
||||
|
||||
def test_apply_config_changes(self):
|
||||
config = {"postgresql": {"parameters": {"work_mem": "4MB"}, "use_pg_rewind": True}, "ttl": 30}
|
||||
|
||||
before_editing = format_config_for_editing(config)
|
||||
|
||||
# Spaces are allowed and stripped, numbers and booleans are interpreted
|
||||
after_editing, changed_config = apply_config_changes(before_editing, config,
|
||||
["postgresql.parameters.work_mem = 5MB",
|
||||
"ttl=15", "postgresql.use_pg_rewind=off", 'a.b=c'])
|
||||
self.assertEquals(changed_config, {"a": {"b": "c"}, "postgresql": {"parameters": {"work_mem": "5MB"},
|
||||
"use_pg_rewind": False}, "ttl": 15})
|
||||
|
||||
# postgresql.parameters namespace is flattened
|
||||
after_editing, changed_config = apply_config_changes(before_editing, config,
|
||||
["postgresql.parameters.work_mem.sub = x"])
|
||||
self.assertEquals(changed_config, {"postgresql": {"parameters": {"work_mem": "4MB", "work_mem.sub": "x"},
|
||||
"use_pg_rewind": True}, "ttl": 30})
|
||||
|
||||
# Setting to null deletes
|
||||
after_editing, changed_config = apply_config_changes(before_editing, config,
|
||||
["postgresql.parameters.work_mem=null"])
|
||||
self.assertEquals(changed_config, {"postgresql": {"use_pg_rewind": True}, "ttl": 30})
|
||||
after_editing, changed_config = apply_config_changes(before_editing, config,
|
||||
["postgresql.use_pg_rewind=null",
|
||||
"postgresql.parameters.work_mem=null"])
|
||||
self.assertEquals(changed_config, {"ttl": 30})
|
||||
|
||||
self.assertRaises(PatroniCtlException, apply_config_changes, before_editing, config, ['a'])
|
||||
|
||||
@patch('sys.stdout.isatty', return_value=False)
|
||||
@patch('cdiff.markup_to_pager')
|
||||
def test_show_diff(self, mock_markup_to_pager, mock_isatty):
|
||||
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
|
||||
mock_markup_to_pager.assert_not_called()
|
||||
|
||||
mock_isatty.return_value = True
|
||||
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
|
||||
mock_markup_to_pager.assert_called_once()
|
||||
|
||||
# Test that unicode handling doesn't fail with an exception
|
||||
show_diff(b"foo:\n bar: \xc3\xb6\xc3\xb6\n".decode('utf-8'),
|
||||
b"foo:\n bar: \xc3\xbc\xc3\xbc\n".decode('utf-8'))
|
||||
|
||||
def test_invoke_editor(self):
|
||||
for e in ('', 'false'):
|
||||
os.environ['EDITOR'] = e
|
||||
self.assertRaises(PatroniCtlException, invoke_editor, 'foo: bar\n', 'test')
|
||||
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
def test_show_config(self, mock_get_dcs):
|
||||
mock_get_dcs.return_value = self.e
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
self.runner.invoke(ctl, ['show-config', 'dummy'])
|
||||
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
def test_edit_config(self, mock_get_dcs):
|
||||
mock_get_dcs.return_value = self.e
|
||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||
os.environ['EDITOR'] = 'true'
|
||||
self.runner.invoke(ctl, ['edit-config', 'dummy'])
|
||||
self.runner.invoke(ctl, ['edit-config', 'dummy', '-s', 'foo=bar'])
|
||||
self.runner.invoke(ctl, ['edit-config', 'dummy', '--replace', 'postgres0.yml'])
|
||||
self.runner.invoke(ctl, ['edit-config', 'dummy', '--apply', '-'], input='foo: bar')
|
||||
self.runner.invoke(ctl, ['edit-config', 'dummy', '--force', '--apply', '-'], input='foo: bar')
|
||||
mock_get_dcs.return_value.set_config_value = Mock(return_value=True)
|
||||
self.runner.invoke(ctl, ['edit-config', 'dummy', '--force', '--apply', '-'], input='foo: bar')
|
||||
|
||||
+40
-132
@@ -7,10 +7,9 @@ from mock import Mock, MagicMock, PropertyMock, patch
|
||||
from patroni.config import Config
|
||||
from patroni.dcs import Cluster, ClusterConfig, Failover, Leader, Member, get_dcs, SyncState
|
||||
from patroni.dcs.etcd import Client
|
||||
from patroni.exceptions import DCSError, PostgresConnectionException, PatroniException
|
||||
from patroni.exceptions import DCSError, PostgresException
|
||||
from patroni.ha import Ha, _MemberStatus
|
||||
from patroni.postgresql import Postgresql
|
||||
from patroni.watchdog import Watchdog
|
||||
from patroni.utils import tzutc
|
||||
from test_etcd import socket_getaddrinfo, etcd_read, etcd_write, requests_get
|
||||
from test_postgresql import psycopg2_connect
|
||||
@@ -35,7 +34,7 @@ def get_cluster_not_initialized_without_leader():
|
||||
def get_cluster_initialized_without_leader(leader=False, failover=None, sync=None):
|
||||
m1 = Member(0, 'leader', 28, {'conn_url': 'postgres://replicator:[email protected]:5435/postgres',
|
||||
'api_url': 'http://127.0.0.1:8008/patroni', 'xlog_location': 4})
|
||||
leader = Leader(0, 0, m1) if leader else None
|
||||
l = Leader(0, 0, m1) if leader else None
|
||||
m2 = Member(0, 'other', 28, {'conn_url': 'postgres://replicator:[email protected]:5436/postgres',
|
||||
'api_url': 'http://127.0.0.1:8011/patroni',
|
||||
'state': 'running',
|
||||
@@ -43,7 +42,7 @@ def get_cluster_initialized_without_leader(leader=False, failover=None, sync=Non
|
||||
'scheduled_restart': {'schedule': "2100-01-01 10:53:07.560445+00:00",
|
||||
'postgres_version': '99.0.0'}})
|
||||
syncstate = SyncState(0 if sync else None, sync and sync[0], sync and sync[1])
|
||||
return get_cluster(True, leader, [m1, m2], failover, syncstate)
|
||||
return get_cluster(True, l, [m1, m2], failover, syncstate)
|
||||
|
||||
|
||||
def get_cluster_initialized_with_leader(failover=None, sync=None):
|
||||
@@ -51,19 +50,18 @@ def get_cluster_initialized_with_leader(failover=None, sync=None):
|
||||
|
||||
|
||||
def get_cluster_initialized_with_only_leader(failover=None):
|
||||
leader = get_cluster_initialized_without_leader(leader=True, failover=failover).leader
|
||||
return get_cluster(True, leader, [leader], failover, None)
|
||||
l = get_cluster_initialized_without_leader(leader=True, failover=failover).leader
|
||||
return get_cluster(True, l, [l], failover, None)
|
||||
|
||||
|
||||
def get_node_status(reachable=True, in_recovery=True, wal_position=10, nofailover=False, watchdog_failed=False):
|
||||
def get_node_status(reachable=True, in_recovery=True, xlog_location=10, nofailover=False):
|
||||
def fetch_node_status(e):
|
||||
tags = {}
|
||||
if nofailover:
|
||||
tags['nofailover'] = True
|
||||
return _MemberStatus(e, reachable, in_recovery, wal_position, tags, watchdog_failed)
|
||||
return _MemberStatus(e, reachable, in_recovery, xlog_location, tags)
|
||||
return fetch_node_status
|
||||
|
||||
|
||||
future_restart_time = datetime.datetime.now(tzutc) + datetime.timedelta(days=5)
|
||||
postmaster_start_time = datetime.datetime.now(tzutc)
|
||||
|
||||
@@ -86,8 +84,6 @@ postgresql:
|
||||
pg_rewind:
|
||||
username: postgres
|
||||
password: postgres
|
||||
watchdog:
|
||||
mode: off
|
||||
zookeeper:
|
||||
exhibitor:
|
||||
hosts: [localhost]
|
||||
@@ -105,7 +101,6 @@ zookeeper:
|
||||
self.nosync = False
|
||||
self.scheduled_restart = {'schedule': future_restart_time,
|
||||
'postmaster_start_time': str(postmaster_start_time)}
|
||||
self.watchdog = Watchdog(self.config)
|
||||
|
||||
|
||||
def run_async(self, func, args=()):
|
||||
@@ -114,13 +109,13 @@ def run_async(self, func, args=()):
|
||||
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'wal_position', Mock(return_value=10))
|
||||
@patch.object(Postgresql, 'xlog_position', Mock(return_value=10))
|
||||
@patch.object(Postgresql, 'call_nowait', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'data_directory_empty', Mock(return_value=False))
|
||||
@patch.object(Postgresql, 'controldata', Mock(return_value={'Database system identifier': '1234567890'}))
|
||||
@patch.object(Postgresql, 'sync_replication_slots', Mock())
|
||||
@patch.object(Postgresql, 'write_pg_hba', Mock())
|
||||
@patch.object(Postgresql, 'write_pgpass', Mock(return_value={}))
|
||||
@patch.object(Postgresql, 'write_pgpass', Mock())
|
||||
@patch.object(Postgresql, 'write_recovery_conf', Mock())
|
||||
@patch.object(Postgresql, 'query', Mock())
|
||||
@patch.object(Postgresql, 'checkpoint', Mock())
|
||||
@@ -131,12 +126,10 @@ def run_async(self, func, args=()):
|
||||
@patch('patroni.async_executor.AsyncExecutor.busy', PropertyMock(return_value=False))
|
||||
@patch('patroni.async_executor.AsyncExecutor.run_async', run_async)
|
||||
@patch('subprocess.call', Mock(return_value=0))
|
||||
@patch('time.sleep', Mock())
|
||||
class TestHa(unittest.TestCase):
|
||||
|
||||
@patch('socket.getaddrinfo', socket_getaddrinfo)
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.dcs.dcs_modules', Mock(return_value=['patroni.dcs.foo', 'patroni.dcs.etcd']))
|
||||
@patch.object(etcd.Client, 'read', etcd_read)
|
||||
def setUp(self):
|
||||
with patch.object(Client, 'machines') as mock_machines:
|
||||
@@ -158,13 +151,14 @@ class TestHa(unittest.TestCase):
|
||||
self.ha.old_cluster = self.e.get_cluster()
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.ha.load_cluster_from_dcs = Mock()
|
||||
self.ha.is_synchronous_mode = false
|
||||
|
||||
def test_update_lock(self):
|
||||
self.p.last_operation = Mock(side_effect=PostgresConnectionException(''))
|
||||
self.p.last_operation = Mock(side_effect=PostgresException(''))
|
||||
self.assertTrue(self.ha.update_lock(True))
|
||||
|
||||
def test_touch_member(self):
|
||||
self.p.wal_position = Mock(side_effect=Exception)
|
||||
self.p.xlog_position = Mock(side_effect=Exception)
|
||||
self.ha.touch_member()
|
||||
|
||||
def test_start_as_replica(self):
|
||||
@@ -172,42 +166,23 @@ class TestHa(unittest.TestCase):
|
||||
self.assertEquals(self.ha.run_cycle(), 'starting as a secondary')
|
||||
|
||||
def test_recover_replica_failed(self):
|
||||
self.p.controldata = lambda: {'Database cluster state': 'in recovery'}
|
||||
self.p.controldata = lambda: {'Database cluster state': 'in production'}
|
||||
self.p.is_healthy = false
|
||||
self.p.is_running = false
|
||||
self.p.follow = false
|
||||
self.assertEquals(self.ha.run_cycle(), 'starting as a secondary')
|
||||
self.assertEquals(self.ha.run_cycle(), 'failed to start postgres')
|
||||
|
||||
def test_recover_former_master(self):
|
||||
def test_recover_master_failed(self):
|
||||
self.p.follow = false
|
||||
self.p.is_healthy = false
|
||||
self.p.is_running = false
|
||||
self.p.name = 'leader'
|
||||
self.p.set_role('master')
|
||||
self.p.controldata = lambda: {'Database cluster state': 'shut down'}
|
||||
self.p.controldata = lambda: {'Database cluster state': 'in production'}
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.assertEquals(self.ha.run_cycle(), 'starting as readonly because i had the session lock')
|
||||
|
||||
@patch.object(Postgresql, 'fix_cluster_state', Mock())
|
||||
def test_crash_recovery(self):
|
||||
self.p.is_running = false
|
||||
self.p.controldata = lambda: {'Database cluster state': 'in production'}
|
||||
self.assertEquals(self.ha.run_cycle(), 'doing crash recovery in a single user mode')
|
||||
|
||||
@patch.object(Postgresql, 'rewind_needed_and_possible', Mock(return_value=True))
|
||||
def test_recover_with_rewind(self):
|
||||
self.p.is_running = false
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.assertEquals(self.ha.run_cycle(), 'running pg_rewind from leader')
|
||||
|
||||
@patch.object(Postgresql, 'can_rewind', PropertyMock(return_value=True))
|
||||
@patch.object(Postgresql, 'fix_cluster_state', Mock())
|
||||
def test_single_user_after_recover_failed(self):
|
||||
self.p.controldata = lambda: {'Database cluster state': 'in recovery'}
|
||||
self.p.is_running = false
|
||||
self.p.follow = false
|
||||
self.assertEquals(self.ha.run_cycle(), 'starting as a secondary')
|
||||
self.assertEquals(self.ha.run_cycle(), 'fixing cluster state in a single user mode')
|
||||
|
||||
@patch('sys.exit', return_value=1)
|
||||
@patch('patroni.ha.Ha.sysid_valid', MagicMock(return_value=True))
|
||||
def test_sysid_no_match(self, exit_mock):
|
||||
@@ -254,15 +229,6 @@ class TestHa(unittest.TestCase):
|
||||
self.p.is_leader = false
|
||||
self.assertEquals(self.ha.run_cycle(), 'promoted self to leader because i had the session lock')
|
||||
|
||||
def test_promote_without_watchdog(self):
|
||||
self.ha.cluster.is_unlocked = false
|
||||
self.ha.has_lock = true
|
||||
self.p.is_leader = true
|
||||
with patch.object(Watchdog, 'activate', Mock(return_value=False)):
|
||||
self.assertEquals(self.ha.run_cycle(), 'Demoting self because watchdog could not be activated')
|
||||
self.p.is_leader = false
|
||||
self.assertEquals(self.ha.run_cycle(), 'Not promoting self because watchdog could not be activated')
|
||||
|
||||
def test_leader_with_lock(self):
|
||||
self.ha.cluster.is_unlocked = false
|
||||
self.ha.has_lock = true
|
||||
@@ -270,8 +236,7 @@ class TestHa(unittest.TestCase):
|
||||
|
||||
def test_demote_because_not_having_lock(self):
|
||||
self.ha.cluster.is_unlocked = false
|
||||
with patch.object(Watchdog, 'is_running', PropertyMock(return_value=True)):
|
||||
self.assertEquals(self.ha.run_cycle(), 'demoting self because i do not have the lock and i was a leader')
|
||||
self.assertEquals(self.ha.run_cycle(), 'demoting self because i do not have the lock and i was a leader')
|
||||
|
||||
def test_demote_because_update_lock_failed(self):
|
||||
self.ha.cluster.is_unlocked = false
|
||||
@@ -295,18 +260,10 @@ class TestHa(unittest.TestCase):
|
||||
self.p.is_leader = false
|
||||
self.assertEquals(self.ha.run_cycle(), 'PAUSE: no action')
|
||||
|
||||
@patch.object(Postgresql, 'rewind_needed_and_possible', Mock(return_value=True))
|
||||
def test_follow_triggers_rewind(self):
|
||||
self.p.is_leader = false
|
||||
self.p.trigger_check_diverged_lsn()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.assertEquals(self.ha.run_cycle(), 'running pg_rewind from leader')
|
||||
|
||||
def test_no_etcd_connection_master_demote(self):
|
||||
self.ha.load_cluster_from_dcs = Mock(side_effect=DCSError('Etcd is not responding properly'))
|
||||
self.assertEquals(self.ha.run_cycle(), 'demoted self because DCS is not accessible and i was a leader')
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
def test_bootstrap_from_another_member(self):
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.assertEquals(self.ha.bootstrap(), 'trying to bootstrap from replica \'other\'')
|
||||
@@ -327,31 +284,14 @@ class TestHa(unittest.TestCase):
|
||||
def test_bootstrap_initialized_new_cluster(self):
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.e.initialize = true
|
||||
self.assertEquals(self.ha.bootstrap(), 'trying to bootstrap a new cluster')
|
||||
self.p.is_leader = false
|
||||
self.assertEquals(self.ha.run_cycle(), 'waiting for end of recovery after bootstrap')
|
||||
self.p.is_leader = true
|
||||
self.assertEquals(self.ha.run_cycle(), 'running post_bootstrap')
|
||||
self.assertEquals(self.ha.run_cycle(), 'initialized a new cluster')
|
||||
self.assertEquals(self.ha.bootstrap(), 'initialized a new cluster')
|
||||
|
||||
def test_bootstrap_release_initialize_key_on_failure(self):
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.e.initialize = true
|
||||
self.ha.bootstrap()
|
||||
self.p.is_running = false
|
||||
self.assertRaises(PatroniException, self.ha.post_bootstrap)
|
||||
self.p.bootstrap = Mock(side_effect=PostgresException("Could not bootstrap master PostgreSQL"))
|
||||
self.assertRaises(PostgresException, self.ha.bootstrap)
|
||||
|
||||
def test_bootstrap_release_initialize_key_on_watchdog_failure(self):
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.e.initialize = true
|
||||
self.ha.bootstrap()
|
||||
self.p.is_running = true
|
||||
self.p.is_leader = true
|
||||
with patch.object(Watchdog, 'activate', Mock(return_value=False)):
|
||||
self.assertEquals(self.ha.post_bootstrap(), 'running post_bootstrap')
|
||||
self.assertRaises(PatroniException, self.ha.post_bootstrap)
|
||||
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
def test_reinitialize(self):
|
||||
self.assertIsNotNone(self.ha.reinitialize())
|
||||
|
||||
@@ -363,7 +303,6 @@ class TestHa(unittest.TestCase):
|
||||
self.ha.state_handler.name = self.ha.cluster.leader.name
|
||||
self.assertIsNotNone(self.ha.reinitialize())
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
def test_restart(self):
|
||||
self.assertEquals(self.ha.restart({}), (True, 'restarted successfully'))
|
||||
self.p.restart = Mock(return_value=None)
|
||||
@@ -376,12 +315,11 @@ class TestHa(unittest.TestCase):
|
||||
with patch.object(self.ha, "restart_matches", return_value=False):
|
||||
self.assertEquals(self.ha.restart({'foo': 'bar'}), (False, "restart conditions are not satisfied"))
|
||||
|
||||
@patch('os.kill', Mock())
|
||||
def test_restart_in_progress(self):
|
||||
with patch('patroni.async_executor.AsyncExecutor.busy', PropertyMock(return_value=True)):
|
||||
self.ha.restart({}, run_async=True)
|
||||
self.assertTrue(self.ha.restart_scheduled())
|
||||
self.assertEquals(self.ha.run_cycle(), 'restart in progress')
|
||||
self.assertEquals(self.ha.run_cycle(), 'not healthy enough for leader race')
|
||||
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.assertEquals(self.ha.run_cycle(), 'restart in progress')
|
||||
@@ -390,13 +328,10 @@ class TestHa(unittest.TestCase):
|
||||
self.assertEquals(self.ha.run_cycle(), 'updated leader lock during restart')
|
||||
|
||||
self.ha.update_lock = false
|
||||
self.p.set_role('master')
|
||||
with patch('patroni.async_executor.CriticalTask.cancel', Mock(return_value=False)):
|
||||
with patch('patroni.postgresql.Postgresql.stop') as stop_mock:
|
||||
self.assertEquals(self.ha.run_cycle(), 'lost leader lock during restart')
|
||||
stop_mock.assert_called()
|
||||
self.assertEquals(self.ha.run_cycle(), 'failed to update leader lock during restart')
|
||||
|
||||
@patch('requests.get', requests_get)
|
||||
@patch('time.sleep', Mock())
|
||||
def test_manual_failover_from_leader(self):
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.ha.has_lock = true
|
||||
@@ -409,13 +344,9 @@ class TestHa(unittest.TestCase):
|
||||
f = Failover(0, self.p.name, '', None)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(f)
|
||||
self.assertEquals(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||
self.p.rewind_needed_and_possible = true
|
||||
self.assertEquals(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True)
|
||||
self.assertEquals(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(watchdog_failed=True)
|
||||
self.assertEquals(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||
self.ha.fetch_node_status = get_node_status(xlog_location=1)
|
||||
self.assertEquals(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
||||
# manual failover from the previous leader to us won't happen if we hold the nofailover flag
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, None))
|
||||
@@ -453,19 +384,7 @@ class TestHa(unittest.TestCase):
|
||||
self.assertEquals('PAUSE: no action. i am the leader with the lock', self.ha.run_cycle())
|
||||
|
||||
@patch('requests.get', requests_get)
|
||||
def test_manual_failover_from_leader_in_synchronous_mode(self):
|
||||
self.p.is_leader = true
|
||||
self.ha.has_lock = true
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.ha.is_failover_possible = false
|
||||
self.ha.process_sync_replication = Mock()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, None))
|
||||
self.assertEquals('no action. i am the leader with the lock', self.ha.run_cycle())
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, 'a'))
|
||||
self.ha.is_failover_possible = true
|
||||
self.assertEquals('manual failover: demoting myself', self.ha.run_cycle())
|
||||
|
||||
@patch('requests.get', requests_get)
|
||||
@patch('time.sleep', Mock())
|
||||
def test_manual_failover_process_no_leader(self):
|
||||
self.p.is_leader = false
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', self.p.name, None))
|
||||
@@ -490,6 +409,7 @@ class TestHa(unittest.TestCase):
|
||||
self.ha.patroni.nofailover = True
|
||||
self.assertEquals(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
def test_manual_failover_process_no_leader_in_pause(self):
|
||||
self.ha.is_paused = true
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None))
|
||||
@@ -508,8 +428,6 @@ class TestHa(unittest.TestCase):
|
||||
self.ha.patroni.nofailover = False
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.assertTrue(self.ha.is_healthiest_node())
|
||||
with patch.object(Watchdog, 'is_healthy', PropertyMock(return_value=False)):
|
||||
self.assertFalse(self.ha.is_healthiest_node())
|
||||
with patch('patroni.postgresql.Postgresql.is_starting', return_value=True):
|
||||
self.assertFalse(self.ha.is_healthiest_node())
|
||||
self.ha.is_paused = true
|
||||
@@ -522,9 +440,9 @@ class TestHa(unittest.TestCase):
|
||||
self.assertTrue(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
||||
self.ha.fetch_node_status = get_node_status(in_recovery=False) # accessible, not in_recovery
|
||||
self.assertFalse(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=11) # accessible, in_recovery, wal position ahead
|
||||
self.ha.fetch_node_status = get_node_status(xlog_location=11) # accessible, in_recovery, xlog location ahead
|
||||
self.assertFalse(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
||||
with patch('patroni.postgresql.Postgresql.wal_position', return_value=1):
|
||||
with patch('patroni.postgresql.Postgresql.xlog_position', return_value=1):
|
||||
self.assertFalse(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
||||
self.ha.patroni.nofailover = True
|
||||
self.assertFalse(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
||||
@@ -651,6 +569,7 @@ class TestHa(unittest.TestCase):
|
||||
self.assertEquals(self.ha.run_cycle(), 'no action. i am a secondary and i am following a leader')
|
||||
check_calls([(update_lock, False), (demote, False)])
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
def test_manual_failover_while_starting(self):
|
||||
self.ha.has_lock = true
|
||||
self.p.check_for_startup = true
|
||||
@@ -662,8 +581,7 @@ class TestHa(unittest.TestCase):
|
||||
@patch('patroni.ha.Ha.demote')
|
||||
def test_failover_immediately_on_zero_master_start_timeout(self, demote):
|
||||
self.p.is_running = false
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(sync=(self.p.name, 'other'))
|
||||
self.ha.cluster.config.data['synchronous_mode'] = True
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.ha.patroni.config.set_dynamic_configuration({'master_start_timeout': 0})
|
||||
self.ha.has_lock = true
|
||||
self.ha.update_lock = true
|
||||
@@ -671,13 +589,15 @@ class TestHa(unittest.TestCase):
|
||||
self.assertEquals(self.ha.run_cycle(), 'stopped PostgreSQL to fail over after a crash')
|
||||
demote.assert_called_once()
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
@patch('patroni.postgresql.Postgresql.follow')
|
||||
def test_demote_immediate(self, follow):
|
||||
self.ha.has_lock = true
|
||||
self.e.get_cluster = Mock(return_value=get_cluster_initialized_without_leader())
|
||||
self.ha.demote('immediate')
|
||||
follow.assert_called_once_with(None)
|
||||
follow.assert_called_once_with(None, None, True, None, True)
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
def test_process_sync_replication(self):
|
||||
self.ha.has_lock = true
|
||||
mock_set_sync = self.p.set_synchronous_standby = Mock()
|
||||
@@ -743,13 +663,6 @@ class TestHa(unittest.TestCase):
|
||||
self.ha.run_cycle()
|
||||
self.assertEquals(self.ha.dcs.write_sync_state.call_count, 1)
|
||||
|
||||
# Test sync set to '*' when synchronous_mode_strict is enabled
|
||||
mock_set_sync.reset_mock()
|
||||
self.ha.is_synchronous_mode_strict = true
|
||||
self.p.pick_synchronous_standby = Mock(return_value=(None, False))
|
||||
self.ha.run_cycle()
|
||||
mock_set_sync.assert_called_once_with('*')
|
||||
|
||||
def test_sync_replication_become_master(self):
|
||||
self.ha.is_synchronous_mode = true
|
||||
|
||||
@@ -804,7 +717,8 @@ class TestHa(unittest.TestCase):
|
||||
mock_promote.assert_called_once()
|
||||
mock_write_sync.assert_called_once_with('other', None, index=0)
|
||||
|
||||
def test_disable_sync_when_restarting(self):
|
||||
@patch('time.sleep')
|
||||
def test_disable_sync_when_restarting(self, mock_sleep):
|
||||
self.ha.is_synchronous_mode = true
|
||||
|
||||
self.p.name = 'other'
|
||||
@@ -817,10 +731,10 @@ class TestHa(unittest.TestCase):
|
||||
get_cluster_initialized_with_leader(sync=('leader', syncstandby))
|
||||
for syncstandby in ['other', None]])
|
||||
|
||||
with patch('time.sleep') as mock_sleep:
|
||||
self.ha.restart({})
|
||||
mock_restart.assert_called_once()
|
||||
mock_sleep.assert_called()
|
||||
self.ha.restart({})
|
||||
|
||||
mock_restart.assert_called_once()
|
||||
mock_sleep.assert_called()
|
||||
|
||||
# Restart is still called when DCS connection fails
|
||||
mock_restart.reset_mock()
|
||||
@@ -858,17 +772,11 @@ class TestHa(unittest.TestCase):
|
||||
def test_wakup(self):
|
||||
self.ha.wakeup()
|
||||
|
||||
def test_shutdown(self):
|
||||
self.p.is_running = false
|
||||
self.ha.shutdown()
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
def test_leader_with_empty_directory(self):
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.ha.has_lock = true
|
||||
self.p.data_directory_empty = true
|
||||
self.assertEquals(self.ha.run_cycle(), 'released leader key voluntarily as data dir empty and currently leader')
|
||||
self.assertEquals(self.p.role, 'uninitialized')
|
||||
|
||||
# as has_lock is mocked out, we need to fake the leader key release
|
||||
self.ha.has_lock = false
|
||||
|
||||
@@ -104,9 +104,7 @@ class TestPatroni(unittest.TestCase):
|
||||
@patch('patroni.config.Config.save_cache', Mock())
|
||||
@patch('patroni.config.Config.reload_local_configuration', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'state', PropertyMock(return_value='running'))
|
||||
@patch.object(Postgresql, 'data_directory_empty', Mock(return_value=False))
|
||||
def test_run(self):
|
||||
self.p.postgresql.set_role('replica')
|
||||
self.p.sighup_handler()
|
||||
self.p.ha.dcs.watch = Mock(side_effect=SleepException)
|
||||
self.p.api.start = Mock()
|
||||
|
||||
+150
-333
@@ -1,16 +1,13 @@
|
||||
import errno
|
||||
import mock # for the mock.call method, importing it without a namespace breaks python3
|
||||
import os
|
||||
import psycopg2
|
||||
import psutil
|
||||
import shutil
|
||||
import subprocess
|
||||
import unittest
|
||||
|
||||
from mock import Mock, MagicMock, PropertyMock, patch, mock_open
|
||||
from patroni.async_executor import CriticalTask
|
||||
from patroni.dcs import Cluster, Leader, Member, SyncState
|
||||
from patroni.exceptions import PostgresConnectionException
|
||||
from patroni.exceptions import PostgresException, PostgresConnectionException
|
||||
from patroni.postgresql import Postgresql, STATE_REJECT, STATE_NO_RESPONSE
|
||||
from patroni.utils import RetryFailedError
|
||||
from six.moves import builtins
|
||||
@@ -26,16 +23,14 @@ class MockCursor(object):
|
||||
self.results = []
|
||||
|
||||
def execute(self, sql, *params):
|
||||
if sql.startswith('blabla'):
|
||||
raise psycopg2.ProgrammingError()
|
||||
elif sql == 'CHECKPOINT':
|
||||
if sql.startswith('blabla') or sql == 'CHECKPOINT':
|
||||
raise psycopg2.OperationalError()
|
||||
elif sql.startswith('RetryFailedError'):
|
||||
raise RetryFailedError('retry')
|
||||
elif sql.startswith('SELECT slot_name'):
|
||||
self.results = [('blabla',), ('foobar',)]
|
||||
elif sql.startswith('SELECT CASE WHEN pg_is_in_recovery()'):
|
||||
self.results = [(2,)]
|
||||
self.results = [(0,)]
|
||||
elif sql == 'SELECT pg_is_in_recovery()':
|
||||
self.results = [(False, )]
|
||||
elif sql.startswith('WITH replication_info AS ('):
|
||||
@@ -47,15 +42,7 @@ class MockCursor(object):
|
||||
('search_path', 'public', None, 'string', 'user'),
|
||||
('port', '5433', None, 'integer', 'postmaster'),
|
||||
('listen_addresses', '*', None, 'string', 'postmaster'),
|
||||
('autovacuum', 'on', None, 'bool', 'sighup'),
|
||||
('unix_socket_directories', '/tmp', None, 'string', 'postmaster')]
|
||||
elif sql.startswith('IDENTIFY_SYSTEM'):
|
||||
self.results = [('1', 2, '0/402EEC0', '')]
|
||||
elif sql.startswith('TIMELINE_HISTORY '):
|
||||
self.results = [('', b'x\t0/40159C0\tno recovery target specified\n\n' +
|
||||
b'1\t0/40159C0\tno recovery target specified\n\n' +
|
||||
b'2\t0/402DD98\tno recovery target specified\n\n' +
|
||||
b'3\t0/403DD98\tno recovery target specified\n')]
|
||||
('autovacuum', 'on', None, 'bool', 'sighup')]
|
||||
else:
|
||||
self.results = [(None, None, None, None, None, None, None, None, None, None)]
|
||||
|
||||
@@ -78,7 +65,7 @@ class MockCursor(object):
|
||||
|
||||
class MockConnect(object):
|
||||
|
||||
server_version = 99999
|
||||
server_version = '99999'
|
||||
autocommit = False
|
||||
closed = 0
|
||||
|
||||
@@ -151,10 +138,21 @@ Data page checksum version: 0
|
||||
"""
|
||||
|
||||
|
||||
def postmaster_opts_string(*args, **kwargs):
|
||||
return '/usr/local/pgsql/bin/postgres "-D" "data/postgresql0" "--listen_addresses=127.0.0.1" \
|
||||
"--port=5432" "--hot_standby=on" "--wal_keep_segments=8" "--wal_level=hot_standby" \
|
||||
"--archive_command=mkdir -p ../wal_archive && cp %p ../wal_archive/%f" "--wal_log_hints=on" \
|
||||
"--max_wal_senders=5" "--archive_timeout=1800s" "--archive_mode=on" "--max_replication_slots=5"\n'
|
||||
|
||||
|
||||
def psycopg2_connect(*args, **kwargs):
|
||||
return MockConnect()
|
||||
|
||||
|
||||
def fake_listdir(path):
|
||||
return ["a", "b", "c"] if path.endswith('pg_xlog/archive_status') else []
|
||||
|
||||
|
||||
@patch('subprocess.call', Mock(return_value=0))
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
class TestPostgresql(unittest.TestCase):
|
||||
@@ -162,30 +160,30 @@ class TestPostgresql(unittest.TestCase):
|
||||
'search_path': 'public', 'hot_standby': 'on', 'max_wal_senders': 5,
|
||||
'wal_keep_segments': 8, 'wal_log_hints': 'on', 'max_locks_per_transaction': 64,
|
||||
'max_worker_processes': 8, 'max_connections': 100, 'max_prepared_transactions': 0,
|
||||
'track_commit_timestamp': 'off', 'unix_socket_directories': '/tmp'}
|
||||
'track_commit_timestamp': 'off'}
|
||||
|
||||
@patch('subprocess.call', Mock(return_value=0))
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('os.rename', Mock())
|
||||
@patch.object(Postgresql, 'get_major_version', Mock(return_value=90600))
|
||||
@patch.object(Postgresql, 'get_major_version', Mock(return_value=9.6))
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||
def setUp(self):
|
||||
self.data_dir = 'data/test0'
|
||||
self.config_dir = self.data_dir
|
||||
if not os.path.exists(self.data_dir):
|
||||
os.makedirs(self.data_dir)
|
||||
self.p = Postgresql({'name': 'test0', 'scope': 'batman', 'data_dir': self.data_dir,
|
||||
'config_dir': self.config_dir, 'retry_timeout': 10, 'pgpass': '/tmp/pgpass0',
|
||||
'listen': '127.0.0.2, 127.0.0.3:5432', 'connect_address': '127.0.0.2:5432',
|
||||
self.p = Postgresql({'name': 'test0', 'scope': 'batman', 'data_dir': self.data_dir, 'retry_timeout': 10,
|
||||
'listen': '127.0.0.1, *:5432', 'connect_address': '127.0.0.2:5432',
|
||||
'authentication': {'superuser': {'username': 'test', 'password': 'test'},
|
||||
'replication': {'username': 'replicator', 'password': 'rep-pass'}},
|
||||
'remove_data_directory_on_rewind_failure': True,
|
||||
'use_pg_rewind': True, 'pg_ctl_timeout': 'bla',
|
||||
'parameters': self._PARAMETERS,
|
||||
'recovery_conf': {'foo': 'bar'},
|
||||
'pg_hba': ['host all all 0.0.0.0/0 md5'],
|
||||
'callbacks': {'on_start': 'true', 'on_stop': 'true', 'on_reload': 'true',
|
||||
'on_restart': 'true', 'on_role_change': 'true'}})
|
||||
'callbacks': {'on_start': 'true', 'on_stop': 'true',
|
||||
'on_restart': 'true', 'on_role_change': 'true',
|
||||
'on_reload': 'true'
|
||||
},
|
||||
'restore': 'true'})
|
||||
self.p._callback_executor = Mock()
|
||||
self.leadermem = Member(0, 'leader', 28, {'conn_url': 'postgres://replicator:[email protected]:5435/postgres'})
|
||||
self.leader = Leader(-1, 28, self.leadermem)
|
||||
@@ -216,13 +214,13 @@ class TestPostgresql(unittest.TestCase):
|
||||
mock_is_running.return_value = True
|
||||
mock_wait_for_port_open.return_value = True
|
||||
mock_wait_for_startup.return_value = False
|
||||
mock_popen.return_value.stdout.readline.return_value = '123'
|
||||
mock_popen.stdout.readline.return_value = '123'
|
||||
self.assertTrue(self.p.start())
|
||||
mock_is_running.return_value = False
|
||||
open(os.path.join(self.data_dir, 'postmaster.pid'), 'w').close()
|
||||
pg_conf = os.path.join(self.data_dir, 'postgresql.conf')
|
||||
open(pg_conf, 'w').close()
|
||||
self.assertFalse(self.p.start(task=CriticalTask()))
|
||||
self.assertFalse(self.p.start())
|
||||
with open(pg_conf) as f:
|
||||
lines = f.readlines()
|
||||
self.assertTrue("f.oo = 'bar'\n" in lines)
|
||||
@@ -233,23 +231,20 @@ class TestPostgresql(unittest.TestCase):
|
||||
|
||||
mock_wait_for_port_open.return_value = False
|
||||
self.assertFalse(self.p.start())
|
||||
task = CriticalTask()
|
||||
task.cancel()
|
||||
self.assertFalse(self.p.start(task=task))
|
||||
|
||||
@patch.object(Postgresql, 'pg_isready')
|
||||
@patch.object(Postgresql, 'read_pid_file')
|
||||
@patch.object(Postgresql, '_is_postmaster_pid_running')
|
||||
@patch.object(Postgresql, 'is_pid_running')
|
||||
@patch('patroni.postgresql.polling_loop', Mock(return_value=range(1)))
|
||||
def test_wait_for_port_open(self, mock_is_postmaster_pid_running, mock_read_pid_file, mock_pg_isready):
|
||||
mock_is_postmaster_pid_running.return_value = False
|
||||
def test_wait_for_port_open(self, mock_is_pid_running, mock_read_pid_file, mock_pg_isready):
|
||||
mock_is_pid_running.return_value = False
|
||||
mock_pg_isready.return_value = STATE_NO_RESPONSE
|
||||
|
||||
# No pid file and postmaster death
|
||||
mock_read_pid_file.return_value = {}
|
||||
self.assertFalse(self.p.wait_for_port_open(42, 100., 1))
|
||||
|
||||
mock_is_postmaster_pid_running.return_value = True
|
||||
mock_is_pid_running.return_value = True
|
||||
|
||||
# timeout
|
||||
mock_read_pid_file.return_value = {'pid', 1}
|
||||
@@ -269,44 +264,13 @@ class TestPostgresql(unittest.TestCase):
|
||||
mock_pg_isready.return_value = 'garbage'
|
||||
self.assertTrue(self.p.wait_for_port_open(42, 100., 1))
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
@patch.object(Postgresql, 'is_running')
|
||||
@patch.object(Postgresql, 'get_pid')
|
||||
def test_stop(self, mock_get_pid, mock_is_running):
|
||||
mock_callback = Mock()
|
||||
mock_is_running.return_value = False
|
||||
self.assertTrue(self.p.stop(on_safepoint=mock_callback))
|
||||
mock_callback.assert_called()
|
||||
|
||||
with patch.object(Postgresql, '_is_postmaster_pid_running', Mock(return_value=False)), \
|
||||
patch.object(Postgresql, 'data_directory_empty', Mock(return_value=True)):
|
||||
with patch('psutil.Process') as mock_psutil:
|
||||
self.p._postmaster_cached_info = {'pid': 1, 'start_time': 1}
|
||||
mock_psutil.return_value.pid = 1
|
||||
mock_psutil.return_value.create_time.return_value = 1
|
||||
self.assertTrue(self.p.stop())
|
||||
self.p._postmaster_cached_info = {'pid': 1, 'start_time': 1}
|
||||
mock_psutil.return_value.create_time.return_value = 100
|
||||
self.assertTrue(self.p.stop())
|
||||
self.p._postmaster_cached_info = {'pid': 1, 'start_time': 1}
|
||||
mock_psutil.side_effect = psutil.NoSuchProcess('')
|
||||
self.assertTrue(self.p.stop())
|
||||
|
||||
def test_stop(self, mock_is_running):
|
||||
mock_is_running.return_value = True
|
||||
mock_get_pid.return_value = 0
|
||||
mock_callback.reset_mock()
|
||||
self.assertTrue(self.p.stop(on_safepoint=mock_callback))
|
||||
mock_callback.assert_called()
|
||||
mock_get_pid.return_value = -1
|
||||
self.assertFalse(self.p.stop())
|
||||
mock_get_pid.return_value = 123
|
||||
with patch('os.kill', Mock(side_effect=[OSError(errno.ESRCH, ''), OSError, None])):
|
||||
self.assertTrue(self.p.stop())
|
||||
with patch('subprocess.call', Mock(return_value=1)):
|
||||
mock_is_running.return_value = False
|
||||
self.assertTrue(self.p.stop())
|
||||
self.assertFalse(self.p.stop())
|
||||
self.assertTrue(self.p.stop())
|
||||
with patch.object(Postgresql, '_signal_postmaster_stop', Mock(return_value=(123, None))):
|
||||
with patch.object(Postgresql, '_is_postmaster_pid_running', Mock(side_effect=[True, False, False])):
|
||||
self.assertTrue(self.p.stop())
|
||||
|
||||
def test_restart(self):
|
||||
self.p.start = Mock(return_value=False)
|
||||
@@ -329,84 +293,39 @@ class TestPostgresql(unittest.TestCase):
|
||||
@patch('patroni.postgresql.Postgresql.write_pgpass', MagicMock(return_value=dict()))
|
||||
def test_pg_rewind(self, mock_call):
|
||||
r = {'user': '', 'host': '', 'port': '', 'database': '', 'password': ''}
|
||||
self.assertTrue(self.p.pg_rewind(r))
|
||||
self.assertTrue(self.p.rewind(r))
|
||||
subprocess.call = mock_call
|
||||
self.assertFalse(self.p.pg_rewind(r))
|
||||
self.assertFalse(self.p.rewind(r))
|
||||
|
||||
def test_check_recovery_conf(self):
|
||||
self.p.write_recovery_conf({'primary_conninfo': 'foo'})
|
||||
self.assertFalse(self.p.check_recovery_conf(None))
|
||||
self.p.write_recovery_conf({})
|
||||
self.assertTrue(self.p.check_recovery_conf(None))
|
||||
|
||||
@patch.object(Postgresql, 'start', Mock())
|
||||
@patch('os.unlink', Mock(return_value=True))
|
||||
@patch('subprocess.check_output', Mock(return_value=0, side_effect=pg_controldata_string))
|
||||
@patch.object(Postgresql, 'remove_data_directory', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'single_user_mode', Mock(return_value=1))
|
||||
@patch.object(Postgresql, 'write_pgpass', Mock(return_value={}))
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'can_rewind', PropertyMock(return_value=True))
|
||||
def test__get_local_timeline_lsn(self):
|
||||
self.p.trigger_check_diverged_lsn()
|
||||
with patch.object(Postgresql, 'controldata',
|
||||
Mock(return_value={'Database cluster state': 'shut down in recovery',
|
||||
'Minimum recovery ending location': '0/0',
|
||||
"Min recovery ending loc's timeline": '0'})):
|
||||
self.p.rewind_needed_and_possible(self.leader)
|
||||
with patch.object(Postgresql, 'is_running', Mock(return_value=True)):
|
||||
with patch.object(MockCursor, 'fetchone', Mock(side_effect=[(False, ), Exception])):
|
||||
self.p.rewind_needed_and_possible(self.leader)
|
||||
@patch.object(Postgresql, 'rewind', return_value=False)
|
||||
def test_follow(self, mock_pg_rewind):
|
||||
with patch.object(Postgresql, 'check_recovery_conf', Mock(return_value=True)):
|
||||
self.assertTrue(self.p.follow(None, None)) # nothing to do, recovery.conf has good primary_conninfo
|
||||
|
||||
@patch.object(Postgresql, 'start', Mock())
|
||||
@patch.object(Postgresql, 'can_rewind', PropertyMock(return_value=True))
|
||||
@patch.object(Postgresql, '_get_local_timeline_lsn', Mock(return_value=(2, '0/40159C1')))
|
||||
@patch.object(Postgresql, 'check_leader_is_not_in_recovery')
|
||||
def test__check_timeline_and_lsn(self, mock_check_leader_is_not_in_recovery):
|
||||
mock_check_leader_is_not_in_recovery.return_value = False
|
||||
self.p.trigger_check_diverged_lsn()
|
||||
self.assertFalse(self.p.rewind_needed_and_possible(self.leader))
|
||||
mock_check_leader_is_not_in_recovery.return_value = True
|
||||
self.assertFalse(self.p.rewind_needed_and_possible(self.leader))
|
||||
self.p.trigger_check_diverged_lsn()
|
||||
with patch('psycopg2.connect', Mock(side_effect=Exception)):
|
||||
self.assertFalse(self.p.rewind_needed_and_possible(self.leader))
|
||||
with patch.object(MockCursor, 'fetchone',
|
||||
Mock(side_effect=[('', 2, '0/0'), ('', b'2\tG/40159C0\tno recovery target specified\n\n')])):
|
||||
self.assertFalse(self.p.rewind_needed_and_possible(self.leader))
|
||||
self.p.trigger_check_diverged_lsn()
|
||||
with patch.object(MockCursor, 'fetchone',
|
||||
Mock(side_effect=[('', 2, '0/0'), ('', b'3\t040159C0\tno recovery target specified\n')])):
|
||||
self.assertFalse(self.p.rewind_needed_and_possible(self.leader))
|
||||
self.p.trigger_check_diverged_lsn()
|
||||
with patch.object(MockCursor, 'fetchone', Mock(return_value=('', 1, '0/0'))):
|
||||
with patch.object(Postgresql, '_get_local_timeline_lsn', Mock(return_value=(1, '0/0'))):
|
||||
self.assertFalse(self.p.rewind_needed_and_possible(self.leader))
|
||||
self.p.trigger_check_diverged_lsn()
|
||||
self.assertTrue(self.p.rewind_needed_and_possible(self.leader))
|
||||
self.p.follow(self.me, self.me) # follow is called when the node is holding leader lock
|
||||
|
||||
@patch.object(MockCursor, 'fetchone', Mock(side_effect=[(True,), Exception]))
|
||||
def test_check_leader_is_not_in_recovery(self):
|
||||
self.p.check_leader_is_not_in_recovery()
|
||||
self.p.check_leader_is_not_in_recovery()
|
||||
with patch.object(Postgresql, 'restart', Mock(return_value=False)):
|
||||
self.p.set_role('replica')
|
||||
self.p.follow(None, None) # restart without rewind
|
||||
|
||||
@patch.object(Postgresql, 'checkpoint', side_effect=['', '1'])
|
||||
@patch.object(Postgresql, 'stop', Mock(return_value=False))
|
||||
@patch.object(Postgresql, 'start', Mock())
|
||||
def test_rewind(self, mock_checkpoint):
|
||||
self.p.rewind(self.leader)
|
||||
with patch.object(Postgresql, 'pg_rewind', Mock(return_value=False)):
|
||||
mock_checkpoint.side_effect = ['1', '', '', '']
|
||||
self.p.rewind(self.leader)
|
||||
self.p.rewind(self.leader)
|
||||
with patch.object(Postgresql, 'check_leader_is_not_in_recovery', Mock(return_value=False)):
|
||||
self.p.rewind(self.leader)
|
||||
self.p.config['remove_data_directory_on_rewind_failure'] = False
|
||||
self.p.trigger_check_diverged_lsn()
|
||||
self.p.rewind(self.leader)
|
||||
with patch.object(Postgresql, 'is_running', Mock(return_value=True)):
|
||||
self.p.rewind(self.leader)
|
||||
self.p.is_leader = Mock(return_value=False)
|
||||
self.p.rewind(self.leader)
|
||||
with patch.object(Postgresql, 'stop', Mock(return_value=False)):
|
||||
self.p.follow(self.leader, self.leader, need_rewind=True) # failed to stop postgres
|
||||
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=False))
|
||||
@patch.object(Postgresql, 'start', Mock())
|
||||
def test_follow(self):
|
||||
self.p.follow(None)
|
||||
self.p.follow(self.leader, self.leader) # "leader" is not accessible or is_in_recovery
|
||||
|
||||
with patch.object(Postgresql, 'checkpoint', Mock(return_value=None)):
|
||||
self.p.follow(self.leader, self.leader)
|
||||
mock_pg_rewind.return_value = True
|
||||
self.p.follow(self.leader, self.leader, need_rewind=True)
|
||||
|
||||
self.p.follow(None, None) # check_recovery_conf...
|
||||
|
||||
@patch('subprocess.check_output', Mock(return_value=0, side_effect=pg_controldata_string))
|
||||
def test_can_rewind(self):
|
||||
@@ -460,7 +379,7 @@ class TestPostgresql(unittest.TestCase):
|
||||
assert "test-3" in errorlog_mock.call_args[0][1]
|
||||
assert "test.3" in errorlog_mock.call_args[0][1]
|
||||
|
||||
@patch.object(MockCursor, 'execute', Mock(side_effect=psycopg2.OperationalError))
|
||||
@patch.object(MockConnect, 'closed', 2)
|
||||
def test__query(self):
|
||||
self.assertRaises(PostgresConnectionException, self.p._query, 'blabla')
|
||||
self.p._state = 'restarting'
|
||||
@@ -469,7 +388,7 @@ class TestPostgresql(unittest.TestCase):
|
||||
def test_query(self):
|
||||
self.p.query('select 1')
|
||||
self.assertRaises(PostgresConnectionException, self.p.query, 'RetryFailedError')
|
||||
self.assertRaises(psycopg2.ProgrammingError, self.p.query, 'blabla')
|
||||
self.assertRaises(psycopg2.OperationalError, self.p.query, 'blabla')
|
||||
|
||||
@patch.object(Postgresql, 'pg_isready', Mock(return_value=STATE_REJECT))
|
||||
def test_is_leader(self):
|
||||
@@ -488,12 +407,12 @@ class TestPostgresql(unittest.TestCase):
|
||||
self.assertFalse(self.p.is_healthy())
|
||||
|
||||
def test_promote(self):
|
||||
self.p.set_role('replica')
|
||||
self.p._role = 'replica'
|
||||
self.assertTrue(self.p.promote())
|
||||
self.assertTrue(self.p.promote())
|
||||
|
||||
def test_last_operation(self):
|
||||
self.assertEquals(self.p.last_operation(), '2')
|
||||
self.assertEquals(self.p.last_operation(), '0')
|
||||
Thread(target=self.p.last_operation).start()
|
||||
|
||||
@patch('os.path.isfile', Mock(return_value=True))
|
||||
@@ -507,9 +426,6 @@ class TestPostgresql(unittest.TestCase):
|
||||
|
||||
@patch('shlex.split', Mock(side_effect=OSError))
|
||||
def test_call_nowait(self):
|
||||
self.p.set_role('replica')
|
||||
self.assertIsNone(self.p.call_nowait('on_start'))
|
||||
self.p.bootstrapping = True
|
||||
self.assertIsNone(self.p.call_nowait('on_start'))
|
||||
|
||||
def test_non_existing_callback(self):
|
||||
@@ -531,74 +447,20 @@ class TestPostgresql(unittest.TestCase):
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||
def test_bootstrap(self):
|
||||
with patch('subprocess.call', Mock(return_value=1)):
|
||||
self.assertFalse(self.p.bootstrap({}))
|
||||
self.assertRaises(PostgresException, self.p.bootstrap, {})
|
||||
|
||||
config = {'users': {'replicator': {'password': 'rep-pass', 'options': ['replication']}}}
|
||||
with patch.object(Postgresql, 'run_bootstrap_post_init', Mock(return_value=False)):
|
||||
self.assertRaises(PostgresException, self.p.bootstrap, {})
|
||||
|
||||
self.p.bootstrap(config)
|
||||
with open(os.path.join(self.config_dir, 'pg_hba.conf')) as f:
|
||||
lines = f.readlines()
|
||||
self.assertTrue('host all all 0.0.0.0/0 md5\n' in lines)
|
||||
|
||||
self.p.config.pop('pg_hba')
|
||||
config.update({'post_init': '/bin/false',
|
||||
'pg_hba': ['host replication replicator 127.0.0.1/32 md5',
|
||||
'hostssl all all 0.0.0.0/0 md5',
|
||||
'host all all 0.0.0.0/0 md5']})
|
||||
self.p.bootstrap(config)
|
||||
self.p.bootstrap({'users': {'replicator': {'password': 'rep-pass', 'options': ['replication']}},
|
||||
'pg_hba': ['host replication replicator 127.0.0.1/32 md5',
|
||||
'hostssl all all 0.0.0.0/0 md5',
|
||||
'host all all 0.0.0.0/0 md5'],
|
||||
'post_init': '/bin/false'})
|
||||
with open(os.path.join(self.data_dir, 'pg_hba.conf')) as f:
|
||||
lines = f.readlines()
|
||||
self.assertTrue('host replication replicator 127.0.0.1/32 md5\n' in lines)
|
||||
|
||||
def test_custom_bootstrap(self):
|
||||
config = {'method': 'foo', 'foo': {'command': 'bar'}}
|
||||
with patch('subprocess.call', Mock(return_value=1)):
|
||||
self.assertFalse(self.p.bootstrap(config))
|
||||
with patch('subprocess.call', Mock(side_effect=Exception)):
|
||||
self.assertFalse(self.p.bootstrap(config))
|
||||
with patch('subprocess.call', Mock(return_value=0)),\
|
||||
patch('subprocess.Popen', Mock(side_effect=Exception("42"))),\
|
||||
patch('os.path.isfile', Mock(return_value=True)),\
|
||||
patch('os.unlink', Mock()),\
|
||||
patch.object(Postgresql, 'save_configuration_files', Mock()),\
|
||||
patch.object(Postgresql, 'restore_configuration_files', Mock()),\
|
||||
patch.object(Postgresql, 'write_recovery_conf', Mock()):
|
||||
with self.assertRaises(Exception) as e:
|
||||
self.p.bootstrap(config)
|
||||
self.assertEqual(str(e.exception), '42')
|
||||
|
||||
config['foo']['recovery_conf'] = {'foo': 'bar'}
|
||||
|
||||
with self.assertRaises(Exception) as e:
|
||||
self.p.bootstrap(config)
|
||||
self.assertEqual(str(e.exception), '42')
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
@patch('os.unlink', Mock())
|
||||
@patch.object(Postgresql, 'run_bootstrap_post_init', Mock(return_value=True))
|
||||
@patch.object(Postgresql, '_custom_bootstrap', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
||||
def test_post_bootstrap(self):
|
||||
config = {'method': 'foo', 'foo': {'command': 'bar'}}
|
||||
self.p.bootstrap(config)
|
||||
|
||||
task = CriticalTask()
|
||||
with patch.object(Postgresql, 'create_or_update_role', Mock(side_effect=Exception)):
|
||||
self.p.post_bootstrap({}, task)
|
||||
self.assertFalse(task.result)
|
||||
|
||||
self.p.config.pop('pg_hba')
|
||||
self.p.post_bootstrap({}, task)
|
||||
self.assertTrue(task.result)
|
||||
|
||||
self.p.bootstrap(config)
|
||||
self.p.set_state('stopped')
|
||||
self.p.reload_config({'authentication': {'superuser': {'username': 'p', 'password': 'p'},
|
||||
'replication': {'username': 'r', 'password': 'r'}},
|
||||
'listen': '*', 'retry_timeout': 10, 'parameters': {'hba_file': 'foo'}})
|
||||
with patch.object(Postgresql, 'restart', Mock()) as mock_restart:
|
||||
self.p.post_bootstrap({}, task)
|
||||
mock_restart.assert_called_once()
|
||||
assert 'host replication replicator 127.0.0.1/32 md5\n' in lines
|
||||
assert 'host all all 0.0.0.0/0 md5\n' in lines
|
||||
|
||||
def test_run_bootstrap_post_init(self):
|
||||
with patch('subprocess.call', Mock(return_value=1)):
|
||||
@@ -610,16 +472,11 @@ class TestPostgresql(unittest.TestCase):
|
||||
with patch('subprocess.call', Mock(return_value=0)) as mock_method:
|
||||
self.p._superuser.pop('username')
|
||||
self.assertTrue(self.p.run_bootstrap_post_init({'post_init': '/bin/false'}))
|
||||
mock_method.assert_called()
|
||||
args, kwargs = mock_method.call_args
|
||||
self.assertTrue('PGPASSFILE' in kwargs['env'])
|
||||
self.assertEquals(args[0], ['/bin/false', 'postgres://127.0.0.2:5432/postgres'])
|
||||
|
||||
mock_method.reset_mock()
|
||||
self.p._local_address.pop('host')
|
||||
self.assertTrue(self.p.run_bootstrap_post_init({'post_init': '/bin/false'}))
|
||||
mock_method.assert_called()
|
||||
self.assertEquals(mock_method.call_args[0][0], ['/bin/false', 'postgres://:5432/postgres'])
|
||||
mock_method.assert_called()
|
||||
args, kwargs = mock_method.call_args
|
||||
assert 'PGPASSFILE' in kwargs['env'].keys()
|
||||
self.assertEquals(args[0], ['/bin/false', 'postgres://localhost:5432/postgres'])
|
||||
|
||||
@patch('patroni.postgresql.Postgresql.create_replica', Mock(return_value=0))
|
||||
def test_clone(self):
|
||||
@@ -651,6 +508,65 @@ class TestPostgresql(unittest.TestCase):
|
||||
with patch('subprocess.check_output', Mock(side_effect=subprocess.CalledProcessError(1, ''))):
|
||||
self.assertEquals(self.p.controldata(), {})
|
||||
|
||||
def test_read_postmaster_opts(self):
|
||||
m = mock_open(read_data=postmaster_opts_string())
|
||||
with patch.object(builtins, 'open', m):
|
||||
data = self.p.read_postmaster_opts()
|
||||
self.assertEquals(data['wal_level'], 'hot_standby')
|
||||
self.assertEquals(int(data['max_replication_slots']), 5)
|
||||
self.assertEqual(data.get('D'), None)
|
||||
|
||||
m.side_effect = IOError
|
||||
data = self.p.read_postmaster_opts()
|
||||
self.assertEqual(data, dict())
|
||||
|
||||
@patch('subprocess.Popen')
|
||||
@patch.object(builtins, 'open', MagicMock(return_value=42))
|
||||
def test_single_user_mode(self, subprocess_popen_mock):
|
||||
subprocess_popen_mock.return_value.wait.return_value = 0
|
||||
self.assertEquals(self.p.single_user_mode(options=dict(archive_mode='on', archive_command='false')), 0)
|
||||
subprocess_popen_mock.assert_called_once_with(['postgres', '--single', '-D', self.data_dir,
|
||||
'-c', 'archive_command=false', '-c', 'archive_mode=on',
|
||||
'postgres'], stdin=subprocess.PIPE,
|
||||
stdout=42,
|
||||
stderr=subprocess.STDOUT)
|
||||
subprocess_popen_mock.reset_mock()
|
||||
self.assertEquals(self.p.single_user_mode(command="CHECKPOINT"), 0)
|
||||
subprocess_popen_mock.assert_called_once_with(['postgres', '--single', '-D', self.data_dir,
|
||||
'postgres'], stdin=subprocess.PIPE,
|
||||
stdout=42,
|
||||
stderr=subprocess.STDOUT)
|
||||
subprocess_popen_mock.return_value = None
|
||||
self.assertEquals(self.p.single_user_mode(), 1)
|
||||
|
||||
@patch('os.listdir', MagicMock(side_effect=fake_listdir))
|
||||
@patch('os.unlink', return_value=True)
|
||||
@patch('os.remove', return_value=True)
|
||||
@patch('os.path.islink', return_value=False)
|
||||
@patch('os.path.isfile', return_value=True)
|
||||
def test_cleanup_archive_status(self, mock_file, mock_link, mock_remove, mock_unlink):
|
||||
ap = os.path.join(self.data_dir, 'pg_xlog', 'archive_status/')
|
||||
self.p.cleanup_archive_status()
|
||||
mock_remove.assert_has_calls([mock.call(ap + 'a'), mock.call(ap + 'b'), mock.call(ap + 'c')])
|
||||
mock_unlink.assert_not_called()
|
||||
|
||||
mock_remove.reset_mock()
|
||||
|
||||
mock_file.return_value = False
|
||||
mock_link.return_value = True
|
||||
self.p.cleanup_archive_status()
|
||||
mock_unlink.assert_has_calls([mock.call(ap + 'a'), mock.call(ap + 'b'), mock.call(ap + 'c')])
|
||||
mock_remove.assert_not_called()
|
||||
|
||||
mock_unlink.reset_mock()
|
||||
mock_remove.reset_mock()
|
||||
|
||||
mock_file.side_effect = OSError
|
||||
mock_link.side_effect = OSError
|
||||
self.p.cleanup_archive_status()
|
||||
mock_unlink.assert_not_called()
|
||||
mock_remove.assert_not_called()
|
||||
|
||||
@patch('patroni.postgresql.Postgresql._version_file_exists', Mock(return_value=True))
|
||||
@patch('subprocess.check_output', MagicMock(return_value=0, side_effect=pg_controldata_string))
|
||||
def test_sysid(self):
|
||||
@@ -685,27 +601,21 @@ class TestPostgresql(unittest.TestCase):
|
||||
def test_reload_config(self):
|
||||
parameters = self._PARAMETERS.copy()
|
||||
parameters.pop('f.oo')
|
||||
config = {'pg_hba': [''], 'use_unix_socket': True, 'authentication': {},
|
||||
'retry_timeout': 10, 'listen': '*', 'parameters': parameters}
|
||||
self.p.reload_config(config)
|
||||
self.p.reload_config({'retry_timeout': 10, 'listen': '*', 'parameters': parameters})
|
||||
parameters['b.ar'] = 'bar'
|
||||
self.p.reload_config(config)
|
||||
self.p.reload_config({'retry_timeout': 10, 'listen': '*', 'parameters': parameters})
|
||||
parameters['autovacuum'] = 'on'
|
||||
self.p.reload_config(config)
|
||||
self.p.reload_config({'retry_timeout': 10, 'listen': '*', 'parameters': parameters})
|
||||
parameters['autovacuum'] = 'off'
|
||||
parameters.pop('search_path')
|
||||
config['listen'] = '*:5433'
|
||||
self.p.reload_config(config)
|
||||
parameters['unix_socket_directories'] = '.'
|
||||
self.p.reload_config(config)
|
||||
self.p.resolve_connection_addresses()
|
||||
self.p.reload_config({'retry_timeout': 10, 'listen': '*:5433', 'parameters': parameters})
|
||||
|
||||
@patch.object(Postgresql, '_version_file_exists', Mock(return_value=True))
|
||||
def test_get_major_version(self):
|
||||
with patch.object(builtins, 'open', mock_open(read_data='9.4')):
|
||||
self.assertEquals(self.p.get_major_version(), 90400)
|
||||
self.assertEquals(self.p.get_major_version(), 9.4)
|
||||
with patch.object(builtins, 'open', Mock(side_effect=Exception)):
|
||||
self.assertEquals(self.p.get_major_version(), 0)
|
||||
self.assertEquals(self.p.get_major_version(), 0.0)
|
||||
|
||||
def test_postmaster_start_time(self):
|
||||
with patch.object(MockCursor, "fetchone", Mock(return_value=('foo', True, '', '', '', '', False))):
|
||||
@@ -788,19 +698,12 @@ class TestPostgresql(unittest.TestCase):
|
||||
os.remove(pidfile)
|
||||
self.assertEquals(self.p.read_pid_file(), {})
|
||||
|
||||
@patch.object(Postgresql, '_version_file_exists', Mock(return_value=True))
|
||||
@patch('os.path.isfile', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'read_pid_file')
|
||||
@patch('psutil.Process')
|
||||
def test_is_postmaster_pid_running(self, mock_psutil, mock_read_pid_file):
|
||||
mock_psutil.return_value.create_time.return_value = 1
|
||||
mock_read_pid_file.return_value = {'pid': -100, 'start_time': 1}
|
||||
self.assertTrue(self.p.is_running())
|
||||
with patch('os.getpid', Mock(return_value=100)):
|
||||
mock_read_pid_file.return_value = {'pid': 100, 'start_time': 1}
|
||||
self.assertFalse(self.p.is_running())
|
||||
mock_read_pid_file.return_value = {'pid': 100, 'start_time': 100}
|
||||
self.assertFalse(self.p.is_running())
|
||||
@patch('os.kill')
|
||||
def test_is_pid_running(self, mock_kill):
|
||||
mock_kill.return_value = True
|
||||
self.assertTrue(self.p.is_pid_running(-100))
|
||||
self.assertFalse(self.p.is_pid_running(0))
|
||||
self.assertFalse(self.p.is_pid_running(None))
|
||||
|
||||
def test_pick_sync_standby(self):
|
||||
cluster = Cluster(True, None, self.leader, 0, [self.me, self.other, self.leadermem], None,
|
||||
@@ -865,91 +768,5 @@ class TestPostgresql(unittest.TestCase):
|
||||
def test_get_server_parameters(self):
|
||||
config = {'synchronous_mode': True, 'parameters': {'wal_level': 'hot_standby'}, 'listen': '0'}
|
||||
self.p.get_server_parameters(config)
|
||||
config['synchronous_mode_strict'] = True
|
||||
self.p.get_server_parameters(config)
|
||||
self.p.set_synchronous_standby('foo')
|
||||
self.p.get_server_parameters(config)
|
||||
|
||||
@patch.object(Postgresql, 'read_pid_file', Mock(return_value={'pid': 'z'}))
|
||||
def test_get_pid(self):
|
||||
self.p.get_pid()
|
||||
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||
@patch.object(Postgresql, '_signal_postmaster_stop', Mock(return_value=(123, None)))
|
||||
@patch.object(Postgresql, 'get_pid', Mock(return_value=123))
|
||||
@patch('time.sleep', Mock())
|
||||
@patch.object(Postgresql, '_is_postmaster_pid_running')
|
||||
def test__wait_for_connection_close(self, mock_is_postmaster_pid_running):
|
||||
mock_is_postmaster_pid_running.side_effect = [True, False, False]
|
||||
mock_callback = Mock()
|
||||
self.p.stop(on_safepoint=mock_callback)
|
||||
|
||||
mock_is_postmaster_pid_running.side_effect = [True, False, False]
|
||||
with patch.object(MockCursor, "execute", Mock(side_effect=psycopg2.Error)):
|
||||
self.p.stop(on_safepoint=mock_callback)
|
||||
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||
@patch.object(Postgresql, '_signal_postmaster_stop', Mock(return_value=(123, None)))
|
||||
@patch.object(Postgresql, 'get_pid', Mock(return_value=123))
|
||||
@patch.object(Postgresql, '_is_postmaster_pid_running', Mock(return_value=False))
|
||||
@patch('psutil.Process')
|
||||
def test__wait_for_user_backends_to_close(self, mock_psutil):
|
||||
child = Mock()
|
||||
child.cmdline.return_value = ['foo']
|
||||
mock_psutil.return_value.children.return_value = [child]
|
||||
mock_callback = Mock()
|
||||
self.p.stop(on_safepoint=mock_callback)
|
||||
|
||||
@patch('os.kill', Mock(side_effect=[OSError(errno.ESRCH, ''), OSError]))
|
||||
@patch('psutil.Process', Mock(side_effect=[psutil.NoSuchProcess]))
|
||||
@patch('time.sleep', Mock())
|
||||
@patch.object(Postgresql, '_is_postmaster_pid_running', Mock(side_effect=[True, False]))
|
||||
def test_terminate_starting_postmaster(self):
|
||||
self.p.terminate_starting_postmaster(123)
|
||||
self.p.terminate_starting_postmaster(123)
|
||||
|
||||
def test_read_postmaster_opts(self):
|
||||
m = mock_open(read_data='/usr/lib/postgres/9.6/bin/postgres "-D" "data/postgresql0" \
|
||||
"--listen_addresses=127.0.0.1" "--port=5432" "--hot_standby=on" "--wal_level=hot_standby" \
|
||||
"--wal_log_hints=on" "--max_wal_senders=5" "--max_replication_slots=5"\n')
|
||||
with patch.object(builtins, 'open', m):
|
||||
data = self.p.read_postmaster_opts()
|
||||
self.assertEquals(data['wal_level'], 'hot_standby')
|
||||
self.assertEquals(int(data['max_replication_slots']), 5)
|
||||
self.assertEqual(data.get('D'), None)
|
||||
|
||||
m.side_effect = IOError
|
||||
data = self.p.read_postmaster_opts()
|
||||
self.assertEqual(data, dict())
|
||||
|
||||
@patch('subprocess.Popen')
|
||||
@patch.object(builtins, 'open', Mock(return_value=42))
|
||||
def test_single_user_mode(self, subprocess_popen_mock):
|
||||
subprocess_popen_mock.return_value.wait.return_value = 0
|
||||
self.assertEquals(self.p.single_user_mode(command="CHECKPOINT"), 0)
|
||||
subprocess_popen_mock.return_value = None
|
||||
self.assertEquals(self.p.single_user_mode(), 1)
|
||||
self.assertEquals(self.p.single_user_mode(options={'archive_mode': 'on'}), 1)
|
||||
|
||||
@patch('os.listdir', Mock(side_effect=[OSError, ['a', 'b']]))
|
||||
@patch('os.unlink', Mock(side_effect=OSError))
|
||||
@patch('os.remove', Mock())
|
||||
@patch('os.path.islink', Mock(side_effect=[True, False]))
|
||||
@patch('os.path.isfile', Mock(return_value=True))
|
||||
def test_cleanup_archive_status(self):
|
||||
self.p.cleanup_archive_status()
|
||||
self.p.cleanup_archive_status()
|
||||
|
||||
@patch('os.unlink', Mock())
|
||||
@patch('os.path.isfile', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'single_user_mode', Mock(return_value=0))
|
||||
def test_fix_cluster_state(self):
|
||||
self.assertTrue(self.p.fix_cluster_state())
|
||||
|
||||
def test__update_postmaster_cached_info(self):
|
||||
with open(os.path.join(self.data_dir, 'postmaster.pid'), 'w') as f:
|
||||
f.write('1\n\n1\n')
|
||||
self.p.read_pid_file()
|
||||
with open(os.path.join(self.data_dir, 'postmaster.pid'), 'w') as f:
|
||||
f.write('a\n\n1\n')
|
||||
self.p.read_pid_file()
|
||||
|
||||
+20
-50
@@ -2,67 +2,41 @@ import psycopg2
|
||||
import subprocess
|
||||
import unittest
|
||||
|
||||
from mock import Mock, PropertyMock, patch, mock_open
|
||||
from patroni.scripts import wale_restore
|
||||
from mock import Mock, MagicMock, patch, mock_open
|
||||
from patroni.scripts.wale_restore import WALERestore, main as _main, get_major_version
|
||||
from six.moves import builtins
|
||||
from test_postgresql import MockConnect, psycopg2_connect
|
||||
|
||||
|
||||
wale_output_header = (
|
||||
b'name\tlast_modified\t'
|
||||
b'expanded_size_bytes\t'
|
||||
b'wal_segment_backup_start\twal_segment_offset_backup_start\t'
|
||||
b'wal_segment_backup_stop\twal_segment_offset_backup_stop\n'
|
||||
)
|
||||
|
||||
wale_output_values = (
|
||||
b'base_00000001000000000000007F_00000040\t2015-05-18T10:13:25.000Z\t'
|
||||
b'167772160\t'
|
||||
b'00000001000000000000007F\t00000040\t'
|
||||
b'00000001000000000000007F\t00000240\n'
|
||||
)
|
||||
|
||||
wale_output = wale_output_header + wale_output_values
|
||||
|
||||
wale_restore.RETRY_SLEEP_INTERVAL = 0.001 # Speed up retries
|
||||
WALE_TEST_RETRIES = 2
|
||||
wale_output = b'name last_modified expanded_size_bytes wal_segment_backup_start ' +\
|
||||
b'wal_segment_offset_backup_start wal_segment_backup_stop wal_segment_offset_backup_stop\n' +\
|
||||
b'base_00000001000000000000007F_00000040 2015-05-18T10:13:25.000Z 167772160 ' +\
|
||||
b'00000001000000000000007F 00000040 00000001000000000000007F 00000240\n'
|
||||
|
||||
|
||||
@patch('os.access', Mock(return_value=True))
|
||||
@patch('os.makedirs', Mock(return_value=True))
|
||||
@patch('os.path.exists', Mock(return_value=True))
|
||||
@patch('os.path.isdir', Mock(return_value=True))
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('psycopg2.extensions.cursor', Mock(autospec=True))
|
||||
@patch('psycopg2.extensions.connection', Mock(autospec=True))
|
||||
@patch('psycopg2.connect', MagicMock(autospec=True))
|
||||
@patch('subprocess.check_output', Mock(return_value=wale_output))
|
||||
class TestWALERestore(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.wale_restore = WALERestore('batman', '/data', 'host=batman port=5432 user=batman',
|
||||
'/etc', 100, 100, 1, 0, WALE_TEST_RETRIES)
|
||||
self.wale_restore = WALERestore("batman", "/data", "host=batman port=5432 user=batman",
|
||||
"/etc", 100, 100, 1, 0, 1)
|
||||
|
||||
def test_should_use_s3_to_create_replica(self):
|
||||
self.assertTrue(self.wale_restore.should_use_s3_to_create_replica())
|
||||
with patch.object(MockConnect, 'server_version', PropertyMock(return_value=100000)):
|
||||
self.assertTrue(self.wale_restore.should_use_s3_to_create_replica())
|
||||
|
||||
with patch('subprocess.check_output', Mock(return_value=wale_output.replace(b'167772160', b'1'))):
|
||||
self.assertFalse(self.wale_restore.should_use_s3_to_create_replica())
|
||||
|
||||
with patch('psycopg2.connect', Mock(side_effect=psycopg2.Error("foo"))):
|
||||
save_no_master = self.wale_restore.no_master
|
||||
save_master_connection = self.wale_restore.master_connection
|
||||
|
||||
self.assertFalse(self.wale_restore.should_use_s3_to_create_replica())
|
||||
|
||||
with patch('time.sleep', Mock(return_value=None)) as mock_sleep:
|
||||
self.wale_restore.no_master = 1
|
||||
assert self.wale_restore.should_use_s3_to_create_replica()
|
||||
# verify retries
|
||||
mock_sleep.assert_has_calls(
|
||||
[((wale_restore.RETRY_SLEEP_INTERVAL,),)] * WALE_TEST_RETRIES
|
||||
)
|
||||
|
||||
self.wale_restore.no_master = 1
|
||||
self.assertTrue(self.wale_restore.should_use_s3_to_create_replica()) # this would do 2 retries 1 sec each
|
||||
self.wale_restore.master_connection = ''
|
||||
self.assertTrue(self.wale_restore.should_use_s3_to_create_replica())
|
||||
|
||||
@@ -71,9 +45,10 @@ class TestWALERestore(unittest.TestCase):
|
||||
|
||||
with patch('subprocess.check_output', Mock(side_effect=subprocess.CalledProcessError(1, "cmd", "foo"))):
|
||||
self.assertFalse(self.wale_restore.should_use_s3_to_create_replica())
|
||||
with patch('subprocess.check_output', Mock(return_value=wale_output_header)):
|
||||
with patch('subprocess.check_output', Mock(return_value=wale_output.split(b'\n')[0])):
|
||||
self.assertFalse(self.wale_restore.should_use_s3_to_create_replica())
|
||||
with patch('subprocess.check_output', Mock(return_value=wale_output + wale_output_values)):
|
||||
with patch('subprocess.check_output',
|
||||
Mock(return_value=wale_output.replace(b' wal_segment_offset_backup_stop', b''))):
|
||||
self.assertFalse(self.wale_restore.should_use_s3_to_create_replica())
|
||||
with patch('subprocess.check_output',
|
||||
Mock(return_value=wale_output.replace(b'expanded_size_bytes', b'expanded_size_foo'))):
|
||||
@@ -95,22 +70,17 @@ class TestWALERestore(unittest.TestCase):
|
||||
with patch.object(self.wale_restore, 'should_use_s3_to_create_replica', Mock(return_value=True)):
|
||||
with patch.object(self.wale_restore, 'create_replica_with_s3', Mock(return_value=0)):
|
||||
self.assertEqual(self.wale_restore.run(), 0)
|
||||
with patch.object(self.wale_restore, 'should_use_s3_to_create_replica', Mock(return_value=False)):
|
||||
self.assertEqual(self.wale_restore.run(), 2)
|
||||
with patch.object(self.wale_restore, 'should_use_s3_to_create_replica', Mock(return_value=None)):
|
||||
self.assertEqual(self.wale_restore.run(), 1)
|
||||
with patch.object(self.wale_restore, 'should_use_s3_to_create_replica', Mock(side_effect=Exception)):
|
||||
self.assertEqual(self.wale_restore.run(), 2)
|
||||
with patch.object(self.wale_restore, 'should_use_s3_to_create_replica', Mock(return_value=None)):
|
||||
self.assertEqual(self.wale_restore.run(), 1)
|
||||
with patch.object(self.wale_restore, 'should_use_s3_to_create_replica', Mock(side_effect=Exception)):
|
||||
self.assertEqual(self.wale_restore.run(), 2)
|
||||
|
||||
@patch('sys.exit', Mock())
|
||||
def test_main(self):
|
||||
with patch.object(WALERestore, 'run', Mock(return_value=0)):
|
||||
self.assertEqual(_main(), 0)
|
||||
|
||||
with patch.object(WALERestore, 'run', Mock(return_value=1)), \
|
||||
patch('time.sleep', Mock(return_value=None)) as mock_sleep:
|
||||
with patch.object(WALERestore, 'run', Mock(return_value=1)):
|
||||
self.assertEqual(_main(), 1)
|
||||
assert mock_sleep.call_count == WALE_TEST_RETRIES
|
||||
|
||||
@patch('os.path.isfile', Mock(return_value=True))
|
||||
def test_get_major_version(self):
|
||||
|
||||
@@ -1,217 +0,0 @@
|
||||
import ctypes
|
||||
import patroni.watchdog.linux as linuxwd
|
||||
import sys
|
||||
import unittest
|
||||
|
||||
from mock import patch, Mock, PropertyMock
|
||||
from patroni.watchdog import Watchdog, WatchdogError
|
||||
from patroni.watchdog.base import NullWatchdog
|
||||
from patroni.watchdog.linux import LinuxWatchdogDevice
|
||||
|
||||
|
||||
class MockDevice(object):
|
||||
def __init__(self, fd, filename, flag):
|
||||
self.fd = fd
|
||||
self.filename = filename
|
||||
self.flag = flag
|
||||
self.timeout = 60
|
||||
self.open = True
|
||||
self.writes = []
|
||||
|
||||
|
||||
mock_devices = [None]
|
||||
|
||||
|
||||
def mock_open(filename, flag):
|
||||
fd = len(mock_devices)
|
||||
mock_devices.append(MockDevice(fd, filename, flag))
|
||||
return fd
|
||||
|
||||
|
||||
def mock_ioctl(fd, op, arg=None, mutate_flag=False):
|
||||
assert 0 < fd < len(mock_devices)
|
||||
dev = mock_devices[fd]
|
||||
sys.stderr.write("Ioctl %d %d %r\n" % (fd, op, arg))
|
||||
if op == linuxwd.WDIOC_GETSUPPORT:
|
||||
sys.stderr.write("Get support\n")
|
||||
assert(mutate_flag is True)
|
||||
arg.options = sum(map(linuxwd.WDIOF.get, ['SETTIMEOUT', 'KEEPALIVEPING']))
|
||||
arg.identity = (ctypes.c_ubyte*32)(*map(ord, 'Mock Watchdog'))
|
||||
elif op == linuxwd.WDIOC_GETTIMEOUT:
|
||||
arg.value = dev.timeout
|
||||
elif op == linuxwd.WDIOC_SETTIMEOUT:
|
||||
sys.stderr.write("Set timeout called with %s\n" % arg.value)
|
||||
assert 0 < arg.value < 65535
|
||||
dev.timeout = arg.value - 1
|
||||
else:
|
||||
raise Exception("Unknown op %d", op)
|
||||
return 0
|
||||
|
||||
|
||||
def mock_write(fd, string):
|
||||
assert 0 < fd < len(mock_devices)
|
||||
assert len(string) == 1
|
||||
assert mock_devices[fd].open
|
||||
mock_devices[fd].writes.append(string)
|
||||
|
||||
|
||||
def mock_close(fd):
|
||||
assert 0 < fd < len(mock_devices)
|
||||
assert mock_devices[fd].open
|
||||
mock_devices[fd].open = False
|
||||
|
||||
|
||||
@patch('os.open', mock_open)
|
||||
@patch('os.write', mock_write)
|
||||
@patch('os.close', mock_close)
|
||||
@patch('fcntl.ioctl', mock_ioctl)
|
||||
class TestWatchdog(unittest.TestCase):
|
||||
def setUp(self):
|
||||
mock_devices[:] = [None]
|
||||
|
||||
@patch('platform.system', Mock(return_value='Linux'))
|
||||
@patch.object(LinuxWatchdogDevice, 'can_be_disabled', PropertyMock(return_value=True))
|
||||
def test_unsafe_timeout_disable_watchdog_and_exit(self):
|
||||
watchdog = Watchdog({'ttl': 30, 'loop_wait': 15, 'watchdog': {'mode': 'required', 'safety_margin': -1}})
|
||||
self.assertEquals(watchdog.activate(), False)
|
||||
self.assertEquals(watchdog.is_running, False)
|
||||
|
||||
@patch('platform.system', Mock(return_value='Linux'))
|
||||
@patch.object(LinuxWatchdogDevice, 'get_timeout', Mock(return_value=16))
|
||||
def test_timeout_does_not_ensure_safe_termination(self):
|
||||
Watchdog({'ttl': 30, 'loop_wait': 15, 'watchdog': {'mode': 'auto', 'safety_margin': -1}}).activate()
|
||||
self.assertEquals(len(mock_devices), 2)
|
||||
|
||||
@patch('platform.system', Mock(return_value='Linux'))
|
||||
@patch.object(Watchdog, 'is_running', PropertyMock(return_value=False))
|
||||
def test_watchdog_not_activated(self):
|
||||
self.assertFalse(Watchdog({'ttl': 30, 'loop_wait': 10, 'watchdog': {'mode': 'required'}}).activate())
|
||||
|
||||
@patch('platform.system', Mock(return_value='Linux'))
|
||||
@patch.object(LinuxWatchdogDevice, 'is_running', PropertyMock(return_value=False))
|
||||
def test_watchdog_activate(self):
|
||||
with patch.object(LinuxWatchdogDevice, 'open', Mock(side_effect=WatchdogError(''))):
|
||||
self.assertTrue(Watchdog({'ttl': 30, 'loop_wait': 10, 'watchdog': {'mode': 'auto'}}).activate())
|
||||
self.assertFalse(Watchdog({'ttl': 30, 'loop_wait': 10, 'watchdog': {'mode': 'required'}}).activate())
|
||||
|
||||
@patch('platform.system', Mock(return_value='Linux'))
|
||||
def test_basic_operation(self):
|
||||
watchdog = Watchdog({'ttl': 30, 'loop_wait': 10, 'watchdog': {'mode': 'required'}})
|
||||
watchdog.activate()
|
||||
|
||||
self.assertEquals(len(mock_devices), 2)
|
||||
device = mock_devices[-1]
|
||||
self.assertTrue(device.open)
|
||||
|
||||
self.assertEquals(device.timeout, 24)
|
||||
|
||||
watchdog.keepalive()
|
||||
self.assertEquals(len(device.writes), 1)
|
||||
|
||||
watchdog.disable()
|
||||
self.assertFalse(device.open)
|
||||
self.assertEquals(device.writes[-1], b'V')
|
||||
|
||||
def test_invalid_timings(self):
|
||||
watchdog = Watchdog({'ttl': 30, 'loop_wait': 20, 'watchdog': {'mode': 'automatic', 'safety_margin': -1}})
|
||||
watchdog.activate()
|
||||
self.assertEquals(len(mock_devices), 1)
|
||||
self.assertFalse(watchdog.is_running)
|
||||
|
||||
def test_parse_mode(self):
|
||||
with patch('patroni.watchdog.base.logger.warning', new_callable=Mock()) as warning_mock:
|
||||
watchdog = Watchdog({'ttl': 30, 'loop_wait': 10, 'watchdog': {'mode': 'bad'}})
|
||||
self.assertEquals(watchdog.config.mode, 'off')
|
||||
warning_mock.assert_called_once()
|
||||
|
||||
@patch('platform.system', Mock(return_value='Unknown'))
|
||||
def test_unsupported_platform(self):
|
||||
self.assertRaises(SystemExit, Watchdog, {'ttl': 30, 'loop_wait': 10,
|
||||
'watchdog': {'mode': 'required', 'driver': 'bad'}})
|
||||
|
||||
def test_exceptions(self):
|
||||
wd = Watchdog({'ttl': 30, 'loop_wait': 10, 'watchdog': {'mode': 'bad'}})
|
||||
wd.impl.close = wd.impl.keepalive = Mock(side_effect=WatchdogError(''))
|
||||
self.assertTrue(wd.activate())
|
||||
self.assertIsNone(wd.keepalive())
|
||||
self.assertIsNone(wd.disable())
|
||||
|
||||
@patch('platform.system', Mock(return_value='Linux'))
|
||||
def test_config_reload(self):
|
||||
watchdog = Watchdog({'ttl': 30, 'loop_wait': 15, 'watchdog': {'mode': 'required'}})
|
||||
self.assertTrue(watchdog.activate())
|
||||
self.assertTrue(watchdog.is_running)
|
||||
|
||||
watchdog.reload_config({'ttl': 30, 'loop_wait': 15, 'watchdog': {'mode': 'off'}})
|
||||
self.assertFalse(watchdog.is_running)
|
||||
|
||||
watchdog.reload_config({'ttl': 30, 'loop_wait': 15, 'watchdog': {'mode': 'required'}})
|
||||
self.assertFalse(watchdog.is_running)
|
||||
watchdog.keepalive()
|
||||
self.assertTrue(watchdog.is_running)
|
||||
|
||||
watchdog.disable()
|
||||
watchdog.reload_config({'ttl': 30, 'loop_wait': 15, 'watchdog': {'mode': 'required', 'driver': 'unknown'}})
|
||||
self.assertFalse(watchdog.is_healthy)
|
||||
|
||||
self.assertFalse(watchdog.activate())
|
||||
watchdog.reload_config({'ttl': 30, 'loop_wait': 15, 'watchdog': {'mode': 'required'}})
|
||||
self.assertFalse(watchdog.is_running)
|
||||
watchdog.keepalive()
|
||||
self.assertTrue(watchdog.is_running)
|
||||
|
||||
watchdog.reload_config({'ttl': 60, 'loop_wait': 15, 'watchdog': {'mode': 'required'}})
|
||||
watchdog.keepalive()
|
||||
|
||||
|
||||
class TestNullWatchdog(unittest.TestCase):
|
||||
|
||||
def test_basics(self):
|
||||
watchdog = NullWatchdog()
|
||||
self.assertTrue(watchdog.can_be_disabled)
|
||||
self.assertRaises(WatchdogError, watchdog.set_timeout, 1)
|
||||
self.assertEquals(watchdog.describe(), 'NullWatchdog')
|
||||
self.assertIsInstance(NullWatchdog.from_config({}), NullWatchdog)
|
||||
|
||||
|
||||
class TestLinuxWatchdogDevice(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.impl = LinuxWatchdogDevice.from_config({})
|
||||
|
||||
@patch('os.open', Mock(return_value=3))
|
||||
@patch('os.write', Mock(side_effect=OSError))
|
||||
@patch('fcntl.ioctl', Mock(return_value=0))
|
||||
def test_basics(self):
|
||||
self.impl.open()
|
||||
try:
|
||||
if self.impl.get_support().has_foo:
|
||||
self.assertFail()
|
||||
except Exception as e:
|
||||
self.assertTrue(isinstance(e, AttributeError))
|
||||
self.assertRaises(WatchdogError, self.impl.close)
|
||||
self.assertRaises(WatchdogError, self.impl.keepalive)
|
||||
self.assertRaises(WatchdogError, self.impl.set_timeout, -1)
|
||||
|
||||
@patch('os.open', Mock(return_value=3))
|
||||
@patch('fcntl.ioctl', Mock(side_effect=OSError))
|
||||
def test__ioctl(self):
|
||||
self.assertRaises(WatchdogError, self.impl.get_support)
|
||||
self.impl.open()
|
||||
self.assertRaises(WatchdogError, self.impl.get_support)
|
||||
|
||||
def test_is_healthy(self):
|
||||
self.assertFalse(self.impl.is_healthy)
|
||||
|
||||
@patch('os.open', Mock(return_value=3))
|
||||
@patch('fcntl.ioctl', Mock(side_effect=OSError))
|
||||
def test_error_handling(self):
|
||||
self.impl.open()
|
||||
self.assertRaises(WatchdogError, self.impl.get_timeout)
|
||||
self.assertRaises(WatchdogError, self.impl.set_timeout, 10)
|
||||
# We still try to output a reasonable string even if getting info errors
|
||||
self.assertEquals(self.impl.describe(), "Linux watchdog device")
|
||||
|
||||
@patch('os.open', Mock(side_effect=OSError))
|
||||
def test_open(self):
|
||||
self.assertRaises(WatchdogError, self.impl.open)
|
||||
Reference in New Issue
Block a user