Compare commits

..
1 Commits
Author SHA1 Message Date
Alexander Kukushkin 76ed5784bb arm64 compatibility 2022-06-24 11:36:20 +02:00
74 changed files with 336 additions and 969 deletions
-1
View File
@@ -1 +0,0 @@
blank_issues_enabled: false
+1 -6
View File
@@ -30,7 +30,6 @@ def install_requirements(what):
requirements.append(r) requirements.append(r)
subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'pip']) subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'pip'])
subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'wheel'])
r = subprocess.call([sys.executable, '-m', 'pip', 'install'] + requirements) r = subprocess.call([sys.executable, '-m', 'pip', 'install'] + requirements)
s = subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'setuptools']) s = subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'setuptools'])
return s | r return s | r
@@ -46,14 +45,10 @@ def install_packages(what):
packages['exhibitor'] = packages['zookeeper'] packages['exhibitor'] = packages['zookeeper']
packages = packages.get(what, []) packages = packages.get(what, [])
ver = versions.get(what) ver = versions.get(what)
subprocess.call(['sudo', 'apt-get', 'update', '-y'])
subprocess.call(['sudo', 'apt-get', 'install', '-y', 'wget', 'ca-certificates', 'gnupg', 'expect-dev'])
subprocess.call(['sudo', 'sh', '-c', "wget -qO - https://www.postgresql.org/media/keys/ACCC4CF8.asc"
" | gpg --dearmor > /etc/apt/trusted.gpg.d/apt.postgresql.org.gpg"])
subprocess.call(['sudo', 'sed', '-i', 's/pgdg main.*$/pgdg main {0}/'.format(ver), subprocess.call(['sudo', 'sed', '-i', 's/pgdg main.*$/pgdg main {0}/'.format(ver),
'/etc/apt/sources.list.d/pgdg.list']) '/etc/apt/sources.list.d/pgdg.list'])
subprocess.call(['sudo', 'apt-get', 'update', '-y']) subprocess.call(['sudo', 'apt-get', 'update', '-y'])
return subprocess.call(['sudo', 'apt-get', 'install', '-y', 'postgresql-' + ver] + packages) return subprocess.call(['sudo', 'apt-get', 'install', '-y', 'postgresql-' + ver, 'expect-dev', 'wget'] + packages)
def get_file(url, name): def get_file(url, name):
+1 -1
View File
@@ -1 +1 @@
versions = {'etcd': '9.6', 'etcd3': '14', 'consul': '13', 'exhibitor': '12', 'raft': '11', 'kubernetes': '15'} versions = {'etcd': '9.6', 'etcd3': '14', 'consul': '13', 'exhibitor': '12', 'raft': '11', 'kubernetes': '14'}
-41
View File
@@ -1,41 +0,0 @@
name: Publish Patroni distributions to PyPI and TestPyPI
on:
push:
tags:
- 'v[0-9]+.[0-9]+.[0-9]+'
release:
types:
- published
jobs:
build-n-publish:
name: Build and publish Patroni distributions to PyPI and TestPyPI
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@master
- name: Set up Python 3.9
uses: actions/setup-python@v4
with:
python-version: 3.9
- name: Install dependencies
run: python .github/workflows/install_deps.py
- name: Run tests and flake8
run: python .github/workflows/run_tests.py
- name: Build a binary wheel and a source tarball
run: python setup.py sdist bdist_wheel
- name: Publish distribution to Test PyPI
if: github.event_name == 'push'
uses: pypa/[email protected]
with:
password: ${{ secrets.TEST_PYPI_API_TOKEN }}
repository_url: https://test.pypi.org/legacy/
- name: Publish distribution to PyPI
if: github.event_name == 'release'
uses: pypa/[email protected]
with:
password: ${{ secrets.PYPI_API_TOKEN }}
+3 -2
View File
@@ -28,15 +28,16 @@ def main():
version = versions.get(what) version = versions.get(what)
path = '/usr/lib/postgresql/{0}/bin:.'.format(version) path = '/usr/lib/postgresql/{0}/bin:.'.format(version)
unbuffer = ['timeout', '900', 'unbuffer'] unbuffer = ['timeout', '900', 'unbuffer']
args = ['--tags=-skip'] if what == 'etcd' else []
else: else:
path = os.path.abspath(os.path.join('pgsql', 'bin')) path = os.path.abspath(os.path.join('pgsql', 'bin'))
if sys.platform == 'darwin': if sys.platform == 'darwin':
path += ':.' path += ':.'
unbuffer = [] args = unbuffer = []
env['PATH'] = path + os.pathsep + env['PATH'] env['PATH'] = path + os.pathsep + env['PATH']
env['DCS'] = what env['DCS'] = what
ret = subprocess.call(unbuffer + [sys.executable, '-m', 'behave'], env=env) ret = subprocess.call(unbuffer + [sys.executable, '-m', 'behave'] + args, env=env)
if ret != 0: if ret != 0:
if subprocess.call('grep . features/output/*_failed/*postgres?.*', shell=True) != 0: if subprocess.call('grep . features/output/*_failed/*postgres?.*', shell=True) != 0:
+11 -11
View File
@@ -17,9 +17,9 @@ jobs:
os: [ubuntu, windows, macos] os: [ubuntu, windows, macos]
steps: steps:
- uses: actions/checkout@v3 - uses: actions/checkout@v1
- name: Set up Python 2.7 - name: Set up Python 2.7
uses: actions/setup-python@v4 uses: actions/setup-python@v2
with: with:
python-version: 2.7 python-version: 2.7
if: matrix.os != 'windows' if: matrix.os != 'windows'
@@ -31,7 +31,7 @@ jobs:
if: matrix.os != 'windows' if: matrix.os != 'windows'
- name: Set up Python 3.6 - name: Set up Python 3.6
uses: actions/setup-python@v4 uses: actions/setup-python@v2
with: with:
python-version: 3.6 python-version: 3.6
- name: Install dependencies - name: Install dependencies
@@ -40,7 +40,7 @@ jobs:
run: python .github/workflows/run_tests.py run: python .github/workflows/run_tests.py
- name: Set up Python 3.7 - name: Set up Python 3.7
uses: actions/setup-python@v4 uses: actions/setup-python@v2
with: with:
python-version: 3.7 python-version: 3.7
- name: Install dependencies - name: Install dependencies
@@ -49,7 +49,7 @@ jobs:
run: python .github/workflows/run_tests.py run: python .github/workflows/run_tests.py
- name: Set up Python 3.8 - name: Set up Python 3.8
uses: actions/setup-python@v4 uses: actions/setup-python@v2
with: with:
python-version: 3.8 python-version: 3.8
- name: Install dependencies - name: Install dependencies
@@ -58,7 +58,7 @@ jobs:
run: python .github/workflows/run_tests.py run: python .github/workflows/run_tests.py
- name: Set up Python 3.9 - name: Set up Python 3.9
uses: actions/setup-python@v4 uses: actions/setup-python@v2
with: with:
python-version: 3.9 python-version: 3.9
- name: Install dependencies - name: Install dependencies
@@ -67,7 +67,7 @@ jobs:
run: python .github/workflows/run_tests.py run: python .github/workflows/run_tests.py
- name: Set up Python 3.10 - name: Set up Python 3.10
uses: actions/setup-python@v4 uses: actions/setup-python@v2
with: with:
python-version: '3.10' python-version: '3.10'
- name: Install dependencies - name: Install dependencies
@@ -115,9 +115,9 @@ jobs:
dcs: etcd3 dcs: etcd3
steps: steps:
- uses: actions/checkout@v3 - uses: actions/checkout@v1
- name: Set up Python - name: Set up Python
uses: actions/setup-python@v4 uses: actions/setup-python@v2
with: with:
python-version: ${{ matrix.python-version }} python-version: ${{ matrix.python-version }}
- name: Add postgresql apt repo - name: Add postgresql apt repo
@@ -127,7 +127,7 @@ jobs:
run: python .github/workflows/install_deps.py run: python .github/workflows/install_deps.py
- name: Run behave tests - name: Run behave tests
run: python .github/workflows/run_tests.py run: python .github/workflows/run_tests.py
- uses: actions/setup-python@v4 - uses: actions/setup-python@v2
with: with:
python-version: '3.10' python-version: '3.10'
- name: Install coveralls - name: Install coveralls
@@ -144,7 +144,7 @@ jobs:
needs: [unit, behave] needs: [unit, behave]
runs-on: ubuntu-latest runs-on: ubuntu-latest
steps: steps:
- uses: actions/setup-python@v4 - uses: actions/setup-python@v2
- run: python -m pip install coveralls - run: python -m pip install coveralls
- run: python -m coveralls --service=github --finish - run: python -m coveralls --service=github --finish
env: env:
-2
View File
@@ -1,2 +0,0 @@
# global owners
* @CyberDem0n @hughcapet
+6 -29
View File
@@ -1,6 +1,6 @@
## This Dockerfile is meant to aid in the building and debugging patroni whilst developing on your local machine ## This Dockerfile is meant to aid in the building and debugging patroni whilst developing on your local machine
## It has all the necessary components to play/debug with a single node appliance, running etcd ## It has all the necessary components to play/debug with a single node appliance, running etcd
ARG PG_MAJOR=15 ARG PG_MAJOR=14
ARG COMPRESS=false ARG COMPRESS=false
ARG PGHOME=/home/postgres ARG PGHOME=/home/postgres
ARG PGDATA=$PGHOME/data ARG PGDATA=$PGHOME/data
@@ -14,7 +14,7 @@ ARG PGDATA
ARG LC_ALL ARG LC_ALL
ARG LANG ARG LANG
ENV ETCDVERSION=3.3.13 CONFDVERSION=0.16.0 ENV ETCDVERSION=3.5.4 CONFDVERSION=0.16.0
RUN set -ex \ RUN set -ex \
&& export DEBIAN_FRONTEND=noninteractive \ && export DEBIAN_FRONTEND=noninteractive \
@@ -50,11 +50,11 @@ RUN set -ex \
&& chown -R postgres:postgres /var/log \ && chown -R postgres:postgres /var/log \
\ \
# Download etcd # Download etcd
&& curl -sL https://github.com/coreos/etcd/releases/download/v${ETCDVERSION}/etcd-v${ETCDVERSION}-linux-amd64.tar.gz \ && curl -L https://github.com/coreos/etcd/releases/download/v${ETCDVERSION}/etcd-v${ETCDVERSION}-linux-$(dpkg --print-architecture).tar.gz \
| tar xz -C /usr/local/bin --strip=1 --wildcards --no-anchored etcd etcdctl \ | tar xz -C /usr/local/bin --strip=1 --wildcards --no-anchored etcd etcdctl \
\ \
# Download confd # Download confd
&& curl -sL https://github.com/kelseyhightower/confd/releases/download/v${CONFDVERSION}/confd-${CONFDVERSION}-linux-amd64 \ && curl -sL https://github.com/kelseyhightower/confd/releases/download/v${CONFDVERSION}/confd-${CONFDVERSION}-linux-$(dpkg --print-architecture) \
> /usr/local/bin/confd && chmod +x /usr/local/bin/confd \ > /usr/local/bin/confd && chmod +x /usr/local/bin/confd \
\ \
# Clean up all useless packages and some files # Clean up all useless packages and some files
@@ -89,35 +89,12 @@ RUN set -ex \
# /var/lib/dpkg/info/* \ # /var/lib/dpkg/info/* \
&& find /usr/bin -xtype l -delete \ && find /usr/bin -xtype l -delete \
&& find /var/log -type f -exec truncate --size 0 {} \; \ && find /var/log -type f -exec truncate --size 0 {} \; \
&& find /usr/lib/python3/dist-packages -name '*test*' | xargs rm -fr \ && find /usr/lib/python3/dist-packages -name '*test*' | xargs rm -fr
&& find /lib/x86_64-linux-gnu/security -type f ! -name pam_env.so ! -name pam_permit.so ! -name pam_unix.so -delete
# perform compression if it is necessary
ARG COMPRESS
RUN if [ "$COMPRESS" = "true" ]; then \
set -ex \
# Allow certain sudo commands from postgres
&& echo 'postgres ALL=(ALL) NOPASSWD: /bin/tar xpJf /a.tar.xz -C /, /bin/rm /a.tar.xz, /bin/ln -snf dash /bin/sh' >> /etc/sudoers \
&& ln -snf busybox /bin/sh \
&& files="/bin/sh /usr/bin/sudo /usr/lib/sudo/sudoers.so /lib/x86_64-linux-gnu/security/pam_*.so" \
&& libs="$(ldd $files | awk '{print $3;}' | grep '^/' | sort -u) /lib/x86_64-linux-gnu/ld-linux-x86-64.so.* /lib/x86_64-linux-gnu/libnsl.so.* /lib/x86_64-linux-gnu/libnss_compat.so.*" \
&& (echo /var/run $files $libs | tr ' ' '\n' && realpath $files $libs) | sort -u | sed 's/^\///' > /exclude \
&& find /etc/alternatives -xtype l -delete \
&& save_dirs="usr lib var bin sbin etc/ssl etc/init.d etc/alternatives etc/apt" \
&& XZ_OPT=-e9v tar -X /exclude -cpJf a.tar.xz $save_dirs \
# we call "cat /exclude" to avoid including files from the $save_dirs that are also among
# the exceptions listed in the /exclude, as "uniq -u" eliminates all non-unique lines.
# By calling "cat /exclude" a second time we guarantee that there will be at least two lines
# for each exception and therefore they will be excluded from the output passed to 'rm'.
&& /bin/busybox sh -c "(find $save_dirs -not -type d && cat /exclude /exclude && echo exclude) | sort | uniq -u | xargs /bin/busybox rm" \
&& /bin/busybox --install -s \
&& /bin/busybox sh -c "find $save_dirs -type d -depth -exec rmdir -p {} \; 2> /dev/null"; \
fi
FROM scratch FROM scratch
COPY --from=builder / / COPY --from=builder / /
LABEL maintainer="Alexander Kukushkin <akukushkin@microsoft.com>" LABEL maintainer="Alexander Kukushkin <alexander.kukushkin@zalando.de>"
ARG PG_MAJOR ARG PG_MAJOR
ARG COMPRESS ARG COMPRESS
+3 -2
View File
@@ -1,2 +1,3 @@
Alexander Kukushkin <akukushkin@microsoft.com> Alexander Kukushkin <alexander.kukushkin@zalando.de>
Polina Bungina <polina.bungina@zalando.de> Feike Steenbergen <feike.steenbergen@zalando.de>
Oleksii Kliukin <[email protected]>
+1 -1
View File
@@ -12,7 +12,7 @@ Patroni is a template for you to create your own customized, high-availability s
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely.
Currently supported PostgreSQL versions: 9.3 to 15. Currently supported PostgreSQL versions: 9.3 to 14.
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the `Kubernetes <https://github.com/zalando/patroni/blob/master/docs/kubernetes.rst>`__ chapter of the Patroni documentation. **Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the `Kubernetes <https://github.com/zalando/patroni/blob/master/docs/kubernetes.rst>`__ chapter of the Patroni documentation.
+1
View File
@@ -15,6 +15,7 @@ services:
ETCD_INITIAL_CLUSTER: etcd1=http://etcd1:2380,etcd2=http://etcd2:2380,etcd3=http://etcd3:2380 ETCD_INITIAL_CLUSTER: etcd1=http://etcd1:2380,etcd2=http://etcd2:2380,etcd3=http://etcd3:2380
ETCD_INITIAL_CLUSTER_STATE: new ETCD_INITIAL_CLUSTER_STATE: new
ETCD_INITIAL_CLUSTER_TOKEN: tutorial ETCD_INITIAL_CLUSTER_TOKEN: tutorial
ETCD_UNSUPPORTED_ARCH: arm64
container_name: demo-etcd1 container_name: demo-etcd1
hostname: etcd1 hostname: etcd1
command: etcd -name etcd1 -initial-advertise-peer-urls http://etcd1:2380 command: etcd -name etcd1 -initial-advertise-peer-urls http://etcd1:2380
+1 -2
View File
@@ -123,12 +123,11 @@ PostgreSQL
---------- ----------
- **PATRONI\_POSTGRESQL\_LISTEN**: IP address + port that Postgres listens to. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node. - **PATRONI\_POSTGRESQL\_LISTEN**: IP address + port that Postgres listens to. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node.
- **PATRONI\_POSTGRESQL\_CONNECT\_ADDRESS**: IP address + port through which Postgres is accessible from other nodes and applications. - **PATRONI\_POSTGRESQL\_CONNECT\_ADDRESS**: IP address + port through which Postgres is accessible from other nodes and applications.
- **PATRONI\_POSTGRESQL\_PROXY\_ADDRESS**: IP address + port through which a connection pool (e.g. pgbouncer) running next to Postgres is accessible. The value is written to the member key in DCS as ``proxy_url`` and could be used/useful for service discovery.
- **PATRONI\_POSTGRESQL\_DATA\_DIR**: The location of the Postgres data directory, either existing or to be initialized by Patroni. - **PATRONI\_POSTGRESQL\_DATA\_DIR**: The location of the Postgres data directory, either existing or to be initialized by Patroni.
- **PATRONI\_POSTGRESQL\_CONFIG\_DIR**: The location of the Postgres configuration directory, defaults to the data directory. Must be writable by Patroni. - **PATRONI\_POSTGRESQL\_CONFIG\_DIR**: The location of the Postgres configuration directory, defaults to the data directory. Must be writable by Patroni.
- **PATRONI\_POSTGRESQL\_BIN_DIR**: Path to PostgreSQL binaries. (pg_ctl, pg_rewind, pg_basebackup, postgres) The default value is an empty string meaning that PATH environment variable will be used to find the executables. - **PATRONI\_POSTGRESQL\_BIN_DIR**: Path to PostgreSQL binaries. (pg_ctl, pg_rewind, pg_basebackup, postgres) The default value is an empty string meaning that PATH environment variable will be used to find the executables.
- **PATRONI\_POSTGRESQL\_PGPASS**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup and under some other circumstances. The location must be writable by Patroni. - **PATRONI\_POSTGRESQL\_PGPASS**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup and under some other circumstances. The location must be writable by Patroni.
- **PATRONI\_REPLICATION\_USERNAME**: replication username; the user will be created during initialization. Replicas will use this user to access the replication source via streaming replication - **PATRONI\_REPLICATION\_USERNAME**: replication username; the user will be created during initialization. Replicas will use this user to access master via streaming replication
- **PATRONI\_REPLICATION\_PASSWORD**: replication password; the user will be created during initialization. - **PATRONI\_REPLICATION\_PASSWORD**: replication password; the user will be created during initialization.
- **PATRONI\_REPLICATION\_SSLMODE**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``. - **PATRONI\_REPLICATION\_SSLMODE**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
- **PATRONI\_REPLICATION\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate. - **PATRONI\_REPLICATION\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
+1 -1
View File
@@ -109,7 +109,7 @@ Planning the Number of PostgreSQL Nodes
--------------------------------------- ---------------------------------------
Patroni/PostgreSQL nodes are decoupled from DCS nodes (except when Patroni implements RAFT on its own) and therefore Patroni/PostgreSQL nodes are decoupled from DCS nodes (except when Patroni implements RAFT on its own) and therefore
there is no requirement on the minimal number of nodes. Running a cluster consisting of one primary and one standby is there is no requirement on the minimal number of nodes. Running a cluster consisting of one master and one standby is
perfectly fine. You can add more standby nodes later. perfectly fine. You can add more standby nodes later.
Running and Configuring Running and Configuring
+11 -12
View File
@@ -17,21 +17,21 @@ Dynamic configuration is stored in the DCS (Distributed Configuration Store) and
- **maximum\_lag\_on\_failover**: the maximum bytes a follower may lag to be able to participate in leader election. - **maximum\_lag\_on\_failover**: the maximum bytes a follower may lag to be able to participate in leader election.
- **maximum\_lag\_on\_syncnode**: the maximum bytes a synchronous follower may lag before it is considered as an unhealthy candidate and swapped by healthy asynchronous follower. Patroni utilize the max replica lsn if there is more than one follower, otherwise it will use leader's current wal lsn. Default is -1, Patroni will not take action to swap synchronous unhealthy follower when the value is set to 0 or below. Please set the value high enough so Patroni won't swap synchrounous follower fequently during high transaction volume. - **maximum\_lag\_on\_syncnode**: the maximum bytes a synchronous follower may lag before it is considered as an unhealthy candidate and swapped by healthy asynchronous follower. Patroni utilize the max replica lsn if there is more than one follower, otherwise it will use leader's current wal lsn. Default is -1, Patroni will not take action to swap synchronous unhealthy follower when the value is set to 0 or below. Please set the value high enough so Patroni won't swap synchrounous follower fequently during high transaction volume.
- **max\_timelines\_history**: maximum number of timeline history items kept in DCS. Default value: 0. When set to 0, it keeps the full history in DCS. - **max\_timelines\_history**: maximum number of timeline history items kept in DCS. Default value: 0. When set to 0, it keeps the full history in DCS.
- **master\_start\_timeout**: the amount of time a primary is allowed to recover from failures before failover is triggered (in seconds). Default is 300 seconds. When set to 0 failover is done immediately after a crash is detected if possible. When using asynchronous replication a failover can cause lost transactions. Worst case failover time for primary failure is: loop\_wait + master\_start\_timeout + loop\_wait, unless master\_start\_timeout is zero, in which case it's just loop\_wait. Set the value according to your durability/availability tradeoff. - **master\_start\_timeout**: the amount of time a master is allowed to recover from failures before failover is triggered (in seconds). Default is 300 seconds. When set to 0 failover is done immediately after a crash is detected if possible. When using asynchronous replication a failover can cause lost transactions. Worst case failover time for master failure is: loop\_wait + master\_start\_timeout + loop\_wait, unless master\_start\_timeout is zero, in which case it's just loop\_wait. Set the value according to your durability/availability tradeoff.
- **master\_stop\_timeout**: The number of seconds Patroni is allowed to wait when stopping Postgres and effective only when synchronous_mode is enabled. When set to > 0 and the synchronous_mode is enabled, Patroni sends SIGKILL to the postmaster if the stop operation is running for more than the value set by master_stop_timeout. Set the value according to your durability/availability tradeoff. If the parameter is not set or set <= 0, master_stop_timeout does not apply. - **master\_stop\_timeout**: The number of seconds Patroni is allowed to wait when stopping Postgres and effective only when synchronous_mode is enabled. When set to > 0 and the synchronous_mode is enabled, Patroni sends SIGKILL to the postmaster if the stop operation is running for more than the value set by master_stop_timeout. Set the value according to your durability/availability tradeoff. If the parameter is not set or set <= 0, master_stop_timeout does not apply.
- **synchronous\_mode**: turns on synchronous replication mode. In this mode a replica will be chosen as synchronous and only the latest leader and synchronous replica are able to participate in leader election. Synchronous mode makes sure that successfully committed transactions will not be lost at failover, at the cost of losing availability for writes when Patroni cannot ensure transaction durability. See :ref:`replication modes documentation <replication_modes>` for details. - **synchronous\_mode**: turns on synchronous replication mode. In this mode a replica will be chosen as synchronous and only the latest leader and synchronous replica are able to participate in leader election. Synchronous mode makes sure that successfully committed transactions will not be lost at failover, at the cost of losing availability for writes when Patroni cannot ensure transaction durability. See :ref:`replication modes documentation <replication_modes>` for details.
- **synchronous\_mode\_strict**: prevents disabling synchronous replication if no synchronous replicas are available, blocking all client writes to the primary. See :ref:`replication modes documentation <replication_modes>` for details. - **synchronous\_mode\_strict**: prevents disabling synchronous replication if no synchronous replicas are available, blocking all client writes to the master. See :ref:`replication modes documentation <replication_modes>` for details.
- **postgresql**: - **postgresql**:
- **use\_pg\_rewind**: whether or not to use pg_rewind. Defaults to `false`. - **use\_pg\_rewind**: whether or not to use pg_rewind. Defaults to `false`.
- **use\_slots**: whether or not to use replication slots. Defaults to `true` on PostgreSQL 9.4+. - **use\_slots**: whether or not to use replication slots. Defaults to `true` on PostgreSQL 9.4+.
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower. There is no recovery.conf anymore in PostgreSQL 12, but you may continue using this section, because Patroni handles it transparently. - **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower. There is no recovery.conf anymore in PostgreSQL 12, but you may continue using this section, because Patroni handles it transparently.
- **parameters**: list of configuration settings for Postgres. - **parameters**: list of configuration settings for Postgres.
- **standby\_cluster**: if this section is defined, we want to bootstrap a standby cluster. - **standby\_cluster**: if this section is defined, we want to bootstrap a standby cluster.
- **host**: an address of remote node - **host**: an address of remote master
- **port**: a port of remote node - **port**: a port of remote master
- **primary\_slot\_name**: which slot on the remote node to use for replication. This parameter is optional, the default value is derived from the instance name (see function `slot_name_from_member_name`). - **primary\_slot\_name**: which slot on the remote master to use for replication. This parameter is optional, the default value is derived from the instance name (see function `slot_name_from_member_name`).
- **create\_replica\_methods**: an ordered list of methods that can be used to bootstrap standby leader from the remote primary, can be different from the list defined in :ref:`postgresql_settings` - **create\_replica\_methods**: an ordered list of methods that can be used to bootstrap standby leader from the remote master, can be different from the list defined in :ref:`postgresql_settings`
- **restore\_command**: command to restore WAL records from the remote primary to nodes in a standby cluster, can be different from the list defined in :ref:`postgresql_settings` - **restore\_command**: command to restore WAL records from the remote master to standby leader, can be different from the list defined in :ref:`postgresql_settings`
- **archive\_cleanup\_command**: cleanup command for standby leader - **archive\_cleanup\_command**: cleanup command for standby leader
- **recovery\_min\_apply\_delay**: how long to wait before actually apply WAL records on a standby leader - **recovery\_min\_apply\_delay**: how long to wait before actually apply WAL records on a standby leader
- **slots**: define permanent replication slots. These slots will be preserved during switchover/failover. The logical slots are copied from the primary to a standby with restart, and after that their position advanced every **loop_wait** seconds (if necessary). Copying logical slot files performed via ``libpq`` connection and using either rewind or superuser credentials (see **postgresql.authentication** section). There is always a chance that the logical slot position on the replica is a bit behind the former primary, therefore application should be prepared that some messages could be received the second time after the failover. The easiest way of doing so - tracking ``confirmed_flush_lsn``. Enabling permanent logical replication slots requires **postgresql.use_slots** to be set and will also automatically enable the ``hot_standby_feedback``. Since the failover of logical replication slots is unsafe on PostgreSQL 9.6 and older and PostgreSQL version 10 is missing some important functions, the feature only works with PostgreSQL 11+. - **slots**: define permanent replication slots. These slots will be preserved during switchover/failover. The logical slots are copied from the primary to a standby with restart, and after that their position advanced every **loop_wait** seconds (if necessary). Copying logical slot files performed via ``libpq`` connection and using either rewind or superuser credentials (see **postgresql.authentication** section). There is always a chance that the logical slot position on the replica is a bit behind the former primary, therefore application should be prepared that some messages could be received the second time after the failover. The easiest way of doing so - tracking ``confirmed_flush_lsn``. Enabling permanent logical replication slots requires **postgresql.use_slots** to be set and will also automatically enable the ``hot_standby_feedback``. Since the failover of logical replication slots is unsafe on PostgreSQL 9.6 and older and PostgreSQL version 10 is missing some important functions, the feature only works with PostgreSQL 11+.
@@ -131,7 +131,7 @@ Most of the parameters are optional, but you have to specify one of the **host**
- **checks**: (optional) list of Consul health checks used for the session. By default an empty list is used. - **checks**: (optional) list of Consul health checks used for the session. By default an empty list is used.
- **register\_service**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, replica or standby-leader depending on the node's role. Defaults to **false**. - **register\_service**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, replica or standby-leader depending on the node's role. Defaults to **false**.
- **service\_tags**: (optional) additional static tags to add to the Consul service apart from the role (``master``/``replica``/``standby-leader``). By default an empty list is used. - **service\_tags**: (optional) additional static tags to add to the Consul service apart from the role (``master``/``replica``/``standby-leader``). By default an empty list is used.
- **service\_check\_interval**: (optional) how often to perform health check against registered url. Defaults to '5s'. - **service\_check\_interval**: (optional) how often to perform health check against registered url.
- **service\_check\_tls\_server\_name**: (optional) overide SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__. - **service\_check\_tls\_server\_name**: (optional) overide SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
The ``token`` needs to have the following ACL permissions: The ``token`` needs to have the following ACL permissions:
@@ -262,7 +262,7 @@ PostgreSQL
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server - **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding. - **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
- **replication**: - **replication**:
- **username**: replication username; the user will be created during initialization. Replicas will use this user to access the replication source via streaming replication - **username**: replication username; the user will be created during initialization. Replicas will use this user to access master via streaming replication
- **password**: replication password; the user will be created during initialization. - **password**: replication password; the user will be created during initialization.
- **sslmode**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``. - **sslmode**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate. - **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
@@ -292,7 +292,6 @@ PostgreSQL
- **on\_start**: run this script when the postgres starts. - **on\_start**: run this script when the postgres starts.
- **on\_stop**: run this script when the postgres stops. - **on\_stop**: run this script when the postgres stops.
- **connect\_address**: IP address + port through which Postgres is accessible from other nodes and applications. - **connect\_address**: IP address + port through which Postgres is accessible from other nodes and applications.
- **proxy\_address**: IP address + port through which a connection pool (e.g. pgbouncer) running next to Postgres is accessible. The value is written to the member key in DCS as ``proxy_url`` and could be used/useful for service discovery.
- **create\_replica\_methods**: an ordered list of the create methods for turning a Patroni node into a new replica. - **create\_replica\_methods**: an ordered list of the create methods for turning a Patroni node into a new replica.
"basebackup" is the default method; other methods are assumed to refer to scripts, each of which is configured as its "basebackup" is the default method; other methods are assumed to refer to scripts, each of which is configured as its
own config item. See :ref:`custom replica creation methods documentation <custom_replica_creation>` for further explanation. own config item. See :ref:`custom replica creation methods documentation <custom_replica_creation>` for further explanation.
@@ -315,14 +314,14 @@ PostgreSQL
- **pg\_ctl\_timeout**: How long should pg_ctl wait when doing ``start``, ``stop`` or ``restart``. Default value is 60 seconds. - **pg\_ctl\_timeout**: How long should pg_ctl wait when doing ``start``, ``stop`` or ``restart``. Default value is 60 seconds.
- **use\_pg\_rewind**: try to use pg\_rewind on the former leader when it joins cluster as a replica. - **use\_pg\_rewind**: try to use pg\_rewind on the former leader when it joins cluster as a replica.
- **remove\_data\_directory\_on\_rewind\_failure**: If this option is enabled, Patroni will remove the PostgreSQL data directory and recreate the replica. Otherwise it will try to follow the new leader. Default value is **false**. - **remove\_data\_directory\_on\_rewind\_failure**: If this option is enabled, Patroni will remove the PostgreSQL data directory and recreate the replica. Otherwise it will try to follow the new leader. Default value is **false**.
- **remove\_data\_directory\_on\_diverged\_timelines**: Patroni will remove the PostgreSQL data directory and recreate the replica if it notices that timelines are diverging and the former primary can not start streaming from the new primary. This option is useful when ``pg_rewind`` can not be used. While performing timelines divergence check on PostgreSQL v10 and older Patroni will try to connect with replication credential to the "postgres" database. Hence, such access should be allowed in the pg_hba.conf. Default value is **false**. - **remove\_data\_directory\_on\_diverged\_timelines**: Patroni will remove the PostgreSQL data directory and recreate the replica if it notices that timelines are diverging and the former master can not start streaming from the new master. This option is useful when ``pg_rewind`` can not be used. While performing timelines divergence check on PostgreSQL v10 and older Patroni will try to connect with replication credential to the "postgres" database. Hence, such access should be allowed in the pg_hba.conf. Default value is **false**.
- **replica\_method**: for each create_replica_methods other than basebackup, you would add a configuration section of the same name. At a minimum, this should include "command" with a full path to the actual script to be executed. Other configuration parameters will be passed along to the script in the form "parameter=value". - **replica\_method**: for each create_replica_methods other than basebackup, you would add a configuration section of the same name. At a minimum, this should include "command" with a full path to the actual script to be executed. Other configuration parameters will be passed along to the script in the form "parameter=value".
- **pre\_promote**: a fencing script that executes during a failover after acquiring the leader lock but before promoting the replica. If the script exits with a non-zero code, Patroni does not promote the replica and removes the leader key from DCS. - **pre\_promote**: a fencing script that executes during a failover after acquiring the leader lock but before promoting the replica. If the script exits with a non-zero code, Patroni does not promote the replica and removes the leader key from DCS.
REST API REST API
-------- --------
- **restapi**: - **restapi**:
- **connect\_address**: IP address (or hostname) and port, to access the Patroni's :ref:`REST API <rest_api>`. All the members of the cluster must be able to connect to this address, so unless the Patroni setup is intended for a demo inside the localhost, this address must be a non "localhost" or loopback address (ie: "localhost" or "127.0.0.1"). It can serve as an endpoint for HTTP health checks (read below about the "listen" REST API parameter), and also for user queries (either directly or via the REST API), as well as for the health checks done by the cluster members during leader elections (for example, to determine whether the leader is still running, or if there is a node which has a WAL position that is ahead of the one doing the query; etc.) The connect_address is put in the member key in DCS, making it possible to translate the member name into the address to connect to its REST API. - **connect\_address**: IP address (or hostname) and port, to access the Patroni's :ref:`REST API <rest_api>`. All the members of the cluster must be able to connect to this address, so unless the Patroni setup is intended for a demo inside the localhost, this address must be a non "localhost" or loopback address (ie: "localhost" or "127.0.0.1"). It can serve as an endpoint for HTTP health checks (read below about the "listen" REST API parameter), and also for user queries (either directly or via the REST API), as well as for the health checks done by the cluster members during leader elections (for example, to determine whether the master is still running, or if there is a node which has a WAL position that is ahead of the one doing the query; etc.) The connect_address is put in the member key in DCS, making it possible to translate the member name into the address to connect to its REST API.
- **listen**: IP address (or hostname) and port that Patroni will listen to for the REST API - to provide also the same health checks and cluster messaging between the participating nodes, as described above. to provide health-check information for HAProxy (or any other load balancer capable of doing a HTTP "OPTION" or "GET" checks). - **listen**: IP address (or hostname) and port that Patroni will listen to for the REST API - to provide also the same health checks and cluster messaging between the participating nodes, as described above. to provide health-check information for HAProxy (or any other load balancer capable of doing a HTTP "OPTION" or "GET" checks).
+3 -3
View File
@@ -22,7 +22,7 @@ Patroni configuration is stored in the DCS (Distributed Configuration Store). Th
The local configuration can be either a single YAML file or a directory. When it is a directory, all YAML files in that directory are loaded one by one in sorted order. In case a key is defined in multiple files, the occurrence in the last file takes precedence. The local configuration can be either a single YAML file or a directory. When it is a directory, all YAML files in that directory are loaded one by one in sorted order. In case a key is defined in multiple files, the occurrence in the last file takes precedence.
Some of the PostgreSQL parameters must hold the same values on the primary and the replicas. For those, values set either in the local patroni configuration files or via the environment variables take no effect. To alter or set their values one must change the shared configuration in the DCS. Below is the actual list of such parameters together with the default values: Some of the PostgreSQL parameters must hold the same values on the master and the replicas. For those, values set either in the local patroni configuration files or via the environment variables take no effect. To alter or set their values one must change the shared configuration in the DCS. Below is the actual list of such parameters together with the default values:
- max_connections: 100 - max_connections: 100
- max_locks_per_transaction: 64 - max_locks_per_transaction: 64
@@ -32,7 +32,7 @@ Some of the PostgreSQL parameters must hold the same values on the primary and t
- wal_log_hints: on - wal_log_hints: on
- track_commit_timestamp: off - track_commit_timestamp: off
For the parameters below, PostgreSQL does not require equal values among the primary and all the replicas. However, considering the possibility of a replica to become the primary at any time, it doesn't really make sense to set them differently; therefore, Patroni restricts setting their values to the Dynamic configuration For the parameters below, PostgreSQL does not require equal values among the master and all the replicas. However, considering the possibility of a replica to become the master at any time, it doesn't really make sense to set them differently; therefore, Patroni restricts setting their values to the Dynamic configuration
- max_wal_senders: 5 - max_wal_senders: 5
- max_replication_slots: 5 - max_replication_slots: 5
@@ -86,4 +86,4 @@ Also, the following Patroni configuration options can be changed only dynamicall
Upon changing these options, Patroni will read the relevant section of the configuration stored in DCS and change its Upon changing these options, Patroni will read the relevant section of the configuration stored in DCS and change its
run-time values. run-time values.
Patroni nodes are dumping the state of the DCS options to disk upon for every change of the configuration into the file ``patroni.dynamic.json`` located in the Postgres data directory. Only the leader is allowed to restore these options from the on-disk dump if these are completely absent from the DCS or if they are invalid. Patroni nodes are dumping the state of the DCS options to disk upon for every change of the configuration into the file ``patroni.dynamic.json`` located in the Postgres data directory. Only the master is allowed to restore these options from the on-disk dump if these are completely absent from the DCS or if they are invalid.
+2 -2
View File
@@ -29,11 +29,11 @@ Major Upgrade of PostgreSQL Version
The only possible way to do a major upgrade currently is: The only possible way to do a major upgrade currently is:
1. Stop Patroni 1. Stop Patroni
2. Upgrade PostgreSQL binaries and perform `pg_upgrade <https://www.postgresql.org/docs/current/pgupgrade.html>`_ on the primary node 2. Upgrade PostgreSQL binaries and perform `pg_upgrade <https://www.postgresql.org/docs/current/pgupgrade.html>`_ on the master node
3. Update patroni.yml 3. Update patroni.yml
4. Remove the initialize key from DCS or wipe complete cluster state from DCS. The second one could be achieved by running ``patronictl remove <cluster-name>``. It is necessary because pg_upgrade runs initdb which actually creates a new database with a new PostgreSQL system identifier. 4. Remove the initialize key from DCS or wipe complete cluster state from DCS. The second one could be achieved by running ``patronictl remove <cluster-name>``. It is necessary because pg_upgrade runs initdb which actually creates a new database with a new PostgreSQL system identifier.
5. If you wiped the cluster state in the previous step, you may wish to copy patroni.dynamic.json from old data dir to the new one. It will help you to retain some PostgreSQL parameters you had set before. 5. If you wiped the cluster state in the previous step, you may wish to copy patroni.dynamic.json from old data dir to the new one. It will help you to retain some PostgreSQL parameters you had set before.
6. Start Patroni on the primary node. 6. Start Patroni on the master node.
7. Upgrade PostgreSQL binaries, update patroni.yml and wipe the data_dir on standby nodes. 7. Upgrade PostgreSQL binaries, update patroni.yml and wipe the data_dir on standby nodes.
8. Start Patroni on the standby nodes and wait for the replication to complete. 8. Start Patroni on the standby nodes and wait for the replication to complete.
+1 -1
View File
@@ -10,7 +10,7 @@ Patroni is a template for you to create your own customized, high-availability s
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__. We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
Currently supported PostgreSQL versions: 9.3 to 15. Currently supported PostgreSQL versions: 9.3 to 14.
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the :ref:`Kubernetes <kubernetes>` chapter of the Patroni documentation. **Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the :ref:`Kubernetes <kubernetes>` chapter of the Patroni documentation.
+1 -1
View File
@@ -23,7 +23,7 @@ Use ConfigMaps
In this mode, Patroni will create ConfigMaps instead of Endpoints and store keys inside meta-data of those ConfigMaps. In this mode, Patroni will create ConfigMaps instead of Endpoints and store keys inside meta-data of those ConfigMaps.
Changing the leader takes at least two updates, one to the leader ConfigMap and another to the respective Endpoint. Changing the leader takes at least two updates, one to the leader ConfigMap and another to the respective Endpoint.
There are two ways to direct the traffic to the Postgres leader: There are two ways to direct the traffic to the Postgres master:
- use the `callback script <https://github.com/zalando/patroni/blob/master/kubernetes/callback.py>`_ provided by Patroni - use the `callback script <https://github.com/zalando/patroni/blob/master/kubernetes/callback.py>`_ provided by Patroni
- configure the Kubernetes Postgres service to use the label selector with the `role_label` (configured in patroni configuration). - configure the Kubernetes Postgres service to use the label selector with the `role_label` (configured in patroni configuration).
+5 -7
View File
@@ -6,7 +6,7 @@ Pause/Resume mode for the cluster
The goal The goal
-------- --------
Under certain circumstances Patroni needs to temporarily step down from managing the cluster, while still retaining the cluster state in DCS. Possible use cases are uncommon activities on the cluster, such as major version upgrades or corruption recovery. During those activities nodes are often started and stopped for reasons unknown to Patroni, some nodes can be even temporarily promoted, violating the assumption of running only one primary. Therefore, Patroni needs to be able to "detach" from the running cluster, implementing an equivalent of the maintenance mode in Pacemaker. Under certain circumstances Patroni needs to temporary step down from managing the cluster, while still retaining the cluster state in DCS. Possible use cases are uncommon activities on the cluster, such as major version upgrades or corruption recovery. During those activities nodes are often started and stopped for the reason unknown to Patroni, some nodes can be even temporary promoted, violating the assumption of running only one master. Therefore, Patroni needs to be able to "detach" from the running cluster, implementing an equivalent of the maintenance mode in Pacemaker.
@@ -17,18 +17,16 @@ When Patroni runs in a paused mode, it does not change the state of PostgreSQL,
- For each node, the member key in DCS is updated with the current information about the cluster. This causes Patroni to run read-only queries on a member node if the member is running. - For each node, the member key in DCS is updated with the current information about the cluster. This causes Patroni to run read-only queries on a member node if the member is running.
- For the Postgres primary with the leader lock Patroni updates the lock. If the node with the leader lock stops being the primary (i.e. is demoted manually), Patroni will release the lock instead of promoting the node back. - For the Postgres master with the leader lock Patroni updates the lock. If the node with the leader lock stops being the master (i.e. is demoted manually), Patroni will release the lock instead of promoting the node back.
- Manual unscheduled restart, reinitialize and manual failover are allowed. Manual failover is only allowed if the node to failover to is specified. In the paused mode, manual failover does not require a running primary node. - Manual unscheduled restart, reinitialize and manual failover are allowed. Manual failover is only allowed if the node to failover to is specified. In the paused mode, manual failover does not require a running master node.
- If 'parallel' primaries are detected by Patroni, it emits a warning, but does not demote the primary without the leader lock. - If 'parallel' masters are detected by Patroni, it emits a warning, but does not demote the masters without the leader lock.
- If there is no leader lock in the cluster, the running primary acquires the lock. If there is more than one primary node, then the first primary to acquire the lock wins. If there are no primary altogether, Patroni does not try to promote any replicas. There is an exception in this rule: if there is no leader lock because the old primary has demoted itself due to the manual promotion, then only the candidate node mentioned in the promotion request may take the leader lock. When the new leader lock is granted (i.e. after promoting a replica manually), Patroni makes sure the replicas that were streaming from the previous leader will switch to the new one. - If there is no leader lock in the cluster, the running master acquires the lock. If there is more than one master node, then the first master to acquire the lock wins. If there are no masters altogether, Patroni does not try to promote any replicas. There is an exception in this rule: if there is no leader lock because the old master has demoted itself due to the manual promotion, then only the candidate node mentioned in the promotion request may take the leader lock. When the new leader lock is granted (i.e. after promoting a replica manually), Patroni makes sure the replicas that were streaming from the previous leader will switch to the new one.
- When Postgres is stopped, Patroni does not try to start it. When Patroni is stopped, it does not try to stop the Postgres instance it is managing. - When Postgres is stopped, Patroni does not try to start it. When Patroni is stopped, it does not try to stop the Postgres instance it is managing.
- Patroni will not try to remove replication slots that don't represent the other cluster member or are not listed in the configuration of the permanent slots.
User guide User guide
---------- ----------
-98
View File
@@ -3,104 +3,6 @@
Release notes Release notes
============= =============
Version 2.1.5
-------------
This version enhances compatibility with PostgreSQL 15 and declares Etcd v3 support as production ready. The Patroni on Raft remains in Beta.
**New features**
- Improve ``patroni --validate-config`` (Denis Laxalde)
Exit with code 1 if config is invalid and print errors to stderr.
- Don't drop replication slots in pause (Alexander Kukushkin)
Patroni is automatically creating/removing physical replication slots when members are joining/leaving the cluster. In pause slots will no longer be removed.
- Support the ``HEAD`` request method for monitoring endpoints (Robert Cutajar)
If used instead of ``GET`` Patroni will return only the HTTP Status Code.
- Support behave tests on Windows (Alexander)
Emulate graceful Patroni shutdown (``SIGTERM``) on Windows by introduce the new REST API endpoint ``POST /sigterm``.
- Introduce ``postgresql.proxy_address`` (Alexander)
It will be written to the member key in DCS as the ``proxy_url`` and could be used/useful for service discovery.
**Stability improvements**
- Call ``pg_replication_slot_advance()`` from a thread (Alexander)
On busy clusters with many logical replication slots the ``pg_replication_slot_advance()`` call was affecting the main HA loop and could result in the member key expiration.
- Archive possibly missing WALs before calling ``pg_rewind`` on the old primary (Polina Bungina)
If the primary crashed and was down during considerable time, some WAL files could be missing from archive and from the new primary. There is a chance that ``pg_rewind`` could remove these WAL files from the old primary making it impossible to start it as a standby. By archiving ``ready`` WAL files we not only mitigate this problem but in general improving continues archiving experience.
- Ignore ``403`` errors when trying to create Kubernetes Service (Nick Hudson, Polina)
Patroni was spamming logs by unsuccessful attempts to create the service, which in fact could already exist.
- Improve liveness probe (Alexander)
The liveness problem will start failing if the heartbeat loop is running longer than `ttl` on the primary or `2*ttl` on the replica. That will allow us to use it as an alternative for :ref:`watchdog <watchdog>` on Kubernetes.
- Make sure only sync node tries to grab the lock when switchover (Alexander, Polina)
Previously there was a slim chance that up-to-date async member could become the leader if the manual switchover was performed without specifying the target.
- Avoid cloning while bootstrap is running (Ants Aasma)
Do not allow a create replica method that does not require a leader to be triggered while the cluster bootstrap is running.
- Compatibility with kazoo-2.9.0 (Alexander)
Depending on python version the ``SequentialThreadingHandler.select()`` method may raise ``TypeError`` and ``IOError`` exceptions if ``select()`` is called on the closed socket.
- Explicitly shut down SSL connection before socket shutdown (Alexander)
Not doing it resulted in ``unexpected eof while reading`` errors with OpenSSL 3.0.
- Compatibility with `prettytable>=2.2.0` (Alexander)
Due to the internal API changes the cluster name header was shown on the incorrect line.
**Bugfixes**
- Handle expired token for Etcd lease_grant (monsterxx03)
In case of error get the new token and retry request.
- Fix bug in the ``GET /read-only-sync`` endpoint (Alexander)
It was introduced in previous release and effectively never worked.
- Handle the case when data dir storage disappeared (Alexander)
Patroni is periodically checking that the PGDATA is there and not empty, but in case of issues with storage the ``os.listdir()`` is raising the ``OSError`` exception, breaking the heart-beat loop.
- Apply ``master_stop_timeout`` when waiting for user backends to close (Alexander)
Something that looks like user backend could be in fact a background worker (e.g., Citus Maintenance Daemon) that is failing to stop.
- Accept ``*:<port>`` for ``postgresql.listen`` (Denis)
The ``patroni --validate-config`` was complaining about it being invalid.
- Timeouts fixes in Raft (Alexander)
When Patroni or patronictl are starting they try to get Raft cluster topology from known members. These calls were made without proper timeouts.
- Forcefully update consul service if token was changed (John A. Lotoski)
Not doing so results in errors "rpc error making call: rpc error making call: ACL not found".
Version 2.1.4 Version 2.1.4
------------- -------------
+6 -18
View File
@@ -63,7 +63,7 @@ Building replicas
----------------- -----------------
Patroni uses tried and proven ``pg_basebackup`` in order to create new replicas. One downside of it is that it requires Patroni uses tried and proven ``pg_basebackup`` in order to create new replicas. One downside of it is that it requires
a running leader node. Another one is the lack of 'on-the-fly' compression for the backup data and no built-in cleanup a running master node. Another one is the lack of 'on-the-fly' compression for the backup data and no built-in cleanup
for outdated backup files. Some people prefer other backup solutions, such as ``WAL-E``, ``pgBackRest``, ``Barman`` and for outdated backup files. Some people prefer other backup solutions, such as ``WAL-E``, ``pgBackRest``, ``Barman`` and
others, or simply roll their own scripts. In order to accommodate all those use-cases Patroni supports running custom others, or simply roll their own scripts. In order to accommodate all those use-cases Patroni supports running custom
scripts to clone a new replica. Those are configured in the ``postgresql`` configuration block: scripts to clone a new replica. Those are configured in the ``postgresql`` configuration block:
@@ -123,11 +123,11 @@ to execute and any custom parameters that should be passed to that command. All
--role --role
Always 'replica' Always 'replica'
--connstring --connstring
Connection string to connect to the cluster member to clone from (primary or other replica). The user in the Connection string to connect to the cluster member to clone from (master or other replica). The user in the
connection string can execute SQL and replication protocol commands. connection string can execute SQL and replication protocol commands.
A special ``no_master`` parameter, if defined, allows Patroni to call the replica creation method even if there is no A special ``no_master`` parameter, if defined, allows Patroni to call the replica creation method even if there is no
running leader or replicas. In that case, an empty string will be passed in a connection string. This is useful for running master or replicas. In that case, an empty string will be passed in a connection string. This is useful for
restoring the formerly running cluster from the binary backup. restoring the formerly running cluster from the binary backup.
A special ``keep_data`` parameter, if defined, will instruct Patroni to not clean PGDATA folder before calling restore. A special ``keep_data`` parameter, if defined, will instruct Patroni to not clean PGDATA folder before calling restore.
@@ -137,7 +137,7 @@ A special ``no_params`` parameter, if defined, restricts passing parameters to c
A ``basebackup`` method is a special case: it will be used if A ``basebackup`` method is a special case: it will be used if
``create_replica_methods`` is empty, although it is possible ``create_replica_methods`` is empty, although it is possible
to list it explicitly among the ``create_replica_methods`` methods. This method initializes a new replica with the to list it explicitly among the ``create_replica_methods`` methods. This method initializes a new replica with the
``pg_basebackup``, the base backup is taken from the leader unless there are replicas with ``clonefrom`` tag, in which case one ``pg_basebackup``, the base backup is taken from the master unless there are replicas with ``clonefrom`` tag, in which case one
of such replicas will be used as the origin for pg_basebackup. It works without any configuration; however, it is of such replicas will be used as the origin for pg_basebackup. It works without any configuration; however, it is
possible to specify a ``basebackup`` configuration section. Same rules as with the other method configuration apply, possible to specify a ``basebackup`` configuration section. Same rules as with the other method configuration apply,
namely, only long (with --) options should be specified there. Not all parameters make sense, if you override a connection namely, only long (with --) options should be specified there. Not all parameters make sense, if you override a connection
@@ -176,10 +176,10 @@ Standby cluster
--------------- ---------------
Another available option is to run a "standby cluster", that contains only of Another available option is to run a "standby cluster", that contains only of
standby nodes replicating from some remote node. This type of clusters has: standby nodes replicating from some remote master. This type of clusters has:
* "standby leader", that behaves pretty much like a regular cluster leader, * "standby leader", that behaves pretty much like a regular cluster leader,
except it replicates from a remote node. except it replicates from a remote master.
* cascade replicas, that are replicating from standby leader. * cascade replicas, that are replicating from standby leader.
@@ -187,13 +187,6 @@ Standby leader holds and updates a leader lock in DCS. If the leader lock
expires, cascade replicas will perform an election to choose another leader expires, cascade replicas will perform an election to choose another leader
from the standbys. from the standbys.
There is no further relationship between the standby cluster and the primary
cluster it replicates from, in particular, they must not share the same DCS
scope if they use the same DCS. They do not know anything else from each other
apart from replication information. Also, the standby cluster is not being
displayed in ``patronictl list`` or ``patronictl topology`` output on the
primary cluster.
For the sake of flexibility, you can specify methods of creating a replica and For the sake of flexibility, you can specify methods of creating a replica and
recovery WAL records when a cluster is in the "standby mode" by providing recovery WAL records when a cluster is in the "standby mode" by providing
`create_replica_methods` key in `standby_cluster` section. It is distinct from `create_replica_methods` key in `standby_cluster` section. It is distinct from
@@ -219,9 +212,4 @@ in a patroni configuration:
Note, that these options will be applied only once during cluster bootstrap, Note, that these options will be applied only once during cluster bootstrap,
and the only way to change them afterwards is through DCS. and the only way to change them afterwards is through DCS.
Patroni expects to find `postgresql.conf` or `postgresql.conf.backup` in PGDATA
of the remote primary and will not start if it does not find it after a
basebackup. If the remote primary keeps its `postgresql.conf` elsewhere, it is
your responsibility to copy it to PGDATA.
If you use replication slots on the standby cluster, you must also create the corresponding replication slot on the primary cluster. It will not be done automatically by the standby cluster implementation. You can use Patroni's permanent replication slots feature on the primary cluster to maintain a replication slot with the same name as ``primary_slot_name``, or its default value if ``primary_slot_name`` is not provided. If you use replication slots on the standby cluster, you must also create the corresponding replication slot on the primary cluster. It will not be done automatically by the standby cluster implementation. You can use Patroni's permanent replication slots feature on the primary cluster to maintain a replication slot with the same name as ``primary_slot_name``, or its default value if ``primary_slot_name`` is not provided.
+1 -1
View File
@@ -13,7 +13,7 @@ In asynchronous mode the cluster is allowed to lose some committed transactions
The amount of transactions that can be lost is controlled via ``maximum_lag_on_failover`` parameter. Because the primary transaction log position is not sampled in real time, in reality the amount of lost data on failover is worst case bounded by ``maximum_lag_on_failover`` bytes of transaction log plus the amount that is written in the last ``ttl`` seconds (``loop_wait``/2 seconds in the average case). However typical steady state replication delay is well under a second. The amount of transactions that can be lost is controlled via ``maximum_lag_on_failover`` parameter. Because the primary transaction log position is not sampled in real time, in reality the amount of lost data on failover is worst case bounded by ``maximum_lag_on_failover`` bytes of transaction log plus the amount that is written in the last ``ttl`` seconds (``loop_wait``/2 seconds in the average case). However typical steady state replication delay is well under a second.
By default, when running leader elections, Patroni does not take into account the current timeline of replicas, what in some cases could be undesirable behavior. You can prevent the node not having the same timeline as a former primary become the new leader by changing the value of ``check_timeline`` parameter to ``true``. By default, when running leader elections, Patroni does not take into account the current timeline of replicas, what in some cases could be undesirable behavior. You can prevent the node not having the same timeline as a former master become the new leader by changing the value of ``check_timeline`` parameter to ``true``.
PostgreSQL synchronous replication PostgreSQL synchronous replication
---------------------------------- ----------------------------------
+2 -2
View File
@@ -7,7 +7,7 @@ Patroni has a rich REST API, which is used by Patroni itself during the leader r
Health check endpoints Health check endpoints
---------------------- ----------------------
For all health check ``GET`` requests Patroni returns a JSON document with the status of the node, along with the HTTP status code. If you don't want or don't need the JSON document, you might consider using the ``HEAD`` or ``OPTIONS`` method instead of ``GET``. For all health check ``GET`` requests Patroni returns a JSON document with the status of the node, along with the HTTP status code. If you don't want or don't need the JSON document, you might consider using the ``OPTIONS`` method instead of ``GET``.
- The following requests to Patroni REST API will return HTTP status code **200** only when the Patroni node is running as the primary with leader lock: - The following requests to Patroni REST API will return HTTP status code **200** only when the Patroni node is running as the primary with leader lock:
@@ -58,7 +58,7 @@ For all health check ``GET`` requests Patroni returns a JSON document with the s
- ``GET /health``: returns HTTP status code **200** only when PostgreSQL is up and running. - ``GET /health``: returns HTTP status code **200** only when PostgreSQL is up and running.
- ``GET /liveness``: returns HTTP status code **200** if Patroni heartbeat loop is properly running and **503** if the last run was more than ``ttl`` seconds ago on the primary or ``2*ttl`` on the replica. Could be used for ``livenessProbe``. - ``GET /liveness``: always returns HTTP status code **200** what only indicates that Patroni is running. Could be used for ``livenessProbe``.
- ``GET /readiness``: returns HTTP status code **200** when the Patroni node is running as the leader or when PostgreSQL is up and running. The endpoint could be used for ``readinessProbe`` when it is not possible to use Kubernetes endpoints for leader elections (OpenShift). - ``GET /readiness``: returns HTTP status code **200** when the Patroni node is running as the leader or when PostgreSQL is up and running. The endpoint could be used for ``readinessProbe`` when it is not possible to use Kubernetes endpoints for leader elections (OpenShift).
+2 -2
View File
@@ -3,7 +3,7 @@
Watchdog support Watchdog support
================ ================
Having multiple PostgreSQL servers running as primary can result in transactions lost due to diverging timelines. This situation is also called a split-brain problem. To avoid split-brain Patroni needs to ensure PostgreSQL will not accept any transaction commits after leader key expires in the DCS. Under normal circumstances Patroni will try to achieve this by stopping PostgreSQL when leader lock update fails for any reason. However, this may fail to happen due to various reasons: Having multiple PostgreSQL servers running as master can result in transactions lost due to diverging timelines. This situation is also called a split-brain problem. To avoid split-brain Patroni needs to ensure PostgreSQL will not accept any transaction commits after leader key expires in the DCS. Under normal circumstances Patroni will try to achieve this by stopping PostgreSQL when leader lock update fails for any reason. However, this may fail to happen due to various reasons:
- Patroni has crashed due to a bug, out-of-memory condition or by being accidentally killed by a system administrator. - Patroni has crashed due to a bug, out-of-memory condition or by being accidentally killed by a system administrator.
@@ -13,7 +13,7 @@ Having multiple PostgreSQL servers running as primary can result in transactions
To guarantee correct behavior under these conditions Patroni supports watchdog devices. Watchdog devices are software or hardware mechanisms that will reset the whole system when they do not get a keepalive heartbeat within a specified timeframe. This adds an additional layer of fail safe in case usual Patroni split-brain protection mechanisms fail. To guarantee correct behavior under these conditions Patroni supports watchdog devices. Watchdog devices are software or hardware mechanisms that will reset the whole system when they do not get a keepalive heartbeat within a specified timeframe. This adds an additional layer of fail safe in case usual Patroni split-brain protection mechanisms fail.
Patroni will try to activate the watchdog before promoting PostgreSQL to primary. If watchdog activation fails and watchdog mode is ``required`` then the node will refuse to become leader. When deciding to participate in leader election Patroni will also check that watchdog configuration will allow it to become leader at all. After demoting PostgreSQL (for example due to a manual failover) Patroni will disable the watchdog again. Watchdog will also be disabled while Patroni is in paused state. Patroni will try to activate the watchdog before promoting PostgreSQL to master. If watchdog activation fails and watchdog mode is ``required`` then the node will refuse to become master. When deciding to participate in leader election Patroni will also check that watchdog configuration will allow it to become leader at all. After demoting PostgreSQL (for example due to a manual failover) Patroni will disable the watchdog again. Watchdog will also be disabled while Patroni is in paused state.
By default Patroni will set up the watchdog to expire 5 seconds before TTL expires. With the default setup of ``loop_wait=10`` and ``ttl=30`` this gives HA loop at least 15 seconds (``ttl`` - ``safety_margin`` - ``loop_wait``) to complete before the system gets forcefully reset. By default accessing DCS is configured to time out after 10 seconds. This means that when DCS is unavailable, for example due to network issues, Patroni and PostgreSQL will have at least 5 seconds (``ttl`` - ``safety_margin`` - ``loop_wait`` - ``retry_timeout``) to come to a state where all client connections are terminated. By default Patroni will set up the watchdog to expire 5 seconds before TTL expires. With the default setup of ``loop_wait=10`` and ``ttl=30`` this gives HA loop at least 15 seconds (``ttl`` - ``safety_margin`` - ``loop_wait``) to complete before the system gets forcefully reset. By default accessing DCS is configured to time out after 10 seconds. This means that when DCS is unavailable, for example due to network issues, Patroni and PostgreSQL will have at least 5 seconds (``ttl`` - ``safety_margin`` - ``loop_wait`` - ``retry_timeout``) to come to a state where all client connections are terminated.
+2 -2
View File
@@ -18,14 +18,14 @@ listen stats
listen master listen master
bind *:5000 bind *:5000
option httpchk HEAD /master option httpchk OPTIONS /master
http-check expect status 200 http-check expect status 200
default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions
{{range gets "/members/*"}} server {{base .Key}} {{$data := json .Value}}{{base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}} maxconn 100 check port {{index (split (index (split $data.api_url "/") 2) ":") 1}} {{range gets "/members/*"}} server {{base .Key}} {{$data := json .Value}}{{base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}} maxconn 100 check port {{index (split (index (split $data.api_url "/") 2) ":") 1}}
{{end}} {{end}}
listen replicas listen replicas
bind *:5001 bind *:5001
option httpchk HEAD /replica option httpchk OPTIONS /replica
http-check expect status 200 http-check expect status 200
default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions
{{range gets "/members/*"}} server {{base .Key}} {{$data := json .Value}}{{base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}} maxconn 100 check port {{index (split (index (split $data.api_url "/") 2) ":") 1}} {{range gets "/members/*"}} server {{base .Key}} {{$data := json .Value}}{{base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}} maxconn 100 check port {{index (split (index (split $data.api_url "/") 2) ":") 1}}
+4 -4
View File
@@ -14,7 +14,7 @@ Group=postgres
# Read in configuration file if it exists, otherwise proceed # Read in configuration file if it exists, otherwise proceed
EnvironmentFile=-/etc/patroni_env.conf EnvironmentFile=-/etc/patroni_env.conf
# The default is the user's home directory, and if you want to change it, you must provide an absolute path. # the default is the user's home directory, and if you want to change it, you must provide an absolute path.
# WorkingDirectory=/home/sameuser # WorkingDirectory=/home/sameuser
# Where to send early-startup messages from the server # Where to send early-startup messages from the server
@@ -32,14 +32,14 @@ ExecStart=/bin/patroni /etc/patroni.yml
# Send HUP to reload from patroni.yml # Send HUP to reload from patroni.yml
ExecReload=/bin/kill -s HUP $MAINPID ExecReload=/bin/kill -s HUP $MAINPID
# Only kill the patroni process, not it's children, so it will gracefully stop postgres # only kill the patroni process, not it's children, so it will gracefully stop postgres
KillMode=process KillMode=process
# Give a reasonable amount of time for the server to start up/shut down # Give a reasonable amount of time for the server to start up/shut down
TimeoutSec=30 TimeoutSec=30
# Restart the service if it crashed # Do not restart the service if it crashes, we want to manually inspect database on failure
Restart=on-failure Restart=no
[Install] [Install]
WantedBy=multi-user.target WantedBy=multi-user.target
+1 -1
View File
@@ -5,7 +5,7 @@ Feature: basic replication
Given I start postgres0 Given I start postgres0
Then postgres0 is a leader after 10 seconds Then postgres0 is a leader after 10 seconds
And there is a non empty initialize key in DCS after 15 seconds And there is a non empty initialize key in DCS after 15 seconds
When I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 20, "synchronous_mode": true} When I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 20, "loop_wait": 2, "synchronous_mode": true}
Then I receive a response code 200 Then I receive a response code 200
When I start postgres1 When I start postgres1
And I configure and start postgres2 with a tag replicatefrom postgres0 And I configure and start postgres2 with a tag replicatefrom postgres0
+21 -54
View File
@@ -14,7 +14,6 @@ import yaml
import patroni.psycopg as psycopg import patroni.psycopg as psycopg
from patroni.request import PatroniRequest
from six.moves.BaseHTTPServer import BaseHTTPRequestHandler, HTTPServer from six.moves.BaseHTTPServer import BaseHTTPRequestHandler, HTTPServer
@@ -139,24 +138,12 @@ class PatroniController(AbstractController):
def _start(self): def _start(self):
if self.watchdog: if self.watchdog:
self.watchdog.start() self.watchdog.start()
env = os.environ.copy()
if isinstance(self._context.dcs_ctl, KubernetesController): if isinstance(self._context.dcs_ctl, KubernetesController):
self._context.dcs_ctl.create_pod(self._name[8:], self._scope) self._context.dcs_ctl.create_pod(self._name[8:], self._scope)
env['PATRONI_KUBERNETES_POD_IP'] = '10.0.0.' + self._name[-1] os.environ['PATRONI_KUBERNETES_POD_IP'] = '10.0.0.' + self._name[-1]
if os.name == 'nt': return subprocess.Popen([sys.executable, '-m', 'coverage', 'run',
env['BEHAVE_DEBUG'] = 'true' '--source=patroni', '-p', 'patroni.py', self._config],
patroni = subprocess.Popen([sys.executable, '-m', 'coverage', 'run', stdout=self._log, stderr=subprocess.STDOUT, cwd=self._work_directory)
'--source=patroni', '-p', 'patroni.py', self._config], env=env,
stdout=self._log, stderr=subprocess.STDOUT, cwd=self._work_directory)
if os.name == 'nt':
patroni.terminate = self.terminate
return patroni
def terminate(self):
try:
self._context.request_executor.request('POST', self._restapi_url + '/sigterm')
except Exception:
pass
def stop(self, kill=False, timeout=15, postgres=False): def stop(self, kill=False, timeout=15, postgres=False):
if postgres: if postgres:
@@ -191,16 +178,15 @@ class PatroniController(AbstractController):
config['postgresql']['listen'] = config['postgresql']['connect_address'] = '{0}:{1}'.format(host, self.__PORT) config['postgresql']['listen'] = config['postgresql']['connect_address'] = '{0}:{1}'.format(host, self.__PORT)
config['name'] = name config['name'] = name
config['postgresql']['data_dir'] = self._data_dir.replace('\\', '/') config['postgresql']['data_dir'] = self._data_dir
config['postgresql']['basebackup'] = [{'checkpoint': 'fast'}] config['postgresql']['basebackup'] = [{'checkpoint': 'fast'}]
config['postgresql']['use_unix_socket'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets config['postgresql']['use_unix_socket'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets
config['postgresql']['use_unix_socket_repl'] = os.name != 'nt' config['postgresql']['use_unix_socket_repl'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets
config['postgresql']['pgpass'] = os.path.join(tempfile.gettempdir(), 'pgpass_' + name).replace('\\', '/') config['postgresql']['pgpass'] = os.path.join(tempfile.gettempdir(), 'pgpass_' + name)
config['postgresql']['parameters'].update({ config['postgresql']['parameters'].update({
'logging_collector': 'on', 'log_destination': 'csvlog', 'logging_collector': 'on', 'log_destination': 'csvlog', 'log_directory': self._output_dir,
'log_directory': self._output_dir.replace('\\', '/'),
'log_filename': name + '.log', 'log_statement': 'all', 'log_min_messages': 'debug1', 'log_filename': name + '.log', 'log_statement': 'all', 'log_min_messages': 'debug1',
'unix_socket_directories': tempfile.gettempdir().replace('\\', '/')}) 'unix_socket_directories': tempfile.gettempdir()})
if 'bootstrap' in config: if 'bootstrap' in config:
config['bootstrap']['post_bootstrap'] = 'psql -w -c "SELECT 1"' config['bootstrap']['post_bootstrap'] = 'psql -w -c "SELECT 1"'
@@ -211,7 +197,7 @@ class PatroniController(AbstractController):
self.recursive_update(config, custom_config) self.recursive_update(config, custom_config)
self.recursive_update(config, { self.recursive_update(config, {
'bootstrap': {'dcs': {'loop_wait': 2, 'postgresql': {'parameters': {'wal_keep_segments': 100}}}}}) 'bootstrap': {'dcs': {'postgresql': {'parameters': {'wal_keep_segments': 100}}}}})
if config['postgresql'].get('callbacks', {}).get('on_role_change'): if config['postgresql'].get('callbacks', {}).get('on_role_change'):
config['postgresql']['callbacks']['on_role_change'] += ' ' + str(self.__PORT) config['postgresql']['callbacks']['on_role_change'] += ' ' + str(self.__PORT)
@@ -224,7 +210,6 @@ class PatroniController(AbstractController):
self._replication = config['postgresql'].get('authentication', config['postgresql']).get('replication', {}) self._replication = config['postgresql'].get('authentication', config['postgresql']).get('replication', {})
self._replication.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres'}) self._replication.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres'})
self._restapi_url = 'http://{0}'.format(config['restapi']['connect_address'])
return patroni_config_path return patroni_config_path
@@ -409,7 +394,7 @@ class AbstractEtcdController(AbstractDcsController):
self._client_cls = client_cls self._client_cls = client_cls
def _start(self): def _start(self):
return subprocess.Popen(["etcd", "--data-dir", self._work_directory], return subprocess.Popen(["etcd", "--debug", "--data-dir", self._work_directory],
stdout=self._log, stderr=subprocess.STDOUT) stdout=self._log, stderr=subprocess.STDOUT)
def _is_running(self): def _is_running(self):
@@ -641,17 +626,15 @@ class RaftController(AbstractDcsController):
self.start() self.start()
ready_event = threading.Event() ready_event = threading.Event()
self._raft = KVStoreTTL(ready_event.set, None, None, self._raft = KVStoreTTL(ready_event.set, None, None, partner_addrs=[self.CONTROLLER_ADDR], password=self.PASSWORD)
partner_addrs=[self.CONTROLLER_ADDR], password=self.PASSWORD)
self._raft.startAutoTick() self._raft.startAutoTick()
ready_event.wait() ready_event.wait()
class PatroniPoolController(object): class PatroniPoolController(object):
PYTHON = sys.executable.replace('\\', '/') BACKUP_SCRIPT = [sys.executable, 'features/backup_create.py']
BACKUP_SCRIPT = [PYTHON, 'features/backup_create.py'] ARCHIVE_RESTORE_SCRIPT = ' '.join((sys.executable, os.path.abspath('features/archive-restore.py')))
ARCHIVE_RESTORE_SCRIPT = ' '.join((PYTHON, os.path.abspath('features/archive-restore.py')))
def __init__(self, context): def __init__(self, context):
self._context = context self._context = context
@@ -727,7 +710,7 @@ class PatroniPoolController(object):
'archive_mode': 'on', 'archive_mode': 'on',
'archive_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode archive ' + 'archive_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode archive ' +
'--dirname {} --filename %f --pathname %p').format( '--dirname {} --filename %f --pathname %p').format(
os.path.join(self.patroni_path, 'data', 'wal_archive').replace('\\', '/')) os.path.join(self.patroni_path, 'data', 'wal_archive'))
}, },
'authentication': { 'authentication': {
'superuser': {'password': 'zalando1'}, 'superuser': {'password': 'zalando1'},
@@ -743,14 +726,14 @@ class PatroniPoolController(object):
'bootstrap': { 'bootstrap': {
'method': 'backup_restore', 'method': 'backup_restore',
'backup_restore': { 'backup_restore': {
'command': (self.PYTHON + ' features/backup_restore.py --sourcedir=' + 'command': (sys.executable + ' features/backup_restore.py --sourcedir=' +
os.path.join(self.patroni_path, 'data', 'basebackup').replace('\\', '/')), os.path.join(self.patroni_path, 'data', 'basebackup')),
'recovery_conf': { 'recovery_conf': {
'recovery_target_action': 'promote', 'recovery_target_action': 'promote',
'recovery_target_timeline': 'latest', 'recovery_target_timeline': 'latest',
'restore_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode restore ' + 'restore_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode restore ' +
'--dirname {} --filename %f --pathname %p').format( '--dirname {} --filename %f --pathname %p').format(
os.path.join(self.patroni_path, 'data', 'wal_archive').replace('\\', '/')) os.path.join(self.patroni_path, 'data', 'wal_archive'))
} }
} }
}, },
@@ -890,10 +873,7 @@ class WatchdogMonitor(object):
# actions to execute on start/stop of the tests and before running individual features # actions to execute on start/stop of the tests and before running individual features
def before_all(context): def before_all(context):
os.environ.update({'PATRONI_RESTAPI_USERNAME': 'username', 'PATRONI_RESTAPI_PASSWORD': 'password'}) os.environ.update({'PATRONI_RESTAPI_USERNAME': 'username', 'PATRONI_RESTAPI_PASSWORD': 'password'})
context.request_executor = PatroniRequest({'ctl': {'auth': os.environ['PATRONI_RESTAPI_USERNAME'] + context.ci = any(a in os.environ for a in ('TRAVIS_BUILD_NUMBER', 'BUILD_NUMBER', 'GITHUB_ACTIONS'))
':' + os.environ['PATRONI_RESTAPI_PASSWORD']}})
context.ci = os.name == 'nt' or\
any(a in os.environ for a in ('TRAVIS_BUILD_NUMBER', 'BUILD_NUMBER', 'GITHUB_ACTIONS'))
context.timeout_multiplier = 5 if context.ci else 1 # MacOS sometimes is VERY slow context.timeout_multiplier = 5 if context.ci else 1 # MacOS sometimes is VERY slow
context.pctl = PatroniPoolController(context) context.pctl = PatroniPoolController(context)
context.dcs_ctl = context.pctl.known_dcs[context.pctl.dcs](context) context.dcs_ctl = context.pctl.known_dcs[context.pctl.dcs](context)
@@ -913,26 +893,13 @@ def after_all(context):
def before_feature(context, feature): def before_feature(context, feature):
""" create per-feature output directory to collect Patroni and PostgreSQL logs """ """ create per-feature output directory to collect Patroni and PostgreSQL logs """
if feature.name == 'watchdog' and os.name == 'nt': context.pctl.create_and_set_output_directory(feature.name)
feature.skip("Watchdog isn't supported on Windows")
else:
context.pctl.create_and_set_output_directory(feature.name)
def after_feature(context, feature): def after_feature(context, feature):
""" stop all Patronis, remove their data directory and cleanup the keys in etcd """ """ stop all Patronis, remove their data directory and cleanup the keys in etcd """
context.pctl.stop_all() context.pctl.stop_all()
data = os.path.join(context.pctl.patroni_path, 'data') shutil.rmtree(os.path.join(context.pctl.patroni_path, 'data'))
if os.path.exists(data):
shutil.rmtree(data)
context.dcs_ctl.cleanup_service_tree() context.dcs_ctl.cleanup_service_tree()
if feature.status == 'failed': if feature.status == 'failed':
shutil.copytree(context.pctl.output_dir, context.pctl.output_dir + '_failed') shutil.copytree(context.pctl.output_dir, context.pctl.output_dir + '_failed')
def before_scenario(context, scenario):
if 'slot-advance' in scenario.effective_tags:
for p in context.pctl._processes.values():
if p._conn and p._conn.server_version < 110000:
scenario.skip('pg_replication_slot_advance() is not supported on {0}'.format(p._conn.server_version))
break
+1 -1
View File
@@ -3,7 +3,7 @@ Feature: ignored slots
Given I start postgres1 Given I start postgres1
Then postgres1 is a leader after 10 seconds Then postgres1 is a leader after 10 seconds
And there is a non empty initialize key in DCS after 15 seconds And there is a non empty initialize key in DCS after 15 seconds
When I issue a PATCH request to http://127.0.0.1:8009/config with {"ignore_slots": [{"name": "unmanaged_slot_0", "database": "postgres", "plugin": "test_decoding", "type": "logical"}, {"name": "unmanaged_slot_1", "database": "postgres", "plugin": "test_decoding"}, {"name": "unmanaged_slot_2", "database": "postgres"}, {"name": "unmanaged_slot_3"}], "postgresql": {"parameters": {"wal_level": "logical"}}} When I issue a PATCH request to http://127.0.0.1:8009/config with {"loop_wait": 2, "ignore_slots": [{"name": "unmanaged_slot_0", "database": "postgres", "plugin": "test_decoding", "type": "logical"}, {"name": "unmanaged_slot_1", "database": "postgres", "plugin": "test_decoding"}, {"name": "unmanaged_slot_2", "database": "postgres"}, {"name": "unmanaged_slot_3"}], "postgresql": {"parameters": {"wal_level": "logical"}}}
Then I receive a response code 200 Then I receive a response code 200
And Response on GET http://127.0.0.1:8009/config contains ignore_slots after 10 seconds And Response on GET http://127.0.0.1:8009/config contains ignore_slots after 10 seconds
# Make sure the wal_level has been changed. # Make sure the wal_level has been changed.
+4 -4
View File
@@ -35,13 +35,13 @@ Scenario: check local configuration reload
Then I receive a response code 202 Then I receive a response code 202
Scenario: check dynamic configuration change via DCS Scenario: check dynamic configuration change via DCS
Given I run patronictl.py edit-config -s 'ttl=10' -p 'max_connections=101' --force batman Given I run patronictl.py edit-config -s 'ttl=10' -s 'loop_wait=2' -p 'max_connections=101' --force batman
Then I receive a response returncode 0 Then I receive a response returncode 0
And I receive a response output "+ttl: 10" And I receive a response output "+loop_wait: 2"
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 11 seconds And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 11 seconds
When I issue a GET request to http://127.0.0.1:8008/config When I issue a GET request to http://127.0.0.1:8008/config
Then I receive a response code 200 Then I receive a response code 200
And I receive a response ttl 10 And I receive a response loop_wait 2
When I issue a GET request to http://127.0.0.1:8008/patroni When I issue a GET request to http://127.0.0.1:8008/patroni
Then I receive a response code 200 Then I receive a response code 200
And I receive a response tags {'new_tag': 'new_value'} And I receive a response tags {'new_tag': 'new_value'}
@@ -109,7 +109,7 @@ Scenario: check the scheduled switchover
And I receive a response output "Can't schedule switchover in the paused state" And I receive a response output "Can't schedule switchover in the paused state"
When I run patronictl.py resume batman When I run patronictl.py resume batman
Then I receive a response returncode 0 Then I receive a response returncode 0
Given I issue a scheduled switchover from postgres1 to postgres0 in 10 seconds Given I issue a scheduled switchover from postgres1 to postgres0 in 5 seconds
Then I receive a response returncode 0 Then I receive a response returncode 0
And postgres0 is a leader after 20 seconds And postgres0 is a leader after 20 seconds
And postgres0 role is the primary after 10 seconds And postgres0 role is the primary after 10 seconds
+2 -2
View File
@@ -3,7 +3,7 @@ Feature: standby cluster
Given I start postgres1 Given I start postgres1
Then postgres1 is a leader after 10 seconds Then postgres1 is a leader after 10 seconds
And there is a non empty initialize key in DCS after 15 seconds And there is a non empty initialize key in DCS after 15 seconds
When I issue a PATCH request to http://127.0.0.1:8009/config with {"slots": {"pm_1": {"type": "physical"}}, "postgresql": {"parameters": {"wal_level": "logical"}}} When I issue a PATCH request to http://127.0.0.1:8009/config with {"loop_wait": 2, "slots": {"pm_1": {"type": "physical"}}, "postgresql": {"parameters": {"wal_level": "logical"}}}
Then I receive a response code 200 Then I receive a response code 200
And Response on GET http://127.0.0.1:8009/config contains slots after 10 seconds And Response on GET http://127.0.0.1:8009/config contains slots after 10 seconds
And I sleep for 3 seconds And I sleep for 3 seconds
@@ -14,7 +14,7 @@ Feature: standby cluster
Then "members/postgres0" key in DCS has state=running after 10 seconds Then "members/postgres0" key in DCS has state=running after 10 seconds
And replication works from postgres1 to postgres0 after 15 seconds And replication works from postgres1 to postgres0 after 15 seconds
@slot-advance @skip
Scenario: check permanent logical slots are synced to the replica Scenario: check permanent logical slots are synced to the replica
Given I run patronictl.py restart batman postgres1 --force Given I run patronictl.py restart batman postgres1 --force
Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds
+5 -3
View File
@@ -10,8 +10,10 @@ import yaml
from behave import register_type, step, then from behave import register_type, step, then
from dateutil import tz from dateutil import tz
from datetime import datetime, timedelta from datetime import datetime, timedelta
from patroni.request import PatroniRequest
tzutc = tz.tzutc() tzutc = tz.tzutc()
request_executor = PatroniRequest({'ctl': {'auth': 'username:password'}})
@parse.with_pattern(r'https?://(?:\w|\.|:|/)+') @parse.with_pattern(r'https?://(?:\w|\.|:|/)+')
@@ -73,9 +75,9 @@ def do_post_empty(context, url):
def do_request(context, request_method, url, data): def do_request(context, request_method, url, data):
data = data and json.loads(data) data = data and json.loads(data)
try: try:
r = context.request_executor.request(request_method, url, data) r = request_executor.request(request_method, url, data)
if request_method == 'PATCH' and r.status == 409: if request_method == 'PATCH' and r.status == 409:
r = context.request_executor.request(request_method, url, data) r = request_executor.request(request_method, url, data)
except Exception: except Exception:
context.status_code = context.response = None context.status_code = context.response = None
else: else:
@@ -137,7 +139,7 @@ def add_tag_to_config(context, tag, value, pg_name):
def check_http_response(context, url, value, timeout, negate=False): def check_http_response(context, url, value, timeout, negate=False):
timeout *= context.timeout_multiplier timeout *= context.timeout_multiplier
for _ in range(int(timeout)): for _ in range(int(timeout)):
r = context.request_executor.request('GET', url) r = request_executor.request('GET', url)
if (value in r.data.decode('utf-8')) != negate: if (value in r.data.decode('utf-8')) != negate:
break break
time.sleep(1) time.sleep(1)
+13 -8
View File
@@ -1,12 +1,17 @@
import os import os
import sys
import time import time
from behave import step from behave import step
def callbacks(context, name): select_replication_query = """
return {c: '{0} features/callback2.py {1}'.format(context.pctl.PYTHON, name) SELECT * FROM pg_catalog.pg_stat_replication
for c in ('on_start', 'on_stop', 'on_restart', 'on_role_change')} WHERE application_name = '{0}'
"""
executable = sys.executable if os.name != 'nt' else sys.executable.replace('\\', '/')
callback = executable + " features/callback2.py "
@step('I start {name:w} in a cluster {cluster_name:w}') @step('I start {name:w} in a cluster {cluster_name:w}')
@@ -14,10 +19,10 @@ def start_patroni(context, name, cluster_name):
return context.pctl.start(name, custom_config={ return context.pctl.start(name, custom_config={
"scope": cluster_name, "scope": cluster_name,
"postgresql": { "postgresql": {
"callbacks": callbacks(context, name), "callbacks": {c: callback + name for c in ('on_start', 'on_stop', 'on_restart', 'on_role_change')},
"backup_restore": { "backup_restore": {
"command": (context.pctl.PYTHON + " features/backup_restore.py --sourcedir=" + "command": (executable + " features/backup_restore.py --sourcedir=" +
os.path.join(context.pctl.patroni_path, 'data', 'basebackup').replace('\\', '/'))} os.path.join(context.pctl.patroni_path, 'data', 'basebackup'))}
} }
}) })
@@ -44,7 +49,7 @@ def start_patroni_standby_cluster(context, name, cluster_name, name2):
} }
}, },
"postgresql": { "postgresql": {
"callbacks": callbacks(context, name) "callbacks": {c: callback + name for c in ('on_start', 'on_stop', 'on_restart', 'on_role_change')}
} }
}) })
return context.pctl.start(name) return context.pctl.start(name)
@@ -57,7 +62,7 @@ def check_replication_status(context, pg_name1, pg_name2, timeout):
while time.time() < bound_time: while time.time() < bound_time:
cur = context.pctl.query( cur = context.pctl.query(
pg_name2, pg_name2,
"SELECT * FROM pg_catalog.pg_stat_replication WHERE application_name = '{0}'".format(pg_name1), select_replication_query.format(pg_name1),
fail_ok=True fail_ok=True
) )
+2 -2
View File
@@ -1,5 +1,5 @@
FROM postgres:15 FROM postgres:11
LABEL maintainer="Alexander Kukushkin <akukushkin@microsoft.com>" MAINTAINER Alexander Kukushkin <alexander.kukushkin@zalando.de>
RUN export DEBIAN_FRONTEND=noninteractive \ RUN export DEBIAN_FRONTEND=noninteractive \
&& echo 'APT::Install-Recommends "0";\nAPT::Install-Suggests "0";' > /etc/apt/apt.conf.d/01norecommend \ && echo 'APT::Install-Recommends "0";\nAPT::Install-Suggests "0";' > /etc/apt/apt.conf.d/01norecommend \
+21 -30
View File
@@ -38,7 +38,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
self.log_request(status_code) self.log_request(status_code)
def _write_response(self, status_code, body, content_type='text/html', headers=None): def _write_response(self, status_code, body, content_type='text/html', headers=None):
# TODO: try-catch ConnectionResetError: [Errno 104] Connection reset by peer and log it in DEBUG level
self.send_response(status_code) self.send_response(status_code)
headers = headers or {} headers = headers or {}
if content_type: if content_type:
@@ -142,7 +141,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
ignore_tags = True ignore_tags = True
elif 'replica' in path: elif 'replica' in path:
status_code = replica_status_code status_code = replica_status_code
elif 'read-only' in path and 'sync' not in path: elif 'read-only' in path:
status_code = 200 if 200 in (primary_status_code, standby_leader_status_code) else replica_status_code status_code = 200 if 200 in (primary_status_code, standby_leader_status_code) else replica_status_code
elif 'health' in path: elif 'health' in path:
status_code = 200 if response.get('state') == 'running' else 503 status_code = 200 if response.get('state') == 'running' else 503
@@ -186,20 +185,8 @@ class RestApiHandler(BaseHTTPRequestHandler):
def do_OPTIONS(self): def do_OPTIONS(self):
self.do_GET(write_status_code_only=True) self.do_GET(write_status_code_only=True)
def do_HEAD(self):
self.do_GET(write_status_code_only=True)
def do_GET_liveness(self): def do_GET_liveness(self):
patroni = self.server.patroni self._write_status_code_only(200)
is_primary = patroni.postgresql.role == 'master' and patroni.postgresql.is_running()
# We can tolerate Patroni problems longer on the replica.
# On the primary the liveness probe most likely will start failing only after the leader key expired.
# It should not be a big problem because replicas will see that the primary is still alive via REST API call.
liveness_threshold = patroni.dcs.ttl * (1 if is_primary else 2)
# In maintenance mode (pause) we are fine if heartbeat loop stuck.
status_code = 200 if patroni.ha.is_paused() or patroni.next_run + liveness_threshold > time.time() else 503
self._write_status_code_only(status_code)
def do_GET_readiness(self): def do_GET_readiness(self):
patroni = self.server.patroni patroni = self.server.patroni
@@ -368,14 +355,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
self.server.patroni.sighup_handler() self.server.patroni.sighup_handler()
self._write_response(202, 'reload scheduled') self._write_response(202, 'reload scheduled')
@check_access
def do_POST_sigterm(self):
"""Only for behave testing on windows"""
if os.name == 'nt' and os.getenv('BEHAVE_DEBUG'):
self.server.patroni.api_sigterm()
self._write_response(202, 'shutdown scheduled')
@staticmethod @staticmethod
def parse_schedule(schedule, action): def parse_schedule(schedule, action):
""" parses the given schedule and validates at """ """ parses the given schedule and validates at """
@@ -833,20 +812,32 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
else: else:
logger.error('Bad value in the "restapi.verify_client": %s', verify_client) logger.error('Bad value in the "restapi.verify_client": %s', verify_client)
self.__ssl_serial_number = self.get_certificate_serial_number() self.__ssl_serial_number = self.get_certificate_serial_number()
self.socket = ctx.wrap_socket(self.socket, server_side=True, do_handshake_on_connect=False) self.socket = ctx.wrap_socket(self.socket, server_side=True)
if reloading_config: if reloading_config:
self.start() self.start()
def process_request_thread(self, request, client_address): def process_request_thread(self, request, client_address):
enable_keepalive(request, 10, 3) if isinstance(request, tuple):
if hasattr(request, 'context'): # SSLSocket sock, newsock = request
request.do_handshake() try:
request = sock.context.wrap_socket(newsock, do_handshake_on_connect=sock.do_handshake_on_connect,
suppress_ragged_eofs=sock.suppress_ragged_eofs, server_side=True)
except socket.error:
return
super(RestApiServer, self).process_request_thread(request, client_address) super(RestApiServer, self).process_request_thread(request, client_address)
def get_request(self):
sock = self.socket
newsock, addr = socket.socket.accept(sock)
enable_keepalive(newsock, 10, 3)
if hasattr(sock, 'context'): # SSLSocket, we want to do the deferred handshake from a thread
newsock = (sock, newsock)
return newsock, addr
def shutdown_request(self, request): def shutdown_request(self, request):
if hasattr(request, 'context'): # SSLSocket if isinstance(request, tuple):
request.unwrap() _, request = request # SSLSocket
super(RestApiServer, self).shutdown_request(request) return super(RestApiServer, self).shutdown_request(request)
def get_certificate_serial_number(self): def get_certificate_serial_number(self):
if self.__ssl_options.get('certfile'): if self.__ssl_options.get('certfile'):
+6 -8
View File
@@ -32,7 +32,7 @@ _AUTH_ALLOWED_PARAMETERS = (
def default_validator(conf): def default_validator(conf):
if not conf: if not conf:
raise ConfigParseError("Config is empty.") return "Config is empty."
class Config(object): class Config(object):
@@ -102,9 +102,9 @@ class Config(object):
config_env = os.environ.pop(self.PATRONI_CONFIG_VARIABLE, None) config_env = os.environ.pop(self.PATRONI_CONFIG_VARIABLE, None)
self._local_configuration = config_env and yaml.safe_load(config_env) or self.__environment_configuration self._local_configuration = config_env and yaml.safe_load(config_env) or self.__environment_configuration
if validator: if validator:
errors = validator(self._local_configuration) error = validator(self._local_configuration)
if errors: if error:
raise ConfigParseError("\n".join(errors)) raise ConfigParseError(error)
self.__effective_configuration = self._build_effective_configuration({}, self._local_configuration) self.__effective_configuration = self._build_effective_configuration({}, self._local_configuration)
self._data_dir = self.__effective_configuration.get('postgresql', {}).get('data_dir', "") self._data_dir = self.__effective_configuration.get('postgresql', {}).get('data_dir', "")
@@ -227,8 +227,7 @@ class Config(object):
for name, value in (value or {}).items(): for name, value in (value or {}).items():
if name == 'parameters': if name == 'parameters':
config['postgresql'][name].update(self._process_postgresql_parameters(value)) config['postgresql'][name].update(self._process_postgresql_parameters(value))
elif name not in ('connect_address', 'proxy_address', 'listen', elif name not in ('connect_address', 'listen', 'data_dir', 'pgpass', 'authentication'):
'config_dir', 'data_dir', 'pgpass', 'authentication'):
config['postgresql'][name] = deepcopy(value) config['postgresql'][name] = deepcopy(value)
elif name == 'standby_cluster': elif name == 'standby_cluster':
for name, value in (value or {}).items(): for name, value in (value or {}).items():
@@ -272,8 +271,7 @@ class Config(object):
'cafile', 'ciphers', 'verify_client', 'http_extra_headers', 'cafile', 'ciphers', 'verify_client', 'http_extra_headers',
'https_extra_headers', 'allowlist', 'allowlist_include_members']) 'https_extra_headers', 'allowlist', 'allowlist_include_members'])
_set_section_values('ctl', ['insecure', 'cacert', 'certfile', 'keyfile', 'keyfile_password']) _set_section_values('ctl', ['insecure', 'cacert', 'certfile', 'keyfile', 'keyfile_password'])
_set_section_values('postgresql', ['listen', 'connect_address', 'proxy_address', _set_section_values('postgresql', ['listen', 'connect_address', 'config_dir', 'data_dir', 'pgpass', 'bin_dir'])
'config_dir', 'data_dir', 'pgpass', 'bin_dir'])
_set_section_values('log', ['level', 'traceback_level', 'format', 'dateformat', 'max_queue_size', _set_section_values('log', ['level', 'traceback_level', 'format', 'dateformat', 'max_queue_size',
'dir', 'file_size', 'file_num', 'loggers']) 'dir', 'file_size', 'file_num', 'loggers'])
_set_section_values('raft', ['data_dir', 'self_addr', 'partner_addrs', 'password', 'bind_addr']) _set_section_values('raft', ['data_dir', 'self_addr', 'partner_addrs', 'password', 'bind_addr'])
+2 -13
View File
@@ -60,18 +60,6 @@ class PatronictlPrettyTable(PrettyTable):
self.__hline_num = 0 self.__hline_num = 0
self.__hline = None self.__hline = None
def __build_header(self, line):
header = self.__table_header[:len(line) - 2]
return "".join([line[0], header, line[1 + len(header):]])
def _stringify_hrule(self, *args, **kwargs):
ret = super(PatronictlPrettyTable, self)._stringify_hrule(*args, **kwargs)
where = args[1] if len(args) > 1 else kwargs.get('where')
if where == 'top_' and self.__table_header:
ret = self.__build_header(ret)
self.__hline_num += 1
return ret
def _is_first_hline(self): def _is_first_hline(self):
return self.__hline_num == 0 return self.__hline_num == 0
@@ -83,7 +71,8 @@ class PatronictlPrettyTable(PrettyTable):
# Inject nice table header # Inject nice table header
if self._is_first_hline() and self.__table_header: if self._is_first_hline() and self.__table_header:
ret = self.__build_header(ret) header = self.__table_header[:len(ret) - 2]
ret = "".join([ret[0], header, ret[1 + len(header):]])
self.__hline_num += 1 self.__hline_num += 1
return ret return ret
+5 -13
View File
@@ -1,5 +1,3 @@
from __future__ import print_function
import abc import abc
import os import os
import signal import signal
@@ -24,15 +22,11 @@ class AbstractPatroniDaemon(object):
def sighup_handler(self, *args): def sighup_handler(self, *args):
self._received_sighup = True self._received_sighup = True
def api_sigterm(self): def sigterm_handler(self, *args):
with self._sigterm_lock: with self._sigterm_lock:
if not self._received_sigterm: if not self._received_sigterm:
self._received_sigterm = True self._received_sigterm = True
return True sys.exit()
def sigterm_handler(self, *args):
if self.api_sigterm():
sys.exit()
def setup_signal_handlers(self): def setup_signal_handlers(self):
self._received_sighup = False self._received_sighup = False
@@ -89,18 +83,16 @@ def abstract_main(cls, validator=None):
help='Patroni may also read the configuration from the {0} environment variable' help='Patroni may also read the configuration from the {0} environment variable'
.format(Config.PATRONI_CONFIG_VARIABLE)) .format(Config.PATRONI_CONFIG_VARIABLE))
args = parser.parse_args() args = parser.parse_args()
validate_config = validator and args.validate_config
try: try:
if validate_config: if validator and args.validate_config:
Config(args.configfile, validator=validator) Config(args.configfile, validator=validator)
sys.exit() sys.exit()
config = Config(args.configfile) config = Config(args.configfile)
except ConfigParseError as e: except ConfigParseError as e:
if e.value: if e.value:
print(e.value, file=sys.stderr) print(e.value)
if not validate_config: parser.print_help()
parser.print_help()
sys.exit(1) sys.exit(1)
controller = cls(config) controller = cls(config)
-3
View File
@@ -236,7 +236,6 @@ class Consul(AbstractDCS):
self._service_check_tls_server_name = config.get('service_check_tls_server_name', None) self._service_check_tls_server_name = config.get('service_check_tls_server_name', None)
if not self._ctl: if not self._ctl:
self.create_session() self.create_session()
self._previous_loop_token = self._client.token
def retry(self, *args, **kwargs): def retry(self, *args, **kwargs):
return self._retry.copy()(*args, **kwargs) return self._retry.copy()(*args, **kwargs)
@@ -465,7 +464,6 @@ class Consul(AbstractDCS):
tags = self._service_tags[:] tags = self._service_tags[:]
tags.append(role) tags.append(role)
self._previous_loop_service_tags = self._service_tags self._previous_loop_service_tags = self._service_tags
self._previous_loop_token = self._client.token
params = { params = {
'service_id': '{0}/{1}'.format(self._scope, self._name), 'service_id': '{0}/{1}'.format(self._scope, self._name),
@@ -502,7 +500,6 @@ class Consul(AbstractDCS):
if ( if (
force or update or self._register_service != self._previous_loop_register_service force or update or self._register_service != self._previous_loop_register_service
or self._service_tags != self._previous_loop_service_tags or self._service_tags != self._previous_loop_service_tags
or self._client.token != self._previous_loop_token
): ):
return self._update_service(new_data) return self._update_service(new_data)
+4 -7
View File
@@ -216,13 +216,10 @@ class AbstractEtcdClientWithFailover(etcd.Client):
return response return response
except (HTTPError, HTTPException, socket.error, socket.timeout) as e: except (HTTPError, HTTPException, socket.error, socket.timeout) as e:
self.http.clear() self.http.clear()
if not retry: # switch to the next etcd node because we don't know exactly what happened,
if len(machines_cache) == 1: # whether the key didn't received an update or there is a network problem.
self.set_base_uri(self._base_uri) # trigger Etcd3 watcher restart if not retry and i + 1 < len(machines_cache):
# switch to the next etcd node because we don't know exactly what happened, self.set_base_uri(machines_cache[i + 1])
# whether the key didn't received an update or there is a network problem.
elif i + 1 < len(machines_cache):
self.set_base_uri(machines_cache[i + 1])
if (isinstance(fields, dict) and fields.get("wait") == "true" and if (isinstance(fields, dict) and fields.get("wait") == "true" and
isinstance(e, (ReadTimeoutError, ProtocolError))): isinstance(e, (ReadTimeoutError, ProtocolError))):
logger.debug("Watch timed out.") logger.debug("Watch timed out.")
+6 -16
View File
@@ -11,7 +11,6 @@ import time
import urllib3 import urllib3
from threading import Condition, Lock, Thread from threading import Condition, Lock, Thread
from urllib3.exceptions import ReadTimeoutError, ProtocolError
from . import ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory from . import ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
from .etcd import AbstractEtcdClientWithFailover, AbstractEtcd, catch_etcd_errors from .etcd import AbstractEtcdClientWithFailover, AbstractEtcd, catch_etcd_errors
@@ -351,7 +350,7 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
def deleteprefix(self, key, retry=None): def deleteprefix(self, key, retry=None):
return self.deleterange(key, prefix_range_end(key), retry=retry) return self.deleterange(key, prefix_range_end(key), retry=retry)
def watchrange(self, key, range_end=None, start_revision=None, filters=None, read_timeout=None): def watchrange(self, key, range_end=None, start_revision=None, filters=None):
"""returns: response object""" """returns: response object"""
params = build_range_request(key, range_end) params = build_range_request(key, range_end)
if start_revision is not None: if start_revision is not None:
@@ -359,11 +358,11 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
params['filters'] = filters or [] params['filters'] = filters or []
kwargs = self._prepare_common_parameters(1, self.read_timeout) kwargs = self._prepare_common_parameters(1, self.read_timeout)
request_executor = self._prepare_request(kwargs, {'create_request': params}) request_executor = self._prepare_request(kwargs, {'create_request': params})
kwargs.update(timeout=urllib3.Timeout(connect=kwargs['timeout'], read=read_timeout), retries=0) kwargs.update(timeout=urllib3.Timeout(connect=kwargs['timeout']), retries=0)
return request_executor(self._MPOST, self._base_uri + self.version_prefix + '/watch', **kwargs) return request_executor(self._MPOST, self._base_uri + self.version_prefix + '/watch', **kwargs)
def watchprefix(self, key, start_revision=None, filters=None, read_timeout=None): def watchprefix(self, key, start_revision=None, filters=None):
return self.watchrange(key, prefix_range_end(key), start_revision, filters, read_timeout) return self.watchrange(key, prefix_range_end(key), start_revision, filters)
class KVCache(Thread): class KVCache(Thread):
@@ -452,14 +451,7 @@ class KVCache(Thread):
def _do_watch(self, revision): def _do_watch(self, revision):
with self._response_lock: with self._response_lock:
self._response = None self._response = None
# We do most of requests with timeouts. The only exception /watch requests to Etcd v3. response = self._client.watchprefix(self._dcs.cluster_prefix, revision)
# In order to interrupt the /watch request we do socket.shutdown() from the main thread,
# which doesn't work on Windows. Therefore we want to use the last resort, `read_timeout`.
# Setting it to TTL will help to partially mitigate the problem.
# Setting it to lower value is not nice because for idling clusters it will increase
# the numbers of interrupts and reconnects.
read_timeout = self._dcs.ttl if os.name == 'nt' else None
response = self._client.watchprefix(self._dcs.cluster_prefix, revision, read_timeout=read_timeout)
with self._response_lock: with self._response_lock:
if self._response is None: if self._response is None:
self._response = response self._response = response
@@ -481,9 +473,7 @@ class KVCache(Thread):
try: try:
self._do_watch(result['header']['revision']) self._do_watch(result['header']['revision'])
except Exception as e: except Exception as e:
# Following exceptions are expected on Windows because the /watch request is done with `read_timeout` logger.error('watchprefix failed: %r', e)
if not (os.name == 'nt' and isinstance(e, (ReadTimeoutError, ProtocolError))):
logger.error('watchprefix failed: %r', e)
finally: finally:
with self.condition: with self.condition:
self._is_ready = False self._is_ready = False
+1 -2
View File
@@ -950,8 +950,7 @@ class Kubernetes(AbstractDCS):
if not self._api.create_namespaced_service(self._namespace, body): if not self._api.create_namespaced_service(self._namespace, body):
return return
except Exception as e: except Exception as e:
# 409 - service already exists, 403 - creation forbidden if not isinstance(e, k8s_client.rest.ApiException) or e.status != 409: # Service already exists
if not isinstance(e, k8s_client.rest.ApiException) or e.status not in (409, 403):
return logger.exception('create_config_service failed') return logger.exception('create_config_service failed')
self._should_create_config_service = False self._should_create_config_service = False
+7 -6
View File
@@ -39,9 +39,9 @@ setattr(TCPNode, 'ip', property(resolve_host))
class SyncObjUtility(object): class SyncObjUtility(object):
def __init__(self, otherNodes, conf, retry_timeout=10): def __init__(self, otherNodes, conf):
self._nodes = otherNodes self._nodes = otherNodes
self._utility = TcpUtility(conf.password, retry_timeout/max(1, len(otherNodes))) self._utility = TcpUtility(conf.password)
def executeCommand(self, command): def executeCommand(self, command):
try: try:
@@ -58,11 +58,11 @@ class SyncObjUtility(object):
class DynMemberSyncObj(SyncObj): class DynMemberSyncObj(SyncObj):
def __init__(self, selfAddress, partnerAddrs, conf, retry_timeout=10): def __init__(self, selfAddress, partnerAddrs, conf):
self.__early_apply_local_log = selfAddress is not None self.__early_apply_local_log = selfAddress is not None
self.applied_local_log = False self.applied_local_log = False
utility = SyncObjUtility(partnerAddrs, conf, retry_timeout) utility = SyncObjUtility(partnerAddrs, conf)
members = utility.getMembers() members = utility.getMembers()
add_self = members and selfAddress not in members add_self = members and selfAddress not in members
@@ -97,7 +97,7 @@ class KVStoreTTL(DynMemberSyncObj):
self.__on_set = on_set self.__on_set = on_set
self.__on_delete = on_delete self.__on_delete = on_delete
self.__limb = {} self.__limb = {}
self.set_retry_timeout(int(config.get('retry_timeout') or 10)) self.__retry_timeout = None
self_addr = config.get('self_addr') self_addr = config.get('self_addr')
partner_addrs = set(config.get('partner_addrs', [])) partner_addrs = set(config.get('partner_addrs', []))
@@ -121,7 +121,7 @@ class KVStoreTTL(DynMemberSyncObj):
journalFile=(file_template + '.journal' if self_addr else None), journalFile=(file_template + '.journal' if self_addr else None),
onReady=on_ready, dynamicMembershipChange=True) onReady=on_ready, dynamicMembershipChange=True)
super(KVStoreTTL, self).__init__(self_addr, partner_addrs, conf, self.__retry_timeout) super(KVStoreTTL, self).__init__(self_addr, partner_addrs, conf)
self.__data = {} self.__data = {}
@staticmethod @staticmethod
@@ -275,6 +275,7 @@ class Raft(AbstractDCS):
break break
else: else:
logger.info('waiting on raft') logger.info('waiting on raft')
self.set_retry_timeout(int(config.get('retry_timeout') or 10))
def _on_set(self, key, value): def _on_set(self, key, value):
leader = (self._sync_obj.get(self.leader_path) or {}).get('value') leader = (self._sync_obj.get(self.leader_path) or {}).get('value')
+3 -14
View File
@@ -1,7 +1,6 @@
import json import json
import logging import logging
import select import select
import six
import time import time
from kazoo.client import KazooClient, KazooState, KazooRetry from kazoo.client import KazooClient, KazooState, KazooRetry
@@ -51,21 +50,11 @@ class PatroniSequentialThreadingHandler(SequentialThreadingHandler):
return super(PatroniSequentialThreadingHandler, self).create_connection(*args, **kwargs) return super(PatroniSequentialThreadingHandler, self).create_connection(*args, **kwargs)
def select(self, *args, **kwargs): def select(self, *args, **kwargs):
""" """Python3 raises `ValueError` if socket is closed, because fd == -1"""
Python 3.XY may raise following exceptions if select/poll are called with an invalid socket:
- `ValueError`: because fd == -1
- `TypeError`: Invalid file descriptor: -1 (starting from kazoo 2.9)
Python 2.7 may raise the `IOError` instead of `socket.error` (starting from kazoo 2.9)
When it is appropriate we map these exceptions to `socket.error`.
"""
try: try:
return super(PatroniSequentialThreadingHandler, self).select(*args, **kwargs) return super(PatroniSequentialThreadingHandler, self).select(*args, **kwargs)
except IOError as e: except ValueError as e:
raise (select.error(e.errno, e.strerror) if six.PY2 else e) raise select.error(9, str(e))
except (TypeError, ValueError) as e:
raise (e if six.PY2 and isinstance(e, TypeError) else select.error(9, str(e)))
class PatroniKazooClient(KazooClient): class PatroniKazooClient(KazooClient):
+5 -28
View File
@@ -192,10 +192,6 @@ class Ha(object):
'version': self.patroni.version 'version': self.patroni.version
} }
proxy_url = self.state_handler.proxy_url
if proxy_url:
data['proxy_url'] = proxy_url
if self.is_leader() and not self._rewind.checkpoint_after_promote(): if self.is_leader() and not self._rewind.checkpoint_after_promote():
data['checkpoint_after_promote'] = False data['checkpoint_after_promote'] = False
tags = self.get_effective_tags() tags = self.get_effective_tags()
@@ -277,9 +273,7 @@ class Ha(object):
else: else:
create_replica_methods = self.get_standby_cluster_config().get('create_replica_methods', []) \ create_replica_methods = self.get_standby_cluster_config().get('create_replica_methods', []) \
if self.is_standby_cluster() else None if self.is_standby_cluster() else None
can_bootstrap = self.state_handler.can_create_replica_without_replication_connection(create_replica_methods) if self.state_handler.can_create_replica_without_replication_connection(create_replica_methods):
concurrent_bootstrap = self.cluster.initialize == ""
if can_bootstrap and not concurrent_bootstrap:
msg = 'bootstrap (without leader)' msg = 'bootstrap (without leader)'
return self._async_executor.try_run_async(msg, self.clone) or 'trying to ' + msg return self._async_executor.try_run_async(msg, self.clone) or 'trying to ' + msg
return 'waiting for {0}leader to bootstrap'.format('standby_' if self.is_standby_cluster() else '') return 'waiting for {0}leader to bootstrap'.format('standby_' if self.is_standby_cluster() else '')
@@ -743,11 +737,6 @@ class Ha(object):
return None return None
return False return False
# in synchronous mode when our name is not in the /sync key
# we shouldn't take any action even if the candidate is unhealthy
if self.is_synchronous_mode() and not self.cluster.sync.matches(self.state_handler.name):
return False
# find specific node and check that it is healthy # find specific node and check that it is healthy
member = self.cluster.get_member(failover.candidate, fallback_to_leader=False) member = self.cluster.get_member(failover.candidate, fallback_to_leader=False)
if member: if member:
@@ -808,7 +797,7 @@ class Ha(object):
if self.cluster.failover: if self.cluster.failover:
# When doing a switchover in synchronous mode only synchronous nodes and former leader are allowed to race # When doing a switchover in synchronous mode only synchronous nodes and former leader are allowed to race
if self.is_synchronous_mode() and self.cluster.failover.leader and \ if self.is_synchronous_mode() and self.cluster.failover.leader and \
not self.cluster.sync.matches(self.state_handler.name): self.cluster.failover.candidate and not self.cluster.sync.matches(self.state_handler.name):
return False return False
return self.manual_failover_process_no_leader() return self.manual_failover_process_no_leader()
@@ -1418,14 +1407,7 @@ class Ha(object):
return 'started as a secondary' return 'started as a secondary'
# is data directory empty? # is data directory empty?
try: if self.state_handler.data_directory_empty():
data_directory_is_empty = self.state_handler.data_directory_empty()
data_directory_is_accessible = True
except OSError as e:
data_directory_is_accessible = False
data_directory_error = e
if not data_directory_is_accessible or data_directory_is_empty:
self.state_handler.set_role('uninitialized') self.state_handler.set_role('uninitialized')
self.state_handler.stop('immediate', stop_timeout=self.patroni.config['retry_timeout']) self.state_handler.stop('immediate', stop_timeout=self.patroni.config['retry_timeout'])
# In case datadir went away while we were master. # In case datadir went away while we were master.
@@ -1434,11 +1416,8 @@ class Ha(object):
# is this instance the leader? # is this instance the leader?
if self.has_lock(): if self.has_lock():
self.release_leader_key_voluntarily() self.release_leader_key_voluntarily()
return 'released leader key voluntarily as data dir {0} and currently leader'.format( return 'released leader key voluntarily as data dir empty and currently leader'
'empty' if data_directory_is_accessible else 'not accessible')
if not data_directory_is_accessible:
return 'data directory is not accessible: {0}'.format(data_directory_error)
if self.is_paused(): if self.is_paused():
return 'running with empty data directory' return 'running with empty data directory'
return self.bootstrap() # new node return self.bootstrap() # new node
@@ -1504,9 +1483,7 @@ class Ha(object):
# asynchronous processes are running (should be always the case for the master) # asynchronous processes are running (should be always the case for the master)
if not self._async_executor.busy and not self.state_handler.is_starting(): if not self._async_executor.busy and not self.state_handler.is_starting():
create_slots = self.state_handler.slots_handler.sync_replication_slots(self.cluster, create_slots = self.state_handler.slots_handler.sync_replication_slots(self.cluster,
self.patroni.nofailover, self.patroni.nofailover)
self.patroni.replicatefrom,
self.is_paused())
if not self.state_handler.cb_called: if not self.state_handler.cb_called:
if not self.state_handler.is_leader(): if not self.state_handler.is_leader():
self._rewind.trigger_check_diverged_lsn() self._rewind.trigger_check_diverged_lsn()
+4 -4
View File
@@ -428,11 +428,11 @@ class Postgresql(object):
# If the cluster is shutdown with archive_mode=on, WAL is switched before writing the checkpoint. # If the cluster is shutdown with archive_mode=on, WAL is switched before writing the checkpoint.
# In this case we want to take the LSN of previous record (switch) as the last known WAL location. # In this case we want to take the LSN of previous record (switch) as the last known WAL location.
if parse_lsn(lsn) == prev and desc.strip() in ('xlog switch', 'SWITCH'): if parse_lsn(lsn) == prev and desc.strip() in ('xlog switch', 'SWITCH'):
return prev return str(prev)
except Exception as e: except Exception as e:
logger.error('Exception when parsing WAL pg_%sdump output: %r', self.wal_name, e) logger.error('Exception when parsing WAL pg_%sdump output: %r', self.wal_name, e)
if isinstance(checkpoint_lsn, six.integer_types): if isinstance(checkpoint_lsn, six.integer_types):
return checkpoint_lsn return str(checkpoint_lsn)
def is_running(self): def is_running(self):
"""Returns PostmasterProcess if one is running on the data directory or None. If most recently seen process """Returns PostmasterProcess if one is running on the data directory or None. If most recently seen process
@@ -659,7 +659,7 @@ class Postgresql(object):
if on_safepoint: if on_safepoint:
# Wait for our connection to terminate so we can be sure that no new connections are being initiated # Wait for our connection to terminate so we can be sure that no new connections are being initiated
self._wait_for_connection_close(postmaster) self._wait_for_connection_close(postmaster)
postmaster.wait_for_user_backends_to_close(stop_timeout) postmaster.wait_for_user_backends_to_close()
on_safepoint() on_safepoint()
if on_shutdown and mode in ('fast', 'smart'): if on_shutdown and mode in ('fast', 'smart'):
@@ -668,7 +668,7 @@ class Postgresql(object):
while postmaster.is_running(): while postmaster.is_running():
data = self.controldata() data = self.controldata()
if data.get('Database cluster state', '') == 'shut down': if data.get('Database cluster state', '') == 'shut down':
on_shutdown(self.latest_checkpoint_location()) on_shutdown(int(self.latest_checkpoint_location()))
break break
elif data.get('Database cluster state', '').startswith('shut down'): # shut down in recovery elif data.get('Database cluster state', '').startswith('shut down'): # shut down in recovery
break break
-3
View File
@@ -1017,9 +1017,6 @@ class ConfigHandler(object):
if not local_connection_address_changed: if not local_connection_address_changed:
self.resolve_connection_addresses() self.resolve_connection_addresses()
proxy_addr = config.get('proxy_address')
self._postgresql.proxy_url = uri('postgres', proxy_addr, self._postgresql.database) if proxy_addr else None
if conf_changed: if conf_changed:
self.write_postgresql_conf() self.write_postgresql_conf()
+7 -11
View File
@@ -171,8 +171,8 @@ class PostmasterProcess(psutil.Process):
else: else:
return not self.is_running() return not self.is_running()
def wait_for_user_backends_to_close(self, stop_timeout): def wait_for_user_backends_to_close(self):
# These regexps are cross checked against versions PostgreSQL 9.1 .. 15 # These regexps are cross checked against versions PostgreSQL 9.1 .. 11
aux_proc_re = re.compile("(?:postgres:)( .*:)? (?:(?:archiver|startup|autovacuum launcher|autovacuum worker|" aux_proc_re = re.compile("(?:postgres:)( .*:)? (?:(?:archiver|startup|autovacuum launcher|autovacuum worker|"
"checkpointer|logger|stats collector|wal receiver|wal writer|writer)(?: process )?|" "checkpointer|logger|stats collector|wal receiver|wal writer|writer)(?: process )?|"
"walreceiver|wal sender process|walsender|walwriter|background writer|" "walreceiver|wal sender process|walsender|walwriter|background writer|"
@@ -184,23 +184,19 @@ class PostmasterProcess(psutil.Process):
return logger.debug('Failed to get list of postmaster children') return logger.debug('Failed to get list of postmaster children')
user_backends = [] user_backends = []
user_backends_cmdlines = {} user_backends_cmdlines = []
for child in children: for child in children:
try: try:
cmdline = child.cmdline() cmdline = child.cmdline()
if cmdline and not aux_proc_re.match(cmdline[0]): if cmdline and not aux_proc_re.match(cmdline[0]):
user_backends.append(child) user_backends.append(child)
user_backends_cmdlines[child.pid] = cmdline[0] user_backends_cmdlines.append(cmdline[0])
except psutil.NoSuchProcess: except psutil.NoSuchProcess:
pass pass
if user_backends: if user_backends:
logger.debug('Waiting for user backends %s to close', ', '.join(user_backends_cmdlines.values())) logger.debug('Waiting for user backends %s to close', ', '.join(user_backends_cmdlines))
gone, live = psutil.wait_procs(user_backends, stop_timeout) psutil.wait_procs(user_backends)
if stop_timeout and live: logger.debug("Backends closed")
live = [user_backends_cmdlines[b.pid] for b in live]
logger.warning('Backends still alive after %s: %s', stop_timeout, ', '.join(live))
else:
logger.debug("Backends closed")
@staticmethod @staticmethod
def start(pgcommand, data_dir, conf, options): def start(pgcommand, data_dir, conf, options):
+8 -58
View File
@@ -1,8 +1,6 @@
import logging import logging
import os import os
import re
import shlex import shlex
import shutil
import six import six
import subprocess import subprocess
@@ -279,37 +277,28 @@ class Rewind(object):
def checkpoint_after_promote(self): def checkpoint_after_promote(self):
return self._state == REWIND_STATUS.CHECKPOINT return self._state == REWIND_STATUS.CHECKPOINT
def _buid_archiver_command(self, command, wal_filename): def _fetch_missing_wal(self, restore_command, wal_filename):
"""Replace placeholders in the given archiver command's template.
Applicable for archive_command and restore_command.
Can also be used for archive_cleanup_command and recovery_end_command,
however %r value is always set to 000000010000000000000001."""
cmd = '' cmd = ''
length = len(command) length = len(restore_command)
i = 0 i = 0
while i < length: while i < length:
if command[i] == '%' and i + 1 < length: if restore_command[i] == '%' and i + 1 < length:
i += 1 i += 1
if command[i] == 'p': if restore_command[i] == 'p':
cmd += os.path.join(self._postgresql.wal_dir, wal_filename) cmd += os.path.join(self._postgresql.wal_dir, wal_filename)
elif command[i] == 'f': elif restore_command[i] == 'f':
cmd += wal_filename cmd += wal_filename
elif command[i] == 'r': elif restore_command[i] == 'r':
cmd += '000000010000000000000001' cmd += '000000010000000000000001'
elif command[i] == '%': elif restore_command[i] == '%':
cmd += '%' cmd += '%'
else: else:
cmd += '%' cmd += '%'
i -= 1 i -= 1
else: else:
cmd += command[i] cmd += restore_command[i]
i += 1 i += 1
return cmd
def _fetch_missing_wal(self, restore_command, wal_filename):
cmd = self._buid_archiver_command(restore_command, wal_filename)
logger.info('Trying to fetch the missing wal: %s', cmd) logger.info('Trying to fetch the missing wal: %s', cmd)
return self._postgresql.cancellable.call(shlex.split(cmd)) == 0 return self._postgresql.cancellable.call(shlex.split(cmd)) == 0
@@ -326,42 +315,6 @@ class Rewind(object):
if waldir.endswith('/pg_' + self._postgresql.wal_name) and len(wal_filename) == 24: if waldir.endswith('/pg_' + self._postgresql.wal_name) and len(wal_filename) == 24:
return wal_filename return wal_filename
def _archive_ready_wals(self):
"""Try to archive WALs that have .ready files just in case
archive_mode was not set to 'always' before promote, while
after it the WALs were recycled on the promoted replica.
With this we prevent the entire loss of such WALs and the
consequent old leader's start failure."""
archive_mode = self._postgresql.get_guc_value('archive_mode')
archive_cmd = self._postgresql.get_guc_value('archive_command')
if archive_mode not in ('on', 'always') or not archive_cmd:
return
walseg_regex = re.compile(r'^[0-9A-F]{24}(\.partial){0,1}\.ready$')
status_dir = os.path.join(self._postgresql.wal_dir, 'archive_status')
try:
wals_to_archive = [f[:-6] for f in os.listdir(status_dir) if walseg_regex.match(f)]
except OSError as e:
return logger.error('Unable to list %s: %r', status_dir, e)
# skip fsync, as postgres --single or pg_rewind will anyway run it
for wal in sorted(wals_to_archive):
old_name = os.path.join(status_dir, wal + '.ready')
# wal file might have alredy been archived
if os.path.isfile(old_name) and os.path.isfile(os.path.join(self._postgresql.wal_dir, wal)):
cmd = self._buid_archiver_command(archive_cmd, wal)
# it is the author of archive_command, who is responsible
# for not overriding the WALs already present in archive
logger.info('Trying to archive %s: %s', wal, cmd)
if self._postgresql.cancellable.call(shlex.split(cmd)) == 0:
new_name = os.path.join(status_dir, wal + '.done')
try:
shutil.move(old_name, new_name)
except Exception as e:
logger.error('Unable to rename %s to %s: %r', old_name, new_name, e)
else:
logger.info('Failed to archive WAL segment %s', wal)
def pg_rewind(self, r): def pg_rewind(self, r):
# prepare pg_rewind connection # prepare pg_rewind connection
env = self._postgresql.config.write_pgpass(r) env = self._postgresql.config.write_pgpass(r)
@@ -414,8 +367,6 @@ class Rewind(object):
if self._postgresql.is_running() and not self._postgresql.stop(checkpoint=False): if self._postgresql.is_running() and not self._postgresql.stop(checkpoint=False):
return logger.warning('Can not run pg_rewind because postgres is still running') return logger.warning('Can not run pg_rewind because postgres is still running')
self._archive_ready_wals()
# prepare pg_rewind connection # prepare pg_rewind connection
r = self._conn_kwargs(leader, self._postgresql.config.rewind_credentials) r = self._conn_kwargs(leader, self._postgresql.config.rewind_credentials)
@@ -514,7 +465,6 @@ class Rewind(object):
logger.exception('Unable to list %s', status_dir) logger.exception('Unable to list %s', status_dir)
def ensure_clean_shutdown(self): def ensure_clean_shutdown(self):
self._archive_ready_wals()
self.cleanup_archive_status() self.cleanup_archive_status()
# Start in a single user mode and stop to produce a clean shutdown # Start in a single user mode and stop to produce a clean shutdown
+23 -108
View File
@@ -5,7 +5,6 @@ import shutil
from collections import defaultdict from collections import defaultdict
from contextlib import contextmanager from contextlib import contextmanager
from threading import Condition, Thread
from .connection import get_connection_cursor from .connection import get_connection_cursor
from .misc import format_lsn from .misc import format_lsn
@@ -32,94 +31,10 @@ def fsync_dir(path):
os.close(fd) os.close(fd)
class SlotsAdvanceThread(Thread):
def __init__(self, slots_handler):
super(SlotsAdvanceThread, self).__init__()
self.daemon = True
self._slots_handler = slots_handler
# _copy_slots and _failed are used to asynchronously give some feedback to the main thread
self._copy_slots = []
self._failed = False
self._scheduled = defaultdict(dict) # {'dbname1': {'slot1': 100, 'slot2': 100}, 'dbname2': {'slot3': 100}}
self._condition = Condition() # protect self._scheduled from concurrent access and to wakeup the run() method
self.start()
def sync_slot(self, cur, database, slot, lsn):
failed = copy = False
try:
cur.execute("SELECT pg_catalog.pg_replication_slot_advance(%s, %s)", (slot, format_lsn(lsn)))
except Exception as e:
logger.error("Failed to advance logical replication slot '%s': %r", slot, e)
failed = True
copy = isinstance(e, OperationalError) and e.diag.sqlstate == '58P01' # WAL file is gone
with self._condition:
if self._scheduled and failed:
if copy and slot not in self._copy_slots:
self._copy_slots.append(slot)
self._failed = True
new_lsn = self._scheduled.get(database, {}).get(slot, 0)
# remove slot from the self._scheduled structure only if it wasn't changed
if new_lsn == lsn and database in self._scheduled:
self._scheduled[database].pop(slot)
if not self._scheduled[database]:
self._scheduled.pop(database)
def sync_slots_in_database(self, database, slots):
with self._slots_handler.get_local_connection_cursor(dbname=database, options='-c statement_timeout=0') as cur:
for slot in slots:
with self._condition:
lsn = self._scheduled.get(database, {}).get(slot, 0)
if lsn:
self.sync_slot(cur, database, slot, lsn)
def sync_slots(self):
with self._condition:
databases = list(self._scheduled.keys())
for database in databases:
with self._condition:
slots = list(self._scheduled.get(database, {}).keys())
if slots:
try:
self.sync_slots_in_database(database, slots)
except Exception as e:
logger.error('Failed to advance replication slots in database %s: %r', database, e)
def run(self):
while True:
with self._condition:
if not self._scheduled:
self._condition.wait()
self.sync_slots()
def schedule(self, advance_slots):
with self._condition:
for database, values in advance_slots.items():
self._scheduled[database].update(values)
ret = (self._failed, self._copy_slots)
self._copy_slots = []
self._failed = False
self._condition.notify()
return ret
def on_promote(self):
with self._condition:
self._scheduled.clear()
self._failed = False
self._copy_slots = []
class SlotsHandler(object): class SlotsHandler(object):
def __init__(self, postgresql): def __init__(self, postgresql):
self._postgresql = postgresql self._postgresql = postgresql
self._advance = None
self._replication_slots = {} # already existing replication slots self._replication_slots = {} # already existing replication slots
self._unready_logical_slots = {} self._unready_logical_slots = {}
self.schedule() self.schedule()
@@ -197,10 +112,10 @@ class SlotsHandler(object):
# In normal situation rowcount should be 1, otherwise either slot doesn't exists or it is still active # In normal situation rowcount should be 1, otherwise either slot doesn't exists or it is still active
return cursor.rowcount == 1 return cursor.rowcount == 1
def _drop_incorrect_slots(self, cluster, slots, paused): def _drop_incorrect_slots(self, cluster, slots):
# drop old replication slots which are not presented in desired slots # drop old replication slots which are not presented in desired slots
for name in set(self._replication_slots) - set(slots): for name in set(self._replication_slots) - set(slots):
if not paused and not self.ignore_replication_slot(cluster, name) and not self.drop_replication_slot(name): if not self.ignore_replication_slot(cluster, name) and not self.drop_replication_slot(name):
logger.error("Failed to drop replication slot '%s'", name) logger.error("Failed to drop replication slot '%s'", name)
self._schedule_load_slots = True self._schedule_load_slots = True
@@ -228,7 +143,7 @@ class SlotsHandler(object):
self._schedule_load_slots = True self._schedule_load_slots = True
@contextmanager @contextmanager
def get_local_connection_cursor(self, **kwargs): def _get_local_connection_cursor(self, **kwargs):
conn_kwargs = self._postgresql.config.local_connect_kwargs conn_kwargs = self._postgresql.config.local_connect_kwargs
conn_kwargs.update(kwargs) conn_kwargs.update(kwargs)
with get_connection_cursor(**conn_kwargs) as cur: with get_connection_cursor(**conn_kwargs) as cur:
@@ -247,7 +162,7 @@ class SlotsHandler(object):
# Create new logical slots # Create new logical slots
for database, values in logical_slots.items(): for database, values in logical_slots.items():
with self.get_local_connection_cursor(dbname=database) as cur: with self._get_local_connection_cursor(dbname=database) as cur:
for name, value in values.items(): for name, value in values.items():
try: try:
cur.execute("SELECT pg_catalog.pg_create_logical_replication_slot(%s, %s)" + cur.execute("SELECT pg_catalog.pg_create_logical_replication_slot(%s, %s)" +
@@ -260,11 +175,6 @@ class SlotsHandler(object):
slots.pop(name) slots.pop(name)
self._schedule_load_slots = True self._schedule_load_slots = True
def schedule_advance_slots(self, slots):
if not self._advance:
self._advance = SlotsAdvanceThread(self)
return self._advance.schedule(slots)
def _ensure_logical_slots_replica(self, cluster, slots): def _ensure_logical_slots_replica(self, cluster, slots):
advance_slots = defaultdict(dict) # Group logical slots to be advanced by database name advance_slots = defaultdict(dict) # Group logical slots to be advanced by database name
create_slots = [] # And collect logical slots to be created on the replica create_slots = [] # And collect logical slots to be created on the replica
@@ -276,18 +186,27 @@ class SlotsHandler(object):
if name in cluster.slots: if name in cluster.slots:
try: # Skip slots that doesn't need to be advanced try: # Skip slots that doesn't need to be advanced
if value['confirmed_flush_lsn'] < int(cluster.slots[name]): if value['confirmed_flush_lsn'] < int(cluster.slots[name]):
advance_slots[value['database']][name] = int(cluster.slots[name]) advance_slots[value['database']][name] = value
except Exception as e: except Exception as e:
logger.error('Failed to parse "%s": %r', cluster.slots[name], e) logger.error('Failed to parse "%s": %r', cluster.slots[name], e)
elif name in cluster.slots: # We want to copy only slots with feedback in a DCS elif name in cluster.slots: # We want to copy only slots with feedback in a DCS
create_slots.append(name) create_slots.append(name)
error, copy_slots = self.schedule_advance_slots(advance_slots) # Advance logical slots
if error: for database, values in advance_slots.items():
self._schedule_load_slots = True with self._get_local_connection_cursor(dbname=database, options='-c statement_timeout=0') as cur:
return create_slots + copy_slots for name, value in values.items():
try:
cur.execute("SELECT pg_catalog.pg_replication_slot_advance(%s, %s)",
(name, format_lsn(int(cluster.slots[name]))))
except Exception as e:
logger.error("Failed to advance logical replication slot '%s': %r", name, e)
if isinstance(e, OperationalError) and e.diag.sqlstate == '58P01': # WAL file is gone
create_slots.append(name)
self._schedule_load_slots = True
return create_slots
def sync_replication_slots(self, cluster, nofailover, replicatefrom=None, paused=False): def sync_replication_slots(self, cluster, nofailover, replicatefrom=None):
ret = None ret = None
if self._postgresql.major_version >= 90400 and cluster.config: if self._postgresql.major_version >= 90400 and cluster.config:
try: try:
@@ -296,7 +215,7 @@ class SlotsHandler(object):
slots = cluster.get_replication_slots(self._postgresql.name, self._postgresql.role, slots = cluster.get_replication_slots(self._postgresql.name, self._postgresql.role,
nofailover, self._postgresql.major_version, True) nofailover, self._postgresql.major_version, True)
self._drop_incorrect_slots(cluster, slots, paused) self._drop_incorrect_slots(cluster, slots)
self._ensure_physical_slots(slots) self._ensure_physical_slots(slots)
@@ -345,11 +264,10 @@ class SlotsHandler(object):
try: try:
cur = self._query("SELECT pg_catalog.current_setting('hot_standby_feedback')::boolean") cur = self._query("SELECT pg_catalog.current_setting('hot_standby_feedback')::boolean")
if not cur.fetchone()[0]: if not cur.fetchone()[0]:
logger.error('Logical slot failover requires "hot_standby_feedback".' return logger.error('Logical slot failover requires "hot_standby_feedback".'
' Please check postgresql.auto.conf') ' Please check postgresql.auto.conf')
except Exception as e: except Exception as e:
logger.error('Failed to check the hot_standby_feedback setting: %r', e) return logger.error('Failed to check the hot_standby_feedback setting: %r', e)
return # since `catalog_xmin` isn't valid further checks don't make any sense
for name in list(self._unready_logical_slots): for name in list(self._unready_logical_slots):
value = self._replication_slots.get(name) value = self._replication_slots.get(name)
@@ -413,9 +331,6 @@ class SlotsHandler(object):
self._schedule_load_slots = self._force_readiness_check = value self._schedule_load_slots = self._force_readiness_check = value
def on_promote(self): def on_promote(self):
if self._advance:
self._advance.on_promote()
if self._unready_logical_slots: if self._unready_logical_slots:
logger.warning('Logical replication slots that might be unsafe to use after promote: %s', logger.warning('Logical replication slots that might be unsafe to use after promote: %s',
set(self._unready_logical_slots)) set(self._unready_logical_slots))
+1
View File
@@ -200,6 +200,7 @@ parameters = CaseInsensitiveDict({
'enable_async_append': Bool(140000, None), 'enable_async_append': Bool(140000, None),
'enable_bitmapscan': Bool(90300, None), 'enable_bitmapscan': Bool(90300, None),
'enable_gathermerge': Bool(100000, None), 'enable_gathermerge': Bool(100000, None),
'enable_group_by_reordering': Bool(150000, None),
'enable_hashagg': Bool(90300, None), 'enable_hashagg': Bool(90300, None),
'enable_hashjoin': Bool(90300, None), 'enable_hashjoin': Bool(90300, None),
'enable_incremental_sort': Bool(130000, None), 'enable_incremental_sort': Bool(130000, None),
+1 -8
View File
@@ -37,10 +37,6 @@ def validate_host_port(host_port, listen=False, multiple_hosts=False):
hosts = hosts.split(",") hosts = hosts.split(",")
else: else:
hosts = [hosts] hosts = [hosts]
if "*" in hosts:
if len(hosts) != 1:
raise ConfigParseError("expecting '*' alone")
hosts = [p[-1][0] for p in socket.getaddrinfo(None, port, 0, socket.SOCK_STREAM, 0, socket.AI_PASSIVE)]
for host in hosts: for host in hosts:
proto = socket.getaddrinfo(host, "", 0, socket.SOCK_STREAM, 0, socket.AI_PASSIVE) proto = socket.getaddrinfo(host, "", 0, socket.SOCK_STREAM, 0, socket.AI_PASSIVE)
s = socket.socket(proto[0][0], socket.SOCK_STREAM) s = socket.socket(proto[0][0], socket.SOCK_STREAM)
@@ -182,11 +178,9 @@ class Schema(object):
self.validator = validator self.validator = validator
def __call__(self, data): def __call__(self, data):
errors = []
for i in self.validate(data): for i in self.validate(data):
if not i.status: if not i.status:
errors.append(str(i)) print(i)
return errors
def validate(self, data): def validate(self, data):
self.data = data self.data = data
@@ -370,7 +364,6 @@ schema = Schema({
"postgresql": { "postgresql": {
"listen": validate_host_port_listen_multiple_hosts, "listen": validate_host_port_listen_multiple_hosts,
"connect_address": validate_connect_address, "connect_address": validate_connect_address,
"proxy_address": validate_connect_address,
"authentication": { "authentication": {
"replication": userattributes, "replication": userattributes,
"superuser": userattributes, "superuser": userattributes,
+1 -1
View File
@@ -1 +1 @@
__version__ = '2.1.5' __version__ = '2.1.4'
-2
View File
@@ -99,8 +99,6 @@ bootstrap:
postgresql: postgresql:
listen: 127.0.0.1:5432 listen: 127.0.0.1:5432
connect_address: 127.0.0.1:5432 connect_address: 127.0.0.1:5432
# proxy_address: 127.0.0.1:5433 # The address of connection pool (e.g., pgbouncer) running next to Patroni/Postgres. Only for service discovery.
data_dir: data/postgresql0 data_dir: data/postgresql0
# bin_dir: # bin_dir:
# config_dir: # config_dir:
-1
View File
@@ -93,7 +93,6 @@ bootstrap:
postgresql: postgresql:
listen: 127.0.0.1:5433 listen: 127.0.0.1:5433
connect_address: 127.0.0.1:5433 connect_address: 127.0.0.1:5433
# proxy_address: 127.0.0.1:5434 # The address of connection pool (e.g., pgbouncer) running next to Patroni/Postgres. Only for service discovery.
data_dir: data/postgresql1 data_dir: data/postgresql1
# bin_dir: # bin_dir:
# config_dir: # config_dir:
-1
View File
@@ -90,7 +90,6 @@ bootstrap:
postgresql: postgresql:
listen: 127.0.0.1:5434 listen: 127.0.0.1:5434
connect_address: 127.0.0.1:5434 connect_address: 127.0.0.1:5434
# proxy_address: 127.0.0.1:5435 # The address of connection pool (e.g., pgbouncer) running next to Patroni/Postgres. Only for service discovery.
data_dir: data/postgresql2 data_dir: data/postgresql2
# bin_dir: # bin_dir:
# config_dir: # config_dir:
+21 -18
View File
@@ -1,28 +1,31 @@
#!/bin/bash #!/bin/sh
# Release process: if [ $# -ne 1 ]; then
# 1. Open a PR that updates release notes and Patroni version >&2 echo "usage: $0 <version>"
# 2. Merge it exit 1
# 3. Run release.sh fi
# 4. After the new tag is pushed, the .github/workflows/release.yaml will run tests and upload the new package to test.pypi.org
# 5. Once the release is created, the .github/workflows/release.yaml will run tests and upload the new package to pypi.org readonly VERSIONFILE="patroni/version.py"
## Bail out on any non-zero exitcode from the called processes ## Bail out on any non-zero exitcode from the called processes
set -xe set -xe
if python3 --version &> /dev/null; then python3 --version
alias python=python3
shopt -s expand_aliases
fi
python --version
git --version git --version
version=$(python -c 'from patroni.version import __version__; print(__version__)') version=$1
python setup.py clean sed -i "s/__version__ = .*/__version__ = '${version}'/" "${VERSIONFILE}"
python setup.py test python3 setup.py clean
python setup.py flake8 python3 setup.py test
python3 setup.py flake8
git tag "v$version" git add "${VERSIONFILE}"
git commit -m "Bumped version to $version"
git push
python3 setup.py sdist bdist_wheel upload
git tag v${version}
git push --tags git push --tags
+1 -1
View File
@@ -1,7 +1,7 @@
psycopg2-binary psycopg2-binary
behave behave
coverage coverage
flake8>=3.0.0 flake8
mock mock
pytest-cov pytest-cov
pytest pytest
+6 -4
View File
@@ -18,8 +18,8 @@ MAIN_PACKAGE = NAME
DESCRIPTION = 'PostgreSQL High-Available orchestrator and CLI' DESCRIPTION = 'PostgreSQL High-Available orchestrator and CLI'
LICENSE = 'The MIT License' LICENSE = 'The MIT License'
URL = 'https://github.com/zalando/patroni' URL = 'https://github.com/zalando/patroni'
AUTHOR = 'Alexander Kukushkin, Polina Bungina' AUTHOR = 'Alexander Kukushkin, Dmitrii Dolgov, Oleksii Kliukin'
AUTHOR_EMAIL = 'akukushkin@microsoft.com, [email protected]' AUTHOR_EMAIL = 'alexander.kukushkin@zalando.de, [email protected], [email protected]'
KEYWORDS = 'etcd governor patroni postgresql postgres ha haproxy confd' +\ KEYWORDS = 'etcd governor patroni postgresql postgres ha haproxy confd' +\
' zookeeper exhibitor consul streaming replication kubernetes k8s' ' zookeeper exhibitor consul streaming replication kubernetes k8s'
@@ -93,10 +93,12 @@ class Flake8(_Command):
return [package for package in self.package_files()] + ['tests', 'setup.py'] return [package for package in self.package_files()] + ['tests', 'setup.py']
def run(self): def run(self):
from flake8.main.cli import main from flake8.main import application
logging.getLogger().setLevel(logging.ERROR) logging.getLogger().setLevel(logging.ERROR)
main(self.targets()) flake8 = application.Application()
flake8.run(self.targets())
flake8.exit()
class PyTest(_Command): class PyTest(_Command):
+1 -2
View File
@@ -188,8 +188,7 @@ class PostgresInit(unittest.TestCase):
self.p = Postgresql({'name': 'postgresql0', 'scope': 'batman', 'data_dir': data_dir, self.p = Postgresql({'name': 'postgresql0', 'scope': 'batman', 'data_dir': data_dir,
'config_dir': data_dir, 'retry_timeout': 10, 'config_dir': data_dir, 'retry_timeout': 10,
'krbsrvname': 'postgres', 'pgpass': os.path.join(data_dir, 'pgpass0'), 'krbsrvname': 'postgres', 'pgpass': os.path.join(data_dir, 'pgpass0'),
'listen': '127.0.0.2, 127.0.0.3:5432', 'listen': '127.0.0.2, 127.0.0.3:5432', 'connect_address': '127.0.0.2:5432',
'connect_address': '127.0.0.2:5432', 'proxy_address': '127.0.0.2:5433',
'authentication': {'superuser': {'username': 'foo', 'password': 'test'}, 'authentication': {'superuser': {'username': 'foo', 'password': 'test'},
'replication': {'username': '', 'password': 'rep-pass'}, 'replication': {'username': '', 'password': 'rep-pass'},
'rewind': {'username': 'rewind', 'password': 'test'}}, 'rewind': {'username': 'rewind', 'password': 'test'}},
+25 -25
View File
@@ -12,7 +12,6 @@ from patroni.ha import _MemberStatus
from patroni.utils import tzutc from patroni.utils import tzutc
from six import BytesIO as IO from six import BytesIO as IO
from six.moves import BaseHTTPServer from six.moves import BaseHTTPServer
from six.moves.socketserver import ThreadingMixIn
from . import psycopg_connect, MockCursor from . import psycopg_connect, MockCursor
from .test_ha import get_cluster_initialized_without_leader from .test_ha import get_cluster_initialized_without_leader
@@ -47,10 +46,6 @@ class MockPostgresql(object):
def replica_cached_timeline(_): def replica_cached_timeline(_):
return 2 return 2
@staticmethod
def is_running():
return True
class MockWatchdog(object): class MockWatchdog(object):
is_healthy = False is_healthy = False
@@ -134,10 +129,6 @@ class MockPatroni(object):
def sighup_handler(): def sighup_handler():
pass pass
@staticmethod
def api_sigterm():
pass
class MockRequest(object): class MockRequest(object):
@@ -178,6 +169,7 @@ class TestRestApiHandler(unittest.TestCase):
MockRestApiServer(RestApiHandler, 'GET /replica?lag=10MB') MockRestApiServer(RestApiHandler, 'GET /replica?lag=10MB')
MockRestApiServer(RestApiHandler, 'GET /replica?lag=10485760') MockRestApiServer(RestApiHandler, 'GET /replica?lag=10485760')
MockRestApiServer(RestApiHandler, 'GET /read-only') MockRestApiServer(RestApiHandler, 'GET /read-only')
MockRestApiServer(RestApiHandler, 'GET /read-only-sync')
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={})): with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={})):
MockRestApiServer(RestApiHandler, 'GET /replica') MockRestApiServer(RestApiHandler, 'GET /replica')
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'master'})): with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'master'})):
@@ -189,13 +181,13 @@ class TestRestApiHandler(unittest.TestCase):
MockPatroni.dcs.cluster.is_synchronous_mode = Mock(return_value=True) MockPatroni.dcs.cluster.is_synchronous_mode = Mock(return_value=True)
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})): with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
MockRestApiServer(RestApiHandler, 'GET /synchronous') MockRestApiServer(RestApiHandler, 'GET /synchronous')
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
MockRestApiServer(RestApiHandler, 'GET /read-only-sync') MockRestApiServer(RestApiHandler, 'GET /read-only-sync')
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})): with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
MockPatroni.dcs.cluster.sync.members = [] MockPatroni.dcs.cluster.sync.members = []
MockRestApiServer(RestApiHandler, 'GET /asynchronous') MockRestApiServer(RestApiHandler, 'GET /asynchronous')
with patch.object(MockHa, 'is_leader', Mock(return_value=True)): with patch.object(MockHa, 'is_leader', Mock(return_value=True)):
MockRestApiServer(RestApiHandler, 'GET /replica') MockRestApiServer(RestApiHandler, 'GET /replica')
MockRestApiServer(RestApiHandler, 'GET /read-only-sync')
with patch.object(MockHa, 'is_standby_cluster', Mock(return_value=True)): with patch.object(MockHa, 'is_standby_cluster', Mock(return_value=True)):
MockRestApiServer(RestApiHandler, 'GET /standby_leader') MockRestApiServer(RestApiHandler, 'GET /standby_leader')
MockPatroni.dcs.cluster = None MockPatroni.dcs.cluster = None
@@ -295,12 +287,7 @@ class TestRestApiHandler(unittest.TestCase):
def test_do_OPTIONS(self): def test_do_OPTIONS(self):
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'OPTIONS / HTTP/1.0')) self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'OPTIONS / HTTP/1.0'))
def test_do_HEAD(self): def test_do_GET_liveness(self):
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'HEAD / HTTP/1.0'))
@patch.object(MockPatroni, 'dcs')
def test_do_GET_liveness(self, mock_dcs):
mock_dcs.ttl.return_value = PropertyMock(30)
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /liveness HTTP/1.0')) self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /liveness HTTP/1.0'))
def test_do_GET_readiness(self): def test_do_GET_readiness(self):
@@ -375,11 +362,6 @@ class TestRestApiHandler(unittest.TestCase):
def test_do_POST_reload(self): def test_do_POST_reload(self):
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'POST /reload HTTP/1.0' + self._authorization)) self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'POST /reload HTTP/1.0' + self._authorization))
@patch('os.environ', {'BEHAVE_DEBUG': 'true'})
@patch('os.name', 'nt')
def test_do_POST_sigterm(self):
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'POST /sigterm HTTP/1.0' + self._authorization))
@patch.object(MockPatroni, 'dcs') @patch.object(MockPatroni, 'dcs')
def test_do_POST_restart(self, mock_dcs): def test_do_POST_restart(self, mock_dcs):
mock_dcs.get_cluster.return_value.is_paused.return_value = False mock_dcs.get_cluster.return_value.is_paused.return_value = False
@@ -596,14 +578,32 @@ class TestRestApiServer(unittest.TestCase):
def test_socket_error(self): def test_socket_error(self):
self.assertRaises(socket.error, MockRestApiServer, Mock(), '', {'listen': '*:8008'}) self.assertRaises(socket.error, MockRestApiServer, Mock(), '', {'listen': '*:8008'})
@patch.object(ThreadingMixIn, 'process_request_thread', Mock()) @patch.object(MockRestApiServer, 'finish_request', Mock())
def test_process_request_thread(self): def test_process_request_thread(self):
self.srv.process_request_thread(Mock(), '2') mock_socket = Mock()
self.srv.process_request_thread((mock_socket, 1), '2')
mock_socket.context.wrap_socket.side_effect = socket.error
self.srv.process_request_thread((mock_socket, 1), '2')
@patch.object(socket.socket, 'accept')
def test_get_request(self, mock_accept):
newsock = Mock()
mock_accept.return_value = (newsock, '2')
self.srv.socket = Mock()
self.assertEqual(self.srv.get_request(), ((self.srv.socket, newsock), '2'))
@patch.object(MockRestApiServer, 'process_request', Mock(side_effect=RuntimeError)) @patch.object(MockRestApiServer, 'process_request', Mock(side_effect=RuntimeError))
@patch.object(MockRestApiServer, 'get_request', Mock(return_value=(Mock(), ('127.0.0.1', 55555))))
def test_process_request_error(self): def test_process_request_error(self):
self.srv._handle_request_noblock() mock_address = ('127.0.0.1', 55555)
mock_socket = Mock()
mock_ssl_socket = (Mock(), Mock())
for mock_request in (mock_socket, mock_ssl_socket):
with patch.object(
MockRestApiServer,
'get_request',
Mock(return_value=(mock_request, mock_address))
):
self.srv._handle_request_noblock()
@patch('ssl._ssl._test_decode_cert', Mock()) @patch('ssl._ssl._test_decode_cert', Mock())
def test_reload_local_certificate(self): def test_reload_local_certificate(self):
-1
View File
@@ -40,7 +40,6 @@ class TestConfig(unittest.TestCase):
'PATRONI_RESTAPI_ALLOWLIST_INCLUDE_MEMBERS': 'on', 'PATRONI_RESTAPI_ALLOWLIST_INCLUDE_MEMBERS': 'on',
'PATRONI_POSTGRESQL_LISTEN': '0.0.0.0:5432', 'PATRONI_POSTGRESQL_LISTEN': '0.0.0.0:5432',
'PATRONI_POSTGRESQL_CONNECT_ADDRESS': '127.0.0.1:5432', 'PATRONI_POSTGRESQL_CONNECT_ADDRESS': '127.0.0.1:5432',
'PATRONI_POSTGRESQL_PROXY_ADDRESS': '127.0.0.1:5433',
'PATRONI_POSTGRESQL_DATA_DIR': 'data/postgres0', 'PATRONI_POSTGRESQL_DATA_DIR': 'data/postgres0',
'PATRONI_POSTGRESQL_CONFIG_DIR': 'data/postgres0', 'PATRONI_POSTGRESQL_CONFIG_DIR': 'data/postgres0',
'PATRONI_POSTGRESQL_PGPASS': '/tmp/pgpass0', 'PATRONI_POSTGRESQL_PGPASS': '/tmp/pgpass0',
-2
View File
@@ -161,7 +161,6 @@ class TestConsul(unittest.TestCase):
self.c.write_leader_optime('1') self.c.write_leader_optime('1')
@patch.object(consul.Consul.Session, 'renew', Mock()) @patch.object(consul.Consul.Session, 'renew', Mock())
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=ConsulException))
def test_update_leader(self): def test_update_leader(self):
self.c.update_leader(12345) self.c.update_leader(12345)
@@ -218,7 +217,6 @@ class TestConsul(unittest.TestCase):
d['role'] = 'bla' d['role'] = 'bla'
self.assertIsNone(self.c.update_service({}, d)) self.assertIsNone(self.c.update_service({}, d))
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=ConsulException))
def test_reload_config(self): def test_reload_config(self):
self.assertEqual([], self.c._service_tags) self.assertEqual([], self.c._service_tags)
self.c.reload_config({'consul': {'token': 'foo', 'register_service': True, 'service_tags': ['foo']}, self.c.reload_config({'consul': {'token': 'foo', 'register_service': True, 'service_tags': ['foo']},
+4 -28
View File
@@ -7,11 +7,10 @@ from datetime import datetime, timedelta
from mock import patch, Mock from mock import patch, Mock
from patroni.ctl import ctl, store_config, load_config, output_members, get_dcs, parse_dcs, \ from patroni.ctl import ctl, store_config, load_config, output_members, get_dcs, parse_dcs, \
get_all_members, get_any_member, get_cursor, query_member, configure, PatroniCtlException, apply_config_changes, \ get_all_members, get_any_member, get_cursor, query_member, configure, PatroniCtlException, apply_config_changes, \
format_config_for_editing, show_diff, invoke_editor, format_pg_version, CONFIG_FILE_PATH, PatronictlPrettyTable format_config_for_editing, show_diff, invoke_editor, format_pg_version, CONFIG_FILE_PATH
from patroni.dcs.etcd import AbstractEtcdClientWithFailover, Failover from patroni.dcs.etcd import AbstractEtcdClientWithFailover, Failover
from patroni.psycopg import OperationalError from patroni.psycopg import OperationalError
from patroni.utils import tzutc from patroni.utils import tzutc
from prettytable import PrettyTable, ALL
from urllib3 import PoolManager from urllib3 import PoolManager
from . import MockConnect, MockCursor, MockResponse, psycopg_connect from . import MockConnect, MockCursor, MockResponse, psycopg_connect
@@ -563,8 +562,7 @@ class TestCtl(unittest.TestCase):
@patch('sys.stdout.isatty', return_value=False) @patch('sys.stdout.isatty', return_value=False)
@patch('patroni.ctl.markup_to_pager') @patch('patroni.ctl.markup_to_pager')
@patch('patroni.ctl.find_executable', return_value=None) def test_show_diff(self, mock_markup_to_pager, mock_isatty):
def test_show_diff(self, mock_find_executable, mock_markup_to_pager, mock_isatty):
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n") show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
mock_markup_to_pager.assert_not_called() mock_markup_to_pager.assert_not_called()
@@ -572,10 +570,10 @@ class TestCtl(unittest.TestCase):
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n") show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
mock_markup_to_pager.assert_called_once() mock_markup_to_pager.assert_called_once()
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n") with patch('patroni.ctl.find_executable', Mock(return_value=None)):
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
# Test that unicode handling doesn't fail with an exception # Test that unicode handling doesn't fail with an exception
mock_find_executable.return_value = '/usr/bin/less'
show_diff(b"foo:\n bar: \xc3\xb6\xc3\xb6\n".decode('utf-8'), show_diff(b"foo:\n bar: \xc3\xb6\xc3\xb6\n".decode('utf-8'),
b"foo:\n bar: \xc3\xbc\xc3\xbc\n".decode('utf-8')) b"foo:\n bar: \xc3\xbc\xc3\xbc\n".decode('utf-8'))
@@ -593,7 +591,6 @@ class TestCtl(unittest.TestCase):
self.runner.invoke(ctl, ['show-config', 'dummy']) self.runner.invoke(ctl, ['show-config', 'dummy'])
@patch('patroni.ctl.get_dcs') @patch('patroni.ctl.get_dcs')
@patch('subprocess.call', Mock(return_value=0))
def test_edit_config(self, mock_get_dcs): def test_edit_config(self, mock_get_dcs):
mock_get_dcs.return_value = self.e mock_get_dcs.return_value = self.e
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
@@ -649,24 +646,3 @@ class TestCtl(unittest.TestCase):
result = self.runner.invoke(ctl, ['reinit', 'alpha', 'other', '--wait'], input='y\ny') result = self.runner.invoke(ctl, ['reinit', 'alpha', 'other', '--wait'], input='y\ny')
self.assertIn("Waiting for reinitialize to complete on: other", result.output) self.assertIn("Waiting for reinitialize to complete on: other", result.output)
self.assertIn("Reinitialize is completed on: other", result.output) self.assertIn("Reinitialize is completed on: other", result.output)
class TestPatronictlPrettyTable(unittest.TestCase):
def setUp(self):
self.pt = PatronictlPrettyTable(' header', ['foo', 'bar'], hrules=ALL)
def test__get_hline(self):
expected = '+-----+-----+'
self.pt._hrule = expected
self.assertEqual(self.pt._hrule, '+ header----+')
self.assertFalse(self.pt._is_first_hline())
self.assertEqual(self.pt._hrule, expected)
@patch.object(PrettyTable, '_stringify_hrule', Mock(return_value='+-----+-----+'))
def test__stringify_hrule(self):
self.assertEqual(self.pt._stringify_hrule((), 'top_'), '+ header----+')
self.assertFalse(self.pt._is_first_hline())
def test_output(self):
self.assertEqual(str(self.pt), '+ header----+\n| foo | bar |\n+-----+-----+')
+7 -67
View File
@@ -45,10 +45,6 @@ def get_cluster_not_initialized_without_leader(cluster_config=None):
return get_cluster(None, None, [], None, SyncState(None, None, None), cluster_config) return get_cluster(None, None, [], None, SyncState(None, None, None), cluster_config)
def get_cluster_bootstrapping_without_leader(cluster_config=None):
return get_cluster("", None, [], None, SyncState(None, None, None), cluster_config)
def get_cluster_initialized_without_leader(leader=False, failover=None, sync=None, cluster_config=None): def get_cluster_initialized_without_leader(leader=False, failover=None, sync=None, cluster_config=None):
m1 = Member(0, 'leader', 28, {'conn_url': 'postgres://replicator:[email protected]:5435/postgres', m1 = Member(0, 'leader', 28, {'conn_url': 'postgres://replicator:[email protected]:5435/postgres',
'api_url': 'http://127.0.0.1:8008/patroni', 'xlog_location': 4}) 'api_url': 'http://127.0.0.1:8008/patroni', 'xlog_location': 4})
@@ -409,7 +405,6 @@ class TestHa(PostgresInit):
self.ha.has_lock = true self.ha.has_lock = true
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock') self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
@patch.object(Postgresql, '_wait_for_connection_close', Mock())
def test_demote_because_not_having_lock(self): def test_demote_because_not_having_lock(self):
self.ha.cluster.is_unlocked = false self.ha.cluster.is_unlocked = false
with patch.object(Watchdog, 'is_running', PropertyMock(return_value=True)): with patch.object(Watchdog, 'is_running', PropertyMock(return_value=True)):
@@ -474,11 +469,6 @@ class TestHa(PostgresInit):
self.p.can_create_replica_without_replication_connection = MagicMock(return_value=True) self.p.can_create_replica_without_replication_connection = MagicMock(return_value=True)
self.assertEqual(self.ha.bootstrap(), 'trying to bootstrap (without leader)') self.assertEqual(self.ha.bootstrap(), 'trying to bootstrap (without leader)')
def test_bootstrap_not_running_concurrently(self):
self.ha.cluster = get_cluster_bootstrapping_without_leader()
self.p.can_create_replica_without_replication_connection = MagicMock(return_value=True)
self.assertEqual(self.ha.bootstrap(), 'waiting for leader to bootstrap')
def test_bootstrap_initialize_lock_failed(self): def test_bootstrap_initialize_lock_failed(self):
self.ha.cluster = get_cluster_not_initialized_without_leader() self.ha.cluster = get_cluster_not_initialized_without_leader()
self.assertEqual(self.ha.bootstrap(), 'failed to acquire initialize lock') self.assertEqual(self.ha.bootstrap(), 'failed to acquire initialize lock')
@@ -653,60 +643,12 @@ class TestHa(PostgresInit):
# same as previous, but set the current member to nofailover. In no case it should be elected as a leader # same as previous, but set the current member to nofailover. In no case it should be elected as a leader
self.ha.patroni.nofailover = True self.ha.patroni.nofailover = True
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote') self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
# in sync mode only the sync node is allowed to take over
def test_manual_failover_process_no_leader_in_synchronous_mode(self):
self.ha.is_synchronous_mode = true
self.p.is_leader = false
# switchover to a specific node, which name doesn't match our name (postgresql0)
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'other', None)) self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'other', None))
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
# switchover to our node (postgresql0), which name is not in sync nodes list
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'postgresql0', None),
sync=('leader1', 'blabla'))
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
# switchover from a specific leader, but our name (postgresql0) is not in the sync nodes list
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', '', None),
sync=('leader', 'blabla'))
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
# switchover from a specific leader, but the only sync node (us, postgresql0) has nofailover tag
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', '', None),
sync=('postgresql0'))
self.ha.patroni.nofailover = True
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
self.ha.patroni.nofailover = False self.ha.patroni.nofailover = False
self.ha.is_synchronous_mode = true
# manual failover when our name (postgresql0) isn't in the /sync key and the `other` node is not available
self.ha.fetch_node_status = get_node_status(nofailover=True) # accessible, in_recovery
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
sync=('leader1', 'blabla'))
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node') self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
# manual failover when the `other` node isn't available but our name is in the /sync key
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
sync=('leader1', 'postgresql0'))
self.p.pick_synchronous_standby = Mock(return_value=([], []))
self.ha.dcs.write_sync_state = true
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
# manual failover to our node (postgresql0),
# which name is not in sync nodes list (the leader and all sync nodes are not available)
self.p.set_role('replica')
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None),
sync=('leader1', 'other'))
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
# manual failover to our node (postgresql0),
# which name is not in sync nodes list (some sync nodes are available)
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None),
sync=('leader1', 'other'))
self.p.set_role('replica')
self.p.pick_synchronous_standby = Mock(return_value=(['leader1'], ['leader1']))
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
def test_manual_failover_process_no_leader_in_pause(self): def test_manual_failover_process_no_leader_in_pause(self):
self.ha.is_paused = true self.ha.is_paused = true
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None)) self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None))
@@ -1191,19 +1133,17 @@ class TestHa(PostgresInit):
self.ha.shutdown() self.ha.shutdown()
@patch('time.sleep', Mock()) @patch('time.sleep', Mock())
def test_leader_with_not_accessible_data_directory(self): def test_leader_with_empty_directory(self):
self.ha.cluster = get_cluster_initialized_with_leader() self.ha.cluster = get_cluster_initialized_with_leader()
self.ha.has_lock = true self.ha.has_lock = true
self.p.data_directory_empty = Mock(side_effect=OSError(5, "Input/output error: '{}'".format(self.p.data_dir))) self.p.data_directory_empty = true
self.assertEqual(self.ha.run_cycle(), self.assertEqual(self.ha.run_cycle(), 'released leader key voluntarily as data dir empty and currently leader')
'released leader key voluntarily as data dir not accessible and currently leader')
self.assertEqual(self.p.role, 'uninitialized') self.assertEqual(self.p.role, 'uninitialized')
# as has_lock is mocked out, we need to fake the leader key release # as has_lock is mocked out, we need to fake the leader key release
self.ha.has_lock = false self.ha.has_lock = false
# will not say bootstrap because data directory is not accessible # will not say bootstrap from leader as replica can't self elect
self.assertEqual(self.ha.run_cycle(), self.assertEqual(self.ha.run_cycle(), "trying to bootstrap from replica 'other'")
"data directory is not accessible: [Errno 5] Input/output error: '{}'".format(self.p.data_dir))
@patch('patroni.postgresql.mtime', Mock(return_value=1588316884)) @patch('patroni.postgresql.mtime', Mock(return_value=1588316884))
@patch.object(builtins, 'open', mock_open(read_data=('1\t0/40159C0\tno recovery target specified\n\n' @patch.object(builtins, 'open', mock_open(read_data=('1\t0/40159C0\tno recovery target specified\n\n'
+2 -23
View File
@@ -315,32 +315,11 @@ class TestKubernetesEndpoints(BaseTestKubernetes):
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_pod', mock_namespaced_kind, create=True) @patch.object(k8s_client.CoreV1Api, 'patch_namespaced_pod', mock_namespaced_kind, create=True)
@patch.object(k8s_client.CoreV1Api, 'create_namespaced_endpoints', mock_namespaced_kind, create=True) @patch.object(k8s_client.CoreV1Api, 'create_namespaced_endpoints', mock_namespaced_kind, create=True)
@patch.object(k8s_client.CoreV1Api, 'create_namespaced_service', @patch.object(k8s_client.CoreV1Api, 'create_namespaced_service',
Mock(side_effect=[True, Mock(side_effect=[True, False, k8s_client.rest.ApiException(500, '')]), create=True)
False, def test__create_config_service(self):
k8s_client.rest.ApiException(409, ''),
k8s_client.rest.ApiException(403, ''),
k8s_client.rest.ApiException(500, ''),
Exception("Unexpected")
]), create=True)
@patch('patroni.dcs.kubernetes.logger.exception')
def test__create_config_service(self, mock_logger_exception):
self.assertIsNotNone(self.k.patch_or_create_config({'foo': 'bar'})) self.assertIsNotNone(self.k.patch_or_create_config({'foo': 'bar'}))
self.assertIsNotNone(self.k.patch_or_create_config({'foo': 'bar'})) self.assertIsNotNone(self.k.patch_or_create_config({'foo': 'bar'}))
self.k.patch_or_create_config({'foo': 'bar'})
mock_logger_exception.assert_not_called()
self.k.patch_or_create_config({'foo': 'bar'})
mock_logger_exception.assert_not_called()
self.k.patch_or_create_config({'foo': 'bar'})
mock_logger_exception.assert_called_once()
self.assertEqual(('create_config_service failed',), mock_logger_exception.call_args[0])
mock_logger_exception.reset_mock()
self.k.touch_member({'state': 'running', 'role': 'replica'}) self.k.touch_member({'state': 'running', 'role': 'replica'})
mock_logger_exception.assert_called_once()
self.assertEqual(('create_config_service failed',), mock_logger_exception.call_args[0])
class TestCacheBuilder(BaseTestKubernetes): class TestCacheBuilder(BaseTestKubernetes):
-4
View File
@@ -50,16 +50,12 @@ class MockFrozenImporter(object):
@patch.object(etcd.Client, 'read', etcd_read) @patch.object(etcd.Client, 'read', etcd_read)
class TestPatroni(unittest.TestCase): class TestPatroni(unittest.TestCase):
@patch('sys.argv', ['patroni.py'])
def test_no_config(self): def test_no_config(self):
self.assertRaises(SystemExit, patroni_main) self.assertRaises(SystemExit, patroni_main)
@patch('sys.argv', ['patroni.py', '--validate-config', 'postgres0.yml']) @patch('sys.argv', ['patroni.py', '--validate-config', 'postgres0.yml'])
@patch('socket.socket.connect_ex', Mock(return_value=1))
def test_validate_config(self): def test_validate_config(self):
self.assertRaises(SystemExit, patroni_main) self.assertRaises(SystemExit, patroni_main)
with patch.object(config.Config, '__init__', Mock(return_value=None)):
self.assertRaises(SystemExit, patroni_main)
@patch('pkgutil.iter_importers', Mock(return_value=[MockFrozenImporter()])) @patch('pkgutil.iter_importers', Mock(return_value=[MockFrozenImporter()]))
@patch('sys.frozen', Mock(return_value=True), create=True) @patch('sys.frozen', Mock(return_value=True), create=True)
+20 -16
View File
@@ -347,7 +347,7 @@ class TestPostgresql(BaseTestPostgresql):
@patch('subprocess.Popen') @patch('subprocess.Popen')
def test_latest_checkpoint_location(self, mock_popen): def test_latest_checkpoint_location(self, mock_popen):
mock_popen.return_value.communicate.return_value = (None, None) mock_popen.return_value.communicate.return_value = (None, None)
self.assertEqual(self.p.latest_checkpoint_location(), 28163096) self.assertEqual(self.p.latest_checkpoint_location(), '28163096')
# 9.3 and 9.4 format # 9.3 and 9.4 format
mock_popen.return_value.communicate.side_effect = [ mock_popen.return_value.communicate.side_effect = [
(b'rmgr: XLOG len (rec/tot): 72/ 104, tx: 0, lsn: 0/01ADBC18, prev 0/01ADBBB8, ' + (b'rmgr: XLOG len (rec/tot): 72/ 104, tx: 0, lsn: 0/01ADBC18, prev 0/01ADBBB8, ' +
@@ -355,14 +355,14 @@ class TestPostgresql(BaseTestPostgresql):
b' 1; offset 0; oldest xid 715 in DB 1; oldest multi 1 in DB 1; oldest running xid 0; shutdown', None), b' 1; offset 0; oldest xid 715 in DB 1; oldest multi 1 in DB 1; oldest running xid 0; shutdown', None),
(b'rmgr: Transaction len (rec/tot): 64/ 96, tx: 726, lsn: 0/01ADBBB8, prev 0/01ADBB70, ' + (b'rmgr: Transaction len (rec/tot): 64/ 96, tx: 726, lsn: 0/01ADBBB8, prev 0/01ADBB70, ' +
b'bkp: 0000, desc: commit: 2021-02-26 11:19:37.900918 CET; inval msgs: catcache 11 catcache 10', None)] b'bkp: 0000, desc: commit: 2021-02-26 11:19:37.900918 CET; inval msgs: catcache 11 catcache 10', None)]
self.assertEqual(self.p.latest_checkpoint_location(), 28163096) self.assertEqual(self.p.latest_checkpoint_location(), '28163096')
mock_popen.return_value.communicate.side_effect = [ mock_popen.return_value.communicate.side_effect = [
(b'rmgr: XLOG len (rec/tot): 72/ 104, tx: 0, lsn: 0/01ADBC18, prev 0/01ADBBB8, ' + (b'rmgr: XLOG len (rec/tot): 72/ 104, tx: 0, lsn: 0/01ADBC18, prev 0/01ADBBB8, ' +
b'bkp: 0000, desc: checkpoint: redo 0/1ADBC18; tli 1; prev tli 1; fpw true; xid 0/727; oid 16386; multi' + b'bkp: 0000, desc: checkpoint: redo 0/1ADBC18; tli 1; prev tli 1; fpw true; xid 0/727; oid 16386; multi' +
b' 1; offset 0; oldest xid 715 in DB 1; oldest multi 1 in DB 1; oldest running xid 0; shutdown', None), b' 1; offset 0; oldest xid 715 in DB 1; oldest multi 1 in DB 1; oldest running xid 0; shutdown', None),
(b'rmgr: XLOG len (rec/tot): 0/ 32, tx: 0, lsn: 0/01ADBBB8, prev 0/01ADBBA0, ' + (b'rmgr: XLOG len (rec/tot): 0/ 32, tx: 0, lsn: 0/01ADBBB8, prev 0/01ADBBA0, ' +
b'bkp: 0000, desc: xlog switch ', None)] b'bkp: 0000, desc: xlog switch ', None)]
self.assertEqual(self.p.latest_checkpoint_location(), 28163000) self.assertEqual(self.p.latest_checkpoint_location(), '28163000')
# 9.5+ format # 9.5+ format
mock_popen.return_value.communicate.side_effect = [ mock_popen.return_value.communicate.side_effect = [
(b'rmgr: XLOG len (rec/tot): 114/ 114, tx: 0, lsn: 0/01ADBC18, prev 0/018260F8, ' + (b'rmgr: XLOG len (rec/tot): 114/ 114, tx: 0, lsn: 0/01ADBC18, prev 0/018260F8, ' +
@@ -371,7 +371,7 @@ class TestPostgresql(BaseTestPostgresql):
b' oldest running xid 0; shutdown', None), b' oldest running xid 0; shutdown', None),
(b'rmgr: XLOG len (rec/tot): 24/ 24, tx: 0, lsn: 0/018260F8, prev 0/01826080, ' + (b'rmgr: XLOG len (rec/tot): 24/ 24, tx: 0, lsn: 0/018260F8, prev 0/01826080, ' +
b'desc: SWITCH ', None)] b'desc: SWITCH ', None)]
self.assertEqual(self.p.latest_checkpoint_location(), 25321720) self.assertEqual(self.p.latest_checkpoint_location(), '25321720')
def test_reload(self): def test_reload(self):
self.assertTrue(self.p.reload()) self.assertTrue(self.p.reload())
@@ -454,20 +454,24 @@ class TestPostgresql(BaseTestPostgresql):
def test_get_postgres_role_from_data_directory(self): def test_get_postgres_role_from_data_directory(self):
self.assertEqual(self.p.get_postgres_role_from_data_directory(), 'replica') self.assertEqual(self.p.get_postgres_role_from_data_directory(), 'replica')
@patch('os.remove', Mock())
@patch('shutil.rmtree', Mock())
@patch('os.unlink', Mock(side_effect=OSError))
@patch('os.path.isdir', Mock(return_value=True))
@patch('os.path.exists', Mock(return_value=True))
def test_remove_data_directory(self): def test_remove_data_directory(self):
with patch('os.path.islink', Mock(return_value=True)): def _symlink(src, dst):
self.p.remove_data_directory() if os.name != 'nt': # os.symlink under Windows needs admin rights skip it
with patch('os.path.isfile', Mock(return_value=True)): os.symlink(src, dst)
self.p.remove_data_directory()
with patch('os.path.islink', Mock(side_effect=[False, False, True, True])),\ os.makedirs(os.path.join(self.p.data_dir, 'foo'))
patch('os.listdir', Mock(return_value=['12345'])),\ _symlink('foo', os.path.join(self.p.data_dir, 'pg_wal'))
patch('os.path.realpath', Mock(side_effect=['../foo', '../foo_tsp'])): os.makedirs(os.path.join(self.p.data_dir, 'foo_tsp'))
pg_tblspc = os.path.join(self.p.data_dir, 'pg_tblspc')
os.makedirs(pg_tblspc)
_symlink('../foo_tsp', os.path.join(pg_tblspc, '12345'))
self.p.remove_data_directory()
open(self.p.data_dir, 'w').close()
self.p.remove_data_directory()
_symlink('unexisting', self.p.data_dir)
with patch('os.unlink', Mock(side_effect=OSError)):
self.p.remove_data_directory() self.p.remove_data_directory()
self.p.remove_data_directory()
@patch('patroni.postgresql.Postgresql._version_file_exists', Mock(return_value=True)) @patch('patroni.postgresql.Postgresql._version_file_exists', Mock(return_value=True))
def test_controldata(self): def test_controldata(self):
+3 -9
View File
@@ -133,20 +133,14 @@ class TestPostmasterProcess(unittest.TestCase):
c2.cmdline = Mock(return_value=["postgres: postgres postgres [local] idle"]) c2.cmdline = Mock(return_value=["postgres: postgres postgres [local] idle"])
c3 = Mock() c3 = Mock()
c3.cmdline = Mock(side_effect=psutil.NoSuchProcess(123)) c3.cmdline = Mock(side_effect=psutil.NoSuchProcess(123))
mock_wait.return_value = ([], [c2])
with patch('psutil.Process.children', Mock(return_value=[c1, c2, c3])): with patch('psutil.Process.children', Mock(return_value=[c1, c2, c3])):
proc = PostmasterProcess(123) proc = PostmasterProcess(123)
self.assertIsNone(proc.wait_for_user_backends_to_close(1)) self.assertIsNone(proc.wait_for_user_backends_to_close())
mock_wait.assert_called_with([c2], 1) mock_wait.assert_called_with([c2])
mock_wait.return_value = ([c2], [])
with patch('psutil.Process.children', Mock(return_value=[c1, c2, c3])):
proc = PostmasterProcess(123)
proc.wait_for_user_backends_to_close(1)
with patch('psutil.Process.children', Mock(side_effect=psutil.NoSuchProcess(123))): with patch('psutil.Process.children', Mock(side_effect=psutil.NoSuchProcess(123))):
proc = PostmasterProcess(123) proc = PostmasterProcess(123)
self.assertIsNone(proc.wait_for_user_backends_to_close(None)) self.assertIsNone(proc.wait_for_user_backends_to_close())
@patch('subprocess.Popen') @patch('subprocess.Popen')
@patch('os.setsid', Mock(), create=True) @patch('os.setsid', Mock(), create=True)
-56
View File
@@ -218,62 +218,6 @@ class TestRewind(BaseTestPostgresql):
self.r.cleanup_archive_status() self.r.cleanup_archive_status()
self.r.cleanup_archive_status() self.r.cleanup_archive_status()
@patch('os.path.isfile', Mock(return_value=True))
@patch('shutil.move', Mock(side_effect=OSError))
@patch('patroni.postgresql.rewind.logger.info')
def test_archive_ready_wals(self, mock_logger_info):
with patch('os.listdir', Mock(side_effect=OSError)), \
patch.object(Postgresql, 'get_guc_value', Mock(side_effect=['on', 'command %f'])):
self.r._archive_ready_wals()
mock_logger_info.assert_not_called()
# each assert_not_called() calls get_guc_value('archive_mode') + get_guc_value('archive_command')
get_guc_value_res = [
'', 'command %f',
'on', '',
]
with patch.object(Postgresql, 'get_guc_value', Mock(side_effect=get_guc_value_res)):
for _ in range(len(get_guc_value_res)//2):
self.r._archive_ready_wals()
mock_logger_info.assert_not_called()
with patch('os.listdir', Mock(return_value=['000000000000000000000000.ready'])):
# successful archive_command call
with patch.object(CancellableSubprocess, 'call', Mock(return_value=0)):
get_guc_value_res = [
'on', 'command %f',
'always', 'command %f',
]
with patch.object(Postgresql, 'get_guc_value', Mock(side_effect=get_guc_value_res)):
for _ in range(len(get_guc_value_res)//2):
self.r._archive_ready_wals()
mock_logger_info.assert_called_once()
self.assertEqual(('Trying to archive %s: %s',
'000000000000000000000000', 'command 000000000000000000000000'),
mock_logger_info.call_args[0])
mock_logger_info.reset_mock()
# failed archive_command call
with patch.object(CancellableSubprocess, 'call', Mock(return_value=1)):
with patch.object(Postgresql, 'get_guc_value', Mock(side_effect=['on', 'command %f'])):
self.r._archive_ready_wals()
self.assertEqual(('Trying to archive %s: %s',
'000000000000000000000000', 'command 000000000000000000000000'),
mock_logger_info.call_args_list[0][0])
self.assertEqual(('Failed to archive WAL segment %s', '000000000000000000000000'),
mock_logger_info.call_args_list[1][0])
mock_logger_info.reset_mock()
wal_files_to_skip = [
'000000000000000000000000.done',
'000000000000000000000001.partial.done',
'002.ready',
'U00000000000000000000001.ready',
]
with patch('os.listdir', Mock(return_value=wal_files_to_skip)):
self.r._archive_ready_wals()
mock_logger_info.assert_not_called()
@patch('os.unlink', Mock()) @patch('os.unlink', Mock())
@patch('os.listdir', Mock(return_value=[])) @patch('os.listdir', Mock(return_value=[]))
@patch('os.path.isfile', Mock(return_value=True)) @patch('os.path.isfile', Mock(return_value=True))
+3 -24
View File
@@ -4,19 +4,17 @@ import unittest
from mock import Mock, PropertyMock, patch from mock import Mock, PropertyMock, patch
from threading import Thread
from patroni import psycopg from patroni import psycopg
from patroni.dcs import Cluster, ClusterConfig, Member from patroni.dcs import Cluster, ClusterConfig, Member
from patroni.postgresql import Postgresql from patroni.postgresql import Postgresql
from patroni.postgresql.slots import SlotsAdvanceThread, SlotsHandler, fsync_dir from patroni.postgresql.slots import SlotsHandler, fsync_dir
from . import BaseTestPostgresql, psycopg_connect, MockCursor from . import BaseTestPostgresql, psycopg_connect, MockCursor
@patch('subprocess.call', Mock(return_value=0)) @patch('subprocess.call', Mock(return_value=0))
@patch('patroni.psycopg.connect', psycopg_connect) @patch('patroni.psycopg.connect', psycopg_connect)
@patch.object(Thread, 'start', Mock())
@patch.object(Postgresql, 'is_running', Mock(return_value=True)) @patch.object(Postgresql, 'is_running', Mock(return_value=True))
class TestSlotsHandler(BaseTestPostgresql): class TestSlotsHandler(BaseTestPostgresql):
@@ -44,10 +42,8 @@ class TestSlotsHandler(BaseTestPostgresql):
self.p.set_role('standby_leader') self.p.set_role('standby_leader')
self.s.sync_replication_slots(cluster, False) self.s.sync_replication_slots(cluster, False)
self.p.set_role('replica') self.p.set_role('replica')
with patch.object(Postgresql, 'is_leader', Mock(return_value=False)),\ with patch.object(Postgresql, 'is_leader', Mock(return_value=False)):
patch.object(SlotsHandler, 'drop_replication_slot') as mock_drop: self.s.sync_replication_slots(cluster, False)
self.s.sync_replication_slots(cluster, False, paused=True)
mock_drop.assert_not_called()
self.p.set_role('master') self.p.set_role('master')
with mock.patch('patroni.postgresql.Postgresql.role', new_callable=PropertyMock(return_value='replica')): with mock.patch('patroni.postgresql.Postgresql.role', new_callable=PropertyMock(return_value='replica')):
self.s.sync_replication_slots(cluster, False) self.s.sync_replication_slots(cluster, False)
@@ -93,7 +89,6 @@ class TestSlotsHandler(BaseTestPostgresql):
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), []) self.assertEqual(self.s.sync_replication_slots(self.cluster, False), [])
self.s._schedule_load_slots = False self.s._schedule_load_slots = False
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)),\ with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)),\
patch.object(SlotsAdvanceThread, 'schedule', Mock(return_value=(True, ['ls']))),\
patch.object(psycopg.OperationalError, 'diag') as mock_diag: patch.object(psycopg.OperationalError, 'diag') as mock_diag:
type(mock_diag).sqlstate = PropertyMock(return_value='58P01') type(mock_diag).sqlstate = PropertyMock(return_value='58P01')
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), ['ls']) self.assertEqual(self.s.sync_replication_slots(self.cluster, False), ['ls'])
@@ -126,7 +121,6 @@ class TestSlotsHandler(BaseTestPostgresql):
@patch.object(Postgresql, 'start', Mock(return_value=True)) @patch.object(Postgresql, 'start', Mock(return_value=True))
@patch.object(Postgresql, 'is_leader', Mock(return_value=False)) @patch.object(Postgresql, 'is_leader', Mock(return_value=False))
def test_on_promote(self): def test_on_promote(self):
self.s.schedule_advance_slots({'foo': {'bar': 100}})
self.s.copy_logical_slots(self.cluster, ['ls']) self.s.copy_logical_slots(self.cluster, ['ls'])
self.s.on_promote() self.s.on_promote()
@@ -136,18 +130,3 @@ class TestSlotsHandler(BaseTestPostgresql):
@patch('os.fsync', Mock(side_effect=OSError)) @patch('os.fsync', Mock(side_effect=OSError))
def test_fsync_dir(self): def test_fsync_dir(self):
self.assertRaises(OSError, fsync_dir, 'foo') self.assertRaises(OSError, fsync_dir, 'foo')
def test_slots_advance_thread(self):
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)),\
patch.object(psycopg.OperationalError, 'diag') as mock_diag:
type(mock_diag).sqlstate = PropertyMock(return_value='58P01')
self.s.schedule_advance_slots({'foo': {'bar': 100}})
self.s._advance.sync_slots()
with patch.object(SlotsAdvanceThread, 'sync_slots', Mock(side_effect=Exception)):
self.s._advance._condition.wait = Mock()
self.assertRaises(Exception, self.s._advance.run)
with patch.object(SlotsHandler, 'get_local_connection_cursor', Mock(side_effect=Exception)):
self.s.schedule_advance_slots({'foo': {'bar': 100}})
self.s._advance.sync_slots()
+17 -20
View File
@@ -63,7 +63,6 @@ config = {
"postgresql": { "postgresql": {
"listen": "127.0.0.2,::1:543", "listen": "127.0.0.2,::1:543",
"connect_address": "127.0.0.2:543", "connect_address": "127.0.0.2:543",
"proxy_address": "127.0.0.2:5433",
"authentication": { "authentication": {
"replication": {"username": "user"}, "replication": {"username": "user"},
"superuser": {"username": "user"}, "superuser": {"username": "user"},
@@ -142,14 +141,14 @@ class TestValidator(unittest.TestCase):
del directories[:] del directories[:]
def test_empty_config(self, mock_out, mock_err): def test_empty_config(self, mock_out, mock_err):
errors = schema({}) schema({})
output = "\n".join(errors) output = mock_out.getvalue()
expected = list(sorted(['name', 'postgresql', 'restapi', 'scope'] + available_dcs)) expected = list(sorted(['name', 'postgresql', 'restapi', 'scope'] + available_dcs))
self.assertEqual(expected, parse_output(output)) self.assertEqual(expected, parse_output(output))
def test_complete_config(self, mock_out, mock_err): def test_complete_config(self, mock_out, mock_err):
errors = schema(config) schema(config)
output = "\n".join(errors) output = mock_out.getvalue()
self.assertEqual(['postgresql.bin_dir', 'raft.bind_addr', 'raft.self_addr'], parse_output(output)) self.assertEqual(['postgresql.bin_dir', 'raft.bind_addr', 'raft.self_addr'], parse_output(output))
def test_bin_dir_is_file(self, mock_out, mock_err): def test_bin_dir_is_file(self, mock_out, mock_err):
@@ -157,11 +156,10 @@ class TestValidator(unittest.TestCase):
files.append(config["postgresql"]["bin_dir"]) files.append(config["postgresql"]["bin_dir"])
c = copy.deepcopy(config) c = copy.deepcopy(config)
c["restapi"]["connect_address"] = 'False:blabla' c["restapi"]["connect_address"] = 'False:blabla'
c["postgresql"]["listen"] = '*:543'
c["etcd"]["hosts"] = ["127.0.0.1:2379", "1244.0.0.1:2379", "127.0.0.1:invalidport"] c["etcd"]["hosts"] = ["127.0.0.1:2379", "1244.0.0.1:2379", "127.0.0.1:invalidport"]
c["kubernetes"]["pod_ip"] = "127.0.0.1111" c["kubernetes"]["pod_ip"] = "127.0.0.1111"
errors = schema(c) schema(c)
output = "\n".join(errors) output = mock_out.getvalue()
self.assertEqual(['etcd.hosts.1', 'etcd.hosts.2', 'kubernetes.pod_ip', 'postgresql.bin_dir', self.assertEqual(['etcd.hosts.1', 'etcd.hosts.2', 'kubernetes.pod_ip', 'postgresql.bin_dir',
'postgresql.data_dir', 'raft.bind_addr', 'raft.self_addr', 'postgresql.data_dir', 'raft.bind_addr', 'raft.self_addr',
'restapi.connect_address'], parse_output(output)) 'restapi.connect_address'], parse_output(output))
@@ -178,8 +176,8 @@ class TestValidator(unittest.TestCase):
c["etcd"]["host"] = "127.0.0.1:237" c["etcd"]["host"] = "127.0.0.1:237"
c["postgresql"]["listen"] = "127.0.0.1:5432" c["postgresql"]["listen"] = "127.0.0.1:5432"
with patch('patroni.validator.open', mock_open(read_data='9')): with patch('patroni.validator.open', mock_open(read_data='9')):
errors = schema(c) schema(c)
output = "\n".join(errors) output = mock_out.getvalue()
self.assertEqual(['consul.host', 'etcd.host', 'postgresql.bin_dir', 'postgresql.data_dir', 'postgresql.listen', self.assertEqual(['consul.host', 'etcd.host', 'postgresql.bin_dir', 'postgresql.data_dir', 'postgresql.listen',
'raft.bind_addr', 'raft.self_addr', 'restapi.connect_address'], parse_output(output)) 'raft.bind_addr', 'raft.self_addr', 'restapi.connect_address'], parse_output(output))
@@ -197,8 +195,8 @@ class TestValidator(unittest.TestCase):
files.append(os.path.join(config["postgresql"]["bin_dir"], "postgres")) files.append(os.path.join(config["postgresql"]["bin_dir"], "postgres"))
files.append(os.path.join(config["postgresql"]["bin_dir"], "pg_isready")) files.append(os.path.join(config["postgresql"]["bin_dir"], "pg_isready"))
with patch('patroni.validator.open', mock_open(read_data='12')): with patch('patroni.validator.open', mock_open(read_data='12')):
errors = schema(config) schema(config)
output = "\n".join(errors) output = mock_out.getvalue()
self.assertEqual(['raft.bind_addr', 'raft.self_addr'], parse_output(output)) self.assertEqual(['raft.bind_addr', 'raft.self_addr'], parse_output(output))
@patch('subprocess.check_output', Mock(return_value=b"postgres (PostgreSQL) 12.1")) @patch('subprocess.check_output', Mock(return_value=b"postgres (PostgreSQL) 12.1"))
@@ -210,12 +208,11 @@ class TestValidator(unittest.TestCase):
files.append(os.path.join(config["postgresql"]["data_dir"], "PG_VERSION")) files.append(os.path.join(config["postgresql"]["data_dir"], "PG_VERSION"))
c = copy.deepcopy(config) c = copy.deepcopy(config)
c["etcd"]["hosts"] = [] c["etcd"]["hosts"] = []
c["postgresql"]["listen"] = '127.0.0.2,*:543'
del c["postgresql"]["bin_dir"] del c["postgresql"]["bin_dir"]
with patch('patroni.validator.open', mock_open(read_data='11')): with patch('patroni.validator.open', mock_open(read_data='11')):
errors = schema(c) schema(c)
output = "\n".join(errors) output = mock_out.getvalue()
self.assertEqual(['etcd.hosts', 'postgresql.data_dir', 'postgresql.listen', self.assertEqual(['etcd.hosts', 'postgresql.data_dir',
'raft.bind_addr', 'raft.self_addr'], parse_output(output)) 'raft.bind_addr', 'raft.self_addr'], parse_output(output))
@patch('subprocess.check_output', Mock(return_value=b"postgres (PostgreSQL) 12.1")) @patch('subprocess.check_output', Mock(return_value=b"postgres (PostgreSQL) 12.1"))
@@ -227,8 +224,8 @@ class TestValidator(unittest.TestCase):
c = copy.deepcopy(config) c = copy.deepcopy(config)
del c["postgresql"]["bin_dir"] del c["postgresql"]["bin_dir"]
with patch('patroni.validator.open', mock_open(read_data='11')): with patch('patroni.validator.open', mock_open(read_data='11')):
errors = schema(c) schema(c)
output = "\n".join(errors) output = mock_out.getvalue()
self.assertEqual(['postgresql.data_dir', 'raft.bind_addr', 'raft.self_addr'], parse_output(output)) self.assertEqual(['postgresql.data_dir', 'raft.bind_addr', 'raft.self_addr'], parse_output(output))
def test_data_dir_is_empty_string(self, mock_out, mock_err): def test_data_dir_is_empty_string(self, mock_out, mock_err):
@@ -239,7 +236,7 @@ class TestValidator(unittest.TestCase):
c["postgresql"]["pg_hba"] = "" c["postgresql"]["pg_hba"] = ""
c["postgresql"]["data_dir"] = "" c["postgresql"]["data_dir"] = ""
c["postgresql"]["bin_dir"] = "" c["postgresql"]["bin_dir"] = ""
errors = schema(c) schema(c)
output = "\n".join(errors) output = mock_out.getvalue()
self.assertEqual(['kubernetes', 'postgresql.bin_dir', 'postgresql.data_dir', self.assertEqual(['kubernetes', 'postgresql.bin_dir', 'postgresql.data_dir',
'postgresql.pg_hba', 'raft.bind_addr', 'raft.self_addr'], parse_output(output)) 'postgresql.pg_hba', 'raft.bind_addr', 'raft.self_addr'], parse_output(output))
+1 -1
View File
@@ -35,7 +35,7 @@ def mock_ioctl(fd, op, arg=None, mutate_flag=False):
sys.stderr.write("Ioctl %d %d %r\n" % (fd, op, arg)) sys.stderr.write("Ioctl %d %d %r\n" % (fd, op, arg))
if op == linuxwd.WDIOC_GETSUPPORT: if op == linuxwd.WDIOC_GETSUPPORT:
sys.stderr.write("Get support\n") sys.stderr.write("Get support\n")
assert (mutate_flag is True) assert(mutate_flag is True)
arg.options = sum(map(linuxwd.WDIOF.get, ['SETTIMEOUT', 'KEEPALIVEPING'])) arg.options = sum(map(linuxwd.WDIOF.get, ['SETTIMEOUT', 'KEEPALIVEPING']))
arg.identity = (ctypes.c_ubyte*32)(*map(ord, 'Mock Watchdog')) arg.identity = (ctypes.c_ubyte*32)(*map(ord, 'Mock Watchdog'))
elif op == linuxwd.WDIOC_GETTIMEOUT: elif op == linuxwd.WDIOC_GETTIMEOUT:
+2 -4
View File
@@ -124,11 +124,9 @@ class TestPatroniSequentialThreadingHandler(unittest.TestCase):
self.assertIsNotNone(self.handler.create_connection((), 40)) self.assertIsNotNone(self.handler.create_connection((), 40))
self.assertIsNotNone(self.handler.create_connection(timeout=40)) self.assertIsNotNone(self.handler.create_connection(timeout=40))
@patch.object(SequentialThreadingHandler, 'select', Mock(side_effect=ValueError))
def test_select(self): def test_select(self):
with patch.object(SequentialThreadingHandler, 'select', Mock(side_effect=ValueError)): self.assertRaises(select.error, self.handler.select)
self.assertRaises(select.error, self.handler.select)
with patch.object(SequentialThreadingHandler, 'select', Mock(side_effect=IOError)):
self.assertRaises(Exception, self.handler.select)
class TestPatroniKazooClient(unittest.TestCase): class TestPatroniKazooClient(unittest.TestCase):