mirror of
https://github.com/outbackdingo/patroni.git
synced 2026-08-27 16:10:10 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
442bd3f434 | ||
|
|
e3e4ad0ada | ||
|
|
bad158046e | ||
|
|
55e1549341 | ||
|
|
2d79757309 | ||
|
|
e5d750e9b8 | ||
|
|
49f1ccf874 | ||
|
|
4d77b444dc | ||
|
|
b6b220dddb | ||
|
|
c152bf319d | ||
|
|
e5027c7a13 | ||
|
|
92d3e1c167 | ||
|
|
6ad5fee99d | ||
|
|
78d3f2cac2 | ||
|
|
ed47224540 | ||
|
|
c7a925a238 | ||
|
|
26244634ce | ||
|
|
b47c50a788 | ||
|
|
2bf7872d64 | ||
|
|
412d508023 | ||
|
|
53f89faaab | ||
|
|
2ed1793bbd | ||
|
|
1b6e23ab6a | ||
|
|
ef2922fe37 | ||
|
|
bda2bedf48 | ||
|
|
5a21ffa3e4 | ||
|
|
4ecaf445fa | ||
|
|
a293b77d25 | ||
|
|
8f8e9c9b81 | ||
|
|
580530b30f | ||
|
|
f4ae55b92a | ||
|
|
816b66311b | ||
|
|
8a227aa743 | ||
|
|
531063f676 | ||
|
|
db9b5962ec | ||
|
|
3dcdb16d2a | ||
|
|
84dc72b031 | ||
|
|
7102346f87 | ||
|
|
6d8d1a2556 | ||
|
|
88db6018ac | ||
|
|
cea1fa869b | ||
|
|
4a854a71c0 | ||
|
|
a2ef950e08 | ||
|
|
f92d975e7b | ||
|
|
2ee09d0a66 | ||
|
|
ae0ede6944 | ||
|
|
b6f057850a | ||
|
|
a0b32379e5 | ||
|
|
2d08e88c3e | ||
|
|
ea2b7d2368 | ||
|
|
f65efecac9 | ||
|
|
a8b73ef021 | ||
|
|
d8d634125c | ||
|
|
ead798d9ac | ||
|
|
cd5d20fa53 | ||
|
|
4c5cce5efd | ||
|
|
5b1fd23776 | ||
|
|
741243695a | ||
|
|
8d7828b079 | ||
|
|
8d773be533 |
@@ -0,0 +1 @@
|
|||||||
|
blank_issues_enabled: false
|
||||||
@@ -5,7 +5,6 @@ import subprocess
|
|||||||
import stat
|
import stat
|
||||||
import sys
|
import sys
|
||||||
import tarfile
|
import tarfile
|
||||||
import time
|
|
||||||
import zipfile
|
import zipfile
|
||||||
|
|
||||||
|
|
||||||
@@ -30,6 +29,7 @@ def install_requirements(what):
|
|||||||
requirements.append(r)
|
requirements.append(r)
|
||||||
|
|
||||||
subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'pip'])
|
subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'pip'])
|
||||||
|
subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'wheel'])
|
||||||
r = subprocess.call([sys.executable, '-m', 'pip', 'install'] + requirements)
|
r = subprocess.call([sys.executable, '-m', 'pip', 'install'] + requirements)
|
||||||
s = subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'setuptools'])
|
s = subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'setuptools'])
|
||||||
return s | r
|
return s | r
|
||||||
@@ -45,10 +45,8 @@ def install_packages(what):
|
|||||||
packages['exhibitor'] = packages['zookeeper']
|
packages['exhibitor'] = packages['zookeeper']
|
||||||
packages = packages.get(what, [])
|
packages = packages.get(what, [])
|
||||||
ver = versions.get(what)
|
ver = versions.get(what)
|
||||||
subprocess.call(['sudo', 'sed', '-i', 's/pgdg main.*$/pgdg main {0}/'.format(ver),
|
|
||||||
'/etc/apt/sources.list.d/pgdg.list'])
|
|
||||||
subprocess.call(['sudo', 'apt-get', 'update', '-y'])
|
subprocess.call(['sudo', 'apt-get', 'update', '-y'])
|
||||||
return subprocess.call(['sudo', 'apt-get', 'install', '-y', 'postgresql-' + ver, 'expect-dev', 'wget'] + packages)
|
return subprocess.call(['sudo', 'apt-get', 'install', '-y', 'postgresql-' + ver, 'expect-dev'] + packages)
|
||||||
|
|
||||||
|
|
||||||
def get_file(url, name):
|
def get_file(url, name):
|
||||||
@@ -122,57 +120,12 @@ def install_postgres():
|
|||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|
||||||
def setup_kubernetes():
|
|
||||||
get_file('https://storage.googleapis.com/minikube/k8sReleases/v1.7.0/localkube-linux-amd64', 'localkube')
|
|
||||||
chmod_755('localkube')
|
|
||||||
|
|
||||||
devnull = open(os.devnull, 'w')
|
|
||||||
subprocess.Popen(['sudo', 'nohup', './localkube', '--logtostderr=true', '--enable-dns=false'],
|
|
||||||
stdout=devnull, stderr=devnull)
|
|
||||||
for _ in range(0, 120):
|
|
||||||
if subprocess.call(['wget', '-qO', '-', 'http://127.0.0.1:8080/'], stdout=devnull, stderr=devnull) == 0:
|
|
||||||
break
|
|
||||||
time.sleep(1)
|
|
||||||
else:
|
|
||||||
print('localkube did not start')
|
|
||||||
return 1
|
|
||||||
|
|
||||||
subprocess.call('sudo chmod 644 /var/lib/localkube/certs/*', shell=True)
|
|
||||||
print('Set up .kube/config')
|
|
||||||
kube = os.path.join(os.path.expanduser('~'), '.kube')
|
|
||||||
os.makedirs(kube)
|
|
||||||
with open(os.path.join(kube, 'config'), 'w') as f:
|
|
||||||
f.write("""apiVersion: v1
|
|
||||||
clusters:
|
|
||||||
- cluster:
|
|
||||||
certificate-authority: /var/lib/localkube/certs/ca.crt
|
|
||||||
server: https://127.0.0.1:8443
|
|
||||||
name: local
|
|
||||||
contexts:
|
|
||||||
- context:
|
|
||||||
cluster: local
|
|
||||||
user: myself
|
|
||||||
name: local
|
|
||||||
current-context: local
|
|
||||||
kind: Config
|
|
||||||
preferences: {}
|
|
||||||
users:
|
|
||||||
- name: myself
|
|
||||||
user:
|
|
||||||
client-certificate: /var/lib/localkube/certs/apiserver.crt
|
|
||||||
client-key: /var/lib/localkube/certs/apiserver.key
|
|
||||||
""")
|
|
||||||
return 0
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
what = os.environ.get('DCS', sys.argv[1] if len(sys.argv) > 1 else 'all')
|
what = os.environ.get('DCS', sys.argv[1] if len(sys.argv) > 1 else 'all')
|
||||||
|
|
||||||
if what != 'all':
|
if what != 'all':
|
||||||
if sys.platform.startswith('linux'):
|
if sys.platform.startswith('linux'):
|
||||||
r = install_packages(what)
|
r = install_packages(what)
|
||||||
if r == 0 and what == 'kubernetes':
|
|
||||||
r = setup_kubernetes()
|
|
||||||
else:
|
else:
|
||||||
r = install_postgres()
|
r = install_postgres()
|
||||||
|
|
||||||
|
|||||||
@@ -1 +1 @@
|
|||||||
versions = {'etcd': '9.6', 'etcd3': '14', 'consul': '13', 'exhibitor': '12', 'raft': '11', 'kubernetes': '14'}
|
versions = {'etcd': '9.6', 'etcd3': '14', 'consul': '13', 'exhibitor': '12', 'raft': '11', 'kubernetes': '15'}
|
||||||
|
|||||||
@@ -0,0 +1,41 @@
|
|||||||
|
name: Publish Patroni distributions to PyPI and TestPyPI
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
tags:
|
||||||
|
- 'v[0-9]+.[0-9]+.[0-9]+'
|
||||||
|
release:
|
||||||
|
types:
|
||||||
|
- published
|
||||||
|
jobs:
|
||||||
|
build-n-publish:
|
||||||
|
name: Build and publish Patroni distributions to PyPI and TestPyPI
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@master
|
||||||
|
|
||||||
|
- name: Set up Python 3.9
|
||||||
|
uses: actions/setup-python@v4
|
||||||
|
with:
|
||||||
|
python-version: 3.9
|
||||||
|
|
||||||
|
- name: Install dependencies
|
||||||
|
run: python .github/workflows/install_deps.py
|
||||||
|
|
||||||
|
- name: Run tests and flake8
|
||||||
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
|
- name: Build a binary wheel and a source tarball
|
||||||
|
run: python setup.py sdist bdist_wheel
|
||||||
|
|
||||||
|
- name: Publish distribution to Test PyPI
|
||||||
|
if: github.event_name == 'push'
|
||||||
|
uses: pypa/[email protected]
|
||||||
|
with:
|
||||||
|
password: ${{ secrets.TEST_PYPI_API_TOKEN }}
|
||||||
|
repository_url: https://test.pypi.org/legacy/
|
||||||
|
|
||||||
|
- name: Publish distribution to PyPI
|
||||||
|
if: github.event_name == 'release'
|
||||||
|
uses: pypa/[email protected]
|
||||||
|
with:
|
||||||
|
password: ${{ secrets.PYPI_API_TOKEN }}
|
||||||
@@ -28,16 +28,17 @@ def main():
|
|||||||
version = versions.get(what)
|
version = versions.get(what)
|
||||||
path = '/usr/lib/postgresql/{0}/bin:.'.format(version)
|
path = '/usr/lib/postgresql/{0}/bin:.'.format(version)
|
||||||
unbuffer = ['timeout', '900', 'unbuffer']
|
unbuffer = ['timeout', '900', 'unbuffer']
|
||||||
args = ['--tags=-skip'] if what == 'etcd' else []
|
|
||||||
else:
|
else:
|
||||||
path = os.path.abspath(os.path.join('pgsql', 'bin'))
|
path = os.path.abspath(os.path.join('pgsql', 'bin'))
|
||||||
if sys.platform == 'darwin':
|
if sys.platform == 'darwin':
|
||||||
path += ':.'
|
path += ':.'
|
||||||
args = unbuffer = []
|
unbuffer = []
|
||||||
env['PATH'] = path + os.pathsep + env['PATH']
|
env['PATH'] = path + os.pathsep + env['PATH']
|
||||||
env['DCS'] = what
|
env['DCS'] = what
|
||||||
|
if what == 'kubernetes':
|
||||||
|
env['PATRONI_KUBERNETES_CONTEXT'] = 'k3d-k3s-default'
|
||||||
|
|
||||||
ret = subprocess.call(unbuffer + [sys.executable, '-m', 'behave'] + args, env=env)
|
ret = subprocess.call(unbuffer + [sys.executable, '-m', 'behave'], env=env)
|
||||||
|
|
||||||
if ret != 0:
|
if ret != 0:
|
||||||
if subprocess.call('grep . features/output/*_failed/*postgres?.*', shell=True) != 0:
|
if subprocess.call('grep . features/output/*_failed/*postgres?.*', shell=True) != 0:
|
||||||
|
|||||||
@@ -10,16 +10,16 @@ on:
|
|||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
unit:
|
unit:
|
||||||
runs-on: ${{ matrix.os }}-latest
|
runs-on: ${{ fromJson('{"ubuntu":"ubuntu-20.04","windows":"windows-latest","macos":"macos-latest"}')[matrix.os] }}
|
||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
os: [ubuntu, windows, macos]
|
os: [ubuntu, windows, macos]
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v1
|
- uses: actions/checkout@v3
|
||||||
- name: Set up Python 2.7
|
- name: Set up Python 2.7
|
||||||
uses: actions/setup-python@v2
|
uses: actions/setup-python@v4
|
||||||
with:
|
with:
|
||||||
python-version: 2.7
|
python-version: 2.7
|
||||||
if: matrix.os != 'windows'
|
if: matrix.os != 'windows'
|
||||||
@@ -31,7 +31,7 @@ jobs:
|
|||||||
if: matrix.os != 'windows'
|
if: matrix.os != 'windows'
|
||||||
|
|
||||||
- name: Set up Python 3.6
|
- name: Set up Python 3.6
|
||||||
uses: actions/setup-python@v2
|
uses: actions/setup-python@v4
|
||||||
with:
|
with:
|
||||||
python-version: 3.6
|
python-version: 3.6
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
@@ -40,7 +40,7 @@ jobs:
|
|||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
- name: Set up Python 3.7
|
- name: Set up Python 3.7
|
||||||
uses: actions/setup-python@v2
|
uses: actions/setup-python@v4
|
||||||
with:
|
with:
|
||||||
python-version: 3.7
|
python-version: 3.7
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
@@ -49,7 +49,7 @@ jobs:
|
|||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
- name: Set up Python 3.8
|
- name: Set up Python 3.8
|
||||||
uses: actions/setup-python@v2
|
uses: actions/setup-python@v4
|
||||||
with:
|
with:
|
||||||
python-version: 3.8
|
python-version: 3.8
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
@@ -58,7 +58,7 @@ jobs:
|
|||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
- name: Set up Python 3.9
|
- name: Set up Python 3.9
|
||||||
uses: actions/setup-python@v2
|
uses: actions/setup-python@v4
|
||||||
with:
|
with:
|
||||||
python-version: 3.9
|
python-version: 3.9
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
@@ -67,7 +67,7 @@ jobs:
|
|||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
- name: Set up Python 3.10
|
- name: Set up Python 3.10
|
||||||
uses: actions/setup-python@v2
|
uses: actions/setup-python@v4
|
||||||
with:
|
with:
|
||||||
python-version: '3.10'
|
python-version: '3.10'
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
@@ -89,7 +89,7 @@ jobs:
|
|||||||
run: python -m coveralls --service=github
|
run: python -m coveralls --service=github
|
||||||
|
|
||||||
behave:
|
behave:
|
||||||
runs-on: ${{ matrix.os }}-latest
|
runs-on: ${{ fromJson('{"ubuntu":"ubuntu-20.04","windows":"windows-latest","macos":"macos-latest"}')[matrix.os] }}
|
||||||
env:
|
env:
|
||||||
DCS: ${{ matrix.dcs }}
|
DCS: ${{ matrix.dcs }}
|
||||||
ETCDVERSION: 3.3.13
|
ETCDVERSION: 3.3.13
|
||||||
@@ -115,19 +115,25 @@ jobs:
|
|||||||
dcs: etcd3
|
dcs: etcd3
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v1
|
- uses: actions/checkout@v3
|
||||||
- name: Set up Python
|
- name: Set up Python
|
||||||
uses: actions/setup-python@v2
|
uses: actions/setup-python@v4
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python-version }}
|
python-version: ${{ matrix.python-version }}
|
||||||
|
- uses: nolar/setup-k3d-k3s@v1
|
||||||
|
if: matrix.dcs == 'kubernetes'
|
||||||
- name: Add postgresql apt repo
|
- name: Add postgresql apt repo
|
||||||
run: sudo sh -c 'echo "deb http://apt.postgresql.org/pub/repos/apt $(lsb_release -cs)-pgdg main" > /etc/apt/sources.list.d/pgdg.list'
|
run: |
|
||||||
|
sudo apt-get update -y
|
||||||
|
sudo apt-get install -y wget ca-certificates gnupg
|
||||||
|
sudo sh -c 'echo "deb http://apt.postgresql.org/pub/repos/apt $(lsb_release -cs)-pgdg main" > /etc/apt/sources.list.d/pgdg.list'
|
||||||
|
sudo sh -c 'wget -qO - https://www.postgresql.org/media/keys/ACCC4CF8.asc | gpg --dearmor > /etc/apt/trusted.gpg.d/apt.postgresql.org.gpg'
|
||||||
if: matrix.os == 'ubuntu'
|
if: matrix.os == 'ubuntu'
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: python .github/workflows/install_deps.py
|
run: python .github/workflows/install_deps.py
|
||||||
- name: Run behave tests
|
- name: Run behave tests
|
||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
- uses: actions/setup-python@v2
|
- uses: actions/setup-python@v4
|
||||||
with:
|
with:
|
||||||
python-version: '3.10'
|
python-version: '3.10'
|
||||||
- name: Install coveralls
|
- name: Install coveralls
|
||||||
@@ -144,7 +150,7 @@ jobs:
|
|||||||
needs: [unit, behave]
|
needs: [unit, behave]
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/setup-python@v2
|
- uses: actions/setup-python@v4
|
||||||
- run: python -m pip install coveralls
|
- run: python -m pip install coveralls
|
||||||
- run: python -m coveralls --service=github --finish
|
- run: python -m coveralls --service=github --finish
|
||||||
env:
|
env:
|
||||||
|
|||||||
@@ -0,0 +1,2 @@
|
|||||||
|
# global owners
|
||||||
|
* @CyberDem0n @hughcapet
|
||||||
+9
-7
@@ -1,6 +1,6 @@
|
|||||||
## This Dockerfile is meant to aid in the building and debugging patroni whilst developing on your local machine
|
## This Dockerfile is meant to aid in the building and debugging patroni whilst developing on your local machine
|
||||||
## It has all the necessary components to play/debug with a single node appliance, running etcd
|
## It has all the necessary components to play/debug with a single node appliance, running etcd
|
||||||
ARG PG_MAJOR=14
|
ARG PG_MAJOR=15
|
||||||
ARG COMPRESS=false
|
ARG COMPRESS=false
|
||||||
ARG PGHOME=/home/postgres
|
ARG PGHOME=/home/postgres
|
||||||
ARG PGDATA=$PGHOME/data
|
ARG PGDATA=$PGHOME/data
|
||||||
@@ -50,11 +50,11 @@ RUN set -ex \
|
|||||||
&& chown -R postgres:postgres /var/log \
|
&& chown -R postgres:postgres /var/log \
|
||||||
\
|
\
|
||||||
# Download etcd
|
# Download etcd
|
||||||
&& curl -sL https://github.com/coreos/etcd/releases/download/v${ETCDVERSION}/etcd-v${ETCDVERSION}-linux-amd64.tar.gz \
|
&& curl -sL https://github.com/coreos/etcd/releases/download/v${ETCDVERSION}/etcd-v${ETCDVERSION}-linux-$(dpkg --print-architecture).tar.gz \
|
||||||
| tar xz -C /usr/local/bin --strip=1 --wildcards --no-anchored etcd etcdctl \
|
| tar xz -C /usr/local/bin --strip=1 --wildcards --no-anchored etcd etcdctl \
|
||||||
\
|
\
|
||||||
# Download confd
|
# Download confd
|
||||||
&& curl -sL https://github.com/kelseyhightower/confd/releases/download/v${CONFDVERSION}/confd-${CONFDVERSION}-linux-amd64 \
|
&& curl -sL https://github.com/kelseyhightower/confd/releases/download/v${CONFDVERSION}/confd-${CONFDVERSION}-linux-$(dpkg --print-architecture) \
|
||||||
> /usr/local/bin/confd && chmod +x /usr/local/bin/confd \
|
> /usr/local/bin/confd && chmod +x /usr/local/bin/confd \
|
||||||
\
|
\
|
||||||
# Clean up all useless packages and some files
|
# Clean up all useless packages and some files
|
||||||
@@ -90,7 +90,7 @@ RUN set -ex \
|
|||||||
&& find /usr/bin -xtype l -delete \
|
&& find /usr/bin -xtype l -delete \
|
||||||
&& find /var/log -type f -exec truncate --size 0 {} \; \
|
&& find /var/log -type f -exec truncate --size 0 {} \; \
|
||||||
&& find /usr/lib/python3/dist-packages -name '*test*' | xargs rm -fr \
|
&& find /usr/lib/python3/dist-packages -name '*test*' | xargs rm -fr \
|
||||||
&& find /lib/x86_64-linux-gnu/security -type f ! -name pam_env.so ! -name pam_permit.so ! -name pam_unix.so -delete
|
&& find /lib/$(uname -m)-linux-gnu/security -type f ! -name pam_env.so ! -name pam_permit.so ! -name pam_unix.so -delete
|
||||||
|
|
||||||
# perform compression if it is necessary
|
# perform compression if it is necessary
|
||||||
ARG COMPRESS
|
ARG COMPRESS
|
||||||
@@ -99,8 +99,10 @@ RUN if [ "$COMPRESS" = "true" ]; then \
|
|||||||
# Allow certain sudo commands from postgres
|
# Allow certain sudo commands from postgres
|
||||||
&& echo 'postgres ALL=(ALL) NOPASSWD: /bin/tar xpJf /a.tar.xz -C /, /bin/rm /a.tar.xz, /bin/ln -snf dash /bin/sh' >> /etc/sudoers \
|
&& echo 'postgres ALL=(ALL) NOPASSWD: /bin/tar xpJf /a.tar.xz -C /, /bin/rm /a.tar.xz, /bin/ln -snf dash /bin/sh' >> /etc/sudoers \
|
||||||
&& ln -snf busybox /bin/sh \
|
&& ln -snf busybox /bin/sh \
|
||||||
&& files="/bin/sh /usr/bin/sudo /usr/lib/sudo/sudoers.so /lib/x86_64-linux-gnu/security/pam_*.so" \
|
&& arch=$(uname -m) \
|
||||||
&& libs="$(ldd $files | awk '{print $3;}' | grep '^/' | sort -u) /lib/x86_64-linux-gnu/ld-linux-x86-64.so.* /lib/x86_64-linux-gnu/libnsl.so.* /lib/x86_64-linux-gnu/libnss_compat.so.*" \
|
&& darch=$(uname -m | sed 's/_/-/') \
|
||||||
|
&& files="/bin/sh /usr/bin/sudo /usr/lib/sudo/sudoers.so /lib/$arch-linux-gnu/security/pam_*.so" \
|
||||||
|
&& libs="$(ldd $files | awk '{print $3;}' | grep '^/' | sort -u) /lib/ld-linux-$darch.so.* /lib/$arch-linux-gnu/ld-linux-$darch.so.* /lib/$arch-linux-gnu/libnsl.so.* /lib/$arch-linux-gnu/libnss_compat.so.* /lib/$arch-linux-gnu/libnss_files.so.*" \
|
||||||
&& (echo /var/run $files $libs | tr ' ' '\n' && realpath $files $libs) | sort -u | sed 's/^\///' > /exclude \
|
&& (echo /var/run $files $libs | tr ' ' '\n' && realpath $files $libs) | sort -u | sed 's/^\///' > /exclude \
|
||||||
&& find /etc/alternatives -xtype l -delete \
|
&& find /etc/alternatives -xtype l -delete \
|
||||||
&& save_dirs="usr lib var bin sbin etc/ssl etc/init.d etc/alternatives etc/apt" \
|
&& save_dirs="usr lib var bin sbin etc/ssl etc/init.d etc/alternatives etc/apt" \
|
||||||
@@ -117,7 +119,7 @@ RUN if [ "$COMPRESS" = "true" ]; then \
|
|||||||
FROM scratch
|
FROM scratch
|
||||||
COPY --from=builder / /
|
COPY --from=builder / /
|
||||||
|
|
||||||
LABEL maintainer="Alexander Kukushkin <alexander.kukushkin@zalando.de>"
|
LABEL maintainer="Alexander Kukushkin <akukushkin@microsoft.com>"
|
||||||
|
|
||||||
ARG PG_MAJOR
|
ARG PG_MAJOR
|
||||||
ARG COMPRESS
|
ARG COMPRESS
|
||||||
|
|||||||
+2
-3
@@ -1,3 +1,2 @@
|
|||||||
Alexander Kukushkin <alexander.kukushkin@zalando.de>
|
Alexander Kukushkin <akukushkin@microsoft.com>
|
||||||
Feike Steenbergen <feike.steenbergen@zalando.de>
|
Polina Bungina <polina.bungina@zalando.de>
|
||||||
Oleksii Kliukin <[email protected]>
|
|
||||||
|
|||||||
+1
-1
@@ -12,7 +12,7 @@ Patroni is a template for you to create your own customized, high-availability s
|
|||||||
|
|
||||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely.
|
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely.
|
||||||
|
|
||||||
Currently supported PostgreSQL versions: 9.3 to 14.
|
Currently supported PostgreSQL versions: 9.3 to 15.
|
||||||
|
|
||||||
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the `Kubernetes <https://github.com/zalando/patroni/blob/master/docs/kubernetes.rst>`__ chapter of the Patroni documentation.
|
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the `Kubernetes <https://github.com/zalando/patroni/blob/master/docs/kubernetes.rst>`__ chapter of the Patroni documentation.
|
||||||
|
|
||||||
|
|||||||
@@ -123,11 +123,12 @@ PostgreSQL
|
|||||||
----------
|
----------
|
||||||
- **PATRONI\_POSTGRESQL\_LISTEN**: IP address + port that Postgres listens to. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node.
|
- **PATRONI\_POSTGRESQL\_LISTEN**: IP address + port that Postgres listens to. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node.
|
||||||
- **PATRONI\_POSTGRESQL\_CONNECT\_ADDRESS**: IP address + port through which Postgres is accessible from other nodes and applications.
|
- **PATRONI\_POSTGRESQL\_CONNECT\_ADDRESS**: IP address + port through which Postgres is accessible from other nodes and applications.
|
||||||
|
- **PATRONI\_POSTGRESQL\_PROXY\_ADDRESS**: IP address + port through which a connection pool (e.g. pgbouncer) running next to Postgres is accessible. The value is written to the member key in DCS as ``proxy_url`` and could be used/useful for service discovery.
|
||||||
- **PATRONI\_POSTGRESQL\_DATA\_DIR**: The location of the Postgres data directory, either existing or to be initialized by Patroni.
|
- **PATRONI\_POSTGRESQL\_DATA\_DIR**: The location of the Postgres data directory, either existing or to be initialized by Patroni.
|
||||||
- **PATRONI\_POSTGRESQL\_CONFIG\_DIR**: The location of the Postgres configuration directory, defaults to the data directory. Must be writable by Patroni.
|
- **PATRONI\_POSTGRESQL\_CONFIG\_DIR**: The location of the Postgres configuration directory, defaults to the data directory. Must be writable by Patroni.
|
||||||
- **PATRONI\_POSTGRESQL\_BIN_DIR**: Path to PostgreSQL binaries. (pg_ctl, pg_rewind, pg_basebackup, postgres) The default value is an empty string meaning that PATH environment variable will be used to find the executables.
|
- **PATRONI\_POSTGRESQL\_BIN_DIR**: Path to PostgreSQL binaries. (pg_ctl, pg_rewind, pg_basebackup, postgres) The default value is an empty string meaning that PATH environment variable will be used to find the executables.
|
||||||
- **PATRONI\_POSTGRESQL\_PGPASS**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup and under some other circumstances. The location must be writable by Patroni.
|
- **PATRONI\_POSTGRESQL\_PGPASS**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup and under some other circumstances. The location must be writable by Patroni.
|
||||||
- **PATRONI\_REPLICATION\_USERNAME**: replication username; the user will be created during initialization. Replicas will use this user to access master via streaming replication
|
- **PATRONI\_REPLICATION\_USERNAME**: replication username; the user will be created during initialization. Replicas will use this user to access the replication source via streaming replication
|
||||||
- **PATRONI\_REPLICATION\_PASSWORD**: replication password; the user will be created during initialization.
|
- **PATRONI\_REPLICATION\_PASSWORD**: replication password; the user will be created during initialization.
|
||||||
- **PATRONI\_REPLICATION\_SSLMODE**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
|
- **PATRONI\_REPLICATION\_SSLMODE**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
|
||||||
- **PATRONI\_REPLICATION\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
- **PATRONI\_REPLICATION\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||||
|
|||||||
+1
-1
@@ -109,7 +109,7 @@ Planning the Number of PostgreSQL Nodes
|
|||||||
---------------------------------------
|
---------------------------------------
|
||||||
|
|
||||||
Patroni/PostgreSQL nodes are decoupled from DCS nodes (except when Patroni implements RAFT on its own) and therefore
|
Patroni/PostgreSQL nodes are decoupled from DCS nodes (except when Patroni implements RAFT on its own) and therefore
|
||||||
there is no requirement on the minimal number of nodes. Running a cluster consisting of one master and one standby is
|
there is no requirement on the minimal number of nodes. Running a cluster consisting of one primary and one standby is
|
||||||
perfectly fine. You can add more standby nodes later.
|
perfectly fine. You can add more standby nodes later.
|
||||||
|
|
||||||
Running and Configuring
|
Running and Configuring
|
||||||
|
|||||||
+12
-11
@@ -17,21 +17,21 @@ Dynamic configuration is stored in the DCS (Distributed Configuration Store) and
|
|||||||
- **maximum\_lag\_on\_failover**: the maximum bytes a follower may lag to be able to participate in leader election.
|
- **maximum\_lag\_on\_failover**: the maximum bytes a follower may lag to be able to participate in leader election.
|
||||||
- **maximum\_lag\_on\_syncnode**: the maximum bytes a synchronous follower may lag before it is considered as an unhealthy candidate and swapped by healthy asynchronous follower. Patroni utilize the max replica lsn if there is more than one follower, otherwise it will use leader's current wal lsn. Default is -1, Patroni will not take action to swap synchronous unhealthy follower when the value is set to 0 or below. Please set the value high enough so Patroni won't swap synchrounous follower fequently during high transaction volume.
|
- **maximum\_lag\_on\_syncnode**: the maximum bytes a synchronous follower may lag before it is considered as an unhealthy candidate and swapped by healthy asynchronous follower. Patroni utilize the max replica lsn if there is more than one follower, otherwise it will use leader's current wal lsn. Default is -1, Patroni will not take action to swap synchronous unhealthy follower when the value is set to 0 or below. Please set the value high enough so Patroni won't swap synchrounous follower fequently during high transaction volume.
|
||||||
- **max\_timelines\_history**: maximum number of timeline history items kept in DCS. Default value: 0. When set to 0, it keeps the full history in DCS.
|
- **max\_timelines\_history**: maximum number of timeline history items kept in DCS. Default value: 0. When set to 0, it keeps the full history in DCS.
|
||||||
- **master\_start\_timeout**: the amount of time a master is allowed to recover from failures before failover is triggered (in seconds). Default is 300 seconds. When set to 0 failover is done immediately after a crash is detected if possible. When using asynchronous replication a failover can cause lost transactions. Worst case failover time for master failure is: loop\_wait + master\_start\_timeout + loop\_wait, unless master\_start\_timeout is zero, in which case it's just loop\_wait. Set the value according to your durability/availability tradeoff.
|
- **master\_start\_timeout**: the amount of time a primary is allowed to recover from failures before failover is triggered (in seconds). Default is 300 seconds. When set to 0 failover is done immediately after a crash is detected if possible. When using asynchronous replication a failover can cause lost transactions. Worst case failover time for primary failure is: loop\_wait + master\_start\_timeout + loop\_wait, unless master\_start\_timeout is zero, in which case it's just loop\_wait. Set the value according to your durability/availability tradeoff.
|
||||||
- **master\_stop\_timeout**: The number of seconds Patroni is allowed to wait when stopping Postgres and effective only when synchronous_mode is enabled. When set to > 0 and the synchronous_mode is enabled, Patroni sends SIGKILL to the postmaster if the stop operation is running for more than the value set by master_stop_timeout. Set the value according to your durability/availability tradeoff. If the parameter is not set or set <= 0, master_stop_timeout does not apply.
|
- **master\_stop\_timeout**: The number of seconds Patroni is allowed to wait when stopping Postgres and effective only when synchronous_mode is enabled. When set to > 0 and the synchronous_mode is enabled, Patroni sends SIGKILL to the postmaster if the stop operation is running for more than the value set by master_stop_timeout. Set the value according to your durability/availability tradeoff. If the parameter is not set or set <= 0, master_stop_timeout does not apply.
|
||||||
- **synchronous\_mode**: turns on synchronous replication mode. In this mode a replica will be chosen as synchronous and only the latest leader and synchronous replica are able to participate in leader election. Synchronous mode makes sure that successfully committed transactions will not be lost at failover, at the cost of losing availability for writes when Patroni cannot ensure transaction durability. See :ref:`replication modes documentation <replication_modes>` for details.
|
- **synchronous\_mode**: turns on synchronous replication mode. In this mode a replica will be chosen as synchronous and only the latest leader and synchronous replica are able to participate in leader election. Synchronous mode makes sure that successfully committed transactions will not be lost at failover, at the cost of losing availability for writes when Patroni cannot ensure transaction durability. See :ref:`replication modes documentation <replication_modes>` for details.
|
||||||
- **synchronous\_mode\_strict**: prevents disabling synchronous replication if no synchronous replicas are available, blocking all client writes to the master. See :ref:`replication modes documentation <replication_modes>` for details.
|
- **synchronous\_mode\_strict**: prevents disabling synchronous replication if no synchronous replicas are available, blocking all client writes to the primary. See :ref:`replication modes documentation <replication_modes>` for details.
|
||||||
- **postgresql**:
|
- **postgresql**:
|
||||||
- **use\_pg\_rewind**: whether or not to use pg_rewind. Defaults to `false`.
|
- **use\_pg\_rewind**: whether or not to use pg_rewind. Defaults to `false`.
|
||||||
- **use\_slots**: whether or not to use replication slots. Defaults to `true` on PostgreSQL 9.4+.
|
- **use\_slots**: whether or not to use replication slots. Defaults to `true` on PostgreSQL 9.4+.
|
||||||
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower. There is no recovery.conf anymore in PostgreSQL 12, but you may continue using this section, because Patroni handles it transparently.
|
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower. There is no recovery.conf anymore in PostgreSQL 12, but you may continue using this section, because Patroni handles it transparently.
|
||||||
- **parameters**: list of configuration settings for Postgres.
|
- **parameters**: list of configuration settings for Postgres.
|
||||||
- **standby\_cluster**: if this section is defined, we want to bootstrap a standby cluster.
|
- **standby\_cluster**: if this section is defined, we want to bootstrap a standby cluster.
|
||||||
- **host**: an address of remote master
|
- **host**: an address of remote node
|
||||||
- **port**: a port of remote master
|
- **port**: a port of remote node
|
||||||
- **primary\_slot\_name**: which slot on the remote master to use for replication. This parameter is optional, the default value is derived from the instance name (see function `slot_name_from_member_name`).
|
- **primary\_slot\_name**: which slot on the remote node to use for replication. This parameter is optional, the default value is derived from the instance name (see function `slot_name_from_member_name`).
|
||||||
- **create\_replica\_methods**: an ordered list of methods that can be used to bootstrap standby leader from the remote master, can be different from the list defined in :ref:`postgresql_settings`
|
- **create\_replica\_methods**: an ordered list of methods that can be used to bootstrap standby leader from the remote primary, can be different from the list defined in :ref:`postgresql_settings`
|
||||||
- **restore\_command**: command to restore WAL records from the remote master to standby leader, can be different from the list defined in :ref:`postgresql_settings`
|
- **restore\_command**: command to restore WAL records from the remote primary to nodes in a standby cluster, can be different from the list defined in :ref:`postgresql_settings`
|
||||||
- **archive\_cleanup\_command**: cleanup command for standby leader
|
- **archive\_cleanup\_command**: cleanup command for standby leader
|
||||||
- **recovery\_min\_apply\_delay**: how long to wait before actually apply WAL records on a standby leader
|
- **recovery\_min\_apply\_delay**: how long to wait before actually apply WAL records on a standby leader
|
||||||
- **slots**: define permanent replication slots. These slots will be preserved during switchover/failover. The logical slots are copied from the primary to a standby with restart, and after that their position advanced every **loop_wait** seconds (if necessary). Copying logical slot files performed via ``libpq`` connection and using either rewind or superuser credentials (see **postgresql.authentication** section). There is always a chance that the logical slot position on the replica is a bit behind the former primary, therefore application should be prepared that some messages could be received the second time after the failover. The easiest way of doing so - tracking ``confirmed_flush_lsn``. Enabling permanent logical replication slots requires **postgresql.use_slots** to be set and will also automatically enable the ``hot_standby_feedback``. Since the failover of logical replication slots is unsafe on PostgreSQL 9.6 and older and PostgreSQL version 10 is missing some important functions, the feature only works with PostgreSQL 11+.
|
- **slots**: define permanent replication slots. These slots will be preserved during switchover/failover. The logical slots are copied from the primary to a standby with restart, and after that their position advanced every **loop_wait** seconds (if necessary). Copying logical slot files performed via ``libpq`` connection and using either rewind or superuser credentials (see **postgresql.authentication** section). There is always a chance that the logical slot position on the replica is a bit behind the former primary, therefore application should be prepared that some messages could be received the second time after the failover. The easiest way of doing so - tracking ``confirmed_flush_lsn``. Enabling permanent logical replication slots requires **postgresql.use_slots** to be set and will also automatically enable the ``hot_standby_feedback``. Since the failover of logical replication slots is unsafe on PostgreSQL 9.6 and older and PostgreSQL version 10 is missing some important functions, the feature only works with PostgreSQL 11+.
|
||||||
@@ -131,7 +131,7 @@ Most of the parameters are optional, but you have to specify one of the **host**
|
|||||||
- **checks**: (optional) list of Consul health checks used for the session. By default an empty list is used.
|
- **checks**: (optional) list of Consul health checks used for the session. By default an empty list is used.
|
||||||
- **register\_service**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, replica or standby-leader depending on the node's role. Defaults to **false**.
|
- **register\_service**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, replica or standby-leader depending on the node's role. Defaults to **false**.
|
||||||
- **service\_tags**: (optional) additional static tags to add to the Consul service apart from the role (``master``/``replica``/``standby-leader``). By default an empty list is used.
|
- **service\_tags**: (optional) additional static tags to add to the Consul service apart from the role (``master``/``replica``/``standby-leader``). By default an empty list is used.
|
||||||
- **service\_check\_interval**: (optional) how often to perform health check against registered url.
|
- **service\_check\_interval**: (optional) how often to perform health check against registered url. Defaults to '5s'.
|
||||||
- **service\_check\_tls\_server\_name**: (optional) overide SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
|
- **service\_check\_tls\_server\_name**: (optional) overide SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
|
||||||
|
|
||||||
The ``token`` needs to have the following ACL permissions:
|
The ``token`` needs to have the following ACL permissions:
|
||||||
@@ -262,7 +262,7 @@ PostgreSQL
|
|||||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
- **replication**:
|
- **replication**:
|
||||||
- **username**: replication username; the user will be created during initialization. Replicas will use this user to access master via streaming replication
|
- **username**: replication username; the user will be created during initialization. Replicas will use this user to access the replication source via streaming replication
|
||||||
- **password**: replication password; the user will be created during initialization.
|
- **password**: replication password; the user will be created during initialization.
|
||||||
- **sslmode**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
|
- **sslmode**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
|
||||||
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||||
@@ -292,6 +292,7 @@ PostgreSQL
|
|||||||
- **on\_start**: run this script when the postgres starts.
|
- **on\_start**: run this script when the postgres starts.
|
||||||
- **on\_stop**: run this script when the postgres stops.
|
- **on\_stop**: run this script when the postgres stops.
|
||||||
- **connect\_address**: IP address + port through which Postgres is accessible from other nodes and applications.
|
- **connect\_address**: IP address + port through which Postgres is accessible from other nodes and applications.
|
||||||
|
- **proxy\_address**: IP address + port through which a connection pool (e.g. pgbouncer) running next to Postgres is accessible. The value is written to the member key in DCS as ``proxy_url`` and could be used/useful for service discovery.
|
||||||
- **create\_replica\_methods**: an ordered list of the create methods for turning a Patroni node into a new replica.
|
- **create\_replica\_methods**: an ordered list of the create methods for turning a Patroni node into a new replica.
|
||||||
"basebackup" is the default method; other methods are assumed to refer to scripts, each of which is configured as its
|
"basebackup" is the default method; other methods are assumed to refer to scripts, each of which is configured as its
|
||||||
own config item. See :ref:`custom replica creation methods documentation <custom_replica_creation>` for further explanation.
|
own config item. See :ref:`custom replica creation methods documentation <custom_replica_creation>` for further explanation.
|
||||||
@@ -314,14 +315,14 @@ PostgreSQL
|
|||||||
- **pg\_ctl\_timeout**: How long should pg_ctl wait when doing ``start``, ``stop`` or ``restart``. Default value is 60 seconds.
|
- **pg\_ctl\_timeout**: How long should pg_ctl wait when doing ``start``, ``stop`` or ``restart``. Default value is 60 seconds.
|
||||||
- **use\_pg\_rewind**: try to use pg\_rewind on the former leader when it joins cluster as a replica.
|
- **use\_pg\_rewind**: try to use pg\_rewind on the former leader when it joins cluster as a replica.
|
||||||
- **remove\_data\_directory\_on\_rewind\_failure**: If this option is enabled, Patroni will remove the PostgreSQL data directory and recreate the replica. Otherwise it will try to follow the new leader. Default value is **false**.
|
- **remove\_data\_directory\_on\_rewind\_failure**: If this option is enabled, Patroni will remove the PostgreSQL data directory and recreate the replica. Otherwise it will try to follow the new leader. Default value is **false**.
|
||||||
- **remove\_data\_directory\_on\_diverged\_timelines**: Patroni will remove the PostgreSQL data directory and recreate the replica if it notices that timelines are diverging and the former master can not start streaming from the new master. This option is useful when ``pg_rewind`` can not be used. While performing timelines divergence check on PostgreSQL v10 and older Patroni will try to connect with replication credential to the "postgres" database. Hence, such access should be allowed in the pg_hba.conf. Default value is **false**.
|
- **remove\_data\_directory\_on\_diverged\_timelines**: Patroni will remove the PostgreSQL data directory and recreate the replica if it notices that timelines are diverging and the former primary can not start streaming from the new primary. This option is useful when ``pg_rewind`` can not be used. While performing timelines divergence check on PostgreSQL v10 and older Patroni will try to connect with replication credential to the "postgres" database. Hence, such access should be allowed in the pg_hba.conf. Default value is **false**.
|
||||||
- **replica\_method**: for each create_replica_methods other than basebackup, you would add a configuration section of the same name. At a minimum, this should include "command" with a full path to the actual script to be executed. Other configuration parameters will be passed along to the script in the form "parameter=value".
|
- **replica\_method**: for each create_replica_methods other than basebackup, you would add a configuration section of the same name. At a minimum, this should include "command" with a full path to the actual script to be executed. Other configuration parameters will be passed along to the script in the form "parameter=value".
|
||||||
- **pre\_promote**: a fencing script that executes during a failover after acquiring the leader lock but before promoting the replica. If the script exits with a non-zero code, Patroni does not promote the replica and removes the leader key from DCS.
|
- **pre\_promote**: a fencing script that executes during a failover after acquiring the leader lock but before promoting the replica. If the script exits with a non-zero code, Patroni does not promote the replica and removes the leader key from DCS.
|
||||||
|
|
||||||
REST API
|
REST API
|
||||||
--------
|
--------
|
||||||
- **restapi**:
|
- **restapi**:
|
||||||
- **connect\_address**: IP address (or hostname) and port, to access the Patroni's :ref:`REST API <rest_api>`. All the members of the cluster must be able to connect to this address, so unless the Patroni setup is intended for a demo inside the localhost, this address must be a non "localhost" or loopback address (ie: "localhost" or "127.0.0.1"). It can serve as an endpoint for HTTP health checks (read below about the "listen" REST API parameter), and also for user queries (either directly or via the REST API), as well as for the health checks done by the cluster members during leader elections (for example, to determine whether the master is still running, or if there is a node which has a WAL position that is ahead of the one doing the query; etc.) The connect_address is put in the member key in DCS, making it possible to translate the member name into the address to connect to its REST API.
|
- **connect\_address**: IP address (or hostname) and port, to access the Patroni's :ref:`REST API <rest_api>`. All the members of the cluster must be able to connect to this address, so unless the Patroni setup is intended for a demo inside the localhost, this address must be a non "localhost" or loopback address (ie: "localhost" or "127.0.0.1"). It can serve as an endpoint for HTTP health checks (read below about the "listen" REST API parameter), and also for user queries (either directly or via the REST API), as well as for the health checks done by the cluster members during leader elections (for example, to determine whether the leader is still running, or if there is a node which has a WAL position that is ahead of the one doing the query; etc.) The connect_address is put in the member key in DCS, making it possible to translate the member name into the address to connect to its REST API.
|
||||||
|
|
||||||
- **listen**: IP address (or hostname) and port that Patroni will listen to for the REST API - to provide also the same health checks and cluster messaging between the participating nodes, as described above. to provide health-check information for HAProxy (or any other load balancer capable of doing a HTTP "OPTION" or "GET" checks).
|
- **listen**: IP address (or hostname) and port that Patroni will listen to for the REST API - to provide also the same health checks and cluster messaging between the participating nodes, as described above. to provide health-check information for HAProxy (or any other load balancer capable of doing a HTTP "OPTION" or "GET" checks).
|
||||||
|
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ Patroni configuration is stored in the DCS (Distributed Configuration Store). Th
|
|||||||
|
|
||||||
The local configuration can be either a single YAML file or a directory. When it is a directory, all YAML files in that directory are loaded one by one in sorted order. In case a key is defined in multiple files, the occurrence in the last file takes precedence.
|
The local configuration can be either a single YAML file or a directory. When it is a directory, all YAML files in that directory are loaded one by one in sorted order. In case a key is defined in multiple files, the occurrence in the last file takes precedence.
|
||||||
|
|
||||||
Some of the PostgreSQL parameters must hold the same values on the master and the replicas. For those, values set either in the local patroni configuration files or via the environment variables take no effect. To alter or set their values one must change the shared configuration in the DCS. Below is the actual list of such parameters together with the default values:
|
Some of the PostgreSQL parameters must hold the same values on the primary and the replicas. For those, values set either in the local patroni configuration files or via the environment variables take no effect. To alter or set their values one must change the shared configuration in the DCS. Below is the actual list of such parameters together with the default values:
|
||||||
|
|
||||||
- max_connections: 100
|
- max_connections: 100
|
||||||
- max_locks_per_transaction: 64
|
- max_locks_per_transaction: 64
|
||||||
@@ -32,7 +32,7 @@ Some of the PostgreSQL parameters must hold the same values on the master and th
|
|||||||
- wal_log_hints: on
|
- wal_log_hints: on
|
||||||
- track_commit_timestamp: off
|
- track_commit_timestamp: off
|
||||||
|
|
||||||
For the parameters below, PostgreSQL does not require equal values among the master and all the replicas. However, considering the possibility of a replica to become the master at any time, it doesn't really make sense to set them differently; therefore, Patroni restricts setting their values to the Dynamic configuration
|
For the parameters below, PostgreSQL does not require equal values among the primary and all the replicas. However, considering the possibility of a replica to become the primary at any time, it doesn't really make sense to set them differently; therefore, Patroni restricts setting their values to the Dynamic configuration
|
||||||
|
|
||||||
- max_wal_senders: 5
|
- max_wal_senders: 5
|
||||||
- max_replication_slots: 5
|
- max_replication_slots: 5
|
||||||
@@ -86,4 +86,4 @@ Also, the following Patroni configuration options can be changed only dynamicall
|
|||||||
Upon changing these options, Patroni will read the relevant section of the configuration stored in DCS and change its
|
Upon changing these options, Patroni will read the relevant section of the configuration stored in DCS and change its
|
||||||
run-time values.
|
run-time values.
|
||||||
|
|
||||||
Patroni nodes are dumping the state of the DCS options to disk upon for every change of the configuration into the file ``patroni.dynamic.json`` located in the Postgres data directory. Only the master is allowed to restore these options from the on-disk dump if these are completely absent from the DCS or if they are invalid.
|
Patroni nodes are dumping the state of the DCS options to disk upon for every change of the configuration into the file ``patroni.dynamic.json`` located in the Postgres data directory. Only the leader is allowed to restore these options from the on-disk dump if these are completely absent from the DCS or if they are invalid.
|
||||||
|
|||||||
@@ -29,11 +29,11 @@ Major Upgrade of PostgreSQL Version
|
|||||||
The only possible way to do a major upgrade currently is:
|
The only possible way to do a major upgrade currently is:
|
||||||
|
|
||||||
1. Stop Patroni
|
1. Stop Patroni
|
||||||
2. Upgrade PostgreSQL binaries and perform `pg_upgrade <https://www.postgresql.org/docs/current/pgupgrade.html>`_ on the master node
|
2. Upgrade PostgreSQL binaries and perform `pg_upgrade <https://www.postgresql.org/docs/current/pgupgrade.html>`_ on the primary node
|
||||||
3. Update patroni.yml
|
3. Update patroni.yml
|
||||||
4. Remove the initialize key from DCS or wipe complete cluster state from DCS. The second one could be achieved by running ``patronictl remove <cluster-name>``. It is necessary because pg_upgrade runs initdb which actually creates a new database with a new PostgreSQL system identifier.
|
4. Remove the initialize key from DCS or wipe complete cluster state from DCS. The second one could be achieved by running ``patronictl remove <cluster-name>``. It is necessary because pg_upgrade runs initdb which actually creates a new database with a new PostgreSQL system identifier.
|
||||||
5. If you wiped the cluster state in the previous step, you may wish to copy patroni.dynamic.json from old data dir to the new one. It will help you to retain some PostgreSQL parameters you had set before.
|
5. If you wiped the cluster state in the previous step, you may wish to copy patroni.dynamic.json from old data dir to the new one. It will help you to retain some PostgreSQL parameters you had set before.
|
||||||
6. Start Patroni on the master node.
|
6. Start Patroni on the primary node.
|
||||||
7. Upgrade PostgreSQL binaries, update patroni.yml and wipe the data_dir on standby nodes.
|
7. Upgrade PostgreSQL binaries, update patroni.yml and wipe the data_dir on standby nodes.
|
||||||
8. Start Patroni on the standby nodes and wait for the replication to complete.
|
8. Start Patroni on the standby nodes and wait for the replication to complete.
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -10,7 +10,7 @@ Patroni is a template for you to create your own customized, high-availability s
|
|||||||
|
|
||||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
|
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
|
||||||
|
|
||||||
Currently supported PostgreSQL versions: 9.3 to 14.
|
Currently supported PostgreSQL versions: 9.3 to 15.
|
||||||
|
|
||||||
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the :ref:`Kubernetes <kubernetes>` chapter of the Patroni documentation.
|
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the :ref:`Kubernetes <kubernetes>` chapter of the Patroni documentation.
|
||||||
|
|
||||||
|
|||||||
+1
-1
@@ -23,7 +23,7 @@ Use ConfigMaps
|
|||||||
In this mode, Patroni will create ConfigMaps instead of Endpoints and store keys inside meta-data of those ConfigMaps.
|
In this mode, Patroni will create ConfigMaps instead of Endpoints and store keys inside meta-data of those ConfigMaps.
|
||||||
Changing the leader takes at least two updates, one to the leader ConfigMap and another to the respective Endpoint.
|
Changing the leader takes at least two updates, one to the leader ConfigMap and another to the respective Endpoint.
|
||||||
|
|
||||||
There are two ways to direct the traffic to the Postgres master:
|
There are two ways to direct the traffic to the Postgres leader:
|
||||||
|
|
||||||
- use the `callback script <https://github.com/zalando/patroni/blob/master/kubernetes/callback.py>`_ provided by Patroni
|
- use the `callback script <https://github.com/zalando/patroni/blob/master/kubernetes/callback.py>`_ provided by Patroni
|
||||||
- configure the Kubernetes Postgres service to use the label selector with the `role_label` (configured in patroni configuration).
|
- configure the Kubernetes Postgres service to use the label selector with the `role_label` (configured in patroni configuration).
|
||||||
|
|||||||
+7
-5
@@ -6,7 +6,7 @@ Pause/Resume mode for the cluster
|
|||||||
The goal
|
The goal
|
||||||
--------
|
--------
|
||||||
|
|
||||||
Under certain circumstances Patroni needs to temporary step down from managing the cluster, while still retaining the cluster state in DCS. Possible use cases are uncommon activities on the cluster, such as major version upgrades or corruption recovery. During those activities nodes are often started and stopped for the reason unknown to Patroni, some nodes can be even temporary promoted, violating the assumption of running only one master. Therefore, Patroni needs to be able to "detach" from the running cluster, implementing an equivalent of the maintenance mode in Pacemaker.
|
Under certain circumstances Patroni needs to temporarily step down from managing the cluster, while still retaining the cluster state in DCS. Possible use cases are uncommon activities on the cluster, such as major version upgrades or corruption recovery. During those activities nodes are often started and stopped for reasons unknown to Patroni, some nodes can be even temporarily promoted, violating the assumption of running only one primary. Therefore, Patroni needs to be able to "detach" from the running cluster, implementing an equivalent of the maintenance mode in Pacemaker.
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
@@ -17,16 +17,18 @@ When Patroni runs in a paused mode, it does not change the state of PostgreSQL,
|
|||||||
|
|
||||||
- For each node, the member key in DCS is updated with the current information about the cluster. This causes Patroni to run read-only queries on a member node if the member is running.
|
- For each node, the member key in DCS is updated with the current information about the cluster. This causes Patroni to run read-only queries on a member node if the member is running.
|
||||||
|
|
||||||
- For the Postgres master with the leader lock Patroni updates the lock. If the node with the leader lock stops being the master (i.e. is demoted manually), Patroni will release the lock instead of promoting the node back.
|
- For the Postgres primary with the leader lock Patroni updates the lock. If the node with the leader lock stops being the primary (i.e. is demoted manually), Patroni will release the lock instead of promoting the node back.
|
||||||
|
|
||||||
- Manual unscheduled restart, reinitialize and manual failover are allowed. Manual failover is only allowed if the node to failover to is specified. In the paused mode, manual failover does not require a running master node.
|
- Manual unscheduled restart, reinitialize and manual failover are allowed. Manual failover is only allowed if the node to failover to is specified. In the paused mode, manual failover does not require a running primary node.
|
||||||
|
|
||||||
- If 'parallel' masters are detected by Patroni, it emits a warning, but does not demote the masters without the leader lock.
|
- If 'parallel' primaries are detected by Patroni, it emits a warning, but does not demote the primary without the leader lock.
|
||||||
|
|
||||||
- If there is no leader lock in the cluster, the running master acquires the lock. If there is more than one master node, then the first master to acquire the lock wins. If there are no masters altogether, Patroni does not try to promote any replicas. There is an exception in this rule: if there is no leader lock because the old master has demoted itself due to the manual promotion, then only the candidate node mentioned in the promotion request may take the leader lock. When the new leader lock is granted (i.e. after promoting a replica manually), Patroni makes sure the replicas that were streaming from the previous leader will switch to the new one.
|
- If there is no leader lock in the cluster, the running primary acquires the lock. If there is more than one primary node, then the first primary to acquire the lock wins. If there are no primary altogether, Patroni does not try to promote any replicas. There is an exception in this rule: if there is no leader lock because the old primary has demoted itself due to the manual promotion, then only the candidate node mentioned in the promotion request may take the leader lock. When the new leader lock is granted (i.e. after promoting a replica manually), Patroni makes sure the replicas that were streaming from the previous leader will switch to the new one.
|
||||||
|
|
||||||
- When Postgres is stopped, Patroni does not try to start it. When Patroni is stopped, it does not try to stop the Postgres instance it is managing.
|
- When Postgres is stopped, Patroni does not try to start it. When Patroni is stopped, it does not try to stop the Postgres instance it is managing.
|
||||||
|
|
||||||
|
- Patroni will not try to remove replication slots that don't represent the other cluster member or are not listed in the configuration of the permanent slots.
|
||||||
|
|
||||||
User guide
|
User guide
|
||||||
----------
|
----------
|
||||||
|
|
||||||
|
|||||||
@@ -3,6 +3,169 @@
|
|||||||
Release notes
|
Release notes
|
||||||
=============
|
=============
|
||||||
|
|
||||||
|
Version 2.1.7
|
||||||
|
-------------
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Fixed little incompatibilities with legacy python modules (Alexander Kukushkin)
|
||||||
|
|
||||||
|
They prevented from building/running Patroni on Debian buster/Ubuntu bionic.
|
||||||
|
|
||||||
|
|
||||||
|
Version 2.1.6
|
||||||
|
-------------
|
||||||
|
|
||||||
|
**Improvements**
|
||||||
|
|
||||||
|
- Fix annoying exceptions on ssl socket shutdown (Alexander Kukushkin)
|
||||||
|
|
||||||
|
The HAProxy is closing connections as soon as it got the HTTP Status code leaving no time for Patroni to properly shutdown SSL connection.
|
||||||
|
|
||||||
|
- Adjust example Dockerfile for arm64 (Polina Bungina)
|
||||||
|
|
||||||
|
Remove explicit ``amd64`` and ``x86_64``, don't remove ``libnss_files.so.*``.
|
||||||
|
|
||||||
|
|
||||||
|
**Security improvements**
|
||||||
|
|
||||||
|
- Enforce ``search_path=pg_catalog`` for non-replication connections (Alexander)
|
||||||
|
|
||||||
|
Since Patroni is heavily relying on superuser connections, we want to protect it from the possible attacks carried out using user-defined functions and/or operators in ``public`` schema with the same name and signature as the corresponding objects in ``pg_catalog``. For that, ``search_path=pg_catalog`` is enforced for all connections created by Patroni (except replication connections).
|
||||||
|
|
||||||
|
- Prevent passwords from being recorded in ``pg_stat_statements`` (Feike Steenbergen)
|
||||||
|
|
||||||
|
It is achieved by setting ``pg_stat_statements.track_utility=off`` when creating users.
|
||||||
|
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Declare ``proxy_address`` as optional (Denis Laxalde)
|
||||||
|
|
||||||
|
As it is effectively a non-required option.
|
||||||
|
|
||||||
|
- Improve behaviour of the insecure option (Alexander)
|
||||||
|
|
||||||
|
Ctl's ``insecure`` option didn't work properly when client certificates were used for REST API requests.
|
||||||
|
|
||||||
|
- Take watchdog configuration from ``bootstrap.dcs`` when the new cluster is bootstrapped (Matt Baker)
|
||||||
|
|
||||||
|
Patroni used to initially configure watchdog with defaults when bootstrapping a new cluster rather than taking configuration used to bootstrap the DCS.
|
||||||
|
|
||||||
|
- Fix the way file extensions are treated while finding executables in WIN32 (Martín Marqués)
|
||||||
|
|
||||||
|
Only add ``.exe`` to a file name if it has no extension yet.
|
||||||
|
|
||||||
|
- Fix Consul TTL setup (Alexander)
|
||||||
|
|
||||||
|
We used ``ttl/2.0`` when setting the value on the HTTPClient, but forgot to multiply the current value by 2 in the class' property. It was resulting in Consul TTL off by twice.
|
||||||
|
|
||||||
|
|
||||||
|
**Removed functionality**
|
||||||
|
|
||||||
|
- Remove ``patronictl configure`` (Polina)
|
||||||
|
|
||||||
|
There is no more need for a separate ``patronictl`` config creation.
|
||||||
|
|
||||||
|
|
||||||
|
Version 2.1.5
|
||||||
|
-------------
|
||||||
|
|
||||||
|
This version enhances compatibility with PostgreSQL 15 and declares Etcd v3 support as production ready. The Patroni on Raft remains in Beta.
|
||||||
|
|
||||||
|
**New features**
|
||||||
|
|
||||||
|
- Improve ``patroni --validate-config`` (Denis Laxalde)
|
||||||
|
|
||||||
|
Exit with code 1 if config is invalid and print errors to stderr.
|
||||||
|
|
||||||
|
- Don't drop replication slots in pause (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Patroni is automatically creating/removing physical replication slots when members are joining/leaving the cluster. In pause slots will no longer be removed.
|
||||||
|
|
||||||
|
- Support the ``HEAD`` request method for monitoring endpoints (Robert Cutajar)
|
||||||
|
|
||||||
|
If used instead of ``GET`` Patroni will return only the HTTP Status Code.
|
||||||
|
|
||||||
|
- Support behave tests on Windows (Alexander)
|
||||||
|
|
||||||
|
Emulate graceful Patroni shutdown (``SIGTERM``) on Windows by introduce the new REST API endpoint ``POST /sigterm``.
|
||||||
|
|
||||||
|
- Introduce ``postgresql.proxy_address`` (Alexander)
|
||||||
|
|
||||||
|
It will be written to the member key in DCS as the ``proxy_url`` and could be used/useful for service discovery.
|
||||||
|
|
||||||
|
|
||||||
|
**Stability improvements**
|
||||||
|
|
||||||
|
- Call ``pg_replication_slot_advance()`` from a thread (Alexander)
|
||||||
|
|
||||||
|
On busy clusters with many logical replication slots the ``pg_replication_slot_advance()`` call was affecting the main HA loop and could result in the member key expiration.
|
||||||
|
|
||||||
|
- Archive possibly missing WALs before calling ``pg_rewind`` on the old primary (Polina Bungina)
|
||||||
|
|
||||||
|
If the primary crashed and was down during considerable time, some WAL files could be missing from archive and from the new primary. There is a chance that ``pg_rewind`` could remove these WAL files from the old primary making it impossible to start it as a standby. By archiving ``ready`` WAL files we not only mitigate this problem but in general improving continues archiving experience.
|
||||||
|
|
||||||
|
- Ignore ``403`` errors when trying to create Kubernetes Service (Nick Hudson, Polina)
|
||||||
|
|
||||||
|
Patroni was spamming logs by unsuccessful attempts to create the service, which in fact could already exist.
|
||||||
|
|
||||||
|
- Improve liveness probe (Alexander)
|
||||||
|
|
||||||
|
The liveness problem will start failing if the heartbeat loop is running longer than `ttl` on the primary or `2*ttl` on the replica. That will allow us to use it as an alternative for :ref:`watchdog <watchdog>` on Kubernetes.
|
||||||
|
|
||||||
|
- Make sure only sync node tries to grab the lock when switchover (Alexander, Polina)
|
||||||
|
|
||||||
|
Previously there was a slim chance that up-to-date async member could become the leader if the manual switchover was performed without specifying the target.
|
||||||
|
|
||||||
|
- Avoid cloning while bootstrap is running (Ants Aasma)
|
||||||
|
|
||||||
|
Do not allow a create replica method that does not require a leader to be triggered while the cluster bootstrap is running.
|
||||||
|
|
||||||
|
- Compatibility with kazoo-2.9.0 (Alexander)
|
||||||
|
|
||||||
|
Depending on python version the ``SequentialThreadingHandler.select()`` method may raise ``TypeError`` and ``IOError`` exceptions if ``select()`` is called on the closed socket.
|
||||||
|
|
||||||
|
- Explicitly shut down SSL connection before socket shutdown (Alexander)
|
||||||
|
|
||||||
|
Not doing it resulted in ``unexpected eof while reading`` errors with OpenSSL 3.0.
|
||||||
|
|
||||||
|
- Compatibility with `prettytable>=2.2.0` (Alexander)
|
||||||
|
|
||||||
|
Due to the internal API changes the cluster name header was shown on the incorrect line.
|
||||||
|
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Handle expired token for Etcd lease_grant (monsterxx03)
|
||||||
|
|
||||||
|
In case of error get the new token and retry request.
|
||||||
|
|
||||||
|
- Fix bug in the ``GET /read-only-sync`` endpoint (Alexander)
|
||||||
|
|
||||||
|
It was introduced in previous release and effectively never worked.
|
||||||
|
|
||||||
|
- Handle the case when data dir storage disappeared (Alexander)
|
||||||
|
|
||||||
|
Patroni is periodically checking that the PGDATA is there and not empty, but in case of issues with storage the ``os.listdir()`` is raising the ``OSError`` exception, breaking the heart-beat loop.
|
||||||
|
|
||||||
|
- Apply ``master_stop_timeout`` when waiting for user backends to close (Alexander)
|
||||||
|
|
||||||
|
Something that looks like user backend could be in fact a background worker (e.g., Citus Maintenance Daemon) that is failing to stop.
|
||||||
|
|
||||||
|
- Accept ``*:<port>`` for ``postgresql.listen`` (Denis)
|
||||||
|
|
||||||
|
The ``patroni --validate-config`` was complaining about it being invalid.
|
||||||
|
|
||||||
|
- Timeouts fixes in Raft (Alexander)
|
||||||
|
|
||||||
|
When Patroni or patronictl are starting they try to get Raft cluster topology from known members. These calls were made without proper timeouts.
|
||||||
|
|
||||||
|
- Forcefully update consul service if token was changed (John A. Lotoski)
|
||||||
|
|
||||||
|
Not doing so results in errors "rpc error making call: rpc error making call: ACL not found".
|
||||||
|
|
||||||
|
|
||||||
Version 2.1.4
|
Version 2.1.4
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
|||||||
@@ -63,7 +63,7 @@ Building replicas
|
|||||||
-----------------
|
-----------------
|
||||||
|
|
||||||
Patroni uses tried and proven ``pg_basebackup`` in order to create new replicas. One downside of it is that it requires
|
Patroni uses tried and proven ``pg_basebackup`` in order to create new replicas. One downside of it is that it requires
|
||||||
a running master node. Another one is the lack of 'on-the-fly' compression for the backup data and no built-in cleanup
|
a running leader node. Another one is the lack of 'on-the-fly' compression for the backup data and no built-in cleanup
|
||||||
for outdated backup files. Some people prefer other backup solutions, such as ``WAL-E``, ``pgBackRest``, ``Barman`` and
|
for outdated backup files. Some people prefer other backup solutions, such as ``WAL-E``, ``pgBackRest``, ``Barman`` and
|
||||||
others, or simply roll their own scripts. In order to accommodate all those use-cases Patroni supports running custom
|
others, or simply roll their own scripts. In order to accommodate all those use-cases Patroni supports running custom
|
||||||
scripts to clone a new replica. Those are configured in the ``postgresql`` configuration block:
|
scripts to clone a new replica. Those are configured in the ``postgresql`` configuration block:
|
||||||
@@ -123,11 +123,11 @@ to execute and any custom parameters that should be passed to that command. All
|
|||||||
--role
|
--role
|
||||||
Always 'replica'
|
Always 'replica'
|
||||||
--connstring
|
--connstring
|
||||||
Connection string to connect to the cluster member to clone from (master or other replica). The user in the
|
Connection string to connect to the cluster member to clone from (primary or other replica). The user in the
|
||||||
connection string can execute SQL and replication protocol commands.
|
connection string can execute SQL and replication protocol commands.
|
||||||
|
|
||||||
A special ``no_master`` parameter, if defined, allows Patroni to call the replica creation method even if there is no
|
A special ``no_master`` parameter, if defined, allows Patroni to call the replica creation method even if there is no
|
||||||
running master or replicas. In that case, an empty string will be passed in a connection string. This is useful for
|
running leader or replicas. In that case, an empty string will be passed in a connection string. This is useful for
|
||||||
restoring the formerly running cluster from the binary backup.
|
restoring the formerly running cluster from the binary backup.
|
||||||
|
|
||||||
A special ``keep_data`` parameter, if defined, will instruct Patroni to not clean PGDATA folder before calling restore.
|
A special ``keep_data`` parameter, if defined, will instruct Patroni to not clean PGDATA folder before calling restore.
|
||||||
@@ -137,7 +137,7 @@ A special ``no_params`` parameter, if defined, restricts passing parameters to c
|
|||||||
A ``basebackup`` method is a special case: it will be used if
|
A ``basebackup`` method is a special case: it will be used if
|
||||||
``create_replica_methods`` is empty, although it is possible
|
``create_replica_methods`` is empty, although it is possible
|
||||||
to list it explicitly among the ``create_replica_methods`` methods. This method initializes a new replica with the
|
to list it explicitly among the ``create_replica_methods`` methods. This method initializes a new replica with the
|
||||||
``pg_basebackup``, the base backup is taken from the master unless there are replicas with ``clonefrom`` tag, in which case one
|
``pg_basebackup``, the base backup is taken from the leader unless there are replicas with ``clonefrom`` tag, in which case one
|
||||||
of such replicas will be used as the origin for pg_basebackup. It works without any configuration; however, it is
|
of such replicas will be used as the origin for pg_basebackup. It works without any configuration; however, it is
|
||||||
possible to specify a ``basebackup`` configuration section. Same rules as with the other method configuration apply,
|
possible to specify a ``basebackup`` configuration section. Same rules as with the other method configuration apply,
|
||||||
namely, only long (with --) options should be specified there. Not all parameters make sense, if you override a connection
|
namely, only long (with --) options should be specified there. Not all parameters make sense, if you override a connection
|
||||||
@@ -176,10 +176,10 @@ Standby cluster
|
|||||||
---------------
|
---------------
|
||||||
|
|
||||||
Another available option is to run a "standby cluster", that contains only of
|
Another available option is to run a "standby cluster", that contains only of
|
||||||
standby nodes replicating from some remote master. This type of clusters has:
|
standby nodes replicating from some remote node. This type of clusters has:
|
||||||
|
|
||||||
* "standby leader", that behaves pretty much like a regular cluster leader,
|
* "standby leader", that behaves pretty much like a regular cluster leader,
|
||||||
except it replicates from a remote master.
|
except it replicates from a remote node.
|
||||||
|
|
||||||
* cascade replicas, that are replicating from standby leader.
|
* cascade replicas, that are replicating from standby leader.
|
||||||
|
|
||||||
@@ -187,6 +187,13 @@ Standby leader holds and updates a leader lock in DCS. If the leader lock
|
|||||||
expires, cascade replicas will perform an election to choose another leader
|
expires, cascade replicas will perform an election to choose another leader
|
||||||
from the standbys.
|
from the standbys.
|
||||||
|
|
||||||
|
There is no further relationship between the standby cluster and the primary
|
||||||
|
cluster it replicates from, in particular, they must not share the same DCS
|
||||||
|
scope if they use the same DCS. They do not know anything else from each other
|
||||||
|
apart from replication information. Also, the standby cluster is not being
|
||||||
|
displayed in ``patronictl list`` or ``patronictl topology`` output on the
|
||||||
|
primary cluster.
|
||||||
|
|
||||||
For the sake of flexibility, you can specify methods of creating a replica and
|
For the sake of flexibility, you can specify methods of creating a replica and
|
||||||
recovery WAL records when a cluster is in the "standby mode" by providing
|
recovery WAL records when a cluster is in the "standby mode" by providing
|
||||||
`create_replica_methods` key in `standby_cluster` section. It is distinct from
|
`create_replica_methods` key in `standby_cluster` section. It is distinct from
|
||||||
@@ -212,4 +219,9 @@ in a patroni configuration:
|
|||||||
Note, that these options will be applied only once during cluster bootstrap,
|
Note, that these options will be applied only once during cluster bootstrap,
|
||||||
and the only way to change them afterwards is through DCS.
|
and the only way to change them afterwards is through DCS.
|
||||||
|
|
||||||
|
Patroni expects to find `postgresql.conf` or `postgresql.conf.backup` in PGDATA
|
||||||
|
of the remote primary and will not start if it does not find it after a
|
||||||
|
basebackup. If the remote primary keeps its `postgresql.conf` elsewhere, it is
|
||||||
|
your responsibility to copy it to PGDATA.
|
||||||
|
|
||||||
If you use replication slots on the standby cluster, you must also create the corresponding replication slot on the primary cluster. It will not be done automatically by the standby cluster implementation. You can use Patroni's permanent replication slots feature on the primary cluster to maintain a replication slot with the same name as ``primary_slot_name``, or its default value if ``primary_slot_name`` is not provided.
|
If you use replication slots on the standby cluster, you must also create the corresponding replication slot on the primary cluster. It will not be done automatically by the standby cluster implementation. You can use Patroni's permanent replication slots feature on the primary cluster to maintain a replication slot with the same name as ``primary_slot_name``, or its default value if ``primary_slot_name`` is not provided.
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ In asynchronous mode the cluster is allowed to lose some committed transactions
|
|||||||
|
|
||||||
The amount of transactions that can be lost is controlled via ``maximum_lag_on_failover`` parameter. Because the primary transaction log position is not sampled in real time, in reality the amount of lost data on failover is worst case bounded by ``maximum_lag_on_failover`` bytes of transaction log plus the amount that is written in the last ``ttl`` seconds (``loop_wait``/2 seconds in the average case). However typical steady state replication delay is well under a second.
|
The amount of transactions that can be lost is controlled via ``maximum_lag_on_failover`` parameter. Because the primary transaction log position is not sampled in real time, in reality the amount of lost data on failover is worst case bounded by ``maximum_lag_on_failover`` bytes of transaction log plus the amount that is written in the last ``ttl`` seconds (``loop_wait``/2 seconds in the average case). However typical steady state replication delay is well under a second.
|
||||||
|
|
||||||
By default, when running leader elections, Patroni does not take into account the current timeline of replicas, what in some cases could be undesirable behavior. You can prevent the node not having the same timeline as a former master become the new leader by changing the value of ``check_timeline`` parameter to ``true``.
|
By default, when running leader elections, Patroni does not take into account the current timeline of replicas, what in some cases could be undesirable behavior. You can prevent the node not having the same timeline as a former primary become the new leader by changing the value of ``check_timeline`` parameter to ``true``.
|
||||||
|
|
||||||
PostgreSQL synchronous replication
|
PostgreSQL synchronous replication
|
||||||
----------------------------------
|
----------------------------------
|
||||||
|
|||||||
+2
-2
@@ -7,7 +7,7 @@ Patroni has a rich REST API, which is used by Patroni itself during the leader r
|
|||||||
|
|
||||||
Health check endpoints
|
Health check endpoints
|
||||||
----------------------
|
----------------------
|
||||||
For all health check ``GET`` requests Patroni returns a JSON document with the status of the node, along with the HTTP status code. If you don't want or don't need the JSON document, you might consider using the ``OPTIONS`` method instead of ``GET``.
|
For all health check ``GET`` requests Patroni returns a JSON document with the status of the node, along with the HTTP status code. If you don't want or don't need the JSON document, you might consider using the ``HEAD`` or ``OPTIONS`` method instead of ``GET``.
|
||||||
|
|
||||||
- The following requests to Patroni REST API will return HTTP status code **200** only when the Patroni node is running as the primary with leader lock:
|
- The following requests to Patroni REST API will return HTTP status code **200** only when the Patroni node is running as the primary with leader lock:
|
||||||
|
|
||||||
@@ -58,7 +58,7 @@ For all health check ``GET`` requests Patroni returns a JSON document with the s
|
|||||||
|
|
||||||
- ``GET /health``: returns HTTP status code **200** only when PostgreSQL is up and running.
|
- ``GET /health``: returns HTTP status code **200** only when PostgreSQL is up and running.
|
||||||
|
|
||||||
- ``GET /liveness``: always returns HTTP status code **200** what only indicates that Patroni is running. Could be used for ``livenessProbe``.
|
- ``GET /liveness``: returns HTTP status code **200** if Patroni heartbeat loop is properly running and **503** if the last run was more than ``ttl`` seconds ago on the primary or ``2*ttl`` on the replica. Could be used for ``livenessProbe``.
|
||||||
|
|
||||||
- ``GET /readiness``: returns HTTP status code **200** when the Patroni node is running as the leader or when PostgreSQL is up and running. The endpoint could be used for ``readinessProbe`` when it is not possible to use Kubernetes endpoints for leader elections (OpenShift).
|
- ``GET /readiness``: returns HTTP status code **200** when the Patroni node is running as the leader or when PostgreSQL is up and running. The endpoint could be used for ``readinessProbe`` when it is not possible to use Kubernetes endpoints for leader elections (OpenShift).
|
||||||
|
|
||||||
|
|||||||
+2
-2
@@ -3,7 +3,7 @@
|
|||||||
Watchdog support
|
Watchdog support
|
||||||
================
|
================
|
||||||
|
|
||||||
Having multiple PostgreSQL servers running as master can result in transactions lost due to diverging timelines. This situation is also called a split-brain problem. To avoid split-brain Patroni needs to ensure PostgreSQL will not accept any transaction commits after leader key expires in the DCS. Under normal circumstances Patroni will try to achieve this by stopping PostgreSQL when leader lock update fails for any reason. However, this may fail to happen due to various reasons:
|
Having multiple PostgreSQL servers running as primary can result in transactions lost due to diverging timelines. This situation is also called a split-brain problem. To avoid split-brain Patroni needs to ensure PostgreSQL will not accept any transaction commits after leader key expires in the DCS. Under normal circumstances Patroni will try to achieve this by stopping PostgreSQL when leader lock update fails for any reason. However, this may fail to happen due to various reasons:
|
||||||
|
|
||||||
- Patroni has crashed due to a bug, out-of-memory condition or by being accidentally killed by a system administrator.
|
- Patroni has crashed due to a bug, out-of-memory condition or by being accidentally killed by a system administrator.
|
||||||
|
|
||||||
@@ -13,7 +13,7 @@ Having multiple PostgreSQL servers running as master can result in transactions
|
|||||||
|
|
||||||
To guarantee correct behavior under these conditions Patroni supports watchdog devices. Watchdog devices are software or hardware mechanisms that will reset the whole system when they do not get a keepalive heartbeat within a specified timeframe. This adds an additional layer of fail safe in case usual Patroni split-brain protection mechanisms fail.
|
To guarantee correct behavior under these conditions Patroni supports watchdog devices. Watchdog devices are software or hardware mechanisms that will reset the whole system when they do not get a keepalive heartbeat within a specified timeframe. This adds an additional layer of fail safe in case usual Patroni split-brain protection mechanisms fail.
|
||||||
|
|
||||||
Patroni will try to activate the watchdog before promoting PostgreSQL to master. If watchdog activation fails and watchdog mode is ``required`` then the node will refuse to become master. When deciding to participate in leader election Patroni will also check that watchdog configuration will allow it to become leader at all. After demoting PostgreSQL (for example due to a manual failover) Patroni will disable the watchdog again. Watchdog will also be disabled while Patroni is in paused state.
|
Patroni will try to activate the watchdog before promoting PostgreSQL to primary. If watchdog activation fails and watchdog mode is ``required`` then the node will refuse to become leader. When deciding to participate in leader election Patroni will also check that watchdog configuration will allow it to become leader at all. After demoting PostgreSQL (for example due to a manual failover) Patroni will disable the watchdog again. Watchdog will also be disabled while Patroni is in paused state.
|
||||||
|
|
||||||
By default Patroni will set up the watchdog to expire 5 seconds before TTL expires. With the default setup of ``loop_wait=10`` and ``ttl=30`` this gives HA loop at least 15 seconds (``ttl`` - ``safety_margin`` - ``loop_wait``) to complete before the system gets forcefully reset. By default accessing DCS is configured to time out after 10 seconds. This means that when DCS is unavailable, for example due to network issues, Patroni and PostgreSQL will have at least 5 seconds (``ttl`` - ``safety_margin`` - ``loop_wait`` - ``retry_timeout``) to come to a state where all client connections are terminated.
|
By default Patroni will set up the watchdog to expire 5 seconds before TTL expires. With the default setup of ``loop_wait=10`` and ``ttl=30`` this gives HA loop at least 15 seconds (``ttl`` - ``safety_margin`` - ``loop_wait``) to complete before the system gets forcefully reset. By default accessing DCS is configured to time out after 10 seconds. This means that when DCS is unavailable, for example due to network issues, Patroni and PostgreSQL will have at least 5 seconds (``ttl`` - ``safety_margin`` - ``loop_wait`` - ``retry_timeout``) to come to a state where all client connections are terminated.
|
||||||
|
|
||||||
|
|||||||
@@ -18,14 +18,14 @@ listen stats
|
|||||||
|
|
||||||
listen master
|
listen master
|
||||||
bind *:5000
|
bind *:5000
|
||||||
option httpchk OPTIONS /master
|
option httpchk HEAD /master
|
||||||
http-check expect status 200
|
http-check expect status 200
|
||||||
default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions
|
default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions
|
||||||
{{range gets "/members/*"}} server {{base .Key}} {{$data := json .Value}}{{base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}} maxconn 100 check port {{index (split (index (split $data.api_url "/") 2) ":") 1}}
|
{{range gets "/members/*"}} server {{base .Key}} {{$data := json .Value}}{{base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}} maxconn 100 check port {{index (split (index (split $data.api_url "/") 2) ":") 1}}
|
||||||
{{end}}
|
{{end}}
|
||||||
listen replicas
|
listen replicas
|
||||||
bind *:5001
|
bind *:5001
|
||||||
option httpchk OPTIONS /replica
|
option httpchk HEAD /replica
|
||||||
http-check expect status 200
|
http-check expect status 200
|
||||||
default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions
|
default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions
|
||||||
{{range gets "/members/*"}} server {{base .Key}} {{$data := json .Value}}{{base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}} maxconn 100 check port {{index (split (index (split $data.api_url "/") 2) ":") 1}}
|
{{range gets "/members/*"}} server {{base .Key}} {{$data := json .Value}}{{base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}} maxconn 100 check port {{index (split (index (split $data.api_url "/") 2) ":") 1}}
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ Group=postgres
|
|||||||
# Read in configuration file if it exists, otherwise proceed
|
# Read in configuration file if it exists, otherwise proceed
|
||||||
EnvironmentFile=-/etc/patroni_env.conf
|
EnvironmentFile=-/etc/patroni_env.conf
|
||||||
|
|
||||||
# the default is the user's home directory, and if you want to change it, you must provide an absolute path.
|
# The default is the user's home directory, and if you want to change it, you must provide an absolute path.
|
||||||
# WorkingDirectory=/home/sameuser
|
# WorkingDirectory=/home/sameuser
|
||||||
|
|
||||||
# Where to send early-startup messages from the server
|
# Where to send early-startup messages from the server
|
||||||
@@ -32,14 +32,14 @@ ExecStart=/bin/patroni /etc/patroni.yml
|
|||||||
# Send HUP to reload from patroni.yml
|
# Send HUP to reload from patroni.yml
|
||||||
ExecReload=/bin/kill -s HUP $MAINPID
|
ExecReload=/bin/kill -s HUP $MAINPID
|
||||||
|
|
||||||
# only kill the patroni process, not it's children, so it will gracefully stop postgres
|
# Only kill the patroni process, not it's children, so it will gracefully stop postgres
|
||||||
KillMode=process
|
KillMode=process
|
||||||
|
|
||||||
# Give a reasonable amount of time for the server to start up/shut down
|
# Give a reasonable amount of time for the server to start up/shut down
|
||||||
TimeoutSec=30
|
TimeoutSec=30
|
||||||
|
|
||||||
# Do not restart the service if it crashes, we want to manually inspect database on failure
|
# Restart the service if it crashed
|
||||||
Restart=no
|
Restart=on-failure
|
||||||
|
|
||||||
[Install]
|
[Install]
|
||||||
WantedBy=multi-user.target
|
WantedBy=multi-user.target
|
||||||
|
|||||||
@@ -5,7 +5,7 @@ Feature: basic replication
|
|||||||
Given I start postgres0
|
Given I start postgres0
|
||||||
Then postgres0 is a leader after 10 seconds
|
Then postgres0 is a leader after 10 seconds
|
||||||
And there is a non empty initialize key in DCS after 15 seconds
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 20, "loop_wait": 2, "synchronous_mode": true}
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 20, "synchronous_mode": true}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
When I start postgres1
|
When I start postgres1
|
||||||
And I configure and start postgres2 with a tag replicatefrom postgres0
|
And I configure and start postgres2 with a tag replicatefrom postgres0
|
||||||
|
|||||||
+125
-31
@@ -2,6 +2,7 @@ import abc
|
|||||||
import datetime
|
import datetime
|
||||||
import os
|
import os
|
||||||
import json
|
import json
|
||||||
|
import re
|
||||||
import shutil
|
import shutil
|
||||||
import signal
|
import signal
|
||||||
import six
|
import six
|
||||||
@@ -14,6 +15,7 @@ import yaml
|
|||||||
|
|
||||||
import patroni.psycopg as psycopg
|
import patroni.psycopg as psycopg
|
||||||
|
|
||||||
|
from patroni.request import PatroniRequest
|
||||||
from six.moves.BaseHTTPServer import BaseHTTPRequestHandler, HTTPServer
|
from six.moves.BaseHTTPServer import BaseHTTPRequestHandler, HTTPServer
|
||||||
|
|
||||||
|
|
||||||
@@ -138,12 +140,24 @@ class PatroniController(AbstractController):
|
|||||||
def _start(self):
|
def _start(self):
|
||||||
if self.watchdog:
|
if self.watchdog:
|
||||||
self.watchdog.start()
|
self.watchdog.start()
|
||||||
|
env = os.environ.copy()
|
||||||
if isinstance(self._context.dcs_ctl, KubernetesController):
|
if isinstance(self._context.dcs_ctl, KubernetesController):
|
||||||
self._context.dcs_ctl.create_pod(self._name[8:], self._scope)
|
self._context.dcs_ctl.create_pod(self._name[8:], self._scope)
|
||||||
os.environ['PATRONI_KUBERNETES_POD_IP'] = '10.0.0.' + self._name[-1]
|
env['PATRONI_KUBERNETES_POD_IP'] = '10.0.0.' + self._name[-1]
|
||||||
return subprocess.Popen([sys.executable, '-m', 'coverage', 'run',
|
if os.name == 'nt':
|
||||||
'--source=patroni', '-p', 'patroni.py', self._config],
|
env['BEHAVE_DEBUG'] = 'true'
|
||||||
|
patroni = subprocess.Popen([sys.executable, '-m', 'coverage', 'run',
|
||||||
|
'--source=patroni', '-p', 'patroni.py', self._config], env=env,
|
||||||
stdout=self._log, stderr=subprocess.STDOUT, cwd=self._work_directory)
|
stdout=self._log, stderr=subprocess.STDOUT, cwd=self._work_directory)
|
||||||
|
if os.name == 'nt':
|
||||||
|
patroni.terminate = self.terminate
|
||||||
|
return patroni
|
||||||
|
|
||||||
|
def terminate(self):
|
||||||
|
try:
|
||||||
|
self._context.request_executor.request('POST', self._restapi_url + '/sigterm')
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
def stop(self, kill=False, timeout=15, postgres=False):
|
def stop(self, kill=False, timeout=15, postgres=False):
|
||||||
if postgres:
|
if postgres:
|
||||||
@@ -164,7 +178,7 @@ class PatroniController(AbstractController):
|
|||||||
patroni_config_name = self.PATRONI_CONFIG.format(name)
|
patroni_config_name = self.PATRONI_CONFIG.format(name)
|
||||||
patroni_config_path = os.path.join(self._output_dir, patroni_config_name)
|
patroni_config_path = os.path.join(self._output_dir, patroni_config_name)
|
||||||
|
|
||||||
with open(patroni_config_name) as f:
|
with open('postgres0.yml') as f:
|
||||||
config = yaml.safe_load(f)
|
config = yaml.safe_load(f)
|
||||||
config.pop('etcd', None)
|
config.pop('etcd', None)
|
||||||
|
|
||||||
@@ -173,20 +187,48 @@ class PatroniController(AbstractController):
|
|||||||
os.environ['RAFT_PORT'] = str(int(raft_port) + 1)
|
os.environ['RAFT_PORT'] = str(int(raft_port) + 1)
|
||||||
config['raft'] = {'data_dir': self._output_dir, 'self_addr': 'localhost:' + os.environ['RAFT_PORT']}
|
config['raft'] = {'data_dir': self._output_dir, 'self_addr': 'localhost:' + os.environ['RAFT_PORT']}
|
||||||
|
|
||||||
host = config['postgresql']['listen'].split(':')[0]
|
host = config['restapi']['listen'].rsplit(':', 1)[0]
|
||||||
|
config['restapi']['listen'] = config['restapi']['connect_address'] = '{0}:{1}'.format(host, 8008+int(name[-1]))
|
||||||
|
|
||||||
|
host = config['postgresql']['listen'].rsplit(':', 1)[0]
|
||||||
config['postgresql']['listen'] = config['postgresql']['connect_address'] = '{0}:{1}'.format(host, self.__PORT)
|
config['postgresql']['listen'] = config['postgresql']['connect_address'] = '{0}:{1}'.format(host, self.__PORT)
|
||||||
|
|
||||||
config['name'] = name
|
config['name'] = name
|
||||||
config['postgresql']['data_dir'] = self._data_dir
|
config['postgresql']['data_dir'] = self._data_dir.replace('\\', '/')
|
||||||
config['postgresql']['basebackup'] = [{'checkpoint': 'fast'}]
|
config['postgresql']['basebackup'] = [{'checkpoint': 'fast'}]
|
||||||
config['postgresql']['use_unix_socket'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets
|
config['postgresql']['use_unix_socket'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets
|
||||||
config['postgresql']['use_unix_socket_repl'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets
|
config['postgresql']['use_unix_socket_repl'] = os.name != 'nt'
|
||||||
config['postgresql']['pgpass'] = os.path.join(tempfile.gettempdir(), 'pgpass_' + name)
|
config['postgresql']['pgpass'] = os.path.join(tempfile.gettempdir(), 'pgpass_' + name).replace('\\', '/')
|
||||||
config['postgresql']['parameters'].update({
|
config['postgresql']['parameters'].update({
|
||||||
'logging_collector': 'on', 'log_destination': 'csvlog', 'log_directory': self._output_dir,
|
'logging_collector': 'on', 'log_destination': 'csvlog',
|
||||||
|
'log_directory': self._output_dir.replace('\\', '/'),
|
||||||
'log_filename': name + '.log', 'log_statement': 'all', 'log_min_messages': 'debug1',
|
'log_filename': name + '.log', 'log_statement': 'all', 'log_min_messages': 'debug1',
|
||||||
'unix_socket_directories': tempfile.gettempdir()})
|
'shared_buffers': '1MB', 'unix_socket_directories': tempfile.gettempdir().replace('\\', '/')})
|
||||||
|
config['postgresql']['pg_hba'] = [
|
||||||
|
'local all all trust',
|
||||||
|
'local replication all trust',
|
||||||
|
'host replication replicator all md5',
|
||||||
|
'host all all all md5'
|
||||||
|
]
|
||||||
|
|
||||||
|
if self._context.postgres_supports_ssl and self._context.certfile:
|
||||||
|
config['postgresql']['parameters'].update({
|
||||||
|
'ssl': 'on',
|
||||||
|
'ssl_ca_file': self._context.certfile.replace('\\', '/'),
|
||||||
|
'ssl_cert_file': self._context.certfile.replace('\\', '/'),
|
||||||
|
'ssl_key_file': self._context.keyfile.replace('\\', '/')
|
||||||
|
})
|
||||||
|
for user in config['postgresql'].get('authentication').keys():
|
||||||
|
config['postgresql'].get('authentication', {}).get(user, {}).update({
|
||||||
|
'sslmode': 'verify-ca',
|
||||||
|
'sslrootcert': self._context.certfile,
|
||||||
|
'sslcert': self._context.certfile,
|
||||||
|
'sslkey': self._context.keyfile
|
||||||
|
})
|
||||||
|
for i, line in enumerate(list(config['postgresql']['pg_hba'])):
|
||||||
|
if line.endswith('md5'):
|
||||||
|
# we want to verify client cert first and than password
|
||||||
|
config['postgresql']['pg_hba'][i] = 'hostssl' + line[4:] + ' clientcert=verify-ca'
|
||||||
|
|
||||||
if 'bootstrap' in config:
|
if 'bootstrap' in config:
|
||||||
config['bootstrap']['post_bootstrap'] = 'psql -w -c "SELECT 1"'
|
config['bootstrap']['post_bootstrap'] = 'psql -w -c "SELECT 1"'
|
||||||
@@ -197,26 +239,28 @@ class PatroniController(AbstractController):
|
|||||||
self.recursive_update(config, custom_config)
|
self.recursive_update(config, custom_config)
|
||||||
|
|
||||||
self.recursive_update(config, {
|
self.recursive_update(config, {
|
||||||
'bootstrap': {'dcs': {'postgresql': {'parameters': {'wal_keep_segments': 100}}}}})
|
'bootstrap': {'dcs': {'loop_wait': 2, 'postgresql': {'parameters': {'wal_keep_segments': 100}}}}})
|
||||||
if config['postgresql'].get('callbacks', {}).get('on_role_change'):
|
if config['postgresql'].get('callbacks', {}).get('on_role_change'):
|
||||||
config['postgresql']['callbacks']['on_role_change'] += ' ' + str(self.__PORT)
|
config['postgresql']['callbacks']['on_role_change'] += ' ' + str(self.__PORT)
|
||||||
|
|
||||||
with open(patroni_config_path, 'w') as f:
|
with open(patroni_config_path, 'w') as f:
|
||||||
yaml.safe_dump(config, f, default_flow_style=False)
|
yaml.safe_dump(config, f, default_flow_style=False)
|
||||||
|
|
||||||
user = config['postgresql'].get('authentication', config['postgresql']).get('superuser', {})
|
self._connkwargs = config['postgresql'].get('authentication', config['postgresql']).get('superuser', {})
|
||||||
self._connkwargs = {k: user[n] for n, k in [('username', 'user'), ('password', 'password')] if n in user}
|
self._connkwargs.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres',
|
||||||
self._connkwargs.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres'})
|
'user': self._connkwargs.pop('username', None)})
|
||||||
|
|
||||||
self._replication = config['postgresql'].get('authentication', config['postgresql']).get('replication', {})
|
self._replication = config['postgresql'].get('authentication', config['postgresql']).get('replication', {})
|
||||||
self._replication.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres'})
|
self._replication.update({'host': host, 'port': self.__PORT, 'user': self._replication.pop('username', None)})
|
||||||
|
self._restapi_url = 'http://{0}'.format(config['restapi']['connect_address'])
|
||||||
|
if self._context.certfile:
|
||||||
|
self._restapi_url = self._restapi_url.replace('http://', 'https://')
|
||||||
|
|
||||||
return patroni_config_path
|
return patroni_config_path
|
||||||
|
|
||||||
def _connection(self):
|
def _connection(self):
|
||||||
if not self._conn or self._conn.closed != 0:
|
if not self._conn or self._conn.closed != 0:
|
||||||
self._conn = psycopg.connect(**self._connkwargs)
|
self._conn = psycopg.connect(**self._connkwargs)
|
||||||
self._conn.autocommit = True
|
|
||||||
return self._conn
|
return self._conn
|
||||||
|
|
||||||
def _cursor(self):
|
def _cursor(self):
|
||||||
@@ -269,7 +313,10 @@ class PatroniController(AbstractController):
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def backup_source(self):
|
def backup_source(self):
|
||||||
return 'postgres://{username}:{password}@{host}:{port}/{dbname}'.format(**self._replication)
|
def escape(value):
|
||||||
|
return re.sub(r'([\'\\ ])', r'\\\1', str(value))
|
||||||
|
|
||||||
|
return ' '.join('{0}={1}'.format(k, escape(v)) for k, v in self._replication.items())
|
||||||
|
|
||||||
def backup(self, dest=os.path.join('data', 'basebackup')):
|
def backup(self, dest=os.path.join('data', 'basebackup')):
|
||||||
subprocess.call(PatroniPoolController.BACKUP_SCRIPT + ['--walmethod=none',
|
subprocess.call(PatroniPoolController.BACKUP_SCRIPT + ['--walmethod=none',
|
||||||
@@ -394,7 +441,7 @@ class AbstractEtcdController(AbstractDcsController):
|
|||||||
self._client_cls = client_cls
|
self._client_cls = client_cls
|
||||||
|
|
||||||
def _start(self):
|
def _start(self):
|
||||||
return subprocess.Popen(["etcd", "--debug", "--data-dir", self._work_directory],
|
return subprocess.Popen(["etcd", "--enable-v2=true", "--data-dir", self._work_directory],
|
||||||
stdout=self._log, stderr=subprocess.STDOUT)
|
stdout=self._log, stderr=subprocess.STDOUT)
|
||||||
|
|
||||||
def _is_running(self):
|
def _is_running(self):
|
||||||
@@ -461,10 +508,10 @@ class KubernetesController(AbstractDcsController):
|
|||||||
self._label_selector = ','.join('{0}={1}'.format(k, v) for k, v in self._labels.items())
|
self._label_selector = ','.join('{0}={1}'.format(k, v) for k, v in self._labels.items())
|
||||||
os.environ['PATRONI_KUBERNETES_LABELS'] = json.dumps(self._labels)
|
os.environ['PATRONI_KUBERNETES_LABELS'] = json.dumps(self._labels)
|
||||||
os.environ['PATRONI_KUBERNETES_USE_ENDPOINTS'] = 'true'
|
os.environ['PATRONI_KUBERNETES_USE_ENDPOINTS'] = 'true'
|
||||||
os.environ['PATRONI_KUBERNETES_BYPASS_API_SERVICE'] = 'true'
|
os.environ.setdefault('PATRONI_KUBERNETES_BYPASS_API_SERVICE', 'true')
|
||||||
|
|
||||||
from patroni.dcs.kubernetes import k8s_client, k8s_config
|
from patroni.dcs.kubernetes import k8s_client, k8s_config
|
||||||
k8s_config.load_kube_config(context='local')
|
k8s_config.load_kube_config(context=os.environ.setdefault('PATRONI_KUBERNETES_CONTEXT', 'kind-kind'))
|
||||||
self._client = k8s_client
|
self._client = k8s_client
|
||||||
self._api = self._client.CoreV1Api()
|
self._api = self._client.CoreV1Api()
|
||||||
|
|
||||||
@@ -626,15 +673,17 @@ class RaftController(AbstractDcsController):
|
|||||||
self.start()
|
self.start()
|
||||||
|
|
||||||
ready_event = threading.Event()
|
ready_event = threading.Event()
|
||||||
self._raft = KVStoreTTL(ready_event.set, None, None, partner_addrs=[self.CONTROLLER_ADDR], password=self.PASSWORD)
|
self._raft = KVStoreTTL(ready_event.set, None, None,
|
||||||
|
partner_addrs=[self.CONTROLLER_ADDR], password=self.PASSWORD)
|
||||||
self._raft.startAutoTick()
|
self._raft.startAutoTick()
|
||||||
ready_event.wait()
|
ready_event.wait()
|
||||||
|
|
||||||
|
|
||||||
class PatroniPoolController(object):
|
class PatroniPoolController(object):
|
||||||
|
|
||||||
BACKUP_SCRIPT = [sys.executable, 'features/backup_create.py']
|
PYTHON = sys.executable.replace('\\', '/')
|
||||||
ARCHIVE_RESTORE_SCRIPT = ' '.join((sys.executable, os.path.abspath('features/archive-restore.py')))
|
BACKUP_SCRIPT = [PYTHON, 'features/backup_create.py']
|
||||||
|
ARCHIVE_RESTORE_SCRIPT = ' '.join((PYTHON, os.path.abspath('features/archive-restore.py')))
|
||||||
|
|
||||||
def __init__(self, context):
|
def __init__(self, context):
|
||||||
self._context = context
|
self._context = context
|
||||||
@@ -643,8 +692,17 @@ class PatroniPoolController(object):
|
|||||||
self._patroni_path = None
|
self._patroni_path = None
|
||||||
self._processes = {}
|
self._processes = {}
|
||||||
self.create_and_set_output_directory('')
|
self.create_and_set_output_directory('')
|
||||||
|
self._check_postgres_ssl()
|
||||||
self.known_dcs = {subclass.name(): subclass for subclass in AbstractDcsController.get_subclasses()}
|
self.known_dcs = {subclass.name(): subclass for subclass in AbstractDcsController.get_subclasses()}
|
||||||
|
|
||||||
|
def _check_postgres_ssl(self):
|
||||||
|
try:
|
||||||
|
subprocess.check_output(['postgres', '-D', os.devnull, '-c', 'ssl=on'], stderr=subprocess.STDOUT)
|
||||||
|
raise Exception # this one should never happen because the previous line will always raise and exception
|
||||||
|
except Exception as e:
|
||||||
|
self._context.postgres_supports_ssl = isinstance(e, subprocess.CalledProcessError)\
|
||||||
|
and 'SSL is not supported by this build' not in e.output.decode()
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def patroni_path(self):
|
def patroni_path(self):
|
||||||
if self._patroni_path is None:
|
if self._patroni_path is None:
|
||||||
@@ -695,7 +753,8 @@ class PatroniPoolController(object):
|
|||||||
'bootstrap': {
|
'bootstrap': {
|
||||||
'method': 'pg_basebackup',
|
'method': 'pg_basebackup',
|
||||||
'pg_basebackup': {
|
'pg_basebackup': {
|
||||||
'command': " ".join(self.BACKUP_SCRIPT) + ' --walmethod=stream --dbname=' + f.backup_source
|
'command': " ".join(self.BACKUP_SCRIPT +
|
||||||
|
['--walmethod=stream', '--dbname="{0}"'.format(f.backup_source)])
|
||||||
},
|
},
|
||||||
'dcs': {
|
'dcs': {
|
||||||
'postgresql': {
|
'postgresql': {
|
||||||
@@ -710,7 +769,7 @@ class PatroniPoolController(object):
|
|||||||
'archive_mode': 'on',
|
'archive_mode': 'on',
|
||||||
'archive_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode archive ' +
|
'archive_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode archive ' +
|
||||||
'--dirname {} --filename %f --pathname %p').format(
|
'--dirname {} --filename %f --pathname %p').format(
|
||||||
os.path.join(self.patroni_path, 'data', 'wal_archive'))
|
os.path.join(self.patroni_path, 'data', 'wal_archive').replace('\\', '/'))
|
||||||
},
|
},
|
||||||
'authentication': {
|
'authentication': {
|
||||||
'superuser': {'password': 'zalando1'},
|
'superuser': {'password': 'zalando1'},
|
||||||
@@ -726,14 +785,14 @@ class PatroniPoolController(object):
|
|||||||
'bootstrap': {
|
'bootstrap': {
|
||||||
'method': 'backup_restore',
|
'method': 'backup_restore',
|
||||||
'backup_restore': {
|
'backup_restore': {
|
||||||
'command': (sys.executable + ' features/backup_restore.py --sourcedir=' +
|
'command': (self.PYTHON + ' features/backup_restore.py --sourcedir=' +
|
||||||
os.path.join(self.patroni_path, 'data', 'basebackup')),
|
os.path.join(self.patroni_path, 'data', 'basebackup').replace('\\', '/')),
|
||||||
'recovery_conf': {
|
'recovery_conf': {
|
||||||
'recovery_target_action': 'promote',
|
'recovery_target_action': 'promote',
|
||||||
'recovery_target_timeline': 'latest',
|
'recovery_target_timeline': 'latest',
|
||||||
'restore_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode restore ' +
|
'restore_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode restore ' +
|
||||||
'--dirname {} --filename %f --pathname %p').format(
|
'--dirname {} --filename %f --pathname %p').format(
|
||||||
os.path.join(self.patroni_path, 'data', 'wal_archive'))
|
os.path.join(self.patroni_path, 'data', 'wal_archive').replace('\\', '/'))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
@@ -872,10 +931,32 @@ class WatchdogMonitor(object):
|
|||||||
|
|
||||||
# actions to execute on start/stop of the tests and before running individual features
|
# actions to execute on start/stop of the tests and before running individual features
|
||||||
def before_all(context):
|
def before_all(context):
|
||||||
os.environ.update({'PATRONI_RESTAPI_USERNAME': 'username', 'PATRONI_RESTAPI_PASSWORD': 'password'})
|
context.ci = os.name == 'nt' or\
|
||||||
context.ci = any(a in os.environ for a in ('TRAVIS_BUILD_NUMBER', 'BUILD_NUMBER', 'GITHUB_ACTIONS'))
|
any(a in os.environ for a in ('TRAVIS_BUILD_NUMBER', 'BUILD_NUMBER', 'GITHUB_ACTIONS'))
|
||||||
context.timeout_multiplier = 5 if context.ci else 1 # MacOS sometimes is VERY slow
|
context.timeout_multiplier = 5 if context.ci else 1 # MacOS sometimes is VERY slow
|
||||||
context.pctl = PatroniPoolController(context)
|
context.pctl = PatroniPoolController(context)
|
||||||
|
|
||||||
|
context.keyfile = os.path.join(context.pctl.output_dir, 'patroni.key')
|
||||||
|
context.certfile = os.path.join(context.pctl.output_dir, 'patroni.crt')
|
||||||
|
try:
|
||||||
|
with open(os.devnull, 'w') as null:
|
||||||
|
ret = subprocess.call(['openssl', 'req', '-nodes', '-new', '-x509', '-subj', '/CN=batman.patroni',
|
||||||
|
'-keyout', context.keyfile, '-out', context.certfile], stdout=null, stderr=null)
|
||||||
|
if ret != 0:
|
||||||
|
raise Exception
|
||||||
|
except Exception:
|
||||||
|
context.keyfile = context.certfile = None
|
||||||
|
|
||||||
|
os.environ.update({'PATRONI_RESTAPI_USERNAME': 'username', 'PATRONI_RESTAPI_PASSWORD': 'password'})
|
||||||
|
ctl = {'auth': os.environ['PATRONI_RESTAPI_USERNAME'] + ':' + os.environ['PATRONI_RESTAPI_PASSWORD']}
|
||||||
|
if context.certfile:
|
||||||
|
os.environ.update({'PATRONI_RESTAPI_CAFILE': context.certfile,
|
||||||
|
'PATRONI_RESTAPI_CERTFILE': context.certfile,
|
||||||
|
'PATRONI_RESTAPI_KEYFILE': context.keyfile,
|
||||||
|
'PATRONI_RESTAPI_VERIFY_CLIENT': 'required',
|
||||||
|
'PATRONI_CTL_INSECURE': 'on'})
|
||||||
|
ctl.update({'cacert': context.certfile, 'certfile': context.certfile, 'keyfile': context.keyfile})
|
||||||
|
context.request_executor = PatroniRequest({'ctl': ctl}, True)
|
||||||
context.dcs_ctl = context.pctl.known_dcs[context.pctl.dcs](context)
|
context.dcs_ctl = context.pctl.known_dcs[context.pctl.dcs](context)
|
||||||
context.dcs_ctl.start()
|
context.dcs_ctl.start()
|
||||||
try:
|
try:
|
||||||
@@ -893,13 +974,26 @@ def after_all(context):
|
|||||||
|
|
||||||
def before_feature(context, feature):
|
def before_feature(context, feature):
|
||||||
""" create per-feature output directory to collect Patroni and PostgreSQL logs """
|
""" create per-feature output directory to collect Patroni and PostgreSQL logs """
|
||||||
|
if feature.name == 'watchdog' and os.name == 'nt':
|
||||||
|
feature.skip("Watchdog isn't supported on Windows")
|
||||||
|
else:
|
||||||
context.pctl.create_and_set_output_directory(feature.name)
|
context.pctl.create_and_set_output_directory(feature.name)
|
||||||
|
|
||||||
|
|
||||||
def after_feature(context, feature):
|
def after_feature(context, feature):
|
||||||
""" stop all Patronis, remove their data directory and cleanup the keys in etcd """
|
""" stop all Patronis, remove their data directory and cleanup the keys in etcd """
|
||||||
context.pctl.stop_all()
|
context.pctl.stop_all()
|
||||||
shutil.rmtree(os.path.join(context.pctl.patroni_path, 'data'))
|
data = os.path.join(context.pctl.patroni_path, 'data')
|
||||||
|
if os.path.exists(data):
|
||||||
|
shutil.rmtree(data)
|
||||||
context.dcs_ctl.cleanup_service_tree()
|
context.dcs_ctl.cleanup_service_tree()
|
||||||
if feature.status == 'failed':
|
if feature.status == 'failed':
|
||||||
shutil.copytree(context.pctl.output_dir, context.pctl.output_dir + '_failed')
|
shutil.copytree(context.pctl.output_dir, context.pctl.output_dir + '_failed')
|
||||||
|
|
||||||
|
|
||||||
|
def before_scenario(context, scenario):
|
||||||
|
if 'slot-advance' in scenario.effective_tags:
|
||||||
|
for p in context.pctl._processes.values():
|
||||||
|
if p._conn and p._conn.server_version < 110000:
|
||||||
|
scenario.skip('pg_replication_slot_advance() is not supported on {0}'.format(p._conn.server_version))
|
||||||
|
break
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ Feature: ignored slots
|
|||||||
Given I start postgres1
|
Given I start postgres1
|
||||||
Then postgres1 is a leader after 10 seconds
|
Then postgres1 is a leader after 10 seconds
|
||||||
And there is a non empty initialize key in DCS after 15 seconds
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8009/config with {"loop_wait": 2, "ignore_slots": [{"name": "unmanaged_slot_0", "database": "postgres", "plugin": "test_decoding", "type": "logical"}, {"name": "unmanaged_slot_1", "database": "postgres", "plugin": "test_decoding"}, {"name": "unmanaged_slot_2", "database": "postgres"}, {"name": "unmanaged_slot_3"}], "postgresql": {"parameters": {"wal_level": "logical"}}}
|
When I issue a PATCH request to http://127.0.0.1:8009/config with {"ignore_slots": [{"name": "unmanaged_slot_0", "database": "postgres", "plugin": "test_decoding", "type": "logical"}, {"name": "unmanaged_slot_1", "database": "postgres", "plugin": "test_decoding"}, {"name": "unmanaged_slot_2", "database": "postgres"}, {"name": "unmanaged_slot_3"}], "postgresql": {"parameters": {"wal_level": "logical"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And Response on GET http://127.0.0.1:8009/config contains ignore_slots after 10 seconds
|
And Response on GET http://127.0.0.1:8009/config contains ignore_slots after 10 seconds
|
||||||
# Make sure the wal_level has been changed.
|
# Make sure the wal_level has been changed.
|
||||||
|
|||||||
@@ -35,13 +35,13 @@ Scenario: check local configuration reload
|
|||||||
Then I receive a response code 202
|
Then I receive a response code 202
|
||||||
|
|
||||||
Scenario: check dynamic configuration change via DCS
|
Scenario: check dynamic configuration change via DCS
|
||||||
Given I run patronictl.py edit-config -s 'ttl=10' -s 'loop_wait=2' -p 'max_connections=101' --force batman
|
Given I run patronictl.py edit-config -s 'ttl=10' -p 'max_connections=101' --force batman
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And I receive a response output "+loop_wait: 2"
|
And I receive a response output "+ttl: 10"
|
||||||
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 11 seconds
|
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 11 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8008/config
|
When I issue a GET request to http://127.0.0.1:8008/config
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I receive a response loop_wait 2
|
And I receive a response ttl 10
|
||||||
When I issue a GET request to http://127.0.0.1:8008/patroni
|
When I issue a GET request to http://127.0.0.1:8008/patroni
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I receive a response tags {'new_tag': 'new_value'}
|
And I receive a response tags {'new_tag': 'new_value'}
|
||||||
@@ -109,7 +109,7 @@ Scenario: check the scheduled switchover
|
|||||||
And I receive a response output "Can't schedule switchover in the paused state"
|
And I receive a response output "Can't schedule switchover in the paused state"
|
||||||
When I run patronictl.py resume batman
|
When I run patronictl.py resume batman
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 5 seconds
|
Given I issue a scheduled switchover from postgres1 to postgres0 in 10 seconds
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And postgres0 is a leader after 20 seconds
|
And postgres0 is a leader after 20 seconds
|
||||||
And postgres0 role is the primary after 10 seconds
|
And postgres0 role is the primary after 10 seconds
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ Feature: standby cluster
|
|||||||
Given I start postgres1
|
Given I start postgres1
|
||||||
Then postgres1 is a leader after 10 seconds
|
Then postgres1 is a leader after 10 seconds
|
||||||
And there is a non empty initialize key in DCS after 15 seconds
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8009/config with {"loop_wait": 2, "slots": {"pm_1": {"type": "physical"}}, "postgresql": {"parameters": {"wal_level": "logical"}}}
|
When I issue a PATCH request to http://127.0.0.1:8009/config with {"slots": {"pm_1": {"type": "physical"}}, "postgresql": {"parameters": {"wal_level": "logical"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And Response on GET http://127.0.0.1:8009/config contains slots after 10 seconds
|
And Response on GET http://127.0.0.1:8009/config contains slots after 10 seconds
|
||||||
And I sleep for 3 seconds
|
And I sleep for 3 seconds
|
||||||
@@ -14,7 +14,7 @@ Feature: standby cluster
|
|||||||
Then "members/postgres0" key in DCS has state=running after 10 seconds
|
Then "members/postgres0" key in DCS has state=running after 10 seconds
|
||||||
And replication works from postgres1 to postgres0 after 15 seconds
|
And replication works from postgres1 to postgres0 after 15 seconds
|
||||||
|
|
||||||
@skip
|
@slot-advance
|
||||||
Scenario: check permanent logical slots are synced to the replica
|
Scenario: check permanent logical slots are synced to the replica
|
||||||
Given I run patronictl.py restart batman postgres1 --force
|
Given I run patronictl.py restart batman postgres1 --force
|
||||||
Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds
|
Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ def stop_postgres(context, name):
|
|||||||
def add_table(context, table_name, pg_name):
|
def add_table(context, table_name, pg_name):
|
||||||
# parse the configuration file and get the port
|
# parse the configuration file and get the port
|
||||||
try:
|
try:
|
||||||
context.pctl.query(pg_name, "CREATE TABLE {0}()".format(table_name))
|
context.pctl.query(pg_name, "CREATE TABLE public.{0}()".format(table_name))
|
||||||
except pg.Error as e:
|
except pg.Error as e:
|
||||||
assert False, "Error creating table {0} on {1}: {2}".format(table_name, pg_name, e)
|
assert False, "Error creating table {0} on {1}: {2}".format(table_name, pg_name, e)
|
||||||
|
|
||||||
@@ -37,9 +37,9 @@ def add_table(context, table_name, pg_name):
|
|||||||
def toggle_wal_replay(context, action, pg_name):
|
def toggle_wal_replay(context, action, pg_name):
|
||||||
# pause or resume the wal replay process
|
# pause or resume the wal replay process
|
||||||
try:
|
try:
|
||||||
version = context.pctl.query(pg_name, "select pg_catalog.pg_read_file('PG_VERSION', 0, 2)").fetchone()
|
version = context.pctl.query(pg_name, "SHOW server_version_num").fetchone()[0]
|
||||||
wal = version and version[0] and int(version[0].split('.')[0]) < 10 and "xlog" or "wal"
|
wal_name = 'xlog' if int(version)/10000 < 10 else 'wal'
|
||||||
context.pctl.query(pg_name, "SELECT pg_{0}_replay_{1}()".format(wal, action))
|
context.pctl.query(pg_name, "SELECT pg_{0}_replay_{1}()".format(wal_name, action))
|
||||||
except pg.Error as e:
|
except pg.Error as e:
|
||||||
assert False, "Error during {0} wal recovery on {1}: {2}".format(action, pg_name, e)
|
assert False, "Error during {0} wal recovery on {1}: {2}".format(action, pg_name, e)
|
||||||
|
|
||||||
@@ -48,9 +48,9 @@ def toggle_wal_replay(context, action, pg_name):
|
|||||||
def crdr_mytest(context, action, pg_name):
|
def crdr_mytest(context, action, pg_name):
|
||||||
try:
|
try:
|
||||||
if (action == "create"):
|
if (action == "create"):
|
||||||
context.pctl.query(pg_name, "create table if not exists mytest(id Numeric)")
|
context.pctl.query(pg_name, "create table if not exists public.mytest(id numeric)")
|
||||||
else:
|
else:
|
||||||
context.pctl.query(pg_name, "drop table if exists mytest")
|
context.pctl.query(pg_name, "drop table if exists public.mytest")
|
||||||
except pg.Error as e:
|
except pg.Error as e:
|
||||||
assert False, "Error {0} table mytest on {1}: {2}".format(action, pg_name, e)
|
assert False, "Error {0} table mytest on {1}: {2}".format(action, pg_name, e)
|
||||||
|
|
||||||
@@ -59,7 +59,7 @@ def crdr_mytest(context, action, pg_name):
|
|||||||
def initiate_load(context, pg_name):
|
def initiate_load(context, pg_name):
|
||||||
# perform dummy load
|
# perform dummy load
|
||||||
try:
|
try:
|
||||||
context.pctl.query(pg_name, "begin; insert into mytest select r::numeric from generate_series(1, 350000) r; commit;")
|
context.pctl.query(pg_name, "insert into public.mytest select r::numeric from generate_series(1, 350000) r")
|
||||||
except pg.Error as e:
|
except pg.Error as e:
|
||||||
assert False, "Error loading test data on {0}: {1}".format(pg_name, e)
|
assert False, "Error loading test data on {0}: {1}".format(pg_name, e)
|
||||||
|
|
||||||
@@ -68,7 +68,7 @@ def initiate_load(context, pg_name):
|
|||||||
def table_is_present_on(context, table_name, pg_name, max_replication_delay):
|
def table_is_present_on(context, table_name, pg_name, max_replication_delay):
|
||||||
max_replication_delay *= context.timeout_multiplier
|
max_replication_delay *= context.timeout_multiplier
|
||||||
for _ in range(int(max_replication_delay)):
|
for _ in range(int(max_replication_delay)):
|
||||||
if context.pctl.query(pg_name, "SELECT 1 FROM {0}".format(table_name), fail_ok=True) is not None:
|
if context.pctl.query(pg_name, "SELECT 1 FROM public.{0}".format(table_name), fail_ok=True) is not None:
|
||||||
break
|
break
|
||||||
sleep(1)
|
sleep(1)
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -1,5 +1,4 @@
|
|||||||
import json
|
import json
|
||||||
import os
|
|
||||||
import parse
|
import parse
|
||||||
import shlex
|
import shlex
|
||||||
import subprocess
|
import subprocess
|
||||||
@@ -10,10 +9,8 @@ import yaml
|
|||||||
from behave import register_type, step, then
|
from behave import register_type, step, then
|
||||||
from dateutil import tz
|
from dateutil import tz
|
||||||
from datetime import datetime, timedelta
|
from datetime import datetime, timedelta
|
||||||
from patroni.request import PatroniRequest
|
|
||||||
|
|
||||||
tzutc = tz.tzutc()
|
tzutc = tz.tzutc()
|
||||||
request_executor = PatroniRequest({'ctl': {'auth': 'username:password'}})
|
|
||||||
|
|
||||||
|
|
||||||
@parse.with_pattern(r'https?://(?:\w|\.|:|/)+')
|
@parse.with_pattern(r'https?://(?:\w|\.|:|/)+')
|
||||||
@@ -73,11 +70,13 @@ def do_post_empty(context, url):
|
|||||||
|
|
||||||
@step('I issue a {request_method:w} request to {url:url} with {data}')
|
@step('I issue a {request_method:w} request to {url:url} with {data}')
|
||||||
def do_request(context, request_method, url, data):
|
def do_request(context, request_method, url, data):
|
||||||
|
if context.certfile:
|
||||||
|
url = url.replace('http://', 'https://')
|
||||||
data = data and json.loads(data)
|
data = data and json.loads(data)
|
||||||
try:
|
try:
|
||||||
r = request_executor.request(request_method, url, data)
|
r = context.request_executor.request(request_method, url, data)
|
||||||
if request_method == 'PATCH' and r.status == 409:
|
if request_method == 'PATCH' and r.status == 409:
|
||||||
r = request_executor.request(request_method, url, data)
|
r = context.request_executor.request(request_method, url, data)
|
||||||
except Exception:
|
except Exception:
|
||||||
context.status_code = context.response = None
|
context.status_code = context.response = None
|
||||||
else:
|
else:
|
||||||
@@ -88,10 +87,7 @@ def do_request(context, request_method, url, data):
|
|||||||
def do_run(context, cmd):
|
def do_run(context, cmd):
|
||||||
cmd = [sys.executable, '-m', 'coverage', 'run', '--source=patroni', '-p'] + shlex.split(cmd)
|
cmd = [sys.executable, '-m', 'coverage', 'run', '--source=patroni', '-p'] + shlex.split(cmd)
|
||||||
try:
|
try:
|
||||||
# XXX: Dirty hack! We need to take name/passwd from the config!
|
response = subprocess.check_output(cmd, stderr=subprocess.STDOUT)
|
||||||
env = os.environ.copy()
|
|
||||||
env.update({'PATRONI_RESTAPI_USERNAME': 'username', 'PATRONI_RESTAPI_PASSWORD': 'password'})
|
|
||||||
response = subprocess.check_output(cmd, stderr=subprocess.STDOUT, env=env)
|
|
||||||
context.status_code = 0
|
context.status_code = 0
|
||||||
except subprocess.CalledProcessError as e:
|
except subprocess.CalledProcessError as e:
|
||||||
response = e.output
|
response = e.output
|
||||||
@@ -137,9 +133,11 @@ def add_tag_to_config(context, tag, value, pg_name):
|
|||||||
|
|
||||||
@then('Response on GET {url} contains {value} after {timeout:d} seconds')
|
@then('Response on GET {url} contains {value} after {timeout:d} seconds')
|
||||||
def check_http_response(context, url, value, timeout, negate=False):
|
def check_http_response(context, url, value, timeout, negate=False):
|
||||||
|
if context.certfile:
|
||||||
|
url = url.replace('http://', 'https://')
|
||||||
timeout *= context.timeout_multiplier
|
timeout *= context.timeout_multiplier
|
||||||
for _ in range(int(timeout)):
|
for _ in range(int(timeout)):
|
||||||
r = request_executor.request('GET', url)
|
r = context.request_executor.request('GET', url)
|
||||||
if (value in r.data.decode('utf-8')) != negate:
|
if (value in r.data.decode('utf-8')) != negate:
|
||||||
break
|
break
|
||||||
time.sleep(1)
|
time.sleep(1)
|
||||||
|
|||||||
@@ -1,17 +1,12 @@
|
|||||||
import os
|
import os
|
||||||
import sys
|
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from behave import step
|
from behave import step
|
||||||
|
|
||||||
|
|
||||||
select_replication_query = """
|
def callbacks(context, name):
|
||||||
SELECT * FROM pg_catalog.pg_stat_replication
|
return {c: '{0} features/callback2.py {1}'.format(context.pctl.PYTHON, name)
|
||||||
WHERE application_name = '{0}'
|
for c in ('on_start', 'on_stop', 'on_restart', 'on_role_change')}
|
||||||
"""
|
|
||||||
|
|
||||||
executable = sys.executable if os.name != 'nt' else sys.executable.replace('\\', '/')
|
|
||||||
callback = executable + " features/callback2.py "
|
|
||||||
|
|
||||||
|
|
||||||
@step('I start {name:w} in a cluster {cluster_name:w}')
|
@step('I start {name:w} in a cluster {cluster_name:w}')
|
||||||
@@ -19,10 +14,10 @@ def start_patroni(context, name, cluster_name):
|
|||||||
return context.pctl.start(name, custom_config={
|
return context.pctl.start(name, custom_config={
|
||||||
"scope": cluster_name,
|
"scope": cluster_name,
|
||||||
"postgresql": {
|
"postgresql": {
|
||||||
"callbacks": {c: callback + name for c in ('on_start', 'on_stop', 'on_restart', 'on_role_change')},
|
"callbacks": callbacks(context, name),
|
||||||
"backup_restore": {
|
"backup_restore": {
|
||||||
"command": (executable + " features/backup_restore.py --sourcedir=" +
|
"command": (context.pctl.PYTHON + " features/backup_restore.py --sourcedir=" +
|
||||||
os.path.join(context.pctl.patroni_path, 'data', 'basebackup'))}
|
os.path.join(context.pctl.patroni_path, 'data', 'basebackup').replace('\\', '/'))}
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
|
|
||||||
@@ -49,7 +44,7 @@ def start_patroni_standby_cluster(context, name, cluster_name, name2):
|
|||||||
}
|
}
|
||||||
},
|
},
|
||||||
"postgresql": {
|
"postgresql": {
|
||||||
"callbacks": {c: callback + name for c in ('on_start', 'on_stop', 'on_restart', 'on_role_change')}
|
"callbacks": callbacks(context, name)
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
return context.pctl.start(name)
|
return context.pctl.start(name)
|
||||||
@@ -62,7 +57,7 @@ def check_replication_status(context, pg_name1, pg_name2, timeout):
|
|||||||
while time.time() < bound_time:
|
while time.time() < bound_time:
|
||||||
cur = context.pctl.query(
|
cur = context.pctl.query(
|
||||||
pg_name2,
|
pg_name2,
|
||||||
select_replication_query.format(pg_name1),
|
"SELECT * FROM pg_catalog.pg_stat_replication WHERE application_name = '{0}'".format(pg_name1),
|
||||||
fail_ok=True
|
fail_ok=True
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -15,7 +15,7 @@ def polling_loop(timeout, interval=1):
|
|||||||
|
|
||||||
@step('I start {name:w} with watchdog')
|
@step('I start {name:w} with watchdog')
|
||||||
def start_patroni_with_watchdog(context, name):
|
def start_patroni_with_watchdog(context, name):
|
||||||
return context.pctl.start(name, custom_config={'watchdog': True})
|
return context.pctl.start(name, custom_config={'watchdog': True, 'bootstrap': {'dcs': {'ttl': 20}}})
|
||||||
|
|
||||||
|
|
||||||
@step('{name:w} watchdog has been pinged after {timeout:d} seconds')
|
@step('{name:w} watchdog has been pinged after {timeout:d} seconds')
|
||||||
@@ -31,6 +31,11 @@ def watchdog_was_closed(context, name):
|
|||||||
assert context.pctl.get_watchdog(name).was_closed
|
assert context.pctl.get_watchdog(name).was_closed
|
||||||
|
|
||||||
|
|
||||||
|
@step('{name:w} watchdog has a {timeout:d} second timeout')
|
||||||
|
def watchdog_has_timeout(context, name, timeout):
|
||||||
|
assert context.pctl.get_watchdog(name).timeout == timeout
|
||||||
|
|
||||||
|
|
||||||
@step('I reset {name:w} watchdog state')
|
@step('I reset {name:w} watchdog state')
|
||||||
def watchdog_reset_pinged(context, name):
|
def watchdog_reset_pinged(context, name):
|
||||||
context.pctl.get_watchdog(name).reset()
|
context.pctl.get_watchdog(name).reset()
|
||||||
|
|||||||
@@ -6,6 +6,14 @@ Feature: watchdog
|
|||||||
Then postgres0 is a leader after 10 seconds
|
Then postgres0 is a leader after 10 seconds
|
||||||
And postgres0 role is the primary after 10 seconds
|
And postgres0 role is the primary after 10 seconds
|
||||||
And postgres0 watchdog has been pinged after 10 seconds
|
And postgres0 watchdog has been pinged after 10 seconds
|
||||||
|
And postgres0 watchdog has a 15 second timeout
|
||||||
|
|
||||||
|
Scenario: watchdog is reconfigured after global ttl changed
|
||||||
|
Given I run patronictl.py edit-config batman -s ttl=30 --force
|
||||||
|
Then I receive a response returncode 0
|
||||||
|
And I receive a response output "+ttl: 30"
|
||||||
|
When I sleep for 4 seconds
|
||||||
|
Then postgres0 watchdog has a 25 second timeout
|
||||||
|
|
||||||
Scenario: watchdog is disabled during pause
|
Scenario: watchdog is disabled during pause
|
||||||
Given I run patronictl.py pause batman
|
Given I run patronictl.py pause batman
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
FROM postgres:11
|
FROM postgres:15
|
||||||
MAINTAINER Alexander Kukushkin <alexander.kukushkin@zalando.de>
|
LABEL maintainer="Alexander Kukushkin <akukushkin@microsoft.com>"
|
||||||
|
|
||||||
RUN export DEBIAN_FRONTEND=noninteractive \
|
RUN export DEBIAN_FRONTEND=noninteractive \
|
||||||
&& echo 'APT::Install-Recommends "0";\nAPT::Install-Suggests "0";' > /etc/apt/apt.conf.d/01norecommend \
|
&& echo 'APT::Install-Recommends "0";\nAPT::Install-Suggests "0";' > /etc/apt/apt.conf.d/01norecommend \
|
||||||
|
|||||||
@@ -47,6 +47,7 @@ class Patroni(AbstractPatroniDaemon):
|
|||||||
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
||||||
if self.config.set_dynamic_configuration(self.config['bootstrap']['dcs']):
|
if self.config.set_dynamic_configuration(self.config['bootstrap']['dcs']):
|
||||||
self.dcs.reload_config(self.config)
|
self.dcs.reload_config(self.config)
|
||||||
|
self.watchdog.reload_config(self.config)
|
||||||
break
|
break
|
||||||
except DCSError:
|
except DCSError:
|
||||||
logger.warning('Can not get cluster from dcs')
|
logger.warning('Can not get cluster from dcs')
|
||||||
|
|||||||
+34
-22
@@ -38,6 +38,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
self.log_request(status_code)
|
self.log_request(status_code)
|
||||||
|
|
||||||
def _write_response(self, status_code, body, content_type='text/html', headers=None):
|
def _write_response(self, status_code, body, content_type='text/html', headers=None):
|
||||||
|
# TODO: try-catch ConnectionResetError: [Errno 104] Connection reset by peer and log it in DEBUG level
|
||||||
self.send_response(status_code)
|
self.send_response(status_code)
|
||||||
headers = headers or {}
|
headers = headers or {}
|
||||||
if content_type:
|
if content_type:
|
||||||
@@ -141,7 +142,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
ignore_tags = True
|
ignore_tags = True
|
||||||
elif 'replica' in path:
|
elif 'replica' in path:
|
||||||
status_code = replica_status_code
|
status_code = replica_status_code
|
||||||
elif 'read-only' in path:
|
elif 'read-only' in path and 'sync' not in path:
|
||||||
status_code = 200 if 200 in (primary_status_code, standby_leader_status_code) else replica_status_code
|
status_code = 200 if 200 in (primary_status_code, standby_leader_status_code) else replica_status_code
|
||||||
elif 'health' in path:
|
elif 'health' in path:
|
||||||
status_code = 200 if response.get('state') == 'running' else 503
|
status_code = 200 if response.get('state') == 'running' else 503
|
||||||
@@ -185,8 +186,20 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
def do_OPTIONS(self):
|
def do_OPTIONS(self):
|
||||||
self.do_GET(write_status_code_only=True)
|
self.do_GET(write_status_code_only=True)
|
||||||
|
|
||||||
|
def do_HEAD(self):
|
||||||
|
self.do_GET(write_status_code_only=True)
|
||||||
|
|
||||||
def do_GET_liveness(self):
|
def do_GET_liveness(self):
|
||||||
self._write_status_code_only(200)
|
patroni = self.server.patroni
|
||||||
|
is_primary = patroni.postgresql.role == 'master' and patroni.postgresql.is_running()
|
||||||
|
# We can tolerate Patroni problems longer on the replica.
|
||||||
|
# On the primary the liveness probe most likely will start failing only after the leader key expired.
|
||||||
|
# It should not be a big problem because replicas will see that the primary is still alive via REST API call.
|
||||||
|
liveness_threshold = patroni.dcs.ttl * (1 if is_primary else 2)
|
||||||
|
|
||||||
|
# In maintenance mode (pause) we are fine if heartbeat loop stuck.
|
||||||
|
status_code = 200 if patroni.ha.is_paused() or patroni.next_run + liveness_threshold > time.time() else 503
|
||||||
|
self._write_status_code_only(status_code)
|
||||||
|
|
||||||
def do_GET_readiness(self):
|
def do_GET_readiness(self):
|
||||||
patroni = self.server.patroni
|
patroni = self.server.patroni
|
||||||
@@ -355,6 +368,14 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
self.server.patroni.sighup_handler()
|
self.server.patroni.sighup_handler()
|
||||||
self._write_response(202, 'reload scheduled')
|
self._write_response(202, 'reload scheduled')
|
||||||
|
|
||||||
|
@check_access
|
||||||
|
def do_POST_sigterm(self):
|
||||||
|
"""Only for behave testing on windows"""
|
||||||
|
|
||||||
|
if os.name == 'nt' and os.getenv('BEHAVE_DEBUG'):
|
||||||
|
self.server.patroni.api_sigterm()
|
||||||
|
self._write_response(202, 'shutdown scheduled')
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def parse_schedule(schedule, action):
|
def parse_schedule(schedule, action):
|
||||||
""" parses the given schedule and validates at """
|
""" parses the given schedule and validates at """
|
||||||
@@ -608,7 +629,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
stmt = ("SELECT " + postgresql.POSTMASTER_START_TIME + ", " + postgresql.TL_LSN + ","
|
stmt = ("SELECT " + postgresql.POSTMASTER_START_TIME + ", " + postgresql.TL_LSN + ","
|
||||||
" pg_catalog.pg_last_xact_replay_timestamp(),"
|
" pg_catalog.pg_last_xact_replay_timestamp(),"
|
||||||
" pg_catalog.array_to_json(pg_catalog.array_agg(pg_catalog.row_to_json(ri))) "
|
" pg_catalog.array_to_json(pg_catalog.array_agg(pg_catalog.row_to_json(ri))) "
|
||||||
"FROM (SELECT (SELECT rolname FROM pg_authid WHERE oid = usesysid) AS usename,"
|
"FROM (SELECT (SELECT rolname FROM pg_catalog.pg_authid WHERE oid = usesysid) AS usename,"
|
||||||
" application_name, client_addr, w.state, sync_state, sync_priority"
|
" application_name, client_addr, w.state, sync_state, sync_priority"
|
||||||
" FROM pg_catalog.pg_stat_get_wal_senders() w, pg_catalog.pg_stat_get_activity(pid)) AS ri")
|
" FROM pg_catalog.pg_stat_get_wal_senders() w, pg_catalog.pg_stat_get_activity(pid)) AS ri")
|
||||||
|
|
||||||
@@ -812,32 +833,23 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
else:
|
else:
|
||||||
logger.error('Bad value in the "restapi.verify_client": %s', verify_client)
|
logger.error('Bad value in the "restapi.verify_client": %s', verify_client)
|
||||||
self.__ssl_serial_number = self.get_certificate_serial_number()
|
self.__ssl_serial_number = self.get_certificate_serial_number()
|
||||||
self.socket = ctx.wrap_socket(self.socket, server_side=True)
|
self.socket = ctx.wrap_socket(self.socket, server_side=True, do_handshake_on_connect=False)
|
||||||
if reloading_config:
|
if reloading_config:
|
||||||
self.start()
|
self.start()
|
||||||
|
|
||||||
def process_request_thread(self, request, client_address):
|
def process_request_thread(self, request, client_address):
|
||||||
if isinstance(request, tuple):
|
enable_keepalive(request, 10, 3)
|
||||||
sock, newsock = request
|
if hasattr(request, 'context'): # SSLSocket
|
||||||
try:
|
request.do_handshake()
|
||||||
request = sock.context.wrap_socket(newsock, do_handshake_on_connect=sock.do_handshake_on_connect,
|
|
||||||
suppress_ragged_eofs=sock.suppress_ragged_eofs, server_side=True)
|
|
||||||
except socket.error:
|
|
||||||
return
|
|
||||||
super(RestApiServer, self).process_request_thread(request, client_address)
|
super(RestApiServer, self).process_request_thread(request, client_address)
|
||||||
|
|
||||||
def get_request(self):
|
|
||||||
sock = self.socket
|
|
||||||
newsock, addr = socket.socket.accept(sock)
|
|
||||||
enable_keepalive(newsock, 10, 3)
|
|
||||||
if hasattr(sock, 'context'): # SSLSocket, we want to do the deferred handshake from a thread
|
|
||||||
newsock = (sock, newsock)
|
|
||||||
return newsock, addr
|
|
||||||
|
|
||||||
def shutdown_request(self, request):
|
def shutdown_request(self, request):
|
||||||
if isinstance(request, tuple):
|
if hasattr(request, 'context'): # SSLSocket
|
||||||
_, request = request # SSLSocket
|
try:
|
||||||
return super(RestApiServer, self).shutdown_request(request)
|
request.unwrap()
|
||||||
|
except Exception as e:
|
||||||
|
logger.debug('Failed to shutdown SSL connection: %r', e)
|
||||||
|
super(RestApiServer, self).shutdown_request(request)
|
||||||
|
|
||||||
def get_certificate_serial_number(self):
|
def get_certificate_serial_number(self):
|
||||||
if self.__ssl_options.get('certfile'):
|
if self.__ssl_options.get('certfile'):
|
||||||
|
|||||||
+8
-6
@@ -32,7 +32,7 @@ _AUTH_ALLOWED_PARAMETERS = (
|
|||||||
|
|
||||||
def default_validator(conf):
|
def default_validator(conf):
|
||||||
if not conf:
|
if not conf:
|
||||||
return "Config is empty."
|
raise ConfigParseError("Config is empty.")
|
||||||
|
|
||||||
|
|
||||||
class Config(object):
|
class Config(object):
|
||||||
@@ -102,9 +102,9 @@ class Config(object):
|
|||||||
config_env = os.environ.pop(self.PATRONI_CONFIG_VARIABLE, None)
|
config_env = os.environ.pop(self.PATRONI_CONFIG_VARIABLE, None)
|
||||||
self._local_configuration = config_env and yaml.safe_load(config_env) or self.__environment_configuration
|
self._local_configuration = config_env and yaml.safe_load(config_env) or self.__environment_configuration
|
||||||
if validator:
|
if validator:
|
||||||
error = validator(self._local_configuration)
|
errors = validator(self._local_configuration)
|
||||||
if error:
|
if errors:
|
||||||
raise ConfigParseError(error)
|
raise ConfigParseError("\n".join(errors))
|
||||||
|
|
||||||
self.__effective_configuration = self._build_effective_configuration({}, self._local_configuration)
|
self.__effective_configuration = self._build_effective_configuration({}, self._local_configuration)
|
||||||
self._data_dir = self.__effective_configuration.get('postgresql', {}).get('data_dir', "")
|
self._data_dir = self.__effective_configuration.get('postgresql', {}).get('data_dir', "")
|
||||||
@@ -227,7 +227,8 @@ class Config(object):
|
|||||||
for name, value in (value or {}).items():
|
for name, value in (value or {}).items():
|
||||||
if name == 'parameters':
|
if name == 'parameters':
|
||||||
config['postgresql'][name].update(self._process_postgresql_parameters(value))
|
config['postgresql'][name].update(self._process_postgresql_parameters(value))
|
||||||
elif name not in ('connect_address', 'listen', 'data_dir', 'pgpass', 'authentication'):
|
elif name not in ('connect_address', 'proxy_address', 'listen',
|
||||||
|
'config_dir', 'data_dir', 'pgpass', 'authentication'):
|
||||||
config['postgresql'][name] = deepcopy(value)
|
config['postgresql'][name] = deepcopy(value)
|
||||||
elif name == 'standby_cluster':
|
elif name == 'standby_cluster':
|
||||||
for name, value in (value or {}).items():
|
for name, value in (value or {}).items():
|
||||||
@@ -271,7 +272,8 @@ class Config(object):
|
|||||||
'cafile', 'ciphers', 'verify_client', 'http_extra_headers',
|
'cafile', 'ciphers', 'verify_client', 'http_extra_headers',
|
||||||
'https_extra_headers', 'allowlist', 'allowlist_include_members'])
|
'https_extra_headers', 'allowlist', 'allowlist_include_members'])
|
||||||
_set_section_values('ctl', ['insecure', 'cacert', 'certfile', 'keyfile', 'keyfile_password'])
|
_set_section_values('ctl', ['insecure', 'cacert', 'certfile', 'keyfile', 'keyfile_password'])
|
||||||
_set_section_values('postgresql', ['listen', 'connect_address', 'config_dir', 'data_dir', 'pgpass', 'bin_dir'])
|
_set_section_values('postgresql', ['listen', 'connect_address', 'proxy_address',
|
||||||
|
'config_dir', 'data_dir', 'pgpass', 'bin_dir'])
|
||||||
_set_section_values('log', ['level', 'traceback_level', 'format', 'dateformat', 'max_queue_size',
|
_set_section_values('log', ['level', 'traceback_level', 'format', 'dateformat', 'max_queue_size',
|
||||||
'dir', 'file_size', 'file_num', 'loggers'])
|
'dir', 'file_size', 'file_num', 'loggers'])
|
||||||
_set_section_values('raft', ['data_dir', 'self_addr', 'partner_addrs', 'password', 'bind_addr'])
|
_set_section_values('raft', ['data_dir', 'self_addr', 'partner_addrs', 'password', 'bind_addr'])
|
||||||
|
|||||||
+20
-26
@@ -60,6 +60,18 @@ class PatronictlPrettyTable(PrettyTable):
|
|||||||
self.__hline_num = 0
|
self.__hline_num = 0
|
||||||
self.__hline = None
|
self.__hline = None
|
||||||
|
|
||||||
|
def __build_header(self, line):
|
||||||
|
header = self.__table_header[:len(line) - 2]
|
||||||
|
return "".join([line[0], header, line[1 + len(header):]])
|
||||||
|
|
||||||
|
def _stringify_hrule(self, *args, **kwargs):
|
||||||
|
ret = super(PatronictlPrettyTable, self)._stringify_hrule(*args, **kwargs)
|
||||||
|
where = args[1] if len(args) > 1 else kwargs.get('where')
|
||||||
|
if where == 'top_' and self.__table_header:
|
||||||
|
ret = self.__build_header(ret)
|
||||||
|
self.__hline_num += 1
|
||||||
|
return ret
|
||||||
|
|
||||||
def _is_first_hline(self):
|
def _is_first_hline(self):
|
||||||
return self.__hline_num == 0
|
return self.__hline_num == 0
|
||||||
|
|
||||||
@@ -71,8 +83,7 @@ class PatronictlPrettyTable(PrettyTable):
|
|||||||
|
|
||||||
# Inject nice table header
|
# Inject nice table header
|
||||||
if self._is_first_hline() and self.__table_header:
|
if self._is_first_hline() and self.__table_header:
|
||||||
header = self.__table_header[:len(ret) - 2]
|
ret = self.__build_header(ret)
|
||||||
ret = "".join([ret[0], header, ret[1 + len(header):]])
|
|
||||||
|
|
||||||
self.__hline_num += 1
|
self.__hline_num += 1
|
||||||
return ret
|
return ret
|
||||||
@@ -99,7 +110,7 @@ def parse_dcs(dcs):
|
|||||||
return yaml.safe_load(default['template'].format(host=parsed.hostname or 'localhost', port=port or default['port']))
|
return yaml.safe_load(default['template'].format(host=parsed.hostname or 'localhost', port=port or default['port']))
|
||||||
|
|
||||||
|
|
||||||
def load_config(path, dcs):
|
def load_config(path, dcs_url):
|
||||||
from patroni.config import Config
|
from patroni.config import Config
|
||||||
|
|
||||||
if not (os.path.exists(path) and os.access(path, os.R_OK)):
|
if not (os.path.exists(path) and os.access(path, os.R_OK)):
|
||||||
@@ -112,22 +123,14 @@ def load_config(path, dcs):
|
|||||||
logging.debug('Loading configuration from file %s', path)
|
logging.debug('Loading configuration from file %s', path)
|
||||||
config = Config(path, validator=None).copy()
|
config = Config(path, validator=None).copy()
|
||||||
|
|
||||||
dcs = parse_dcs(dcs) or parse_dcs(config.get('dcs_api')) or {}
|
dcs_url = parse_dcs(dcs_url) or {}
|
||||||
if dcs:
|
if dcs_url:
|
||||||
for d in DCS_DEFAULTS:
|
for d in DCS_DEFAULTS:
|
||||||
config.pop(d, None)
|
config.pop(d, None)
|
||||||
config.update(dcs)
|
config.update(dcs_url)
|
||||||
return config
|
return config
|
||||||
|
|
||||||
|
|
||||||
def store_config(config, path):
|
|
||||||
dir_path = os.path.dirname(path)
|
|
||||||
if dir_path and not os.path.isdir(dir_path):
|
|
||||||
os.makedirs(dir_path)
|
|
||||||
with open(path, 'w') as fd:
|
|
||||||
yaml.dump(config, fd)
|
|
||||||
|
|
||||||
|
|
||||||
option_format = click.option('--format', '-f', 'fmt', help='Output format (pretty, tsv, json, yaml)', default='pretty')
|
option_format = click.option('--format', '-f', 'fmt', help='Output format (pretty, tsv, json, yaml)', default='pretty')
|
||||||
option_watchrefresh = click.option('-w', '--watch', type=float, help='Auto update the screen every X seconds')
|
option_watchrefresh = click.option('-w', '--watch', type=float, help='Auto update the screen every X seconds')
|
||||||
option_watch = click.option('-W', is_flag=True, help='Auto update the screen every 2 seconds')
|
option_watch = click.option('-W', is_flag=True, help='Auto update the screen every 2 seconds')
|
||||||
@@ -140,16 +143,16 @@ option_insecure = click.option('-k', '--insecure', is_flag=True, help='Allow con
|
|||||||
@click.group()
|
@click.group()
|
||||||
@click.option('--config-file', '-c', help='Configuration file',
|
@click.option('--config-file', '-c', help='Configuration file',
|
||||||
envvar='PATRONICTL_CONFIG_FILE', default=CONFIG_FILE_PATH)
|
envvar='PATRONICTL_CONFIG_FILE', default=CONFIG_FILE_PATH)
|
||||||
@click.option('--dcs', '-d', help='Use this DCS', envvar='DCS')
|
@click.option('--dcs-url', '--dcs', '-d', 'dcs_url', help='The DCS connect url', envvar='DCS_URL')
|
||||||
@option_insecure
|
@option_insecure
|
||||||
@click.pass_context
|
@click.pass_context
|
||||||
def ctl(ctx, config_file, dcs, insecure):
|
def ctl(ctx, config_file, dcs_url, insecure):
|
||||||
level = 'WARNING'
|
level = 'WARNING'
|
||||||
for name in ('LOGLEVEL', 'PATRONI_LOGLEVEL', 'PATRONI_LOG_LEVEL'):
|
for name in ('LOGLEVEL', 'PATRONI_LOGLEVEL', 'PATRONI_LOG_LEVEL'):
|
||||||
level = os.environ.get(name, level)
|
level = os.environ.get(name, level)
|
||||||
logging.basicConfig(format='%(asctime)s - %(levelname)s - %(message)s', level=level)
|
logging.basicConfig(format='%(asctime)s - %(levelname)s - %(message)s', level=level)
|
||||||
logging.captureWarnings(True) # Capture eventual SSL warning
|
logging.captureWarnings(True) # Capture eventual SSL warning
|
||||||
ctx.obj = load_config(config_file, dcs)
|
ctx.obj = load_config(config_file, dcs_url)
|
||||||
# backward compatibility for configuration file where ctl section is not define
|
# backward compatibility for configuration file where ctl section is not define
|
||||||
ctx.obj.setdefault('ctl', {})['insecure'] = ctx.obj.get('ctl', {}).get('insecure') or insecure
|
ctx.obj.setdefault('ctl', {})['insecure'] = ctx.obj.get('ctl', {}).get('insecure') or insecure
|
||||||
|
|
||||||
@@ -271,7 +274,6 @@ def get_cursor(cluster, connect_parameters, role='master', member=None):
|
|||||||
|
|
||||||
from . import psycopg
|
from . import psycopg
|
||||||
conn = psycopg.connect(**params)
|
conn = psycopg.connect(**params)
|
||||||
conn.autocommit = True
|
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
if role == 'any':
|
if role == 'any':
|
||||||
return cursor
|
return cursor
|
||||||
@@ -886,14 +888,6 @@ def timestamp(precision=6):
|
|||||||
return datetime.datetime.now().strftime('%Y-%m-%d %H:%M:%S.%f')[:precision - 7]
|
return datetime.datetime.now().strftime('%Y-%m-%d %H:%M:%S.%f')[:precision - 7]
|
||||||
|
|
||||||
|
|
||||||
@ctl.command('configure', help='Create configuration file')
|
|
||||||
@click.option('--config-file', '-c', help='Configuration file', prompt='Configuration file', default=CONFIG_FILE_PATH)
|
|
||||||
@click.option('--dcs', '-d', help='The DCS connect url', prompt='DCS connect url', default='etcd://localhost:2379')
|
|
||||||
@click.option('--namespace', '-n', help='The namespace', prompt='Namespace', default='/service/')
|
|
||||||
def configure(config_file, dcs, namespace):
|
|
||||||
store_config({'dcs_api': str(dcs), 'namespace': str(namespace)}, config_file)
|
|
||||||
|
|
||||||
|
|
||||||
def touch_member(config, dcs):
|
def touch_member(config, dcs):
|
||||||
''' Rip-off of the ha.touch_member without inter-class dependencies '''
|
''' Rip-off of the ha.touch_member without inter-class dependencies '''
|
||||||
p = Postgresql(config['postgresql'])
|
p = Postgresql(config['postgresql'])
|
||||||
|
|||||||
+11
-3
@@ -1,3 +1,5 @@
|
|||||||
|
from __future__ import print_function
|
||||||
|
|
||||||
import abc
|
import abc
|
||||||
import os
|
import os
|
||||||
import signal
|
import signal
|
||||||
@@ -22,10 +24,14 @@ class AbstractPatroniDaemon(object):
|
|||||||
def sighup_handler(self, *args):
|
def sighup_handler(self, *args):
|
||||||
self._received_sighup = True
|
self._received_sighup = True
|
||||||
|
|
||||||
def sigterm_handler(self, *args):
|
def api_sigterm(self):
|
||||||
with self._sigterm_lock:
|
with self._sigterm_lock:
|
||||||
if not self._received_sigterm:
|
if not self._received_sigterm:
|
||||||
self._received_sigterm = True
|
self._received_sigterm = True
|
||||||
|
return True
|
||||||
|
|
||||||
|
def sigterm_handler(self, *args):
|
||||||
|
if self.api_sigterm():
|
||||||
sys.exit()
|
sys.exit()
|
||||||
|
|
||||||
def setup_signal_handlers(self):
|
def setup_signal_handlers(self):
|
||||||
@@ -83,15 +89,17 @@ def abstract_main(cls, validator=None):
|
|||||||
help='Patroni may also read the configuration from the {0} environment variable'
|
help='Patroni may also read the configuration from the {0} environment variable'
|
||||||
.format(Config.PATRONI_CONFIG_VARIABLE))
|
.format(Config.PATRONI_CONFIG_VARIABLE))
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
validate_config = validator and args.validate_config
|
||||||
try:
|
try:
|
||||||
if validator and args.validate_config:
|
if validate_config:
|
||||||
Config(args.configfile, validator=validator)
|
Config(args.configfile, validator=validator)
|
||||||
sys.exit()
|
sys.exit()
|
||||||
|
|
||||||
config = Config(args.configfile)
|
config = Config(args.configfile)
|
||||||
except ConfigParseError as e:
|
except ConfigParseError as e:
|
||||||
if e.value:
|
if e.value:
|
||||||
print(e.value)
|
print(e.value, file=sys.stderr)
|
||||||
|
if not validate_config:
|
||||||
parser.print_help()
|
parser.print_help()
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|
||||||
|
|||||||
+51
-10
@@ -444,7 +444,7 @@ class TimelineHistory(namedtuple('TimelineHistory', 'index,value,lines')):
|
|||||||
return TimelineHistory(index, value, lines)
|
return TimelineHistory(index, value, lines)
|
||||||
|
|
||||||
|
|
||||||
class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_lsn,members,failover,sync,history,slots')):
|
class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_lsn,members,failover,sync,history,slots,failsafe')):
|
||||||
|
|
||||||
"""Immutable object (namedtuple) which represents PostgreSQL cluster.
|
"""Immutable object (namedtuple) which represents PostgreSQL cluster.
|
||||||
Consists of the following fields:
|
Consists of the following fields:
|
||||||
@@ -606,11 +606,11 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_lsn,members,f
|
|||||||
@property
|
@property
|
||||||
def timeline(self):
|
def timeline(self):
|
||||||
"""
|
"""
|
||||||
>>> Cluster(0, 0, 0, 0, 0, 0, 0, 0, 0).timeline
|
>>> Cluster(0, 0, 0, 0, 0, 0, 0, 0, 0, None).timeline
|
||||||
0
|
0
|
||||||
>>> Cluster(0, 0, 0, 0, 0, 0, 0, TimelineHistory.from_node(1, '[]'), 0).timeline
|
>>> Cluster(0, 0, 0, 0, 0, 0, 0, TimelineHistory.from_node(1, '[]'), 0, None).timeline
|
||||||
1
|
1
|
||||||
>>> Cluster(0, 0, 0, 0, 0, 0, 0, TimelineHistory.from_node(1, '[["a"]]'), 0).timeline
|
>>> Cluster(0, 0, 0, 0, 0, 0, 0, TimelineHistory.from_node(1, '[["a"]]'), 0, None).timeline
|
||||||
0
|
0
|
||||||
"""
|
"""
|
||||||
if self.history:
|
if self.history:
|
||||||
@@ -628,6 +628,20 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_lsn,members,f
|
|||||||
return next(iter(sorted(filter(lambda v: v, [m.version for m in self.members])) + [None]))
|
return next(iter(sorted(filter(lambda v: v, [m.version for m in self.members])) + [None]))
|
||||||
|
|
||||||
|
|
||||||
|
class ReturnFalseException(Exception):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def catch_return_false_exception(func):
|
||||||
|
def wrapper(*args, **kwargs):
|
||||||
|
try:
|
||||||
|
return func(*args, **kwargs)
|
||||||
|
except ReturnFalseException:
|
||||||
|
return False
|
||||||
|
|
||||||
|
return wrapper
|
||||||
|
|
||||||
|
|
||||||
@six.add_metaclass(abc.ABCMeta)
|
@six.add_metaclass(abc.ABCMeta)
|
||||||
class AbstractDCS(object):
|
class AbstractDCS(object):
|
||||||
|
|
||||||
@@ -641,6 +655,7 @@ class AbstractDCS(object):
|
|||||||
_STATUS = 'status' # JSON, contains "leader_lsn" and confirmed_flush_lsn of logical "slots" on the leader
|
_STATUS = 'status' # JSON, contains "leader_lsn" and confirmed_flush_lsn of logical "slots" on the leader
|
||||||
_LEADER_OPTIME = _OPTIME + '/' + _LEADER # legacy
|
_LEADER_OPTIME = _OPTIME + '/' + _LEADER # legacy
|
||||||
_SYNC = 'sync'
|
_SYNC = 'sync'
|
||||||
|
_FAILSAFE = 'failsafe'
|
||||||
|
|
||||||
def __init__(self, config):
|
def __init__(self, config):
|
||||||
"""
|
"""
|
||||||
@@ -658,6 +673,7 @@ class AbstractDCS(object):
|
|||||||
self._last_lsn = ''
|
self._last_lsn = ''
|
||||||
self._last_seen = 0
|
self._last_seen = 0
|
||||||
self._last_status = {}
|
self._last_status = {}
|
||||||
|
self._last_failsafe = {}
|
||||||
self.event = Event()
|
self.event = Event()
|
||||||
|
|
||||||
def client_path(self, path):
|
def client_path(self, path):
|
||||||
@@ -703,6 +719,10 @@ class AbstractDCS(object):
|
|||||||
def sync_path(self):
|
def sync_path(self):
|
||||||
return self.client_path(self._SYNC)
|
return self.client_path(self._SYNC)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def failsafe_path(self):
|
||||||
|
return self.client_path(self._FAILSAFE)
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
def set_ttl(self, ttl):
|
def set_ttl(self, ttl):
|
||||||
"""Set the new ttl value for leader key"""
|
"""Set the new ttl value for leader key"""
|
||||||
@@ -754,6 +774,8 @@ class AbstractDCS(object):
|
|||||||
raise
|
raise
|
||||||
|
|
||||||
self._last_seen = int(time.time())
|
self._last_seen = int(time.time())
|
||||||
|
self._last_status = {self._OPTIME: cluster.last_lsn, 'slots': cluster.slots}
|
||||||
|
self._last_failsafe = cluster.failsafe
|
||||||
|
|
||||||
with self._cluster_thread_lock:
|
with self._cluster_thread_lock:
|
||||||
self._cluster = cluster
|
self._cluster = cluster
|
||||||
@@ -796,23 +818,35 @@ class AbstractDCS(object):
|
|||||||
self._last_lsn = value[self._OPTIME]
|
self._last_lsn = value[self._OPTIME]
|
||||||
self._write_leader_optime(str(value[self._OPTIME]))
|
self._write_leader_optime(str(value[self._OPTIME]))
|
||||||
|
|
||||||
|
@abc.abstractmethod
|
||||||
|
def _write_failsafe(self, value):
|
||||||
|
"""Write current cluster topology to DCS that will be used by failsafe mechanism (if enabled).
|
||||||
|
|
||||||
|
:param value: failsafe topology serialized in JSON format
|
||||||
|
:returns: `!True` on success."""
|
||||||
|
|
||||||
|
def write_failsafe(self, value):
|
||||||
|
if not (isinstance(self._last_failsafe, dict) and deep_compare(self._last_failsafe, value))\
|
||||||
|
and self._write_failsafe(json.dumps(value, separators=(',', ':'))):
|
||||||
|
self._last_failsafe = value
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
"""Update leader key (or session) ttl
|
"""Update leader key (or session) ttl
|
||||||
|
|
||||||
:returns: `!True` if leader key (or session) has been updated successfully.
|
:returns: `!True` if leader key (or session) has been updated successfully.
|
||||||
If not, `!False` must be returned and current instance would be demoted.
|
|
||||||
|
|
||||||
You have to use CAS (Compare And Swap) operation in order to update leader key,
|
You have to use CAS (Compare And Swap) operation in order to update leader key,
|
||||||
for example for etcd `prevValue` parameter must be used."""
|
for example for etcd `prevValue` parameter must be used.
|
||||||
|
If update fails due to DCS not being accessible or because it is not able to
|
||||||
|
process requests (hopefuly temporary), the ~DCSError exception should be raised."""
|
||||||
|
|
||||||
def update_leader(self, last_lsn, slots=None):
|
def update_leader(self, last_lsn, slots=None, failsafe=None):
|
||||||
"""Update leader key (or session) ttl and optime/leader
|
"""Update leader key (or session) ttl and optime/leader
|
||||||
|
|
||||||
:param last_lsn: absolute WAL LSN in bytes
|
:param last_lsn: absolute WAL LSN in bytes
|
||||||
:param slots: dict with permanent slots confirmed_flush_lsn
|
:param slots: dict with permanent slots confirmed_flush_lsn
|
||||||
:returns: `!True` if leader key (or session) has been updated successfully.
|
:returns: `!True` if leader key (or session) has been updated successfully."""
|
||||||
If not, `!False` must be returned and current instance would be demoted."""
|
|
||||||
|
|
||||||
ret = self._update_leader()
|
ret = self._update_leader()
|
||||||
if ret and last_lsn:
|
if ret and last_lsn:
|
||||||
@@ -820,6 +854,10 @@ class AbstractDCS(object):
|
|||||||
if slots:
|
if slots:
|
||||||
status['slots'] = slots
|
status['slots'] = slots
|
||||||
self.write_status(status)
|
self.write_status(status)
|
||||||
|
|
||||||
|
if ret and failsafe is not None:
|
||||||
|
self.write_failsafe(failsafe)
|
||||||
|
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
@@ -831,7 +869,10 @@ class AbstractDCS(object):
|
|||||||
:returns: `!True` if key has been created successfully.
|
:returns: `!True` if key has been created successfully.
|
||||||
|
|
||||||
Key must be created atomically. In case if key already exists it should not be
|
Key must be created atomically. In case if key already exists it should not be
|
||||||
overwritten and `!False` must be returned"""
|
overwritten and `!False` must be returned.
|
||||||
|
|
||||||
|
If key creation fails due to DCS not being accessible or because it is not able to
|
||||||
|
process requests (hopefuly temporary), the ~DCSError exception should be raised"""
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
def set_failover_value(self, value, index=None):
|
def set_failover_value(self, value, index=None):
|
||||||
|
|||||||
+72
-22
@@ -14,7 +14,8 @@ from urllib3.exceptions import HTTPError
|
|||||||
from six.moves.urllib.parse import urlencode, urlparse, quote
|
from six.moves.urllib.parse import urlencode, urlparse, quote
|
||||||
from six.moves.http_client import HTTPException
|
from six.moves.http_client import HTTPException
|
||||||
|
|
||||||
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member, SyncState, TimelineHistory
|
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member,\
|
||||||
|
SyncState, TimelineHistory, ReturnFalseException, catch_return_false_exception
|
||||||
from ..exceptions import DCSError
|
from ..exceptions import DCSError
|
||||||
from ..utils import deep_compare, parse_bool, Retry, RetryFailedError, split_host_port, uri, USER_AGENT
|
from ..utils import deep_compare, parse_bool, Retry, RetryFailedError, split_host_port, uri, USER_AGENT
|
||||||
|
|
||||||
@@ -236,6 +237,7 @@ class Consul(AbstractDCS):
|
|||||||
self._service_check_tls_server_name = config.get('service_check_tls_server_name', None)
|
self._service_check_tls_server_name = config.get('service_check_tls_server_name', None)
|
||||||
if not self._ctl:
|
if not self._ctl:
|
||||||
self.create_session()
|
self.create_session()
|
||||||
|
self._previous_loop_token = self._client.token
|
||||||
|
|
||||||
def retry(self, *args, **kwargs):
|
def retry(self, *args, **kwargs):
|
||||||
return self._retry.copy()(*args, **kwargs)
|
return self._retry.copy()(*args, **kwargs)
|
||||||
@@ -270,7 +272,7 @@ class Consul(AbstractDCS):
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def ttl(self):
|
def ttl(self):
|
||||||
return self._client.http.ttl
|
return self._client.http.ttl * 2 # we multiply the value by 2 because it was divided in the `set_ttl()` method
|
||||||
|
|
||||||
def set_retry_timeout(self, retry_timeout):
|
def set_retry_timeout(self, retry_timeout):
|
||||||
self._retry.deadline = retry_timeout
|
self._retry.deadline = retry_timeout
|
||||||
@@ -285,9 +287,9 @@ class Consul(AbstractDCS):
|
|||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('adjust_ttl')
|
logger.exception('adjust_ttl')
|
||||||
|
|
||||||
def _do_refresh_session(self):
|
def _do_refresh_session(self, force=False):
|
||||||
""":returns: `!True` if it had to create new session"""
|
""":returns: `!True` if it had to create new session"""
|
||||||
if self._session and self._last_session_refresh + self._loop_wait > time.time():
|
if not force and self._session and self._last_session_refresh + self._loop_wait > time.time():
|
||||||
return False
|
return False
|
||||||
|
|
||||||
if self._session:
|
if self._session:
|
||||||
@@ -372,11 +374,6 @@ class Consul(AbstractDCS):
|
|||||||
|
|
||||||
# get leader
|
# get leader
|
||||||
leader = nodes.get(self._LEADER)
|
leader = nodes.get(self._LEADER)
|
||||||
if not self._ctl and leader and leader['Value'] == self._name \
|
|
||||||
and self._session != leader.get('Session', 'x'):
|
|
||||||
logger.info('I am leader but not owner of the session. Removing leader node')
|
|
||||||
self._client.kv.delete(self.leader_path, cas=leader['ModifyIndex'])
|
|
||||||
leader = None
|
|
||||||
|
|
||||||
if leader:
|
if leader:
|
||||||
member = Member(-1, leader['Value'], None, {})
|
member = Member(-1, leader['Value'], None, {})
|
||||||
@@ -392,9 +389,16 @@ class Consul(AbstractDCS):
|
|||||||
sync = nodes.get(self._SYNC)
|
sync = nodes.get(self._SYNC)
|
||||||
sync = SyncState.from_node(sync and sync['ModifyIndex'], sync and sync['Value'])
|
sync = SyncState.from_node(sync and sync['ModifyIndex'], sync and sync['Value'])
|
||||||
|
|
||||||
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
# get failsafe topology
|
||||||
|
failsafe = nodes.get(self._FAILSAFE)
|
||||||
|
try:
|
||||||
|
failsafe = json.loads(failsafe['Value']) if failsafe else None
|
||||||
|
except Exception:
|
||||||
|
failsafe = None
|
||||||
|
|
||||||
|
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots, failsafe)
|
||||||
except NotFound:
|
except NotFound:
|
||||||
return Cluster(None, None, None, None, [], None, None, None, None)
|
return Cluster(None, None, None, None, [], None, None, None, None, None)
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('get_cluster')
|
logger.exception('get_cluster')
|
||||||
raise ConsulError('Consul is not responding properly')
|
raise ConsulError('Consul is not responding properly')
|
||||||
@@ -464,6 +468,7 @@ class Consul(AbstractDCS):
|
|||||||
tags = self._service_tags[:]
|
tags = self._service_tags[:]
|
||||||
tags.append(role)
|
tags.append(role)
|
||||||
self._previous_loop_service_tags = self._service_tags
|
self._previous_loop_service_tags = self._service_tags
|
||||||
|
self._previous_loop_token = self._client.token
|
||||||
|
|
||||||
params = {
|
params = {
|
||||||
'service_id': '{0}/{1}'.format(self._scope, self._name),
|
'service_id': '{0}/{1}'.format(self._scope, self._name),
|
||||||
@@ -500,25 +505,38 @@ class Consul(AbstractDCS):
|
|||||||
if (
|
if (
|
||||||
force or update or self._register_service != self._previous_loop_register_service
|
force or update or self._register_service != self._previous_loop_register_service
|
||||||
or self._service_tags != self._previous_loop_service_tags
|
or self._service_tags != self._previous_loop_service_tags
|
||||||
|
or self._client.token != self._previous_loop_token
|
||||||
):
|
):
|
||||||
return self._update_service(new_data)
|
return self._update_service(new_data)
|
||||||
|
|
||||||
@catch_consul_errors
|
def _do_attempt_to_acquire_leader(self, permanent, retry):
|
||||||
def _do_attempt_to_acquire_leader(self, permanent):
|
|
||||||
try:
|
try:
|
||||||
kwargs = {} if permanent else {'acquire': self._session}
|
kwargs = {} if permanent else {'acquire': self._session}
|
||||||
return self.retry(self._client.kv.put, self.leader_path, self._name, **kwargs)
|
return retry(self._client.kv.put, self.leader_path, self._name, **kwargs)
|
||||||
except InvalidSession:
|
except InvalidSession:
|
||||||
self._session = None
|
|
||||||
logger.error('Our session disappeared from Consul. Will try to get a new one and retry attempt')
|
logger.error('Our session disappeared from Consul. Will try to get a new one and retry attempt')
|
||||||
self.refresh_session()
|
self._session = None
|
||||||
return self.retry(self._client.kv.put, self.leader_path, self._name, acquire=self._session)
|
retry.deadline = retry.stoptime - time.time()
|
||||||
|
|
||||||
|
retry(self._do_refresh_session)
|
||||||
|
|
||||||
|
retry.deadline = retry.stoptime - time.time()
|
||||||
|
if retry.deadline < 1:
|
||||||
|
raise ConsulError('_do_attempt_to_acquire_leader timeout')
|
||||||
|
|
||||||
|
return retry(self._client.kv.put, self.leader_path, self._name, acquire=self._session)
|
||||||
|
|
||||||
|
@catch_return_false_exception
|
||||||
def attempt_to_acquire_leader(self, permanent=False):
|
def attempt_to_acquire_leader(self, permanent=False):
|
||||||
if not self._session and not permanent:
|
retry = self._retry.copy()
|
||||||
self.refresh_session()
|
if not permanent:
|
||||||
|
self._run_and_handle_exceptions(self._do_refresh_session, retry=retry)
|
||||||
|
|
||||||
ret = self._do_attempt_to_acquire_leader(permanent)
|
retry.deadline = retry.stoptime - time.time()
|
||||||
|
if retry.deadline < 1:
|
||||||
|
raise ConsulError('attempt_to_acquire_leader timeout')
|
||||||
|
|
||||||
|
ret = self._run_and_handle_exceptions(self._do_attempt_to_acquire_leader, permanent, retry, retry=None)
|
||||||
if not ret:
|
if not ret:
|
||||||
logger.info('Could not take out TTL lock')
|
logger.info('Could not take out TTL lock')
|
||||||
|
|
||||||
@@ -544,10 +562,42 @@ class Consul(AbstractDCS):
|
|||||||
return self._client.kv.put(self.status_path, value)
|
return self._client.kv.put(self.status_path, value)
|
||||||
|
|
||||||
@catch_consul_errors
|
@catch_consul_errors
|
||||||
|
def _write_failsafe(self, value):
|
||||||
|
return self._client.kv.put(self.failsafe_path, value)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _run_and_handle_exceptions(method, *args, **kwargs):
|
||||||
|
retry = kwargs.pop('retry', None)
|
||||||
|
try:
|
||||||
|
return retry(method, *args, **kwargs) if retry else method(*args, **kwargs)
|
||||||
|
except (RetryFailedError, InvalidSession, HTTPException, HTTPError, socket.error, socket.timeout) as e:
|
||||||
|
raise ConsulError(e)
|
||||||
|
except ConsulException:
|
||||||
|
raise ReturnFalseException
|
||||||
|
|
||||||
|
@catch_return_false_exception
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
|
retry = self._retry.copy()
|
||||||
|
|
||||||
|
self._run_and_handle_exceptions(self._do_refresh_session, True, retry=retry)
|
||||||
|
|
||||||
if self._session:
|
if self._session:
|
||||||
self.retry(self._client.session.renew, self._session)
|
cluster = self.cluster
|
||||||
self._last_session_refresh = time.time()
|
leader_session = cluster and isinstance(cluster.leader, Leader) and cluster.leader.session
|
||||||
|
if leader_session != self._session:
|
||||||
|
retry.deadline = retry.stoptime - time.time()
|
||||||
|
if retry.deadline < 1:
|
||||||
|
raise ConsulError('update_leader timeout')
|
||||||
|
logger.warning('Recreating the leader key due to session mismatch')
|
||||||
|
if cluster.leader:
|
||||||
|
self._run_and_handle_exceptions(self._client.kv.delete, self.leader_path, cas=cluster.leader.index)
|
||||||
|
|
||||||
|
retry.deadline = retry.stoptime - time.time()
|
||||||
|
if retry.deadline < 0.5:
|
||||||
|
raise ConsulError('update_leader timeout')
|
||||||
|
self._run_and_handle_exceptions(self._client.kv.put, self.leader_path,
|
||||||
|
self._name, acquire=self._session)
|
||||||
|
|
||||||
return bool(self._session)
|
return bool(self._session)
|
||||||
|
|
||||||
@catch_consul_errors
|
@catch_consul_errors
|
||||||
|
|||||||
+44
-8
@@ -19,7 +19,8 @@ from six.moves.http_client import HTTPException
|
|||||||
from six.moves.urllib_parse import urlparse
|
from six.moves.urllib_parse import urlparse
|
||||||
from threading import Thread
|
from threading import Thread
|
||||||
|
|
||||||
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member, SyncState, TimelineHistory
|
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member,\
|
||||||
|
SyncState, TimelineHistory, ReturnFalseException, catch_return_false_exception
|
||||||
from ..exceptions import DCSError
|
from ..exceptions import DCSError
|
||||||
from ..request import get as requests_get
|
from ..request import get as requests_get
|
||||||
from ..utils import Retry, RetryFailedError, split_host_port, uri, USER_AGENT
|
from ..utils import Retry, RetryFailedError, split_host_port, uri, USER_AGENT
|
||||||
@@ -216,9 +217,12 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
|||||||
return response
|
return response
|
||||||
except (HTTPError, HTTPException, socket.error, socket.timeout) as e:
|
except (HTTPError, HTTPException, socket.error, socket.timeout) as e:
|
||||||
self.http.clear()
|
self.http.clear()
|
||||||
|
if not retry:
|
||||||
|
if len(machines_cache) == 1:
|
||||||
|
self.set_base_uri(self._base_uri) # trigger Etcd3 watcher restart
|
||||||
# switch to the next etcd node because we don't know exactly what happened,
|
# switch to the next etcd node because we don't know exactly what happened,
|
||||||
# whether the key didn't received an update or there is a network problem.
|
# whether the key didn't received an update or there is a network problem.
|
||||||
if not retry and i + 1 < len(machines_cache):
|
elif i + 1 < len(machines_cache):
|
||||||
self.set_base_uri(machines_cache[i + 1])
|
self.set_base_uri(machines_cache[i + 1])
|
||||||
if (isinstance(fields, dict) and fields.get("wait") == "true" and
|
if (isinstance(fields, dict) and fields.get("wait") == "true" and
|
||||||
isinstance(e, (ReadTimeoutError, ProtocolError))):
|
isinstance(e, (ReadTimeoutError, ProtocolError))):
|
||||||
@@ -457,6 +461,18 @@ class AbstractEtcd(AbstractDCS):
|
|||||||
if isinstance(raise_ex, Exception):
|
if isinstance(raise_ex, Exception):
|
||||||
raise raise_ex
|
raise raise_ex
|
||||||
|
|
||||||
|
def _run_and_handle_exceptions(self, method, *args, **kwargs):
|
||||||
|
retry = kwargs.pop('retry', self.retry)
|
||||||
|
try:
|
||||||
|
return retry(method, *args, **kwargs) if retry else method(*args, **kwargs)
|
||||||
|
except (RetryFailedError, etcd.EtcdConnectionFailed) as e:
|
||||||
|
raise self._client.ERROR_CLS(e)
|
||||||
|
except etcd.EtcdException as e:
|
||||||
|
self._handle_exception(e)
|
||||||
|
raise ReturnFalseException
|
||||||
|
except Exception as e:
|
||||||
|
self._handle_exception(e, raise_ex=self._client.ERROR_CLS('unexpected error'))
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def set_socket_options(sock, socket_options):
|
def set_socket_options(sock, socket_options):
|
||||||
if socket_options:
|
if socket_options:
|
||||||
@@ -645,9 +661,16 @@ class Etcd(AbstractEtcd):
|
|||||||
sync = nodes.get(self._SYNC)
|
sync = nodes.get(self._SYNC)
|
||||||
sync = SyncState.from_node(sync and sync.modifiedIndex, sync and sync.value)
|
sync = SyncState.from_node(sync and sync.modifiedIndex, sync and sync.value)
|
||||||
|
|
||||||
cluster = Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
# get failsafe topology
|
||||||
|
failsafe = nodes.get(self._FAILSAFE)
|
||||||
|
try:
|
||||||
|
failsafe = json.loads(failsafe.value) if failsafe else None
|
||||||
|
except Exception:
|
||||||
|
failsafe = None
|
||||||
|
|
||||||
|
cluster = Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots, failsafe)
|
||||||
except etcd.EtcdKeyNotFound:
|
except etcd.EtcdKeyNotFound:
|
||||||
cluster = Cluster(None, None, None, None, [], None, None, None, None)
|
cluster = Cluster(None, None, None, None, [], None, None, None, None, None)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
self._handle_exception(e, 'get_cluster', raise_ex=EtcdError('Etcd is not responding properly'))
|
self._handle_exception(e, 'get_cluster', raise_ex=EtcdError('Etcd is not responding properly'))
|
||||||
self._has_failed = False
|
self._has_failed = False
|
||||||
@@ -662,7 +685,7 @@ class Etcd(AbstractEtcd):
|
|||||||
def take_leader(self):
|
def take_leader(self):
|
||||||
return self.retry(self._client.write, self.leader_path, self._name, ttl=self._ttl)
|
return self.retry(self._client.write, self.leader_path, self._name, ttl=self._ttl)
|
||||||
|
|
||||||
def attempt_to_acquire_leader(self, permanent=False):
|
def _do_attempt_to_acquire_leader(self, permanent=False):
|
||||||
try:
|
try:
|
||||||
return bool(self.retry(self._client.write,
|
return bool(self.retry(self._client.write,
|
||||||
self.leader_path,
|
self.leader_path,
|
||||||
@@ -671,10 +694,12 @@ class Etcd(AbstractEtcd):
|
|||||||
prevExist=False))
|
prevExist=False))
|
||||||
except etcd.EtcdAlreadyExist:
|
except etcd.EtcdAlreadyExist:
|
||||||
logger.info('Could not take out TTL lock')
|
logger.info('Could not take out TTL lock')
|
||||||
except (RetryFailedError, etcd.EtcdException):
|
|
||||||
pass
|
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
@catch_return_false_exception
|
||||||
|
def attempt_to_acquire_leader(self, permanent=False):
|
||||||
|
return self._run_and_handle_exceptions(self._do_attempt_to_acquire_leader, permanent=permanent, retry=None)
|
||||||
|
|
||||||
@catch_etcd_errors
|
@catch_etcd_errors
|
||||||
def set_failover_value(self, value, index=None):
|
def set_failover_value(self, value, index=None):
|
||||||
return self._client.write(self.failover_path, value, prevIndex=index or 0)
|
return self._client.write(self.failover_path, value, prevIndex=index or 0)
|
||||||
@@ -691,9 +716,20 @@ class Etcd(AbstractEtcd):
|
|||||||
def _write_status(self, value):
|
def _write_status(self, value):
|
||||||
return self._client.set(self.status_path, value)
|
return self._client.set(self.status_path, value)
|
||||||
|
|
||||||
|
def _do_update_leader(self):
|
||||||
|
try:
|
||||||
|
return self.retry(self._client.write, self.leader_path, self._name,
|
||||||
|
prevValue=self._name, ttl=self._ttl) is not None
|
||||||
|
except etcd.EtcdKeyNotFound:
|
||||||
|
return self._do_attempt_to_acquire_leader()
|
||||||
|
|
||||||
@catch_etcd_errors
|
@catch_etcd_errors
|
||||||
|
def _write_failsafe(self, value):
|
||||||
|
return self._client.set(self.failsafe_path, value)
|
||||||
|
|
||||||
|
@catch_return_false_exception
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
return self.retry(self._client.write, self.leader_path, self._name, prevValue=self._name, ttl=self._ttl)
|
return self._run_and_handle_exceptions(self._do_update_leader, retry=None)
|
||||||
|
|
||||||
@catch_etcd_errors
|
@catch_etcd_errors
|
||||||
def initialize(self, create_new=True, sysid=""):
|
def initialize(self, create_new=True, sysid=""):
|
||||||
|
|||||||
+77
-23
@@ -11,8 +11,10 @@ import time
|
|||||||
import urllib3
|
import urllib3
|
||||||
|
|
||||||
from threading import Condition, Lock, Thread
|
from threading import Condition, Lock, Thread
|
||||||
|
from urllib3.exceptions import ReadTimeoutError, ProtocolError
|
||||||
|
|
||||||
from . import ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
from . import ClusterConfig, Cluster, Failover, Leader, Member,\
|
||||||
|
SyncState, TimelineHistory, ReturnFalseException, catch_return_false_exception
|
||||||
from .etcd import AbstractEtcdClientWithFailover, AbstractEtcd, catch_etcd_errors
|
from .etcd import AbstractEtcdClientWithFailover, AbstractEtcd, catch_etcd_errors
|
||||||
from ..exceptions import DCSError, PatroniException
|
from ..exceptions import DCSError, PatroniException
|
||||||
from ..utils import deep_compare, enable_keepalive, iter_response_objects, RetryFailedError, USER_AGENT
|
from ..utils import deep_compare, enable_keepalive, iter_response_objects, RetryFailedError, USER_AGENT
|
||||||
@@ -350,7 +352,7 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
|
|||||||
def deleteprefix(self, key, retry=None):
|
def deleteprefix(self, key, retry=None):
|
||||||
return self.deleterange(key, prefix_range_end(key), retry=retry)
|
return self.deleterange(key, prefix_range_end(key), retry=retry)
|
||||||
|
|
||||||
def watchrange(self, key, range_end=None, start_revision=None, filters=None):
|
def watchrange(self, key, range_end=None, start_revision=None, filters=None, read_timeout=None):
|
||||||
"""returns: response object"""
|
"""returns: response object"""
|
||||||
params = build_range_request(key, range_end)
|
params = build_range_request(key, range_end)
|
||||||
if start_revision is not None:
|
if start_revision is not None:
|
||||||
@@ -358,11 +360,11 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
|
|||||||
params['filters'] = filters or []
|
params['filters'] = filters or []
|
||||||
kwargs = self._prepare_common_parameters(1, self.read_timeout)
|
kwargs = self._prepare_common_parameters(1, self.read_timeout)
|
||||||
request_executor = self._prepare_request(kwargs, {'create_request': params})
|
request_executor = self._prepare_request(kwargs, {'create_request': params})
|
||||||
kwargs.update(timeout=urllib3.Timeout(connect=kwargs['timeout']), retries=0)
|
kwargs.update(timeout=urllib3.Timeout(connect=kwargs['timeout'], read=read_timeout), retries=0)
|
||||||
return request_executor(self._MPOST, self._base_uri + self.version_prefix + '/watch', **kwargs)
|
return request_executor(self._MPOST, self._base_uri + self.version_prefix + '/watch', **kwargs)
|
||||||
|
|
||||||
def watchprefix(self, key, start_revision=None, filters=None):
|
def watchprefix(self, key, start_revision=None, filters=None, read_timeout=None):
|
||||||
return self.watchrange(key, prefix_range_end(key), start_revision, filters)
|
return self.watchrange(key, prefix_range_end(key), start_revision, filters, read_timeout)
|
||||||
|
|
||||||
|
|
||||||
class KVCache(Thread):
|
class KVCache(Thread):
|
||||||
@@ -451,7 +453,14 @@ class KVCache(Thread):
|
|||||||
def _do_watch(self, revision):
|
def _do_watch(self, revision):
|
||||||
with self._response_lock:
|
with self._response_lock:
|
||||||
self._response = None
|
self._response = None
|
||||||
response = self._client.watchprefix(self._dcs.cluster_prefix, revision)
|
# We do most of requests with timeouts. The only exception /watch requests to Etcd v3.
|
||||||
|
# In order to interrupt the /watch request we do socket.shutdown() from the main thread,
|
||||||
|
# which doesn't work on Windows. Therefore we want to use the last resort, `read_timeout`.
|
||||||
|
# Setting it to TTL will help to partially mitigate the problem.
|
||||||
|
# Setting it to lower value is not nice because for idling clusters it will increase
|
||||||
|
# the numbers of interrupts and reconnects.
|
||||||
|
read_timeout = self._dcs.ttl if os.name == 'nt' else None
|
||||||
|
response = self._client.watchprefix(self._dcs.cluster_prefix, revision, read_timeout=read_timeout)
|
||||||
with self._response_lock:
|
with self._response_lock:
|
||||||
if self._response is None:
|
if self._response is None:
|
||||||
self._response = response
|
self._response = response
|
||||||
@@ -473,6 +482,8 @@ class KVCache(Thread):
|
|||||||
try:
|
try:
|
||||||
self._do_watch(result['header']['revision'])
|
self._do_watch(result['header']['revision'])
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
|
# Following exceptions are expected on Windows because the /watch request is done with `read_timeout`
|
||||||
|
if not (os.name == 'nt' and isinstance(e, (ReadTimeoutError, ProtocolError))):
|
||||||
logger.error('watchprefix failed: %r', e)
|
logger.error('watchprefix failed: %r', e)
|
||||||
finally:
|
finally:
|
||||||
with self.condition:
|
with self.condition:
|
||||||
@@ -599,8 +610,8 @@ class Etcd3(AbstractEtcd):
|
|||||||
if self.__do_not_watch:
|
if self.__do_not_watch:
|
||||||
self._lease = None
|
self._lease = None
|
||||||
|
|
||||||
def _do_refresh_lease(self, retry=None):
|
def _do_refresh_lease(self, force=False, retry=None):
|
||||||
if self._lease and self._last_lease_refresh + self._loop_wait > time.time():
|
if not force and self._lease and self._last_lease_refresh + self._loop_wait > time.time():
|
||||||
return False
|
return False
|
||||||
|
|
||||||
if self._lease and not self._client.lease_keepalive(self._lease, retry):
|
if self._lease and not self._client.lease_keepalive(self._lease, retry):
|
||||||
@@ -701,7 +712,14 @@ class Etcd3(AbstractEtcd):
|
|||||||
sync = nodes.get(self._SYNC)
|
sync = nodes.get(self._SYNC)
|
||||||
sync = SyncState.from_node(sync and sync['mod_revision'], sync and sync['value'])
|
sync = SyncState.from_node(sync and sync['mod_revision'], sync and sync['value'])
|
||||||
|
|
||||||
cluster = Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
# get failsafe topology
|
||||||
|
failsafe = nodes.get(self._FAILSAFE)
|
||||||
|
try:
|
||||||
|
failsafe = json.loads(failsafe['value']) if failsafe else None
|
||||||
|
except Exception:
|
||||||
|
failsafe = None
|
||||||
|
|
||||||
|
cluster = Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots, failsafe)
|
||||||
except UnsupportedEtcdVersion:
|
except UnsupportedEtcdVersion:
|
||||||
raise
|
raise
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
@@ -734,21 +752,42 @@ class Etcd3(AbstractEtcd):
|
|||||||
def take_leader(self):
|
def take_leader(self):
|
||||||
return self.retry(self._client.put, self.leader_path, self._name, self._lease)
|
return self.retry(self._client.put, self.leader_path, self._name, self._lease)
|
||||||
|
|
||||||
@catch_etcd_errors
|
def _do_attempt_to_acquire_leader(self, permanent, retry):
|
||||||
def _do_attempt_to_acquire_leader(self, permanent):
|
def _retry(*args, **kwargs):
|
||||||
|
kwargs['retry'] = retry
|
||||||
|
return retry(*args, **kwargs)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
return self.retry(self._client.put, self.leader_path, self._name, None if permanent else self._lease, 0)
|
return _retry(self._client.put, self.leader_path, self._name, None if permanent else self._lease, 0)
|
||||||
except LeaseNotFound:
|
except LeaseNotFound:
|
||||||
self._lease = None
|
|
||||||
logger.error('Our lease disappeared from Etcd. Will try to get a new one and retry attempt')
|
logger.error('Our lease disappeared from Etcd. Will try to get a new one and retry attempt')
|
||||||
self.refresh_lease()
|
self._lease = None
|
||||||
return self.retry(self._client.put, self.leader_path, self._name, None if permanent else self._lease, 0)
|
retry.deadline = retry.stoptime - time.time()
|
||||||
|
|
||||||
|
_retry(self._do_refresh_lease)
|
||||||
|
|
||||||
|
retry.deadline = retry.stoptime - time.time()
|
||||||
|
if retry.deadline < 1:
|
||||||
|
raise Etcd3Error('_do_attempt_to_acquire_leader timeout')
|
||||||
|
|
||||||
|
return _retry(self._client.put, self.leader_path, self._name, None if permanent else self._lease, 0)
|
||||||
|
|
||||||
|
@catch_return_false_exception
|
||||||
def attempt_to_acquire_leader(self, permanent=False):
|
def attempt_to_acquire_leader(self, permanent=False):
|
||||||
if not self._lease and not permanent:
|
retry = self._retry.copy()
|
||||||
self.refresh_lease()
|
|
||||||
|
|
||||||
ret = self._do_attempt_to_acquire_leader(permanent)
|
def _retry(*args, **kwargs):
|
||||||
|
kwargs['retry'] = retry
|
||||||
|
return retry(*args, **kwargs)
|
||||||
|
|
||||||
|
if not permanent:
|
||||||
|
self._run_and_handle_exceptions(self._do_refresh_lease, retry=_retry)
|
||||||
|
|
||||||
|
retry.deadline = retry.stoptime - time.time()
|
||||||
|
if retry.deadline < 1:
|
||||||
|
raise Etcd3Error('attempt_to_acquire_leader timeout')
|
||||||
|
|
||||||
|
ret = self._run_and_handle_exceptions(self._do_attempt_to_acquire_leader, permanent, retry, retry=None)
|
||||||
if not ret:
|
if not ret:
|
||||||
logger.info('Could not take out TTL lock')
|
logger.info('Could not take out TTL lock')
|
||||||
return ret
|
return ret
|
||||||
@@ -770,17 +809,32 @@ class Etcd3(AbstractEtcd):
|
|||||||
return self._client.put(self.status_path, value)
|
return self._client.put(self.status_path, value)
|
||||||
|
|
||||||
@catch_etcd_errors
|
@catch_etcd_errors
|
||||||
|
def _write_failsafe(self, value):
|
||||||
|
return self._client.put(self.failsafe_path, value)
|
||||||
|
|
||||||
|
@catch_return_false_exception
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
if not self._lease:
|
retry = self._retry.copy()
|
||||||
self.refresh_lease()
|
|
||||||
elif self.retry(self._client.lease_keepalive, self._lease):
|
def _retry(*args, **kwargs):
|
||||||
self._last_lease_refresh = time.time()
|
kwargs['retry'] = retry
|
||||||
|
return retry(*args, **kwargs)
|
||||||
|
|
||||||
|
self._run_and_handle_exceptions(self._do_refresh_lease, True, retry=_retry)
|
||||||
|
|
||||||
if self._lease:
|
if self._lease:
|
||||||
cluster = self.cluster
|
cluster = self.cluster
|
||||||
leader_lease = cluster and isinstance(cluster.leader, Leader) and cluster.leader.session
|
leader_lease = cluster and isinstance(cluster.leader, Leader) and cluster.leader.session
|
||||||
if leader_lease != self._lease:
|
if leader_lease != self._lease:
|
||||||
self.take_leader()
|
retry.deadline = retry.stoptime - time.time()
|
||||||
|
if retry.deadline < 1:
|
||||||
|
raise Etcd3Error('update_leader timeout')
|
||||||
|
|
||||||
|
try:
|
||||||
|
self._run_and_handle_exceptions(self._client.put, self.leader_path,
|
||||||
|
self._name, self._lease, retry=_retry)
|
||||||
|
except ReturnFalseException:
|
||||||
|
pass
|
||||||
return bool(self._lease)
|
return bool(self._lease)
|
||||||
|
|
||||||
@catch_etcd_errors
|
@catch_etcd_errors
|
||||||
|
|||||||
+91
-23
@@ -1,3 +1,5 @@
|
|||||||
|
import atexit
|
||||||
|
import base64
|
||||||
import datetime
|
import datetime
|
||||||
import functools
|
import functools
|
||||||
import json
|
import json
|
||||||
@@ -7,6 +9,7 @@ import random
|
|||||||
import socket
|
import socket
|
||||||
import six
|
import six
|
||||||
import sys
|
import sys
|
||||||
|
import tempfile
|
||||||
import time
|
import time
|
||||||
import urllib3
|
import urllib3
|
||||||
import yaml
|
import yaml
|
||||||
@@ -28,12 +31,34 @@ SERVICE_HOST_ENV_NAME = 'KUBERNETES_SERVICE_HOST'
|
|||||||
SERVICE_PORT_ENV_NAME = 'KUBERNETES_SERVICE_PORT'
|
SERVICE_PORT_ENV_NAME = 'KUBERNETES_SERVICE_PORT'
|
||||||
SERVICE_TOKEN_FILENAME = '/var/run/secrets/kubernetes.io/serviceaccount/token'
|
SERVICE_TOKEN_FILENAME = '/var/run/secrets/kubernetes.io/serviceaccount/token'
|
||||||
SERVICE_CERT_FILENAME = '/var/run/secrets/kubernetes.io/serviceaccount/ca.crt'
|
SERVICE_CERT_FILENAME = '/var/run/secrets/kubernetes.io/serviceaccount/ca.crt'
|
||||||
|
__temp_files = []
|
||||||
|
|
||||||
|
|
||||||
class KubernetesError(DCSError):
|
class KubernetesError(DCSError):
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def _cleanup_temp_files():
|
||||||
|
global __temp_files
|
||||||
|
for temp_file in __temp_files:
|
||||||
|
try:
|
||||||
|
os.remove(temp_file)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
__temp_files = []
|
||||||
|
|
||||||
|
|
||||||
|
def _create_temp_file(content):
|
||||||
|
if len(__temp_files) == 0:
|
||||||
|
atexit.register(_cleanup_temp_files)
|
||||||
|
|
||||||
|
fd, name = tempfile.mkstemp()
|
||||||
|
os.write(fd, content)
|
||||||
|
os.close(fd)
|
||||||
|
__temp_files.append(name)
|
||||||
|
return name
|
||||||
|
|
||||||
|
|
||||||
# this function does the same mapping of snake_case => camelCase for > 97% of cases as autogenerated swagger code
|
# this function does the same mapping of snake_case => camelCase for > 97% of cases as autogenerated swagger code
|
||||||
def to_camel_case(value):
|
def to_camel_case(value):
|
||||||
reserved = {'api', 'apiv3', 'cidr', 'cpu', 'csi', 'id', 'io', 'ip', 'ipc', 'pid', 'tls', 'uri', 'url', 'uuid'}
|
reserved = {'api', 'apiv3', 'cidr', 'cpu', 'csi', 'id', 'io', 'ip', 'ipc', 'pid', 'tls', 'uri', 'url', 'uuid'}
|
||||||
@@ -93,6 +118,13 @@ class K8sConfig(object):
|
|||||||
if c['name'] == name:
|
if c['name'] == name:
|
||||||
return c[section]
|
return c[section]
|
||||||
|
|
||||||
|
def _pool_config_from_file_or_data(self, config, file_key_name, pool_key_name):
|
||||||
|
data_key_name = file_key_name + '-data'
|
||||||
|
if data_key_name in config:
|
||||||
|
self.pool_config[pool_key_name] = _create_temp_file(base64.b64decode(config[data_key_name]))
|
||||||
|
elif file_key_name in config:
|
||||||
|
self.pool_config[pool_key_name] = config[file_key_name]
|
||||||
|
|
||||||
def load_kube_config(self, context=None):
|
def load_kube_config(self, context=None):
|
||||||
with open(os.path.expanduser(KUBE_CONFIG_DEFAULT_LOCATION)) as f:
|
with open(os.path.expanduser(KUBE_CONFIG_DEFAULT_LOCATION)) as f:
|
||||||
config = yaml.safe_load(f)
|
config = yaml.safe_load(f)
|
||||||
@@ -103,10 +135,9 @@ class K8sConfig(object):
|
|||||||
|
|
||||||
self._server = cluster['server'].rstrip('/')
|
self._server = cluster['server'].rstrip('/')
|
||||||
if self._server.startswith('https'):
|
if self._server.startswith('https'):
|
||||||
self.pool_config.update({v: user[k] for k, v in {'client-certificate': 'cert_file',
|
self._pool_config_from_file_or_data(user, 'client-certificate', 'cert_file')
|
||||||
'client-key': 'key_file'}.items() if k in user})
|
self._pool_config_from_file_or_data(user, 'client-key', 'key_file')
|
||||||
if 'certificate-authority' in cluster:
|
self._pool_config_from_file_or_data(cluster, 'certificate-authority', 'ca_certs')
|
||||||
self.pool_config['ca_certs'] = cluster['certificate-authority']
|
|
||||||
self.pool_config['cert_reqs'] = 'CERT_NONE' if cluster.get('insecure-skip-tls-verify') else 'CERT_REQUIRED'
|
self.pool_config['cert_reqs'] = 'CERT_NONE' if cluster.get('insecure-skip-tls-verify') else 'CERT_REQUIRED'
|
||||||
if user.get('token'):
|
if user.get('token'):
|
||||||
self._make_headers(token=user['token'])
|
self._make_headers(token=user['token'])
|
||||||
@@ -493,16 +524,10 @@ class CoreV1ApiProxy(object):
|
|||||||
|
|
||||||
|
|
||||||
def catch_kubernetes_errors(func):
|
def catch_kubernetes_errors(func):
|
||||||
def wrapper(*args, **kwargs):
|
def wrapper(self, *args, **kwargs):
|
||||||
try:
|
try:
|
||||||
return func(*args, **kwargs)
|
return self._run_and_handle_exceptions(func, self, *args, **kwargs)
|
||||||
except k8s_client.rest.ApiException as e:
|
except KubernetesError:
|
||||||
if e.status == 403:
|
|
||||||
logger.exception('Permission denied')
|
|
||||||
elif e.status != 409: # Object exists or conflict in resource_version
|
|
||||||
logger.exception('Unexpected error from Kubernetes API')
|
|
||||||
return False
|
|
||||||
except (RetryFailedError, K8sException):
|
|
||||||
return False
|
return False
|
||||||
return wrapper
|
return wrapper
|
||||||
|
|
||||||
@@ -677,7 +702,7 @@ class Kubernetes(AbstractDCS):
|
|||||||
try:
|
try:
|
||||||
k8s_config.load_incluster_config(ca_certs=self._ca_certs)
|
k8s_config.load_incluster_config(ca_certs=self._ca_certs)
|
||||||
except k8s_config.ConfigException:
|
except k8s_config.ConfigException:
|
||||||
k8s_config.load_kube_config(context=config.get('context', 'local'))
|
k8s_config.load_kube_config(context=config.get('context', 'kind-kind'))
|
||||||
|
|
||||||
self.__my_pod = None
|
self.__my_pod = None
|
||||||
self.__ips = [] if config.get('patronictl') else [config.get('pod_ip')]
|
self.__ips = [] if config.get('patronictl') else [config.get('pod_ip')]
|
||||||
@@ -712,6 +737,19 @@ class Kubernetes(AbstractDCS):
|
|||||||
kwargs['_retry'] = retry
|
kwargs['_retry'] = retry
|
||||||
return retry(*args, **kwargs)
|
return retry(*args, **kwargs)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _run_and_handle_exceptions(method, *args, **kwargs):
|
||||||
|
try:
|
||||||
|
return method(*args, **kwargs)
|
||||||
|
except k8s_client.rest.ApiException as e:
|
||||||
|
if e.status == 403:
|
||||||
|
logger.exception('Permission denied')
|
||||||
|
elif e.status != 409: # Object exists or conflict in resource_version
|
||||||
|
logger.exception('Unexpected error from Kubernetes API')
|
||||||
|
return False
|
||||||
|
except (RetryFailedError, K8sException) as e:
|
||||||
|
raise KubernetesError(e)
|
||||||
|
|
||||||
def client_path(self, path):
|
def client_path(self, path):
|
||||||
return super(Kubernetes, self).client_path(path)[1:].replace('/', '-')
|
return super(Kubernetes, self).client_path(path)[1:].replace('/', '-')
|
||||||
|
|
||||||
@@ -794,6 +832,13 @@ class Kubernetes(AbstractDCS):
|
|||||||
except Exception:
|
except Exception:
|
||||||
slots = None
|
slots = None
|
||||||
|
|
||||||
|
# get failsafe topology
|
||||||
|
failsafe = annotations.get(self._FAILSAFE)
|
||||||
|
try:
|
||||||
|
failsafe = json.loads(failsafe) if failsafe else None
|
||||||
|
except Exception:
|
||||||
|
failsafe = None
|
||||||
|
|
||||||
# get leader
|
# get leader
|
||||||
leader_record = {n: annotations.get(n) for n in (self._LEADER, 'acquireTime',
|
leader_record = {n: annotations.get(n) for n in (self._LEADER, 'acquireTime',
|
||||||
'ttl', 'renewTime', 'transitions') if n in annotations}
|
'ttl', 'renewTime', 'transitions') if n in annotations}
|
||||||
@@ -826,7 +871,7 @@ class Kubernetes(AbstractDCS):
|
|||||||
metadata = sync and sync.metadata
|
metadata = sync and sync.metadata
|
||||||
sync = SyncState.from_node(metadata and metadata.resource_version, metadata and metadata.annotations)
|
sync = SyncState.from_node(metadata and metadata.resource_version, metadata and metadata.annotations)
|
||||||
|
|
||||||
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots, failsafe)
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('get_cluster')
|
logger.exception('get_cluster')
|
||||||
raise KubernetesError('Kubernetes API is not responding properly')
|
raise KubernetesError('Kubernetes API is not responding properly')
|
||||||
@@ -950,7 +995,8 @@ class Kubernetes(AbstractDCS):
|
|||||||
if not self._api.create_namespaced_service(self._namespace, body):
|
if not self._api.create_namespaced_service(self._namespace, body):
|
||||||
return
|
return
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
if not isinstance(e, k8s_client.rest.ApiException) or e.status != 409: # Service already exists
|
# 409 - service already exists, 403 - creation forbidden
|
||||||
|
if not isinstance(e, k8s_client.rest.ApiException) or e.status not in (409, 403):
|
||||||
return logger.exception('create_config_service failed')
|
return logger.exception('create_config_service failed')
|
||||||
self._should_create_config_service = False
|
self._should_create_config_service = False
|
||||||
|
|
||||||
@@ -960,6 +1006,9 @@ class Kubernetes(AbstractDCS):
|
|||||||
def _write_status(self, value):
|
def _write_status(self, value):
|
||||||
"""Unused"""
|
"""Unused"""
|
||||||
|
|
||||||
|
def _write_failsafe(self, value):
|
||||||
|
"""Unused"""
|
||||||
|
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
"""Unused"""
|
"""Unused"""
|
||||||
|
|
||||||
@@ -978,16 +1027,19 @@ class Kubernetes(AbstractDCS):
|
|||||||
else:
|
else:
|
||||||
logger.exception('Permission denied' if e.status == 403 else 'Unexpected error from Kubernetes API')
|
logger.exception('Permission denied' if e.status == 403 else 'Unexpected error from Kubernetes API')
|
||||||
return False
|
return False
|
||||||
except (RetryFailedError, K8sException):
|
except (RetryFailedError, K8sException) as e:
|
||||||
return False
|
raise KubernetesError(e)
|
||||||
|
|
||||||
|
# if we are here, that means update failed with 409
|
||||||
retry.deadline = retry.stoptime - time.time()
|
retry.deadline = retry.stoptime - time.time()
|
||||||
if retry.deadline < 1:
|
if retry.deadline < 1:
|
||||||
return False
|
return False # No time for retry. Tell ha.py that we have to demote due to failed update.
|
||||||
|
|
||||||
# Try to get the latest version directly from K8s API instead of relying on async cache
|
# Try to get the latest version directly from K8s API instead of relying on async cache
|
||||||
try:
|
try:
|
||||||
kind = _retry(self._api.read_namespaced_kind, self.leader_path, self._namespace)
|
kind = _retry(self._api.read_namespaced_kind, self.leader_path, self._namespace)
|
||||||
|
except (RetryFailedError, K8sException) as e:
|
||||||
|
raise KubernetesError(e)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error('Failed to get the leader object "%s": %r', self.leader_path, e)
|
logger.error('Failed to get the leader object "%s": %r', self.leader_path, e)
|
||||||
return False
|
return False
|
||||||
@@ -1005,9 +1057,10 @@ class Kubernetes(AbstractDCS):
|
|||||||
if kind and (kind_annotations.get(self._LEADER) != self._name or kind_resource_version == resource_version):
|
if kind and (kind_annotations.get(self._LEADER) != self._name or kind_resource_version == resource_version):
|
||||||
return False
|
return False
|
||||||
|
|
||||||
return self.patch_or_create(self.leader_path, annotations, kind_resource_version, ips=ips, retry=_retry)
|
return self._run_and_handle_exceptions(self._patch_or_create, self.leader_path, annotations,
|
||||||
|
kind_resource_version, ips=ips, retry=_retry)
|
||||||
|
|
||||||
def update_leader(self, last_lsn, slots=None):
|
def update_leader(self, last_lsn, slots=None, failsafe=None):
|
||||||
kind = self._kinds.get(self.leader_path)
|
kind = self._kinds.get(self.leader_path)
|
||||||
kind_annotations = kind and kind.metadata.annotations or {}
|
kind_annotations = kind and kind.metadata.annotations or {}
|
||||||
|
|
||||||
@@ -1021,7 +1074,10 @@ class Kubernetes(AbstractDCS):
|
|||||||
'transitions': leader_observed_record.get('transitions') or '0'}
|
'transitions': leader_observed_record.get('transitions') or '0'}
|
||||||
if last_lsn:
|
if last_lsn:
|
||||||
annotations[self._OPTIME] = str(last_lsn)
|
annotations[self._OPTIME] = str(last_lsn)
|
||||||
annotations['slots'] = json.dumps(slots) if slots else None
|
annotations['slots'] = json.dumps(slots, separators=(',', ':')) if slots else None
|
||||||
|
|
||||||
|
if failsafe is not None:
|
||||||
|
annotations[self._FAILSAFE] = json.dumps(failsafe, separators=(',', ':')) if failsafe else None
|
||||||
|
|
||||||
resource_version = kind and kind.metadata.resource_version
|
resource_version = kind and kind.metadata.resource_version
|
||||||
return self._update_leader_with_retry(annotations, resource_version, self.__ips)
|
return self._update_leader_with_retry(annotations, resource_version, self.__ips)
|
||||||
@@ -1042,7 +1098,19 @@ class Kubernetes(AbstractDCS):
|
|||||||
annotations['acquireTime'] = self._leader_observed_record.get('acquireTime') or now
|
annotations['acquireTime'] = self._leader_observed_record.get('acquireTime') or now
|
||||||
annotations['transitions'] = str(transitions)
|
annotations['transitions'] = str(transitions)
|
||||||
ips = [] if self._api.use_endpoints else None
|
ips = [] if self._api.use_endpoints else None
|
||||||
ret = self.patch_or_create(self.leader_path, annotations, self._leader_resource_version, ips=ips)
|
|
||||||
|
try:
|
||||||
|
ret = self._patch_or_create(self.leader_path, annotations,
|
||||||
|
self._leader_resource_version, retry=self.retry, ips=ips)
|
||||||
|
except k8s_client.rest.ApiException as e:
|
||||||
|
if e.status == 409 and self._leader_resource_version: # Conflict in resource_version
|
||||||
|
# Terminate watchers, it could be a sign that K8s API is in a failed state
|
||||||
|
self._kinds.kill_stream()
|
||||||
|
self._pods.kill_stream()
|
||||||
|
ret = False
|
||||||
|
except (RetryFailedError, K8sException) as e:
|
||||||
|
raise KubernetesError(e)
|
||||||
|
|
||||||
if not ret:
|
if not ret:
|
||||||
logger.info('Could not take out TTL lock')
|
logger.info('Could not take out TTL lock')
|
||||||
return ret
|
return ret
|
||||||
|
|||||||
+37
-14
@@ -11,11 +11,16 @@ from pysyncobj.transport import TCPTransport, CONNECTION_STATE
|
|||||||
from pysyncobj.utility import TcpUtility
|
from pysyncobj.utility import TcpUtility
|
||||||
|
|
||||||
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
||||||
|
from ..exceptions import DCSError
|
||||||
from ..utils import validate_directory
|
from ..utils import validate_directory
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class RaftError(DCSError):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
class _TCPTransport(TCPTransport):
|
class _TCPTransport(TCPTransport):
|
||||||
|
|
||||||
def __init__(self, syncObj, selfNode, otherNodes):
|
def __init__(self, syncObj, selfNode, otherNodes):
|
||||||
@@ -39,9 +44,9 @@ setattr(TCPNode, 'ip', property(resolve_host))
|
|||||||
|
|
||||||
class SyncObjUtility(object):
|
class SyncObjUtility(object):
|
||||||
|
|
||||||
def __init__(self, otherNodes, conf):
|
def __init__(self, otherNodes, conf, retry_timeout=10):
|
||||||
self._nodes = otherNodes
|
self._nodes = otherNodes
|
||||||
self._utility = TcpUtility(conf.password)
|
self._utility = TcpUtility(conf.password, retry_timeout/max(1, len(otherNodes)))
|
||||||
|
|
||||||
def executeCommand(self, command):
|
def executeCommand(self, command):
|
||||||
try:
|
try:
|
||||||
@@ -58,11 +63,11 @@ class SyncObjUtility(object):
|
|||||||
|
|
||||||
class DynMemberSyncObj(SyncObj):
|
class DynMemberSyncObj(SyncObj):
|
||||||
|
|
||||||
def __init__(self, selfAddress, partnerAddrs, conf):
|
def __init__(self, selfAddress, partnerAddrs, conf, retry_timeout=10):
|
||||||
self.__early_apply_local_log = selfAddress is not None
|
self.__early_apply_local_log = selfAddress is not None
|
||||||
self.applied_local_log = False
|
self.applied_local_log = False
|
||||||
|
|
||||||
utility = SyncObjUtility(partnerAddrs, conf)
|
utility = SyncObjUtility(partnerAddrs, conf, retry_timeout)
|
||||||
members = utility.getMembers()
|
members = utility.getMembers()
|
||||||
add_self = members and selfAddress not in members
|
add_self = members and selfAddress not in members
|
||||||
|
|
||||||
@@ -97,7 +102,7 @@ class KVStoreTTL(DynMemberSyncObj):
|
|||||||
self.__on_set = on_set
|
self.__on_set = on_set
|
||||||
self.__on_delete = on_delete
|
self.__on_delete = on_delete
|
||||||
self.__limb = {}
|
self.__limb = {}
|
||||||
self.__retry_timeout = None
|
self.set_retry_timeout(int(config.get('retry_timeout') or 10))
|
||||||
|
|
||||||
self_addr = config.get('self_addr')
|
self_addr = config.get('self_addr')
|
||||||
partner_addrs = set(config.get('partner_addrs', []))
|
partner_addrs = set(config.get('partner_addrs', []))
|
||||||
@@ -121,7 +126,7 @@ class KVStoreTTL(DynMemberSyncObj):
|
|||||||
journalFile=(file_template + '.journal' if self_addr else None),
|
journalFile=(file_template + '.journal' if self_addr else None),
|
||||||
onReady=on_ready, dynamicMembershipChange=True)
|
onReady=on_ready, dynamicMembershipChange=True)
|
||||||
|
|
||||||
super(KVStoreTTL, self).__init__(self_addr, partner_addrs, conf)
|
super(KVStoreTTL, self).__init__(self_addr, partner_addrs, conf, self.__retry_timeout)
|
||||||
self.__data = {}
|
self.__data = {}
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
@@ -156,7 +161,7 @@ class KVStoreTTL(DynMemberSyncObj):
|
|||||||
elif deadline:
|
elif deadline:
|
||||||
timeout = deadline - time.time()
|
timeout = deadline - time.time()
|
||||||
if timeout <= 0:
|
if timeout <= 0:
|
||||||
break
|
raise RaftError('timeout')
|
||||||
time.sleep(1)
|
time.sleep(1)
|
||||||
return False
|
return False
|
||||||
|
|
||||||
@@ -175,7 +180,7 @@ class KVStoreTTL(DynMemberSyncObj):
|
|||||||
self.__on_set(key, value)
|
self.__on_set(key, value)
|
||||||
return True
|
return True
|
||||||
|
|
||||||
def set(self, key, value, ttl=None, **kwargs):
|
def set(self, key, value, ttl=None, handle_raft_error=True, **kwargs):
|
||||||
old_value = self.__data.get(key, {})
|
old_value = self.__data.get(key, {})
|
||||||
if not self.__check_requirements(old_value, **kwargs):
|
if not self.__check_requirements(old_value, **kwargs):
|
||||||
return False
|
return False
|
||||||
@@ -184,7 +189,12 @@ class KVStoreTTL(DynMemberSyncObj):
|
|||||||
value['created'] = old_value.get('created', value['updated'])
|
value['created'] = old_value.get('created', value['updated'])
|
||||||
if ttl:
|
if ttl:
|
||||||
value['expire'] = value['updated'] + ttl
|
value['expire'] = value['updated'] + ttl
|
||||||
|
try:
|
||||||
return self.retry(self._set, key, value, **kwargs)
|
return self.retry(self._set, key, value, **kwargs)
|
||||||
|
except RaftError:
|
||||||
|
if not handle_raft_error:
|
||||||
|
raise
|
||||||
|
return False
|
||||||
|
|
||||||
def __pop(self, key):
|
def __pop(self, key):
|
||||||
self.__data.pop(key)
|
self.__data.pop(key)
|
||||||
@@ -206,7 +216,10 @@ class KVStoreTTL(DynMemberSyncObj):
|
|||||||
def delete(self, key, recursive=False, **kwargs):
|
def delete(self, key, recursive=False, **kwargs):
|
||||||
if not recursive and not self.__check_requirements(self.__data.get(key, {}), **kwargs):
|
if not recursive and not self.__check_requirements(self.__data.get(key, {}), **kwargs):
|
||||||
return False
|
return False
|
||||||
|
try:
|
||||||
return self.retry(self._delete, key, recursive=recursive, **kwargs)
|
return self.retry(self._delete, key, recursive=recursive, **kwargs)
|
||||||
|
except RaftError:
|
||||||
|
return False
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def __values_match(old, new):
|
def __values_match(old, new):
|
||||||
@@ -275,7 +288,6 @@ class Raft(AbstractDCS):
|
|||||||
break
|
break
|
||||||
else:
|
else:
|
||||||
logger.info('waiting on raft')
|
logger.info('waiting on raft')
|
||||||
self.set_retry_timeout(int(config.get('retry_timeout') or 10))
|
|
||||||
|
|
||||||
def _on_set(self, key, value):
|
def _on_set(self, key, value):
|
||||||
leader = (self._sync_obj.get(self.leader_path) or {}).get('value')
|
leader = (self._sync_obj.get(self.leader_path) or {}).get('value')
|
||||||
@@ -311,7 +323,7 @@ class Raft(AbstractDCS):
|
|||||||
prefix = self.client_path('')
|
prefix = self.client_path('')
|
||||||
response = self._sync_obj.get(prefix, recursive=True)
|
response = self._sync_obj.get(prefix, recursive=True)
|
||||||
if not response:
|
if not response:
|
||||||
return Cluster(None, None, None, None, [], None, None, None, None)
|
return Cluster(None, None, None, None, [], None, None, None, None, None)
|
||||||
nodes = {os.path.relpath(key, prefix).replace('\\', '/'): value for key, value in response.items()}
|
nodes = {os.path.relpath(key, prefix).replace('\\', '/'): value for key, value in response.items()}
|
||||||
|
|
||||||
# get initialize flag
|
# get initialize flag
|
||||||
@@ -364,7 +376,14 @@ class Raft(AbstractDCS):
|
|||||||
sync = nodes.get(self._SYNC)
|
sync = nodes.get(self._SYNC)
|
||||||
sync = SyncState.from_node(sync and sync['index'], sync and sync['value'])
|
sync = SyncState.from_node(sync and sync['index'], sync and sync['value'])
|
||||||
|
|
||||||
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
# get failsafe topology
|
||||||
|
failsafe = nodes.get(self._FAILSAFE)
|
||||||
|
try:
|
||||||
|
failsafe = json.loads(failsafe['value']) if failsafe else None
|
||||||
|
except Exception:
|
||||||
|
failsafe = None
|
||||||
|
|
||||||
|
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots, failsafe)
|
||||||
|
|
||||||
def _write_leader_optime(self, last_lsn):
|
def _write_leader_optime(self, last_lsn):
|
||||||
return self._sync_obj.set(self.leader_optime_path, last_lsn, timeout=1)
|
return self._sync_obj.set(self.leader_optime_path, last_lsn, timeout=1)
|
||||||
@@ -372,15 +391,19 @@ class Raft(AbstractDCS):
|
|||||||
def _write_status(self, value):
|
def _write_status(self, value):
|
||||||
return self._sync_obj.set(self.status_path, value, timeout=1)
|
return self._sync_obj.set(self.status_path, value, timeout=1)
|
||||||
|
|
||||||
|
def _write_failsafe(self, value):
|
||||||
|
return self._sync_obj.set(self.failsafe_path, value, timeout=1)
|
||||||
|
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
ret = self._sync_obj.set(self.leader_path, self._name, ttl=self._ttl, prevValue=self._name)
|
ret = self._sync_obj.set(self.leader_path, self._name, ttl=self._ttl,
|
||||||
|
handle_raft_error=False, prevValue=self._name)
|
||||||
if not ret and self._sync_obj.get(self.leader_path) is None:
|
if not ret and self._sync_obj.get(self.leader_path) is None:
|
||||||
ret = self.attempt_to_acquire_leader()
|
ret = self.attempt_to_acquire_leader()
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
def attempt_to_acquire_leader(self, permanent=False):
|
def attempt_to_acquire_leader(self, permanent=False):
|
||||||
return self._sync_obj.set(self.leader_path, self._name, prevExist=False,
|
return self._sync_obj.set(self.leader_path, self._name, ttl=None if permanent else self._ttl,
|
||||||
ttl=None if permanent else self._ttl)
|
handle_raft_error=False, prevExist=False)
|
||||||
|
|
||||||
def set_failover_value(self, value, index=None):
|
def set_failover_value(self, value, index=None):
|
||||||
return self._sync_obj.set(self.failover_path, value, prevIndex=index)
|
return self._sync_obj.set(self.failover_path, value, prevIndex=index)
|
||||||
|
|||||||
+61
-18
@@ -1,12 +1,14 @@
|
|||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import select
|
import select
|
||||||
|
import six
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from kazoo.client import KazooClient, KazooState, KazooRetry
|
from kazoo.client import KazooClient, KazooState, KazooRetry
|
||||||
from kazoo.exceptions import NoNodeError, NodeExistsError, SessionExpiredError
|
from kazoo.exceptions import ConnectionClosedError, NoNodeError, NodeExistsError, SessionExpiredError
|
||||||
from kazoo.handlers.threading import SequentialThreadingHandler
|
from kazoo.handlers.threading import SequentialThreadingHandler
|
||||||
from kazoo.protocol.states import KeeperState
|
from kazoo.protocol.states import KeeperState
|
||||||
|
from kazoo.retry import RetryFailedError
|
||||||
from kazoo.security import make_acl
|
from kazoo.security import make_acl
|
||||||
|
|
||||||
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
||||||
@@ -50,11 +52,21 @@ class PatroniSequentialThreadingHandler(SequentialThreadingHandler):
|
|||||||
return super(PatroniSequentialThreadingHandler, self).create_connection(*args, **kwargs)
|
return super(PatroniSequentialThreadingHandler, self).create_connection(*args, **kwargs)
|
||||||
|
|
||||||
def select(self, *args, **kwargs):
|
def select(self, *args, **kwargs):
|
||||||
"""Python3 raises `ValueError` if socket is closed, because fd == -1"""
|
"""
|
||||||
|
Python 3.XY may raise following exceptions if select/poll are called with an invalid socket:
|
||||||
|
- `ValueError`: because fd == -1
|
||||||
|
- `TypeError`: Invalid file descriptor: -1 (starting from kazoo 2.9)
|
||||||
|
Python 2.7 may raise the `IOError` instead of `socket.error` (starting from kazoo 2.9)
|
||||||
|
|
||||||
|
When it is appropriate we map these exceptions to `socket.error`.
|
||||||
|
"""
|
||||||
|
|
||||||
try:
|
try:
|
||||||
return super(PatroniSequentialThreadingHandler, self).select(*args, **kwargs)
|
return super(PatroniSequentialThreadingHandler, self).select(*args, **kwargs)
|
||||||
except ValueError as e:
|
except IOError as e:
|
||||||
raise select.error(9, str(e))
|
raise (select.error(e.errno, e.strerror) if six.PY2 else e)
|
||||||
|
except (TypeError, ValueError) as e:
|
||||||
|
raise (e if six.PY2 and isinstance(e, TypeError) else select.error(9, str(e)))
|
||||||
|
|
||||||
|
|
||||||
class PatroniKazooClient(KazooClient):
|
class PatroniKazooClient(KazooClient):
|
||||||
@@ -251,14 +263,6 @@ class ZooKeeper(AbstractDCS):
|
|||||||
|
|
||||||
# get leader
|
# get leader
|
||||||
leader = self.get_node(self.leader_path) if self._LEADER in nodes else None
|
leader = self.get_node(self.leader_path) if self._LEADER in nodes else None
|
||||||
if leader:
|
|
||||||
client_id = self._client.client_id
|
|
||||||
if not self._ctl and leader[0] == self._name and client_id is not None \
|
|
||||||
and client_id[0] != leader[1].ephemeralOwner:
|
|
||||||
logger.info('I am leader but not owner of the session. Removing leader node')
|
|
||||||
self._client.delete(self.leader_path)
|
|
||||||
leader = None
|
|
||||||
|
|
||||||
if leader:
|
if leader:
|
||||||
member = Member(-1, leader[0], None, {})
|
member = Member(-1, leader[0], None, {})
|
||||||
member = ([m for m in members if m.name == leader[0]] or [member])[0]
|
member = ([m for m in members if m.name == leader[0]] or [member])[0]
|
||||||
@@ -272,7 +276,14 @@ class ZooKeeper(AbstractDCS):
|
|||||||
failover = self.get_node(self.failover_path, watch=self.cluster_watcher) if self._FAILOVER in nodes else None
|
failover = self.get_node(self.failover_path, watch=self.cluster_watcher) if self._FAILOVER in nodes else None
|
||||||
failover = failover and Failover.from_node(failover[1].version, failover[0])
|
failover = failover and Failover.from_node(failover[1].version, failover[0])
|
||||||
|
|
||||||
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
# get failsafe topology
|
||||||
|
failsafe = self.get_node(self.failsafe_path, watch=self.cluster_watcher) if self._FAILSAFE in nodes else None
|
||||||
|
try:
|
||||||
|
failsafe = json.loads(failsafe[0]) if failsafe else None
|
||||||
|
except Exception:
|
||||||
|
failsafe = None
|
||||||
|
|
||||||
|
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots, failsafe)
|
||||||
|
|
||||||
def _load_cluster(self):
|
def _load_cluster(self):
|
||||||
cluster = self.cluster
|
cluster = self.cluster
|
||||||
@@ -293,8 +304,8 @@ class ZooKeeper(AbstractDCS):
|
|||||||
try:
|
try:
|
||||||
last_lsn, slots = self.get_status(cluster.leader)
|
last_lsn, slots = self.get_status(cluster.leader)
|
||||||
self.event.clear()
|
self.event.clear()
|
||||||
cluster = Cluster(cluster.initialize, cluster.config, cluster.leader, last_lsn,
|
cluster = Cluster(cluster.initialize, cluster.config, cluster.leader, last_lsn, cluster.members,
|
||||||
cluster.members, cluster.failover, cluster.sync, cluster.history, slots)
|
cluster.failover, cluster.sync, cluster.history, slots, cluster.failsafe)
|
||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
return cluster
|
return cluster
|
||||||
@@ -314,10 +325,17 @@ class ZooKeeper(AbstractDCS):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
def attempt_to_acquire_leader(self, permanent=False):
|
def attempt_to_acquire_leader(self, permanent=False):
|
||||||
ret = self._create(self.leader_path, self._name.encode('utf-8'), retry=True, ephemeral=not permanent)
|
try:
|
||||||
if not ret:
|
self._client.retry(self._client.create, self.leader_path, self._name.encode('utf-8'),
|
||||||
|
makepath=True, ephemeral=not permanent)
|
||||||
|
return True
|
||||||
|
except (ConnectionClosedError, RetryFailedError) as e:
|
||||||
|
raise ZooKeeperError(e)
|
||||||
|
except Exception as e:
|
||||||
|
if not isinstance(e, NodeExistsError):
|
||||||
|
logger.error('Failed to create %s: %r', self.leader_path, e)
|
||||||
logger.info('Could not take out TTL lock')
|
logger.info('Could not take out TTL lock')
|
||||||
return ret
|
return False
|
||||||
|
|
||||||
def _set_or_create(self, key, value, index=None, retry=False, do_not_create_empty=False):
|
def _set_or_create(self, key, value, index=None, retry=False, do_not_create_empty=False):
|
||||||
value = value.encode('utf-8')
|
value = value.encode('utf-8')
|
||||||
@@ -397,7 +415,32 @@ class ZooKeeper(AbstractDCS):
|
|||||||
def _write_status(self, value):
|
def _write_status(self, value):
|
||||||
return self._set_or_create(self.status_path, value)
|
return self._set_or_create(self.status_path, value)
|
||||||
|
|
||||||
|
def _write_failsafe(self, value):
|
||||||
|
return self._set_or_create(self.failsafe_path, value)
|
||||||
|
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
|
cluster = self.cluster
|
||||||
|
session = cluster and isinstance(cluster.leader, Leader) and cluster.leader.session
|
||||||
|
if self._client.client_id and self._client.client_id[0] != session:
|
||||||
|
logger.warning('Recreating the leader ZNode due to ownership mismatch')
|
||||||
|
try:
|
||||||
|
self._client.retry(self._client.delete, self.leader_path)
|
||||||
|
except NoNodeError:
|
||||||
|
pass
|
||||||
|
except (ConnectionClosedError, RetryFailedError) as e:
|
||||||
|
raise ZooKeeperError(e)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Failed to remove %s: %r', self.leader_path, e)
|
||||||
|
return False
|
||||||
|
|
||||||
|
try:
|
||||||
|
self._client.retry(self._client.create, self.leader_path,
|
||||||
|
self._name.encode('utf-8'), makepath=True, ephemeral=True)
|
||||||
|
except (ConnectionClosedError, RetryFailedError) as e:
|
||||||
|
raise ZooKeeperError(e)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Failed to create %s: %r', self.leader_path, e)
|
||||||
|
return False
|
||||||
return True
|
return True
|
||||||
|
|
||||||
def _delete_leader(self):
|
def _delete_leader(self):
|
||||||
|
|||||||
+53
-9
@@ -39,11 +39,20 @@ class _MemberStatus(namedtuple('_MemberStatus', ['member', 'reachable', 'in_reco
|
|||||||
"""
|
"""
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_api_response(cls, member, json):
|
def from_api_response(cls, member, json):
|
||||||
is_master = json['role'] == 'master'
|
"""
|
||||||
|
:param member: dcs.Member object
|
||||||
|
:param json: RestApiHandler.get_postgresql_status() result
|
||||||
|
:returns: _MemberStatus object
|
||||||
|
"""
|
||||||
|
# If one of those is not in a response we want to count the node as not healthy/reachable
|
||||||
|
assert 'wal' in json or 'xlog' in json
|
||||||
|
|
||||||
|
wal = json.get('wal', json.get('xlog'))
|
||||||
|
in_recovery = not bool(wal.get('location')) # abuse difference in primary/replica response format
|
||||||
timeline = json.get('timeline', 0)
|
timeline = json.get('timeline', 0)
|
||||||
dcs_last_seen = json.get('dcs_last_seen', 0)
|
dcs_last_seen = json.get('dcs_last_seen', 0)
|
||||||
wal = not is_master and max(json['xlog'].get('received_location', 0), json['xlog'].get('replayed_location', 0))
|
wal = in_recovery and max(wal.get('received_location', 0), wal.get('replayed_location', 0))
|
||||||
return cls(member, True, not is_master, dcs_last_seen, timeline, wal,
|
return cls(member, True, in_recovery, dcs_last_seen, timeline, wal,
|
||||||
json.get('tags', {}), json.get('watchdog_failed', False))
|
json.get('tags', {}), json.get('watchdog_failed', False))
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -146,7 +155,13 @@ class Ha(object):
|
|||||||
self._leader_timeline = None if cluster.is_unlocked() else cluster.leader.timeline
|
self._leader_timeline = None if cluster.is_unlocked() else cluster.leader.timeline
|
||||||
|
|
||||||
def acquire_lock(self):
|
def acquire_lock(self):
|
||||||
|
try:
|
||||||
ret = self.dcs.attempt_to_acquire_leader()
|
ret = self.dcs.attempt_to_acquire_leader()
|
||||||
|
except DCSError:
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
logger.exception('Unexpected exception raised from attempt_to_acquire_leader, please report it as a BUG')
|
||||||
|
ret = False
|
||||||
self.set_is_leader(ret)
|
self.set_is_leader(ret)
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
@@ -160,6 +175,8 @@ class Ha(object):
|
|||||||
logger.exception('Exception when called state_handler.last_operation()')
|
logger.exception('Exception when called state_handler.last_operation()')
|
||||||
try:
|
try:
|
||||||
ret = self.dcs.update_leader(last_lsn, slots)
|
ret = self.dcs.update_leader(last_lsn, slots)
|
||||||
|
except DCSError:
|
||||||
|
raise
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('Unexpected exception raised from update_leader, please report it as a BUG')
|
logger.exception('Unexpected exception raised from update_leader, please report it as a BUG')
|
||||||
ret = False
|
ret = False
|
||||||
@@ -192,6 +209,10 @@ class Ha(object):
|
|||||||
'version': self.patroni.version
|
'version': self.patroni.version
|
||||||
}
|
}
|
||||||
|
|
||||||
|
proxy_url = self.state_handler.proxy_url
|
||||||
|
if proxy_url:
|
||||||
|
data['proxy_url'] = proxy_url
|
||||||
|
|
||||||
if self.is_leader() and not self._rewind.checkpoint_after_promote():
|
if self.is_leader() and not self._rewind.checkpoint_after_promote():
|
||||||
data['checkpoint_after_promote'] = False
|
data['checkpoint_after_promote'] = False
|
||||||
tags = self.get_effective_tags()
|
tags = self.get_effective_tags()
|
||||||
@@ -273,7 +294,9 @@ class Ha(object):
|
|||||||
else:
|
else:
|
||||||
create_replica_methods = self.get_standby_cluster_config().get('create_replica_methods', []) \
|
create_replica_methods = self.get_standby_cluster_config().get('create_replica_methods', []) \
|
||||||
if self.is_standby_cluster() else None
|
if self.is_standby_cluster() else None
|
||||||
if self.state_handler.can_create_replica_without_replication_connection(create_replica_methods):
|
can_bootstrap = self.state_handler.can_create_replica_without_replication_connection(create_replica_methods)
|
||||||
|
concurrent_bootstrap = self.cluster.initialize == ""
|
||||||
|
if can_bootstrap and not concurrent_bootstrap:
|
||||||
msg = 'bootstrap (without leader)'
|
msg = 'bootstrap (without leader)'
|
||||||
return self._async_executor.try_run_async(msg, self.clone) or 'trying to ' + msg
|
return self._async_executor.try_run_async(msg, self.clone) or 'trying to ' + msg
|
||||||
return 'waiting for {0}leader to bootstrap'.format('standby_' if self.is_standby_cluster() else '')
|
return 'waiting for {0}leader to bootstrap'.format('standby_' if self.is_standby_cluster() else '')
|
||||||
@@ -737,6 +760,11 @@ class Ha(object):
|
|||||||
return None
|
return None
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
# in synchronous mode when our name is not in the /sync key
|
||||||
|
# we shouldn't take any action even if the candidate is unhealthy
|
||||||
|
if self.is_synchronous_mode() and not self.cluster.sync.matches(self.state_handler.name):
|
||||||
|
return False
|
||||||
|
|
||||||
# find specific node and check that it is healthy
|
# find specific node and check that it is healthy
|
||||||
member = self.cluster.get_member(failover.candidate, fallback_to_leader=False)
|
member = self.cluster.get_member(failover.candidate, fallback_to_leader=False)
|
||||||
if member:
|
if member:
|
||||||
@@ -797,7 +825,7 @@ class Ha(object):
|
|||||||
if self.cluster.failover:
|
if self.cluster.failover:
|
||||||
# When doing a switchover in synchronous mode only synchronous nodes and former leader are allowed to race
|
# When doing a switchover in synchronous mode only synchronous nodes and former leader are allowed to race
|
||||||
if self.is_synchronous_mode() and self.cluster.failover.leader and \
|
if self.is_synchronous_mode() and self.cluster.failover.leader and \
|
||||||
self.cluster.failover.candidate and not self.cluster.sync.matches(self.state_handler.name):
|
not self.cluster.sync.matches(self.state_handler.name):
|
||||||
return False
|
return False
|
||||||
return self.manual_failover_process_no_leader()
|
return self.manual_failover_process_no_leader()
|
||||||
|
|
||||||
@@ -1407,7 +1435,14 @@ class Ha(object):
|
|||||||
return 'started as a secondary'
|
return 'started as a secondary'
|
||||||
|
|
||||||
# is data directory empty?
|
# is data directory empty?
|
||||||
if self.state_handler.data_directory_empty():
|
try:
|
||||||
|
data_directory_is_empty = self.state_handler.data_directory_empty()
|
||||||
|
data_directory_is_accessible = True
|
||||||
|
except OSError as e:
|
||||||
|
data_directory_is_accessible = False
|
||||||
|
data_directory_error = e
|
||||||
|
|
||||||
|
if not data_directory_is_accessible or data_directory_is_empty:
|
||||||
self.state_handler.set_role('uninitialized')
|
self.state_handler.set_role('uninitialized')
|
||||||
self.state_handler.stop('immediate', stop_timeout=self.patroni.config['retry_timeout'])
|
self.state_handler.stop('immediate', stop_timeout=self.patroni.config['retry_timeout'])
|
||||||
# In case datadir went away while we were master.
|
# In case datadir went away while we were master.
|
||||||
@@ -1416,8 +1451,11 @@ class Ha(object):
|
|||||||
# is this instance the leader?
|
# is this instance the leader?
|
||||||
if self.has_lock():
|
if self.has_lock():
|
||||||
self.release_leader_key_voluntarily()
|
self.release_leader_key_voluntarily()
|
||||||
return 'released leader key voluntarily as data dir empty and currently leader'
|
return 'released leader key voluntarily as data dir {0} and currently leader'.format(
|
||||||
|
'empty' if data_directory_is_accessible else 'not accessible')
|
||||||
|
|
||||||
|
if not data_directory_is_accessible:
|
||||||
|
return 'data directory is not accessible: {0}'.format(data_directory_error)
|
||||||
if self.is_paused():
|
if self.is_paused():
|
||||||
return 'running with empty data directory'
|
return 'running with empty data directory'
|
||||||
return self.bootstrap() # new node
|
return self.bootstrap() # new node
|
||||||
@@ -1483,7 +1521,9 @@ class Ha(object):
|
|||||||
# asynchronous processes are running (should be always the case for the master)
|
# asynchronous processes are running (should be always the case for the master)
|
||||||
if not self._async_executor.busy and not self.state_handler.is_starting():
|
if not self._async_executor.busy and not self.state_handler.is_starting():
|
||||||
create_slots = self.state_handler.slots_handler.sync_replication_slots(self.cluster,
|
create_slots = self.state_handler.slots_handler.sync_replication_slots(self.cluster,
|
||||||
self.patroni.nofailover)
|
self.patroni.nofailover,
|
||||||
|
self.patroni.replicatefrom,
|
||||||
|
self.is_paused())
|
||||||
if not self.state_handler.cb_called:
|
if not self.state_handler.cb_called:
|
||||||
if not self.state_handler.is_leader():
|
if not self.state_handler.is_leader():
|
||||||
self._rewind.trigger_check_diverged_lsn()
|
self._rewind.trigger_check_diverged_lsn()
|
||||||
@@ -1499,8 +1539,12 @@ class Ha(object):
|
|||||||
dcs_failed = True
|
dcs_failed = True
|
||||||
logger.error('Error communicating with DCS')
|
logger.error('Error communicating with DCS')
|
||||||
if not self.is_paused() and self.state_handler.is_running() and self.state_handler.is_leader():
|
if not self.is_paused() and self.state_handler.is_running() and self.state_handler.is_leader():
|
||||||
|
msg = 'demoting self because DCS is not accessible and I was a leader'
|
||||||
|
if not self._async_executor.try_run_async(msg, self.demote, ('offline',)):
|
||||||
|
return msg
|
||||||
|
logger.warning('AsyncExecutor is busy, demoting from the main thread')
|
||||||
self.demote('offline')
|
self.demote('offline')
|
||||||
return 'demoted self because DCS is not accessible and i was a leader'
|
return 'demoted self because DCS is not accessible and I was a leader'
|
||||||
return 'DCS is not accessible'
|
return 'DCS is not accessible'
|
||||||
except (psycopg.Error, PostgresConnectionException):
|
except (psycopg.Error, PostgresConnectionException):
|
||||||
return 'Error communicating with PostgreSQL. Will try again later'
|
return 'Error communicating with PostgreSQL. Will try again later'
|
||||||
|
|||||||
@@ -428,11 +428,11 @@ class Postgresql(object):
|
|||||||
# If the cluster is shutdown with archive_mode=on, WAL is switched before writing the checkpoint.
|
# If the cluster is shutdown with archive_mode=on, WAL is switched before writing the checkpoint.
|
||||||
# In this case we want to take the LSN of previous record (switch) as the last known WAL location.
|
# In this case we want to take the LSN of previous record (switch) as the last known WAL location.
|
||||||
if parse_lsn(lsn) == prev and desc.strip() in ('xlog switch', 'SWITCH'):
|
if parse_lsn(lsn) == prev and desc.strip() in ('xlog switch', 'SWITCH'):
|
||||||
return str(prev)
|
return prev
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error('Exception when parsing WAL pg_%sdump output: %r', self.wal_name, e)
|
logger.error('Exception when parsing WAL pg_%sdump output: %r', self.wal_name, e)
|
||||||
if isinstance(checkpoint_lsn, six.integer_types):
|
if isinstance(checkpoint_lsn, six.integer_types):
|
||||||
return str(checkpoint_lsn)
|
return checkpoint_lsn
|
||||||
|
|
||||||
def is_running(self):
|
def is_running(self):
|
||||||
"""Returns PostmasterProcess if one is running on the data directory or None. If most recently seen process
|
"""Returns PostmasterProcess if one is running on the data directory or None. If most recently seen process
|
||||||
@@ -659,7 +659,7 @@ class Postgresql(object):
|
|||||||
if on_safepoint:
|
if on_safepoint:
|
||||||
# Wait for our connection to terminate so we can be sure that no new connections are being initiated
|
# Wait for our connection to terminate so we can be sure that no new connections are being initiated
|
||||||
self._wait_for_connection_close(postmaster)
|
self._wait_for_connection_close(postmaster)
|
||||||
postmaster.wait_for_user_backends_to_close()
|
postmaster.wait_for_user_backends_to_close(stop_timeout)
|
||||||
on_safepoint()
|
on_safepoint()
|
||||||
|
|
||||||
if on_shutdown and mode in ('fast', 'smart'):
|
if on_shutdown and mode in ('fast', 'smart'):
|
||||||
@@ -668,7 +668,7 @@ class Postgresql(object):
|
|||||||
while postmaster.is_running():
|
while postmaster.is_running():
|
||||||
data = self.controldata()
|
data = self.controldata()
|
||||||
if data.get('Database cluster state', '') == 'shut down':
|
if data.get('Database cluster state', '') == 'shut down':
|
||||||
on_shutdown(int(self.latest_checkpoint_location()))
|
on_shutdown(self.latest_checkpoint_location())
|
||||||
break
|
break
|
||||||
elif data.get('Database cluster state', '').startswith('shut down'): # shut down in recovery
|
elif data.get('Database cluster state', '').startswith('shut down'): # shut down in recovery
|
||||||
break
|
break
|
||||||
|
|||||||
@@ -315,12 +315,14 @@ END;$$""".format(quote_literal(name), quote_ident(name, self._postgresql.connect
|
|||||||
self._postgresql.query('SET log_statement TO none')
|
self._postgresql.query('SET log_statement TO none')
|
||||||
self._postgresql.query('SET log_min_duration_statement TO -1')
|
self._postgresql.query('SET log_min_duration_statement TO -1')
|
||||||
self._postgresql.query("SET log_min_error_statement TO 'log'")
|
self._postgresql.query("SET log_min_error_statement TO 'log'")
|
||||||
|
self._postgresql.query("SET pg_stat_statements.track_utility to 'off'")
|
||||||
try:
|
try:
|
||||||
self._postgresql.query(sql)
|
self._postgresql.query(sql)
|
||||||
finally:
|
finally:
|
||||||
self._postgresql.query('RESET log_min_error_statement')
|
self._postgresql.query('RESET log_min_error_statement')
|
||||||
self._postgresql.query('RESET log_min_duration_statement')
|
self._postgresql.query('RESET log_min_duration_statement')
|
||||||
self._postgresql.query('RESET log_statement')
|
self._postgresql.query('RESET log_statement')
|
||||||
|
self._postgresql.query('RESET pg_stat_statements.track_utility')
|
||||||
|
|
||||||
def post_bootstrap(self, config, task):
|
def post_bootstrap(self, config, task):
|
||||||
try:
|
try:
|
||||||
|
|||||||
@@ -1017,6 +1017,9 @@ class ConfigHandler(object):
|
|||||||
if not local_connection_address_changed:
|
if not local_connection_address_changed:
|
||||||
self.resolve_connection_addresses()
|
self.resolve_connection_addresses()
|
||||||
|
|
||||||
|
proxy_addr = config.get('proxy_address')
|
||||||
|
self._postgresql.proxy_url = uri('postgres', proxy_addr, self._postgresql.database) if proxy_addr else None
|
||||||
|
|
||||||
if conf_changed:
|
if conf_changed:
|
||||||
self.write_postgresql_conf()
|
self.write_postgresql_conf()
|
||||||
|
|
||||||
|
|||||||
@@ -22,7 +22,6 @@ class Connection(object):
|
|||||||
with self._lock:
|
with self._lock:
|
||||||
if not self._connection or self._connection.closed != 0:
|
if not self._connection or self._connection.closed != 0:
|
||||||
self._connection = psycopg.connect(**self._conn_kwargs)
|
self._connection = psycopg.connect(**self._conn_kwargs)
|
||||||
self._connection.autocommit = True
|
|
||||||
self.server_version = self._connection.server_version
|
self.server_version = self._connection.server_version
|
||||||
return self._connection
|
return self._connection
|
||||||
|
|
||||||
@@ -42,7 +41,6 @@ class Connection(object):
|
|||||||
@contextmanager
|
@contextmanager
|
||||||
def get_connection_cursor(**kwargs):
|
def get_connection_cursor(**kwargs):
|
||||||
conn = psycopg.connect(**kwargs)
|
conn = psycopg.connect(**kwargs)
|
||||||
conn.autocommit = True
|
|
||||||
with conn.cursor() as cur:
|
with conn.cursor() as cur:
|
||||||
yield cur
|
yield cur
|
||||||
conn.close()
|
conn.close()
|
||||||
|
|||||||
@@ -171,8 +171,8 @@ class PostmasterProcess(psutil.Process):
|
|||||||
else:
|
else:
|
||||||
return not self.is_running()
|
return not self.is_running()
|
||||||
|
|
||||||
def wait_for_user_backends_to_close(self):
|
def wait_for_user_backends_to_close(self, stop_timeout):
|
||||||
# These regexps are cross checked against versions PostgreSQL 9.1 .. 11
|
# These regexps are cross checked against versions PostgreSQL 9.1 .. 15
|
||||||
aux_proc_re = re.compile("(?:postgres:)( .*:)? (?:(?:archiver|startup|autovacuum launcher|autovacuum worker|"
|
aux_proc_re = re.compile("(?:postgres:)( .*:)? (?:(?:archiver|startup|autovacuum launcher|autovacuum worker|"
|
||||||
"checkpointer|logger|stats collector|wal receiver|wal writer|writer)(?: process )?|"
|
"checkpointer|logger|stats collector|wal receiver|wal writer|writer)(?: process )?|"
|
||||||
"walreceiver|wal sender process|walsender|walwriter|background writer|"
|
"walreceiver|wal sender process|walsender|walwriter|background writer|"
|
||||||
@@ -184,18 +184,22 @@ class PostmasterProcess(psutil.Process):
|
|||||||
return logger.debug('Failed to get list of postmaster children')
|
return logger.debug('Failed to get list of postmaster children')
|
||||||
|
|
||||||
user_backends = []
|
user_backends = []
|
||||||
user_backends_cmdlines = []
|
user_backends_cmdlines = {}
|
||||||
for child in children:
|
for child in children:
|
||||||
try:
|
try:
|
||||||
cmdline = child.cmdline()
|
cmdline = child.cmdline()
|
||||||
if cmdline and not aux_proc_re.match(cmdline[0]):
|
if cmdline and not aux_proc_re.match(cmdline[0]):
|
||||||
user_backends.append(child)
|
user_backends.append(child)
|
||||||
user_backends_cmdlines.append(cmdline[0])
|
user_backends_cmdlines[child.pid] = cmdline[0]
|
||||||
except psutil.NoSuchProcess:
|
except psutil.NoSuchProcess:
|
||||||
pass
|
pass
|
||||||
if user_backends:
|
if user_backends:
|
||||||
logger.debug('Waiting for user backends %s to close', ', '.join(user_backends_cmdlines))
|
logger.debug('Waiting for user backends %s to close', ', '.join(user_backends_cmdlines.values()))
|
||||||
psutil.wait_procs(user_backends)
|
gone, live = psutil.wait_procs(user_backends, stop_timeout)
|
||||||
|
if stop_timeout and live:
|
||||||
|
live = [user_backends_cmdlines[b.pid] for b in live]
|
||||||
|
logger.warning('Backends still alive after %s: %s', stop_timeout, ', '.join(live))
|
||||||
|
else:
|
||||||
logger.debug("Backends closed")
|
logger.debug("Backends closed")
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
|
|||||||
@@ -1,6 +1,8 @@
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
import shlex
|
import shlex
|
||||||
|
import shutil
|
||||||
import six
|
import six
|
||||||
import subprocess
|
import subprocess
|
||||||
|
|
||||||
@@ -277,28 +279,37 @@ class Rewind(object):
|
|||||||
def checkpoint_after_promote(self):
|
def checkpoint_after_promote(self):
|
||||||
return self._state == REWIND_STATUS.CHECKPOINT
|
return self._state == REWIND_STATUS.CHECKPOINT
|
||||||
|
|
||||||
def _fetch_missing_wal(self, restore_command, wal_filename):
|
def _buid_archiver_command(self, command, wal_filename):
|
||||||
|
"""Replace placeholders in the given archiver command's template.
|
||||||
|
Applicable for archive_command and restore_command.
|
||||||
|
Can also be used for archive_cleanup_command and recovery_end_command,
|
||||||
|
however %r value is always set to 000000010000000000000001."""
|
||||||
cmd = ''
|
cmd = ''
|
||||||
length = len(restore_command)
|
length = len(command)
|
||||||
i = 0
|
i = 0
|
||||||
while i < length:
|
while i < length:
|
||||||
if restore_command[i] == '%' and i + 1 < length:
|
if command[i] == '%' and i + 1 < length:
|
||||||
i += 1
|
i += 1
|
||||||
if restore_command[i] == 'p':
|
if command[i] == 'p':
|
||||||
cmd += os.path.join(self._postgresql.wal_dir, wal_filename)
|
cmd += os.path.join(self._postgresql.wal_dir, wal_filename)
|
||||||
elif restore_command[i] == 'f':
|
elif command[i] == 'f':
|
||||||
cmd += wal_filename
|
cmd += wal_filename
|
||||||
elif restore_command[i] == 'r':
|
elif command[i] == 'r':
|
||||||
cmd += '000000010000000000000001'
|
cmd += '000000010000000000000001'
|
||||||
elif restore_command[i] == '%':
|
elif command[i] == '%':
|
||||||
cmd += '%'
|
cmd += '%'
|
||||||
else:
|
else:
|
||||||
cmd += '%'
|
cmd += '%'
|
||||||
i -= 1
|
i -= 1
|
||||||
else:
|
else:
|
||||||
cmd += restore_command[i]
|
cmd += command[i]
|
||||||
i += 1
|
i += 1
|
||||||
|
|
||||||
|
return cmd
|
||||||
|
|
||||||
|
def _fetch_missing_wal(self, restore_command, wal_filename):
|
||||||
|
cmd = self._buid_archiver_command(restore_command, wal_filename)
|
||||||
|
|
||||||
logger.info('Trying to fetch the missing wal: %s', cmd)
|
logger.info('Trying to fetch the missing wal: %s', cmd)
|
||||||
return self._postgresql.cancellable.call(shlex.split(cmd)) == 0
|
return self._postgresql.cancellable.call(shlex.split(cmd)) == 0
|
||||||
|
|
||||||
@@ -315,6 +326,42 @@ class Rewind(object):
|
|||||||
if waldir.endswith('/pg_' + self._postgresql.wal_name) and len(wal_filename) == 24:
|
if waldir.endswith('/pg_' + self._postgresql.wal_name) and len(wal_filename) == 24:
|
||||||
return wal_filename
|
return wal_filename
|
||||||
|
|
||||||
|
def _archive_ready_wals(self):
|
||||||
|
"""Try to archive WALs that have .ready files just in case
|
||||||
|
archive_mode was not set to 'always' before promote, while
|
||||||
|
after it the WALs were recycled on the promoted replica.
|
||||||
|
With this we prevent the entire loss of such WALs and the
|
||||||
|
consequent old leader's start failure."""
|
||||||
|
archive_mode = self._postgresql.get_guc_value('archive_mode')
|
||||||
|
archive_cmd = self._postgresql.get_guc_value('archive_command')
|
||||||
|
if archive_mode not in ('on', 'always') or not archive_cmd:
|
||||||
|
return
|
||||||
|
|
||||||
|
walseg_regex = re.compile(r'^[0-9A-F]{24}(\.partial){0,1}\.ready$')
|
||||||
|
status_dir = os.path.join(self._postgresql.wal_dir, 'archive_status')
|
||||||
|
try:
|
||||||
|
wals_to_archive = [f[:-6] for f in os.listdir(status_dir) if walseg_regex.match(f)]
|
||||||
|
except OSError as e:
|
||||||
|
return logger.error('Unable to list %s: %r', status_dir, e)
|
||||||
|
|
||||||
|
# skip fsync, as postgres --single or pg_rewind will anyway run it
|
||||||
|
for wal in sorted(wals_to_archive):
|
||||||
|
old_name = os.path.join(status_dir, wal + '.ready')
|
||||||
|
# wal file might have alredy been archived
|
||||||
|
if os.path.isfile(old_name) and os.path.isfile(os.path.join(self._postgresql.wal_dir, wal)):
|
||||||
|
cmd = self._buid_archiver_command(archive_cmd, wal)
|
||||||
|
# it is the author of archive_command, who is responsible
|
||||||
|
# for not overriding the WALs already present in archive
|
||||||
|
logger.info('Trying to archive %s: %s', wal, cmd)
|
||||||
|
if self._postgresql.cancellable.call(shlex.split(cmd)) == 0:
|
||||||
|
new_name = os.path.join(status_dir, wal + '.done')
|
||||||
|
try:
|
||||||
|
shutil.move(old_name, new_name)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Unable to rename %s to %s: %r', old_name, new_name, e)
|
||||||
|
else:
|
||||||
|
logger.info('Failed to archive WAL segment %s', wal)
|
||||||
|
|
||||||
def pg_rewind(self, r):
|
def pg_rewind(self, r):
|
||||||
# prepare pg_rewind connection
|
# prepare pg_rewind connection
|
||||||
env = self._postgresql.config.write_pgpass(r)
|
env = self._postgresql.config.write_pgpass(r)
|
||||||
@@ -367,6 +414,8 @@ class Rewind(object):
|
|||||||
if self._postgresql.is_running() and not self._postgresql.stop(checkpoint=False):
|
if self._postgresql.is_running() and not self._postgresql.stop(checkpoint=False):
|
||||||
return logger.warning('Can not run pg_rewind because postgres is still running')
|
return logger.warning('Can not run pg_rewind because postgres is still running')
|
||||||
|
|
||||||
|
self._archive_ready_wals()
|
||||||
|
|
||||||
# prepare pg_rewind connection
|
# prepare pg_rewind connection
|
||||||
r = self._conn_kwargs(leader, self._postgresql.config.rewind_credentials)
|
r = self._conn_kwargs(leader, self._postgresql.config.rewind_credentials)
|
||||||
|
|
||||||
@@ -465,6 +514,7 @@ class Rewind(object):
|
|||||||
logger.exception('Unable to list %s', status_dir)
|
logger.exception('Unable to list %s', status_dir)
|
||||||
|
|
||||||
def ensure_clean_shutdown(self):
|
def ensure_clean_shutdown(self):
|
||||||
|
self._archive_ready_wals()
|
||||||
self.cleanup_archive_status()
|
self.cleanup_archive_status()
|
||||||
|
|
||||||
# Start in a single user mode and stop to produce a clean shutdown
|
# Start in a single user mode and stop to produce a clean shutdown
|
||||||
|
|||||||
+106
-21
@@ -5,6 +5,7 @@ import shutil
|
|||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
|
from threading import Condition, Thread
|
||||||
|
|
||||||
from .connection import get_connection_cursor
|
from .connection import get_connection_cursor
|
||||||
from .misc import format_lsn
|
from .misc import format_lsn
|
||||||
@@ -31,10 +32,94 @@ def fsync_dir(path):
|
|||||||
os.close(fd)
|
os.close(fd)
|
||||||
|
|
||||||
|
|
||||||
|
class SlotsAdvanceThread(Thread):
|
||||||
|
|
||||||
|
def __init__(self, slots_handler):
|
||||||
|
super(SlotsAdvanceThread, self).__init__()
|
||||||
|
self.daemon = True
|
||||||
|
self._slots_handler = slots_handler
|
||||||
|
|
||||||
|
# _copy_slots and _failed are used to asynchronously give some feedback to the main thread
|
||||||
|
self._copy_slots = []
|
||||||
|
self._failed = False
|
||||||
|
|
||||||
|
self._scheduled = defaultdict(dict) # {'dbname1': {'slot1': 100, 'slot2': 100}, 'dbname2': {'slot3': 100}}
|
||||||
|
self._condition = Condition() # protect self._scheduled from concurrent access and to wakeup the run() method
|
||||||
|
|
||||||
|
self.start()
|
||||||
|
|
||||||
|
def sync_slot(self, cur, database, slot, lsn):
|
||||||
|
failed = copy = False
|
||||||
|
try:
|
||||||
|
cur.execute("SELECT pg_catalog.pg_replication_slot_advance(%s, %s)", (slot, format_lsn(lsn)))
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("Failed to advance logical replication slot '%s': %r", slot, e)
|
||||||
|
failed = True
|
||||||
|
copy = isinstance(e, OperationalError) and e.diag.sqlstate == '58P01' # WAL file is gone
|
||||||
|
with self._condition:
|
||||||
|
if self._scheduled and failed:
|
||||||
|
if copy and slot not in self._copy_slots:
|
||||||
|
self._copy_slots.append(slot)
|
||||||
|
self._failed = True
|
||||||
|
|
||||||
|
new_lsn = self._scheduled.get(database, {}).get(slot, 0)
|
||||||
|
# remove slot from the self._scheduled structure only if it wasn't changed
|
||||||
|
if new_lsn == lsn and database in self._scheduled:
|
||||||
|
self._scheduled[database].pop(slot)
|
||||||
|
if not self._scheduled[database]:
|
||||||
|
self._scheduled.pop(database)
|
||||||
|
|
||||||
|
def sync_slots_in_database(self, database, slots):
|
||||||
|
with self._slots_handler.get_local_connection_cursor(dbname=database, options='-c statement_timeout=0') as cur:
|
||||||
|
for slot in slots:
|
||||||
|
with self._condition:
|
||||||
|
lsn = self._scheduled.get(database, {}).get(slot, 0)
|
||||||
|
if lsn:
|
||||||
|
self.sync_slot(cur, database, slot, lsn)
|
||||||
|
|
||||||
|
def sync_slots(self):
|
||||||
|
with self._condition:
|
||||||
|
databases = list(self._scheduled.keys())
|
||||||
|
for database in databases:
|
||||||
|
with self._condition:
|
||||||
|
slots = list(self._scheduled.get(database, {}).keys())
|
||||||
|
if slots:
|
||||||
|
try:
|
||||||
|
self.sync_slots_in_database(database, slots)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Failed to advance replication slots in database %s: %r', database, e)
|
||||||
|
|
||||||
|
def run(self):
|
||||||
|
while True:
|
||||||
|
with self._condition:
|
||||||
|
if not self._scheduled:
|
||||||
|
self._condition.wait()
|
||||||
|
|
||||||
|
self.sync_slots()
|
||||||
|
|
||||||
|
def schedule(self, advance_slots):
|
||||||
|
with self._condition:
|
||||||
|
for database, values in advance_slots.items():
|
||||||
|
self._scheduled[database].update(values)
|
||||||
|
ret = (self._failed, self._copy_slots)
|
||||||
|
self._copy_slots = []
|
||||||
|
self._failed = False
|
||||||
|
self._condition.notify()
|
||||||
|
|
||||||
|
return ret
|
||||||
|
|
||||||
|
def on_promote(self):
|
||||||
|
with self._condition:
|
||||||
|
self._scheduled.clear()
|
||||||
|
self._failed = False
|
||||||
|
self._copy_slots = []
|
||||||
|
|
||||||
|
|
||||||
class SlotsHandler(object):
|
class SlotsHandler(object):
|
||||||
|
|
||||||
def __init__(self, postgresql):
|
def __init__(self, postgresql):
|
||||||
self._postgresql = postgresql
|
self._postgresql = postgresql
|
||||||
|
self._advance = None
|
||||||
self._replication_slots = {} # already existing replication slots
|
self._replication_slots = {} # already existing replication slots
|
||||||
self._unready_logical_slots = {}
|
self._unready_logical_slots = {}
|
||||||
self.schedule()
|
self.schedule()
|
||||||
@@ -112,10 +197,10 @@ class SlotsHandler(object):
|
|||||||
# In normal situation rowcount should be 1, otherwise either slot doesn't exists or it is still active
|
# In normal situation rowcount should be 1, otherwise either slot doesn't exists or it is still active
|
||||||
return cursor.rowcount == 1
|
return cursor.rowcount == 1
|
||||||
|
|
||||||
def _drop_incorrect_slots(self, cluster, slots):
|
def _drop_incorrect_slots(self, cluster, slots, paused):
|
||||||
# drop old replication slots which are not presented in desired slots
|
# drop old replication slots which are not presented in desired slots
|
||||||
for name in set(self._replication_slots) - set(slots):
|
for name in set(self._replication_slots) - set(slots):
|
||||||
if not self.ignore_replication_slot(cluster, name) and not self.drop_replication_slot(name):
|
if not paused and not self.ignore_replication_slot(cluster, name) and not self.drop_replication_slot(name):
|
||||||
logger.error("Failed to drop replication slot '%s'", name)
|
logger.error("Failed to drop replication slot '%s'", name)
|
||||||
self._schedule_load_slots = True
|
self._schedule_load_slots = True
|
||||||
|
|
||||||
@@ -143,7 +228,7 @@ class SlotsHandler(object):
|
|||||||
self._schedule_load_slots = True
|
self._schedule_load_slots = True
|
||||||
|
|
||||||
@contextmanager
|
@contextmanager
|
||||||
def _get_local_connection_cursor(self, **kwargs):
|
def get_local_connection_cursor(self, **kwargs):
|
||||||
conn_kwargs = self._postgresql.config.local_connect_kwargs
|
conn_kwargs = self._postgresql.config.local_connect_kwargs
|
||||||
conn_kwargs.update(kwargs)
|
conn_kwargs.update(kwargs)
|
||||||
with get_connection_cursor(**conn_kwargs) as cur:
|
with get_connection_cursor(**conn_kwargs) as cur:
|
||||||
@@ -162,7 +247,7 @@ class SlotsHandler(object):
|
|||||||
|
|
||||||
# Create new logical slots
|
# Create new logical slots
|
||||||
for database, values in logical_slots.items():
|
for database, values in logical_slots.items():
|
||||||
with self._get_local_connection_cursor(dbname=database) as cur:
|
with self.get_local_connection_cursor(dbname=database) as cur:
|
||||||
for name, value in values.items():
|
for name, value in values.items():
|
||||||
try:
|
try:
|
||||||
cur.execute("SELECT pg_catalog.pg_create_logical_replication_slot(%s, %s)" +
|
cur.execute("SELECT pg_catalog.pg_create_logical_replication_slot(%s, %s)" +
|
||||||
@@ -175,6 +260,11 @@ class SlotsHandler(object):
|
|||||||
slots.pop(name)
|
slots.pop(name)
|
||||||
self._schedule_load_slots = True
|
self._schedule_load_slots = True
|
||||||
|
|
||||||
|
def schedule_advance_slots(self, slots):
|
||||||
|
if not self._advance:
|
||||||
|
self._advance = SlotsAdvanceThread(self)
|
||||||
|
return self._advance.schedule(slots)
|
||||||
|
|
||||||
def _ensure_logical_slots_replica(self, cluster, slots):
|
def _ensure_logical_slots_replica(self, cluster, slots):
|
||||||
advance_slots = defaultdict(dict) # Group logical slots to be advanced by database name
|
advance_slots = defaultdict(dict) # Group logical slots to be advanced by database name
|
||||||
create_slots = [] # And collect logical slots to be created on the replica
|
create_slots = [] # And collect logical slots to be created on the replica
|
||||||
@@ -186,27 +276,18 @@ class SlotsHandler(object):
|
|||||||
if name in cluster.slots:
|
if name in cluster.slots:
|
||||||
try: # Skip slots that doesn't need to be advanced
|
try: # Skip slots that doesn't need to be advanced
|
||||||
if value['confirmed_flush_lsn'] < int(cluster.slots[name]):
|
if value['confirmed_flush_lsn'] < int(cluster.slots[name]):
|
||||||
advance_slots[value['database']][name] = value
|
advance_slots[value['database']][name] = int(cluster.slots[name])
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error('Failed to parse "%s": %r', cluster.slots[name], e)
|
logger.error('Failed to parse "%s": %r', cluster.slots[name], e)
|
||||||
elif name in cluster.slots: # We want to copy only slots with feedback in a DCS
|
elif name in cluster.slots: # We want to copy only slots with feedback in a DCS
|
||||||
create_slots.append(name)
|
create_slots.append(name)
|
||||||
|
|
||||||
# Advance logical slots
|
error, copy_slots = self.schedule_advance_slots(advance_slots)
|
||||||
for database, values in advance_slots.items():
|
if error:
|
||||||
with self._get_local_connection_cursor(dbname=database, options='-c statement_timeout=0') as cur:
|
|
||||||
for name, value in values.items():
|
|
||||||
try:
|
|
||||||
cur.execute("SELECT pg_catalog.pg_replication_slot_advance(%s, %s)",
|
|
||||||
(name, format_lsn(int(cluster.slots[name]))))
|
|
||||||
except Exception as e:
|
|
||||||
logger.error("Failed to advance logical replication slot '%s': %r", name, e)
|
|
||||||
if isinstance(e, OperationalError) and e.diag.sqlstate == '58P01': # WAL file is gone
|
|
||||||
create_slots.append(name)
|
|
||||||
self._schedule_load_slots = True
|
self._schedule_load_slots = True
|
||||||
return create_slots
|
return create_slots + copy_slots
|
||||||
|
|
||||||
def sync_replication_slots(self, cluster, nofailover, replicatefrom=None):
|
def sync_replication_slots(self, cluster, nofailover, replicatefrom=None, paused=False):
|
||||||
ret = None
|
ret = None
|
||||||
if self._postgresql.major_version >= 90400 and cluster.config:
|
if self._postgresql.major_version >= 90400 and cluster.config:
|
||||||
try:
|
try:
|
||||||
@@ -215,7 +296,7 @@ class SlotsHandler(object):
|
|||||||
slots = cluster.get_replication_slots(self._postgresql.name, self._postgresql.role,
|
slots = cluster.get_replication_slots(self._postgresql.name, self._postgresql.role,
|
||||||
nofailover, self._postgresql.major_version, True)
|
nofailover, self._postgresql.major_version, True)
|
||||||
|
|
||||||
self._drop_incorrect_slots(cluster, slots)
|
self._drop_incorrect_slots(cluster, slots, paused)
|
||||||
|
|
||||||
self._ensure_physical_slots(slots)
|
self._ensure_physical_slots(slots)
|
||||||
|
|
||||||
@@ -264,10 +345,11 @@ class SlotsHandler(object):
|
|||||||
try:
|
try:
|
||||||
cur = self._query("SELECT pg_catalog.current_setting('hot_standby_feedback')::boolean")
|
cur = self._query("SELECT pg_catalog.current_setting('hot_standby_feedback')::boolean")
|
||||||
if not cur.fetchone()[0]:
|
if not cur.fetchone()[0]:
|
||||||
return logger.error('Logical slot failover requires "hot_standby_feedback".'
|
logger.error('Logical slot failover requires "hot_standby_feedback".'
|
||||||
' Please check postgresql.auto.conf')
|
' Please check postgresql.auto.conf')
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
return logger.error('Failed to check the hot_standby_feedback setting: %r', e)
|
logger.error('Failed to check the hot_standby_feedback setting: %r', e)
|
||||||
|
return # since `catalog_xmin` isn't valid further checks don't make any sense
|
||||||
|
|
||||||
for name in list(self._unready_logical_slots):
|
for name in list(self._unready_logical_slots):
|
||||||
value = self._replication_slots.get(name)
|
value = self._replication_slots.get(name)
|
||||||
@@ -331,6 +413,9 @@ class SlotsHandler(object):
|
|||||||
self._schedule_load_slots = self._force_readiness_check = value
|
self._schedule_load_slots = self._force_readiness_check = value
|
||||||
|
|
||||||
def on_promote(self):
|
def on_promote(self):
|
||||||
|
if self._advance:
|
||||||
|
self._advance.on_promote()
|
||||||
|
|
||||||
if self._unready_logical_slots:
|
if self._unready_logical_slots:
|
||||||
logger.warning('Logical replication slots that might be unsafe to use after promote: %s',
|
logger.warning('Logical replication slots that might be unsafe to use after promote: %s',
|
||||||
set(self._unready_logical_slots))
|
set(self._unready_logical_slots))
|
||||||
|
|||||||
@@ -200,7 +200,6 @@ parameters = CaseInsensitiveDict({
|
|||||||
'enable_async_append': Bool(140000, None),
|
'enable_async_append': Bool(140000, None),
|
||||||
'enable_bitmapscan': Bool(90300, None),
|
'enable_bitmapscan': Bool(90300, None),
|
||||||
'enable_gathermerge': Bool(100000, None),
|
'enable_gathermerge': Bool(100000, None),
|
||||||
'enable_group_by_reordering': Bool(150000, None),
|
|
||||||
'enable_hashagg': Bool(90300, None),
|
'enable_hashagg': Bool(90300, None),
|
||||||
'enable_hashjoin': Bool(90300, None),
|
'enable_hashjoin': Bool(90300, None),
|
||||||
'enable_incremental_sort': Bool(130000, None),
|
'enable_incremental_sort': Bool(130000, None),
|
||||||
|
|||||||
+14
-4
@@ -6,7 +6,7 @@ try:
|
|||||||
from . import MIN_PSYCOPG2, parse_version
|
from . import MIN_PSYCOPG2, parse_version
|
||||||
if parse_version(__version__) < MIN_PSYCOPG2:
|
if parse_version(__version__) < MIN_PSYCOPG2:
|
||||||
raise ImportError
|
raise ImportError
|
||||||
from psycopg2 import connect, Error, DatabaseError, OperationalError, ProgrammingError
|
from psycopg2 import connect as _connect, Error, DatabaseError, OperationalError, ProgrammingError
|
||||||
from psycopg2.extensions import adapt
|
from psycopg2.extensions import adapt
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@@ -20,10 +20,10 @@ try:
|
|||||||
value.prepare(conn)
|
value.prepare(conn)
|
||||||
return value.getquoted().decode('utf-8')
|
return value.getquoted().decode('utf-8')
|
||||||
except ImportError:
|
except ImportError:
|
||||||
from psycopg import connect as _connect, sql, Error, DatabaseError, OperationalError, ProgrammingError
|
from psycopg import connect as __connect, sql, Error, DatabaseError, OperationalError, ProgrammingError
|
||||||
|
|
||||||
def connect(*args, **kwargs):
|
def _connect(*args, **kwargs):
|
||||||
ret = _connect(*args, **kwargs)
|
ret = __connect(*args, **kwargs)
|
||||||
ret.server_version = ret.pgconn.server_version # compatibility with psycopg2
|
ret.server_version = ret.pgconn.server_version # compatibility with psycopg2
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
@@ -34,6 +34,16 @@ except ImportError:
|
|||||||
return sql.Literal(value).as_string(conn)
|
return sql.Literal(value).as_string(conn)
|
||||||
|
|
||||||
|
|
||||||
|
def connect(*args, **kwargs):
|
||||||
|
if kwargs and 'replication' not in kwargs and kwargs.get('fallback_application_name') != 'Patroni ctl':
|
||||||
|
options = [kwargs['options']] if 'options' in kwargs else []
|
||||||
|
options.append('-c search_path=pg_catalog')
|
||||||
|
kwargs['options'] = ' '.join(options)
|
||||||
|
ret = _connect(*args, **kwargs)
|
||||||
|
ret.autocommit = True
|
||||||
|
return ret
|
||||||
|
|
||||||
|
|
||||||
def quote_ident(value, conn=None):
|
def quote_ident(value, conn=None):
|
||||||
if _legacy or conn is None:
|
if _legacy or conn is None:
|
||||||
return '"{0}"'.format(value.replace('"', '""'))
|
return '"{0}"'.format(value.replace('"', '""'))
|
||||||
|
|||||||
+10
-3
@@ -9,9 +9,9 @@ from .utils import USER_AGENT
|
|||||||
|
|
||||||
class PatroniRequest(object):
|
class PatroniRequest(object):
|
||||||
|
|
||||||
def __init__(self, config, insecure=False):
|
def __init__(self, config, insecure=None):
|
||||||
cert_reqs = 'CERT_NONE' if insecure or config.get('ctl', {}).get('insecure', False) else 'CERT_REQUIRED'
|
self._insecure = insecure
|
||||||
self._pool = urllib3.PoolManager(num_pools=10, maxsize=10, cert_reqs=cert_reqs)
|
self._pool = urllib3.PoolManager(num_pools=10, maxsize=10)
|
||||||
self.reload_config(config)
|
self.reload_config(config)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
@@ -32,12 +32,19 @@ class PatroniRequest(object):
|
|||||||
def reload_config(self, config):
|
def reload_config(self, config):
|
||||||
self._pool.headers = urllib3.make_headers(basic_auth=self._get_cfg_value(config, 'auth'), user_agent=USER_AGENT)
|
self._pool.headers = urllib3.make_headers(basic_auth=self._get_cfg_value(config, 'auth'), user_agent=USER_AGENT)
|
||||||
|
|
||||||
|
insecure = self._insecure if isinstance(self._insecure, bool) else config.get('ctl', {}).get('insecure', False)
|
||||||
if self._apply_ssl_file_param(config, 'cert'):
|
if self._apply_ssl_file_param(config, 'cert'):
|
||||||
|
# With client certificate the cert_reqs must be set to CERT_REQUIRED even if insecure option is used
|
||||||
|
self._pool.connection_pool_kw['cert_reqs'] = 'CERT_REQUIRED'
|
||||||
|
# The assert_hostname = False helps to silence warnings
|
||||||
|
self._pool.connection_pool_kw['assert_hostname'] = False if insecure else None
|
||||||
|
|
||||||
self._apply_ssl_file_param(config, 'key')
|
self._apply_ssl_file_param(config, 'key')
|
||||||
|
|
||||||
password = self._get_cfg_value(config, 'keyfile_password')
|
password = self._get_cfg_value(config, 'keyfile_password')
|
||||||
self._apply_pool_param('key_password', password)
|
self._apply_pool_param('key_password', password)
|
||||||
else:
|
else:
|
||||||
|
self._pool.connection_pool_kw['cert_reqs'] = 'CERT_NONE' if insecure else 'CERT_REQUIRED'
|
||||||
self._pool.connection_pool_kw.pop('key_file', None)
|
self._pool.connection_pool_kw.pop('key_file', None)
|
||||||
|
|
||||||
cacert = config.get('ctl', {}).get('cacert') or config.get('restapi', {}).get('cafile')
|
cacert = config.get('ctl', {}).get('cacert') or config.get('restapi', {}).get('cafile')
|
||||||
|
|||||||
@@ -214,16 +214,16 @@ class WALERestore(object):
|
|||||||
attempts_no = 0
|
attempts_no = 0
|
||||||
while True:
|
while True:
|
||||||
if self.master_connection:
|
if self.master_connection:
|
||||||
|
con = None
|
||||||
try:
|
try:
|
||||||
# get the difference in bytes between the current WAL location and the backup start offset
|
# get the difference in bytes between the current WAL location and the backup start offset
|
||||||
with psycopg.connect(self.master_connection) as con:
|
con = psycopg.connect(self.master_connection)
|
||||||
if con.server_version >= 100000:
|
if con.server_version >= 100000:
|
||||||
wal_name = 'wal'
|
wal_name = 'wal'
|
||||||
lsn_name = 'lsn'
|
lsn_name = 'lsn'
|
||||||
else:
|
else:
|
||||||
wal_name = 'xlog'
|
wal_name = 'xlog'
|
||||||
lsn_name = 'location'
|
lsn_name = 'location'
|
||||||
con.autocommit = True
|
|
||||||
with con.cursor() as cur:
|
with con.cursor() as cur:
|
||||||
cur.execute(("SELECT CASE WHEN pg_catalog.pg_is_in_recovery()"
|
cur.execute(("SELECT CASE WHEN pg_catalog.pg_is_in_recovery()"
|
||||||
" THEN GREATEST(pg_catalog.pg_{0}_{1}_diff(COALESCE("
|
" THEN GREATEST(pg_catalog.pg_{0}_{1}_diff(COALESCE("
|
||||||
@@ -246,6 +246,9 @@ class WALERestore(object):
|
|||||||
logger.info("continue with base backup from S3 since master is not available")
|
logger.info("continue with base backup from S3 since master is not available")
|
||||||
diff_in_bytes = 0
|
diff_in_bytes = 0
|
||||||
break
|
break
|
||||||
|
finally:
|
||||||
|
if con:
|
||||||
|
con.close()
|
||||||
else:
|
else:
|
||||||
# always try to use WAL-E if master connection string is not available
|
# always try to use WAL-E if master connection string is not available
|
||||||
diff_in_bytes = 0
|
diff_in_bytes = 0
|
||||||
|
|||||||
+2
-2
@@ -519,8 +519,8 @@ def enable_keepalive(sock, timeout, idle, cnt=3):
|
|||||||
def find_executable(executable, path=None):
|
def find_executable(executable, path=None):
|
||||||
_, ext = os.path.splitext(executable)
|
_, ext = os.path.splitext(executable)
|
||||||
|
|
||||||
if (sys.platform == 'win32') and (ext != '.exe'):
|
if (sys.platform == 'win32') and (ext == ''):
|
||||||
executable = executable + '.exe'
|
executable = executable + '.exe' # Set default WIN extension
|
||||||
|
|
||||||
if os.path.isfile(executable):
|
if os.path.isfile(executable):
|
||||||
return executable
|
return executable
|
||||||
|
|||||||
@@ -37,6 +37,10 @@ def validate_host_port(host_port, listen=False, multiple_hosts=False):
|
|||||||
hosts = hosts.split(",")
|
hosts = hosts.split(",")
|
||||||
else:
|
else:
|
||||||
hosts = [hosts]
|
hosts = [hosts]
|
||||||
|
if "*" in hosts:
|
||||||
|
if len(hosts) != 1:
|
||||||
|
raise ConfigParseError("expecting '*' alone")
|
||||||
|
hosts = [p[-1][0] for p in socket.getaddrinfo(None, port, 0, socket.SOCK_STREAM, 0, socket.AI_PASSIVE)]
|
||||||
for host in hosts:
|
for host in hosts:
|
||||||
proto = socket.getaddrinfo(host, "", 0, socket.SOCK_STREAM, 0, socket.AI_PASSIVE)
|
proto = socket.getaddrinfo(host, "", 0, socket.SOCK_STREAM, 0, socket.AI_PASSIVE)
|
||||||
s = socket.socket(proto[0][0], socket.SOCK_STREAM)
|
s = socket.socket(proto[0][0], socket.SOCK_STREAM)
|
||||||
@@ -178,9 +182,11 @@ class Schema(object):
|
|||||||
self.validator = validator
|
self.validator = validator
|
||||||
|
|
||||||
def __call__(self, data):
|
def __call__(self, data):
|
||||||
|
errors = []
|
||||||
for i in self.validate(data):
|
for i in self.validate(data):
|
||||||
if not i.status:
|
if not i.status:
|
||||||
print(i)
|
errors.append(str(i))
|
||||||
|
return errors
|
||||||
|
|
||||||
def validate(self, data):
|
def validate(self, data):
|
||||||
self.data = data
|
self.data = data
|
||||||
@@ -364,6 +370,7 @@ schema = Schema({
|
|||||||
"postgresql": {
|
"postgresql": {
|
||||||
"listen": validate_host_port_listen_multiple_hosts,
|
"listen": validate_host_port_listen_multiple_hosts,
|
||||||
"connect_address": validate_connect_address,
|
"connect_address": validate_connect_address,
|
||||||
|
Optional("proxy_address"): validate_connect_address,
|
||||||
"authentication": {
|
"authentication": {
|
||||||
"replication": userattributes,
|
"replication": userattributes,
|
||||||
"superuser": userattributes,
|
"superuser": userattributes,
|
||||||
|
|||||||
+1
-1
@@ -1 +1 @@
|
|||||||
__version__ = '2.1.4'
|
__version__ = '2.1.7'
|
||||||
|
|||||||
@@ -215,6 +215,10 @@ class Watchdog(object):
|
|||||||
self._activate()
|
self._activate()
|
||||||
if self.config.timeout != self.active_config.timeout:
|
if self.config.timeout != self.active_config.timeout:
|
||||||
self.impl.set_timeout(self.config.timeout)
|
self.impl.set_timeout(self.config.timeout)
|
||||||
|
if self.is_running:
|
||||||
|
logger.info("{0} updated with {1} second timeout, timing slack {2} seconds"
|
||||||
|
.format(self.impl.describe(), self.impl.get_timeout(), self.config.timing_slack))
|
||||||
|
self.active_config = self.config
|
||||||
except WatchdogError as e:
|
except WatchdogError as e:
|
||||||
logger.error("Error while sending keepalive: %s", e)
|
logger.error("Error while sending keepalive: %s", e)
|
||||||
|
|
||||||
|
|||||||
+6
-2
@@ -5,15 +5,17 @@ name: postgresql0
|
|||||||
restapi:
|
restapi:
|
||||||
listen: 127.0.0.1:8008
|
listen: 127.0.0.1:8008
|
||||||
connect_address: 127.0.0.1:8008
|
connect_address: 127.0.0.1:8008
|
||||||
|
# cafile: /etc/ssl/certs/ssl-cacert-snakeoil.pem
|
||||||
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
||||||
# keyfile: /etc/ssl/private/ssl-cert-snakeoil.key
|
# keyfile: /etc/ssl/private/ssl-cert-snakeoil.key
|
||||||
# authentication:
|
# authentication:
|
||||||
# username: username
|
# username: username
|
||||||
# password: password
|
# password: password
|
||||||
|
|
||||||
# ctl:
|
#ctl:
|
||||||
# insecure: false # Allow connections to SSL sites without certs
|
# insecure: false # Allow connections to Patroni REST API without verifying certificates
|
||||||
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
||||||
|
# keyfile: /etc/ssl/private/ssl-cert-snakeoil.key
|
||||||
# cacert: /etc/ssl/certs/ssl-cacert-snakeoil.pem
|
# cacert: /etc/ssl/certs/ssl-cacert-snakeoil.pem
|
||||||
|
|
||||||
etcd:
|
etcd:
|
||||||
@@ -99,6 +101,8 @@ bootstrap:
|
|||||||
postgresql:
|
postgresql:
|
||||||
listen: 127.0.0.1:5432
|
listen: 127.0.0.1:5432
|
||||||
connect_address: 127.0.0.1:5432
|
connect_address: 127.0.0.1:5432
|
||||||
|
|
||||||
|
# proxy_address: 127.0.0.1:5433 # The address of connection pool (e.g., pgbouncer) running next to Patroni/Postgres. Only for service discovery.
|
||||||
data_dir: data/postgresql0
|
data_dir: data/postgresql0
|
||||||
# bin_dir:
|
# bin_dir:
|
||||||
# config_dir:
|
# config_dir:
|
||||||
|
|||||||
+5
-2
@@ -5,15 +5,17 @@ name: postgresql1
|
|||||||
restapi:
|
restapi:
|
||||||
listen: 127.0.0.1:8009
|
listen: 127.0.0.1:8009
|
||||||
connect_address: 127.0.0.1:8009
|
connect_address: 127.0.0.1:8009
|
||||||
|
# cafile: /etc/ssl/certs/ssl-cacert-snakeoil.pem
|
||||||
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
||||||
# keyfile: /etc/ssl/private/ssl-cert-snakeoil.key
|
# keyfile: /etc/ssl/private/ssl-cert-snakeoil.key
|
||||||
# authentication:
|
# authentication:
|
||||||
# username: username
|
# username: username
|
||||||
# password: password
|
# password: password
|
||||||
|
|
||||||
# ctl:
|
#ctl:
|
||||||
# insecure: false # Allow connections to SSL sites without certs
|
# insecure: false # Allow connections to Patroni REST API without verifying certificates
|
||||||
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
||||||
|
# keyfile: /etc/ssl/private/ssl-cert-snakeoil.key
|
||||||
# cacert: /etc/ssl/certs/ssl-cacert-snakeoil.pem
|
# cacert: /etc/ssl/certs/ssl-cacert-snakeoil.pem
|
||||||
|
|
||||||
etcd:
|
etcd:
|
||||||
@@ -93,6 +95,7 @@ bootstrap:
|
|||||||
postgresql:
|
postgresql:
|
||||||
listen: 127.0.0.1:5433
|
listen: 127.0.0.1:5433
|
||||||
connect_address: 127.0.0.1:5433
|
connect_address: 127.0.0.1:5433
|
||||||
|
# proxy_address: 127.0.0.1:5434 # The address of connection pool (e.g., pgbouncer) running next to Patroni/Postgres. Only for service discovery.
|
||||||
data_dir: data/postgresql1
|
data_dir: data/postgresql1
|
||||||
# bin_dir:
|
# bin_dir:
|
||||||
# config_dir:
|
# config_dir:
|
||||||
|
|||||||
+5
-2
@@ -5,15 +5,17 @@ name: postgresql2
|
|||||||
restapi:
|
restapi:
|
||||||
listen: 127.0.0.1:8010
|
listen: 127.0.0.1:8010
|
||||||
connect_address: 127.0.0.1:8010
|
connect_address: 127.0.0.1:8010
|
||||||
|
# cafile: /etc/ssl/certs/ssl-cacert-snakeoil.pem
|
||||||
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
||||||
# keyfile: /etc/ssl/private/ssl-cert-snakeoil.key
|
# keyfile: /etc/ssl/private/ssl-cert-snakeoil.key
|
||||||
authentication:
|
authentication:
|
||||||
username: username
|
username: username
|
||||||
password: password
|
password: password
|
||||||
|
|
||||||
# ctl:
|
#ctl:
|
||||||
# insecure: false # Allow connections to SSL sites without certs
|
# insecure: false # Allow connections to Patroni REST API without verifying certificates
|
||||||
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
||||||
|
# keyfile: /etc/ssl/private/ssl-cert-snakeoil.key
|
||||||
# cacert: /etc/ssl/certs/ssl-cacert-snakeoil.pem
|
# cacert: /etc/ssl/certs/ssl-cacert-snakeoil.pem
|
||||||
|
|
||||||
etcd:
|
etcd:
|
||||||
@@ -90,6 +92,7 @@ bootstrap:
|
|||||||
postgresql:
|
postgresql:
|
||||||
listen: 127.0.0.1:5434
|
listen: 127.0.0.1:5434
|
||||||
connect_address: 127.0.0.1:5434
|
connect_address: 127.0.0.1:5434
|
||||||
|
# proxy_address: 127.0.0.1:5435 # The address of connection pool (e.g., pgbouncer) running next to Patroni/Postgres. Only for service discovery.
|
||||||
data_dir: data/postgresql2
|
data_dir: data/postgresql2
|
||||||
# bin_dir:
|
# bin_dir:
|
||||||
# config_dir:
|
# config_dir:
|
||||||
|
|||||||
+18
-21
@@ -1,31 +1,28 @@
|
|||||||
#!/bin/sh
|
#!/bin/bash
|
||||||
|
|
||||||
if [ $# -ne 1 ]; then
|
# Release process:
|
||||||
>&2 echo "usage: $0 <version>"
|
# 1. Open a PR that updates release notes and Patroni version
|
||||||
exit 1
|
# 2. Merge it
|
||||||
fi
|
# 3. Run release.sh
|
||||||
|
# 4. After the new tag is pushed, the .github/workflows/release.yaml will run tests and upload the new package to test.pypi.org
|
||||||
readonly VERSIONFILE="patroni/version.py"
|
# 5. Once the release is created, the .github/workflows/release.yaml will run tests and upload the new package to pypi.org
|
||||||
|
|
||||||
## Bail out on any non-zero exitcode from the called processes
|
## Bail out on any non-zero exitcode from the called processes
|
||||||
set -xe
|
set -xe
|
||||||
|
|
||||||
python3 --version
|
if python3 --version &> /dev/null; then
|
||||||
|
alias python=python3
|
||||||
|
shopt -s expand_aliases
|
||||||
|
fi
|
||||||
|
|
||||||
|
python --version
|
||||||
git --version
|
git --version
|
||||||
|
|
||||||
version=$1
|
version=$(python -c 'from patroni.version import __version__; print(__version__)')
|
||||||
|
|
||||||
sed -i "s/__version__ = .*/__version__ = '${version}'/" "${VERSIONFILE}"
|
python setup.py clean
|
||||||
python3 setup.py clean
|
python setup.py test
|
||||||
python3 setup.py test
|
python setup.py flake8
|
||||||
python3 setup.py flake8
|
|
||||||
|
|
||||||
git add "${VERSIONFILE}"
|
git tag "v$version"
|
||||||
|
|
||||||
git commit -m "Bumped version to $version"
|
|
||||||
git push
|
|
||||||
|
|
||||||
python3 setup.py sdist bdist_wheel upload
|
|
||||||
|
|
||||||
git tag v${version}
|
|
||||||
git push --tags
|
git push --tags
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
psycopg2-binary
|
psycopg2-binary
|
||||||
behave
|
behave
|
||||||
coverage
|
coverage
|
||||||
flake8
|
flake8>=3.0.0
|
||||||
mock
|
mock
|
||||||
pytest-cov
|
pytest-cov
|
||||||
pytest
|
pytest
|
||||||
|
|||||||
@@ -18,8 +18,8 @@ MAIN_PACKAGE = NAME
|
|||||||
DESCRIPTION = 'PostgreSQL High-Available orchestrator and CLI'
|
DESCRIPTION = 'PostgreSQL High-Available orchestrator and CLI'
|
||||||
LICENSE = 'The MIT License'
|
LICENSE = 'The MIT License'
|
||||||
URL = 'https://github.com/zalando/patroni'
|
URL = 'https://github.com/zalando/patroni'
|
||||||
AUTHOR = 'Alexander Kukushkin, Dmitrii Dolgov, Oleksii Kliukin'
|
AUTHOR = 'Alexander Kukushkin, Polina Bungina'
|
||||||
AUTHOR_EMAIL = 'alexander.kukushkin@zalando.de, [email protected], [email protected]'
|
AUTHOR_EMAIL = 'akukushkin@microsoft.com, [email protected]'
|
||||||
KEYWORDS = 'etcd governor patroni postgresql postgres ha haproxy confd' +\
|
KEYWORDS = 'etcd governor patroni postgresql postgres ha haproxy confd' +\
|
||||||
' zookeeper exhibitor consul streaming replication kubernetes k8s'
|
' zookeeper exhibitor consul streaming replication kubernetes k8s'
|
||||||
|
|
||||||
@@ -93,12 +93,10 @@ class Flake8(_Command):
|
|||||||
return [package for package in self.package_files()] + ['tests', 'setup.py']
|
return [package for package in self.package_files()] + ['tests', 'setup.py']
|
||||||
|
|
||||||
def run(self):
|
def run(self):
|
||||||
from flake8.main import application
|
from flake8.main.cli import main
|
||||||
|
|
||||||
logging.getLogger().setLevel(logging.ERROR)
|
logging.getLogger().setLevel(logging.ERROR)
|
||||||
flake8 = application.Application()
|
main(self.targets())
|
||||||
flake8.run(self.targets())
|
|
||||||
flake8.exit()
|
|
||||||
|
|
||||||
|
|
||||||
class PyTest(_Command):
|
class PyTest(_Command):
|
||||||
|
|||||||
+4
-3
@@ -50,7 +50,7 @@ def requests_get(url, **kwargs):
|
|||||||
if url.startswith('http://local'):
|
if url.startswith('http://local'):
|
||||||
raise urllib3.exceptions.HTTPError()
|
raise urllib3.exceptions.HTTPError()
|
||||||
elif ':8011/patroni' in url:
|
elif ':8011/patroni' in url:
|
||||||
response.content = '{"role": "replica", "xlog": {"received_location": 0}, "tags": {}}'
|
response.content = '{"role": "replica", "wal": {"received_location": 0}, "tags": {}}'
|
||||||
elif url.endswith('/members'):
|
elif url.endswith('/members'):
|
||||||
response.content = '[{}]' if url.startswith('http://error') else members
|
response.content = '[{}]' if url.startswith('http://error') else members
|
||||||
elif url.startswith('http://exhibitor'):
|
elif url.startswith('http://exhibitor'):
|
||||||
@@ -177,7 +177,7 @@ class PostgresInit(unittest.TestCase):
|
|||||||
'force_parallel_mode': '1', 'constraint_exclusion': '',
|
'force_parallel_mode': '1', 'constraint_exclusion': '',
|
||||||
'max_stack_depth': 'Z', 'vacuum_cost_limit': -1, 'vacuum_cost_delay': 200}
|
'max_stack_depth': 'Z', 'vacuum_cost_limit': -1, 'vacuum_cost_delay': 200}
|
||||||
|
|
||||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
@patch('patroni.psycopg._connect', psycopg_connect)
|
||||||
@patch('patroni.postgresql.CallbackExecutor', Mock())
|
@patch('patroni.postgresql.CallbackExecutor', Mock())
|
||||||
@patch.object(ConfigHandler, 'write_postgresql_conf', Mock())
|
@patch.object(ConfigHandler, 'write_postgresql_conf', Mock())
|
||||||
@patch.object(ConfigHandler, 'replace_pg_hba', Mock())
|
@patch.object(ConfigHandler, 'replace_pg_hba', Mock())
|
||||||
@@ -188,7 +188,8 @@ class PostgresInit(unittest.TestCase):
|
|||||||
self.p = Postgresql({'name': 'postgresql0', 'scope': 'batman', 'data_dir': data_dir,
|
self.p = Postgresql({'name': 'postgresql0', 'scope': 'batman', 'data_dir': data_dir,
|
||||||
'config_dir': data_dir, 'retry_timeout': 10,
|
'config_dir': data_dir, 'retry_timeout': 10,
|
||||||
'krbsrvname': 'postgres', 'pgpass': os.path.join(data_dir, 'pgpass0'),
|
'krbsrvname': 'postgres', 'pgpass': os.path.join(data_dir, 'pgpass0'),
|
||||||
'listen': '127.0.0.2, 127.0.0.3:5432', 'connect_address': '127.0.0.2:5432',
|
'listen': '127.0.0.2, 127.0.0.3:5432',
|
||||||
|
'connect_address': '127.0.0.2:5432', 'proxy_address': '127.0.0.2:5433',
|
||||||
'authentication': {'superuser': {'username': 'foo', 'password': 'test'},
|
'authentication': {'superuser': {'username': 'foo', 'password': 'test'},
|
||||||
'replication': {'username': '', 'password': 'rep-pass'},
|
'replication': {'username': '', 'password': 'rep-pass'},
|
||||||
'rewind': {'username': 'rewind', 'password': 'test'}},
|
'rewind': {'username': 'rewind', 'password': 'test'}},
|
||||||
|
|||||||
+28
-25
@@ -12,6 +12,7 @@ from patroni.ha import _MemberStatus
|
|||||||
from patroni.utils import tzutc
|
from patroni.utils import tzutc
|
||||||
from six import BytesIO as IO
|
from six import BytesIO as IO
|
||||||
from six.moves import BaseHTTPServer
|
from six.moves import BaseHTTPServer
|
||||||
|
from six.moves.socketserver import ThreadingMixIn
|
||||||
from . import psycopg_connect, MockCursor
|
from . import psycopg_connect, MockCursor
|
||||||
from .test_ha import get_cluster_initialized_without_leader
|
from .test_ha import get_cluster_initialized_without_leader
|
||||||
|
|
||||||
@@ -46,6 +47,10 @@ class MockPostgresql(object):
|
|||||||
def replica_cached_timeline(_):
|
def replica_cached_timeline(_):
|
||||||
return 2
|
return 2
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def is_running():
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
class MockWatchdog(object):
|
class MockWatchdog(object):
|
||||||
is_healthy = False
|
is_healthy = False
|
||||||
@@ -129,6 +134,10 @@ class MockPatroni(object):
|
|||||||
def sighup_handler():
|
def sighup_handler():
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def api_sigterm():
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
class MockRequest(object):
|
class MockRequest(object):
|
||||||
|
|
||||||
@@ -169,7 +178,6 @@ class TestRestApiHandler(unittest.TestCase):
|
|||||||
MockRestApiServer(RestApiHandler, 'GET /replica?lag=10MB')
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=10MB')
|
||||||
MockRestApiServer(RestApiHandler, 'GET /replica?lag=10485760')
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=10485760')
|
||||||
MockRestApiServer(RestApiHandler, 'GET /read-only')
|
MockRestApiServer(RestApiHandler, 'GET /read-only')
|
||||||
MockRestApiServer(RestApiHandler, 'GET /read-only-sync')
|
|
||||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={})):
|
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={})):
|
||||||
MockRestApiServer(RestApiHandler, 'GET /replica')
|
MockRestApiServer(RestApiHandler, 'GET /replica')
|
||||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'master'})):
|
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'master'})):
|
||||||
@@ -181,13 +189,13 @@ class TestRestApiHandler(unittest.TestCase):
|
|||||||
MockPatroni.dcs.cluster.is_synchronous_mode = Mock(return_value=True)
|
MockPatroni.dcs.cluster.is_synchronous_mode = Mock(return_value=True)
|
||||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
|
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
|
||||||
MockRestApiServer(RestApiHandler, 'GET /synchronous')
|
MockRestApiServer(RestApiHandler, 'GET /synchronous')
|
||||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
|
|
||||||
MockRestApiServer(RestApiHandler, 'GET /read-only-sync')
|
MockRestApiServer(RestApiHandler, 'GET /read-only-sync')
|
||||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
|
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
|
||||||
MockPatroni.dcs.cluster.sync.members = []
|
MockPatroni.dcs.cluster.sync.members = []
|
||||||
MockRestApiServer(RestApiHandler, 'GET /asynchronous')
|
MockRestApiServer(RestApiHandler, 'GET /asynchronous')
|
||||||
with patch.object(MockHa, 'is_leader', Mock(return_value=True)):
|
with patch.object(MockHa, 'is_leader', Mock(return_value=True)):
|
||||||
MockRestApiServer(RestApiHandler, 'GET /replica')
|
MockRestApiServer(RestApiHandler, 'GET /replica')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /read-only-sync')
|
||||||
with patch.object(MockHa, 'is_standby_cluster', Mock(return_value=True)):
|
with patch.object(MockHa, 'is_standby_cluster', Mock(return_value=True)):
|
||||||
MockRestApiServer(RestApiHandler, 'GET /standby_leader')
|
MockRestApiServer(RestApiHandler, 'GET /standby_leader')
|
||||||
MockPatroni.dcs.cluster = None
|
MockPatroni.dcs.cluster = None
|
||||||
@@ -287,7 +295,12 @@ class TestRestApiHandler(unittest.TestCase):
|
|||||||
def test_do_OPTIONS(self):
|
def test_do_OPTIONS(self):
|
||||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'OPTIONS / HTTP/1.0'))
|
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'OPTIONS / HTTP/1.0'))
|
||||||
|
|
||||||
def test_do_GET_liveness(self):
|
def test_do_HEAD(self):
|
||||||
|
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'HEAD / HTTP/1.0'))
|
||||||
|
|
||||||
|
@patch.object(MockPatroni, 'dcs')
|
||||||
|
def test_do_GET_liveness(self, mock_dcs):
|
||||||
|
mock_dcs.ttl.return_value = PropertyMock(30)
|
||||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /liveness HTTP/1.0'))
|
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /liveness HTTP/1.0'))
|
||||||
|
|
||||||
def test_do_GET_readiness(self):
|
def test_do_GET_readiness(self):
|
||||||
@@ -362,6 +375,11 @@ class TestRestApiHandler(unittest.TestCase):
|
|||||||
def test_do_POST_reload(self):
|
def test_do_POST_reload(self):
|
||||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'POST /reload HTTP/1.0' + self._authorization))
|
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'POST /reload HTTP/1.0' + self._authorization))
|
||||||
|
|
||||||
|
@patch('os.environ', {'BEHAVE_DEBUG': 'true'})
|
||||||
|
@patch('os.name', 'nt')
|
||||||
|
def test_do_POST_sigterm(self):
|
||||||
|
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'POST /sigterm HTTP/1.0' + self._authorization))
|
||||||
|
|
||||||
@patch.object(MockPatroni, 'dcs')
|
@patch.object(MockPatroni, 'dcs')
|
||||||
def test_do_POST_restart(self, mock_dcs):
|
def test_do_POST_restart(self, mock_dcs):
|
||||||
mock_dcs.get_cluster.return_value.is_paused.return_value = False
|
mock_dcs.get_cluster.return_value.is_paused.return_value = False
|
||||||
@@ -578,31 +596,16 @@ class TestRestApiServer(unittest.TestCase):
|
|||||||
def test_socket_error(self):
|
def test_socket_error(self):
|
||||||
self.assertRaises(socket.error, MockRestApiServer, Mock(), '', {'listen': '*:8008'})
|
self.assertRaises(socket.error, MockRestApiServer, Mock(), '', {'listen': '*:8008'})
|
||||||
|
|
||||||
@patch.object(MockRestApiServer, 'finish_request', Mock())
|
@patch.object(ThreadingMixIn, 'process_request_thread', Mock())
|
||||||
def test_process_request_thread(self):
|
def test_process_request_thread(self):
|
||||||
mock_socket = Mock()
|
self.srv.process_request_thread(Mock(), '2')
|
||||||
self.srv.process_request_thread((mock_socket, 1), '2')
|
|
||||||
mock_socket.context.wrap_socket.side_effect = socket.error
|
|
||||||
self.srv.process_request_thread((mock_socket, 1), '2')
|
|
||||||
|
|
||||||
@patch.object(socket.socket, 'accept')
|
|
||||||
def test_get_request(self, mock_accept):
|
|
||||||
newsock = Mock()
|
|
||||||
mock_accept.return_value = (newsock, '2')
|
|
||||||
self.srv.socket = Mock()
|
|
||||||
self.assertEqual(self.srv.get_request(), ((self.srv.socket, newsock), '2'))
|
|
||||||
|
|
||||||
@patch.object(MockRestApiServer, 'process_request', Mock(side_effect=RuntimeError))
|
@patch.object(MockRestApiServer, 'process_request', Mock(side_effect=RuntimeError))
|
||||||
def test_process_request_error(self):
|
@patch.object(MockRestApiServer, 'get_request')
|
||||||
mock_address = ('127.0.0.1', 55555)
|
def test_process_request_error(self, mock_get_request):
|
||||||
mock_socket = Mock()
|
mock_request = Mock()
|
||||||
mock_ssl_socket = (Mock(), Mock())
|
mock_request.unwrap.side_effect = Exception
|
||||||
for mock_request in (mock_socket, mock_ssl_socket):
|
mock_get_request.return_value = (mock_request, ('127.0.0.1', 55555))
|
||||||
with patch.object(
|
|
||||||
MockRestApiServer,
|
|
||||||
'get_request',
|
|
||||||
Mock(return_value=(mock_request, mock_address))
|
|
||||||
):
|
|
||||||
self.srv._handle_request_noblock()
|
self.srv._handle_request_noblock()
|
||||||
|
|
||||||
@patch('ssl._ssl._test_decode_cert', Mock())
|
@patch('ssl._ssl._test_decode_cert', Mock())
|
||||||
|
|||||||
@@ -40,6 +40,7 @@ class TestConfig(unittest.TestCase):
|
|||||||
'PATRONI_RESTAPI_ALLOWLIST_INCLUDE_MEMBERS': 'on',
|
'PATRONI_RESTAPI_ALLOWLIST_INCLUDE_MEMBERS': 'on',
|
||||||
'PATRONI_POSTGRESQL_LISTEN': '0.0.0.0:5432',
|
'PATRONI_POSTGRESQL_LISTEN': '0.0.0.0:5432',
|
||||||
'PATRONI_POSTGRESQL_CONNECT_ADDRESS': '127.0.0.1:5432',
|
'PATRONI_POSTGRESQL_CONNECT_ADDRESS': '127.0.0.1:5432',
|
||||||
|
'PATRONI_POSTGRESQL_PROXY_ADDRESS': '127.0.0.1:5433',
|
||||||
'PATRONI_POSTGRESQL_DATA_DIR': 'data/postgres0',
|
'PATRONI_POSTGRESQL_DATA_DIR': 'data/postgres0',
|
||||||
'PATRONI_POSTGRESQL_CONFIG_DIR': 'data/postgres0',
|
'PATRONI_POSTGRESQL_CONFIG_DIR': 'data/postgres0',
|
||||||
'PATRONI_POSTGRESQL_PGPASS': '/tmp/pgpass0',
|
'PATRONI_POSTGRESQL_PGPASS': '/tmp/pgpass0',
|
||||||
|
|||||||
+31
-10
@@ -4,7 +4,7 @@ import unittest
|
|||||||
from consul import ConsulException, NotFound
|
from consul import ConsulException, NotFound
|
||||||
from mock import Mock, PropertyMock, patch
|
from mock import Mock, PropertyMock, patch
|
||||||
from patroni.dcs.consul import AbstractDCS, Cluster, Consul, ConsulInternalError, \
|
from patroni.dcs.consul import AbstractDCS, Cluster, Consul, ConsulInternalError, \
|
||||||
ConsulError, ConsulClient, HTTPClient, InvalidSessionTTL, InvalidSession
|
ConsulError, ConsulClient, HTTPClient, InvalidSessionTTL, InvalidSession, RetryFailedError
|
||||||
from . import SleepException
|
from . import SleepException
|
||||||
|
|
||||||
|
|
||||||
@@ -34,6 +34,8 @@ def kv_get(self, key, **kwargs):
|
|||||||
'ModifyIndex': 6429, 'Value': b'4496294792'},
|
'ModifyIndex': 6429, 'Value': b'4496294792'},
|
||||||
{'CreateIndex': 1085, 'Flags': 0, 'Key': key + 'sync', 'LockIndex': 0,
|
{'CreateIndex': 1085, 'Flags': 0, 'Key': key + 'sync', 'LockIndex': 0,
|
||||||
'ModifyIndex': 6429, 'Value': b'{"leader": "leader", "sync_standby": null}'},
|
'ModifyIndex': 6429, 'Value': b'{"leader": "leader", "sync_standby": null}'},
|
||||||
|
{'CreateIndex': 1085, 'Flags': 0, 'Key': key + 'failsafe', 'LockIndex': 0,
|
||||||
|
'ModifyIndex': 6429, 'Value': b'{'},
|
||||||
{'CreateIndex': 1085, 'Flags': 0, 'Key': key + 'status', 'LockIndex': 0,
|
{'CreateIndex': 1085, 'Flags': 0, 'Key': key + 'status', 'LockIndex': 0,
|
||||||
'ModifyIndex': 6429, 'Value': b'{"optime":4496294792, "slots":{"ls":12345}}'}])
|
'ModifyIndex': 6429, 'Value': b'{"optime":4496294792, "slots":{"ls":12345}}'}])
|
||||||
if key == 'service/good/':
|
if key == 'service/good/':
|
||||||
@@ -122,9 +124,6 @@ class TestConsul(unittest.TestCase):
|
|||||||
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
||||||
self.c._base_path = '/service/legacy'
|
self.c._base_path = '/service/legacy'
|
||||||
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
||||||
self.c._base_path = '/service/good'
|
|
||||||
self.c._session = 'fd4f44fe-2cac-bba5-a60b-304b51ff39b8'
|
|
||||||
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
|
||||||
|
|
||||||
@patch.object(consul.Consul.KV, 'delete', Mock(side_effect=[ConsulException, True, True, True]))
|
@patch.object(consul.Consul.KV, 'delete', Mock(side_effect=[ConsulException, True, True, True]))
|
||||||
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=[True, ConsulException, InvalidSession]))
|
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=[True, ConsulException, InvalidSession]))
|
||||||
@@ -140,11 +139,15 @@ class TestConsul(unittest.TestCase):
|
|||||||
self.c.refresh_session = Mock(side_effect=ConsulError('foo'))
|
self.c.refresh_session = Mock(side_effect=ConsulError('foo'))
|
||||||
self.assertFalse(self.c.touch_member({'balbla': 'blabla'}))
|
self.assertFalse(self.c.touch_member({'balbla': 'blabla'}))
|
||||||
|
|
||||||
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=InvalidSession))
|
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=[InvalidSession, False, InvalidSession]))
|
||||||
def test_take_leader(self):
|
def test_take_leader(self):
|
||||||
self.c.set_ttl(20)
|
self.c.set_ttl(20)
|
||||||
self.c.refresh_session = Mock()
|
self.c._do_refresh_session = Mock()
|
||||||
self.c.take_leader()
|
self.assertFalse(self.c.take_leader())
|
||||||
|
with patch('time.time', Mock(side_effect=[0, 100])):
|
||||||
|
self.assertRaises(ConsulError, self.c.take_leader)
|
||||||
|
with patch('time.time', Mock(side_effect=[0, 0, 0, 0, 0, 0, 100])):
|
||||||
|
self.assertRaises(ConsulError, self.c.take_leader)
|
||||||
|
|
||||||
@patch.object(consul.Consul.KV, 'put', Mock(return_value=True))
|
@patch.object(consul.Consul.KV, 'put', Mock(return_value=True))
|
||||||
def test_set_failover_value(self):
|
def test_set_failover_value(self):
|
||||||
@@ -160,9 +163,26 @@ class TestConsul(unittest.TestCase):
|
|||||||
self.c.get_cluster()
|
self.c.get_cluster()
|
||||||
self.c.write_leader_optime('1')
|
self.c.write_leader_optime('1')
|
||||||
|
|
||||||
@patch.object(consul.Consul.Session, 'renew', Mock())
|
@patch.object(consul.Consul.Session, 'renew')
|
||||||
def test_update_leader(self):
|
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=ConsulException))
|
||||||
self.c.update_leader(12345)
|
def test_update_leader(self, mock_renew):
|
||||||
|
self.c._session = 'fd4f44fe-2cac-bba5-a60b-304b51ff39b8'
|
||||||
|
with patch.object(consul.Consul.KV, 'delete', Mock(return_value=True)):
|
||||||
|
with patch.object(consul.Consul.KV, 'put', Mock(return_value=True)):
|
||||||
|
self.assertTrue(self.c.update_leader(12345, failsafe={'foo': 'bar'}))
|
||||||
|
with patch.object(consul.Consul.KV, 'put', Mock(side_effect=ConsulException)):
|
||||||
|
self.assertFalse(self.c.update_leader(12345))
|
||||||
|
with patch('time.time', Mock(side_effect=[0, 0, 0, 0, 0, 100, 200, 300])):
|
||||||
|
self.assertRaises(ConsulError, self.c.update_leader, 12345)
|
||||||
|
with patch('time.time', Mock(side_effect=[0, 100, 200, 300])):
|
||||||
|
self.assertRaises(ConsulError, self.c.update_leader, 12345)
|
||||||
|
with patch.object(consul.Consul.KV, 'delete', Mock(side_effect=ConsulException)):
|
||||||
|
self.assertFalse(self.c.update_leader(12347))
|
||||||
|
mock_renew.side_effect = RetryFailedError('')
|
||||||
|
self.c._last_session_refresh = 0
|
||||||
|
self.assertRaises(ConsulError, self.c.update_leader, 12346)
|
||||||
|
mock_renew.side_effect = ConsulException
|
||||||
|
self.assertFalse(self.c.update_leader(12347))
|
||||||
|
|
||||||
@patch.object(consul.Consul.KV, 'delete', Mock(return_value=True))
|
@patch.object(consul.Consul.KV, 'delete', Mock(return_value=True))
|
||||||
def test_delete_leader(self):
|
def test_delete_leader(self):
|
||||||
@@ -217,6 +237,7 @@ class TestConsul(unittest.TestCase):
|
|||||||
d['role'] = 'bla'
|
d['role'] = 'bla'
|
||||||
self.assertIsNone(self.c.update_service({}, d))
|
self.assertIsNone(self.c.update_service({}, d))
|
||||||
|
|
||||||
|
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=ConsulException))
|
||||||
def test_reload_config(self):
|
def test_reload_config(self):
|
||||||
self.assertEqual([], self.c._service_tags)
|
self.assertEqual([], self.c._service_tags)
|
||||||
self.c.reload_config({'consul': {'token': 'foo', 'register_service': True, 'service_tags': ['foo']},
|
self.c.reload_config({'consul': {'token': 'foo', 'register_service': True, 'service_tags': ['foo']},
|
||||||
|
|||||||
+47
-22
@@ -5,12 +5,13 @@ import unittest
|
|||||||
from click.testing import CliRunner
|
from click.testing import CliRunner
|
||||||
from datetime import datetime, timedelta
|
from datetime import datetime, timedelta
|
||||||
from mock import patch, Mock
|
from mock import patch, Mock
|
||||||
from patroni.ctl import ctl, store_config, load_config, output_members, get_dcs, parse_dcs, \
|
from patroni.ctl import ctl, load_config, output_members, get_dcs, parse_dcs, \
|
||||||
get_all_members, get_any_member, get_cursor, query_member, configure, PatroniCtlException, apply_config_changes, \
|
get_all_members, get_any_member, get_cursor, query_member, PatroniCtlException, apply_config_changes, \
|
||||||
format_config_for_editing, show_diff, invoke_editor, format_pg_version, CONFIG_FILE_PATH
|
format_config_for_editing, show_diff, invoke_editor, format_pg_version, CONFIG_FILE_PATH, PatronictlPrettyTable
|
||||||
from patroni.dcs.etcd import AbstractEtcdClientWithFailover, Failover
|
from patroni.dcs.etcd import AbstractEtcdClientWithFailover, Failover
|
||||||
from patroni.psycopg import OperationalError
|
from patroni.psycopg import OperationalError
|
||||||
from patroni.utils import tzutc
|
from patroni.utils import tzutc
|
||||||
|
from prettytable import PrettyTable, ALL
|
||||||
from urllib3 import PoolManager
|
from urllib3 import PoolManager
|
||||||
|
|
||||||
from . import MockConnect, MockCursor, MockResponse, psycopg_connect
|
from . import MockConnect, MockCursor, MockResponse, psycopg_connect
|
||||||
@@ -19,17 +20,6 @@ from .test_ha import get_cluster_initialized_without_leader, get_cluster_initial
|
|||||||
get_cluster_initialized_with_only_leader, get_cluster_not_initialized_without_leader, get_cluster, Member
|
get_cluster_initialized_with_only_leader, get_cluster_not_initialized_without_leader, get_cluster, Member
|
||||||
|
|
||||||
|
|
||||||
def test_rw_config():
|
|
||||||
runner = CliRunner()
|
|
||||||
with runner.isolated_filesystem():
|
|
||||||
load_config(CONFIG_FILE_PATH, None)
|
|
||||||
CONFIG_PATH = './test-ctl.yaml'
|
|
||||||
store_config({'etcd': {'host': 'localhost:2379'}}, CONFIG_PATH + '/dummy')
|
|
||||||
load_config(CONFIG_PATH + '/dummy', '0.0.0.0')
|
|
||||||
os.remove(CONFIG_PATH + '/dummy')
|
|
||||||
os.rmdir(CONFIG_PATH)
|
|
||||||
|
|
||||||
|
|
||||||
@patch('patroni.ctl.load_config', Mock(return_value={
|
@patch('patroni.ctl.load_config', Mock(return_value={
|
||||||
'scope': 'alpha', 'restapi': {'listen': '::', 'certfile': 'a'}, 'etcd': {'host': 'localhost:2379'},
|
'scope': 'alpha', 'restapi': {'listen': '::', 'certfile': 'a'}, 'etcd': {'host': 'localhost:2379'},
|
||||||
'postgresql': {'data_dir': '.', 'pgpass': './pgpass', 'parameters': {}, 'retry_timeout': 5}}))
|
'postgresql': {'data_dir': '.', 'pgpass': './pgpass', 'parameters': {}, 'retry_timeout': 5}}))
|
||||||
@@ -42,11 +32,27 @@ class TestCtl(unittest.TestCase):
|
|||||||
self.runner = CliRunner()
|
self.runner = CliRunner()
|
||||||
self.e = get_dcs({'etcd': {'ttl': 30, 'host': 'ok:2379', 'retry_timeout': 10}}, 'foo')
|
self.e = get_dcs({'etcd': {'ttl': 30, 'host': 'ok:2379', 'retry_timeout': 10}}, 'foo')
|
||||||
|
|
||||||
def test_load_config(self):
|
@patch('patroni.ctl.logging.debug')
|
||||||
|
def test_load_config(self, mock_logger_debug):
|
||||||
runner = CliRunner()
|
runner = CliRunner()
|
||||||
with runner.isolated_filesystem():
|
with runner.isolated_filesystem():
|
||||||
self.assertRaises(PatroniCtlException, load_config, './non-existing-config-file', None)
|
self.assertRaises(PatroniCtlException, load_config, './non-existing-config-file', None)
|
||||||
self.assertRaises(PatroniCtlException, load_config, './non-existing-config-file', None)
|
|
||||||
|
with patch('os.path.exists', Mock(return_value=True)), \
|
||||||
|
patch('patroni.config.Config._load_config_path', Mock(return_value={})):
|
||||||
|
load_config(CONFIG_FILE_PATH, None)
|
||||||
|
mock_logger_debug.assert_called_once()
|
||||||
|
self.assertEqual(('Ignoring configuration file "%s". It does not exists or is not readable.',
|
||||||
|
CONFIG_FILE_PATH),
|
||||||
|
mock_logger_debug.call_args[0])
|
||||||
|
mock_logger_debug.reset_mock()
|
||||||
|
|
||||||
|
with patch('os.access', Mock(return_value=True)):
|
||||||
|
load_config(CONFIG_FILE_PATH, '')
|
||||||
|
mock_logger_debug.assert_called_once()
|
||||||
|
self.assertEqual(('Loading configuration from file %s', CONFIG_FILE_PATH),
|
||||||
|
mock_logger_debug.call_args[0])
|
||||||
|
mock_logger_debug.reset_mock()
|
||||||
|
|
||||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
def test_get_cursor(self):
|
def test_get_cursor(self):
|
||||||
@@ -379,10 +385,6 @@ class TestCtl(unittest.TestCase):
|
|||||||
with patch('patroni.ctl.load_config', Mock(return_value={})):
|
with patch('patroni.ctl.load_config', Mock(return_value={})):
|
||||||
self.runner.invoke(ctl, ['list'])
|
self.runner.invoke(ctl, ['list'])
|
||||||
|
|
||||||
def test_configure(self):
|
|
||||||
result = self.runner.invoke(configure, ['--dcs', 'abc', '-c', 'dummy', '-n', 'bla'])
|
|
||||||
assert result.exit_code == 0
|
|
||||||
|
|
||||||
@patch('patroni.ctl.get_dcs')
|
@patch('patroni.ctl.get_dcs')
|
||||||
def test_scaffold(self, mock_get_dcs):
|
def test_scaffold(self, mock_get_dcs):
|
||||||
mock_get_dcs.return_value = self.e
|
mock_get_dcs.return_value = self.e
|
||||||
@@ -562,7 +564,8 @@ class TestCtl(unittest.TestCase):
|
|||||||
|
|
||||||
@patch('sys.stdout.isatty', return_value=False)
|
@patch('sys.stdout.isatty', return_value=False)
|
||||||
@patch('patroni.ctl.markup_to_pager')
|
@patch('patroni.ctl.markup_to_pager')
|
||||||
def test_show_diff(self, mock_markup_to_pager, mock_isatty):
|
@patch('patroni.ctl.find_executable', return_value=None)
|
||||||
|
def test_show_diff(self, mock_find_executable, mock_markup_to_pager, mock_isatty):
|
||||||
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
|
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
|
||||||
mock_markup_to_pager.assert_not_called()
|
mock_markup_to_pager.assert_not_called()
|
||||||
|
|
||||||
@@ -570,10 +573,10 @@ class TestCtl(unittest.TestCase):
|
|||||||
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
|
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
|
||||||
mock_markup_to_pager.assert_called_once()
|
mock_markup_to_pager.assert_called_once()
|
||||||
|
|
||||||
with patch('patroni.ctl.find_executable', Mock(return_value=None)):
|
|
||||||
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
|
show_diff("foo:\n bar: 1\n", "foo:\n bar: 2\n")
|
||||||
|
|
||||||
# Test that unicode handling doesn't fail with an exception
|
# Test that unicode handling doesn't fail with an exception
|
||||||
|
mock_find_executable.return_value = '/usr/bin/less'
|
||||||
show_diff(b"foo:\n bar: \xc3\xb6\xc3\xb6\n".decode('utf-8'),
|
show_diff(b"foo:\n bar: \xc3\xb6\xc3\xb6\n".decode('utf-8'),
|
||||||
b"foo:\n bar: \xc3\xbc\xc3\xbc\n".decode('utf-8'))
|
b"foo:\n bar: \xc3\xbc\xc3\xbc\n".decode('utf-8'))
|
||||||
|
|
||||||
@@ -591,6 +594,7 @@ class TestCtl(unittest.TestCase):
|
|||||||
self.runner.invoke(ctl, ['show-config', 'dummy'])
|
self.runner.invoke(ctl, ['show-config', 'dummy'])
|
||||||
|
|
||||||
@patch('patroni.ctl.get_dcs')
|
@patch('patroni.ctl.get_dcs')
|
||||||
|
@patch('subprocess.call', Mock(return_value=0))
|
||||||
def test_edit_config(self, mock_get_dcs):
|
def test_edit_config(self, mock_get_dcs):
|
||||||
mock_get_dcs.return_value = self.e
|
mock_get_dcs.return_value = self.e
|
||||||
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
mock_get_dcs.return_value.get_cluster = get_cluster_initialized_with_leader
|
||||||
@@ -646,3 +650,24 @@ class TestCtl(unittest.TestCase):
|
|||||||
result = self.runner.invoke(ctl, ['reinit', 'alpha', 'other', '--wait'], input='y\ny')
|
result = self.runner.invoke(ctl, ['reinit', 'alpha', 'other', '--wait'], input='y\ny')
|
||||||
self.assertIn("Waiting for reinitialize to complete on: other", result.output)
|
self.assertIn("Waiting for reinitialize to complete on: other", result.output)
|
||||||
self.assertIn("Reinitialize is completed on: other", result.output)
|
self.assertIn("Reinitialize is completed on: other", result.output)
|
||||||
|
|
||||||
|
|
||||||
|
class TestPatronictlPrettyTable(unittest.TestCase):
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self.pt = PatronictlPrettyTable(' header', ['foo', 'bar'], hrules=ALL)
|
||||||
|
|
||||||
|
def test__get_hline(self):
|
||||||
|
expected = '+-----+-----+'
|
||||||
|
self.pt._hrule = expected
|
||||||
|
self.assertEqual(self.pt._hrule, '+ header----+')
|
||||||
|
self.assertFalse(self.pt._is_first_hline())
|
||||||
|
self.assertEqual(self.pt._hrule, expected)
|
||||||
|
|
||||||
|
@patch.object(PrettyTable, '_stringify_hrule', Mock(return_value='+-----+-----+'))
|
||||||
|
def test__stringify_hrule(self):
|
||||||
|
self.assertEqual(self.pt._stringify_hrule((), 'top_'), '+ header----+')
|
||||||
|
self.assertFalse(self.pt._is_first_hline())
|
||||||
|
|
||||||
|
def test_output(self):
|
||||||
|
self.assertEqual(str(self.pt), '+ header----+\n| foo | bar |\n+-----+-----+')
|
||||||
|
|||||||
+12
-1
@@ -67,6 +67,7 @@ def etcd_read(self, key, **kwargs):
|
|||||||
"expiration": "2015-05-15T09:11:09.611860899Z", "ttl": 30,
|
"expiration": "2015-05-15T09:11:09.611860899Z", "ttl": 30,
|
||||||
"modifiedIndex": 20730, "createdIndex": 20730}],
|
"modifiedIndex": 20730, "createdIndex": 20730}],
|
||||||
"modifiedIndex": 1581, "createdIndex": 1581},
|
"modifiedIndex": 1581, "createdIndex": 1581},
|
||||||
|
{"key": "/service/batman5/failsafe", "value": '{', "modifiedIndex": 1582, "createdIndex": 1582},
|
||||||
{"key": "/service/batman5/status", "value": '{"optime":2164261704,"slots":{"ls":12345}}',
|
{"key": "/service/batman5/status", "value": '{"optime":2164261704,"slots":{"ls":12345}}',
|
||||||
"modifiedIndex": 1582, "createdIndex": 1582}], "modifiedIndex": 1581, "createdIndex": 1581}}
|
"modifiedIndex": 1582, "createdIndex": 1582}], "modifiedIndex": 1581, "createdIndex": 1581}}
|
||||||
if key == '/service/legacy/':
|
if key == '/service/legacy/':
|
||||||
@@ -276,6 +277,9 @@ class TestEtcd(unittest.TestCase):
|
|||||||
self.assertFalse(self.etcd.attempt_to_acquire_leader())
|
self.assertFalse(self.etcd.attempt_to_acquire_leader())
|
||||||
self.etcd._base_path = '/service/failed'
|
self.etcd._base_path = '/service/failed'
|
||||||
self.assertFalse(self.etcd.attempt_to_acquire_leader())
|
self.assertFalse(self.etcd.attempt_to_acquire_leader())
|
||||||
|
with patch.object(EtcdClient, 'write', Mock(side_effect=[etcd.EtcdConnectionFailed, Exception])):
|
||||||
|
self.assertRaises(EtcdError, self.etcd.attempt_to_acquire_leader)
|
||||||
|
self.assertRaises(EtcdError, self.etcd.attempt_to_acquire_leader)
|
||||||
|
|
||||||
@patch.object(Cluster, 'min_version', PropertyMock(return_value=(2, 0)))
|
@patch.object(Cluster, 'min_version', PropertyMock(return_value=(2, 0)))
|
||||||
def test_write_leader_optime(self):
|
def test_write_leader_optime(self):
|
||||||
@@ -283,7 +287,14 @@ class TestEtcd(unittest.TestCase):
|
|||||||
self.etcd.write_leader_optime('0')
|
self.etcd.write_leader_optime('0')
|
||||||
|
|
||||||
def test_update_leader(self):
|
def test_update_leader(self):
|
||||||
self.assertTrue(self.etcd.update_leader(None))
|
self.assertTrue(self.etcd.update_leader(None, failsafe={'foo': 'bar'}))
|
||||||
|
with patch.object(etcd.Client, 'write',
|
||||||
|
Mock(side_effect=[etcd.EtcdConnectionFailed, etcd.EtcdClusterIdChanged, Exception])):
|
||||||
|
self.assertRaises(EtcdError, self.etcd.update_leader, None)
|
||||||
|
self.assertFalse(self.etcd.update_leader(None))
|
||||||
|
self.assertRaises(EtcdError, self.etcd.update_leader, None)
|
||||||
|
with patch.object(etcd.Client, 'write', Mock(side_effect=etcd.EtcdKeyNotFound)):
|
||||||
|
self.assertFalse(self.etcd.update_leader(None))
|
||||||
|
|
||||||
def test_initialize(self):
|
def test_initialize(self):
|
||||||
self.assertFalse(self.etcd.initialize())
|
self.assertFalse(self.etcd.initialize())
|
||||||
|
|||||||
+19
-3
@@ -36,7 +36,8 @@ def mock_urlopen(self, method, url, **kwargs):
|
|||||||
"value": base64_encode('{}'), "lease": "123", "mod_revision": '1'},
|
"value": base64_encode('{}'), "lease": "123", "mod_revision": '1'},
|
||||||
{"key": base64_encode('/patroni/test/members/bar'),
|
{"key": base64_encode('/patroni/test/members/bar'),
|
||||||
"value": base64_encode('{"version":"1.6.5"}'), "lease": "123", "mod_revision": '1'},
|
"value": base64_encode('{"version":"1.6.5"}'), "lease": "123", "mod_revision": '1'},
|
||||||
{"key": base64_encode('/patroni/test/failover'), "value": base64_encode('{}'), "mod_revision": '1'}
|
{"key": base64_encode('/patroni/test/failover'), "value": base64_encode('{}'), "mod_revision": '1'},
|
||||||
|
{"key": base64_encode('/patroni/test/failsafe'), "value": base64_encode('{'), "mod_revision": '1'}
|
||||||
]
|
]
|
||||||
})
|
})
|
||||||
elif url.endswith('/watch'):
|
elif url.endswith('/watch'):
|
||||||
@@ -215,11 +216,26 @@ class TestEtcd3(BaseTestEtcd3):
|
|||||||
|
|
||||||
def test__update_leader(self):
|
def test__update_leader(self):
|
||||||
self.etcd3._lease = None
|
self.etcd3._lease = None
|
||||||
self.etcd3.update_leader('123')
|
self.etcd3.update_leader('123', failsafe={'foo': 'bar'})
|
||||||
|
self.etcd3._last_lease_refresh = 0
|
||||||
self.etcd3.update_leader('124')
|
self.etcd3.update_leader('124')
|
||||||
|
with patch.object(PatroniEtcd3Client, 'lease_keepalive', Mock(return_value=True)),\
|
||||||
|
patch('time.time', Mock(side_effect=[0, 100, 200, 300])):
|
||||||
|
self.assertRaises(Etcd3Error, self.etcd3.update_leader, '126')
|
||||||
|
self.etcd3._last_lease_refresh = 0
|
||||||
|
with patch.object(PatroniEtcd3Client, 'lease_keepalive', Mock(side_effect=Unknown)):
|
||||||
|
self.assertFalse(self.etcd3.update_leader('125'))
|
||||||
|
|
||||||
|
def test_take_leader(self):
|
||||||
|
self.assertFalse(self.etcd3.take_leader())
|
||||||
|
|
||||||
def test_attempt_to_acquire_leader(self):
|
def test_attempt_to_acquire_leader(self):
|
||||||
self.etcd3._lease = None
|
self.assertFalse(self.etcd3.attempt_to_acquire_leader())
|
||||||
|
with patch('time.time', Mock(side_effect=[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 100, 200])):
|
||||||
|
self.assertRaises(Etcd3Error, self.etcd3.attempt_to_acquire_leader)
|
||||||
|
with patch('time.time', Mock(side_effect=[0, 100, 200, 300, 400])):
|
||||||
|
self.assertRaises(Etcd3Error, self.etcd3.attempt_to_acquire_leader)
|
||||||
|
with patch.object(PatroniEtcd3Client, 'put', Mock(return_value=False)):
|
||||||
self.assertFalse(self.etcd3.attempt_to_acquire_leader())
|
self.assertFalse(self.etcd3.attempt_to_acquire_leader())
|
||||||
|
|
||||||
def test_set_ttl(self):
|
def test_set_ttl(self):
|
||||||
|
|||||||
@@ -30,5 +30,6 @@ class TestExhibitor(unittest.TestCase):
|
|||||||
'name': 'foo', 'ttl': 30, 'retry_timeout': 10})
|
'name': 'foo', 'ttl': 30, 'retry_timeout': 10})
|
||||||
|
|
||||||
@patch.object(ExhibitorEnsembleProvider, 'poll', Mock(return_value=True))
|
@patch.object(ExhibitorEnsembleProvider, 'poll', Mock(return_value=True))
|
||||||
|
@patch.object(MockKazooClient, 'get_children', Mock(side_effect=Exception))
|
||||||
def test_get_cluster(self):
|
def test_get_cluster(self):
|
||||||
self.assertRaises(ZooKeeperError, self.e.get_cluster)
|
self.assertRaises(ZooKeeperError, self.e.get_cluster)
|
||||||
|
|||||||
+79
-11
@@ -38,13 +38,17 @@ def get_cluster(initialize, leader, members, failover, sync, cluster_config=None
|
|||||||
history = TimelineHistory(1, '[[1,67197376,"no recovery target specified","' + t + '","foo"]]',
|
history = TimelineHistory(1, '[[1,67197376,"no recovery target specified","' + t + '","foo"]]',
|
||||||
[(1, 67197376, 'no recovery target specified', t, 'foo')])
|
[(1, 67197376, 'no recovery target specified', t, 'foo')])
|
||||||
cluster_config = cluster_config or ClusterConfig(1, {'check_timeline': True}, 1)
|
cluster_config = cluster_config or ClusterConfig(1, {'check_timeline': True}, 1)
|
||||||
return Cluster(initialize, cluster_config, leader, 10, members, failover, sync, history, None)
|
return Cluster(initialize, cluster_config, leader, 10, members, failover, sync, history, None, None)
|
||||||
|
|
||||||
|
|
||||||
def get_cluster_not_initialized_without_leader(cluster_config=None):
|
def get_cluster_not_initialized_without_leader(cluster_config=None):
|
||||||
return get_cluster(None, None, [], None, SyncState(None, None, None), cluster_config)
|
return get_cluster(None, None, [], None, SyncState(None, None, None), cluster_config)
|
||||||
|
|
||||||
|
|
||||||
|
def get_cluster_bootstrapping_without_leader(cluster_config=None):
|
||||||
|
return get_cluster("", None, [], None, SyncState(None, None, None), cluster_config)
|
||||||
|
|
||||||
|
|
||||||
def get_cluster_initialized_without_leader(leader=False, failover=None, sync=None, cluster_config=None):
|
def get_cluster_initialized_without_leader(leader=False, failover=None, sync=None, cluster_config=None):
|
||||||
m1 = Member(0, 'leader', 28, {'conn_url': 'postgres://replicator:[email protected]:5435/postgres',
|
m1 = Member(0, 'leader', 28, {'conn_url': 'postgres://replicator:[email protected]:5435/postgres',
|
||||||
'api_url': 'http://127.0.0.1:8008/patroni', 'xlog_location': 4})
|
'api_url': 'http://127.0.0.1:8008/patroni', 'xlog_location': 4})
|
||||||
@@ -202,7 +206,8 @@ class TestHa(PostgresInit):
|
|||||||
|
|
||||||
def test_update_lock(self):
|
def test_update_lock(self):
|
||||||
self.p.last_operation = Mock(side_effect=PostgresConnectionException(''))
|
self.p.last_operation = Mock(side_effect=PostgresConnectionException(''))
|
||||||
self.ha.dcs.update_leader = Mock(side_effect=Exception)
|
self.ha.dcs.update_leader = Mock(side_effect=[DCSError(''), Exception])
|
||||||
|
self.assertRaises(DCSError, self.ha.update_lock)
|
||||||
self.assertFalse(self.ha.update_lock(True))
|
self.assertFalse(self.ha.update_lock(True))
|
||||||
|
|
||||||
@patch.object(Postgresql, 'received_timeline', Mock(return_value=None))
|
@patch.object(Postgresql, 'received_timeline', Mock(return_value=None))
|
||||||
@@ -405,6 +410,7 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
|
|
||||||
|
@patch.object(Postgresql, '_wait_for_connection_close', Mock())
|
||||||
def test_demote_because_not_having_lock(self):
|
def test_demote_because_not_having_lock(self):
|
||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
with patch.object(Watchdog, 'is_running', PropertyMock(return_value=True)):
|
with patch.object(Watchdog, 'is_running', PropertyMock(return_value=True)):
|
||||||
@@ -453,7 +459,9 @@ class TestHa(PostgresInit):
|
|||||||
|
|
||||||
def test_no_etcd_connection_master_demote(self):
|
def test_no_etcd_connection_master_demote(self):
|
||||||
self.ha.load_cluster_from_dcs = Mock(side_effect=DCSError('Etcd is not responding properly'))
|
self.ha.load_cluster_from_dcs = Mock(side_effect=DCSError('Etcd is not responding properly'))
|
||||||
self.assertEqual(self.ha.run_cycle(), 'demoted self because DCS is not accessible and i was a leader')
|
self.assertEqual(self.ha.run_cycle(), 'demoting self because DCS is not accessible and I was a leader')
|
||||||
|
self.ha._async_executor.schedule('dummy')
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'demoted self because DCS is not accessible and I was a leader')
|
||||||
|
|
||||||
@patch('time.sleep', Mock())
|
@patch('time.sleep', Mock())
|
||||||
def test_bootstrap_from_another_member(self):
|
def test_bootstrap_from_another_member(self):
|
||||||
@@ -469,6 +477,11 @@ class TestHa(PostgresInit):
|
|||||||
self.p.can_create_replica_without_replication_connection = MagicMock(return_value=True)
|
self.p.can_create_replica_without_replication_connection = MagicMock(return_value=True)
|
||||||
self.assertEqual(self.ha.bootstrap(), 'trying to bootstrap (without leader)')
|
self.assertEqual(self.ha.bootstrap(), 'trying to bootstrap (without leader)')
|
||||||
|
|
||||||
|
def test_bootstrap_not_running_concurrently(self):
|
||||||
|
self.ha.cluster = get_cluster_bootstrapping_without_leader()
|
||||||
|
self.p.can_create_replica_without_replication_connection = MagicMock(return_value=True)
|
||||||
|
self.assertEqual(self.ha.bootstrap(), 'waiting for leader to bootstrap')
|
||||||
|
|
||||||
def test_bootstrap_initialize_lock_failed(self):
|
def test_bootstrap_initialize_lock_failed(self):
|
||||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||||
self.assertEqual(self.ha.bootstrap(), 'failed to acquire initialize lock')
|
self.assertEqual(self.ha.bootstrap(), 'failed to acquire initialize lock')
|
||||||
@@ -643,12 +656,60 @@ class TestHa(PostgresInit):
|
|||||||
# same as previous, but set the current member to nofailover. In no case it should be elected as a leader
|
# same as previous, but set the current member to nofailover. In no case it should be elected as a leader
|
||||||
self.ha.patroni.nofailover = True
|
self.ha.patroni.nofailover = True
|
||||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
||||||
# in sync mode only the sync node is allowed to take over
|
|
||||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'other', None))
|
def test_manual_failover_process_no_leader_in_synchronous_mode(self):
|
||||||
self.ha.patroni.nofailover = False
|
|
||||||
self.ha.is_synchronous_mode = true
|
self.ha.is_synchronous_mode = true
|
||||||
|
self.p.is_leader = false
|
||||||
|
|
||||||
|
# switchover to a specific node, which name doesn't match our name (postgresql0)
|
||||||
|
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'other', None))
|
||||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||||
|
|
||||||
|
# switchover to our node (postgresql0), which name is not in sync nodes list
|
||||||
|
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'postgresql0', None),
|
||||||
|
sync=('leader1', 'blabla'))
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||||
|
|
||||||
|
# switchover from a specific leader, but our name (postgresql0) is not in the sync nodes list
|
||||||
|
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', '', None),
|
||||||
|
sync=('leader', 'blabla'))
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||||
|
|
||||||
|
# switchover from a specific leader, but the only sync node (us, postgresql0) has nofailover tag
|
||||||
|
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', '', None),
|
||||||
|
sync=('postgresql0'))
|
||||||
|
self.ha.patroni.nofailover = True
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
||||||
|
self.ha.patroni.nofailover = False
|
||||||
|
|
||||||
|
# manual failover when our name (postgresql0) isn't in the /sync key and the `other` node is not available
|
||||||
|
self.ha.fetch_node_status = get_node_status(nofailover=True) # accessible, in_recovery
|
||||||
|
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
|
||||||
|
sync=('leader1', 'blabla'))
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||||
|
|
||||||
|
# manual failover when the `other` node isn't available but our name is in the /sync key
|
||||||
|
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None),
|
||||||
|
sync=('leader1', 'postgresql0'))
|
||||||
|
self.p.pick_synchronous_standby = Mock(return_value=([], []))
|
||||||
|
self.ha.dcs.write_sync_state = true
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||||
|
|
||||||
|
# manual failover to our node (postgresql0),
|
||||||
|
# which name is not in sync nodes list (the leader and all sync nodes are not available)
|
||||||
|
self.p.set_role('replica')
|
||||||
|
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None),
|
||||||
|
sync=('leader1', 'other'))
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||||
|
|
||||||
|
# manual failover to our node (postgresql0),
|
||||||
|
# which name is not in sync nodes list (some sync nodes are available)
|
||||||
|
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'postgresql0', None),
|
||||||
|
sync=('leader1', 'other'))
|
||||||
|
self.p.set_role('replica')
|
||||||
|
self.p.pick_synchronous_standby = Mock(return_value=(['leader1'], ['leader1']))
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock')
|
||||||
|
|
||||||
def test_manual_failover_process_no_leader_in_pause(self):
|
def test_manual_failover_process_no_leader_in_pause(self):
|
||||||
self.ha.is_paused = true
|
self.ha.is_paused = true
|
||||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None))
|
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None))
|
||||||
@@ -1133,17 +1194,19 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.shutdown()
|
self.ha.shutdown()
|
||||||
|
|
||||||
@patch('time.sleep', Mock())
|
@patch('time.sleep', Mock())
|
||||||
def test_leader_with_empty_directory(self):
|
def test_leader_with_not_accessible_data_directory(self):
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.p.data_directory_empty = true
|
self.p.data_directory_empty = Mock(side_effect=OSError(5, "Input/output error: '{}'".format(self.p.data_dir)))
|
||||||
self.assertEqual(self.ha.run_cycle(), 'released leader key voluntarily as data dir empty and currently leader')
|
self.assertEqual(self.ha.run_cycle(),
|
||||||
|
'released leader key voluntarily as data dir not accessible and currently leader')
|
||||||
self.assertEqual(self.p.role, 'uninitialized')
|
self.assertEqual(self.p.role, 'uninitialized')
|
||||||
|
|
||||||
# as has_lock is mocked out, we need to fake the leader key release
|
# as has_lock is mocked out, we need to fake the leader key release
|
||||||
self.ha.has_lock = false
|
self.ha.has_lock = false
|
||||||
# will not say bootstrap from leader as replica can't self elect
|
# will not say bootstrap because data directory is not accessible
|
||||||
self.assertEqual(self.ha.run_cycle(), "trying to bootstrap from replica 'other'")
|
self.assertEqual(self.ha.run_cycle(),
|
||||||
|
"data directory is not accessible: [Errno 5] Input/output error: '{}'".format(self.p.data_dir))
|
||||||
|
|
||||||
@patch('patroni.postgresql.mtime', Mock(return_value=1588316884))
|
@patch('patroni.postgresql.mtime', Mock(return_value=1588316884))
|
||||||
@patch.object(builtins, 'open', mock_open(read_data=('1\t0/40159C0\tno recovery target specified\n\n'
|
@patch.object(builtins, 'open', mock_open(read_data=('1\t0/40159C0\tno recovery target specified\n\n'
|
||||||
@@ -1225,3 +1288,8 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.fetch_node_status = Mock(return_value=_MemberStatus(self.ha.cluster.members[0],
|
self.ha.fetch_node_status = Mock(return_value=_MemberStatus(self.ha.cluster.members[0],
|
||||||
True, True, 0, 2, None, {}, False))
|
True, True, 0, 2, None, {}, False))
|
||||||
self.assertFalse(self.ha.is_failover_possible(self.ha.cluster.members))
|
self.assertFalse(self.ha.is_failover_possible(self.ha.cluster.members))
|
||||||
|
|
||||||
|
def test_acquire_lock(self):
|
||||||
|
self.ha.dcs.attempt_to_acquire_leader = Mock(side_effect=[DCSError('foo'), Exception])
|
||||||
|
self.assertRaises(DCSError, self.ha.acquire_lock)
|
||||||
|
self.assertFalse(self.ha.acquire_lock())
|
||||||
|
|||||||
@@ -1,5 +1,7 @@
|
|||||||
|
import base64
|
||||||
import datetime
|
import datetime
|
||||||
import json
|
import json
|
||||||
|
import mock
|
||||||
import socket
|
import socket
|
||||||
import time
|
import time
|
||||||
import unittest
|
import unittest
|
||||||
@@ -18,7 +20,7 @@ def mock_list_namespaced_config_map(*args, **kwargs):
|
|||||||
'annotations': {'initialize': '123', 'config': '{}'}}
|
'annotations': {'initialize': '123', 'config': '{}'}}
|
||||||
items = [k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata))]
|
items = [k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata))]
|
||||||
metadata.update({'name': 'test-leader',
|
metadata.update({'name': 'test-leader',
|
||||||
'annotations': {'optime': '1234x', 'leader': 'p-0', 'ttl': '30s', 'slots': '{'}})
|
'annotations': {'optime': '1234x', 'leader': 'p-0', 'ttl': '30s', 'slots': '{', 'failsafe': '{'}})
|
||||||
items.append(k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata)))
|
items.append(k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata)))
|
||||||
metadata.update({'name': 'test-failover', 'annotations': {'leader': 'p-0'}})
|
metadata.update({'name': 'test-failover', 'annotations': {'leader': 'p-0'}})
|
||||||
items.append(k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata)))
|
items.append(k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata)))
|
||||||
@@ -121,6 +123,20 @@ class TestK8sConfig(unittest.TestCase):
|
|||||||
k8s_config.load_kube_config()
|
k8s_config.load_kube_config()
|
||||||
self.assertEqual(k8s_config.headers.get('authorization'), 'Bearer token')
|
self.assertEqual(k8s_config.headers.get('authorization'), 'Bearer token')
|
||||||
|
|
||||||
|
config["users"][0]["user"]["client-key-data"] = base64.b64encode(b'foobar').decode('utf-8')
|
||||||
|
config["clusters"][0]["cluster"]["certificate-authority-data"] = base64.b64encode(b'foobar').decode('utf-8')
|
||||||
|
with patch.object(builtins, 'open', mock_open(read_data=json.dumps(config))),\
|
||||||
|
patch('os.write', Mock()), patch('os.close', Mock()),\
|
||||||
|
patch('os.remove') as mock_remove,\
|
||||||
|
patch('atexit.register') as mock_atexit,\
|
||||||
|
patch('tempfile.mkstemp') as mock_mkstemp:
|
||||||
|
mock_mkstemp.side_effect = [(3, '1.tmp'), (4, '2.tmp')]
|
||||||
|
k8s_config.load_kube_config()
|
||||||
|
mock_atexit.assert_called_once()
|
||||||
|
mock_remove.side_effect = OSError
|
||||||
|
mock_atexit.call_args[0][0]() # call _cleanup_temp_files
|
||||||
|
mock_remove.assert_has_calls([mock.call('1.tmp'), mock.call('2.tmp')])
|
||||||
|
|
||||||
|
|
||||||
@patch('urllib3.PoolManager.request')
|
@patch('urllib3.PoolManager.request')
|
||||||
class TestApiClient(unittest.TestCase):
|
class TestApiClient(unittest.TestCase):
|
||||||
@@ -223,6 +239,13 @@ class TestKubernetesConfigMaps(BaseTestKubernetes):
|
|||||||
with patch.object(Kubernetes, '_wait_caches', Mock(side_effect=Exception)):
|
with patch.object(Kubernetes, '_wait_caches', Mock(side_effect=Exception)):
|
||||||
self.assertRaises(KubernetesError, self.k.get_cluster)
|
self.assertRaises(KubernetesError, self.k.get_cluster)
|
||||||
|
|
||||||
|
def test_attempt_to_acquire_leader(self):
|
||||||
|
with patch.object(k8s_client.CoreV1Api, 'patch_namespaced_config_map', create=True) as mock_patch:
|
||||||
|
mock_patch.side_effect = K8sException
|
||||||
|
self.assertRaises(KubernetesError, self.k.attempt_to_acquire_leader)
|
||||||
|
mock_patch.side_effect = k8s_client.rest.ApiException(409, '')
|
||||||
|
self.assertFalse(self.k.attempt_to_acquire_leader())
|
||||||
|
|
||||||
def test_take_leader(self):
|
def test_take_leader(self):
|
||||||
self.k.take_leader()
|
self.k.take_leader()
|
||||||
self.k._leader_observed_record['leader'] = 'test'
|
self.k._leader_observed_record['leader'] = 'test'
|
||||||
@@ -278,7 +301,7 @@ class TestKubernetesEndpoints(BaseTestKubernetes):
|
|||||||
|
|
||||||
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_endpoints', create=True)
|
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_endpoints', create=True)
|
||||||
def test_update_leader(self, mock_patch_namespaced_endpoints):
|
def test_update_leader(self, mock_patch_namespaced_endpoints):
|
||||||
self.assertIsNotNone(self.k.update_leader('123'))
|
self.assertIsNotNone(self.k.update_leader('123', failsafe={'foo': 'bar'}))
|
||||||
args = mock_patch_namespaced_endpoints.call_args[0]
|
args = mock_patch_namespaced_endpoints.call_args[0]
|
||||||
self.assertEqual(args[2].subsets[0].addresses[0].target_ref.resource_version, '10')
|
self.assertEqual(args[2].subsets[0].addresses[0].target_ref.resource_version, '10')
|
||||||
self.k._kinds._object_cache['test'].subsets[:] = []
|
self.k._kinds._object_cache['test'].subsets[:] = []
|
||||||
@@ -293,7 +316,7 @@ class TestKubernetesEndpoints(BaseTestKubernetes):
|
|||||||
mock_patch.side_effect = k8s_client.rest.ApiException(502, '')
|
mock_patch.side_effect = k8s_client.rest.ApiException(502, '')
|
||||||
self.assertFalse(self.k.update_leader('123'))
|
self.assertFalse(self.k.update_leader('123'))
|
||||||
mock_patch.side_effect = RetryFailedError('')
|
mock_patch.side_effect = RetryFailedError('')
|
||||||
self.assertFalse(self.k.update_leader('123'))
|
self.assertRaises(KubernetesError, self.k.update_leader, '123')
|
||||||
mock_patch.side_effect = k8s_client.rest.ApiException(409, '')
|
mock_patch.side_effect = k8s_client.rest.ApiException(409, '')
|
||||||
with patch('time.time', Mock(side_effect=[0, 100, 200, 0, 0, 0, 0, 100, 200])):
|
with patch('time.time', Mock(side_effect=[0, 100, 200, 0, 0, 0, 0, 100, 200])):
|
||||||
self.assertFalse(self.k.update_leader('123'))
|
self.assertFalse(self.k.update_leader('123'))
|
||||||
@@ -303,6 +326,8 @@ class TestKubernetesEndpoints(BaseTestKubernetes):
|
|||||||
mock_read.return_value.metadata.resource_version = '2'
|
mock_read.return_value.metadata.resource_version = '2'
|
||||||
self.assertIsNotNone(self.k._update_leader_with_retry({}, '1', []))
|
self.assertIsNotNone(self.k._update_leader_with_retry({}, '1', []))
|
||||||
mock_patch.side_effect = k8s_client.rest.ApiException(409, '')
|
mock_patch.side_effect = k8s_client.rest.ApiException(409, '')
|
||||||
|
mock_read.side_effect = RetryFailedError('')
|
||||||
|
self.assertRaises(KubernetesError, self.k.update_leader, '123')
|
||||||
mock_read.side_effect = Exception
|
mock_read.side_effect = Exception
|
||||||
self.assertFalse(self.k.update_leader('123'))
|
self.assertFalse(self.k.update_leader('123'))
|
||||||
|
|
||||||
@@ -315,11 +340,32 @@ class TestKubernetesEndpoints(BaseTestKubernetes):
|
|||||||
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_pod', mock_namespaced_kind, create=True)
|
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_pod', mock_namespaced_kind, create=True)
|
||||||
@patch.object(k8s_client.CoreV1Api, 'create_namespaced_endpoints', mock_namespaced_kind, create=True)
|
@patch.object(k8s_client.CoreV1Api, 'create_namespaced_endpoints', mock_namespaced_kind, create=True)
|
||||||
@patch.object(k8s_client.CoreV1Api, 'create_namespaced_service',
|
@patch.object(k8s_client.CoreV1Api, 'create_namespaced_service',
|
||||||
Mock(side_effect=[True, False, k8s_client.rest.ApiException(500, '')]), create=True)
|
Mock(side_effect=[True,
|
||||||
def test__create_config_service(self):
|
False,
|
||||||
|
k8s_client.rest.ApiException(409, ''),
|
||||||
|
k8s_client.rest.ApiException(403, ''),
|
||||||
|
k8s_client.rest.ApiException(500, ''),
|
||||||
|
Exception("Unexpected")
|
||||||
|
]), create=True)
|
||||||
|
@patch('patroni.dcs.kubernetes.logger.exception')
|
||||||
|
def test__create_config_service(self, mock_logger_exception):
|
||||||
self.assertIsNotNone(self.k.patch_or_create_config({'foo': 'bar'}))
|
self.assertIsNotNone(self.k.patch_or_create_config({'foo': 'bar'}))
|
||||||
self.assertIsNotNone(self.k.patch_or_create_config({'foo': 'bar'}))
|
self.assertIsNotNone(self.k.patch_or_create_config({'foo': 'bar'}))
|
||||||
|
|
||||||
|
self.k.patch_or_create_config({'foo': 'bar'})
|
||||||
|
mock_logger_exception.assert_not_called()
|
||||||
|
|
||||||
|
self.k.patch_or_create_config({'foo': 'bar'})
|
||||||
|
mock_logger_exception.assert_not_called()
|
||||||
|
|
||||||
|
self.k.patch_or_create_config({'foo': 'bar'})
|
||||||
|
mock_logger_exception.assert_called_once()
|
||||||
|
self.assertEqual(('create_config_service failed',), mock_logger_exception.call_args[0])
|
||||||
|
mock_logger_exception.reset_mock()
|
||||||
|
|
||||||
self.k.touch_member({'state': 'running', 'role': 'replica'})
|
self.k.touch_member({'state': 'running', 'role': 'replica'})
|
||||||
|
mock_logger_exception.assert_called_once()
|
||||||
|
self.assertEqual(('create_config_service failed',), mock_logger_exception.call_args[0])
|
||||||
|
|
||||||
|
|
||||||
class TestCacheBuilder(BaseTestKubernetes):
|
class TestCacheBuilder(BaseTestKubernetes):
|
||||||
|
|||||||
@@ -50,12 +50,16 @@ class MockFrozenImporter(object):
|
|||||||
@patch.object(etcd.Client, 'read', etcd_read)
|
@patch.object(etcd.Client, 'read', etcd_read)
|
||||||
class TestPatroni(unittest.TestCase):
|
class TestPatroni(unittest.TestCase):
|
||||||
|
|
||||||
|
@patch('sys.argv', ['patroni.py'])
|
||||||
def test_no_config(self):
|
def test_no_config(self):
|
||||||
self.assertRaises(SystemExit, patroni_main)
|
self.assertRaises(SystemExit, patroni_main)
|
||||||
|
|
||||||
@patch('sys.argv', ['patroni.py', '--validate-config', 'postgres0.yml'])
|
@patch('sys.argv', ['patroni.py', '--validate-config', 'postgres0.yml'])
|
||||||
|
@patch('socket.socket.connect_ex', Mock(return_value=1))
|
||||||
def test_validate_config(self):
|
def test_validate_config(self):
|
||||||
self.assertRaises(SystemExit, patroni_main)
|
self.assertRaises(SystemExit, patroni_main)
|
||||||
|
with patch.object(config.Config, '__init__', Mock(return_value=None)):
|
||||||
|
self.assertRaises(SystemExit, patroni_main)
|
||||||
|
|
||||||
@patch('pkgutil.iter_importers', Mock(return_value=[MockFrozenImporter()]))
|
@patch('pkgutil.iter_importers', Mock(return_value=[MockFrozenImporter()]))
|
||||||
@patch('sys.frozen', Mock(return_value=True), create=True)
|
@patch('sys.frozen', Mock(return_value=True), create=True)
|
||||||
|
|||||||
+15
-19
@@ -347,7 +347,7 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
@patch('subprocess.Popen')
|
@patch('subprocess.Popen')
|
||||||
def test_latest_checkpoint_location(self, mock_popen):
|
def test_latest_checkpoint_location(self, mock_popen):
|
||||||
mock_popen.return_value.communicate.return_value = (None, None)
|
mock_popen.return_value.communicate.return_value = (None, None)
|
||||||
self.assertEqual(self.p.latest_checkpoint_location(), '28163096')
|
self.assertEqual(self.p.latest_checkpoint_location(), 28163096)
|
||||||
# 9.3 and 9.4 format
|
# 9.3 and 9.4 format
|
||||||
mock_popen.return_value.communicate.side_effect = [
|
mock_popen.return_value.communicate.side_effect = [
|
||||||
(b'rmgr: XLOG len (rec/tot): 72/ 104, tx: 0, lsn: 0/01ADBC18, prev 0/01ADBBB8, ' +
|
(b'rmgr: XLOG len (rec/tot): 72/ 104, tx: 0, lsn: 0/01ADBC18, prev 0/01ADBBB8, ' +
|
||||||
@@ -355,14 +355,14 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
b' 1; offset 0; oldest xid 715 in DB 1; oldest multi 1 in DB 1; oldest running xid 0; shutdown', None),
|
b' 1; offset 0; oldest xid 715 in DB 1; oldest multi 1 in DB 1; oldest running xid 0; shutdown', None),
|
||||||
(b'rmgr: Transaction len (rec/tot): 64/ 96, tx: 726, lsn: 0/01ADBBB8, prev 0/01ADBB70, ' +
|
(b'rmgr: Transaction len (rec/tot): 64/ 96, tx: 726, lsn: 0/01ADBBB8, prev 0/01ADBB70, ' +
|
||||||
b'bkp: 0000, desc: commit: 2021-02-26 11:19:37.900918 CET; inval msgs: catcache 11 catcache 10', None)]
|
b'bkp: 0000, desc: commit: 2021-02-26 11:19:37.900918 CET; inval msgs: catcache 11 catcache 10', None)]
|
||||||
self.assertEqual(self.p.latest_checkpoint_location(), '28163096')
|
self.assertEqual(self.p.latest_checkpoint_location(), 28163096)
|
||||||
mock_popen.return_value.communicate.side_effect = [
|
mock_popen.return_value.communicate.side_effect = [
|
||||||
(b'rmgr: XLOG len (rec/tot): 72/ 104, tx: 0, lsn: 0/01ADBC18, prev 0/01ADBBB8, ' +
|
(b'rmgr: XLOG len (rec/tot): 72/ 104, tx: 0, lsn: 0/01ADBC18, prev 0/01ADBBB8, ' +
|
||||||
b'bkp: 0000, desc: checkpoint: redo 0/1ADBC18; tli 1; prev tli 1; fpw true; xid 0/727; oid 16386; multi' +
|
b'bkp: 0000, desc: checkpoint: redo 0/1ADBC18; tli 1; prev tli 1; fpw true; xid 0/727; oid 16386; multi' +
|
||||||
b' 1; offset 0; oldest xid 715 in DB 1; oldest multi 1 in DB 1; oldest running xid 0; shutdown', None),
|
b' 1; offset 0; oldest xid 715 in DB 1; oldest multi 1 in DB 1; oldest running xid 0; shutdown', None),
|
||||||
(b'rmgr: XLOG len (rec/tot): 0/ 32, tx: 0, lsn: 0/01ADBBB8, prev 0/01ADBBA0, ' +
|
(b'rmgr: XLOG len (rec/tot): 0/ 32, tx: 0, lsn: 0/01ADBBB8, prev 0/01ADBBA0, ' +
|
||||||
b'bkp: 0000, desc: xlog switch ', None)]
|
b'bkp: 0000, desc: xlog switch ', None)]
|
||||||
self.assertEqual(self.p.latest_checkpoint_location(), '28163000')
|
self.assertEqual(self.p.latest_checkpoint_location(), 28163000)
|
||||||
# 9.5+ format
|
# 9.5+ format
|
||||||
mock_popen.return_value.communicate.side_effect = [
|
mock_popen.return_value.communicate.side_effect = [
|
||||||
(b'rmgr: XLOG len (rec/tot): 114/ 114, tx: 0, lsn: 0/01ADBC18, prev 0/018260F8, ' +
|
(b'rmgr: XLOG len (rec/tot): 114/ 114, tx: 0, lsn: 0/01ADBC18, prev 0/018260F8, ' +
|
||||||
@@ -371,7 +371,7 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
b' oldest running xid 0; shutdown', None),
|
b' oldest running xid 0; shutdown', None),
|
||||||
(b'rmgr: XLOG len (rec/tot): 24/ 24, tx: 0, lsn: 0/018260F8, prev 0/01826080, ' +
|
(b'rmgr: XLOG len (rec/tot): 24/ 24, tx: 0, lsn: 0/018260F8, prev 0/01826080, ' +
|
||||||
b'desc: SWITCH ', None)]
|
b'desc: SWITCH ', None)]
|
||||||
self.assertEqual(self.p.latest_checkpoint_location(), '25321720')
|
self.assertEqual(self.p.latest_checkpoint_location(), 25321720)
|
||||||
|
|
||||||
def test_reload(self):
|
def test_reload(self):
|
||||||
self.assertTrue(self.p.reload())
|
self.assertTrue(self.p.reload())
|
||||||
@@ -454,23 +454,19 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
def test_get_postgres_role_from_data_directory(self):
|
def test_get_postgres_role_from_data_directory(self):
|
||||||
self.assertEqual(self.p.get_postgres_role_from_data_directory(), 'replica')
|
self.assertEqual(self.p.get_postgres_role_from_data_directory(), 'replica')
|
||||||
|
|
||||||
|
@patch('os.remove', Mock())
|
||||||
|
@patch('shutil.rmtree', Mock())
|
||||||
|
@patch('os.unlink', Mock(side_effect=OSError))
|
||||||
|
@patch('os.path.isdir', Mock(return_value=True))
|
||||||
|
@patch('os.path.exists', Mock(return_value=True))
|
||||||
def test_remove_data_directory(self):
|
def test_remove_data_directory(self):
|
||||||
def _symlink(src, dst):
|
with patch('os.path.islink', Mock(return_value=True)):
|
||||||
if os.name != 'nt': # os.symlink under Windows needs admin rights skip it
|
|
||||||
os.symlink(src, dst)
|
|
||||||
|
|
||||||
os.makedirs(os.path.join(self.p.data_dir, 'foo'))
|
|
||||||
_symlink('foo', os.path.join(self.p.data_dir, 'pg_wal'))
|
|
||||||
os.makedirs(os.path.join(self.p.data_dir, 'foo_tsp'))
|
|
||||||
pg_tblspc = os.path.join(self.p.data_dir, 'pg_tblspc')
|
|
||||||
os.makedirs(pg_tblspc)
|
|
||||||
_symlink('../foo_tsp', os.path.join(pg_tblspc, '12345'))
|
|
||||||
self.p.remove_data_directory()
|
self.p.remove_data_directory()
|
||||||
open(self.p.data_dir, 'w').close()
|
with patch('os.path.isfile', Mock(return_value=True)):
|
||||||
self.p.remove_data_directory()
|
|
||||||
_symlink('unexisting', self.p.data_dir)
|
|
||||||
with patch('os.unlink', Mock(side_effect=OSError)):
|
|
||||||
self.p.remove_data_directory()
|
self.p.remove_data_directory()
|
||||||
|
with patch('os.path.islink', Mock(side_effect=[False, False, True, True])),\
|
||||||
|
patch('os.listdir', Mock(return_value=['12345'])),\
|
||||||
|
patch('os.path.realpath', Mock(side_effect=['../foo', '../foo_tsp'])):
|
||||||
self.p.remove_data_directory()
|
self.p.remove_data_directory()
|
||||||
|
|
||||||
@patch('patroni.postgresql.Postgresql._version_file_exists', Mock(return_value=True))
|
@patch('patroni.postgresql.Postgresql._version_file_exists', Mock(return_value=True))
|
||||||
@@ -644,7 +640,7 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
|
|
||||||
def test_pick_sync_standby(self):
|
def test_pick_sync_standby(self):
|
||||||
cluster = Cluster(True, None, self.leader, 0, [self.me, self.other, self.leadermem], None,
|
cluster = Cluster(True, None, self.leader, 0, [self.me, self.other, self.leadermem], None,
|
||||||
SyncState(0, self.me.name, self.leadermem.name), None, None)
|
SyncState(0, self.me.name, self.leadermem.name), None, None, None)
|
||||||
mock_cursor = Mock()
|
mock_cursor = Mock()
|
||||||
mock_cursor.fetchone.return_value = ('remote_apply',)
|
mock_cursor.fetchone.return_value = ('remote_apply',)
|
||||||
|
|
||||||
|
|||||||
@@ -133,14 +133,20 @@ class TestPostmasterProcess(unittest.TestCase):
|
|||||||
c2.cmdline = Mock(return_value=["postgres: postgres postgres [local] idle"])
|
c2.cmdline = Mock(return_value=["postgres: postgres postgres [local] idle"])
|
||||||
c3 = Mock()
|
c3 = Mock()
|
||||||
c3.cmdline = Mock(side_effect=psutil.NoSuchProcess(123))
|
c3.cmdline = Mock(side_effect=psutil.NoSuchProcess(123))
|
||||||
|
mock_wait.return_value = ([], [c2])
|
||||||
with patch('psutil.Process.children', Mock(return_value=[c1, c2, c3])):
|
with patch('psutil.Process.children', Mock(return_value=[c1, c2, c3])):
|
||||||
proc = PostmasterProcess(123)
|
proc = PostmasterProcess(123)
|
||||||
self.assertIsNone(proc.wait_for_user_backends_to_close())
|
self.assertIsNone(proc.wait_for_user_backends_to_close(1))
|
||||||
mock_wait.assert_called_with([c2])
|
mock_wait.assert_called_with([c2], 1)
|
||||||
|
|
||||||
|
mock_wait.return_value = ([c2], [])
|
||||||
|
with patch('psutil.Process.children', Mock(return_value=[c1, c2, c3])):
|
||||||
|
proc = PostmasterProcess(123)
|
||||||
|
proc.wait_for_user_backends_to_close(1)
|
||||||
|
|
||||||
with patch('psutil.Process.children', Mock(side_effect=psutil.NoSuchProcess(123))):
|
with patch('psutil.Process.children', Mock(side_effect=psutil.NoSuchProcess(123))):
|
||||||
proc = PostmasterProcess(123)
|
proc = PostmasterProcess(123)
|
||||||
self.assertIsNone(proc.wait_for_user_backends_to_close())
|
self.assertIsNone(proc.wait_for_user_backends_to_close(None))
|
||||||
|
|
||||||
@patch('subprocess.Popen')
|
@patch('subprocess.Popen')
|
||||||
@patch('os.setsid', Mock(), create=True)
|
@patch('os.setsid', Mock(), create=True)
|
||||||
|
|||||||
+9
-3
@@ -4,7 +4,7 @@ import tempfile
|
|||||||
import time
|
import time
|
||||||
|
|
||||||
from mock import Mock, PropertyMock, patch
|
from mock import Mock, PropertyMock, patch
|
||||||
from patroni.dcs.raft import DynMemberSyncObj, KVStoreTTL, Raft, SyncObjUtility, TCPTransport, _TCPTransport
|
from patroni.dcs.raft import DynMemberSyncObj, KVStoreTTL, Raft, RaftError, SyncObjUtility, TCPTransport, _TCPTransport
|
||||||
from pysyncobj import SyncObjConf, FAIL_REASON
|
from pysyncobj import SyncObjConf, FAIL_REASON
|
||||||
|
|
||||||
|
|
||||||
@@ -79,6 +79,9 @@ class TestKVStoreTTL(unittest.TestCase):
|
|||||||
self.assertFalse(self.so.set('foo', 'bar', prevExist=False, ttl=30))
|
self.assertFalse(self.so.set('foo', 'bar', prevExist=False, ttl=30))
|
||||||
self.assertFalse(self.so.retry(self.so._set, 'foo', {'value': 'buz', 'created': 1, 'updated': 1}, prevValue=''))
|
self.assertFalse(self.so.retry(self.so._set, 'foo', {'value': 'buz', 'created': 1, 'updated': 1}, prevValue=''))
|
||||||
self.assertTrue(self.so.retry(self.so._set, 'foo', {'value': 'buz', 'created': 1, 'updated': 1}))
|
self.assertTrue(self.so.retry(self.so._set, 'foo', {'value': 'buz', 'created': 1, 'updated': 1}))
|
||||||
|
with patch.object(KVStoreTTL, 'retry', Mock(side_effect=RaftError(''))):
|
||||||
|
self.assertFalse(self.so.set('foo', 'bar'))
|
||||||
|
self.assertRaises(RaftError, self.so.set, 'foo', 'bar', handle_raft_error=False)
|
||||||
|
|
||||||
def test_delete(self):
|
def test_delete(self):
|
||||||
self.so.autoTickPeriod = 0.2
|
self.so.autoTickPeriod = 0.2
|
||||||
@@ -87,6 +90,8 @@ class TestKVStoreTTL(unittest.TestCase):
|
|||||||
self.assertFalse(self.so.delete('foo', prevValue='buz'))
|
self.assertFalse(self.so.delete('foo', prevValue='buz'))
|
||||||
self.assertTrue(self.so.delete('foo', recursive=True))
|
self.assertTrue(self.so.delete('foo', recursive=True))
|
||||||
self.assertFalse(self.so.retry(self.so._delete, 'foo', prevValue=''))
|
self.assertFalse(self.so.retry(self.so._delete, 'foo', prevValue=''))
|
||||||
|
with patch.object(KVStoreTTL, 'retry', Mock(side_effect=RaftError(''))):
|
||||||
|
self.assertFalse(self.so.delete('foo'))
|
||||||
|
|
||||||
def test_expire(self):
|
def test_expire(self):
|
||||||
self.so.set('foo', 'bar', ttl=0.001)
|
self.so.set('foo', 'bar', ttl=0.001)
|
||||||
@@ -102,7 +107,7 @@ class TestKVStoreTTL(unittest.TestCase):
|
|||||||
callback(True, return_values.pop(0))
|
callback(True, return_values.pop(0))
|
||||||
|
|
||||||
with patch('time.time', Mock(side_effect=[1, 100])):
|
with patch('time.time', Mock(side_effect=[1, 100])):
|
||||||
self.assertFalse(self.so.retry(test))
|
self.assertRaises(RaftError, self.so.retry, test)
|
||||||
|
|
||||||
self.assertTrue(self.so.retry(test))
|
self.assertTrue(self.so.retry(test))
|
||||||
self.assertFalse(self.so.retry(test))
|
self.assertFalse(self.so.retry(test))
|
||||||
@@ -135,7 +140,8 @@ class TestRaft(unittest.TestCase):
|
|||||||
raft.get_cluster()
|
raft.get_cluster()
|
||||||
self.assertTrue(raft._sync_obj.set(raft.status_path, '{"optime":1234567,"slots":{"ls":12345}}'))
|
self.assertTrue(raft._sync_obj.set(raft.status_path, '{"optime":1234567,"slots":{"ls":12345}}'))
|
||||||
raft.get_cluster()
|
raft.get_cluster()
|
||||||
self.assertTrue(raft.update_leader('1'))
|
self.assertTrue(raft.update_leader('1', failsafe={'foo': 'bat'}))
|
||||||
|
self.assertTrue(raft._sync_obj.set(raft.failsafe_path, '{"foo"}'))
|
||||||
self.assertTrue(raft._sync_obj.set(raft.status_path, '{'))
|
self.assertTrue(raft._sync_obj.set(raft.status_path, '{'))
|
||||||
raft.get_cluster()
|
raft.get_cluster()
|
||||||
self.assertTrue(raft.delete_sync_state())
|
self.assertTrue(raft.delete_sync_state())
|
||||||
|
|||||||
@@ -218,6 +218,62 @@ class TestRewind(BaseTestPostgresql):
|
|||||||
self.r.cleanup_archive_status()
|
self.r.cleanup_archive_status()
|
||||||
self.r.cleanup_archive_status()
|
self.r.cleanup_archive_status()
|
||||||
|
|
||||||
|
@patch('os.path.isfile', Mock(return_value=True))
|
||||||
|
@patch('shutil.move', Mock(side_effect=OSError))
|
||||||
|
@patch('patroni.postgresql.rewind.logger.info')
|
||||||
|
def test_archive_ready_wals(self, mock_logger_info):
|
||||||
|
with patch('os.listdir', Mock(side_effect=OSError)), \
|
||||||
|
patch.object(Postgresql, 'get_guc_value', Mock(side_effect=['on', 'command %f'])):
|
||||||
|
self.r._archive_ready_wals()
|
||||||
|
mock_logger_info.assert_not_called()
|
||||||
|
|
||||||
|
# each assert_not_called() calls get_guc_value('archive_mode') + get_guc_value('archive_command')
|
||||||
|
get_guc_value_res = [
|
||||||
|
'', 'command %f',
|
||||||
|
'on', '',
|
||||||
|
]
|
||||||
|
with patch.object(Postgresql, 'get_guc_value', Mock(side_effect=get_guc_value_res)):
|
||||||
|
for _ in range(len(get_guc_value_res)//2):
|
||||||
|
self.r._archive_ready_wals()
|
||||||
|
mock_logger_info.assert_not_called()
|
||||||
|
|
||||||
|
with patch('os.listdir', Mock(return_value=['000000000000000000000000.ready'])):
|
||||||
|
# successful archive_command call
|
||||||
|
with patch.object(CancellableSubprocess, 'call', Mock(return_value=0)):
|
||||||
|
get_guc_value_res = [
|
||||||
|
'on', 'command %f',
|
||||||
|
'always', 'command %f',
|
||||||
|
]
|
||||||
|
with patch.object(Postgresql, 'get_guc_value', Mock(side_effect=get_guc_value_res)):
|
||||||
|
for _ in range(len(get_guc_value_res)//2):
|
||||||
|
self.r._archive_ready_wals()
|
||||||
|
mock_logger_info.assert_called_once()
|
||||||
|
self.assertEqual(('Trying to archive %s: %s',
|
||||||
|
'000000000000000000000000', 'command 000000000000000000000000'),
|
||||||
|
mock_logger_info.call_args[0])
|
||||||
|
mock_logger_info.reset_mock()
|
||||||
|
|
||||||
|
# failed archive_command call
|
||||||
|
with patch.object(CancellableSubprocess, 'call', Mock(return_value=1)):
|
||||||
|
with patch.object(Postgresql, 'get_guc_value', Mock(side_effect=['on', 'command %f'])):
|
||||||
|
self.r._archive_ready_wals()
|
||||||
|
self.assertEqual(('Trying to archive %s: %s',
|
||||||
|
'000000000000000000000000', 'command 000000000000000000000000'),
|
||||||
|
mock_logger_info.call_args_list[0][0])
|
||||||
|
self.assertEqual(('Failed to archive WAL segment %s', '000000000000000000000000'),
|
||||||
|
mock_logger_info.call_args_list[1][0])
|
||||||
|
mock_logger_info.reset_mock()
|
||||||
|
|
||||||
|
wal_files_to_skip = [
|
||||||
|
'000000000000000000000000.done',
|
||||||
|
'000000000000000000000001.partial.done',
|
||||||
|
'002.ready',
|
||||||
|
'U00000000000000000000001.ready',
|
||||||
|
]
|
||||||
|
with patch('os.listdir', Mock(return_value=wal_files_to_skip)):
|
||||||
|
self.r._archive_ready_wals()
|
||||||
|
mock_logger_info.assert_not_called()
|
||||||
|
|
||||||
@patch('os.unlink', Mock())
|
@patch('os.unlink', Mock())
|
||||||
@patch('os.listdir', Mock(return_value=[]))
|
@patch('os.listdir', Mock(return_value=[]))
|
||||||
@patch('os.path.isfile', Mock(return_value=True))
|
@patch('os.path.isfile', Mock(return_value=True))
|
||||||
|
|||||||
+28
-6
@@ -4,17 +4,19 @@ import unittest
|
|||||||
|
|
||||||
|
|
||||||
from mock import Mock, PropertyMock, patch
|
from mock import Mock, PropertyMock, patch
|
||||||
|
from threading import Thread
|
||||||
|
|
||||||
from patroni import psycopg
|
from patroni import psycopg
|
||||||
from patroni.dcs import Cluster, ClusterConfig, Member
|
from patroni.dcs import Cluster, ClusterConfig, Member
|
||||||
from patroni.postgresql import Postgresql
|
from patroni.postgresql import Postgresql
|
||||||
from patroni.postgresql.slots import SlotsHandler, fsync_dir
|
from patroni.postgresql.slots import SlotsAdvanceThread, SlotsHandler, fsync_dir
|
||||||
|
|
||||||
from . import BaseTestPostgresql, psycopg_connect, MockCursor
|
from . import BaseTestPostgresql, psycopg_connect, MockCursor
|
||||||
|
|
||||||
|
|
||||||
@patch('subprocess.call', Mock(return_value=0))
|
@patch('subprocess.call', Mock(return_value=0))
|
||||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
|
@patch.object(Thread, 'start', Mock())
|
||||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||||
class TestSlotsHandler(BaseTestPostgresql):
|
class TestSlotsHandler(BaseTestPostgresql):
|
||||||
|
|
||||||
@@ -29,21 +31,23 @@ class TestSlotsHandler(BaseTestPostgresql):
|
|||||||
self.p.start()
|
self.p.start()
|
||||||
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}}}, 1)
|
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}}}, 1)
|
||||||
self.cluster = Cluster(True, config, self.leader, 0,
|
self.cluster = Cluster(True, config, self.leader, 0,
|
||||||
[self.me, self.other, self.leadermem], None, None, None, {'ls': 12345})
|
[self.me, self.other, self.leadermem], None, None, None, {'ls': 12345}, None)
|
||||||
|
|
||||||
def test_sync_replication_slots(self):
|
def test_sync_replication_slots(self):
|
||||||
config = ClusterConfig(1, {'slots': {'test_3': {'database': 'a', 'plugin': 'b'},
|
config = ClusterConfig(1, {'slots': {'test_3': {'database': 'a', 'plugin': 'b'},
|
||||||
'A': 0, 'ls': 0, 'b': {'type': 'logical', 'plugin': '1'}},
|
'A': 0, 'ls': 0, 'b': {'type': 'logical', 'plugin': '1'}},
|
||||||
'ignore_slots': [{'name': 'blabla'}]}, 1)
|
'ignore_slots': [{'name': 'blabla'}]}, 1)
|
||||||
cluster = Cluster(True, config, self.leader, 0,
|
cluster = Cluster(True, config, self.leader, 0,
|
||||||
[self.me, self.other, self.leadermem], None, None, None, {'test_3': 10})
|
[self.me, self.other, self.leadermem], None, None, None, {'test_3': 10}, None)
|
||||||
with mock.patch('patroni.postgresql.Postgresql._query', Mock(side_effect=psycopg.OperationalError)):
|
with mock.patch('patroni.postgresql.Postgresql._query', Mock(side_effect=psycopg.OperationalError)):
|
||||||
self.s.sync_replication_slots(cluster, False)
|
self.s.sync_replication_slots(cluster, False)
|
||||||
self.p.set_role('standby_leader')
|
self.p.set_role('standby_leader')
|
||||||
self.s.sync_replication_slots(cluster, False)
|
self.s.sync_replication_slots(cluster, False)
|
||||||
self.p.set_role('replica')
|
self.p.set_role('replica')
|
||||||
with patch.object(Postgresql, 'is_leader', Mock(return_value=False)):
|
with patch.object(Postgresql, 'is_leader', Mock(return_value=False)),\
|
||||||
self.s.sync_replication_slots(cluster, False)
|
patch.object(SlotsHandler, 'drop_replication_slot') as mock_drop:
|
||||||
|
self.s.sync_replication_slots(cluster, False, paused=True)
|
||||||
|
mock_drop.assert_not_called()
|
||||||
self.p.set_role('master')
|
self.p.set_role('master')
|
||||||
with mock.patch('patroni.postgresql.Postgresql.role', new_callable=PropertyMock(return_value='replica')):
|
with mock.patch('patroni.postgresql.Postgresql.role', new_callable=PropertyMock(return_value='replica')):
|
||||||
self.s.sync_replication_slots(cluster, False)
|
self.s.sync_replication_slots(cluster, False)
|
||||||
@@ -63,7 +67,8 @@ class TestSlotsHandler(BaseTestPostgresql):
|
|||||||
def test_process_permanent_slots(self):
|
def test_process_permanent_slots(self):
|
||||||
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}},
|
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}},
|
||||||
'ignore_slots': [{'name': 'blabla'}]}, 1)
|
'ignore_slots': [{'name': 'blabla'}]}, 1)
|
||||||
cluster = Cluster(True, config, self.leader, 0, [self.me, self.other, self.leadermem], None, None, None, None)
|
cluster = Cluster(True, config, self.leader, 0,
|
||||||
|
[self.me, self.other, self.leadermem], None, None, None, None, None)
|
||||||
|
|
||||||
self.s.sync_replication_slots(cluster, False)
|
self.s.sync_replication_slots(cluster, False)
|
||||||
with patch.object(Postgresql, '_query') as mock_query:
|
with patch.object(Postgresql, '_query') as mock_query:
|
||||||
@@ -89,6 +94,7 @@ class TestSlotsHandler(BaseTestPostgresql):
|
|||||||
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), [])
|
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), [])
|
||||||
self.s._schedule_load_slots = False
|
self.s._schedule_load_slots = False
|
||||||
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)),\
|
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)),\
|
||||||
|
patch.object(SlotsAdvanceThread, 'schedule', Mock(return_value=(True, ['ls']))),\
|
||||||
patch.object(psycopg.OperationalError, 'diag') as mock_diag:
|
patch.object(psycopg.OperationalError, 'diag') as mock_diag:
|
||||||
type(mock_diag).sqlstate = PropertyMock(return_value='58P01')
|
type(mock_diag).sqlstate = PropertyMock(return_value='58P01')
|
||||||
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), ['ls'])
|
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), ['ls'])
|
||||||
@@ -121,6 +127,7 @@ class TestSlotsHandler(BaseTestPostgresql):
|
|||||||
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
||||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||||
def test_on_promote(self):
|
def test_on_promote(self):
|
||||||
|
self.s.schedule_advance_slots({'foo': {'bar': 100}})
|
||||||
self.s.copy_logical_slots(self.cluster, ['ls'])
|
self.s.copy_logical_slots(self.cluster, ['ls'])
|
||||||
self.s.on_promote()
|
self.s.on_promote()
|
||||||
|
|
||||||
@@ -130,3 +137,18 @@ class TestSlotsHandler(BaseTestPostgresql):
|
|||||||
@patch('os.fsync', Mock(side_effect=OSError))
|
@patch('os.fsync', Mock(side_effect=OSError))
|
||||||
def test_fsync_dir(self):
|
def test_fsync_dir(self):
|
||||||
self.assertRaises(OSError, fsync_dir, 'foo')
|
self.assertRaises(OSError, fsync_dir, 'foo')
|
||||||
|
|
||||||
|
def test_slots_advance_thread(self):
|
||||||
|
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)),\
|
||||||
|
patch.object(psycopg.OperationalError, 'diag') as mock_diag:
|
||||||
|
type(mock_diag).sqlstate = PropertyMock(return_value='58P01')
|
||||||
|
self.s.schedule_advance_slots({'foo': {'bar': 100}})
|
||||||
|
self.s._advance.sync_slots()
|
||||||
|
|
||||||
|
with patch.object(SlotsAdvanceThread, 'sync_slots', Mock(side_effect=Exception)):
|
||||||
|
self.s._advance._condition.wait = Mock()
|
||||||
|
self.assertRaises(Exception, self.s._advance.run)
|
||||||
|
|
||||||
|
with patch.object(SlotsHandler, 'get_local_connection_cursor', Mock(side_effect=Exception)):
|
||||||
|
self.s.schedule_advance_slots({'foo': {'bar': 100}})
|
||||||
|
self.s._advance.sync_slots()
|
||||||
|
|||||||
+20
-17
@@ -63,6 +63,7 @@ config = {
|
|||||||
"postgresql": {
|
"postgresql": {
|
||||||
"listen": "127.0.0.2,::1:543",
|
"listen": "127.0.0.2,::1:543",
|
||||||
"connect_address": "127.0.0.2:543",
|
"connect_address": "127.0.0.2:543",
|
||||||
|
"proxy_address": "127.0.0.2:5433",
|
||||||
"authentication": {
|
"authentication": {
|
||||||
"replication": {"username": "user"},
|
"replication": {"username": "user"},
|
||||||
"superuser": {"username": "user"},
|
"superuser": {"username": "user"},
|
||||||
@@ -141,14 +142,14 @@ class TestValidator(unittest.TestCase):
|
|||||||
del directories[:]
|
del directories[:]
|
||||||
|
|
||||||
def test_empty_config(self, mock_out, mock_err):
|
def test_empty_config(self, mock_out, mock_err):
|
||||||
schema({})
|
errors = schema({})
|
||||||
output = mock_out.getvalue()
|
output = "\n".join(errors)
|
||||||
expected = list(sorted(['name', 'postgresql', 'restapi', 'scope'] + available_dcs))
|
expected = list(sorted(['name', 'postgresql', 'restapi', 'scope'] + available_dcs))
|
||||||
self.assertEqual(expected, parse_output(output))
|
self.assertEqual(expected, parse_output(output))
|
||||||
|
|
||||||
def test_complete_config(self, mock_out, mock_err):
|
def test_complete_config(self, mock_out, mock_err):
|
||||||
schema(config)
|
errors = schema(config)
|
||||||
output = mock_out.getvalue()
|
output = "\n".join(errors)
|
||||||
self.assertEqual(['postgresql.bin_dir', 'raft.bind_addr', 'raft.self_addr'], parse_output(output))
|
self.assertEqual(['postgresql.bin_dir', 'raft.bind_addr', 'raft.self_addr'], parse_output(output))
|
||||||
|
|
||||||
def test_bin_dir_is_file(self, mock_out, mock_err):
|
def test_bin_dir_is_file(self, mock_out, mock_err):
|
||||||
@@ -156,10 +157,11 @@ class TestValidator(unittest.TestCase):
|
|||||||
files.append(config["postgresql"]["bin_dir"])
|
files.append(config["postgresql"]["bin_dir"])
|
||||||
c = copy.deepcopy(config)
|
c = copy.deepcopy(config)
|
||||||
c["restapi"]["connect_address"] = 'False:blabla'
|
c["restapi"]["connect_address"] = 'False:blabla'
|
||||||
|
c["postgresql"]["listen"] = '*:543'
|
||||||
c["etcd"]["hosts"] = ["127.0.0.1:2379", "1244.0.0.1:2379", "127.0.0.1:invalidport"]
|
c["etcd"]["hosts"] = ["127.0.0.1:2379", "1244.0.0.1:2379", "127.0.0.1:invalidport"]
|
||||||
c["kubernetes"]["pod_ip"] = "127.0.0.1111"
|
c["kubernetes"]["pod_ip"] = "127.0.0.1111"
|
||||||
schema(c)
|
errors = schema(c)
|
||||||
output = mock_out.getvalue()
|
output = "\n".join(errors)
|
||||||
self.assertEqual(['etcd.hosts.1', 'etcd.hosts.2', 'kubernetes.pod_ip', 'postgresql.bin_dir',
|
self.assertEqual(['etcd.hosts.1', 'etcd.hosts.2', 'kubernetes.pod_ip', 'postgresql.bin_dir',
|
||||||
'postgresql.data_dir', 'raft.bind_addr', 'raft.self_addr',
|
'postgresql.data_dir', 'raft.bind_addr', 'raft.self_addr',
|
||||||
'restapi.connect_address'], parse_output(output))
|
'restapi.connect_address'], parse_output(output))
|
||||||
@@ -176,8 +178,8 @@ class TestValidator(unittest.TestCase):
|
|||||||
c["etcd"]["host"] = "127.0.0.1:237"
|
c["etcd"]["host"] = "127.0.0.1:237"
|
||||||
c["postgresql"]["listen"] = "127.0.0.1:5432"
|
c["postgresql"]["listen"] = "127.0.0.1:5432"
|
||||||
with patch('patroni.validator.open', mock_open(read_data='9')):
|
with patch('patroni.validator.open', mock_open(read_data='9')):
|
||||||
schema(c)
|
errors = schema(c)
|
||||||
output = mock_out.getvalue()
|
output = "\n".join(errors)
|
||||||
self.assertEqual(['consul.host', 'etcd.host', 'postgresql.bin_dir', 'postgresql.data_dir', 'postgresql.listen',
|
self.assertEqual(['consul.host', 'etcd.host', 'postgresql.bin_dir', 'postgresql.data_dir', 'postgresql.listen',
|
||||||
'raft.bind_addr', 'raft.self_addr', 'restapi.connect_address'], parse_output(output))
|
'raft.bind_addr', 'raft.self_addr', 'restapi.connect_address'], parse_output(output))
|
||||||
|
|
||||||
@@ -195,8 +197,8 @@ class TestValidator(unittest.TestCase):
|
|||||||
files.append(os.path.join(config["postgresql"]["bin_dir"], "postgres"))
|
files.append(os.path.join(config["postgresql"]["bin_dir"], "postgres"))
|
||||||
files.append(os.path.join(config["postgresql"]["bin_dir"], "pg_isready"))
|
files.append(os.path.join(config["postgresql"]["bin_dir"], "pg_isready"))
|
||||||
with patch('patroni.validator.open', mock_open(read_data='12')):
|
with patch('patroni.validator.open', mock_open(read_data='12')):
|
||||||
schema(config)
|
errors = schema(config)
|
||||||
output = mock_out.getvalue()
|
output = "\n".join(errors)
|
||||||
self.assertEqual(['raft.bind_addr', 'raft.self_addr'], parse_output(output))
|
self.assertEqual(['raft.bind_addr', 'raft.self_addr'], parse_output(output))
|
||||||
|
|
||||||
@patch('subprocess.check_output', Mock(return_value=b"postgres (PostgreSQL) 12.1"))
|
@patch('subprocess.check_output', Mock(return_value=b"postgres (PostgreSQL) 12.1"))
|
||||||
@@ -208,11 +210,12 @@ class TestValidator(unittest.TestCase):
|
|||||||
files.append(os.path.join(config["postgresql"]["data_dir"], "PG_VERSION"))
|
files.append(os.path.join(config["postgresql"]["data_dir"], "PG_VERSION"))
|
||||||
c = copy.deepcopy(config)
|
c = copy.deepcopy(config)
|
||||||
c["etcd"]["hosts"] = []
|
c["etcd"]["hosts"] = []
|
||||||
|
c["postgresql"]["listen"] = '127.0.0.2,*:543'
|
||||||
del c["postgresql"]["bin_dir"]
|
del c["postgresql"]["bin_dir"]
|
||||||
with patch('patroni.validator.open', mock_open(read_data='11')):
|
with patch('patroni.validator.open', mock_open(read_data='11')):
|
||||||
schema(c)
|
errors = schema(c)
|
||||||
output = mock_out.getvalue()
|
output = "\n".join(errors)
|
||||||
self.assertEqual(['etcd.hosts', 'postgresql.data_dir',
|
self.assertEqual(['etcd.hosts', 'postgresql.data_dir', 'postgresql.listen',
|
||||||
'raft.bind_addr', 'raft.self_addr'], parse_output(output))
|
'raft.bind_addr', 'raft.self_addr'], parse_output(output))
|
||||||
|
|
||||||
@patch('subprocess.check_output', Mock(return_value=b"postgres (PostgreSQL) 12.1"))
|
@patch('subprocess.check_output', Mock(return_value=b"postgres (PostgreSQL) 12.1"))
|
||||||
@@ -224,8 +227,8 @@ class TestValidator(unittest.TestCase):
|
|||||||
c = copy.deepcopy(config)
|
c = copy.deepcopy(config)
|
||||||
del c["postgresql"]["bin_dir"]
|
del c["postgresql"]["bin_dir"]
|
||||||
with patch('patroni.validator.open', mock_open(read_data='11')):
|
with patch('patroni.validator.open', mock_open(read_data='11')):
|
||||||
schema(c)
|
errors = schema(c)
|
||||||
output = mock_out.getvalue()
|
output = "\n".join(errors)
|
||||||
self.assertEqual(['postgresql.data_dir', 'raft.bind_addr', 'raft.self_addr'], parse_output(output))
|
self.assertEqual(['postgresql.data_dir', 'raft.bind_addr', 'raft.self_addr'], parse_output(output))
|
||||||
|
|
||||||
def test_data_dir_is_empty_string(self, mock_out, mock_err):
|
def test_data_dir_is_empty_string(self, mock_out, mock_err):
|
||||||
@@ -236,7 +239,7 @@ class TestValidator(unittest.TestCase):
|
|||||||
c["postgresql"]["pg_hba"] = ""
|
c["postgresql"]["pg_hba"] = ""
|
||||||
c["postgresql"]["data_dir"] = ""
|
c["postgresql"]["data_dir"] = ""
|
||||||
c["postgresql"]["bin_dir"] = ""
|
c["postgresql"]["bin_dir"] = ""
|
||||||
schema(c)
|
errors = schema(c)
|
||||||
output = mock_out.getvalue()
|
output = "\n".join(errors)
|
||||||
self.assertEqual(['kubernetes', 'postgresql.bin_dir', 'postgresql.data_dir',
|
self.assertEqual(['kubernetes', 'postgresql.bin_dir', 'postgresql.data_dir',
|
||||||
'postgresql.pg_hba', 'raft.bind_addr', 'raft.self_addr'], parse_output(output))
|
'postgresql.pg_hba', 'raft.bind_addr', 'raft.self_addr'], parse_output(output))
|
||||||
|
|||||||
@@ -35,7 +35,7 @@ def mock_ioctl(fd, op, arg=None, mutate_flag=False):
|
|||||||
sys.stderr.write("Ioctl %d %d %r\n" % (fd, op, arg))
|
sys.stderr.write("Ioctl %d %d %r\n" % (fd, op, arg))
|
||||||
if op == linuxwd.WDIOC_GETSUPPORT:
|
if op == linuxwd.WDIOC_GETSUPPORT:
|
||||||
sys.stderr.write("Get support\n")
|
sys.stderr.write("Get support\n")
|
||||||
assert(mutate_flag is True)
|
assert (mutate_flag is True)
|
||||||
arg.options = sum(map(linuxwd.WDIOF.get, ['SETTIMEOUT', 'KEEPALIVEPING']))
|
arg.options = sum(map(linuxwd.WDIOF.get, ['SETTIMEOUT', 'KEEPALIVEPING']))
|
||||||
arg.identity = (ctypes.c_ubyte*32)(*map(ord, 'Mock Watchdog'))
|
arg.identity = (ctypes.c_ubyte*32)(*map(ord, 'Mock Watchdog'))
|
||||||
elif op == linuxwd.WDIOC_GETTIMEOUT:
|
elif op == linuxwd.WDIOC_GETTIMEOUT:
|
||||||
@@ -164,6 +164,13 @@ class TestWatchdog(unittest.TestCase):
|
|||||||
|
|
||||||
watchdog.reload_config({'ttl': 60, 'loop_wait': 15, 'watchdog': {'mode': 'required'}})
|
watchdog.reload_config({'ttl': 60, 'loop_wait': 15, 'watchdog': {'mode': 'required'}})
|
||||||
watchdog.keepalive()
|
watchdog.keepalive()
|
||||||
|
self.assertTrue(watchdog.is_running)
|
||||||
|
self.assertEqual(watchdog.config.timeout, 60 - 5)
|
||||||
|
|
||||||
|
watchdog.reload_config({'ttl': 60, 'loop_wait': 15, 'watchdog': {'mode': 'required', 'safety_margin': -1}})
|
||||||
|
watchdog.keepalive()
|
||||||
|
self.assertTrue(watchdog.is_running)
|
||||||
|
self.assertEqual(watchdog.config.timeout, 60 // 2)
|
||||||
|
|
||||||
|
|
||||||
class TestNullWatchdog(unittest.TestCase):
|
class TestNullWatchdog(unittest.TestCase):
|
||||||
|
|||||||
+20
-4
@@ -6,6 +6,7 @@ from kazoo.client import KazooClient, KazooState
|
|||||||
from kazoo.exceptions import NoNodeError, NodeExistsError
|
from kazoo.exceptions import NoNodeError, NodeExistsError
|
||||||
from kazoo.handlers.threading import SequentialThreadingHandler
|
from kazoo.handlers.threading import SequentialThreadingHandler
|
||||||
from kazoo.protocol.states import KeeperState, ZnodeStat
|
from kazoo.protocol.states import KeeperState, ZnodeStat
|
||||||
|
from kazoo.retry import RetryFailedError
|
||||||
from mock import Mock, PropertyMock, patch
|
from mock import Mock, PropertyMock, patch
|
||||||
from patroni.dcs.zookeeper import Cluster, Leader, PatroniKazooClient,\
|
from patroni.dcs.zookeeper import Cluster, Leader, PatroniKazooClient,\
|
||||||
PatroniSequentialThreadingHandler, ZooKeeper, ZooKeeperError
|
PatroniSequentialThreadingHandler, ZooKeeper, ZooKeeperError
|
||||||
@@ -50,6 +51,8 @@ class MockKazooClient(Mock):
|
|||||||
return (b'foo', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0))
|
return (b'foo', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0))
|
||||||
elif path.endswith('/status'):
|
elif path.endswith('/status'):
|
||||||
return (b'{"optime":500,"slots":{"ls":1234567}}', ZnodeStat(0, 0, 0, 0, 0, 0, 0, -1, 0, 0, 0))
|
return (b'{"optime":500,"slots":{"ls":1234567}}', ZnodeStat(0, 0, 0, 0, 0, 0, 0, -1, 0, 0, 0))
|
||||||
|
elif path.endswith('/failsafe'):
|
||||||
|
return (b'{a}', ZnodeStat(0, 0, 0, 0, 0, 0, 0, -1, 0, 0, 0))
|
||||||
return (b'', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0))
|
return (b'', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0))
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
@@ -59,7 +62,7 @@ class MockKazooClient(Mock):
|
|||||||
if path.startswith('/no_node'):
|
if path.startswith('/no_node'):
|
||||||
raise NoNodeError
|
raise NoNodeError
|
||||||
elif path in ['/service/bla/', '/service/test/']:
|
elif path in ['/service/bla/', '/service/test/']:
|
||||||
return ['initialize', 'leader', 'members', 'optime', 'failover', 'sync']
|
return ['initialize', 'leader', 'members', 'optime', 'failover', 'sync', 'failsafe']
|
||||||
return ['foo', 'bar', 'buzz']
|
return ['foo', 'bar', 'buzz']
|
||||||
|
|
||||||
def create(self, path, value=b"", acl=None, ephemeral=False, sequence=False, makepath=False):
|
def create(self, path, value=b"", acl=None, ephemeral=False, sequence=False, makepath=False):
|
||||||
@@ -124,9 +127,11 @@ class TestPatroniSequentialThreadingHandler(unittest.TestCase):
|
|||||||
self.assertIsNotNone(self.handler.create_connection((), 40))
|
self.assertIsNotNone(self.handler.create_connection((), 40))
|
||||||
self.assertIsNotNone(self.handler.create_connection(timeout=40))
|
self.assertIsNotNone(self.handler.create_connection(timeout=40))
|
||||||
|
|
||||||
@patch.object(SequentialThreadingHandler, 'select', Mock(side_effect=ValueError))
|
|
||||||
def test_select(self):
|
def test_select(self):
|
||||||
|
with patch.object(SequentialThreadingHandler, 'select', Mock(side_effect=ValueError)):
|
||||||
self.assertRaises(select.error, self.handler.select)
|
self.assertRaises(select.error, self.handler.select)
|
||||||
|
with patch.object(SequentialThreadingHandler, 'select', Mock(side_effect=IOError)):
|
||||||
|
self.assertRaises(Exception, self.handler.select)
|
||||||
|
|
||||||
|
|
||||||
class TestPatroniKazooClient(unittest.TestCase):
|
class TestPatroniKazooClient(unittest.TestCase):
|
||||||
@@ -171,7 +176,6 @@ class TestZooKeeper(unittest.TestCase):
|
|||||||
self.zk._inner_load_cluster()
|
self.zk._inner_load_cluster()
|
||||||
|
|
||||||
def test_get_cluster(self):
|
def test_get_cluster(self):
|
||||||
self.assertRaises(ZooKeeperError, self.zk.get_cluster)
|
|
||||||
cluster = self.zk.get_cluster(True)
|
cluster = self.zk.get_cluster(True)
|
||||||
self.assertIsInstance(cluster.leader, Leader)
|
self.assertIsInstance(cluster.leader, Leader)
|
||||||
self.zk.status_watcher(None)
|
self.zk.status_watcher(None)
|
||||||
@@ -220,13 +224,25 @@ class TestZooKeeper(unittest.TestCase):
|
|||||||
self.zk.touch_member({'conn_url': 'postgres://repuser:rep-pass@localhost:5434/postgres',
|
self.zk.touch_member({'conn_url': 'postgres://repuser:rep-pass@localhost:5434/postgres',
|
||||||
'api_url': 'http://127.0.0.1:8009/patroni'})
|
'api_url': 'http://127.0.0.1:8009/patroni'})
|
||||||
|
|
||||||
|
@patch.object(MockKazooClient, 'create', Mock(side_effect=[RetryFailedError, Exception]))
|
||||||
|
def test_attempt_to_acquire_leader(self):
|
||||||
|
self.assertRaises(ZooKeeperError, self.zk.attempt_to_acquire_leader)
|
||||||
|
self.assertFalse(self.zk.attempt_to_acquire_leader())
|
||||||
|
|
||||||
def test_take_leader(self):
|
def test_take_leader(self):
|
||||||
self.zk.take_leader()
|
self.zk.take_leader()
|
||||||
with patch.object(MockKazooClient, 'create', Mock(side_effect=Exception)):
|
with patch.object(MockKazooClient, 'create', Mock(side_effect=Exception)):
|
||||||
self.zk.take_leader()
|
self.zk.take_leader()
|
||||||
|
|
||||||
def test_update_leader(self):
|
def test_update_leader(self):
|
||||||
self.assertTrue(self.zk.update_leader(12345))
|
self.assertFalse(self.zk.update_leader(12345))
|
||||||
|
with patch.object(MockKazooClient, 'delete', Mock(side_effect=RetryFailedError)):
|
||||||
|
self.assertRaises(ZooKeeperError, self.zk.update_leader, 12345)
|
||||||
|
with patch.object(MockKazooClient, 'delete', Mock(side_effect=NoNodeError)):
|
||||||
|
self.assertTrue(self.zk.update_leader(12345, failsafe={'foo': 'bar'}))
|
||||||
|
with patch.object(MockKazooClient, 'create', Mock(side_effect=[RetryFailedError, Exception])):
|
||||||
|
self.assertRaises(ZooKeeperError, self.zk.update_leader, 12345)
|
||||||
|
self.assertFalse(self.zk.update_leader(12345))
|
||||||
|
|
||||||
@patch.object(Cluster, 'min_version', PropertyMock(return_value=(2, 0)))
|
@patch.object(Cluster, 'min_version', PropertyMock(return_value=(2, 0)))
|
||||||
def test_write_leader_optime(self):
|
def test_write_leader_optime(self):
|
||||||
|
|||||||
Reference in New Issue
Block a user