mirror of
https://github.com/outbackdingo/patroni.git
synced 2026-08-26 07:30:14 +00:00
Compare commits
@@ -0,0 +1,48 @@
|
||||
---
|
||||
name: Bug report
|
||||
about: Create a report to help us improve
|
||||
title: ''
|
||||
labels: ''
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**Describe the bug**
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
**To Reproduce**
|
||||
Steps to reproduce the behavior:
|
||||
|
||||
**Expected behavior**
|
||||
A clear and concise description of what you expected to happen.
|
||||
|
||||
**Screenshots**
|
||||
If applicable, add screenshots to help explain your problem.
|
||||
|
||||
**Environment**
|
||||
- Patroni version:
|
||||
- PostgreSQL version:
|
||||
- DCS (and its version):
|
||||
|
||||
**Patroni configuration file**
|
||||
```
|
||||
Please copy&paste your Patroni configuration file here
|
||||
```
|
||||
|
||||
**patronictl show-config**
|
||||
```
|
||||
Please copy&paste the output of "patronictl show-config" command here
|
||||
```
|
||||
|
||||
**Have you checked Patroni logs?**
|
||||
Please provide a snippet of Patroni log files here
|
||||
|
||||
**Have you checked PostgreSQL logs?**
|
||||
Please provide a snippet here
|
||||
|
||||
**Have you tried to use GitHub issue search?**
|
||||
Maybe there is already a similar issue solved.
|
||||
|
||||
|
||||
**Additional context**
|
||||
Add any other context about the problem here.
|
||||
@@ -0,0 +1,189 @@
|
||||
import inspect
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import stat
|
||||
import sys
|
||||
import tarfile
|
||||
import time
|
||||
import zipfile
|
||||
|
||||
|
||||
def install_requirements(what):
|
||||
old_path = sys.path[:]
|
||||
w = os.path.join(os.getcwd(), os.path.dirname(inspect.getfile(inspect.currentframe())))
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(w)))
|
||||
try:
|
||||
from setup import EXTRAS_REQUIRE, read
|
||||
finally:
|
||||
sys.path = old_path
|
||||
requirements = ['mock>=2.0.0', 'flake8', 'pytest', 'pytest-cov'] if what == 'all' else ['behave']
|
||||
requirements += ['coverage']
|
||||
# try to split tests between psycopg2 and psycopg3
|
||||
requirements += ['psycopg[binary]'] if sys.version_info >= (3, 6, 0) and\
|
||||
(sys.platform != 'darwin' or what == 'etcd3') else ['psycopg2-binary']
|
||||
for r in read('requirements.txt').split('\n'):
|
||||
r = r.strip()
|
||||
if r != '':
|
||||
extras = {e for e, v in EXTRAS_REQUIRE.items() if v and any(r.startswith(x) for x in v)}
|
||||
if not extras or what == 'all' or what in extras:
|
||||
requirements.append(r)
|
||||
|
||||
subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'pip'])
|
||||
r = subprocess.call([sys.executable, '-m', 'pip', 'install'] + requirements)
|
||||
s = subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'setuptools'])
|
||||
return s | r
|
||||
|
||||
|
||||
def install_packages(what):
|
||||
from mapping import versions
|
||||
|
||||
packages = {
|
||||
'zookeeper': ['zookeeper', 'zookeeper-bin', 'zookeeperd'],
|
||||
'consul': ['consul'],
|
||||
}
|
||||
packages['exhibitor'] = packages['zookeeper']
|
||||
packages = packages.get(what, [])
|
||||
ver = versions.get(what)
|
||||
subprocess.call(['sudo', 'sed', '-i', 's/pgdg main.*$/pgdg main {0}/'.format(ver),
|
||||
'/etc/apt/sources.list.d/pgdg.list'])
|
||||
subprocess.call(['sudo', 'apt-get', 'update', '-y'])
|
||||
return subprocess.call(['sudo', 'apt-get', 'install', '-y', 'postgresql-' + ver, 'expect-dev', 'wget'] + packages)
|
||||
|
||||
|
||||
def get_file(url, name):
|
||||
try:
|
||||
from urllib.request import urlretrieve
|
||||
except ImportError:
|
||||
from urllib import urlretrieve
|
||||
|
||||
print('Downloading ' + url)
|
||||
urlretrieve(url, name)
|
||||
|
||||
|
||||
def untar(archive, name):
|
||||
with tarfile.open(archive) as tar:
|
||||
f = tar.extractfile(name)
|
||||
dest = os.path.basename(name)
|
||||
with open(dest, 'wb') as d:
|
||||
shutil.copyfileobj(f, d)
|
||||
return dest
|
||||
|
||||
|
||||
def unzip(archive, name):
|
||||
with zipfile.ZipFile(archive, 'r') as z:
|
||||
name = z.extract(name)
|
||||
dest = os.path.basename(name)
|
||||
shutil.move(name, dest)
|
||||
return dest
|
||||
|
||||
|
||||
def unzip_all(archive):
|
||||
print('Extracting ' + archive)
|
||||
with zipfile.ZipFile(archive, 'r') as z:
|
||||
z.extractall()
|
||||
|
||||
|
||||
def chmod_755(name):
|
||||
os.chmod(name, stat.S_IRUSR | stat.S_IWUSR | stat.S_IXUSR |
|
||||
stat.S_IRGRP | stat.S_IXGRP | stat.S_IROTH | stat.S_IXOTH)
|
||||
|
||||
|
||||
def unpack(archive, name):
|
||||
print('Extracting {0} from {1}'.format(name, archive))
|
||||
func = unzip if archive.endswith('.zip') else untar
|
||||
name = func(archive, name)
|
||||
chmod_755(name)
|
||||
return name
|
||||
|
||||
|
||||
def install_etcd():
|
||||
version = os.environ.get('ETCDVERSION', '3.3.13')
|
||||
platform = {'linux2': 'linux', 'win32': 'windows', 'cygwin': 'windows'}.get(sys.platform, sys.platform)
|
||||
dirname = 'etcd-v{0}-{1}-amd64'.format(version, platform)
|
||||
ext = 'tar.gz' if platform == 'linux' else 'zip'
|
||||
name = '{0}.{1}'.format(dirname, ext)
|
||||
url = 'https://github.com/etcd-io/etcd/releases/download/v{0}/{1}'.format(version, name)
|
||||
get_file(url, name)
|
||||
ext = '.exe' if platform == 'windows' else ''
|
||||
return int(unpack(name, '{0}/etcd{1}'.format(dirname, ext)) is None)
|
||||
|
||||
|
||||
def install_postgres():
|
||||
version = os.environ.get('PGVERSION', '14.1-1')
|
||||
platform = {'darwin': 'osx', 'win32': 'windows-x64', 'cygwin': 'windows-x64'}[sys.platform]
|
||||
name = 'postgresql-{0}-{1}-binaries.zip'.format(version, platform)
|
||||
get_file('http://get.enterprisedb.com/postgresql/' + name, name)
|
||||
unzip_all(name)
|
||||
bin_dir = os.path.join('pgsql', 'bin')
|
||||
for f in os.listdir(bin_dir):
|
||||
chmod_755(os.path.join(bin_dir, f))
|
||||
subprocess.call(['pgsql/bin/postgres', '-V'])
|
||||
return 0
|
||||
|
||||
|
||||
def setup_kubernetes():
|
||||
get_file('https://storage.googleapis.com/minikube/k8sReleases/v1.7.0/localkube-linux-amd64', 'localkube')
|
||||
chmod_755('localkube')
|
||||
|
||||
devnull = open(os.devnull, 'w')
|
||||
subprocess.Popen(['sudo', 'nohup', './localkube', '--logtostderr=true', '--enable-dns=false'],
|
||||
stdout=devnull, stderr=devnull)
|
||||
for _ in range(0, 120):
|
||||
if subprocess.call(['wget', '-qO', '-', 'http://127.0.0.1:8080/'], stdout=devnull, stderr=devnull) == 0:
|
||||
break
|
||||
time.sleep(1)
|
||||
else:
|
||||
print('localkube did not start')
|
||||
return 1
|
||||
|
||||
subprocess.call('sudo chmod 644 /var/lib/localkube/certs/*', shell=True)
|
||||
print('Set up .kube/config')
|
||||
kube = os.path.join(os.path.expanduser('~'), '.kube')
|
||||
os.makedirs(kube)
|
||||
with open(os.path.join(kube, 'config'), 'w') as f:
|
||||
f.write("""apiVersion: v1
|
||||
clusters:
|
||||
- cluster:
|
||||
certificate-authority: /var/lib/localkube/certs/ca.crt
|
||||
server: https://127.0.0.1:8443
|
||||
name: local
|
||||
contexts:
|
||||
- context:
|
||||
cluster: local
|
||||
user: myself
|
||||
name: local
|
||||
current-context: local
|
||||
kind: Config
|
||||
preferences: {}
|
||||
users:
|
||||
- name: myself
|
||||
user:
|
||||
client-certificate: /var/lib/localkube/certs/apiserver.crt
|
||||
client-key: /var/lib/localkube/certs/apiserver.key
|
||||
""")
|
||||
return 0
|
||||
|
||||
|
||||
def main():
|
||||
what = os.environ.get('DCS', sys.argv[1] if len(sys.argv) > 1 else 'all')
|
||||
|
||||
if what != 'all':
|
||||
if sys.platform.startswith('linux'):
|
||||
r = install_packages(what)
|
||||
if r == 0 and what == 'kubernetes':
|
||||
r = setup_kubernetes()
|
||||
else:
|
||||
r = install_postgres()
|
||||
|
||||
if r == 0 and what.startswith('etcd'):
|
||||
r = install_etcd()
|
||||
|
||||
if r != 0:
|
||||
return r
|
||||
|
||||
return install_requirements(what)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1 @@
|
||||
versions = {'etcd': '9.6', 'etcd3': '14', 'consul': '13', 'exhibitor': '12', 'raft': '11', 'kubernetes': '14'}
|
||||
@@ -0,0 +1,50 @@
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
|
||||
def main():
|
||||
what = os.environ.get('DCS', sys.argv[1] if len(sys.argv) > 1 else 'all')
|
||||
|
||||
if what == 'all':
|
||||
flake8 = subprocess.call([sys.executable, 'setup.py', 'flake8'])
|
||||
test = subprocess.call([sys.executable, 'setup.py', 'test'])
|
||||
version = '.'.join(map(str, sys.version_info[:2]))
|
||||
shutil.move('.coverage', os.path.join(tempfile.gettempdir(), '.coverage.' + version))
|
||||
return flake8 | test
|
||||
elif what == 'combine':
|
||||
tmp = tempfile.gettempdir()
|
||||
for name in os.listdir(tmp):
|
||||
if name.startswith('.coverage.'):
|
||||
shutil.move(os.path.join(tmp, name), name)
|
||||
return subprocess.call([sys.executable, '-m', 'coverage', 'combine'])
|
||||
|
||||
env = os.environ.copy()
|
||||
if sys.platform.startswith('linux'):
|
||||
from mapping import versions
|
||||
|
||||
version = versions.get(what)
|
||||
path = '/usr/lib/postgresql/{0}/bin:.'.format(version)
|
||||
unbuffer = ['timeout', '900', 'unbuffer']
|
||||
args = ['--tags=-skip'] if what == 'etcd' else []
|
||||
else:
|
||||
path = os.path.abspath(os.path.join('pgsql', 'bin'))
|
||||
if sys.platform == 'darwin':
|
||||
path += ':.'
|
||||
args = unbuffer = []
|
||||
env['PATH'] = path + os.pathsep + env['PATH']
|
||||
env['DCS'] = what
|
||||
|
||||
ret = subprocess.call(unbuffer + [sys.executable, '-m', 'behave'] + args, env=env)
|
||||
|
||||
if ret != 0:
|
||||
if subprocess.call('grep . features/output/*_failed/*postgres?.*', shell=True) != 0:
|
||||
subprocess.call('grep . features/output/*/*postgres?.*', shell=True)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,151 @@
|
||||
name: Tests
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
tags:
|
||||
- v.*
|
||||
|
||||
jobs:
|
||||
unit:
|
||||
runs-on: ${{ matrix.os }}-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu, windows, macos]
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v1
|
||||
- name: Set up Python 2.7
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: 2.7
|
||||
if: matrix.os != 'windows'
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
if: matrix.os != 'windows'
|
||||
- name: Run tests and flake8
|
||||
run: python .github/workflows/run_tests.py
|
||||
if: matrix.os != 'windows'
|
||||
|
||||
- name: Set up Python 3.6
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: 3.6
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
- name: Run tests and flake8
|
||||
run: python .github/workflows/run_tests.py
|
||||
|
||||
- name: Set up Python 3.7
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: 3.7
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
- name: Run tests and flake8
|
||||
run: python .github/workflows/run_tests.py
|
||||
|
||||
- name: Set up Python 3.8
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: 3.8
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
- name: Run tests and flake8
|
||||
run: python .github/workflows/run_tests.py
|
||||
|
||||
- name: Set up Python 3.9
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: 3.9
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
- name: Run tests and flake8
|
||||
run: python .github/workflows/run_tests.py
|
||||
|
||||
- name: Set up Python 3.10
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: '3.10'
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
- name: Run tests and flake8
|
||||
run: python .github/workflows/run_tests.py
|
||||
|
||||
- name: Combine coverage
|
||||
run: python .github/workflows/run_tests.py combine
|
||||
|
||||
- name: Install coveralls
|
||||
run: python -m pip install coveralls
|
||||
|
||||
- name: Upload Coverage
|
||||
env:
|
||||
COVERALLS_FLAG_NAME: unit-${{ matrix.os }}
|
||||
COVERALLS_PARALLEL: 'true'
|
||||
GITHUB_TOKEN: ${{ secrets.github_token }}
|
||||
run: python -m coveralls --service=github
|
||||
|
||||
behave:
|
||||
runs-on: ${{ matrix.os }}-latest
|
||||
env:
|
||||
DCS: ${{ matrix.dcs }}
|
||||
ETCDVERSION: 3.3.13
|
||||
PGVERSION: 12.1-1 # for windows and macos
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu]
|
||||
python-version: [2.7, 3.6, 3.9]
|
||||
dcs: [etcd, etcd3, consul, exhibitor, kubernetes, raft]
|
||||
exclude:
|
||||
- dcs: kubernetes
|
||||
python-version: 2.7
|
||||
include:
|
||||
- os: macos
|
||||
python-version: 3.7
|
||||
dcs: raft
|
||||
- os: macos
|
||||
python-version: 3.8
|
||||
dcs: etcd
|
||||
- os: macos
|
||||
python-version: '3.10'
|
||||
dcs: etcd3
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v1
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
- name: Add postgresql apt repo
|
||||
run: sudo sh -c 'echo "deb http://apt.postgresql.org/pub/repos/apt $(lsb_release -cs)-pgdg main" > /etc/apt/sources.list.d/pgdg.list'
|
||||
if: matrix.os == 'ubuntu'
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
- name: Run behave tests
|
||||
run: python .github/workflows/run_tests.py
|
||||
- uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: '3.10'
|
||||
- name: Install coveralls
|
||||
run: python -m pip install coveralls
|
||||
- name: Upload Coverage
|
||||
env:
|
||||
COVERALLS_FLAG_NAME: behave-${{ matrix.os }}-${{ matrix.dcs }}-${{ matrix.python-version }}
|
||||
COVERALLS_PARALLEL: 'true'
|
||||
GITHUB_TOKEN: ${{ secrets.github_token }}
|
||||
run: python -m coveralls --service=github
|
||||
|
||||
coveralls-finish:
|
||||
name: Finalize coveralls.io
|
||||
needs: [unit, behave]
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/setup-python@v2
|
||||
- run: python -m pip install coveralls
|
||||
- run: python -m coveralls --service=github --finish
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.github_token }}
|
||||
+53
-6
@@ -1,12 +1,59 @@
|
||||
data/*
|
||||
*.pyc
|
||||
*.egg/
|
||||
*.egg-info/
|
||||
*.py[cod]
|
||||
|
||||
# vi(m) swap files:
|
||||
*.sw?
|
||||
|
||||
# C extensions
|
||||
*.so
|
||||
|
||||
# Packages
|
||||
.cache/
|
||||
*.egg
|
||||
*.eggs
|
||||
*.egg-info
|
||||
dist
|
||||
build
|
||||
eggs
|
||||
parts
|
||||
bin
|
||||
var
|
||||
sdist
|
||||
develop-eggs
|
||||
.installed.cfg
|
||||
lib
|
||||
lib64
|
||||
|
||||
# Installer logs
|
||||
pip-log.txt
|
||||
|
||||
# Unit test / coverage reports
|
||||
.coverage
|
||||
.eggs/
|
||||
build/
|
||||
.tox
|
||||
nosetests.xml
|
||||
coverage.xml
|
||||
htmlcov
|
||||
junit.xml
|
||||
features/output
|
||||
dummy
|
||||
|
||||
# Translations
|
||||
*.mo
|
||||
|
||||
# Mr Developer
|
||||
.mr.developer.cfg
|
||||
.project
|
||||
.pydevproject
|
||||
|
||||
pgpass
|
||||
scm-source.json
|
||||
|
||||
# Sphinx-generated documentation
|
||||
docs/build/
|
||||
docs/source/_static/
|
||||
docs/source/_templates/
|
||||
|
||||
# Pycharm IDE
|
||||
.idea/
|
||||
|
||||
#VSCode IDE
|
||||
.vscode/
|
||||
|
||||
-82
@@ -1,82 +0,0 @@
|
||||
sudo: false
|
||||
language: python
|
||||
python:
|
||||
- "3.5"
|
||||
addons:
|
||||
apt:
|
||||
packages:
|
||||
- postgresql-contrib-9.5
|
||||
postgresql: "9.5"
|
||||
env:
|
||||
global:
|
||||
- ETCDVERSION=2.3.2 ZKVERSION=3.4.6 CONSULVERSION=0.6.4
|
||||
matrix:
|
||||
- TEST_SUITE="python setup.py"
|
||||
- DCS="etcd" TEST_SUITE="behave"
|
||||
- DCS="exhibitor" TEST_SUITE="behave"
|
||||
- DCS="consul" TEST_SUITE="behave"
|
||||
cache:
|
||||
directories:
|
||||
- $HOME/virtualenv/python2.7.9
|
||||
- $HOME/virtualenv/python3.4.2
|
||||
- $HOME/virtualenv/python3.5.0
|
||||
install:
|
||||
- |
|
||||
set -e
|
||||
|
||||
if [[ $TEST_SUITE == "behave" ]]; then
|
||||
if [[ $DCS == "consul" ]]; then
|
||||
curl -L https://releases.hashicorp.com/consul/${CONSULVERSION}/consul_${CONSULVERSION}_linux_amd64.zip \
|
||||
| gunzip > consul
|
||||
chmod +x consul
|
||||
fi
|
||||
|
||||
if [[ $DCS == "etcd" ]]; then
|
||||
curl -L https://github.com/coreos/etcd/releases/download/v${ETCDVERSION}/etcd-v${ETCDVERSION}-linux-amd64.tar.gz \
|
||||
| tar xz -C . --strip=1 --wildcards --no-anchored etcd
|
||||
fi
|
||||
|
||||
if [[ $DCS == "exhibitor" ]]; then
|
||||
curl -L http://www.apache.org/dist/zookeeper/zookeeper-${ZKVERSION}/zookeeper-${ZKVERSION}.tar.gz | tar xz
|
||||
mv zookeeper-${ZKVERSION}/conf/zoo_sample.cfg zookeeper-${ZKVERSION}/conf/zoo.cfg
|
||||
zookeeper-${ZKVERSION}/bin/zkServer.sh start
|
||||
# following lines are 'emulating' exhibitor REST API
|
||||
while true; do
|
||||
echo -e 'HTTP/1.0 200 OK\nContent-Type: application/json\n\n{"servers":["127.0.0.1"],"port":2181}' \
|
||||
| nc -l 8181 &> /dev/null
|
||||
done&
|
||||
fi
|
||||
fi
|
||||
|
||||
for pv in "2.7" "3.4" "3.5"; do
|
||||
source ~/virtualenv/python${pv}/bin/activate
|
||||
# explicitly install all needed python modules to cache them
|
||||
for p in '-r requirements.txt' 'behave codacy-coverage coverage coveralls flake8 mock>=2.0.0 pytest-cov pytest'; do
|
||||
pip install $p
|
||||
done
|
||||
done
|
||||
script:
|
||||
- |
|
||||
for pv in "2.7" "3.4" "3.5"; do
|
||||
source ~/virtualenv/python${pv}/bin/activate
|
||||
|
||||
if [[ $TEST_SUITE == "behave" ]]; then
|
||||
if [[ $pv != "3.4" ]]; then
|
||||
echo Running acceptance tests using python${pv}
|
||||
if ! PATH=.:$PATH $TEST_SUITE; then
|
||||
# output all log files when tests are failing
|
||||
grep . features/output/*/*postgres?.*
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
else
|
||||
echo Running unit tests using python${pv}
|
||||
$TEST_SUITE test
|
||||
$TEST_SUITE flake8
|
||||
fi
|
||||
done
|
||||
|
||||
set +e
|
||||
after_success:
|
||||
- coveralls
|
||||
- if [[ $TEST_SUITE != "behave" ]]; then python-codacy-coverage -r coverage.xml; fi
|
||||
-12
@@ -1,12 +0,0 @@
|
||||
approvals:
|
||||
# PR needs at least 4 approvals
|
||||
minimum: 1
|
||||
# approval = comment that matches this regex
|
||||
pattern: "^\\s*(:?\\+1:?|👍)\\s*$"
|
||||
from:
|
||||
# commenter must be either one of:
|
||||
# a public zalando org member
|
||||
orgs:
|
||||
- zalando
|
||||
# a collaborator of the repo
|
||||
collaborators: true
|
||||
+122
-31
@@ -1,44 +1,135 @@
|
||||
## This Dockerfile is meant to aid in the building and debugging patroni whilst developing on your local machine
|
||||
## It has all the necessary components to play/debug with a single node appliance, running etcd
|
||||
FROM ubuntu:16.04
|
||||
MAINTAINER Feike Steenbergen <[email protected]>
|
||||
ARG PG_MAJOR=14
|
||||
ARG COMPRESS=false
|
||||
ARG PGHOME=/home/postgres
|
||||
ARG PGDATA=$PGHOME/data
|
||||
ARG LC_ALL=C.UTF-8
|
||||
ARG LANG=C.UTF-8
|
||||
|
||||
RUN echo 'APT::Install-Recommends "0";' > /etc/apt/apt.conf.d/01norecommend \
|
||||
&& echo 'APT::Install-Suggests "0";' >> /etc/apt/apt.conf.d/01norecommend
|
||||
FROM postgres:$PG_MAJOR as builder
|
||||
|
||||
ENV PGVERSION 9.5
|
||||
ENV PATH /usr/lib/postgresql/${PGVERSION}/bin:$PATH
|
||||
RUN apt-get update -y \
|
||||
&& apt-get upgrade -y \
|
||||
&& apt-get install -y curl jq haproxy zookeeper postgresql-${PGVERSION} python-psycopg2 python-yaml \
|
||||
python-requests python-six python-click python-dateutil python-tzlocal python-urllib3 \
|
||||
python-dnspython python-pip python-setuptools python-kazoo python-prettytable python \
|
||||
&& pip install python-etcd==0.4.3 python-consul==0.6.0 --upgrade \
|
||||
&& apt-get remove -y python-pip python-setuptools \
|
||||
ARG PGHOME
|
||||
ARG PGDATA
|
||||
ARG LC_ALL
|
||||
ARG LANG
|
||||
|
||||
ENV ETCDVERSION=3.5.4 CONFDVERSION=0.16.0
|
||||
|
||||
RUN set -ex \
|
||||
&& export DEBIAN_FRONTEND=noninteractive \
|
||||
&& echo 'APT::Install-Recommends "0";\nAPT::Install-Suggests "0";' > /etc/apt/apt.conf.d/01norecommend \
|
||||
&& apt-get update -y \
|
||||
# postgres:10 is based on debian, which has the patroni package. We will install all required dependencies
|
||||
&& apt-cache depends patroni | sed -n -e 's/.*Depends: \(python3-.\+\)$/\1/p' \
|
||||
| grep -Ev '^python3-(sphinx|etcd|consul|kazoo|kubernetes)' \
|
||||
| xargs apt-get install -y vim curl less jq locales haproxy sudo \
|
||||
python3-etcd python3-kazoo python3-pip busybox \
|
||||
net-tools iputils-ping --fix-missing \
|
||||
&& pip3 install dumb-init \
|
||||
\
|
||||
# Cleanup all locales but en_US.UTF-8
|
||||
&& find /usr/share/i18n/charmaps/ -type f ! -name UTF-8.gz -delete \
|
||||
&& find /usr/share/i18n/locales/ -type f ! -name en_US ! -name en_GB ! -name i18n* ! -name iso14651_t1 ! -name iso14651_t1_common ! -name 'translit_*' -delete \
|
||||
&& echo 'en_US.UTF-8 UTF-8' > /usr/share/i18n/SUPPORTED \
|
||||
\
|
||||
# Make sure we have a en_US.UTF-8 locale available
|
||||
&& localedef -i en_US -c -f UTF-8 -A /usr/share/locale/locale.alias en_US.UTF-8 \
|
||||
\
|
||||
# haproxy dummy config
|
||||
&& echo 'global\n stats socket /run/haproxy/admin.sock mode 660 level admin' > /etc/haproxy/haproxy.cfg \
|
||||
\
|
||||
# vim config
|
||||
&& echo 'syntax on\nfiletype plugin indent on\nset mouse-=a\nautocmd FileType yaml setlocal ts=2 sts=2 sw=2 expandtab' > /etc/vim/vimrc.local \
|
||||
\
|
||||
# Prepare postgres/patroni/haproxy environment
|
||||
&& mkdir -p $PGHOME/.config/patroni /patroni /run/haproxy \
|
||||
&& ln -s ../../postgres0.yml $PGHOME/.config/patroni/patronictl.yaml \
|
||||
&& ln -s /patronictl.py /usr/local/bin/patronictl \
|
||||
&& sed -i "s|/var/lib/postgresql.*|$PGHOME:/bin/bash|" /etc/passwd \
|
||||
&& chown -R postgres:postgres /var/log \
|
||||
\
|
||||
# Download etcd
|
||||
&& curl -L https://github.com/coreos/etcd/releases/download/v${ETCDVERSION}/etcd-v${ETCDVERSION}-linux-$(dpkg --print-architecture).tar.gz \
|
||||
| tar xz -C /usr/local/bin --strip=1 --wildcards --no-anchored etcd etcdctl \
|
||||
\
|
||||
# Download confd
|
||||
&& curl -sL https://github.com/kelseyhightower/confd/releases/download/v${CONFDVERSION}/confd-${CONFDVERSION}-linux-$(dpkg --print-architecture) \
|
||||
> /usr/local/bin/confd && chmod +x /usr/local/bin/confd \
|
||||
\
|
||||
# Clean up all useless packages and some files
|
||||
&& apt-get purge -y --allow-remove-essential python3-pip gzip bzip2 util-linux e2fsprogs \
|
||||
libmagic1 bsdmainutils login ncurses-bin libmagic-mgc e2fslibs bsdutils \
|
||||
exim4-config gnupg-agent dirmngr libpython2.7-stdlib libpython2.7-minimal \
|
||||
&& apt-get autoremove -y \
|
||||
# Clean up
|
||||
&& apt-get clean -y \
|
||||
&& rm -rf /var/lib/apt/lists/* /root/.cache
|
||||
&& rm -rf /var/lib/apt/lists/* \
|
||||
/root/.cache \
|
||||
/var/cache/debconf/* \
|
||||
/etc/rc?.d \
|
||||
/etc/systemd \
|
||||
/docker-entrypoint* \
|
||||
/sbin/pam* \
|
||||
/sbin/swap* \
|
||||
/sbin/unix* \
|
||||
/usr/local/bin/gosu \
|
||||
/usr/sbin/[acgipr]* \
|
||||
/usr/sbin/*user* \
|
||||
/usr/share/doc* \
|
||||
/usr/share/man \
|
||||
/usr/share/info \
|
||||
/usr/share/i18n/locales/translit_hangul \
|
||||
/usr/share/locale/?? \
|
||||
/usr/share/locale/??_?? \
|
||||
/usr/share/postgresql/*/man \
|
||||
/usr/share/postgresql-common/pg_wrapper \
|
||||
/usr/share/vim/vim80/doc \
|
||||
/usr/share/vim/vim80/lang \
|
||||
/usr/share/vim/vim80/tutor \
|
||||
# /var/lib/dpkg/info/* \
|
||||
&& find /usr/bin -xtype l -delete \
|
||||
&& find /var/log -type f -exec truncate --size 0 {} \; \
|
||||
&& find /usr/lib/python3/dist-packages -name '*test*' | xargs rm -fr
|
||||
|
||||
ENV ETCDVERSION 2.3.6
|
||||
RUN curl -L https://github.com/coreos/etcd/releases/download/v${ETCDVERSION}/etcd-v${ETCDVERSION}-linux-amd64.tar.gz \
|
||||
| tar xz -C /usr/local/bin --strip=1 --wildcards --no-anchored etcd etcdctl
|
||||
FROM scratch
|
||||
COPY --from=builder / /
|
||||
|
||||
ENV CONFDVERSION 0.11.0
|
||||
RUN curl -L https://github.com/kelseyhightower/confd/releases/download/v${CONFDVERSION}/confd-${CONFDVERSION}-linux-amd64 > /usr/local/bin/confd \
|
||||
&& chmod +x /usr/local/bin/confd
|
||||
LABEL maintainer="Alexander Kukushkin <[email protected]>"
|
||||
|
||||
ADD patronictl.py patroni.py docker/entrypoint.sh /
|
||||
ADD patroni /patroni/
|
||||
ADD extras/confd /etc/confd
|
||||
RUN ln -s /patronictl.py /usr/local/bin/patronictl
|
||||
ARG PG_MAJOR
|
||||
ARG COMPRESS
|
||||
ARG PGHOME
|
||||
ARG PGDATA
|
||||
ARG LC_ALL
|
||||
ARG LANG
|
||||
|
||||
### Setting up a simple script that will serve as an entrypoint
|
||||
RUN mkdir /data/ && touch /pgpass /patroni.yml \
|
||||
&& chown postgres:postgres -R /patroni/ /data/ /pgpass /patroni.yml /etc/haproxy /var/run/ /var/lib/ /var/log/ \
|
||||
&& echo 1 > /etc/zookeeper/conf/myid
|
||||
ARG PGBIN=/usr/lib/postgresql/$PG_MAJOR/bin
|
||||
|
||||
EXPOSE 2379 5432 8008
|
||||
ENV LC_ALL=$LC_ALL LANG=$LANG EDITOR=/usr/bin/editor
|
||||
ENV PGDATA=$PGDATA PATH=$PATH:$PGBIN
|
||||
|
||||
COPY patroni /patroni/
|
||||
COPY extras/confd/conf.d/haproxy.toml /etc/confd/conf.d/
|
||||
COPY extras/confd/templates/haproxy.tmpl /etc/confd/templates/
|
||||
COPY patroni*.py docker/entrypoint.sh /
|
||||
COPY postgres?.yml $PGHOME/
|
||||
|
||||
WORKDIR $PGHOME
|
||||
|
||||
RUN sed -i 's/env python/&3/' /patroni*.py \
|
||||
# "fix" patroni configs
|
||||
&& sed -i 's/^\( connect_address:\| - host\)/#&/' postgres?.yml \
|
||||
&& sed -i 's/^ listen: 127.0.0.1/ listen: 0.0.0.0/' postgres?.yml \
|
||||
&& sed -i "s|^\( data_dir: \).*|\1$PGDATA|" postgres?.yml \
|
||||
&& sed -i "s|^#\( bin_dir: \).*|\1$PGBIN|" postgres?.yml \
|
||||
&& sed -i 's/^ - encoding: UTF8/ - locale: en_US.UTF-8\n&/' postgres?.yml \
|
||||
&& sed -i 's/^\(scope\|name\|etcd\| host\| authentication\| pg_hba\| parameters\):/#&/' postgres?.yml \
|
||||
&& sed -i 's/^ \(replication\|superuser\|rewind\|unix_socket_directories\|\(\( \)\{0,1\}\(username\|password\)\)\):/#&/' postgres?.yml \
|
||||
&& sed -i 's/^ parameters:/ pg_hba:\n - local all all trust\n - host replication all all md5\n - host all all all md5\n&\n max_connections: 100/' postgres?.yml \
|
||||
&& if [ "$COMPRESS" = "true" ]; then chmod u+s /usr/bin/sudo; fi \
|
||||
&& chmod +s /bin/ping \
|
||||
&& chown -R postgres:postgres $PGHOME /run /etc/haproxy
|
||||
|
||||
ENTRYPOINT ["/bin/bash", "/entrypoint.sh"]
|
||||
USER postgres
|
||||
|
||||
ENTRYPOINT ["/bin/sh", "/entrypoint.sh"]
|
||||
|
||||
+109
-54
@@ -1,64 +1,144 @@
|
||||
|Build Status| |Coverage Status|
|
||||
|Tests Status| |Coverage Status|
|
||||
|
||||
Patroni: A Template for PostgreSQL HA with ZooKeeper, etcd or Consul
|
||||
------------------------------------------------------------
|
||||
--------------------------------------------------------------------
|
||||
|
||||
You can find a version of this documentation that is searchable and also easier to navigate at `patroni.readthedocs.io <https://patroni.readthedocs.io>`__.
|
||||
|
||||
|
||||
There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
|
||||
|
||||
Patroni is a template for you to create your own customized, high-availability solution using Python and — for maximum accessibility — a distributed configuration store like `ZooKeeper <https://zookeeper.apache.org/>`__, `etcd <https://github.com/coreos/etcd>`__ or `Consul <https://github.com/hashicorp/consul>`__. Database engineers, DBAs, DevOps engineers, and SREs who are looking to quickly deploy HA PostgreSQL in the datacenter—or anywhere else—will hopefully find it useful.
|
||||
Patroni is a template for you to create your own customized, high-availability solution using Python and - for maximum accessibility - a distributed configuration store like `ZooKeeper <https://zookeeper.apache.org/>`__, `etcd <https://github.com/coreos/etcd>`__, `Consul <https://github.com/hashicorp/consul>`__ or `Kubernetes <https://kubernetes.io>`__. Database engineers, DBAs, DevOps engineers, and SREs who are looking to quickly deploy HA PostgreSQL in the datacenter-or anywhere else-will hopefully find it useful.
|
||||
|
||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely.
|
||||
|
||||
**Note to Kubernetes users**: We're currently developing Patroni to be as useful as possible for teams running Kubernetes on top of Google Compute Engine; Patroni can be the HA solution for Postgres in such an environment. Please contact us via our Issues Tracker if this describes your team's current setup, and we'll follow up.
|
||||
Currently supported PostgreSQL versions: 9.3 to 14.
|
||||
|
||||
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the `Kubernetes <https://github.com/zalando/patroni/blob/master/docs/kubernetes.rst>`__ chapter of the Patroni documentation.
|
||||
|
||||
.. contents::
|
||||
:local:
|
||||
:depth: 1
|
||||
:backlinks: none
|
||||
|
||||
==============
|
||||
=================
|
||||
How Patroni Works
|
||||
==============
|
||||
=================
|
||||
|
||||
Patroni originated as a fork of `Governor <https://github.com/compose/governor>`__, the project from Compose. It includes plenty of new features.
|
||||
Patroni originated as a fork of `Governor <https://github.com/compose/governor>`__, the project from Compose. It includes plenty of new features.
|
||||
|
||||
For an example of a Docker-based deployment with Patroni, see `Spilo <https://github.com/zalando/spilo>`__, currently in use at Zalando.
|
||||
|
||||
For additional background info, see:
|
||||
|
||||
* `Elephants on Automatic: HA Clustered PostgreSQL with Helm <https://www.youtube.com/watch?v=CftcVhFMGSY>`_, talk by Josh Berkus and Oleksii Kliukin at KubeCon Berlin 2017
|
||||
* `PostgreSQL HA with Kubernetes and Patroni <https://www.youtube.com/watch?v=iruaCgeG7qs>`__, talk by Josh Berkus at KubeCon 2016 (video)
|
||||
* `Feb. 2016 Zalando Tech blog post <https://tech.zalando.de/blog/zalandos-patroni-a-template-for-high-availability-postgresql/>`__
|
||||
|
||||
================
|
||||
==================
|
||||
Development Status
|
||||
================
|
||||
==================
|
||||
|
||||
Patroni is in active development and accepts contributions. See our `Contributing <https://github.com/zalando/patroni/blob/master/README.rst#contributing>`__ section below for more details.
|
||||
Patroni is in active development and accepts contributions. See our `Contributing <https://github.com/zalando/patroni/blob/master/docs/CONTRIBUTING.rst>`__ section below for more details.
|
||||
|
||||
===========================
|
||||
We report new releases information `here <https://github.com/zalando/patroni/releases>`__.
|
||||
|
||||
=========
|
||||
Community
|
||||
=========
|
||||
|
||||
There are two places to connect with the Patroni community: `on github <https://github.com/zalando/patroni>`__, via Issues and PRs, and on channel #patroni in the `PostgreSQL Slack <https://postgres-slack.herokuapp.com/>`__. If you're using Patroni, or just interested, please join us.
|
||||
|
||||
===================================
|
||||
Technical Requirements/Installation
|
||||
===========================
|
||||
===================================
|
||||
|
||||
**For Mac**
|
||||
**Pre-requirements for Mac OS**
|
||||
|
||||
To install requirements on a Mac, run the following:
|
||||
|
||||
::
|
||||
|
||||
brew install postgresql etcd haproxy libyaml python
|
||||
pip install psycopg2 pyyaml
|
||||
|
||||
===================
|
||||
**Psycopg**
|
||||
|
||||
Starting from `psycopg2-2.8 <http://initd.org/psycopg/articles/2019/04/04/psycopg-28-released/>`__ the binary version of psycopg2 will no longer be installed by default. Installing it from the source code requires C compiler and postgres+python dev packages.
|
||||
Since in the python world it is not possible to specify dependency as ``psycopg2 OR psycopg2-binary`` you will have to decide how to install it.
|
||||
|
||||
There are a few options available:
|
||||
|
||||
1. Use the package manager from your distro
|
||||
|
||||
::
|
||||
|
||||
sudo apt-get install python-psycopg2 # install python2 psycopg2 module on Debian/Ubuntu
|
||||
sudo apt-get install python3-psycopg2 # install python3 psycopg2 module on Debian/Ubuntu
|
||||
sudo yum install python-psycopg2 # install python2 psycopg2 on RedHat/Fedora/CentOS
|
||||
|
||||
2. Install psycopg2 from the binary package
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg2-binary
|
||||
|
||||
3. Install psycopg2 from source
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg2>=2.5.4
|
||||
|
||||
4. Use psycopg 3.0 instead of psycopg2
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg[binary]
|
||||
|
||||
**General installation for pip**
|
||||
|
||||
Patroni can be installed with pip:
|
||||
|
||||
::
|
||||
|
||||
pip install patroni[dependencies]
|
||||
|
||||
where dependencies can be either empty, or consist of one or more of the following:
|
||||
|
||||
etcd or etcd3
|
||||
`python-etcd` module in order to use Etcd as DCS
|
||||
consul
|
||||
`python-consul` module in order to use Consul as DCS
|
||||
zookeeper
|
||||
`kazoo` module in order to use Zookeeper as DCS
|
||||
exhibitor
|
||||
`kazoo` module in order to use Exhibitor as DCS (same dependencies as for Zookeeper)
|
||||
kubernetes
|
||||
`kubernetes` module in order to use Kubernetes as DCS in Patroni
|
||||
raft
|
||||
`pysyncobj` module in order to use python Raft implementation as DCS
|
||||
aws
|
||||
`boto` in order to use AWS callbacks
|
||||
|
||||
For example, the command in order to install Patroni together with dependencies for Etcd as a DCS and AWS callbacks is:
|
||||
|
||||
::
|
||||
|
||||
pip install patroni[etcd,aws]
|
||||
|
||||
Note that external tools to call in the replica creation or custom bootstrap scripts (i.e. WAL-E) should be installed independently of Patroni.
|
||||
|
||||
=======================
|
||||
Running and Configuring
|
||||
===================
|
||||
=======================
|
||||
|
||||
To get started, do the following from different terminals:
|
||||
::
|
||||
|
||||
> etcd --data-dir=data/etcd
|
||||
> etcd --data-dir=data/etcd --enable-v2=true
|
||||
> ./patroni.py postgres0.yml
|
||||
> ./patroni.py postgres1.yml
|
||||
|
||||
You will then see a high-availability cluster start up. Test different settings in the YAML files to see how the cluster’s behavior changes. Kill some of the components to see how the system behaves.
|
||||
You will then see a high-availability cluster start up. Test different settings in the YAML files to see how the cluster's behavior changes. Kill some of the components to see how the system behaves.
|
||||
|
||||
Add more ``postgres*.yml`` files to create an even larger cluster.
|
||||
|
||||
@@ -73,11 +153,11 @@ run:
|
||||
|
||||
> psql --host 127.0.0.1 --port 5000 postgres
|
||||
|
||||
===============
|
||||
==================
|
||||
YAML Configuration
|
||||
===============
|
||||
==================
|
||||
|
||||
Go `here <https://github.com/zalando/patroni/blob/master/docs/SETTINGS.rst>`__ for comprehensive information about settings for etcd, consul, and ZooKeeper. And for an example, see `postgres0.yml <https://github.com/zalando/patroni/blob/master/postgres0.yml>`__.
|
||||
Go `here <https://github.com/zalando/patroni/blob/master/docs/SETTINGS.rst>`__ for comprehensive information about settings for etcd, consul, and ZooKeeper. And for an example, see `postgres0.yml <https://github.com/zalando/patroni/blob/master/postgres0.yml>`__.
|
||||
|
||||
=========================
|
||||
Environment Configuration
|
||||
@@ -85,44 +165,19 @@ Environment Configuration
|
||||
|
||||
Go `here <https://github.com/zalando/patroni/blob/master/docs/ENVIRONMENT.rst>`__ for comprehensive information about configuring(overriding) settings via environment variables.
|
||||
|
||||
===============
|
||||
===================
|
||||
Replication Choices
|
||||
===============
|
||||
===================
|
||||
|
||||
Patroni uses Postgres' streaming replication, which is asynchronous by default. For more information, see the `Postgres documentation on streaming replication <http://www.postgresql.org/docs/current/static/warm-standby.html#STREAMING-REPLICATION>`__.
|
||||
Patroni uses Postgres' streaming replication, which is asynchronous by default. Patroni's asynchronous replication configuration allows for ``maximum_lag_on_failover`` settings. This setting ensures failover will not occur if a follower is more than a certain number of bytes behind the leader. This setting should be increased or decreased based on business requirements. It's also possible to use synchronous replication for better durability guarantees. See `replication modes documentation <https://github.com/zalando/patroni/blob/master/docs/replication_modes.rst>`__ for details.
|
||||
|
||||
Patroni's asynchronous replication configuration allows for ``maximum_lag_on_failover`` settings. This setting ensures failover will not occur if a follower is more than a certain number of bytes behind the follower. This setting should be increased or decreased based on business requirements.
|
||||
|
||||
When asynchronous replication is not optimal for your use case, investigate Postgres's `synchronous replication <http://www.postgresql.org/docs/current/static/warm-standby.html#SYNCHRONOUS-REPLICATION>`__. Synchronous replication ensures consistency across a cluster by confirming that writes are written to a secondary before returning to the connecting client with a success. The cost of synchronous replication: reduced throughput on writes. This throughput will be entirely based on network performance.
|
||||
|
||||
In hosted datacenter environments (like AWS, Rackspace, or any network you do not control), synchronous replication significantly increases the variability of write performance. If followers become inaccessible from the leader, the leader effectively becomes read-only.
|
||||
|
||||
To enable a simple synchronous replication test, add the follow lines to the ``parameters`` section of your YAML configuration files:
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
synchronous_commit: "on"
|
||||
synchronous_standby_names: "*"
|
||||
|
||||
When using synchronous replication, use at least three Postgres data nodes to ensure write availability if one host fails.
|
||||
|
||||
Choosing your replication schema is dependent on your business considerations. Investigate both async and sync replication, as well as other HA solutions, to determine which solution is best for you.
|
||||
|
||||
===============================
|
||||
======================================
|
||||
Applications Should Not Use Superusers
|
||||
===============================
|
||||
======================================
|
||||
|
||||
When connecting from an application, always use a non-superuser. Patroni requires access to the database to function properly. By using a superuser from an application, you can potentially use the entire connection pool, including the connections reserved for superusers, with the ``superuser_reserved_connections`` setting. If Patroni cannot access the Primary because the connection pool is full, behavior will be undesirable.
|
||||
|
||||
================
|
||||
Contributing
|
||||
================
|
||||
Patroni accepts contributions from the open-source community; see the `Issues Tracker <https://github.com/zalando/patroni/issues>`__ for current needs.
|
||||
|
||||
Before making a contribution, please let us know by posting a comment to the relevant issue.
|
||||
If you would like to propose a new feature, please first file a new issue explaining the feature you’d like to create.
|
||||
|
||||
.. |Build Status| image:: https://travis-ci.org/zalando/patroni.svg?branch=master
|
||||
:target: https://travis-ci.org/zalando/patroni
|
||||
.. |Tests Status| image:: https://github.com/zalando/patroni/actions/workflows/tests.yaml/badge.svg
|
||||
:target: https://github.com/zalando/patroni/actions/workflows/tests.yaml?query=branch%3Amaster
|
||||
.. |Coverage Status| image:: https://coveralls.io/repos/zalando/patroni/badge.svg?branch=master
|
||||
:target: https://coveralls.io/r/zalando/patroni?branch=master
|
||||
:target: https://coveralls.io/github/zalando/patroni?branch=master
|
||||
|
||||
+72
-52
@@ -1,58 +1,78 @@
|
||||
# docker compose file for running a 3-node PostgreSQL cluster
|
||||
# with etcd as the SIS
|
||||
# with 3-node etcd cluster as the DCS and one haproxy node
|
||||
version: "2"
|
||||
|
||||
patroni_etcd:
|
||||
container_name: patroni_etcd
|
||||
image: patroni
|
||||
command: --etcd
|
||||
networks:
|
||||
demo:
|
||||
|
||||
dbnode1:
|
||||
image: patroni
|
||||
hostname: dbnode1
|
||||
links:
|
||||
- patroni_etcd:patroni_etcd
|
||||
volumes:
|
||||
- ./patroni:/patroni
|
||||
env_file: docker/patroni-secrets.env
|
||||
environment:
|
||||
PATRONI_ETCD_HOST: patroni_etcd:2379
|
||||
PATRONI_NAME: dbnode1
|
||||
PATRONI_SCOPE: testcluster
|
||||
services:
|
||||
etcd1: &etcd
|
||||
image: patroni
|
||||
networks: [ demo ]
|
||||
environment:
|
||||
ETCD_LISTEN_PEER_URLS: http://0.0.0.0:2380
|
||||
ETCD_LISTEN_CLIENT_URLS: http://0.0.0.0:2379
|
||||
ETCD_INITIAL_CLUSTER: etcd1=http://etcd1:2380,etcd2=http://etcd2:2380,etcd3=http://etcd3:2380
|
||||
ETCD_INITIAL_CLUSTER_STATE: new
|
||||
ETCD_INITIAL_CLUSTER_TOKEN: tutorial
|
||||
ETCD_UNSUPPORTED_ARCH: arm64
|
||||
container_name: demo-etcd1
|
||||
hostname: etcd1
|
||||
command: etcd -name etcd1 -initial-advertise-peer-urls http://etcd1:2380
|
||||
|
||||
dbnode2:
|
||||
image: patroni
|
||||
hostname: dbnode2
|
||||
links:
|
||||
- patroni_etcd:patroni_etcd
|
||||
volumes:
|
||||
- ./patroni:/patroni
|
||||
env_file: docker/patroni-secrets.env
|
||||
environment:
|
||||
PATRONI_ETCD_HOST: patroni_etcd:2379
|
||||
PATRONI_NAME: dbnode2
|
||||
PATRONI_SCOPE: testcluster
|
||||
etcd2:
|
||||
<<: *etcd
|
||||
container_name: demo-etcd2
|
||||
hostname: etcd2
|
||||
command: etcd -name etcd2 -initial-advertise-peer-urls http://etcd2:2380
|
||||
|
||||
dbnode3:
|
||||
image: patroni
|
||||
hostname: dbnode3
|
||||
links:
|
||||
- patroni_etcd:patroni_etcd
|
||||
volumes:
|
||||
- ./patroni:/patroni
|
||||
env_file: docker/patroni-secrets.env
|
||||
environment:
|
||||
PATRONI_ETCD_HOST: patroni_etcd:2379
|
||||
PATRONI_NAME: dbnode3
|
||||
PATRONI_SCOPE: testcluster
|
||||
etcd3:
|
||||
<<: *etcd
|
||||
container_name: demo-etcd3
|
||||
hostname: etcd3
|
||||
command: etcd -name etcd3 -initial-advertise-peer-urls http://etcd3:2380
|
||||
|
||||
haproxy:
|
||||
image: patroni
|
||||
links:
|
||||
- patroni_etcd:patroni_etcd
|
||||
ports:
|
||||
- "5000"
|
||||
- "5001"
|
||||
environment:
|
||||
PATRONI_ETCD_HOST: patroni_etcd:2379
|
||||
PATRONI_SCOPE: testcluster
|
||||
command: --confd
|
||||
haproxy:
|
||||
image: patroni
|
||||
networks: [ demo ]
|
||||
env_file: docker/patroni.env
|
||||
hostname: haproxy
|
||||
container_name: demo-haproxy
|
||||
ports:
|
||||
- "5000:5000"
|
||||
- "5001:5001"
|
||||
command: haproxy
|
||||
environment: &haproxy_env
|
||||
ETCDCTL_ENDPOINTS: http://etcd1:2379,http://etcd2:2379,http://etcd3:2379
|
||||
PATRONI_ETCD3_HOSTS: "'etcd1:2379','etcd2:2379','etcd3:2379'"
|
||||
PATRONI_SCOPE: demo
|
||||
|
||||
patroni1:
|
||||
image: patroni
|
||||
networks: [ demo ]
|
||||
env_file: docker/patroni.env
|
||||
hostname: patroni1
|
||||
container_name: demo-patroni1
|
||||
environment:
|
||||
<<: *haproxy_env
|
||||
PATRONI_NAME: patroni1
|
||||
|
||||
patroni2:
|
||||
image: patroni
|
||||
networks: [ demo ]
|
||||
env_file: docker/patroni.env
|
||||
hostname: patroni2
|
||||
container_name: demo-patroni2
|
||||
environment:
|
||||
<<: *haproxy_env
|
||||
PATRONI_NAME: patroni2
|
||||
|
||||
patroni3:
|
||||
image: patroni
|
||||
networks: [ demo ]
|
||||
env_file: docker/patroni.env
|
||||
hostname: patroni3
|
||||
container_name: demo-patroni3
|
||||
environment:
|
||||
<<: *haproxy_env
|
||||
PATRONI_NAME: patroni3
|
||||
|
||||
+103
-33
@@ -1,47 +1,117 @@
|
||||
# Patroni Dockerfile
|
||||
You can run Patroni in a docker container using this Dockerfile, or by using one of the Docker image at
|
||||
|
||||
https://registry.opensource.zalan.do/v1/repositories/acid/patroni/tags
|
||||
You can run Patroni in a docker container using this Dockerfile
|
||||
|
||||
This Dockerfile is meant in aiding development of Patroni and quick testing of features. It is not a production-worthy
|
||||
Dockerfile
|
||||
|
||||
docker build -t patroni .
|
||||
|
||||
# Examples
|
||||
|
||||
## Standalone Patroni
|
||||
|
||||
docker run -d registry.opensource.zalan.do/acid/patroni:1.0-SNAPSHOT
|
||||
docker run -d patroni
|
||||
|
||||
## Multiple Patroni's communicating with a standalone etcd inside Docker
|
||||
|
||||
Basically what you would do would be:
|
||||
|
||||
* Run 1 container which provides etcd
|
||||
|
||||
docker run -d <IMAGE> --etcd-only
|
||||
|
||||
* Run n containers running Patroni, passing the `--etcd` option to the `docker run` command
|
||||
|
||||
docker run -d <IMAGE> --etcd=<IP FROM etcd CONTAINER:PORT>
|
||||
|
||||
To automate this you can run the following script:
|
||||
|
||||
dev_patroni_cluster.sh [OPTIONS]
|
||||
|
||||
Options:
|
||||
|
||||
--image IMAGE The Docker image to use for the cluster
|
||||
--members INT The number of members for the cluster
|
||||
--name NAME The name of the new cluster
|
||||
## Three-node Patroni cluster with three-node etcd cluster and one haproxy container using docker-compose
|
||||
|
||||
Example session:
|
||||
|
||||
$ ./dev_patroni_cluster.sh --image registry.opensource.zalan.do/acid/patroni:1.0-SNAPSHOT --members=2 --name=bravo
|
||||
The etcd container is 6be871a11cb373406ca5ea1c6b39e1.0-SNAPSHOTfdde9fb1d6177212d6ad0c0d1bd9b563, ip=172.17.1.24
|
||||
Started Patroni container 67e611f2eca7c40f9e6e0e24a4a8f2cba7e3e56d22a420e15ab9240a37a9d7a4, ip=172.17.1.25
|
||||
Started Patroni container 47dd12ae635ab83b039f5889e250048b606ed5e48e3650b69e365e7e1d4acbcf, ip=172.17.1.26
|
||||
$ docker-compose up -d
|
||||
Creating demo-haproxy ...
|
||||
Creating demo-patroni2 ...
|
||||
Creating demo-patroni1 ...
|
||||
Creating demo-patroni3 ...
|
||||
Creating demo-etcd2 ...
|
||||
Creating demo-etcd1 ...
|
||||
Creating demo-etcd3 ...
|
||||
Creating demo-haproxy
|
||||
Creating demo-patroni2
|
||||
Creating demo-patroni1
|
||||
Creating demo-patroni3
|
||||
Creating demo-etcd1
|
||||
Creating demo-etcd2
|
||||
Creating demo-etcd2 ... done
|
||||
|
||||
$ docker ps
|
||||
CONTAINER ID IMAGE COMMAND CREATED STATUS PORTS NAMES
|
||||
47dd12ae635a registry.opensource.zalan.do/acid/patroni:1.0-SNAPSHOT "/bin/bash /entrypoi 10 seconds ago Up 8 seconds 4001/tcp, 5432/tcp, 2380/tcp bravo_OR64g8bx
|
||||
67e611f2eca7 registry.opensource.zalan.do/acid/patroni:1.0-SNAPSHOT "/bin/bash /entrypoi 11 seconds ago Up 10 seconds 2380/tcp, 4001/tcp, 5432/tcp bravo_si9no8iz
|
||||
6be871a11cb3 registry.opensource.zalan.do/acid/patroni:1.0-SNAPSHOT "/bin/bash /entrypoi 12 seconds ago Up 10 seconds 4001/tcp, 5432/tcp, 2380/tcp bravo_etcd
|
||||
CONTAINER ID IMAGE COMMAND CREATED STATUS PORTS NAMES
|
||||
5b7a90b4cfbf patroni "/bin/sh /entrypoint…" 29 seconds ago Up 27 seconds demo-etcd2
|
||||
e30eea5222f2 patroni "/bin/sh /entrypoint…" 29 seconds ago Up 27 seconds demo-etcd1
|
||||
83bcf3cb208f patroni "/bin/sh /entrypoint…" 29 seconds ago Up 27 seconds demo-etcd3
|
||||
922532c56e7d patroni "/bin/sh /entrypoint…" 29 seconds ago Up 28 seconds demo-patroni3
|
||||
14f875e445f3 patroni "/bin/sh /entrypoint…" 29 seconds ago Up 28 seconds demo-patroni2
|
||||
110d1073b383 patroni "/bin/sh /entrypoint…" 29 seconds ago Up 28 seconds demo-patroni1
|
||||
5af5e6e36028 patroni "/bin/sh /entrypoint…" 29 seconds ago Up 28 seconds 0.0.0.0:5000-5001->5000-5001/tcp demo-haproxy
|
||||
|
||||
$ docker logs demo-patroni1
|
||||
2019-02-20 08:19:32,714 INFO: Failed to import patroni.dcs.consul
|
||||
2019-02-20 08:19:32,737 INFO: Selected new etcd server http://etcd3:2379
|
||||
2019-02-20 08:19:35,140 INFO: Lock owner: None; I am patroni1
|
||||
2019-02-20 08:19:35,174 INFO: trying to bootstrap a new cluster
|
||||
...
|
||||
2019-02-20 08:19:39,310 INFO: postmaster pid=37
|
||||
2019-02-20 08:19:39.314 UTC [37] LOG: listening on IPv4 address "0.0.0.0", port 5432
|
||||
2019-02-20 08:19:39.321 UTC [37] LOG: listening on Unix socket "/var/run/postgresql/.s.PGSQL.5432"
|
||||
2019-02-20 08:19:39.353 UTC [39] LOG: database system was shut down at 2019-02-20 08:19:36 UTC
|
||||
2019-02-20 08:19:39.354 UTC [40] FATAL: the database system is starting up
|
||||
localhost:5432 - rejecting connections
|
||||
2019-02-20 08:19:39.369 UTC [37] LOG: database system is ready to accept connections
|
||||
localhost:5432 - accepting connections
|
||||
2019-02-20 08:19:39,383 INFO: establishing a new patroni connection to the postgres cluster
|
||||
2019-02-20 08:19:39,408 INFO: running post_bootstrap
|
||||
2019-02-20 08:19:39,432 WARNING: Could not activate Linux watchdog device: "Can't open watchdog device: [Errno 2] No such file or directory: '/dev/watchdog'"
|
||||
2019-02-20 08:19:39,515 INFO: initialized a new cluster
|
||||
2019-02-20 08:19:49,424 INFO: Lock owner: patroni1; I am patroni1
|
||||
2019-02-20 08:19:49,447 INFO: Lock owner: patroni1; I am patroni1
|
||||
2019-02-20 08:19:49,480 INFO: no action. i am the leader with the lock
|
||||
2019-02-20 08:19:59,422 INFO: Lock owner: patroni1; I am patroni1
|
||||
|
||||
$ docker exec -ti demo-patroni1 bash
|
||||
postgres@patroni1:~$ patronictl list
|
||||
+---------+----------+------------+--------+---------+----+-----------+
|
||||
| Cluster | Member | Host | Role | State | TL | Lag in MB |
|
||||
+---------+----------+------------+--------+---------+----+-----------+
|
||||
| demo | patroni1 | 172.22.0.3 | Leader | running | 1 | 0 |
|
||||
| demo | patroni2 | 172.22.0.7 | | running | 1 | 0 |
|
||||
| demo | patroni3 | 172.22.0.4 | | running | 1 | 0 |
|
||||
+---------+----------+------------+--------+---------+----+-----------+
|
||||
|
||||
postgres@patroni1:~$ etcdctl ls --recursive --sort -p /service/demo
|
||||
/service/demo/config
|
||||
/service/demo/initialize
|
||||
/service/demo/leader
|
||||
/service/demo/members/
|
||||
/service/demo/members/patroni1
|
||||
/service/demo/members/patroni2
|
||||
/service/demo/members/patroni3
|
||||
/service/demo/optime/
|
||||
/service/demo/optime/leader
|
||||
|
||||
postgres@patroni1:~$ etcdctl member list
|
||||
1bab629f01fa9065: name=etcd3 peerURLs=http://etcd3:2380 clientURLs=http://etcd3:2379 isLeader=false
|
||||
8ecb6af518d241cc: name=etcd2 peerURLs=http://etcd2:2380 clientURLs=http://etcd2:2379 isLeader=true
|
||||
b2e169fcb8a34028: name=etcd1 peerURLs=http://etcd1:2380 clientURLs=http://etcd1:2379 isLeader=false
|
||||
postgres@patroni1:~$ exit
|
||||
|
||||
$ psql -h localhost -p 5000 -U postgres -W
|
||||
Password: postgres
|
||||
psql (11.2 (Ubuntu 11.2-1.pgdg18.04+1), server 10.7 (Debian 10.7-1.pgdg90+1))
|
||||
Type "help" for help.
|
||||
|
||||
localhost/postgres=# select pg_is_in_recovery();
|
||||
pg_is_in_recovery
|
||||
───────────────────
|
||||
f
|
||||
(1 row)
|
||||
|
||||
localhost/postgres=# \q
|
||||
|
||||
$ psql -h localhost -p 5001 -U postgres -W
|
||||
Password: postgres
|
||||
psql (11.2 (Ubuntu 11.2-1.pgdg18.04+1), server 10.7 (Debian 10.7-1.pgdg90+1))
|
||||
Type "help" for help.
|
||||
|
||||
localhost/postgres=# select pg_is_in_recovery();
|
||||
pg_is_in_recovery
|
||||
───────────────────
|
||||
t
|
||||
(1 row)
|
||||
|
||||
@@ -1,104 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
DOCKER_IMAGE="registry.opensource.zalan.do/acid/patroni:1.0-SNAPSHOT"
|
||||
MEMBERS=3
|
||||
|
||||
|
||||
function usage()
|
||||
{
|
||||
cat <<__EOF__
|
||||
Usage: $0
|
||||
|
||||
Options:
|
||||
|
||||
--image IMAGE The Docker image to use for the cluster
|
||||
--members INT The number of members for the cluster
|
||||
--name NAME The name of the new cluster
|
||||
|
||||
Examples:
|
||||
|
||||
$0 --image ${DOCKER_IMAGE}
|
||||
$0
|
||||
$0 --image ${DOCKER_IMAGE} --members=2
|
||||
__EOF__
|
||||
}
|
||||
|
||||
|
||||
optspec=":-:"
|
||||
while getopts "$optspec" optchar; do
|
||||
case "${optchar}" in
|
||||
-)
|
||||
case "${OPTARG}" in
|
||||
help)
|
||||
usage
|
||||
exit 0
|
||||
;;
|
||||
name)
|
||||
PATRONI_SCOPE="${!OPTIND}"; OPTIND=$(( $OPTIND + 1 ))
|
||||
;;
|
||||
name=*)
|
||||
PATRONI_SCOPE="${OPTARG#*=}"
|
||||
;;
|
||||
image)
|
||||
DOCKER_IMAGE="${!OPTIND}"; OPTIND=$(( $OPTIND + 1 ))
|
||||
;;
|
||||
image=*)
|
||||
DOCKER_IMAGE="${OPTARG#*=}"
|
||||
;;
|
||||
members)
|
||||
MEMBERS="${!OPTIND}"; OPTIND=$(( $OPTIND + 1 ))
|
||||
;;
|
||||
members=*)
|
||||
MEMBERS="${OPTARG#*=}"
|
||||
;;
|
||||
*)
|
||||
if [ "$OPTERR" = 1 ] && [ "${optspec:0:1}" != ":" ]; then
|
||||
echo "Unknown option --${OPTARG}" >&2
|
||||
fi
|
||||
;;
|
||||
esac;;
|
||||
*)
|
||||
if [ "$OPTERR" != 1 ] || [ "${optspec:0:1}" = ":" ]; then
|
||||
echo "Non-option argument: '-${OPTARG}'" >&2
|
||||
usage
|
||||
exit 1
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
if [ -z ${PATRONI_SCOPE} ]; then
|
||||
PATRONI_SCOPE=$(cat /dev/urandom | LC_ALL=C tr -dc 'a-z0-9' | head -c 8)
|
||||
fi
|
||||
|
||||
function docker_run()
|
||||
{
|
||||
local name=$1
|
||||
shift
|
||||
container=$(docker run -d --name=$name $*)
|
||||
container_ip=$(docker inspect --format '{{ .NetworkSettings.IPAddress }}' ${container})
|
||||
echo "Started container ${name}, ip=${container_ip}"
|
||||
}
|
||||
|
||||
|
||||
ETCD_CONTAINER="${PATRONI_SCOPE}_etcd"
|
||||
docker_run ${ETCD_CONTAINER} ${DOCKER_IMAGE} --etcd
|
||||
|
||||
DOCKER_ARGS="--link=${ETCD_CONTAINER}:${ETCD_CONTAINER} -e PATRONI_SCOPE=${PATRONI_SCOPE} -e PATRONI_ETCD_HOST=${ETCD_CONTAINER}:2379"
|
||||
PATRONI_ENV=$(sed 's/#.*//g' docker/patroni-secrets.env | sed -n 's/^PATRONI_.*$/-e &/p' | tr '\n' ' ')
|
||||
PATRONI_VOLUME="-v $(dirname $(dirname $(realpath $0)))/patroni:/patroni"
|
||||
|
||||
for i in $(seq 1 "${MEMBERS}"); do
|
||||
container_name=postgres${i}
|
||||
docker_run "${PATRONI_SCOPE}_${container_name}" \
|
||||
$PATRONI_VOLUME \
|
||||
$DOCKER_ARGS \
|
||||
$PATRONI_ENV \
|
||||
-e PATRONI_NAME=${container_name} \
|
||||
${DOCKER_IMAGE}
|
||||
done
|
||||
|
||||
docker_run "${PATRONI_SCOPE}_haproxy" \
|
||||
-p=5000 -p=5001 \
|
||||
$DOCKER_ARGS \
|
||||
${DOCKER_IMAGE} --confd
|
||||
+45
-100
@@ -1,115 +1,60 @@
|
||||
#!/bin/bash
|
||||
#!/bin/sh
|
||||
|
||||
function usage()
|
||||
{
|
||||
cat <<__EOF__
|
||||
Usage: $0
|
||||
if [ -f /a.tar.xz ]; then
|
||||
echo "decompressing image..."
|
||||
sudo tar xpJf /a.tar.xz -C / > /dev/null 2>&1
|
||||
sudo rm /a.tar.xz
|
||||
sudo ln -snf dash /bin/sh
|
||||
fi
|
||||
|
||||
Options:
|
||||
readonly PATRONI_SCOPE=${PATRONI_SCOPE:-batman}
|
||||
PATRONI_NAMESPACE=${PATRONI_NAMESPACE:-/service}
|
||||
readonly PATRONI_NAMESPACE=${PATRONI_NAMESPACE%/}
|
||||
readonly DOCKER_IP=$(hostname --ip-address)
|
||||
|
||||
--etcd Do not run Patroni, run a standalone etcd
|
||||
--confd Do not run Patroni, run a standalone confd
|
||||
--zookeeper Do not run Patroni, run a standalone zookeeper
|
||||
|
||||
Examples:
|
||||
|
||||
$0 --etcd
|
||||
$0 --confd
|
||||
$0 --zookeeper
|
||||
$0
|
||||
__EOF__
|
||||
}
|
||||
|
||||
DOCKER_IP=$(hostname --ip-address)
|
||||
PATRONI_SCOPE=${PATRONI_SCOPE:-batman}
|
||||
ETCD_ARGS="--data-dir /tmp/etcd.data -advertise-client-urls=http://${DOCKER_IP}:2379 -listen-client-urls=http://0.0.0.0:2379 -listen-peer-urls=http://0.0.0.0:2380"
|
||||
|
||||
optspec=":vh-:"
|
||||
while getopts "$optspec" optchar; do
|
||||
case "${optchar}" in
|
||||
-)
|
||||
case "${OPTARG}" in
|
||||
confd)
|
||||
haproxy -f /etc/haproxy/haproxy.cfg -p /var/run/haproxy.pid -D
|
||||
CONFD="confd -prefix=${PATRONI_NAMESPACE:-/service}/$PATRONI_SCOPE -interval=10 -backend"
|
||||
if [ ! -z ${PATRONI_ZOOKEEPER_HOSTS} ]; then
|
||||
while ! /usr/share/zookeeper/bin/zkCli.sh -server ${PATRONI_ZOOKEEPER_HOSTS} ls /; do
|
||||
sleep 1
|
||||
done
|
||||
exec $CONFD zookeeper -node ${PATRONI_ZOOKEEPER_HOSTS}
|
||||
else
|
||||
while ! curl -s ${PATRONI_ETCD_HOST}/v2/members | jq -r '.members[0].clientURLs[0]' | grep -q http; do
|
||||
sleep 1
|
||||
done
|
||||
exec $CONFD etcd -node $PATRONI_ETCD_HOST
|
||||
fi
|
||||
;;
|
||||
etcd)
|
||||
exec etcd $ETCD_ARGS
|
||||
;;
|
||||
zookeeper)
|
||||
exec /usr/share/zookeeper/bin/zkServer.sh start-foreground
|
||||
;;
|
||||
cheat)
|
||||
CHEAT=1
|
||||
;;
|
||||
help)
|
||||
usage
|
||||
exit 0
|
||||
;;
|
||||
*)
|
||||
if [ "$OPTERR" = 1 ] && [ "${optspec:0:1}" != ":" ]; then
|
||||
echo "Unknown option --${OPTARG}" >&2
|
||||
fi
|
||||
;;
|
||||
esac;;
|
||||
*)
|
||||
if [ "$OPTERR" != 1 ] || [ "${optspec:0:1}" = ":" ]; then
|
||||
echo "Non-option argument: '-${OPTARG}'" >&2
|
||||
usage
|
||||
exit 1
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
done
|
||||
case "$1" in
|
||||
haproxy)
|
||||
haproxy -f /etc/haproxy/haproxy.cfg -p /var/run/haproxy.pid -D
|
||||
CONFD="confd -prefix=$PATRONI_NAMESPACE/$PATRONI_SCOPE -interval=10 -backend"
|
||||
if [ ! -z "$PATRONI_ZOOKEEPER_HOSTS" ]; then
|
||||
while ! /usr/share/zookeeper/bin/zkCli.sh -server $PATRONI_ZOOKEEPER_HOSTS ls /; do
|
||||
sleep 1
|
||||
done
|
||||
exec dumb-init $CONFD zookeeper -node $PATRONI_ZOOKEEPER_HOSTS
|
||||
else
|
||||
while ! etcdctl cluster-health 2> /dev/null; do
|
||||
sleep 1
|
||||
done
|
||||
exec dumb-init $CONFD etcdv3 -node $(echo $ETCDCTL_ENDPOINTS | sed 's/,/ -node /g')
|
||||
fi
|
||||
;;
|
||||
etcd)
|
||||
exec "$@" -advertise-client-urls http://$DOCKER_IP:2379
|
||||
;;
|
||||
zookeeper)
|
||||
exec /usr/share/zookeeper/bin/zkServer.sh start-foreground
|
||||
;;
|
||||
esac
|
||||
|
||||
## We start an etcd
|
||||
if [[ -z ${PATRONI_ETCD_HOST} && -z ${PATRONI_ZOOKEEPER_HOSTS} ]]; then
|
||||
etcd $ETCD_ARGS > /var/log/etcd.log 2> /var/log/etcd.err &
|
||||
export PATRONI_ETCD_HOST="127.0.0.1:2379"
|
||||
if [ -z "$PATRONI_ETCD3_HOSTS" ] && [ -z "$PATRONI_ZOOKEEPER_HOSTS" ]; then
|
||||
export PATRONI_ETCD_URL="http://127.0.0.1:2379"
|
||||
etcd --data-dir /tmp/etcd.data -advertise-client-urls=$PATRONI_ETCD_URL -listen-client-urls=http://0.0.0.0:2379 > /var/log/etcd.log 2> /var/log/etcd.err &
|
||||
fi
|
||||
|
||||
export PATRONI_SCOPE
|
||||
export PATRONI_NAME="${PATRONI_NAME:-${HOSTNAME}}"
|
||||
export PATRONI_RESTAPI_CONNECT_ADDRESS="${DOCKER_IP}:8008"
|
||||
export PATRONI_NAMESPACE
|
||||
export PATRONI_NAME="${PATRONI_NAME:-$(hostname)}"
|
||||
export PATRONI_RESTAPI_CONNECT_ADDRESS="$DOCKER_IP:8008"
|
||||
export PATRONI_RESTAPI_LISTEN="0.0.0.0:8008"
|
||||
export PATRONI_admin_PASSWORD="${PATRONI_admin_PASSWORD:=admin}"
|
||||
export PATRONI_admin_PASSWORD="${PATRONI_admin_PASSWORD:-admin}"
|
||||
export PATRONI_admin_OPTIONS="${PATRONI_admin_OPTIONS:-createdb, createrole}"
|
||||
export PATRONI_POSTGRESQL_CONNECT_ADDRESS="${DOCKER_IP}:5432"
|
||||
export PATRONI_POSTGRESQL_CONNECT_ADDRESS="$DOCKER_IP:5432"
|
||||
export PATRONI_POSTGRESQL_LISTEN="0.0.0.0:5432"
|
||||
export PATRONI_POSTGRESQL_DATA_DIR="data/${PATRONI_SCOPE}"
|
||||
export PATRONI_POSTGRESQL_DATA_DIR="${PATRONI_POSTGRESQL_DATA_DIR:-$PGDATA}"
|
||||
export PATRONI_REPLICATION_USERNAME="${PATRONI_REPLICATION_USERNAME:-replicator}"
|
||||
export PATRONI_REPLICATION_PASSWORD="${PATRONI_REPLICATION_PASSWORD:-abcd}"
|
||||
export PATRONI_REPLICATION_PASSWORD="${PATRONI_REPLICATION_PASSWORD:-replicate}"
|
||||
export PATRONI_SUPERUSER_USERNAME="${PATRONI_SUPERUSER_USERNAME:-postgres}"
|
||||
export PATRONI_SUPERUSER_PASSWORD="${PATRONI_SUPERUSER_PASSWORD:-postgres}"
|
||||
export PATRONI_POSTGRESQL_PGPASS="$HOME/.pgpass"
|
||||
|
||||
cat > /patroni.yml <<__EOF__
|
||||
bootstrap:
|
||||
dcs:
|
||||
postgresql:
|
||||
use_pg_rewind: true
|
||||
|
||||
pg_hba:
|
||||
- host all all 0.0.0.0/0 md5
|
||||
- host replication replicator ${DOCKER_IP}/16 md5
|
||||
__EOF__
|
||||
|
||||
mkdir -p "$HOME/.config/patroni"
|
||||
[ -h "$HOME/.config/patroni/patronictl.yaml" ] || ln -s /patroni.yml "$HOME/.config/patroni/patronictl.yaml"
|
||||
|
||||
[ -z $CHEAT ] && exec python /patroni.py /patroni.yml
|
||||
|
||||
while true; do
|
||||
sleep 60
|
||||
done
|
||||
exec python3 /patroni.py postgres0.yml
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
.. _contributing:
|
||||
|
||||
Contributing guidelines
|
||||
=======================
|
||||
|
||||
Wanna contribute to Patroni? Yay - here is how!
|
||||
|
||||
Chatting
|
||||
--------
|
||||
|
||||
Just want to chat with other Patroni users? Looking for interactive troubleshooting help? Join us on channel #patroni in the `PostgreSQL Slack <https://postgres-slack.herokuapp.com/>`__.
|
||||
|
||||
Running tests
|
||||
-------------
|
||||
|
||||
Requirements for running behave tests:
|
||||
|
||||
1. PostgreSQL packages need to be installed.
|
||||
2. PostgreSQL binaries must be available in your `PATH`. You may need to add them to the path with something like `PATH=/usr/lib/postgresql/11/bin:$PATH python -m behave`.
|
||||
3. If you'd like to test with external DCSs (e.g., Etcd, Consul, and Zookeeper) you'll need the packages installed and respective services running and accepting unencrypted/unprotected connections on localhost and default port. In the case of Etcd or Consul, the behave test suite could start them up if binaries are available in the `PATH`.
|
||||
|
||||
Install dependencies:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# You may want to use Virtualenv or specify pip3.
|
||||
pip install -r requirements.txt
|
||||
pip install -r requirements.dev.txt
|
||||
|
||||
After you have all dependencies installed, you can run the various test suites:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# You may want to use Virtualenv or specify python3.
|
||||
|
||||
# Run flake8 to check syntax and formatting:
|
||||
python setup.py flake8
|
||||
|
||||
# Run the pytest suite in tests/:
|
||||
python setup.py test
|
||||
|
||||
# Run the behave (https://behave.readthedocs.io/en/latest/) test suite in features/;
|
||||
# modify DCS as desired (raft has no dependencies so is the easiest to start with):
|
||||
DCS=raft python -m behave
|
||||
|
||||
Reporting issues
|
||||
----------------
|
||||
|
||||
If you have a question about patroni or have a problem using it, please read the :ref:`README <readme>` before filing an issue.
|
||||
Also double check with the current issues on our `Issues Tracker <https://github.com/zalando/patroni/issues>`__.
|
||||
|
||||
Contributing a pull request
|
||||
---------------------------
|
||||
|
||||
1) Submit a comment to the relevant issue or create a new issue describing your proposed change.
|
||||
2) Do a fork, develop and test your code changes.
|
||||
3) Include documentation
|
||||
4) Submit a pull request.
|
||||
|
||||
You'll get feedback about your pull request as soon as possible.
|
||||
|
||||
Happy Patroni hacking ;-)
|
||||
+135
-6
@@ -1,4 +1,5 @@
|
||||
==================================
|
||||
.. _environment:
|
||||
|
||||
Environment Configuration Settings
|
||||
==================================
|
||||
|
||||
@@ -11,6 +12,18 @@ Global/Universal
|
||||
- **PATRONI\_NAMESPACE**: path within the configuration store where Patroni will keep information about the cluster. Default value: "/service"
|
||||
- **PATRONI\_SCOPE**: cluster name
|
||||
|
||||
Log
|
||||
---
|
||||
- **PATRONI\_LOG\_LEVEL**: sets the general logging level. Default value is **INFO** (see `the docs for Python logging <https://docs.python.org/3.6/library/logging.html#levels>`_)
|
||||
- **PATRONI\_LOG\_TRACEBACK\_LEVEL**: sets the level where tracebacks will be visible. Default value is **ERROR**. Set it to **DEBUG** if you want to see tracebacks only if you enable **PATRONI\_LOG\_LEVEL=DEBUG**.
|
||||
- **PATRONI\_LOG\_FORMAT**: sets the log formatting string. Default value is **%(asctime)s %(levelname)s: %(message)s** (see `the LogRecord attributes <https://docs.python.org/3.6/library/logging.html#logrecord-attributes>`_)
|
||||
- **PATRONI\_LOG\_DATEFORMAT**: sets the datetime formatting string. (see the `formatTime() documentation <https://docs.python.org/3.6/library/logging.html#logging.Formatter.formatTime>`_)
|
||||
- **PATRONI\_LOG\_MAX\_QUEUE\_SIZE**: Patroni is using two-step logging. Log records are written into the in-memory queue and there is a separate thread which pulls them from the queue and writes to stderr or file. The maximum size of the internal queue is limited by default by **1000** records, which is enough to keep logs for the past 1h20m.
|
||||
- **PATRONI\_LOG\_DIR**: Directory to write application logs to. The directory must exist and be writable by the user executing Patroni. If you set this env variable, the application will retain 4 25MB logs by default. You can tune those retention values with `PATRONI_LOG_FILE_NUM` and `PATRONI_LOG_FILE_SIZE` (see below).
|
||||
- **PATRONI\_LOG\_FILE\_NUM**: The number of application logs to retain.
|
||||
- **PATRONI\_LOG\_FILE\_SIZE**: Size of patroni.log file (in bytes) that triggers a log rolling.
|
||||
- **PATRONI\_LOG\_LOGGERS**: Redefine logging level per python module. Example ``PATRONI_LOG_LOGGERS="{patroni.postmaster: WARNING, urllib3: DEBUG}"``
|
||||
|
||||
Bootstrap configuration
|
||||
-----------------------
|
||||
It is possible to create new database users right after the successful initialization of a new cluster. This process is defined by the following variables:
|
||||
@@ -22,37 +35,153 @@ Example: defining ``PATRONI_admin_PASSWORD=strongpasswd`` and ``PATRONI_admin_OP
|
||||
|
||||
Consul
|
||||
------
|
||||
- **PATRONI\_CONSUL\_HOST**: the host:port for the Consul endpoint.
|
||||
- **PATRONI\_CONSUL\_HOST**: the host:port for the Consul local agent.
|
||||
- **PATRONI\_CONSUL\_URL**: url for the Consul local agent, in format: http(s)://host:port
|
||||
- **PATRONI\_CONSUL\_PORT**: (optional) Consul port
|
||||
- **PATRONI\_CONSUL\_SCHEME**: (optional) **http** or **https**, defaults to **http**
|
||||
- **PATRONI\_CONSUL\_TOKEN**: (optional) ACL token
|
||||
- **PATRONI\_CONSUL\_VERIFY**: (optional) whether to verify the SSL certificate for HTTPS requests
|
||||
- **PATRONI\_CONSUL\_CACERT**: (optional) The ca certificate. If present it will enable validation.
|
||||
- **PATRONI\_CONSUL\_CERT**: (optional) File with the client certificate
|
||||
- **PATRONI\_CONSUL\_KEY**: (optional) File with the client key. Can be empty if the key is part of certificate.
|
||||
- **PATRONI\_CONSUL\_DC**: (optional) Datacenter to communicate with. By default the datacenter of the host is used.
|
||||
- **PATRONI\_CONSUL\_CONSISTENCY**: (optional) Select consul consistency mode. Possible values are ``default``, ``consistent``, or ``stale`` (more details in `consul API reference <https://www.consul.io/api/features/consistency.html/>`__)
|
||||
- **PATRONI\_CONSUL\_CHECKS**: (optional) list of Consul health checks used for the session. By default an empty list is used.
|
||||
- **PATRONI\_CONSUL\_REGISTER\_SERVICE**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, replica or standby-leader depending on the node's role. Defaults to **false**
|
||||
- **PATRONI\_CONSUL\_SERVICE\_CHECK\_INTERVAL**: (optional) how often to perform health check against registered url
|
||||
- **PATRONI\_CONSUL\_SERVICE\_CHECK\_TLS\_SERVER\_NAME**: (optional) overide SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
|
||||
|
||||
Etcd
|
||||
----
|
||||
|
||||
- **PATRONI\_ETCD\_PROXY**: proxy url for the etcd. If you are connecting to the etcd using proxy, use this parameter instead of **PATRONI\_ETCD\_URL**
|
||||
- **PATRONI\_ETCD\_URL**: url for the etcd, in format: http(s)://(username:password@)host:port
|
||||
- **PATRONI\_ETCD\_HOSTS**: list of etcd endpoints in format 'host1:port1','host2:port2',etc...
|
||||
- **PATRONI\_ETCD\_USE\_PROXIES**: If this parameter is set to true, Patroni will consider **hosts** as a list of proxies and will not perform a topology discovery of etcd cluster but stick to a fixed list of **hosts**.
|
||||
- **PATRONI\_ETCD\_PROTOCOL**: http or https, if not specified http is used. If the **url** or **proxy** is specified - will take protocol from them.
|
||||
- **PATRONI\_ETCD\_HOST**: the host:port for the etcd endpoint.
|
||||
- **PATRONI\_ETCD\_SRV**: Domain to search the SRV record(s) for cluster autodiscovery. Patroni will try to query these SRV service names for specified domain (in that order until first success): ``_etcd-client-ssl``, ``_etcd-client``, ``_etcd-ssl``, ``_etcd``, ``_etcd-server-ssl``, ``_etcd-server``. If SRV records for ``_etcd-server-ssl`` or ``_etcd-server`` are retrieved then ETCD peer protocol is used do query ETCD for available members. Otherwise hosts from SRV records will be used.
|
||||
- **PATRONI\_ETCD\_SRV\_SUFFIX**: Configures a suffix to the SRV name that is queried during discovery. Use this flag to differentiate between multiple etcd clusters under the same domain. Works only with conjunction with **PATRONI\_ETCD\_SRV**. For example, if ``PATRONI_ETCD_SRV_SUFFIX=foo`` and ``PATRONI_ETCD_SRV=example.org`` are set, the following DNS SRV query is made:``_etcd-client-ssl-foo._tcp.example.com`` (and so on for every possible ETCD SRV service name).
|
||||
- **PATRONI\_ETCD\_USERNAME**: username for etcd authentication.
|
||||
- **PATRONI\_ETCD\_PASSWORD**: password for etcd authentication.
|
||||
- **PATRONI\_ETCD\_CACERT**: The ca certificate. If present it will enable validation.
|
||||
- **PATRONI\_ETCD\_CERT**: File with the client certificate.
|
||||
- **PATRONI\_ETCD\_KEY**: File with the client key. Can be empty if the key is part of certificate.
|
||||
|
||||
Etcdv3
|
||||
------
|
||||
Environment names for Etcdv3 are similar as for Etcd, you just need to use ``ETCD3`` instead of ``ETCD`` in the variable name. Example: ``PATRONI_ETCD3_HOST``, ``PATRONI_ETCD3_CACERT``, and so on.
|
||||
|
||||
.. warning::
|
||||
Keys created with protocol version 2 are not visible with protocol version 3 and the other way around, therefore it is not possible to switch from Etcd to Etcdv3 just by updating Patroni configuration.
|
||||
|
||||
|
||||
ZooKeeper
|
||||
---------
|
||||
- **PATRONI\_ZOOKEEPER\_HOSTS**: Comma separated list of ZooKeeper cluster members: "'host1:port1','host2:port2','etc...'". It is important to quote every single entity!
|
||||
- **PATRONI\_ZOOKEEPER\_USE\_SSL**: (optional) Whether SSL is used or not. Defaults to ``false``. If set to ``false``, all SSL specific parameters are ignored.
|
||||
- **PATRONI\_ZOOKEEPER\_CACERT**: (optional) The CA certificate. If present it will enable validation.
|
||||
- **PATRONI\_ZOOKEEPER\_CERT**: (optional) File with the client certificate.
|
||||
- **PATRONI\_ZOOKEEPER\_KEY**: (optional) File with the client key.
|
||||
- **PATRONI\_ZOOKEEPER\_KEY\_PASSWORD**: (optional) The client key password.
|
||||
- **PATRONI\_ZOOKEEPER\_VERIFY**: (optional) Whether to verify certificate or not. Defaults to ``true``.
|
||||
- **PATRONI\_ZOOKEEPER\_SET\_ACLS**: (optional) If set, configure Kazoo to apply a default ACL to each ZNode that it creates. ACLs will assume 'x509' schema and should be specified as a dictionary with the principal as the key and one or more permissions as a list in the value. Permissions may be one of ``CREATE``, ``READ``, ``WRITE``, ``DELETE`` or ``ADMIN``. For example, ``set_acls: {CN=principal1: [CREATE, READ], CN=principal2: [ALL]}``.
|
||||
|
||||
.. note::
|
||||
It is required to install ``kazoo>=2.6.0`` to support SSL.
|
||||
|
||||
|
||||
Exhibitor
|
||||
---------
|
||||
- **PATRONI\_EXHIBITOR\_HOSTS**: initial list of Exhibitor (ZooKeeper) nodes in format: 'host1,host2,etc...'. This list updates automatically whenever the Exhibitor (ZooKeeper) cluster topology changes.
|
||||
- **PATRONI\_EXHIBITOR\_PORT**: Exhibitor port.
|
||||
|
||||
.. _kubernetes_environment:
|
||||
|
||||
Kubernetes
|
||||
----------
|
||||
- **PATRONI\_KUBERNETES\_BYPASS\_API\_SERVICE**: (optional) When communicating with the Kubernetes API, Patroni is usually relying on the `kubernetes` service, the address of which is exposed in the pods via the `KUBERNETES_SERVICE_HOST` environment variable. If `PATRONI_KUBERNETES_BYPASS_API_SERVICE` is set to ``true``, Patroni will resolve the list of API nodes behind the service and connect directly to them.
|
||||
- **PATRONI\_KUBERNETES\_NAMESPACE**: (optional) Kubernetes namespace where the Patroni pod is running. Default value is `default`.
|
||||
- **PATRONI\_KUBERNETES\_LABELS**: Labels in format ``{label1: value1, label2: value2}``. These labels will be used to find existing objects (Pods and either Endpoints or ConfigMaps) associated with the current cluster. Also Patroni will set them on every object (Endpoint or ConfigMap) it creates.
|
||||
- **PATRONI\_KUBERNETES\_SCOPE\_LABEL**: (optional) name of the label containing cluster name. Default value is `cluster-name`.
|
||||
- **PATRONI\_KUBERNETES\_ROLE\_LABEL**: (optional) name of the label containing Postgres role (`master` or `replica`). Patroni will set this label on the pod it is running in. Default value is `role`.
|
||||
- **PATRONI\_KUBERNETES\_USE\_ENDPOINTS**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||
- **PATRONI\_KUBERNETES\_POD\_IP**: (optional) IP address of the pod Patroni is running in. This value is required when `PATRONI_KUBERNETES_USE_ENDPOINTS` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||
- **PATRONI\_KUBERNETES\_PORTS**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``PATRONI_KUBERNETES_PORTS='[{"name": "postgresql", "port": 5432}]'`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `PATRONI_KUBERNETES_USE_ENDPOINTS` is set.
|
||||
- **PATRONI\_KUBERNETES\_CACERT**: (optional) Specifies the file with the CA_BUNDLE file with certificates of trusted CAs to use while verifying Kubernetes API SSL certs. If not provided, patroni will use the value provided by the ServiceAccount secret.
|
||||
|
||||
Raft
|
||||
----
|
||||
|
||||
- **PATRONI\_RAFT\_SELF\_ADDR**: ``ip:port`` to listen on for Raft connections. The ``self_addr`` must be accessible from other nodes of the cluster. If not set, the node will not participate in consensus.
|
||||
- **PATRONI\_RAFT\_BIND\_ADDR**: (optional) ``ip:port`` to listen on for Raft connections. If not specified the ``self_addr`` will be used.
|
||||
- **PATRONI\_RAFT\_PARTNER\_ADDRS**: list of other Patroni nodes in the cluster in format ``"'ip1:port1','ip2:port2'"``. It is important to quote every single entity!
|
||||
- **PATRONI\_RAFT\_DATA\_DIR**: directory where to store Raft log and snapshot. If not specified the current working directory is used.
|
||||
- **PATRONI\_RAFT\_PASSWORD**: (optional) Encrypt Raft traffic with a specified password, requires ``cryptography`` python module.
|
||||
|
||||
PostgreSQL
|
||||
----------
|
||||
- **PATRONI\_POSTGRESQL\_LISTEN**: IP address + port that Postgres listens to. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node.
|
||||
- **PATRONI\_POSTGRESQL\_CONNECT\_ADDRESS**: IP address + port through which Postgres is accessible from other nodes and applications.
|
||||
- **PATRONI\_POSTGRESQL\_DATA\_DIR**: The location of the Postgres data directory, either existing or to be initialized by Patroni.
|
||||
- **PATRONI\_POSTGRESQL\_CONFIG\_DIR**: The location of the Postgres configuration directory, defaults to the data directory. Must be writable by Patroni.
|
||||
- **PATRONI\_POSTGRESQL\_BIN_DIR**: Path to PostgreSQL binaries. (pg_ctl, pg_rewind, pg_basebackup, postgres) The default value is an empty string meaning that PATH environment variable will be used to find the executables.
|
||||
- **PATRONI\_POSTGRESQL\_PGPASS**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup and under some other circumstances. The location must be writable by Patroni.
|
||||
- **PATRONI\_REPLICATION\_USERNAME**: replication username; the user will be created during initialization. Replicas will use this user to access master via streaming replication
|
||||
- **PATRONI\_REPLICATION\_PASSWORD**: replication password; the user will be created during initialization.
|
||||
- **PATRONI\_REPLICATION\_SSLMODE**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
|
||||
- **PATRONI\_REPLICATION\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||
- **PATRONI\_REPLICATION\_SSLPASSWORD**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``PATRONI_REPLICATION_SSLKEY``.
|
||||
- **PATRONI\_REPLICATION\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **PATRONI\_REPLICATION\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **PATRONI\_REPLICATION\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_REPLICATION\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_REPLICATION\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **PATRONI\_REPLICATION\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
- **PATRONI\_SUPERUSER\_USERNAME**: name for the superuser, set during initialization (initdb) and later used by Patroni to connect to the postgres. Also this user is used by pg_rewind.
|
||||
- **PATRONI\_SUPERUSER\_PASSWORD**: password for the superuser, set during initialization (initdb).
|
||||
- **PATRONI\_SUPERUSER\_SSLMODE**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
|
||||
- **PATRONI\_SUPERUSER\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||
- **PATRONI\_SUPERUSER\_SSLPASSWORD**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``PATRONI_SUPERUSER_SSLKEY``.
|
||||
- **PATRONI\_SUPERUSER\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **PATRONI\_SUPERUSER\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **PATRONI\_SUPERUSER\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_SUPERUSER\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_SUPERUSER\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **PATRONI\_SUPERUSER\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
- **PATRONI\_REWIND\_USERNAME**: name for the user for ``pg_rewind``; the user will be created during initialization of postgres 11+ and all necessary `permissions <https://www.postgresql.org/docs/11/app-pgrewind.html#id-1.9.5.8.8>`__ will be granted.
|
||||
- **PATRONI\_REWIND\_PASSWORD**: password for the user for ``pg_rewind``; the user will be created during initialization.
|
||||
- **PATRONI\_REWIND\_SSLMODE**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
|
||||
- **PATRONI\_REWIND\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||
- **PATRONI\_REWIND\_SSLPASSWORD**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``PATRONI_REWIND_SSLKEY``.
|
||||
- **PATRONI\_REWIND\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **PATRONI\_REWIND\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **PATRONI\_REWIND\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_REWIND\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_REWIND\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **PATRONI\_REWIND\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
|
||||
REST API
|
||||
--------
|
||||
--------
|
||||
- **PATRONI\_RESTAPI\_CONNECT\_ADDRESS**: IP address and port to access the REST API.
|
||||
- **PATRONI\_RESTAPI\_LISTEN**: IP address and port that Patroni will listen to, to provide health-check information for HAProxy.
|
||||
- **PATRONI\_RESTAPI\_USERNAME**: Basic-auth username to protect unsafe REST API endpoints.
|
||||
- **PATRONI\_RESTAPI\_PASSWORD**: Basic-auth password to protect unsafe REST API endpoints.
|
||||
- **PATRONI\_RESTAPI\_CERTFILE**: Specifies the file with the certificate in the PEM format. If the certfile is not specified or is left empty, the API server will work without SSL.
|
||||
- **PATRONI\_RESTAPI\_KEYFILE**: Specifies the file with the secret key in the PEM format.
|
||||
- **PATRONI\_RESTAPI\_KEYFILE\_PASSWORD**: Specifies a password for decrypting the keyfile.
|
||||
- **PATRONI\_RESTAPI\_CAFILE**: Specifies the file with the CA_BUNDLE with certificates of trusted CAs to use while verifying client certs.
|
||||
- **PATRONI\_RESTAPI\_CIPHERS**: (optional) Specifies the permitted cipher suites (e.g. "ECDHE-RSA-AES256-GCM-SHA384:DHE-RSA-AES256-GCM-SHA384:ECDHE-RSA-AES128-GCM-SHA256:DHE-RSA-AES128-GCM-SHA256:!SSLv1:!SSLv2:!SSLv3:!TLSv1:!TLSv1.1")
|
||||
- **PATRONI\_RESTAPI\_VERIFY\_CLIENT**: ``none`` (default), ``optional`` or ``required``. When ``none`` REST API will not check client certificates. When ``required`` client certificates are required for all REST API calls. When ``optional`` client certificates are required for all unsafe REST API endpoints. When ``required`` is used, then client authentication succeeds, if the certificate signature verification succeeds. For ``optional`` the client cert will only be checked for ``PUT``, ``POST``, ``PATCH``, and ``DELETE`` requests.
|
||||
- **PATRONI\_RESTAPI\_ALLOWLIST**: (optional): Specifies the set of hosts that are allowed to call unsafe REST API endpoints. The single element could be a host name, an IP address or a network address using CIDR notation. By default ``allow all`` is used. In case if ``allowlist`` or ``allowlist_include_members`` are set, anything that is not included is rejected.
|
||||
- **PATRONI\_RESTAPI\_ALLOWLIST\_INCLUDE\_MEMBERS**: (optional): If set to ``true`` it allows accessing unsafe REST API endpoints from other cluster members registered in DCS (IP address or hostname is taken from the members ``api_url``). Be careful, it might happen that OS will use a different IP for outgoing connections.
|
||||
- **PATRONI\_RESTAPI\_HTTP\_EXTRA\_HEADERS**: (optional) HTTP headers let the REST API server pass additional information with an HTTP response.
|
||||
- **PATRONI\_RESTAPI\_HTTPS\_EXTRA\_HEADERS**: (optional) HTTPS headers let the REST API server pass additional information with an HTTP response when TLS is enabled. This will also pass additional information set in ``http_extra_headers``.
|
||||
|
||||
ZooKeeper
|
||||
---------
|
||||
- **PATRONI\_ZOOKEEPER\_HOSTS**: comma separated list of ZooKeeper cluster members: "'host1:port1','host2:port2','etc...'". It is important to quote every single entity!
|
||||
CTL
|
||||
---
|
||||
- **PATRONICTL\_CONFIG\_FILE**: location of the configuration file.
|
||||
- **PATRONI\_CTL\_INSECURE**: Allow connections to REST API without verifying SSL certs.
|
||||
- **PATRONI\_CTL\_CACERT**: Specifies the file with the CA_BUNDLE file or directory with certificates of trusted CAs to use while verifying REST API SSL certs. If not provided patronictl will use the value provided for REST API "cafile" parameter.
|
||||
- **PATRONI\_CTL\_CERTFILE**: Specifies the file with the client certificate in the PEM format. If not provided patronictl will use the value provided for REST API "certfile" parameter.
|
||||
- **PATRONI\_CTL\_KEYFILE**: Specifies the file with the client secret key in the PEM format. If not provided patronictl will use the value provided for REST API "keyfile" parameter.
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
# Minimal makefile for Sphinx documentation
|
||||
#
|
||||
|
||||
# You can set these variables from the command line.
|
||||
SPHINXOPTS =
|
||||
SPHINXBUILD = sphinx-build
|
||||
SPHINXPROJ = Patroni
|
||||
SOURCEDIR = .
|
||||
BUILDDIR = build
|
||||
|
||||
# Put it first so that "make" without argument is like "make help".
|
||||
help:
|
||||
@$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
|
||||
|
||||
.PHONY: help Makefile
|
||||
|
||||
# Catch-all target: route all unknown targets to Sphinx using the new
|
||||
# "make mode" option. $(O) is meant as a shortcut for $(SPHINXOPTS).
|
||||
%: Makefile
|
||||
@$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
|
||||
+188
@@ -0,0 +1,188 @@
|
||||
.. _readme:
|
||||
|
||||
============
|
||||
Introduction
|
||||
============
|
||||
|
||||
Patroni originated as a fork of `Governor <https://github.com/compose/governor>`__, the project from Compose. It includes plenty of new features.
|
||||
|
||||
For an example of a Docker-based deployment with Patroni, see `Spilo <https://github.com/zalando/spilo>`__, currently in use at Zalando.
|
||||
|
||||
For additional background info, see:
|
||||
|
||||
* `PostgreSQL HA with Kubernetes and Patroni <https://www.youtube.com/watch?v=iruaCgeG7qs>`__, talk by Josh Berkus at KubeCon 2016 (video)
|
||||
* `Feb. 2016 Zalando Tech blog post <https://tech.zalando.de/blog/zalandos-patroni-a-template-for-high-availability-postgresql/>`__
|
||||
|
||||
|
||||
Development Status
|
||||
------------------
|
||||
|
||||
Patroni is in active development and accepts contributions. See our :ref:`Contributing <contributing>` section below for more details.
|
||||
|
||||
We report new releases information :ref:`here <releases>`.
|
||||
|
||||
|
||||
Technical Requirements/Installation
|
||||
-----------------------------------
|
||||
|
||||
**Pre-requirements for Mac OS**
|
||||
|
||||
To install requirements on a Mac, run the following:
|
||||
|
||||
::
|
||||
|
||||
brew install postgresql etcd haproxy libyaml python
|
||||
|
||||
.. _psycopg2_install_options:
|
||||
|
||||
**Psycopg**
|
||||
|
||||
Starting from `psycopg2-2.8 <http://initd.org/psycopg/articles/2019/04/04/psycopg-28-released/>`__ the binary version of psycopg2 will no longer be installed by default. Installing it from the source code requires C compiler and postgres+python dev packages.
|
||||
Since in the python world it is not possible to specify dependency as ``psycopg2 OR psycopg2-binary`` you will have to decide how to install it.
|
||||
|
||||
There are a few options available:
|
||||
|
||||
1. Use the package manager from your distro
|
||||
|
||||
::
|
||||
|
||||
sudo apt-get install python-psycopg2 # install python2 psycopg2 module on Debian/Ubuntu
|
||||
sudo apt-get install python3-psycopg2 # install python3 psycopg2 module on Debian/Ubuntu
|
||||
sudo yum install python-psycopg2 # install python2 psycopg2 on RedHat/Fedora/CentOS
|
||||
|
||||
2. Install psycopg2 from the binary package
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg2-binary
|
||||
|
||||
3. Install psycopg2 from source
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg2>=2.5.4
|
||||
|
||||
4. Use psycopg 3.0 instead of psycopg2
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg[binary]>=3.0.0
|
||||
|
||||
**General installation for pip**
|
||||
|
||||
Patroni can be installed with pip:
|
||||
|
||||
::
|
||||
|
||||
pip install patroni[dependencies]
|
||||
|
||||
where dependencies can be either empty, or consist of one or more of the following:
|
||||
|
||||
etcd or etcd3
|
||||
`python-etcd` module in order to use Etcd as Distributed Configuration Store (DCS)
|
||||
consul
|
||||
`python-consul` module in order to use Consul as DCS
|
||||
zookeeper
|
||||
`kazoo` module in order to use Zookeeper as DCS
|
||||
exhibitor
|
||||
`kazoo` module in order to use Exhibitor as DCS (same dependencies as for Zookeeper)
|
||||
kubernetes
|
||||
`kubernetes` module in order to use Kubernetes as DCS in Patroni
|
||||
raft
|
||||
`pysyncobj` module in order to use python Raft implementation as DCS
|
||||
aws
|
||||
`boto` in order to use AWS callbacks
|
||||
|
||||
For example, the command in order to install Patroni together with dependencies for Etcd as a DCS and AWS callbacks is:
|
||||
|
||||
::
|
||||
|
||||
pip install patroni[etcd,aws]
|
||||
|
||||
Note that external tools to call in the replica creation or custom bootstrap scripts (i.e. WAL-E) should be installed
|
||||
independently of Patroni.
|
||||
|
||||
|
||||
.. _running_configuring:
|
||||
|
||||
Planning the Number of PostgreSQL Nodes
|
||||
---------------------------------------
|
||||
|
||||
Patroni/PostgreSQL nodes are decoupled from DCS nodes (except when Patroni implements RAFT on its own) and therefore
|
||||
there is no requirement on the minimal number of nodes. Running a cluster consisting of one master and one standby is
|
||||
perfectly fine. You can add more standby nodes later.
|
||||
|
||||
Running and Configuring
|
||||
-----------------------
|
||||
|
||||
The following section assumes Patroni repository as being cloned from https://github.com/zalando/patroni. Namely, you
|
||||
will need example configuration files `postgres0.yml` and `postgres1.yml`. If you installed Patroni with pip, you can
|
||||
obtain those files from the git repository and replace `./patroni.py` below with `patroni` command.
|
||||
|
||||
To get started, do the following from different terminals:
|
||||
::
|
||||
|
||||
> etcd --data-dir=data/etcd --enable-v2=true
|
||||
> ./patroni.py postgres0.yml
|
||||
> ./patroni.py postgres1.yml
|
||||
|
||||
You will then see a high-availability cluster start up. Test different settings in the YAML files to see how the cluster's behavior changes. Kill some of the components to see how the system behaves.
|
||||
|
||||
Add more ``postgres*.yml`` files to create an even larger cluster.
|
||||
|
||||
Patroni provides an `HAProxy <http://www.haproxy.org/>`__ configuration, which will give your application a single endpoint for connecting to the cluster's leader. To configure,
|
||||
run:
|
||||
|
||||
::
|
||||
|
||||
> haproxy -f haproxy.cfg
|
||||
|
||||
::
|
||||
|
||||
> psql --host 127.0.0.1 --port 5000 postgres
|
||||
|
||||
|
||||
YAML Configuration
|
||||
------------------
|
||||
|
||||
Go :ref:`here <settings>` for comprehensive information about settings for etcd, consul, and ZooKeeper. And for an example, see `postgres0.yml <https://github.com/zalando/patroni/blob/master/postgres0.yml>`__.
|
||||
|
||||
|
||||
Environment Configuration
|
||||
-------------------------
|
||||
|
||||
Go :ref:`here <environment>` for comprehensive information about configuring(overriding) settings via environment variables.
|
||||
|
||||
|
||||
Replication Choices
|
||||
-------------------
|
||||
|
||||
Patroni uses Postgres' streaming replication, which is asynchronous by default. Patroni's asynchronous replication configuration allows for ``maximum_lag_on_failover`` settings. This setting ensures failover will not occur if a follower is more than a certain number of bytes behind the leader. This setting should be increased or decreased based on business requirements. It's also possible to use synchronous replication for better durability guarantees. See :ref:`replication modes documentation <replication_modes>` for details.
|
||||
|
||||
|
||||
Applications Should Not Use Superusers
|
||||
--------------------------------------
|
||||
|
||||
When connecting from an application, always use a non-superuser. Patroni requires access to the database to function properly. By using a superuser from an application, you can potentially use the entire connection pool, including the connections reserved for superusers, with the ``superuser_reserved_connections`` setting. If Patroni cannot access the Primary because the connection pool is full, behavior will be undesirable.
|
||||
|
||||
.. |Build Status| image:: https://travis-ci.org/zalando/patroni.svg?branch=master
|
||||
:target: https://travis-ci.org/zalando/patroni
|
||||
.. |Coverage Status| image:: https://coveralls.io/repos/zalando/patroni/badge.svg?branch=master
|
||||
:target: https://coveralls.io/r/zalando/patroni?branch=master
|
||||
|
||||
Testing Your HA Solution
|
||||
--------------------------------------
|
||||
Testing an HA solution is a time consuming process, with many variables. This is particularly true considering a cross-platform application. You need a trained system administrator or a consultant to do this work. It is not something we can cover in depth in the documentation.
|
||||
|
||||
That said, here are some pieces of your infrastructure you should be sure to test:
|
||||
|
||||
* Network (the network in front of your system as well as the NICs [physical or virtual] themselves)
|
||||
* Disk IO
|
||||
* file limits (nofile in Linux)
|
||||
* RAM. Even if you have oomkiller turned off as suggested, the unavailability of RAM could cause issues.
|
||||
* CPU
|
||||
* Virtualization Contention (overcommitting the hypervisor)
|
||||
* Any cgroup limitation (likely to be related to the above)
|
||||
* ``kill -9`` of any postgres process (except postmaster!). This is a decent simulation of a segfault.
|
||||
|
||||
One thing that you should not do is run ``kill -9`` on a postmaster process. This is because doing so does not mimic any real life scenario. If you are concerned your infrastructure is insecure and an attacker could run ``kill -9``, no amount of HA process is going to fix that. The attacker will simply kill the process again, or cause chaos in another way.
|
||||
+363
-58
@@ -1,91 +1,396 @@
|
||||
.. _settings:
|
||||
|
||||
===========================
|
||||
YAML Configuration Settings
|
||||
===========================
|
||||
|
||||
.. _dynamic_configuration_settings:
|
||||
|
||||
Dynamic configuration settings
|
||||
------------------------------
|
||||
|
||||
Dynamic configuration is stored in the DCS (Distributed Configuration Store) and applied on all cluster nodes. Some parameters, like **loop_wait**, **ttl**, **postgresql.parameters.max_connections**, **postgresql.parameters.max_worker_processes** and so on could be set only in the dynamic configuration. Some other parameters like **postgresql.listen**, **postgresql.data_dir** could be set only locally, i.e. in the Patroni config file or via :ref:`configuration <environment>` variable. In most cases the local configuration will override the dynamic configuration. In order to change the dynamic configuration you can use either ``patronictl edit-config`` tool or Patroni :ref:`REST API <rest_api>`.
|
||||
|
||||
- **loop\_wait**: the number of seconds the loop will sleep. Default value: 10
|
||||
- **ttl**: the TTL to acquire the leader lock (in seconds). Think of it as the length of time before initiation of the automatic failover process. Default value: 30
|
||||
- **retry\_timeout**: timeout for DCS and PostgreSQL operation retries (in seconds). DCS or network issues shorter than this will not cause Patroni to demote the leader. Default value: 10
|
||||
- **maximum\_lag\_on\_failover**: the maximum bytes a follower may lag to be able to participate in leader election.
|
||||
- **maximum\_lag\_on\_syncnode**: the maximum bytes a synchronous follower may lag before it is considered as an unhealthy candidate and swapped by healthy asynchronous follower. Patroni utilize the max replica lsn if there is more than one follower, otherwise it will use leader's current wal lsn. Default is -1, Patroni will not take action to swap synchronous unhealthy follower when the value is set to 0 or below. Please set the value high enough so Patroni won't swap synchrounous follower fequently during high transaction volume.
|
||||
- **max\_timelines\_history**: maximum number of timeline history items kept in DCS. Default value: 0. When set to 0, it keeps the full history in DCS.
|
||||
- **master\_start\_timeout**: the amount of time a master is allowed to recover from failures before failover is triggered (in seconds). Default is 300 seconds. When set to 0 failover is done immediately after a crash is detected if possible. When using asynchronous replication a failover can cause lost transactions. Worst case failover time for master failure is: loop\_wait + master\_start\_timeout + loop\_wait, unless master\_start\_timeout is zero, in which case it's just loop\_wait. Set the value according to your durability/availability tradeoff.
|
||||
- **master\_stop\_timeout**: The number of seconds Patroni is allowed to wait when stopping Postgres and effective only when synchronous_mode is enabled. When set to > 0 and the synchronous_mode is enabled, Patroni sends SIGKILL to the postmaster if the stop operation is running for more than the value set by master_stop_timeout. Set the value according to your durability/availability tradeoff. If the parameter is not set or set <= 0, master_stop_timeout does not apply.
|
||||
- **synchronous\_mode**: turns on synchronous replication mode. In this mode a replica will be chosen as synchronous and only the latest leader and synchronous replica are able to participate in leader election. Synchronous mode makes sure that successfully committed transactions will not be lost at failover, at the cost of losing availability for writes when Patroni cannot ensure transaction durability. See :ref:`replication modes documentation <replication_modes>` for details.
|
||||
- **synchronous\_mode\_strict**: prevents disabling synchronous replication if no synchronous replicas are available, blocking all client writes to the master. See :ref:`replication modes documentation <replication_modes>` for details.
|
||||
- **postgresql**:
|
||||
- **use\_pg\_rewind**: whether or not to use pg_rewind. Defaults to `false`.
|
||||
- **use\_slots**: whether or not to use replication slots. Defaults to `true` on PostgreSQL 9.4+.
|
||||
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower. There is no recovery.conf anymore in PostgreSQL 12, but you may continue using this section, because Patroni handles it transparently.
|
||||
- **parameters**: list of configuration settings for Postgres.
|
||||
- **standby\_cluster**: if this section is defined, we want to bootstrap a standby cluster.
|
||||
- **host**: an address of remote master
|
||||
- **port**: a port of remote master
|
||||
- **primary\_slot\_name**: which slot on the remote master to use for replication. This parameter is optional, the default value is derived from the instance name (see function `slot_name_from_member_name`).
|
||||
- **create\_replica\_methods**: an ordered list of methods that can be used to bootstrap standby leader from the remote master, can be different from the list defined in :ref:`postgresql_settings`
|
||||
- **restore\_command**: command to restore WAL records from the remote master to standby leader, can be different from the list defined in :ref:`postgresql_settings`
|
||||
- **archive\_cleanup\_command**: cleanup command for standby leader
|
||||
- **recovery\_min\_apply\_delay**: how long to wait before actually apply WAL records on a standby leader
|
||||
- **slots**: define permanent replication slots. These slots will be preserved during switchover/failover. The logical slots are copied from the primary to a standby with restart, and after that their position advanced every **loop_wait** seconds (if necessary). Copying logical slot files performed via ``libpq`` connection and using either rewind or superuser credentials (see **postgresql.authentication** section). There is always a chance that the logical slot position on the replica is a bit behind the former primary, therefore application should be prepared that some messages could be received the second time after the failover. The easiest way of doing so - tracking ``confirmed_flush_lsn``. Enabling permanent logical replication slots requires **postgresql.use_slots** to be set and will also automatically enable the ``hot_standby_feedback``. Since the failover of logical replication slots is unsafe on PostgreSQL 9.6 and older and PostgreSQL version 10 is missing some important functions, the feature only works with PostgreSQL 11+.
|
||||
- **my_slot_name**: the name of replication slot. If the permanent slot name matches with the name of the current primary it will not be created. Everything else is the responsibility of the operator to make sure that there are no clashes in names between replication slots automatically created by Patroni for members and permanent replication slots.
|
||||
- **type**: slot type. Could be ``physical`` or ``logical``. If the slot is logical, you have to additionally define ``database`` and ``plugin``.
|
||||
- **database**: the database name where logical slots should be created.
|
||||
- **plugin**: the plugin name for the logical slot.
|
||||
- **ignore_slots**: list of sets of replication slot properties for which Patroni should ignore matching slots. This configuration/feature/etc. is useful when some replication slots are managed outside of Patroni. Any subset of matching properties will cause a slot to be ignored.
|
||||
- **name**: the name of the replication slot.
|
||||
- **type**: slot type. Can be ``physical`` or ``logical``. If the slot is logical, you may additionally define ``database`` and/or ``plugin``.
|
||||
- **database**: the database name (when matching a ``logical`` slot).
|
||||
- **plugin**: the logical decoding plugin (when matching a ``logical`` slot).
|
||||
|
||||
Note: **slots** is a hashmap while **ignore_slots** is an array. For example:
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
slots:
|
||||
permanent_logical_slot_name:
|
||||
type: logical
|
||||
database: my_db
|
||||
plugin: test_decoding
|
||||
permanent_physical_slot_name:
|
||||
type: physical
|
||||
...
|
||||
ignore_slots:
|
||||
- name: ignored_logical_slot_name
|
||||
type: logical
|
||||
database: my_db
|
||||
plugin: test_decoding
|
||||
- name: ignored_physical_slot_name
|
||||
type: physical
|
||||
...
|
||||
|
||||
Global/Universal
|
||||
----------------
|
||||
- **name**: the name of the host. Must be unique for the cluster.
|
||||
- **namespace**: path within the configuration store where Patroni will keep information about the cluster. Default value: "/service"
|
||||
- **scope**: cluster name
|
||||
|
||||
Log
|
||||
---
|
||||
- **level**: sets the general logging level. Default value is **INFO** (see `the docs for Python logging <https://docs.python.org/3.6/library/logging.html#levels>`_)
|
||||
- **traceback\_level**: sets the level where tracebacks will be visible. Default value is **ERROR**. Set it to **DEBUG** if you want to see tracebacks only if you enable **log.level=DEBUG**.
|
||||
- **format**: sets the log formatting string. Default value is **%(asctime)s %(levelname)s: %(message)s** (see `the LogRecord attributes <https://docs.python.org/3.6/library/logging.html#logrecord-attributes>`_)
|
||||
- **dateformat**: sets the datetime formatting string. (see the `formatTime() documentation <https://docs.python.org/3.6/library/logging.html#logging.Formatter.formatTime>`_)
|
||||
- **max\_queue\_size**: Patroni is using two-step logging. Log records are written into the in-memory queue and there is a separate thread which pulls them from the queue and writes to stderr or file. The maximum size of the internal queue is limited by default by **1000** records, which is enough to keep logs for the past 1h20m.
|
||||
- **dir**: Directory to write application logs to. The directory must exist and be writable by the user executing Patroni. If you set this value, the application will retain 4 25MB logs by default. You can tune those retention values with `file_num` and `file_size` (see below).
|
||||
- **file\_num**: The number of application logs to retain.
|
||||
- **file\_size**: Size of patroni.log file (in bytes) that triggers a log rolling.
|
||||
- **loggers**: This section allows redefining logging level per python module
|
||||
- **patroni.postmaster: WARNING**
|
||||
- **urllib3: DEBUG**
|
||||
|
||||
.. _bootstrap_settings:
|
||||
|
||||
Bootstrap configuration
|
||||
-----------------------
|
||||
- **dcs**: This section will be written into `/<namespace>/<scope>/config` of a given configuration store after initializing of new cluster. This is the global configuration for the cluster. If you want to change some parameters for all cluster nodes - just do it in DCS (or via Patroni API) and all nodes will apply this configuration.
|
||||
- **loop\_wait**: the number of seconds the loop will sleep. Default value: 10
|
||||
- **ttl**: the TTL to acquire the leader lock. Think of it as the length of time before initiation of the automatic failover process. Default value: 30
|
||||
- **maximum\_lag\_on\_failover**: the maximum bytes a follower may lag to be able to participate in leader election.
|
||||
- **postgresql**:
|
||||
- **use\_pg\_rewind**:whether or not to use pg_rewind
|
||||
- **use\_slots**: whether or not to use replication_slots. Must be False for PostgreSQL 9.3. You should comment out max_replication_slots before it becomes ineligible for leader status.
|
||||
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower.
|
||||
- **parameters**: list of configuration settings for Postgres. Many of these are required for replication to work.
|
||||
- **initdb**: List options to be passed on to initdb.
|
||||
- **- data-checksums**: Must be enabled when pg_rewind is needed on 9.3.
|
||||
- **- encoding: UTF8**: default encoding for new databases.
|
||||
- **- locale: UTF8**: default locale for new databases.
|
||||
- **pg\_hba**: list of lines that you should add to pg\_hba.conf.
|
||||
- **- host all all 0.0.0.0/0 md5**.
|
||||
- **- host replication replicator 127.0.0.1/32 md5**: A line like this is required for replication.
|
||||
- **users**: Some additional users users which needs to be created after initializing new cluster
|
||||
- **admin**: the name of user
|
||||
- **password: zalando**:
|
||||
- **options**: list of options for CREATE USER statement
|
||||
- **- createrole**
|
||||
- **- createdb**
|
||||
- **bootstrap**:
|
||||
- **dcs**: This section will be written into `/<namespace>/<scope>/config` of the given configuration store after initializing of new cluster. The global dynamic configuration for the cluster. Under the ``bootstrap.dcs`` you can put any of the parameters described in the :ref:`Dynamic Configuration settings <dynamic_configuration_settings>` and after Patroni initialized (bootstrapped) the new cluster, it will write this section into `/<namespace>/<scope>/config` of the configuration store. All later changes of ``bootstrap.dcs`` will not take any effect! If you want to change them please use either ``patronictl edit-config`` or Patroni :ref:`REST API <rest_api>`.
|
||||
- **method**: custom script to use for bootstrapping this cluster.
|
||||
See :ref:`custom bootstrap methods documentation <custom_bootstrap>` for details.
|
||||
When ``initdb`` is specified revert to the default ``initdb`` command. ``initdb`` is also triggered when no ``method``
|
||||
parameter is present in the configuration file.
|
||||
- **initdb**: List options to be passed on to initdb.
|
||||
- **- data-checksums**: Must be enabled when pg_rewind is needed on 9.3.
|
||||
- **- encoding: UTF8**: default encoding for new databases.
|
||||
- **- locale: UTF8**: default locale for new databases.
|
||||
- **pg\_hba**: list of lines that you should add to pg\_hba.conf.
|
||||
- **- host all all 0.0.0.0/0 md5**.
|
||||
- **- host replication replicator 127.0.0.1/32 md5**: A line like this is required for replication.
|
||||
- **users**: Some additional users which need to be created after initializing new cluster
|
||||
- **admin**: the name of user
|
||||
- **password: zalando**:
|
||||
- **options**: list of options for CREATE USER statement
|
||||
- **- createrole**
|
||||
- **- createdb**
|
||||
- **post\_bootstrap** or **post\_init**: An additional script that will be executed after initializing the cluster. The script receives a connection string URL (with the cluster superuser as a user name). The PGPASSFILE variable is set to the location of pgpass file.
|
||||
|
||||
.. _consul_settings:
|
||||
|
||||
Consul
|
||||
------
|
||||
- **host**: the host:port for the Consul endpoint.
|
||||
Most of the parameters are optional, but you have to specify one of the **host** or **url**
|
||||
|
||||
- **host**: the host:port for the Consul local agent.
|
||||
- **url**: url for the Consul local agent, in format: http(s)://host:port.
|
||||
- **port**: (optional) Consul port.
|
||||
- **scheme**: (optional) **http** or **https**, defaults to **http**.
|
||||
- **token**: (optional) ACL token.
|
||||
- **verify**: (optional) whether to verify the SSL certificate for HTTPS requests.
|
||||
- **cacert**: (optional) The ca certificate. If present it will enable validation.
|
||||
- **cert**: (optional) file with the client certificate.
|
||||
- **key**: (optional) file with the client key. Can be empty if the key is part of **cert**.
|
||||
- **dc**: (optional) Datacenter to communicate with. By default the datacenter of the host is used.
|
||||
- **consistency**: (optional) Select consul consistency mode. Possible values are ``default``, ``consistent``, or ``stale`` (more details in `consul API reference <https://www.consul.io/api/features/consistency.html/>`__)
|
||||
- **checks**: (optional) list of Consul health checks used for the session. By default an empty list is used.
|
||||
- **register\_service**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, replica or standby-leader depending on the node's role. Defaults to **false**.
|
||||
- **service\_tags**: (optional) additional static tags to add to the Consul service apart from the role (``master``/``replica``/``standby-leader``). By default an empty list is used.
|
||||
- **service\_check\_interval**: (optional) how often to perform health check against registered url.
|
||||
- **service\_check\_tls\_server\_name**: (optional) overide SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
|
||||
|
||||
The ``token`` needs to have the following ACL permissions:
|
||||
|
||||
::
|
||||
|
||||
service_prefix "${scope}" {
|
||||
policy = "write"
|
||||
}
|
||||
key_prefix "${namespace}/${scope}" {
|
||||
policy = "write"
|
||||
}
|
||||
session_prefix "" {
|
||||
policy = "write"
|
||||
}
|
||||
|
||||
Etcd
|
||||
----
|
||||
Most of the parameters are optional, but you have to specify one of the **host**, **hosts**, **url**, **proxy** or **srv**
|
||||
|
||||
- **host**: the host:port for the etcd endpoint.
|
||||
- **hosts**: list of etcd endpoint in format host1:port1,host2:port2,etc... Could be a comma separated string or an actual yaml list.
|
||||
- **use\_proxies**: If this parameter is set to true, Patroni will consider **hosts** as a list of proxies and will not perform a topology discovery of etcd cluster.
|
||||
- **url**: url for the etcd.
|
||||
- **proxy**: proxy url for the etcd. If you are connecting to the etcd using proxy, use this parameter instead of **url**.
|
||||
- **srv**: Domain to search the SRV record(s) for cluster autodiscovery. Patroni will try to query these SRV service names for specified domain (in that order until first success): ``_etcd-client-ssl``, ``_etcd-client``, ``_etcd-ssl``, ``_etcd``, ``_etcd-server-ssl``, ``_etcd-server``. If SRV records for ``_etcd-server-ssl`` or ``_etcd-server`` are retrieved then ETCD peer protocol is used do query ETCD for available members. Otherwise hosts from SRV records will be used.
|
||||
- **srv\_suffix**: Configures a suffix to the SRV name that is queried during discovery. Use this flag to differentiate between multiple etcd clusters under the same domain. Works only with conjunction with **srv**. For example, if ``srv_suffix: foo`` and ``srv: example.org`` are set, the following DNS SRV query is made:``_etcd-client-ssl-foo._tcp.example.com`` (and so on for every possible ETCD SRV service name).
|
||||
- **protocol**: (optional) http or https, if not specified http is used. If the **url** or **proxy** is specified - will take protocol from them.
|
||||
- **username**: (optional) username for etcd authentication.
|
||||
- **password**: (optional) password for etcd authentication.
|
||||
- **cacert**: (optional) The ca certificate. If present it will enable validation.
|
||||
- **cert**: (optional) file with the client certificate.
|
||||
- **key**: (optional) file with the client key. Can be empty if the key is part of **cert**.
|
||||
|
||||
Etcdv3
|
||||
------
|
||||
If you want that Patroni works with Etcd cluster via protocol version 3, you need to use the ``etcd3`` section in the Patroni configuration file. All configuration parameters are the same as for ``etcd``.
|
||||
|
||||
.. warning::
|
||||
Keys created with protocol version 2 are not visible with protocol version 3 and the other way around, therefore it is not possible to switch from ``etcd`` to ``etcd3`` just by updating Patroni config file.
|
||||
|
||||
|
||||
ZooKeeper
|
||||
----------
|
||||
- **hosts**: List of ZooKeeper cluster members in format: ['host1:port1', 'host2:port2', 'etc...'].
|
||||
- **use_ssl**: (optional) Whether SSL is used or not. Defaults to ``false``. If set to ``false``, all SSL specific parameters are ignored.
|
||||
- **cacert**: (optional) The CA certificate. If present it will enable validation.
|
||||
- **cert**: (optional) File with the client certificate.
|
||||
- **key**: (optional) File with the client key.
|
||||
- **key_password**: (optional) The client key password.
|
||||
- **verify**: (optional) Whether to verify certificate or not. Defaults to ``true``.
|
||||
- **set_acls**: (optional) If set, configure Kazoo to apply a default ACL to each ZNode that it creates. ACLs will assume 'x509' schema and should be specified as a dictionary with the principal as the key and one or more permissions as a list in the value. Permissions may be one of ``CREATE``, ``READ``, ``WRITE``, ``DELETE`` or ``ADMIN``. For example, ``set_acls: {CN=principal1: [CREATE, READ], CN=principal2: [ALL]}``.
|
||||
|
||||
.. note::
|
||||
It is required to install ``kazoo>=2.6.0`` to support SSL.
|
||||
|
||||
|
||||
Exhibitor
|
||||
---------
|
||||
- **hosts**: initial list of Exhibitor (ZooKeeper) nodes in format: 'host1,host2,etc...'. This list updates automatically whenever the Exhibitor (ZooKeeper) cluster topology changes.
|
||||
- **poll\_interval**: how often the list of ZooKeeper and Exhibitor nodes should be updated from Exhibitor
|
||||
- **poll\_interval**: how often the list of ZooKeeper and Exhibitor nodes should be updated from Exhibitor.
|
||||
- **port**: Exhibitor port.
|
||||
|
||||
.. _kubernetes_settings:
|
||||
|
||||
Kubernetes
|
||||
----------
|
||||
- **bypass\_api\_service**: (optional) When communicating with the Kubernetes API, Patroni is usually relying on the `kubernetes` service, the address of which is exposed in the pods via the `KUBERNETES_SERVICE_HOST` environment variable. If `bypass_api_service` is set to ``true``, Patroni will resolve the list of API nodes behind the service and connect directly to them.
|
||||
- **namespace**: (optional) Kubernetes namespace where Patroni pod is running. Default value is `default`.
|
||||
- **labels**: Labels in format ``{label1: value1, label2: value2}``. These labels will be used to find existing objects (Pods and either Endpoints or ConfigMaps) associated with the current cluster. Also Patroni will set them on every object (Endpoint or ConfigMap) it creates.
|
||||
- **scope\_label**: (optional) name of the label containing cluster name. Default value is `cluster-name`.
|
||||
- **role\_label**: (optional) name of the label containing role (master or replica). Patroni will set this label on the pod it runs in. Default value is ``role``.
|
||||
- **use\_endpoints**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||
- **pod\_ip**: (optional) IP address of the pod Patroni is running in. This value is required when `use_endpoints` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||
- **ports**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``kubernetes.ports: [{"name": "postgresql", "port": 5432}]`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `kubernetes.use_endpoints` is set.
|
||||
- **cacert**: (optional) Specifies the file with the CA_BUNDLE file with certificates of trusted CAs to use while verifying Kubernetes API SSL certs. If not provided, patroni will use the value provided by the ServiceAccount secret.
|
||||
|
||||
|
||||
.. _raft_settings:
|
||||
|
||||
Raft
|
||||
----
|
||||
- **self\_addr**: ``ip:port`` to listen on for Raft connections. The ``self_addr`` must be accessible from other nodes of the cluster. If not set, the node will not participate in consensus.
|
||||
- **bind\_addr**: (optional) ``ip:port`` to listen on for Raft connections. If not specified the ``self_addr`` will be used.
|
||||
- **partner\_addrs**: list of other Patroni nodes in the cluster in format: ['ip1:port', 'ip2:port', 'etc...']
|
||||
- **data\_dir**: directory where to store Raft log and snapshot. If not specified the current working directory is used.
|
||||
- **password**: (optional) Encrypt Raft traffic with a specified password, requires ``cryptography`` python module.
|
||||
|
||||
Short FAQ about Raft implementation
|
||||
|
||||
- Q: How to list all the nodes providing consensus?
|
||||
|
||||
A: ``syncobj_admin -conn host:port -status`` where the host:port is the address of one of the cluster nodes
|
||||
|
||||
- Q: Node that was a part of consensus and has gone and I can't reuse the same IP for other node. How to remove this node from the consensus?
|
||||
|
||||
A: ``syncobj_admin -conn host:port -remove host2:port2`` where the ``host2:port2`` is the address of the node you want to remove from consensus.
|
||||
|
||||
- Q: Where to get the ``syncobj_admin`` utility?
|
||||
|
||||
A: It is installed together with ``pysyncobj`` module (python RAFT implementation), which is Patroni dependency.
|
||||
|
||||
- Q: it is possible to run Patroni node without adding in to the consensus?
|
||||
|
||||
A: Yes, just comment out or remove ``raft.self_addr`` from Patroni configuration.
|
||||
|
||||
- Q: It is possible to run Patroni and PostgreSQL only on two nodes?
|
||||
|
||||
A: Yes, on the third node you can run ``patroni_raft_controller`` (without Patroni and PostgreSQL). In such a setup, one can temporarily lose one node without affecting the primary.
|
||||
|
||||
|
||||
.. _postgresql_settings:
|
||||
|
||||
PostgreSQL
|
||||
----------
|
||||
- **authentication**:
|
||||
- **superuser**:
|
||||
- **username**: name for the superuser, set during initialization (initdb) and later used by Patroni to connect to the postgres.
|
||||
- **password**: password for the superuser, set during initialization (initdb).
|
||||
- **replication**:
|
||||
- **username**: replication username; the user will be created during initialization. Replicas will use this user to access master via streaming replication
|
||||
- **password**: replication password; the user will be created during initialization.
|
||||
- **callbacks**: callback scripts to run on certain actions. Patroni will pass the action, role and cluster name. (See scripts/aws.py as an example of how to write them.)
|
||||
- **on\_reload**: run this script when configuration reload is triggered.
|
||||
- **on\_restart**: run this script when the cluster restarts.
|
||||
- **on\_role\_change**: run this script when the cluster is being promoted or demoted.
|
||||
- **on\_start**: run this script when the cluster starts.
|
||||
- **on\_stop**: run this script when the cluster stops.
|
||||
- **connect\_address**: IP address + port through which Postgres is accessible from other nodes and applications.
|
||||
- **create\_replica\_methods**: an ordered list of the create methods for turning a Patroni node into a new replica. "basebackup" is the default method; other methods are assumed to refer to scripts, each of which is configured as its own config item.
|
||||
- **data\_dir**: The location of the Postgres data directory, either existing or to be initialized by Patroni.
|
||||
- **listen**: IP address + port that Postgres listens to; must be accessible from other nodes in the cluster, if you're using streaming replication. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node.
|
||||
- **pgpass**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup and under some other circumstances. The location must be writable by Patroni.
|
||||
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower.
|
||||
- **parameters**: list of configuration settings for Postgres. Many of these are required for replication to work.
|
||||
- **pg\_ctl\_timeout**: How long should pg_ctl wait when doing ``start``, ``stop`` or ``restart``. Default value is 60 seconds.
|
||||
- **use\_pg\_rewind**: try to use pg\_rewind on the former leader when it joins cluster as a replica.
|
||||
- **remove\_data\_directory\_on\_rewind\_failure**: If this option is enabled, Patroni will remove postgres data directory and recreate replica. Otherwise it will try to follow the new leader. Default value is **false**.
|
||||
- **replica\_method** for each create_replica_method other than basebackup, you would add a configuration section of the same name. At a minimum, this should include "command" with a full path to the actual script to be executed. Other configuration parameters will be passed along to the script in the form "parameter=value".
|
||||
- **postgresql**:
|
||||
- **authentication**:
|
||||
- **superuser**:
|
||||
- **username**: name for the superuser, set during initialization (initdb) and later used by Patroni to connect to the postgres.
|
||||
- **password**: password for the superuser, set during initialization (initdb).
|
||||
- **sslmode**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
|
||||
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||
- **sslpassword**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``sslkey``.
|
||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
- **replication**:
|
||||
- **username**: replication username; the user will be created during initialization. Replicas will use this user to access master via streaming replication
|
||||
- **password**: replication password; the user will be created during initialization.
|
||||
- **sslmode**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
|
||||
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||
- **sslpassword**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``sslkey``.
|
||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
- **rewind**:
|
||||
- **username**: name for the user for ``pg_rewind``; the user will be created during initialization of postgres 11+ and all necessary `permissions <https://www.postgresql.org/docs/11/app-pgrewind.html#id-1.9.5.8.8>`__ will be granted.
|
||||
- **password**: password for the user for ``pg_rewind``; the user will be created during initialization.
|
||||
- **sslmode**: (optional) maps to the `sslmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLMODE>`__ connection parameter, which allows a client to specify the type of TLS negotiation mode with the server. For more information on how each mode works, please visit the `PostgreSQL documentation <https://www.postgresql.org/docs/current/libpq-ssl.html#LIBPQ-SSL-SSLMODE-STATEMENTS>`__. The default mode is ``prefer``.
|
||||
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||
- **sslpassword**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``sslkey``.
|
||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
- **callbacks**: callback scripts to run on certain actions. Patroni will pass the action, role and cluster name. (See scripts/aws.py as an example of how to write them.)
|
||||
- **on\_reload**: run this script when configuration reload is triggered.
|
||||
- **on\_restart**: run this script when the postgres restarts (without changing role).
|
||||
- **on\_role\_change**: run this script when the postgres is being promoted or demoted.
|
||||
- **on\_start**: run this script when the postgres starts.
|
||||
- **on\_stop**: run this script when the postgres stops.
|
||||
- **connect\_address**: IP address + port through which Postgres is accessible from other nodes and applications.
|
||||
- **create\_replica\_methods**: an ordered list of the create methods for turning a Patroni node into a new replica.
|
||||
"basebackup" is the default method; other methods are assumed to refer to scripts, each of which is configured as its
|
||||
own config item. See :ref:`custom replica creation methods documentation <custom_replica_creation>` for further explanation.
|
||||
- **data\_dir**: The location of the Postgres data directory, either :ref:`existing <existing_data>` or to be initialized by Patroni.
|
||||
- **config\_dir**: The location of the Postgres configuration directory, defaults to the data directory. Must be writable by Patroni.
|
||||
- **bin\_dir**: Path to PostgreSQL binaries (pg_ctl, pg_rewind, pg_basebackup, postgres). The default value is an empty string meaning that PATH environment variable will be used to find the executables.
|
||||
- **listen**: IP address + port that Postgres listens to; must be accessible from other nodes in the cluster, if you're using streaming replication. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node.
|
||||
- **use\_unix\_socket**: specifies that Patroni should prefer to use unix sockets to connect to the cluster. Default value is ``false``. If ``unix_socket_directories`` is defined, Patroni will use the first suitable value from it to connect to the cluster and fallback to tcp if nothing is suitable. If ``unix_socket_directories`` is not specified in ``postgresql.parameters``, Patroni will assume that the default value should be used and omit ``host`` from the connection parameters.
|
||||
- **use\_unix\_socket\_repl**: specifies that Patroni should prefer to use unix sockets for replication user cluster connection. Default value is ``false``. If ``unix_socket_directories`` is defined, Patroni will use the first suitable value from it to connect to the cluster and fallback to tcp if nothing is suitable. If ``unix_socket_directories`` is not specified in ``postgresql.parameters``, Patroni will assume that the default value should be used and omit ``host`` from the connection parameters.
|
||||
- **pgpass**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup, the post_init script and under some other circumstances. The location must be writable by Patroni.
|
||||
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower.
|
||||
- **custom\_conf** : path to an optional custom ``postgresql.conf`` file, that will be used in place of ``postgresql.base.conf``. The file must exist on all cluster nodes, be readable by PostgreSQL and will be included from its location on the real ``postgresql.conf``. Note that Patroni will not monitor this file for changes, nor backup it. However, its settings can still be overridden by Patroni's own configuration facilities - see :ref:`dynamic configuration <dynamic_configuration>` for details.
|
||||
- **parameters**: list of configuration settings for Postgres. Many of these are required for replication to work.
|
||||
- **pg\_hba**: list of lines that Patroni will use to generate ``pg_hba.conf``. This parameter has higher priority than ``bootstrap.pg_hba``. Together with :ref:`dynamic configuration <dynamic_configuration>` it simplifies management of ``pg_hba.conf``.
|
||||
- **- host all all 0.0.0.0/0 md5**.
|
||||
- **- host replication replicator 127.0.0.1/32 md5**: A line like this is required for replication.
|
||||
- **pg\_ident**: list of lines that Patroni will use to generate ``pg_ident.conf``. Together with :ref:`dynamic configuration <dynamic_configuration>` it simplifies management of ``pg_ident.conf``.
|
||||
- **- mapname1 systemname1 pguser1**.
|
||||
- **- mapname1 systemname2 pguser2**.
|
||||
- **pg\_ctl\_timeout**: How long should pg_ctl wait when doing ``start``, ``stop`` or ``restart``. Default value is 60 seconds.
|
||||
- **use\_pg\_rewind**: try to use pg\_rewind on the former leader when it joins cluster as a replica.
|
||||
- **remove\_data\_directory\_on\_rewind\_failure**: If this option is enabled, Patroni will remove the PostgreSQL data directory and recreate the replica. Otherwise it will try to follow the new leader. Default value is **false**.
|
||||
- **remove\_data\_directory\_on\_diverged\_timelines**: Patroni will remove the PostgreSQL data directory and recreate the replica if it notices that timelines are diverging and the former master can not start streaming from the new master. This option is useful when ``pg_rewind`` can not be used. While performing timelines divergence check on PostgreSQL v10 and older Patroni will try to connect with replication credential to the "postgres" database. Hence, such access should be allowed in the pg_hba.conf. Default value is **false**.
|
||||
- **replica\_method**: for each create_replica_methods other than basebackup, you would add a configuration section of the same name. At a minimum, this should include "command" with a full path to the actual script to be executed. Other configuration parameters will be passed along to the script in the form "parameter=value".
|
||||
- **pre\_promote**: a fencing script that executes during a failover after acquiring the leader lock but before promoting the replica. If the script exits with a non-zero code, Patroni does not promote the replica and removes the leader key from DCS.
|
||||
|
||||
REST API
|
||||
--------
|
||||
- **connect\_address**: IP address and port to access the REST API.
|
||||
- **listen**: IP address and port that Patroni will listen to, to provide health-check information for HAProxy.
|
||||
- **Optional**:
|
||||
- **authentication**:
|
||||
--------
|
||||
- **restapi**:
|
||||
- **connect\_address**: IP address (or hostname) and port, to access the Patroni's :ref:`REST API <rest_api>`. All the members of the cluster must be able to connect to this address, so unless the Patroni setup is intended for a demo inside the localhost, this address must be a non "localhost" or loopback address (ie: "localhost" or "127.0.0.1"). It can serve as an endpoint for HTTP health checks (read below about the "listen" REST API parameter), and also for user queries (either directly or via the REST API), as well as for the health checks done by the cluster members during leader elections (for example, to determine whether the master is still running, or if there is a node which has a WAL position that is ahead of the one doing the query; etc.) The connect_address is put in the member key in DCS, making it possible to translate the member name into the address to connect to its REST API.
|
||||
|
||||
- **listen**: IP address (or hostname) and port that Patroni will listen to for the REST API - to provide also the same health checks and cluster messaging between the participating nodes, as described above. to provide health-check information for HAProxy (or any other load balancer capable of doing a HTTP "OPTION" or "GET" checks).
|
||||
|
||||
- **authentication**: (optional)
|
||||
- **username**: Basic-auth username to protect unsafe REST API endpoints.
|
||||
- **password**: Basic-auth password to protect unsafe REST API endpoints.
|
||||
- **certfile**: (optional): Specifies the file with the certificate in the PEM format. If the certfile is not specified or is left empty, the API server will work without SSL.
|
||||
- **keyfile**: (optional): Specifies the file with the secret key in the PEM format.
|
||||
- **keyfile\_password**: (optional): Specifies a password for decrypting the keyfile.
|
||||
- **cafile**: (optional): Specifies the file with the CA_BUNDLE with certificates of trusted CAs to use while verifying client certs.
|
||||
- **ciphers**: (optional): Specifies the permitted cipher suites (e.g. "ECDHE-RSA-AES256-GCM-SHA384:DHE-RSA-AES256-GCM-SHA384:ECDHE-RSA-AES128-GCM-SHA256:DHE-RSA-AES128-GCM-SHA256:!SSLv1:!SSLv2:!SSLv3:!TLSv1:!TLSv1.1")
|
||||
- **verify\_client**: (optional): ``none`` (default), ``optional`` or ``required``. When ``none`` REST API will not check client certificates. When ``required`` client certificates are required for all REST API calls. When ``optional`` client certificates are required for all unsafe REST API endpoints. When ``required`` is used, then client authentication succeeds, if the certificate signature verification succeeds. For ``optional`` the client cert will only be checked for ``PUT``, ``POST``, ``PATCH``, and ``DELETE`` requests.
|
||||
- **allowlist**: (optional): Specifies the set of hosts that are allowed to call unsafe REST API endpoints. The single element could be a host name, an IP address or a network address using CIDR notation. By default ``allow all`` is used. In case if ``allowlist`` or ``allowlist_include_members`` are set, anything that is not included is rejected.
|
||||
- **allowlist\_include\_members**: (optional): If set to ``true`` it allows accessing unsafe REST API endpoints from other cluster members registered in DCS (IP address or hostname is taken from the members ``api_url``). Be careful, it might happen that OS will use a different IP for outgoing connections.
|
||||
- **http\_extra\_headers**: (optional): HTTP headers let the REST API server pass additional information with an HTTP response.
|
||||
- **https\_extra\_headers**: (optional): HTTPS headers let the REST API server pass additional information with an HTTP response when TLS is enabled. This will also pass additional information set in ``http_extra_headers``.
|
||||
|
||||
- **certfile**: Specifies the file with the certificate in the PEM format. If the certfile is not specified or is left empty, the API server will work without SSL.
|
||||
- **keyfile**: Specifies the file with the secret key in the PEM format.
|
||||
Here is an example of both **http_extra_headers** and **https_extra_headers**:
|
||||
|
||||
ZooKeeper
|
||||
----------
|
||||
- **hosts**: list of ZooKeeper cluster members in format: ['host1:port1', 'host2:port2', 'etc...'].
|
||||
.. code:: YAML
|
||||
|
||||
restapi:
|
||||
listen: <listen>
|
||||
connect_address: <connect_address>
|
||||
authentication:
|
||||
username: <username>
|
||||
password: <password>
|
||||
http_extra_headers:
|
||||
'X-Frame-Options': 'SAMEORIGIN'
|
||||
'X-XSS-Protection': '1; mode=block'
|
||||
'X-Content-Type-Options': 'nosniff'
|
||||
cafile: <ca file>
|
||||
certfile: <cert>
|
||||
keyfile: <key>
|
||||
https_extra_headers:
|
||||
'Strict-Transport-Security': 'max-age=31536000; includeSubDomains'
|
||||
|
||||
.. _patronictl_settings:
|
||||
|
||||
CTL
|
||||
---
|
||||
- **ctl**: (optional)
|
||||
- **insecure**: Allow connections to REST API without verifying SSL certs.
|
||||
- **cacert**: Specifies the file with the CA_BUNDLE file or directory with certificates of trusted CAs to use while verifying REST API SSL certs. If not provided patronictl will use the value provided for REST API "cafile" parameter.
|
||||
- **certfile**: Specifies the file with the client certificate in the PEM format. If not provided patronictl will use the value provided for REST API "certfile" parameter.
|
||||
- **keyfile**: Specifies the file with the client secret key in the PEM format. If not provided patronictl will use the value provided for REST API "keyfile" parameter.
|
||||
- **keyfile\_password**: Specifies a password for decrypting the keyfile. If not provided patronictl will use the value provided for REST API "keyfile\_password" parameter.
|
||||
|
||||
Watchdog
|
||||
--------
|
||||
- **mode**: ``off``, ``automatic`` or ``required``. When ``off`` watchdog is disabled. When ``automatic`` watchdog will be used if available, but ignored if it is not. When ``required`` the node will not become a leader unless watchdog can be successfully enabled.
|
||||
- **device**: Path to watchdog device. Defaults to ``/dev/watchdog``.
|
||||
- **safety_margin**: Number of seconds of safety margin between watchdog triggering and leader key expiration.
|
||||
|
||||
.. _tags_settings:
|
||||
|
||||
Tags
|
||||
----
|
||||
- **nofailover**: ``true`` or ``false``, controls whether this node is allowed to participate in the leader race and become a leader. Defaults to ``false``
|
||||
- **clonefrom**: ``true`` or ``false``. If set to ``true`` other nodes might prefer to use this node for bootstrap (take ``pg_basebackup`` from). If there are several nodes with ``clonefrom`` tag set to ``true`` the node to bootstrap from will be chosen randomly. The default value is ``false``.
|
||||
- **noloadbalance**: ``true`` or ``false``. If set to ``true`` the node will return HTTP Status Code 503 for the ``GET /replica`` REST API health-check and therefore will be excluded from the load-balancing. Defaults to ``false``.
|
||||
- **replicatefrom**: The IP address/hostname of another replica. Used to support cascading replication.
|
||||
- **nosync**: ``true`` or ``false``. If set to ``true`` the node will never be selected as a synchronous replica.
|
||||
|
||||
In addition to these predefined tags, you can also add your own ones:
|
||||
|
||||
- **key1**: ``true``
|
||||
- **key2**: ``false``
|
||||
- **key3**: ``1.4``
|
||||
- **key4**: ``"RandomString"``
|
||||
|
||||
Tags are visible in the :ref:`REST API <rest_api>` and ``patronictl list`` You can also check for an instance health using these tags. If the tag isn't defined for an instance, or if the respective value doesn't match the querying value, it will return HTTP Status Code 503.
|
||||
|
||||
Vendored
+3
@@ -0,0 +1,3 @@
|
||||
li {
|
||||
margin-bottom: 0.5em
|
||||
}
|
||||
+200
@@ -0,0 +1,200 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# Patroni documentation build configuration file, created by
|
||||
# sphinx-quickstart on Mon Dec 19 16:54:09 2016.
|
||||
#
|
||||
# This file is execfile()d with the current directory set to its
|
||||
# containing dir.
|
||||
#
|
||||
# Note that not all possible configuration values are present in this
|
||||
# autogenerated file.
|
||||
#
|
||||
# All configuration values have a default; values that are commented out
|
||||
# serve to show the default.
|
||||
|
||||
# If extensions (or modules to document with autodoc) are in another directory,
|
||||
# add these directories to sys.path here. If the directory is relative to the
|
||||
# documentation root, use os.path.abspath to make it absolute, like shown here.
|
||||
#
|
||||
import os
|
||||
|
||||
import sys
|
||||
sys.path.insert(0, os.path.abspath('..'))
|
||||
|
||||
from patroni.version import __version__
|
||||
|
||||
# -- General configuration ------------------------------------------------
|
||||
|
||||
# If your documentation needs a minimal Sphinx version, state it here.
|
||||
#
|
||||
# needs_sphinx = '1.0'
|
||||
|
||||
# Add any Sphinx extension module names here, as strings. They can be
|
||||
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
|
||||
# ones.
|
||||
extensions = ['sphinx.ext.intersphinx',
|
||||
'sphinx.ext.todo',
|
||||
'sphinx.ext.mathjax',
|
||||
'sphinx.ext.ifconfig',
|
||||
'sphinx.ext.viewcode']
|
||||
|
||||
# Add any paths that contain templates here, relative to this directory.
|
||||
templates_path = ['_templates']
|
||||
|
||||
# The suffix(es) of source filenames.
|
||||
# You can specify multiple suffix as a list of string:
|
||||
#
|
||||
# source_suffix = ['.rst', '.md']
|
||||
source_suffix = '.rst'
|
||||
|
||||
# The master toctree document.
|
||||
master_doc = 'index'
|
||||
|
||||
# General information about the project.
|
||||
project = 'Patroni'
|
||||
copyright = '2015 Compose, Zalando SE'
|
||||
author = 'Zalando SE'
|
||||
|
||||
# The version info for the project you're documenting, acts as replacement for
|
||||
# |version| and |release|, also used in various other places throughout the
|
||||
# built documents.
|
||||
#
|
||||
# The short X.Y version.
|
||||
version = __version__[:__version__.rfind('.')]
|
||||
# The full version, including alpha/beta/rc tags.
|
||||
release = __version__
|
||||
|
||||
# The language for content autogenerated by Sphinx. Refer to documentation
|
||||
# for a list of supported languages.
|
||||
#
|
||||
# This is also used if you do content translation via gettext catalogs.
|
||||
# Usually you set "language" from the command line for these cases.
|
||||
language = None
|
||||
|
||||
# List of patterns, relative to source directory, that match files and
|
||||
# directories to ignore when looking for source files.
|
||||
# This patterns also effect to html_static_path and html_extra_path
|
||||
exclude_patterns = []
|
||||
|
||||
# The name of the Pygments (syntax highlighting) style to use.
|
||||
pygments_style = 'sphinx'
|
||||
|
||||
# If true, `todo` and `todoList` produce output, else they produce nothing.
|
||||
todo_include_todos = True
|
||||
|
||||
|
||||
# -- Options for HTML output ----------------------------------------------
|
||||
|
||||
# The theme to use for HTML and HTML Help pages. See the documentation for
|
||||
# a list of builtin themes.
|
||||
#
|
||||
|
||||
on_rtd = os.environ.get('READTHEDOCS', None) == 'True'
|
||||
if not on_rtd: # only import and set the theme if we're building docs locally
|
||||
import sphinx_rtd_theme
|
||||
html_theme = 'sphinx_rtd_theme'
|
||||
html_theme_path = [sphinx_rtd_theme.get_html_theme_path()]
|
||||
|
||||
# Theme options are theme-specific and customize the look and feel of a theme
|
||||
# further. For a list of options available for each theme, see the
|
||||
# documentation.
|
||||
#
|
||||
# html_theme_options = {}
|
||||
|
||||
# Add any paths that contain custom static files (such as style sheets) here,
|
||||
# relative to this directory. They are copied after the builtin static files,
|
||||
# so a file named "default.css" will overwrite the builtin "default.css".
|
||||
html_static_path = ['_static']
|
||||
|
||||
|
||||
# -- Options for HTMLHelp output ------------------------------------------
|
||||
|
||||
# Output file base name for HTML help builder.
|
||||
htmlhelp_basename = 'Patronidoc'
|
||||
|
||||
|
||||
# -- Options for LaTeX output ---------------------------------------------
|
||||
|
||||
latex_elements = {
|
||||
# The paper size ('letterpaper' or 'a4paper').
|
||||
#
|
||||
# 'papersize': 'letterpaper',
|
||||
|
||||
# The font size ('10pt', '11pt' or '12pt').
|
||||
#
|
||||
# 'pointsize': '10pt',
|
||||
|
||||
# Additional stuff for the LaTeX preamble.
|
||||
#
|
||||
# 'preamble': '',
|
||||
|
||||
# Latex figure (float) alignment
|
||||
#
|
||||
# 'figure_align': 'htbp',
|
||||
}
|
||||
|
||||
# Grouping the document tree into LaTeX files. List of tuples
|
||||
# (source start file, target name, title,
|
||||
# author, documentclass [howto, manual, or own class]).
|
||||
latex_documents = [
|
||||
(master_doc, 'Patroni.tex', 'Patroni Documentation',
|
||||
'Zalando SE', 'manual'),
|
||||
]
|
||||
|
||||
|
||||
# -- Options for manual page output ---------------------------------------
|
||||
|
||||
# One entry per manual page. List of tuples
|
||||
# (source start file, name, description, authors, manual section).
|
||||
man_pages = [
|
||||
(master_doc, 'patroni', 'Patroni Documentation',
|
||||
[author], 1)
|
||||
]
|
||||
|
||||
|
||||
# -- Options for Texinfo output -------------------------------------------
|
||||
|
||||
# Grouping the document tree into Texinfo files. List of tuples
|
||||
# (source start file, target name, title, author,
|
||||
# dir menu entry, description, category)
|
||||
texinfo_documents = [
|
||||
(master_doc, 'Patroni', 'Patroni Documentation',
|
||||
author, 'Patroni', 'One line description of project.',
|
||||
'Miscellaneous'),
|
||||
]
|
||||
|
||||
|
||||
|
||||
# -- Options for Epub output ----------------------------------------------
|
||||
|
||||
# Bibliographic Dublin Core info.
|
||||
epub_title = project
|
||||
epub_author = author
|
||||
epub_publisher = author
|
||||
epub_copyright = copyright
|
||||
|
||||
# The unique identifier of the text. This can be a ISBN number
|
||||
# or the project homepage.
|
||||
#
|
||||
# epub_identifier = ''
|
||||
|
||||
# A unique identification for the text.
|
||||
#
|
||||
# epub_uid = ''
|
||||
|
||||
# A list of files that should not be packed into the epub file.
|
||||
epub_exclude_files = ['search.html']
|
||||
|
||||
|
||||
|
||||
# Example configuration for intersphinx: refer to the Python standard library.
|
||||
intersphinx_mapping = {'https://docs.python.org/': None}
|
||||
|
||||
# A possibility to have an own stylesheet, to add new rules or override existing ones
|
||||
# For the latter case, the CSS specificity of the rules should be higher than the default ones
|
||||
def setup(app):
|
||||
if hasattr(app, 'add_css_file'):
|
||||
app.add_css_file('custom.css')
|
||||
else:
|
||||
app.add_stylesheet('custom.css')
|
||||
+19
-155
@@ -1,3 +1,5 @@
|
||||
.. _dynamic_configuration:
|
||||
|
||||
Patroni configuration
|
||||
=====================
|
||||
|
||||
@@ -10,13 +12,15 @@ Patroni configuration is stored in the DCS (Distributed Configuration Store). Th
|
||||
have changed), a special flag, ``pending_restart`` indicating this, is set in the members.data JSON.
|
||||
Additionally, the node status also indicates this, by showing ``"restart_pending": true``.
|
||||
|
||||
- Local `configuration <https://github.com/zalando/patroni/blob/master/docs/SETTINGS.rst>`__ (patroni.yml).
|
||||
- Local :ref:`configuration <settings>` (patroni.yml).
|
||||
These options are defined in the configuration file and take precedence over dynamic configuration.
|
||||
patroni.yml could be changed and reload in runtime (without restart of Patroni) by sending SIGHUP to the Patroni process or by performing ``POST /reload`` REST-API request.
|
||||
patroni.yml could be changed and reloaded in runtime (without restart of Patroni) by sending SIGHUP to the Patroni process, performing ``POST /reload`` REST-API request or executing ``patronictl reload``.
|
||||
|
||||
- Environment `configuration <https://github.com/zalando/patroni/blob/master/docs/ENVIRONMENT.rst>`__ .
|
||||
- Environment :ref:`configuration <environment>`.
|
||||
It is possible to set/override some of the "Local" configuration parameters with environment variables.
|
||||
Environment configuration is very useful when you are running in a dynamic environment and you don't know some of the parameters in advance (for example it's not possible to know you external IP address when you are running inside ``docker``).
|
||||
Environment configuration is very useful when you are running in a dynamic environment and you don't know some of the parameters in advance (for example it's not possible to know your external IP address when you are running inside ``docker``).
|
||||
|
||||
The local configuration can be either a single YAML file or a directory. When it is a directory, all YAML files in that directory are loaded one by one in sorted order. In case a key is defined in multiple files, the occurrence in the last file takes precedence.
|
||||
|
||||
Some of the PostgreSQL parameters must hold the same values on the master and the replicas. For those, values set either in the local patroni configuration files or via the environment variables take no effect. To alter or set their values one must change the shared configuration in the DCS. Below is the actual list of such parameters together with the default values:
|
||||
|
||||
@@ -33,6 +37,7 @@ For the parameters below, PostgreSQL does not require equal values among the mas
|
||||
- max_wal_senders: 5
|
||||
- max_replication_slots: 5
|
||||
- wal_keep_segments: 8
|
||||
- wal_keep_size: 128MB
|
||||
|
||||
These parameters are validated to ensure they are sane, or meet a minimum value.
|
||||
|
||||
@@ -40,7 +45,7 @@ There are some other Postgres parameters controlled by Patroni:
|
||||
|
||||
- listen_addresses - is set either from ``postgresql.listen`` or from ``PATRONI_POSTGRESQL_LISTEN`` environment variable
|
||||
- port - is set either from ``postgresql.listen`` or from ``PATRONI_POSTGRESQL_LISTEN`` environment variable
|
||||
- cluster_name - is set either from ``scope`` or from ``PATRRONI_SCOPE`` environment variable
|
||||
- cluster_name - is set either from ``scope`` or from ``PATRONI_SCOPE`` environment variable
|
||||
- hot_standby: on
|
||||
|
||||
To be on the safe side parameters from the above lists are not written into ``postgresql.conf``, but passed as a list of arguments to the ``pg_ctl start`` which gives them the highest precedence, even above `ALTER SYSTEM <https://www.postgresql.org/docs/current/static/sql-altersystem.html>`__
|
||||
@@ -48,23 +53,24 @@ To be on the safe side parameters from the above lists are not written into ``po
|
||||
|
||||
When applying the local or dynamic configuration options, the following actions are taken:
|
||||
|
||||
- The node first checks if there is a postgresql.base.conf.
|
||||
- If it exists, it contains the renamed "original" configuration.
|
||||
- If it doesn't, the original postgresql.conf is taken and renamed to postgresql.base.conf.
|
||||
- The node first checks if there is a postgresql.base.conf or if the ``custom_conf`` parameter is set.
|
||||
- If the `custom_conf` parameter is set, it will take the file specified on it as a base configuration, ignoring `postgresql.base.conf` and `postgresql.conf`.
|
||||
- If the `custom_conf` parameter is not set and `postgresql.base.conf` exists, it contains the renamed "original" configuration and it will be used as a base configuration.
|
||||
- If there is no `custom_conf` nor `postgresql.base.conf`, the original postgresql.conf is taken and renamed to postgresql.base.conf.
|
||||
- The dynamic options (with the exceptions above) are dumped into the postgresql.conf and an include is set in
|
||||
postgresql.conf to postgresql.base.conf. Therefore, we would be able to apply new options without re-reading the configuration file to check if the include is present not.
|
||||
postgresql.conf to the used base configuration (either postgresql.base.conf or what is on ``custom_conf``). Therefore, we would be able to apply new options without re-reading the configuration file to check if the include is present not.
|
||||
- Some parameters that are essential for Patroni to manage the cluster are overridden using the command line.
|
||||
- If some of the options that require restart are changed (we should look at the context in pg_settings and at the actual
|
||||
values of those options), a pending_restart flag of a given node is set. This flag is reset on any restart.
|
||||
|
||||
The parameters would be applied in the following order (run-time are given the highest priority):
|
||||
|
||||
1. load parameters from file `postgresql.base.conf`
|
||||
1. load parameters from file `postgresql.base.conf` (or from a `custom_conf` file, if set)
|
||||
2. load parameters from file `postgresql.conf`
|
||||
3. load parameters from file `postgresql.auto.conf`
|
||||
4. run-time parameter using `-o --name=value`
|
||||
|
||||
This allows configuration for all the nodes (2), configuration for a specific node using `ALTER SYSTEM` (3) and ensures that parameters essential to the running of Patroni are enforced. (4)
|
||||
This allows configuration for all the nodes (2), configuration for a specific node using `ALTER SYSTEM` (3) and ensures that parameters essential to the running of Patroni are enforced (4), as well as leaves room for configuration tools that manage `postgresql.conf` directly without involving Patroni (1).
|
||||
|
||||
|
||||
Also, the following Patroni configuration options can be changed only dynamically:
|
||||
@@ -73,153 +79,11 @@ Also, the following Patroni configuration options can be changed only dynamicall
|
||||
- loop_wait: 10
|
||||
- retry_timeouts: 10
|
||||
- maximum_lag_on_failover: 1048576
|
||||
- max_timelines_history: 0
|
||||
- check_timeline: false
|
||||
- postgresql.use_slots: true
|
||||
|
||||
Upon changing these options, Patroni will read the relevant section of the configuration stored in DCS and change its
|
||||
run-time values.
|
||||
|
||||
Patroni nodes are dumping the state of the DCS options to disk upon for every change of the configuration into the file ``patroni.dynamic.json`` located in the Postgres data directory. Only the master is allowed to restore these options from the on-disk dump if these are completely absent from the DCS or if they are invalid.
|
||||
|
||||
REST API
|
||||
========
|
||||
|
||||
We provide a REST API endpoint for working with dynamic configuration.
|
||||
|
||||
GET /config
|
||||
-----------
|
||||
Get current version of dynamic configuration.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s localhost:8008/config | jq .
|
||||
{
|
||||
"ttl": 30,
|
||||
"loop_wait": 10,
|
||||
"retry_timeout": 10,
|
||||
"maximum_lag_on_failover": 1048576,
|
||||
"postgresql": {
|
||||
"use_slots": true,
|
||||
"use_pg_rewind": true,
|
||||
"parameters": {
|
||||
"hot_standby": "on",
|
||||
"wal_log_hints": "on",
|
||||
"wal_keep_segments": 8,
|
||||
"wal_level": "hot_standby",
|
||||
"max_wal_senders": 5,
|
||||
"max_replication_slots": 5,
|
||||
"max_connections": "100"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
PATCH /config
|
||||
-------------
|
||||
Change existing configuration.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s -XPATCH -d \
|
||||
'{"loop_wait":5,"ttl":20,"postgresql":{"parameters":{"max_connections":"101"}}}' \
|
||||
http://localhost:8008/config | jq .
|
||||
{
|
||||
"ttl": 20,
|
||||
"loop_wait": 5,
|
||||
"maximum_lag_on_failover": 1048576,
|
||||
"retry_timeout": 10,
|
||||
"postgresql": {
|
||||
"use_slots": true,
|
||||
"use_pg_rewind": true,
|
||||
"parameters": {
|
||||
"hot_standby": "on",
|
||||
"wal_log_hints": "on",
|
||||
"wal_keep_segments": 8,
|
||||
"wal_level": "hot_standby",
|
||||
"max_wal_senders": 5,
|
||||
"max_replication_slots": 5,
|
||||
"max_connections": "101"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
The above REST API call patches the existing configuration and returns the new configuration.
|
||||
|
||||
Let's check that the node processed this configuration. First of all it should start printing log lines every 5 seconds (loop_wait=5). The change of "max_connections" requires a restart, so the "restart_pending" flag should be exposed:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/patroni | jq .
|
||||
{
|
||||
"pending_restart": true,
|
||||
"database_system_identifier": "6287881213849985952",
|
||||
"postmaster_start_time": "2016-06-13 13:13:05.211 CEST",
|
||||
"xlog": {
|
||||
"location": 2197818976
|
||||
},
|
||||
"patroni": {
|
||||
"scope": "batman",
|
||||
"version": "1.0"
|
||||
},
|
||||
"state": "running",
|
||||
"role": "master",
|
||||
"server_version": 90503
|
||||
}
|
||||
|
||||
Removing parameters:
|
||||
|
||||
If you want to remove (reset) some setting just patch it with ``null``:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s -XPATCH -d \
|
||||
'{"postgresql":{"parameters":{"max_connections":null}}}' \
|
||||
http://localhost:8008/config | jq .
|
||||
{
|
||||
"ttl": 20,
|
||||
"loop_wait": 5,
|
||||
"retry_timeout": 10,
|
||||
"maximum_lag_on_failover": 1048576,
|
||||
"postgresql": {
|
||||
"use_slots": true,
|
||||
"use_pg_rewind": true,
|
||||
"parameters": {
|
||||
"hot_standby": "on",
|
||||
"unix_socket_directories": ".",
|
||||
"wal_keep_segments": 8,
|
||||
"wal_level": "hot_standby",
|
||||
"wal_log_hints": "on",
|
||||
"max_wal_senders": 5,
|
||||
"max_replication_slots": 5
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Above call removes ``postgresql.parameters.max_connections`` from the dynamic configuration.
|
||||
|
||||
PUT /config
|
||||
-----------
|
||||
|
||||
It's also possible to perform the full rewrite of an existing dynamic configuration unconditionally:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s -XPUT -d \
|
||||
'{"maximum_lag_on_failover":1048576,"retry_timeout":10,"postgresql":{"use_slots":true,"use_pg_rewind":true,"parameters":{"hot_standby":"on","wal_log_hints":"on","wal_keep_segments":8,"wal_level":"hot_standby","unix_socket_directories":".","max_wal_senders":5}},"loop_wait":3,"ttl":20}' \
|
||||
http://localhost:8008/config | jq .
|
||||
{
|
||||
"ttl": 20,
|
||||
"maximum_lag_on_failover": 1048576,
|
||||
"retry_timeout": 10,
|
||||
"postgresql": {
|
||||
"use_slots": true,
|
||||
"parameters": {
|
||||
"hot_standby": "on",
|
||||
"unix_socket_directories": ".",
|
||||
"wal_keep_segments": 8,
|
||||
"wal_level": "hot_standby",
|
||||
"wal_log_hints": "on",
|
||||
"max_wal_senders": 5
|
||||
},
|
||||
"use_pg_rewind": true
|
||||
},
|
||||
"loop_wait": 3
|
||||
}
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
.. _existing_data:
|
||||
|
||||
Convert a Standalone to a Patroni Cluster
|
||||
=========================================
|
||||
|
||||
This section describes the process for converting a standalone PostgreSQL instance into a Patroni cluster.
|
||||
|
||||
To deploy a Patroni cluster without using a pre-existing PostgreSQL instance, see :ref:`Running and Configuring <running_configuring>` instead.
|
||||
|
||||
Procedure
|
||||
---------
|
||||
|
||||
A Patroni cluster can be started with a data directory from a single-node PostgreSQL database. This is achieved by following closely these steps:
|
||||
|
||||
1. Manually start PostgreSQL daemon
|
||||
2. Create Patroni superuser and replication users as defined in the :ref:`authentication <postgresql_settings>` section of the Patroni configuration. If this user is created in SQL, the following queries achieve this:
|
||||
|
||||
.. code-block:: sql
|
||||
|
||||
CREATE USER $PATRONI_SUPERUSER_USERNAME WITH SUPERUSER ENCRYPTED PASSWORD '$PATRONI_SUPERUSER_PASSWORD';
|
||||
CREATE USER $PATRONI_REPLICATION_USERNAME WITH REPLICATION ENCRYPTED PASSWORD '$PATRONI_REPLICATION_PASSWORD';
|
||||
|
||||
3. Start Patroni (e.g. ``patroni /etc/patroni/patroni.yml``). It automatically detects that PostgreSQL daemon is already running but its configuration might be out-of-date.
|
||||
4. Ask Patroni to restart the node with ``patronictl restart cluster-name node-name``. This step is only required if PostgreSQL configuration is out-of-date.
|
||||
|
||||
Major Upgrade of PostgreSQL Version
|
||||
===================================
|
||||
|
||||
The only possible way to do a major upgrade currently is:
|
||||
|
||||
1. Stop Patroni
|
||||
2. Upgrade PostgreSQL binaries and perform `pg_upgrade <https://www.postgresql.org/docs/current/pgupgrade.html>`_ on the master node
|
||||
3. Update patroni.yml
|
||||
4. Remove the initialize key from DCS or wipe complete cluster state from DCS. The second one could be achieved by running ``patronictl remove <cluster-name>``. It is necessary because pg_upgrade runs initdb which actually creates a new database with a new PostgreSQL system identifier.
|
||||
5. If you wiped the cluster state in the previous step, you may wish to copy patroni.dynamic.json from old data dir to the new one. It will help you to retain some PostgreSQL parameters you had set before.
|
||||
6. Start Patroni on the master node.
|
||||
7. Upgrade PostgreSQL binaries, update patroni.yml and wipe the data_dir on standby nodes.
|
||||
8. Start Patroni on the standby nodes and wait for the replication to complete.
|
||||
|
||||
Running pg_upgrade on standby nodes is not supported by PostgreSQL. If you know what you are doing, you can try the rsync procedure described in https://www.postgresql.org/docs/current/pgupgrade.html instead of wiping data_dir on standby nodes. The safest way is however to let Patroni replicate the data for you.
|
||||
|
||||
FAQ
|
||||
---
|
||||
|
||||
- During Patroni startup, Patroni complains that it cannot bind to the PostgreSQL port.
|
||||
|
||||
You need to verify ``listen_addresses`` and ``port`` in ``postgresql.conf`` and ``postgresql.listen`` in ``patroni.yml``. Don't forget that ``pg_hba.conf`` should allow such access.
|
||||
|
||||
- After asking Patroni to restart the node, PostgreSQL displays the error message ``could not open configuration file "/etc/postgresql/10/main/pg_hba.conf": No such file or directory``
|
||||
|
||||
It can mean various things depending on how you manage PostgreSQL configuration. If you specified `postgresql.config_dir`, Patroni generates the ``pg_hba.conf`` based on the settings in the :ref:`bootstrap <bootstrap_settings>` section only when it bootstraps a new cluster. In this scenario the ``PGDATA`` was not empty, therefore no bootstrap happened. This file must exist beforehand.
|
||||
@@ -0,0 +1,148 @@
|
||||
// Graphviz source for ha_loop_diagram.png
|
||||
// recompile with:
|
||||
// dot -Tpng ha_loop_diagram.dot -o ha_loop_diagram.png
|
||||
|
||||
digraph G {
|
||||
rankdir=TB;
|
||||
fontname="sans-serif";
|
||||
penwidth="0.3";
|
||||
layout="dot";
|
||||
newrank=true;
|
||||
edge [fontname="sans-serif",
|
||||
fontsize=12,
|
||||
color=black,
|
||||
fontcolor=black];
|
||||
node [fontname=serif,
|
||||
fontsize=12,
|
||||
fillcolor=white,
|
||||
color=black,
|
||||
fontcolor=black,
|
||||
style=filled];
|
||||
"start" [label=Start, shape="rectangle", fillcolor="green"]
|
||||
"start" -> "load_cluster_from_dcs";
|
||||
"update_member" [label="Persist node state in DCS"]
|
||||
"update_member" -> "start"
|
||||
|
||||
subgraph cluster_run_cycle {
|
||||
label="run_cycle"
|
||||
"load_cluster_from_dcs" [label="Load cluster from DCS"];
|
||||
"touch_member" [label="Persist node in DCS"];
|
||||
"cluster.has_member" [shape="diamond", label="Is node registered on DCS?"]
|
||||
"cluster.has_member" -> "touch_member" [label="no" color="red"]
|
||||
"long_action_in_progress?" [shape="diamond" label="Is the PostgreSQL currently being\nstopping/starting/restarting/reinitializing?"]
|
||||
"load_cluster_from_dcs" -> "cluster.has_member";
|
||||
"touch_member" -> "long_action_in_progress?";
|
||||
"cluster.has_member" -> "long_action_in_progress?" [label="yes" color="green"];
|
||||
"long_action_in_progress?" -> "recovering?" [label="no" color="red"]
|
||||
"recovering?" [label="Was cluster recovering and failed?", shape="diamond"];
|
||||
"recovering?" -> "post_recover" [label="yes" color="green"];
|
||||
"recovering?" -> "data_directory_empty" [label="no" color="red"];
|
||||
"post_recover" [label="Remove leader key (if I was the leader)"];
|
||||
"data_directory_empty" [label="Is data folder empty?", shape="diamond"];
|
||||
"data_directory_empty" -> "cluster_initialize" [label="no" color="red"];
|
||||
"data_belongs_to_cluster" [label="Does data dir belong to cluster?", shape="diamond"];
|
||||
"data_belongs_to_cluster" -> "exit" [label="no" color="red"];
|
||||
"data_belongs_to_cluster" -> "is_healthy" [label="yes" color="green"]
|
||||
"exit" [label="Fail and exit", fillcolor=red];
|
||||
"cluster_initialize" [label="Is cluster initialized on DCS?" shape="diamond"]
|
||||
"cluster_initialize" -> "cluster.has_leader" [label="no" color="red"]
|
||||
"cluster.has_leader" [label="Does the cluster has leader?", shape="diamond"]
|
||||
"cluster.has_leader" -> "dcs.initialize" [label="no", color="red"]
|
||||
"cluster.has_leader" -> "is_healthy" [label="yes", color="green"]
|
||||
"cluster_initialize" -> "data_belongs_to_cluster" [label="yes" color="green"]
|
||||
"dcs.initialize" [label="Initialize new cluster"];
|
||||
"dcs.initialize" -> "is_healthy"
|
||||
"is_healthy" [label="Is node healthy?\n(running Postgres)", shape="diamond"];
|
||||
"recover" [label="Start as read-only\nand set Recover flag"]
|
||||
"is_healthy" -> "recover" [label="no" color="red"];
|
||||
"is_healthy" -> "cluster.is_unlocked" [label="yes" color="green"];
|
||||
"cluster.is_unlocked" [label="Does the cluster has a leader?", shape="diamond"]
|
||||
}
|
||||
|
||||
"post_recover" -> "update_member"
|
||||
"recover" -> "update_member"
|
||||
"long_action_in_progress?" -> "async_has_lock?" [label="yes" color="green"];
|
||||
"cluster.is_unlocked" -> "unhealthy_is_healthiest" [label="no" color="red"]
|
||||
"cluster.is_unlocked" -> "healthy_has_lock" [label="yes" color="green"]
|
||||
"data_directory_empty" -> "bootstrap.is_unlocked" [label="yes" color="green"]
|
||||
|
||||
subgraph cluster_async {
|
||||
label = "Long action in progress\n(Start/Stop/Restart/Reinitialize)"
|
||||
"async_has_lock?" [label="Do I have the leader lock?", shape="diamond"]
|
||||
"async_update_lock" [label="Renew leader lock"]
|
||||
"async_has_lock?" -> "async_update_lock" [label="yes" color="green"]
|
||||
}
|
||||
"async_update_lock" -> "update_member"
|
||||
"async_has_lock?" -> "update_member" [label="no" color="red"]
|
||||
|
||||
subgraph cluster_bootstrap {
|
||||
label = "Node bootstrap";
|
||||
"bootstrap.is_unlocked" [label="Does the cluster has a leader?", shape="diamond"]
|
||||
"bootstrap.is_initialized" [label="Does the cluster has an initialize key?", shape="diamond"]
|
||||
"bootstrap.is_unlocked" -> "bootstrap.is_initialized" [label="no" color="red"]
|
||||
"bootstrap.is_unlocked" -> "bootstrap.select_node" [label="yes" color="green"]
|
||||
"bootstrap.select_node" [label="Select a node to take a backup from"]
|
||||
"bootstrap.do_bootstrap" [label="Run pg_basebackup\n(async)"]
|
||||
"bootstrap.select_node" -> "bootstrap.do_bootstrap"
|
||||
"bootstrap.is_initialized" -> "bootstrap.initialization_race" [label="no" color="red"]
|
||||
"bootstrap.is_initialized" -> "bootstrap.wait_for_leader" [label="yes" color="green"]
|
||||
"bootstrap.initialization_race" [label="Race for initialize key"]
|
||||
"bootstrap.initialization_race" -> "bootstrap.won_initialize_race?"
|
||||
"bootstrap.won_initialize_race?" [label="Do I won initialize race?", shape="diamond"]
|
||||
"bootstrap.won_initialize_race?" -> "bootstrap.initdb_and_start" [label="yes" color="green"]
|
||||
"bootstrap.won_initialize_race?" -> "bootstrap.wait_for_leader" [label="no" color="red"]
|
||||
"bootstrap.wait_for_leader" [label="Need to wait for leader key"]
|
||||
"bootstrap.initdb_and_start" [label="Run initdb, start postgres and create roles"]
|
||||
"bootstrap.initdb_and_start" -> "bootstrap.success?"
|
||||
"bootstrap.success?" [label="Success", shape="diamond"]
|
||||
"bootstrap.success?" -> "bootstrap.take_leader_key" [label="yes" color="green"]
|
||||
"bootstrap.success?" -> "bootstrap.clean" [label="no" color="red"]
|
||||
"bootstrap.clean" [label="Remove initialize key from DCS\nand data directory from filesystem"]
|
||||
"bootstrap.take_leader_key" [label="Take a leader key in DCS"]
|
||||
}
|
||||
|
||||
"bootstrap.do_bootstrap" -> "update_member"
|
||||
"bootstrap.wait_for_leader" -> "update_member"
|
||||
"bootstrap.clean" -> "update_member"
|
||||
"bootstrap.take_leader_key" -> "update_member"
|
||||
|
||||
subgraph cluster_process_healthy_cluster {
|
||||
label = "process_healthy_cluster"
|
||||
"healthy_has_lock" [label="Am I the owner of the leader lock?", shape=diamond]
|
||||
"healthy_is_leader" [label="Is Postgres running as master?", shape=diamond]
|
||||
"healthy_no_lock" [label="Follow the leader (async,\ncreate/update recovery.conf and restart if necessary)"]
|
||||
"healthy_has_lock" -> "healthy_no_lock" [label="no" color="red"]
|
||||
"healthy_has_lock" -> "healthy_update_leader_lock" [label="yes" color="green"]
|
||||
"healthy_update_leader_lock" [label="Try to update leader lock"]
|
||||
"healthy_update_leader_lock" -> "healthy_update_success"
|
||||
"healthy_update_success" [label="Success?", shape=diamond]
|
||||
"healthy_update_success" -> "healthy_is_leader" [label="yes" color="green"]
|
||||
"healthy_update_success" -> "healthy_demote" [label="no" color="red"]
|
||||
"healthy_demote" [label="Demote (async,\nrestart in read-only)"]
|
||||
"healthy_failover" [label="Promote Postgres to master"]
|
||||
"healthy_is_leader" -> "healthy_failover" [label="no" color="red"]
|
||||
}
|
||||
"healthy_demote" -> "update_member"
|
||||
"healthy_is_leader" -> "update_member" [label="yes" color="green"]
|
||||
"healthy_failover" -> "update_member"
|
||||
"healthy_no_lock" -> "update_member"
|
||||
|
||||
subgraph cluster_process_unhealthy_cluster {
|
||||
label = "process_unhealthy_cluster"
|
||||
"unhealthy_is_healthiest" [label="Am I the healthiest node?", shape="diamond"]
|
||||
"unhealthy_is_healthiest" -> "unhealthy_leader_race" [label="yes", color="green"]
|
||||
"unhealthy_leader_race" [label="Try to create leader key"]
|
||||
"unhealthy_leader_race" -> "unhealthy_acquire_lock"
|
||||
"unhealthy_acquire_lock" [label="Was I able to get the lock?", shape="diamond"]
|
||||
"unhealthy_is_leader" [label="Is Postgres running as master?", shape=diamond]
|
||||
"unhealthy_acquire_lock" -> "unhealthy_is_leader" [label="yes" color="green"]
|
||||
"unhealthy_is_leader" -> "unhealthy_promote" [label="no" color="red"]
|
||||
"unhealthy_promote" [label="Promote to master"]
|
||||
"unhealthy_is_healthiest" -> "unhealthy_follow" [label="no" color="red"]
|
||||
"unhealthy_follow" [label="try to follow somebody else()"]
|
||||
"unhealthy_acquire_lock" -> "unhealthy_follow" [label="no" color="red"]
|
||||
}
|
||||
"unhealthy_follow" -> "update_member"
|
||||
"unhealthy_promote" -> "update_member"
|
||||
"unhealthy_is_leader" -> "update_member" [label="yes" color="green"]
|
||||
}
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 507 KiB |
@@ -0,0 +1,42 @@
|
||||
.. Patroni documentation master file, created by
|
||||
sphinx-quickstart on Mon Dec 19 16:54:09 2016.
|
||||
You can adapt this file completely to your liking, but it should at least
|
||||
contain the root `toctree` directive.
|
||||
|
||||
Introduction
|
||||
============
|
||||
|
||||
Patroni is a template for you to create your own customized, high-availability solution using Python and - for maximum accessibility - a distributed configuration store like `ZooKeeper <https://zookeeper.apache.org/>`__, `etcd <https://github.com/coreos/etcd>`__, `Consul <https://github.com/hashicorp/consul>`__ or `Kubernetes <https://kubernetes.io>`__. Database engineers, DBAs, DevOps engineers, and SREs who are looking to quickly deploy HA PostgreSQL in the datacenter-or anywhere else-will hopefully find it useful.
|
||||
|
||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
|
||||
|
||||
Currently supported PostgreSQL versions: 9.3 to 14.
|
||||
|
||||
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the :ref:`Kubernetes <kubernetes>` chapter of the Patroni documentation.
|
||||
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 2
|
||||
:caption: Contents:
|
||||
|
||||
README
|
||||
dynamic_configuration
|
||||
rest_api
|
||||
existing_data
|
||||
ENVIRONMENT
|
||||
SETTINGS
|
||||
security
|
||||
replica_bootstrap
|
||||
replication_modes
|
||||
pause
|
||||
kubernetes
|
||||
watchdog
|
||||
releases
|
||||
CONTRIBUTING
|
||||
|
||||
Indices and tables
|
||||
==================
|
||||
|
||||
* :ref:`genindex`
|
||||
* :ref:`modindex`
|
||||
* :ref:`search`
|
||||
@@ -0,0 +1,53 @@
|
||||
.. _kubernetes:
|
||||
|
||||
Using Patroni with Kubernetes
|
||||
=============================
|
||||
|
||||
Patroni can use Kubernetes objects in order to store the state of the cluster and manage the leader key. That makes it
|
||||
capable of operating Postgres in Kubernetes environment without any consistency store, namely, one doesn't
|
||||
need to run an extra Etcd deployment. There are two different type of Kubernetes objects Patroni can use to store the
|
||||
leader and the configuration keys, they are configured with the `kubernetes.use_endpoints` or `PATRONI_KUBERNETES_USE_ENDPOINTS`
|
||||
environment variable.
|
||||
|
||||
Use Endpoints
|
||||
-------------
|
||||
|
||||
Despite the fact that this is the recommended mode, it is turned off by default for compatibility reasons. When it is on, Patroni stores
|
||||
the cluster configuration and the leader key in the `metadata: annotations` fields of the respective `Endpoints` it creates.
|
||||
Changing the leader is safer than when using `ConfigMaps`, since both the annotations, containing the leader information, and the actual addresses
|
||||
pointing to the running leader pod are updated simultaneously in one go.
|
||||
|
||||
Use ConfigMaps
|
||||
--------------
|
||||
|
||||
In this mode, Patroni will create ConfigMaps instead of Endpoints and store keys inside meta-data of those ConfigMaps.
|
||||
Changing the leader takes at least two updates, one to the leader ConfigMap and another to the respective Endpoint.
|
||||
|
||||
There are two ways to direct the traffic to the Postgres master:
|
||||
|
||||
- use the `callback script <https://github.com/zalando/patroni/blob/master/kubernetes/callback.py>`_ provided by Patroni
|
||||
- configure the Kubernetes Postgres service to use the label selector with the `role_label` (configured in patroni configuration).
|
||||
|
||||
Note that in some cases, for instance, when running on OpenShift, there is no alternative to using ConfigMaps.
|
||||
|
||||
Configuration
|
||||
-------------
|
||||
|
||||
Patroni Kubernetes :ref:`settings <kubernetes_settings>` and :ref:`environment variables <kubernetes_environment>` are described in the general chapters of the documentation.
|
||||
|
||||
Examples
|
||||
--------
|
||||
|
||||
- The `kubernetes <https://github.com/zalando/patroni/tree/master/kubernetes>`__ folder of the Patroni repository contains
|
||||
examples of the Docker image, the Kubernetes manifest and the callback script in order to test Patroni Kubernetes setup.
|
||||
Note that in the current state it will not be able to use PersistentVolumes because of permission issues.
|
||||
|
||||
- You can find the full-featured Docker image that can use Persistent Volumes in the
|
||||
`Spilo Project <https://github.com/zalando/spilo>`_.
|
||||
|
||||
- There is also a `Helm chart <https://github.com/kubernetes/charts/tree/master/incubator/patroni>`_
|
||||
to deploy the Spilo image configured with Patroni running using Kubernetes.
|
||||
|
||||
- In order to run your database clusters at scale using Patroni and Spilo, take a look at the
|
||||
`postgres-operator <https://github.com/zalando-incubator/postgres-operator>`_ project. It implements the operator pattern
|
||||
to manage Spilo clusters.
|
||||
@@ -0,0 +1,35 @@
|
||||
.. _pause:
|
||||
|
||||
Pause/Resume mode for the cluster
|
||||
=================================
|
||||
|
||||
The goal
|
||||
--------
|
||||
|
||||
Under certain circumstances Patroni needs to temporary step down from managing the cluster, while still retaining the cluster state in DCS. Possible use cases are uncommon activities on the cluster, such as major version upgrades or corruption recovery. During those activities nodes are often started and stopped for the reason unknown to Patroni, some nodes can be even temporary promoted, violating the assumption of running only one master. Therefore, Patroni needs to be able to "detach" from the running cluster, implementing an equivalent of the maintenance mode in Pacemaker.
|
||||
|
||||
|
||||
|
||||
The implementation
|
||||
------------------
|
||||
|
||||
When Patroni runs in a paused mode, it does not change the state of PostgreSQL, except for the following cases:
|
||||
|
||||
- For each node, the member key in DCS is updated with the current information about the cluster. This causes Patroni to run read-only queries on a member node if the member is running.
|
||||
|
||||
- For the Postgres master with the leader lock Patroni updates the lock. If the node with the leader lock stops being the master (i.e. is demoted manually), Patroni will release the lock instead of promoting the node back.
|
||||
|
||||
- Manual unscheduled restart, reinitialize and manual failover are allowed. Manual failover is only allowed if the node to failover to is specified. In the paused mode, manual failover does not require a running master node.
|
||||
|
||||
- If 'parallel' masters are detected by Patroni, it emits a warning, but does not demote the masters without the leader lock.
|
||||
|
||||
- If there is no leader lock in the cluster, the running master acquires the lock. If there is more than one master node, then the first master to acquire the lock wins. If there are no masters altogether, Patroni does not try to promote any replicas. There is an exception in this rule: if there is no leader lock because the old master has demoted itself due to the manual promotion, then only the candidate node mentioned in the promotion request may take the leader lock. When the new leader lock is granted (i.e. after promoting a replica manually), Patroni makes sure the replicas that were streaming from the previous leader will switch to the new one.
|
||||
|
||||
- When Postgres is stopped, Patroni does not try to start it. When Patroni is stopped, it does not try to stop the Postgres instance it is managing.
|
||||
|
||||
User guide
|
||||
----------
|
||||
|
||||
``patronictl`` supports ``pause`` and ``resume`` commands.
|
||||
|
||||
One can also issue a ``PATCH`` request to the ``{namespace}/{cluster}/config`` key with ``{"pause": true/false/null}``
|
||||
+2433
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,215 @@
|
||||
Replica imaging and bootstrap
|
||||
=============================
|
||||
|
||||
Patroni allows customizing creation of a new replica. It also supports defining what happens when the new empty cluster
|
||||
is being bootstrapped. The distinction between two is well defined: Patroni creates replicas only if the ``initialize``
|
||||
key is present in DCS for the cluster. If there is no ``initialize`` key - Patroni calls bootstrap exclusively on the
|
||||
first node that takes the initialize key lock.
|
||||
|
||||
.. _custom_bootstrap:
|
||||
|
||||
Bootstrap
|
||||
---------
|
||||
|
||||
PostgreSQL provides ``initdb`` command to initialize a new cluster and Patroni calls it by default. In certain cases,
|
||||
particularly when creating a new cluster as a copy of an existing one, it is necessary to replace a built-in method with
|
||||
custom actions. Patroni supports executing user-defined scripts to bootstrap new clusters, supplying some required
|
||||
arguments to them, i.e. the name of the cluster and the path to the data directory. This is configured in the
|
||||
``bootstrap`` section of the Patroni configuration. For example:
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
bootstrap:
|
||||
method: <custom_bootstrap_method_name>
|
||||
<custom_bootstrap_method_name>:
|
||||
command: <path_to_custom_bootstrap_script> [param1 [, ...]]
|
||||
keep_existing_recovery_conf: False
|
||||
no_params: False
|
||||
recovery_conf:
|
||||
recovery_target_action: promote
|
||||
recovery_target_timeline: latest
|
||||
restore_command: <method_specific_restore_command>
|
||||
|
||||
|
||||
Each bootstrap method must define at least a ``name`` and a ``command``. A special ``initdb`` method is available to trigger
|
||||
the default behavior, in which case ``method`` parameter can be omitted altogether. The ``command`` can be specified using either
|
||||
an absolute path, or the one relative to the ``patroni`` command location. In addition to the fixed parameters defined
|
||||
in the configuration files, Patroni supplies two cluster-specific ones:
|
||||
|
||||
--scope
|
||||
Name of the cluster to be bootstrapped
|
||||
--datadir
|
||||
Path to the data directory of the cluster instance to be bootstrapped
|
||||
|
||||
Passing these two additional flags can be disabled by setting a special ``no_params`` parameter to ``True``.
|
||||
|
||||
If the bootstrap script returns 0, Patroni tries to configure and start the PostgreSQL instance produced by it. If any
|
||||
of the intermediate steps fail, or the script returns a non-zero value, Patroni assumes that the bootstrap has failed,
|
||||
cleans up after itself and releases the initialize lock to give another node the opportunity to bootstrap.
|
||||
|
||||
If a ``recovery_conf`` block is defined in the same section as the custom bootstrap method, Patroni will generate a
|
||||
``recovery.conf`` before starting the newly bootstrapped instance. Typically, such recovery.conf should contain at least
|
||||
one of the ``recovery_target_*`` parameters, together with the ``recovery_target_timeline`` set to ``promote``.
|
||||
|
||||
If ``keep_existing_recovery_conf`` is defined and set to ``True``, Patroni will not remove the existing ``recovery.conf`` file if it exists.
|
||||
This is useful when bootstrapping from a backup with tools like pgBackRest that generate the appropriate ``recovery.conf`` for you.
|
||||
|
||||
.. note:: Bootstrap methods are neither chained, nor fallen-back to the default one in case the primary one fails
|
||||
|
||||
|
||||
.. _custom_replica_creation:
|
||||
|
||||
Building replicas
|
||||
-----------------
|
||||
|
||||
Patroni uses tried and proven ``pg_basebackup`` in order to create new replicas. One downside of it is that it requires
|
||||
a running master node. Another one is the lack of 'on-the-fly' compression for the backup data and no built-in cleanup
|
||||
for outdated backup files. Some people prefer other backup solutions, such as ``WAL-E``, ``pgBackRest``, ``Barman`` and
|
||||
others, or simply roll their own scripts. In order to accommodate all those use-cases Patroni supports running custom
|
||||
scripts to clone a new replica. Those are configured in the ``postgresql`` configuration block:
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
postgresql:
|
||||
create_replica_methods:
|
||||
- <method name>
|
||||
<method name>:
|
||||
command: <command name>
|
||||
keep_data: True
|
||||
no_params: True
|
||||
no_master: 1
|
||||
|
||||
example: wal_e
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
postgresql:
|
||||
create_replica_methods:
|
||||
- wal_e
|
||||
- basebackup
|
||||
wal_e:
|
||||
command: patroni_wale_restore
|
||||
no_master: 1
|
||||
envdir: {{WALE_ENV_DIR}}
|
||||
use_iam: 1
|
||||
basebackup:
|
||||
max-rate: '100M'
|
||||
|
||||
example: pgbackrest
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
postgresql:
|
||||
create_replica_methods:
|
||||
- pgbackrest
|
||||
- basebackup
|
||||
pgbackrest:
|
||||
command: /usr/bin/pgbackrest --stanza=<scope> --delta restore
|
||||
keep_data: True
|
||||
no_params: True
|
||||
basebackup:
|
||||
max-rate: '100M'
|
||||
|
||||
|
||||
The ``create_replica_methods`` defines available replica creation methods and the order of executing them. Patroni will
|
||||
stop on the first one that returns 0. Each method should define a separate section in the configuration file, listing the command
|
||||
to execute and any custom parameters that should be passed to that command. All parameters will be passed in a
|
||||
``--name=value`` format. Besides user-defined parameters, Patroni supplies a couple of cluster-specific ones:
|
||||
|
||||
--scope
|
||||
Which cluster this replica belongs to
|
||||
--datadir
|
||||
Path to the data directory of the replica
|
||||
--role
|
||||
Always 'replica'
|
||||
--connstring
|
||||
Connection string to connect to the cluster member to clone from (master or other replica). The user in the
|
||||
connection string can execute SQL and replication protocol commands.
|
||||
|
||||
A special ``no_master`` parameter, if defined, allows Patroni to call the replica creation method even if there is no
|
||||
running master or replicas. In that case, an empty string will be passed in a connection string. This is useful for
|
||||
restoring the formerly running cluster from the binary backup.
|
||||
|
||||
A special ``keep_data`` parameter, if defined, will instruct Patroni to not clean PGDATA folder before calling restore.
|
||||
|
||||
A special ``no_params`` parameter, if defined, restricts passing parameters to custom command.
|
||||
|
||||
A ``basebackup`` method is a special case: it will be used if
|
||||
``create_replica_methods`` is empty, although it is possible
|
||||
to list it explicitly among the ``create_replica_methods`` methods. This method initializes a new replica with the
|
||||
``pg_basebackup``, the base backup is taken from the master unless there are replicas with ``clonefrom`` tag, in which case one
|
||||
of such replicas will be used as the origin for pg_basebackup. It works without any configuration; however, it is
|
||||
possible to specify a ``basebackup`` configuration section. Same rules as with the other method configuration apply,
|
||||
namely, only long (with --) options should be specified there. Not all parameters make sense, if you override a connection
|
||||
string or provide an option to created tar-ed or compressed base backups, patroni won't be able to make a replica out
|
||||
of it. There is no validation performed on the names or values of the parameters passed to the ``basebackup`` section.
|
||||
Also note that in case symlinks are used for the WAL folder it is up to the user to specify the correct ``--waldir``
|
||||
path as an option, so that after replica buildup or re-initialization the symlink would persist. This option is supported
|
||||
only since v10 though.
|
||||
|
||||
You can specify basebackup parameters as either a map (key-value pairs) or a list of elements, where each element
|
||||
could be either a key-value pair or a single key (for options that does not receive any values, for instance, ``--verbose``).
|
||||
Consider those 2 examples:
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
postgresql:
|
||||
basebackup:
|
||||
max-rate: '100M'
|
||||
checkpoint: 'fast'
|
||||
|
||||
and
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
postgresql:
|
||||
basebackup:
|
||||
- verbose
|
||||
- max-rate: '100M'
|
||||
- waldir: /pg-wal-mount/external-waldir
|
||||
|
||||
If all replica creation methods fail, Patroni will try again all methods in order during the next event loop cycle.
|
||||
|
||||
.. _standby_cluster:
|
||||
|
||||
Standby cluster
|
||||
---------------
|
||||
|
||||
Another available option is to run a "standby cluster", that contains only of
|
||||
standby nodes replicating from some remote master. This type of clusters has:
|
||||
|
||||
* "standby leader", that behaves pretty much like a regular cluster leader,
|
||||
except it replicates from a remote master.
|
||||
|
||||
* cascade replicas, that are replicating from standby leader.
|
||||
|
||||
Standby leader holds and updates a leader lock in DCS. If the leader lock
|
||||
expires, cascade replicas will perform an election to choose another leader
|
||||
from the standbys.
|
||||
|
||||
For the sake of flexibility, you can specify methods of creating a replica and
|
||||
recovery WAL records when a cluster is in the "standby mode" by providing
|
||||
`create_replica_methods` key in `standby_cluster` section. It is distinct from
|
||||
creating replicas, when cluster is detached and functions as a normal cluster,
|
||||
which is controlled by `create_replica_methods` in `postgresql` section. Both
|
||||
"standby" and "normal" `create_replica_methods` reference keys in `postgresql`
|
||||
section.
|
||||
|
||||
To configure such cluster you need to specify the section ``standby_cluster``
|
||||
in a patroni configuration:
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
bootstrap:
|
||||
dcs:
|
||||
standby_cluster:
|
||||
host: 1.2.3.4
|
||||
port: 5432
|
||||
primary_slot_name: patroni
|
||||
create_replica_methods:
|
||||
- basebackup
|
||||
|
||||
Note, that these options will be applied only once during cluster bootstrap,
|
||||
and the only way to change them afterwards is through DCS.
|
||||
|
||||
If you use replication slots on the standby cluster, you must also create the corresponding replication slot on the primary cluster. It will not be done automatically by the standby cluster implementation. You can use Patroni's permanent replication slots feature on the primary cluster to maintain a replication slot with the same name as ``primary_slot_name``, or its default value if ``primary_slot_name`` is not provided.
|
||||
@@ -0,0 +1,84 @@
|
||||
.. _replication_modes:
|
||||
|
||||
=================
|
||||
Replication modes
|
||||
=================
|
||||
|
||||
Patroni uses PostgreSQL streaming replication. For more information about streaming replication, see the `Postgres documentation <http://www.postgresql.org/docs/current/static/warm-standby.html#STREAMING-REPLICATION>`__. By default Patroni configures PostgreSQL for asynchronous replication. Choosing your replication schema is dependent on your business considerations. Investigate both async and sync replication, as well as other HA solutions, to determine which solution is best for you.
|
||||
|
||||
Asynchronous mode durability
|
||||
----------------------------
|
||||
|
||||
In asynchronous mode the cluster is allowed to lose some committed transactions to ensure availability. When the primary server fails or becomes unavailable for any other reason Patroni will automatically promote a sufficiently healthy standby to primary. Any transactions that have not been replicated to that standby remain in a "forked timeline" on the primary, and are effectively unrecoverable [1]_.
|
||||
|
||||
The amount of transactions that can be lost is controlled via ``maximum_lag_on_failover`` parameter. Because the primary transaction log position is not sampled in real time, in reality the amount of lost data on failover is worst case bounded by ``maximum_lag_on_failover`` bytes of transaction log plus the amount that is written in the last ``ttl`` seconds (``loop_wait``/2 seconds in the average case). However typical steady state replication delay is well under a second.
|
||||
|
||||
By default, when running leader elections, Patroni does not take into account the current timeline of replicas, what in some cases could be undesirable behavior. You can prevent the node not having the same timeline as a former master become the new leader by changing the value of ``check_timeline`` parameter to ``true``.
|
||||
|
||||
PostgreSQL synchronous replication
|
||||
----------------------------------
|
||||
|
||||
You can use Postgres's `synchronous replication <http://www.postgresql.org/docs/current/static/warm-standby.html#SYNCHRONOUS-REPLICATION>`__ with Patroni. Synchronous replication ensures consistency across a cluster by confirming that writes are written to a secondary before returning to the connecting client with a success. The cost of synchronous replication: reduced throughput on writes. This throughput will be entirely based on network performance.
|
||||
|
||||
In hosted datacenter environments (like AWS, Rackspace, or any network you do not control), synchronous replication significantly increases the variability of write performance. If followers become inaccessible from the leader, the leader effectively becomes read-only.
|
||||
|
||||
To enable a simple synchronous replication test, add the following lines to the ``parameters`` section of your YAML configuration files:
|
||||
|
||||
.. code:: YAML
|
||||
|
||||
synchronous_commit: "on"
|
||||
synchronous_standby_names: "*"
|
||||
|
||||
When using PostgreSQL synchronous replication, use at least three Postgres data nodes to ensure write availability if one host fails.
|
||||
|
||||
Using PostgreSQL synchronous replication does not guarantee zero lost transactions under all circumstances. When the primary and the secondary that is currently acting as a synchronous replica fail simultaneously a third node that might not contain all transactions will be promoted.
|
||||
|
||||
.. _synchronous_mode:
|
||||
|
||||
Synchronous mode
|
||||
----------------
|
||||
|
||||
For use cases where losing committed transactions is not permissible you can turn on Patroni's ``synchronous_mode``. When ``synchronous_mode`` is turned on Patroni will not promote a standby unless it is certain that the standby contains all transactions that may have returned a successful commit status to client [2]_. This means that the system may be unavailable for writes even though some servers are available. System administrators can still use manual failover commands to promote a standby even if it results in transaction loss.
|
||||
|
||||
Turning on ``synchronous_mode`` does not guarantee multi node durability of commits under all circumstances. When no suitable standby is available, primary server will still accept writes, but does not guarantee their replication. When the primary fails in this mode no standby will be promoted. When the host that used to be the primary comes back it will get promoted automatically, unless system administrator performed a manual failover. This behavior makes synchronous mode usable with 2 node clusters.
|
||||
|
||||
When ``synchronous_mode`` is on and a standby crashes, commits will block until next iteration of Patroni runs and switches the primary to standalone mode (worst case delay for writes ``ttl`` seconds, average case ``loop_wait``/2 seconds). Manually shutting down or restarting a standby will not cause a commit service interruption. Standby will signal the primary to release itself from synchronous standby duties before PostgreSQL shutdown is initiated.
|
||||
|
||||
When it is absolutely necessary to guarantee that each write is stored durably
|
||||
on at least two nodes, enable ``synchronous_mode_strict`` in addition to the
|
||||
``synchronous_mode``. This parameter prevents Patroni from switching off the
|
||||
synchronous replication on the primary when no synchronous standby candidates
|
||||
are available. As a downside, the primary is not be available for writes
|
||||
(unless the Postgres transaction explicitly turns of ``synchronous_mode``),
|
||||
blocking all client write requests until at least one synchronous replica comes
|
||||
up.
|
||||
|
||||
You can ensure that a standby never becomes the synchronous standby by setting ``nosync`` tag to true. This is recommended to set for standbys that are behind slow network connections and would cause performance degradation when becoming a synchronous standby.
|
||||
|
||||
Synchronous mode can be switched on and off via Patroni REST interface. See :ref:`dynamic configuration <dynamic_configuration>` for instructions.
|
||||
|
||||
Note: Because of the way synchronous replication is implemented in PostgreSQL it is still possible to lose transactions even when using ``synchronous_mode_strict``. If the PostgreSQL backend is cancelled while waiting to acknowledge replication (as a result of packet cancellation due to client timeout or backend failure) transaction changes become visible for other backends. Such changes are not yet replicated and may be lost in case of standby promotion.
|
||||
|
||||
Synchronous Replication Factor
|
||||
------------------------------
|
||||
The parameter ``synchronous_node_count`` is used by Patroni to manage number of synchronous standby databases. It is set to 1 by default. It has no effect when ``synchronous_mode`` is set to off. When enabled, Patroni manages precise number of synchronous standby databases based on parameter ``synchronous_node_count`` and adjusts the state in DCS & synchronous_standby_names as members join and leave.
|
||||
|
||||
Synchronous mode implementation
|
||||
-------------------------------
|
||||
|
||||
When in synchronous mode Patroni maintains synchronization state in the DCS, containing the latest primary and current synchronous standby databases. This state is updated with strict ordering constraints to ensure the following invariants:
|
||||
|
||||
- A node must be marked as the latest leader whenever it can accept write transactions. Patroni crashing or PostgreSQL not shutting down can cause violations of this invariant.
|
||||
|
||||
- A node must be set as the synchronous standby in PostgreSQL as long as it is published as the synchronous standby.
|
||||
|
||||
- A node that is not the leader or current synchronous standby is not allowed to promote itself automatically.
|
||||
|
||||
Patroni will only assign one or more synchronous standby nodes based on ``synchronous_node_count`` parameter to ``synchronous_standby_names``.
|
||||
|
||||
On each HA loop iteration Patroni re-evaluates synchronous standby nodes choice. If the current list of synchronous standby nodes are connected and has not requested its synchronous status to be removed it remains picked. Otherwise the cluster member available for sync that is furthest ahead in replication is picked.
|
||||
|
||||
|
||||
.. [1] The data is still there, but recovering it requires a manual recovery effort by data recovery specialists. When Patroni is allowed to rewind with ``use_pg_rewind`` the forked timeline will be automatically erased to rejoin the failed primary with the cluster.
|
||||
|
||||
.. [2] Clients can change the behavior per transaction using PostgreSQL's ``synchronous_commit`` setting. Transactions with ``synchronous_commit`` values of ``off`` and ``local`` may be lost on fail over, but will not be blocked by replication delays.
|
||||
@@ -0,0 +1,394 @@
|
||||
.. _rest_api:
|
||||
|
||||
Patroni REST API
|
||||
================
|
||||
|
||||
Patroni has a rich REST API, which is used by Patroni itself during the leader race, by the ``patronictl`` tool in order to perform failovers/switchovers/reinitialize/restarts/reloads, by HAProxy or any other kind of load balancer to perform HTTP health checks, and of course could also be used for monitoring. Below you will find the list of Patroni REST API endpoints.
|
||||
|
||||
Health check endpoints
|
||||
----------------------
|
||||
For all health check ``GET`` requests Patroni returns a JSON document with the status of the node, along with the HTTP status code. If you don't want or don't need the JSON document, you might consider using the ``OPTIONS`` method instead of ``GET``.
|
||||
|
||||
- The following requests to Patroni REST API will return HTTP status code **200** only when the Patroni node is running as the primary with leader lock:
|
||||
|
||||
- ``GET /``
|
||||
- ``GET /master``
|
||||
- ``GET /primary``
|
||||
- ``GET /read-write``
|
||||
|
||||
- ``GET /standby-leader``: returns HTTP status code **200** only when the Patroni node is running as the leader in a :ref:`standby cluster <standby_cluster>`.
|
||||
|
||||
- ``GET /leader``: returns HTTP status code **200** when the Patroni node has the leader lock. The major difference from the two previous endpoints is that it doesn't take into account whether PostgreSQL is running as the ``primary`` or the ``standby_leader``.
|
||||
|
||||
- ``GET /replica``: replica health check endpoint. It returns HTTP status code **200** only when the Patroni node is in the state ``running``, the role is ``replica`` and ``noloadbalance`` tag is not set.
|
||||
|
||||
- ``GET /replica?lag=<max-lag>``: replica check endpoint. In addition to checks from ``replica``, it also checks replication latency and returns status code **200** only when it is below specified value. The key cluster.last_leader_operation from DCS is used for Leader wal position and compute latency on replica for performance reasons. max-lag can be specified in bytes (integer) or in human readable values, for e.g. 16kB, 64MB, 1GB.
|
||||
|
||||
- ``GET /replica?lag=1048576``
|
||||
- ``GET /replica?lag=1024kB``
|
||||
- ``GET /replica?lag=10MB``
|
||||
- ``GET /replica?lag=1GB``
|
||||
|
||||
- ``GET /replica?tag_key1=value1&tag_key2=value2``: replica check endpoint. In addition, It will also check for user defined tags ``key1`` and ``key2`` and their respective values in the **tags** section of the yaml configuration management. If the tag isn't defined for an instance, or if the value in the yaml configuration doesn't match the querying value, it will return HTTP Status Code 503.
|
||||
|
||||
In the following requests, since we are checking for the leader or standby-leader status, Patroni doesn't apply any of the user defined tags and they will be ignored.
|
||||
- ``GET /?tag_key1=value1&tag_key2=value2``
|
||||
- ``GET /master?tag_key1=value1&tag_key2=value2``
|
||||
- ``GET /leader?tag_key1=value1&tag_key2=value2``
|
||||
- ``GET /primary?tag_key1=value1&tag_key2=value2``
|
||||
- ``GET /read-write?tag_key1=value1&tag_key2=value2``
|
||||
- ``GET /standby_leader?tag_key1=value1&tag_key2=value2``
|
||||
- ``GET /standby-leader?tag_key1=value1&tag_key2=value2``
|
||||
|
||||
- ``GET /read-only``: like the above endpoint, but also includes the primary.
|
||||
|
||||
- ``GET /synchronous`` or ``GET /sync``: returns HTTP status code **200** only when the Patroni node is running as a synchronous standby.
|
||||
|
||||
- ``GET /read-only-sync``: like the above endpoint, but also includes the primary.
|
||||
|
||||
- ``GET /asynchronous`` or ``GET /async``: returns HTTP status code **200** only when the Patroni node is running as an asynchronous standby.
|
||||
|
||||
|
||||
- ``GET /asynchronous?lag=<max-lag>`` or ``GET /async?lag=<max-lag>``: asynchronous standby check endpoint. In addition to checks from ``asynchronous`` or ``async``, it also checks replication latency and returns status code **200** only when it is below specified value. The key cluster.last_leader_operation from DCS is used for Leader wal position and compute latency on replica for performance reasons. max-lag can be specified in bytes (integer) or in human readable values, for e.g. 16kB, 64MB, 1GB.
|
||||
|
||||
- ``GET /async?lag=1048576``
|
||||
- ``GET /async?lag=1024kB``
|
||||
- ``GET /async?lag=10MB``
|
||||
- ``GET /async?lag=1GB``
|
||||
|
||||
- ``GET /health``: returns HTTP status code **200** only when PostgreSQL is up and running.
|
||||
|
||||
- ``GET /liveness``: always returns HTTP status code **200** what only indicates that Patroni is running. Could be used for ``livenessProbe``.
|
||||
|
||||
- ``GET /readiness``: returns HTTP status code **200** when the Patroni node is running as the leader or when PostgreSQL is up and running. The endpoint could be used for ``readinessProbe`` when it is not possible to use Kubernetes endpoints for leader elections (OpenShift).
|
||||
|
||||
Both, ``readiness`` and ``liveness`` endpoints are very light-weight and not executing any SQL. Probes should be configured in such a way that they start failing about time when the leader key is expiring. With the default value of ``ttl``, which is ``30s`` example probes would look like:
|
||||
|
||||
.. code-block:: yaml
|
||||
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
scheme: HTTP
|
||||
path: /readiness
|
||||
port: 8008
|
||||
initialDelaySeconds: 3
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
successThreshold: 1
|
||||
failureThreshold: 3
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
scheme: HTTP
|
||||
path: /liveness
|
||||
port: 8008
|
||||
initialDelaySeconds: 3
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
successThreshold: 1
|
||||
failureThreshold: 3
|
||||
|
||||
|
||||
Monitoring endpoint
|
||||
-------------------
|
||||
|
||||
The ``GET /patroni`` is used by Patroni during the leader race. It also could be used by your monitoring system. The JSON document produced by this endpoint has the same structure as the JSON produced by the health check endpoints.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/patroni | jq .
|
||||
{
|
||||
"state": "running",
|
||||
"postmaster_start_time": "2019-09-24 09:22:32.555 CEST",
|
||||
"role": "master",
|
||||
"server_version": 110005,
|
||||
"cluster_unlocked": false,
|
||||
"xlog": {
|
||||
"location": 25624640
|
||||
},
|
||||
"timeline": 3,
|
||||
"database_system_identifier": "6739877027151648096",
|
||||
"patroni": {
|
||||
"version": "1.6.0",
|
||||
"scope": "batman"
|
||||
}
|
||||
}
|
||||
|
||||
Cluster status endpoints
|
||||
------------------------
|
||||
|
||||
- The ``GET /cluster`` endpoint generates a JSON document describing the current cluster topology and state:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/cluster | jq .
|
||||
{
|
||||
"members": [
|
||||
{
|
||||
"name": "postgresql0",
|
||||
"host": "127.0.0.1",
|
||||
"port": 5432,
|
||||
"role": "leader",
|
||||
"state": "running",
|
||||
"api_url": "http://127.0.0.1:8008/patroni",
|
||||
"timeline": 5,
|
||||
"tags": {
|
||||
"clonefrom": true
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "postgresql1",
|
||||
"host": "127.0.0.1",
|
||||
"port": 5433,
|
||||
"role": "replica",
|
||||
"state": "running",
|
||||
"api_url": "http://127.0.0.1:8009/patroni",
|
||||
"timeline": 5,
|
||||
"tags": {
|
||||
"clonefrom": true
|
||||
},
|
||||
"lag": 0
|
||||
}
|
||||
],
|
||||
"scheduled_switchover": {
|
||||
"at": "2019-09-24T10:36:00+02:00",
|
||||
"from": "postgresql0"
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
- The ``GET /history`` endpoint provides a view on the history of cluster switchovers/failovers. The format is very similar to the content of history files in the ``pg_wal`` directory. The only difference is the timestamp field showing when the new timeline was created.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/history | jq .
|
||||
[
|
||||
[
|
||||
1,
|
||||
25623960,
|
||||
"no recovery target specified",
|
||||
"2019-09-23T16:57:57+02:00"
|
||||
],
|
||||
[
|
||||
2,
|
||||
25624344,
|
||||
"no recovery target specified",
|
||||
"2019-09-24T09:22:33+02:00"
|
||||
],
|
||||
[
|
||||
3,
|
||||
25624752,
|
||||
"no recovery target specified",
|
||||
"2019-09-24T09:26:15+02:00"
|
||||
],
|
||||
[
|
||||
4,
|
||||
50331856,
|
||||
"no recovery target specified",
|
||||
"2019-09-24T09:35:52+02:00"
|
||||
]
|
||||
]
|
||||
|
||||
|
||||
Config endpoint
|
||||
---------------
|
||||
|
||||
``GET /config``: Get the current version of the dynamic configuration:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s localhost:8008/config | jq .
|
||||
{
|
||||
"ttl": 30,
|
||||
"loop_wait": 10,
|
||||
"retry_timeout": 10,
|
||||
"maximum_lag_on_failover": 1048576,
|
||||
"postgresql": {
|
||||
"use_slots": true,
|
||||
"use_pg_rewind": true,
|
||||
"parameters": {
|
||||
"hot_standby": "on",
|
||||
"wal_log_hints": "on",
|
||||
"wal_level": "hot_standby",
|
||||
"max_wal_senders": 5,
|
||||
"max_replication_slots": 5,
|
||||
"max_connections": "100"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
``PATCH /config``: Change the existing configuration.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s -XPATCH -d \
|
||||
'{"loop_wait":5,"ttl":20,"postgresql":{"parameters":{"max_connections":"101"}}}' \
|
||||
http://localhost:8008/config | jq .
|
||||
{
|
||||
"ttl": 20,
|
||||
"loop_wait": 5,
|
||||
"maximum_lag_on_failover": 1048576,
|
||||
"retry_timeout": 10,
|
||||
"postgresql": {
|
||||
"use_slots": true,
|
||||
"use_pg_rewind": true,
|
||||
"parameters": {
|
||||
"hot_standby": "on",
|
||||
"wal_log_hints": "on",
|
||||
"wal_level": "hot_standby",
|
||||
"max_wal_senders": 5,
|
||||
"max_replication_slots": 5,
|
||||
"max_connections": "101"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
The above REST API call patches the existing configuration and returns the new configuration.
|
||||
|
||||
Let's check that the node processed this configuration. First of all it should start printing log lines every 5 seconds (loop_wait=5). The change of "max_connections" requires a restart, so the "pending_restart" flag should be exposed:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/patroni | jq .
|
||||
{
|
||||
"pending_restart": true,
|
||||
"database_system_identifier": "6287881213849985952",
|
||||
"postmaster_start_time": "2016-06-13 13:13:05.211 CEST",
|
||||
"xlog": {
|
||||
"location": 2197818976
|
||||
},
|
||||
"patroni": {
|
||||
"scope": "batman",
|
||||
"version": "1.0"
|
||||
},
|
||||
"state": "running",
|
||||
"role": "master",
|
||||
"server_version": 90503
|
||||
}
|
||||
|
||||
Removing parameters:
|
||||
|
||||
If you want to remove (reset) some setting just patch it with ``null``:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s -XPATCH -d \
|
||||
'{"postgresql":{"parameters":{"max_connections":null}}}' \
|
||||
http://localhost:8008/config | jq .
|
||||
{
|
||||
"ttl": 20,
|
||||
"loop_wait": 5,
|
||||
"retry_timeout": 10,
|
||||
"maximum_lag_on_failover": 1048576,
|
||||
"postgresql": {
|
||||
"use_slots": true,
|
||||
"use_pg_rewind": true,
|
||||
"parameters": {
|
||||
"hot_standby": "on",
|
||||
"unix_socket_directories": ".",
|
||||
"wal_level": "hot_standby",
|
||||
"wal_log_hints": "on",
|
||||
"max_wal_senders": 5,
|
||||
"max_replication_slots": 5
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
The above call removes ``postgresql.parameters.max_connections`` from the dynamic configuration.
|
||||
|
||||
``PUT /config``: It's also possible to perform the full rewrite of an existing dynamic configuration unconditionally:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s -XPUT -d \
|
||||
'{"maximum_lag_on_failover":1048576,"retry_timeout":10,"postgresql":{"use_slots":true,"use_pg_rewind":true,"parameters":{"hot_standby":"on","wal_log_hints":"on","wal_level":"hot_standby","unix_socket_directories":".","max_wal_senders":5}},"loop_wait":3,"ttl":20}' \
|
||||
http://localhost:8008/config | jq .
|
||||
{
|
||||
"ttl": 20,
|
||||
"maximum_lag_on_failover": 1048576,
|
||||
"retry_timeout": 10,
|
||||
"postgresql": {
|
||||
"use_slots": true,
|
||||
"parameters": {
|
||||
"hot_standby": "on",
|
||||
"unix_socket_directories": ".",
|
||||
"wal_level": "hot_standby",
|
||||
"wal_log_hints": "on",
|
||||
"max_wal_senders": 5
|
||||
},
|
||||
"use_pg_rewind": true
|
||||
},
|
||||
"loop_wait": 3
|
||||
}
|
||||
|
||||
|
||||
Switchover and failover endpoints
|
||||
---------------------------------
|
||||
|
||||
``POST /switchover`` or ``POST /failover``. These endpoints are very similar to each other. There are a couple of minor differences though:
|
||||
|
||||
1. The failover endpoint allows to perform a manual failover when there are no healthy nodes, but at the same time it will not allow you to schedule a switchover.
|
||||
|
||||
2. The switchover endpoint is the opposite. It works only when the cluster is healthy (there is a leader) and allows to schedule a switchover at a given time.
|
||||
|
||||
|
||||
In the JSON body of the ``POST`` request you must specify at least the ``leader`` or ``candidate`` fields and optionally the ``scheduled_at`` field if you want to schedule a switchover at a specific time.
|
||||
|
||||
|
||||
Example: perform a failover to the specific node:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8009/failover -XPOST -d '{"candidate":"postgresql1"}'
|
||||
Successfully failed over to "postgresql1"
|
||||
|
||||
|
||||
Example: schedule a switchover from the leader to any other healthy replica in the cluster at a specific time:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
$ curl -s http://localhost:8008/switchover -XPOST -d \
|
||||
'{"leader":"postgresql0","scheduled_at":"2019-09-24T12:00+00"}'
|
||||
Switchover scheduled
|
||||
|
||||
|
||||
Depending on the situation the request might finish with a different HTTP status code and body. The status code **200** is returned when the switchover or failover successfully completed. If the switchover was successfully scheduled, Patroni will return HTTP status code **202**. In case something went wrong, the error status code (one of **400**, **412** or **503**) will be returned with some details in the response body. For more information please check the source code of ``patroni/api.py:do_POST_failover()`` method.
|
||||
|
||||
- ``DELETE /switchover``: delete the scheduled switchover
|
||||
|
||||
The ``POST /switchover`` and ``POST failover`` endpoints are used by ``patronictl switchover`` and ``patronictl failover``, respectively.
|
||||
The ``DELETE /switchover`` is used by ``patronictl flush <cluster-name> switchover``.
|
||||
|
||||
|
||||
Restart endpoint
|
||||
----------------
|
||||
|
||||
- ``POST /restart``: You can restart Postgres on the specific node by performing the ``POST /restart`` call. In the JSON body of ``POST`` request it is possible to optionally specify some restart conditions:
|
||||
|
||||
- **restart_pending**: boolean, if set to ``true`` Patroni will restart PostgreSQL only when restart is pending in order to apply some changes in the PostgreSQL config.
|
||||
- **role**: perform restart only if the current role of the node matches with the role from the POST request.
|
||||
- **postgres_version**: perform restart only if the current version of postgres is smaller than specified in the POST request.
|
||||
- **timeout**: how long we should wait before PostgreSQL starts accepting connections. Overrides ``master_start_timeout``.
|
||||
- **schedule**: timestamp with time zone, schedule the restart somewhere in the future.
|
||||
|
||||
- ``DELETE /restart``: delete the scheduled restart
|
||||
|
||||
``POST /restart`` and ``DELETE /restart`` endpoints are used by ``patronictl restart`` and ``patronictl flush <cluster-name> restart`` respectively.
|
||||
|
||||
|
||||
Reload endpoint
|
||||
---------------
|
||||
|
||||
The ``POST /reload`` call will order Patroni to re-read and apply the configuration file. This is the equivalent of sending the ``SIGHUP`` signal to the Patroni process. In case you changed some of the Postgres parameters which require a restart (like **shared_buffers**), you still have to explicitly do the restart of Postgres by either calling the ``POST /restart`` endpoint or with the help of ``patronictl restart``.
|
||||
|
||||
The reload endpoint is used by ``patronictl reload``.
|
||||
|
||||
|
||||
Reinitialize endpoint
|
||||
---------------------
|
||||
|
||||
``POST /reinitialize``: reinitialize the PostgreSQL data directory on the specified node. It is allowed to be executed only on replicas. Once called, it will remove the data directory and start ``pg_basebackup`` or some alternative :ref:`replica creation method <custom_replica_creation>`.
|
||||
|
||||
The call might fail if Patroni is in a loop trying to recover (restart) a failed Postgres. In order to overcome this problem one can specify ``{"force":true}`` in the request body.
|
||||
|
||||
The reinitialize endpoint is used by ``patronictl reinit``.
|
||||
@@ -0,0 +1,37 @@
|
||||
.. _security:
|
||||
|
||||
=======================
|
||||
Security Considerations
|
||||
=======================
|
||||
|
||||
A Patroni cluster has two interfaces to be protected from unauthorized access: the distributed configuration storage (DCS) and the Patroni REST API.
|
||||
|
||||
Protecting DCS
|
||||
==============
|
||||
|
||||
Patroni and patronictl both store and retrieve data to/from the DCS.
|
||||
|
||||
Despite DCS doesn't contain any sensitive information, it allows changing some of Patroni/Postgres configuration. Therefore the very first thing that should be protected is DCS itself.
|
||||
|
||||
The details of protection depend on the type of DCS used. The authentication and encryption parameters (tokens/basic-auth/client certificates) for the supported types of DCS are covered in :ref:`SETTINGS <bootstrap_settings>`
|
||||
|
||||
The general recommendation is to enable TLS for all DCS communication.
|
||||
|
||||
Protecting the REST API
|
||||
=======================
|
||||
|
||||
Protecting the REST API is a more complicated task.
|
||||
|
||||
The Patroni REST API is used by Patroni itself during the leader race, by the ``patronictl`` tool in order to perform failovers/switchovers/reinitialize/restarts/reloads, by HAProxy or any other kind of load balancer to perform HTTP health checks, and of course could also be used for monitoring.
|
||||
|
||||
From the point of view of security, REST API contains safe (``GET`` requests, only retrieve information) and unsafe (``PUT``, ``POST``, ``PATCH`` and ``DELETE`` requests, change the state of nodes) endpoints.
|
||||
|
||||
The unsafe endpoints can be protected with HTTP basic-auth by setting the ``restapi.authentication.username`` and ``restapi.authentication.password`` parameters. There is no way to protect the safe endpoints without enabling TLS.
|
||||
|
||||
When TLS for the REST API is enabled and a PKI is established, mutual authentication of the API server and API client is possible for all endpoints.
|
||||
|
||||
The ``restapi`` section parameters enable TLS client authentication to the server. Depending on the value of the ``verify_client`` parameter, the API server requires a successful client certificate verification for both safe and unsafe API calls (``verify_client: required``), or only for unsafe API calls (``verify_client: optional``), or for no API calls (``verify_client: none``).
|
||||
|
||||
The ``ctl`` section parameters enable TLS server authentication to the client (the ``patronictl`` tool which uses the same config as patroni). Set ``insecure: true`` to disable the server certificate verification by the client. See :ref:`SETTINGS <patronictl_settings>` for a detailed description of the TLS client parameters.
|
||||
|
||||
Protecting the PostgreSQL database proper from unauthorized access is beyond the scope of this document and is covered in https://www.postgresql.org/docs/current/client-authentication.html
|
||||
@@ -0,0 +1,39 @@
|
||||
.. _watchdog:
|
||||
|
||||
Watchdog support
|
||||
================
|
||||
|
||||
Having multiple PostgreSQL servers running as master can result in transactions lost due to diverging timelines. This situation is also called a split-brain problem. To avoid split-brain Patroni needs to ensure PostgreSQL will not accept any transaction commits after leader key expires in the DCS. Under normal circumstances Patroni will try to achieve this by stopping PostgreSQL when leader lock update fails for any reason. However, this may fail to happen due to various reasons:
|
||||
|
||||
- Patroni has crashed due to a bug, out-of-memory condition or by being accidentally killed by a system administrator.
|
||||
|
||||
- Shutting down PostgreSQL is too slow.
|
||||
|
||||
- Patroni does not get to run due to high load on the system, the VM being paused by the hypervisor, or other infrastructure issues.
|
||||
|
||||
To guarantee correct behavior under these conditions Patroni supports watchdog devices. Watchdog devices are software or hardware mechanisms that will reset the whole system when they do not get a keepalive heartbeat within a specified timeframe. This adds an additional layer of fail safe in case usual Patroni split-brain protection mechanisms fail.
|
||||
|
||||
Patroni will try to activate the watchdog before promoting PostgreSQL to master. If watchdog activation fails and watchdog mode is ``required`` then the node will refuse to become master. When deciding to participate in leader election Patroni will also check that watchdog configuration will allow it to become leader at all. After demoting PostgreSQL (for example due to a manual failover) Patroni will disable the watchdog again. Watchdog will also be disabled while Patroni is in paused state.
|
||||
|
||||
By default Patroni will set up the watchdog to expire 5 seconds before TTL expires. With the default setup of ``loop_wait=10`` and ``ttl=30`` this gives HA loop at least 15 seconds (``ttl`` - ``safety_margin`` - ``loop_wait``) to complete before the system gets forcefully reset. By default accessing DCS is configured to time out after 10 seconds. This means that when DCS is unavailable, for example due to network issues, Patroni and PostgreSQL will have at least 5 seconds (``ttl`` - ``safety_margin`` - ``loop_wait`` - ``retry_timeout``) to come to a state where all client connections are terminated.
|
||||
|
||||
Safety margin is the amount of time that Patroni reserves for time between leader key update and watchdog keepalive. Patroni will try to send a keepalive immediately after confirmation of leader key update. If Patroni process is suspended for extended amount of time at exactly the right moment the keepalive may be delayed for more than the safety margin without triggering the watchdog. This results in a window of time where watchdog will not trigger before leader key expiration, invalidating the guarantee. To be absolutely sure that watchdog will trigger under all circumstances set up the watchdog to expire after half of TTL by setting ``safety_margin`` to -1 to set watchdog timeout to ``ttl // 2``. If you need this guarantee you probably should increase ``ttl`` and/or reduce ``loop_wait`` and ``retry_timeout``.
|
||||
|
||||
Currently watchdogs are only supported using Linux watchdog device interface.
|
||||
|
||||
Setting up software watchdog on Linux
|
||||
-------------------------------------
|
||||
|
||||
Default Patroni configuration will try to use ``/dev/watchdog`` on Linux if it is accessible to Patroni. For most use cases using software watchdog built into the Linux kernel is secure enough.
|
||||
|
||||
To enable software watchdog issue the following commands as root before starting Patroni:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
modprobe softdog
|
||||
# Replace postgres with the user you will be running patroni under
|
||||
chown postgres /dev/watchdog
|
||||
|
||||
For testing it may be helpful to disable rebooting by adding ``soft_noboot=1`` to the modprobe command line. In this case the watchdog will just log a line in kernel ring buffer, visible via `dmesg`.
|
||||
|
||||
Patroni will log information about the watchdog when it is successfully enabled.
|
||||
+3
-3
@@ -1,11 +1,11 @@
|
||||
### confd
|
||||
|
||||
`confd` directory contains haproxy template files for the [confd](https://github.com/kelseyhightower/confd) -- lightweight configuration management tool
|
||||
`confd` directory contains haproxy and pgbouncer template files for the [confd](https://github.com/kelseyhightower/confd) -- lightweight configuration management tool
|
||||
You need to copy content of `confd` directory into /etcd/confd and run confd service:
|
||||
```bash
|
||||
$ confd -prefix=/service/$PATRONI_SCOPE -backend etcd -node $PATRONI_ETCD_HOST -interval=10
|
||||
$ confd -prefix=/service/$PATRONI_SCOPE -backend etcd -node $PATRONI_ETCD_URL -interval=10
|
||||
```
|
||||
It will periodically update haproxy.cfg with the actual list of Patroni nodes from `etcd` and "reload" haproxy when it is necessary.
|
||||
It will periodically update haproxy.cfg and pgbouncer.ini with the actual list of Patroni nodes from `etcd` and "reload" haproxy and pgbouncer.ini when it is necessary.
|
||||
|
||||
|
||||
### startup-scripts
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
[template]
|
||||
prefix = "/service/batman"
|
||||
owner = "postgres"
|
||||
mode = "0644"
|
||||
src = "pgbouncer.tmpl"
|
||||
dest = "/etc/pgbouncer/pgbouncer.ini"
|
||||
|
||||
reload_cmd = "systemctl reload pgbouncer"
|
||||
|
||||
keys = [
|
||||
"/members/","/leader"
|
||||
]
|
||||
@@ -10,19 +10,23 @@ defaults
|
||||
timeout server 30m
|
||||
timeout check 5s
|
||||
|
||||
frontend master_postgresql
|
||||
listen stats
|
||||
mode http
|
||||
bind *:7000
|
||||
stats enable
|
||||
stats uri /
|
||||
|
||||
listen master
|
||||
bind *:5000
|
||||
default_backend backend_master
|
||||
|
||||
frontend replicas_postgresql
|
||||
bind *:5001
|
||||
default_backend backend_replicas
|
||||
|
||||
backend backend_master
|
||||
option httpchk OPTIONS /master
|
||||
http-check expect status 200
|
||||
default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions
|
||||
{{range gets "/members/*"}} server {{base .Key}} {{$data := json .Value}}{{base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}} maxconn 100 check port {{index (split (index (split $data.api_url "/") 2) ":") 1}}
|
||||
{{end}}
|
||||
backend backend_replicas
|
||||
listen replicas
|
||||
bind *:5001
|
||||
option httpchk OPTIONS /replica
|
||||
http-check expect status 200
|
||||
default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions
|
||||
{{range gets "/members/*"}} server {{base .Key}} {{$data := json .Value}}{{base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}} maxconn 100 check port {{index (split (index (split $data.api_url "/") 2) ":") 1}}
|
||||
{{end}}
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
[databases]
|
||||
{{with get "/leader"}}{{$leader := .Value}}{{$leadkey := printf "/members/%s" $leader}}{{with get $leadkey}}{{$data := json .Value}}{{$hostport := base (replace (index (split $data.conn_url "/") 2) "@" "/" -1)}}{{ $host := base (index (split $hostport ":") 0)}}{{ $port := base (index (split $hostport ":") 1)}}* = host={{ $host }} port={{ $port }} pool_size=10{{end}}{{end}}
|
||||
|
||||
[pgbouncer]
|
||||
logfile = /var/log/postgresql/pgbouncer.log
|
||||
pidfile = /var/run/postgresql/pgbouncer.pid
|
||||
listen_addr = *
|
||||
listen_port = 6432
|
||||
unix_socket_dir = /var/run/postgresql
|
||||
auth_type = trust
|
||||
auth_file = /etc/pgbouncer/userlist.txt
|
||||
auth_hba_file = /etc/pgbouncer/pg_hba.txt
|
||||
admin_users = pgbouncer
|
||||
stats_users = pgbouncer
|
||||
pool_mode = session
|
||||
max_client_conn = 100
|
||||
default_pool_size = 20
|
||||
@@ -11,3 +11,12 @@ Upstart job for Ubuntu 12.04 or 14.04. Requires Upstart > 1.4. Intended for sys
|
||||
|
||||
### patroni.service
|
||||
Systemd service file, to be copied to /etc/systemd/system/patroni.service, tested on Centos 7.1 with Patroni installed from pip.
|
||||
|
||||
### patroni
|
||||
Init.d service file for Debian-like distributions. Copy it to /etc/init.d/, make executable:
|
||||
```chmod 755 /etc/init.d/patroni``` and run with ```service patroni start```, or make it starting on boot with ```update-rc.d patroni defaults```. Also you might edit some configuration variables in it:
|
||||
PATRONI for patroni.py location
|
||||
CONF for configuration file
|
||||
LOGFILE for log (script creates it if does not exist)
|
||||
|
||||
Note. If you have several versions of Postgres installed, please add to POSTGRES_VERSION the release number which you wish to run. Script uses this value to append PATH environment with correct path to Postgres bin.
|
||||
|
||||
@@ -0,0 +1,144 @@
|
||||
#!/bin/sh
|
||||
#
|
||||
### BEGIN INIT INFO
|
||||
# Provides: patroni
|
||||
# Required-Start: $remote_fs $syslog
|
||||
# Required-Stop: $remote_fs $syslog
|
||||
# Default-Start: 2 3 4 5
|
||||
# Default-Stop: 0 1 6
|
||||
# Short-Description: Patroni init script
|
||||
# Description: Runners to orchestrate a high-availability PostgreSQL
|
||||
### END INIT INFO
|
||||
|
||||
### BEGIN USER CONFIGURATION
|
||||
|
||||
CONF="/etc/patroni/postgres.yml"
|
||||
LOGFILE="/var/log/patroni.log"
|
||||
USER="postgres"
|
||||
GROUP="postgres"
|
||||
|
||||
NAME=patroni
|
||||
PATRONI="/opt/patroni/$NAME.py"
|
||||
PIDFILE="/var/run/$NAME.pid"
|
||||
|
||||
# Set this parameter, if you have several Postgres versions installed
|
||||
# POSTGRES_VERSION="9.4"
|
||||
POSTGRES_VERSION=""
|
||||
|
||||
### END USER CONFIGURATION
|
||||
|
||||
. /lib/lsb/init-functions
|
||||
|
||||
# Loading this library for get_versions() function
|
||||
if test ! -e /usr/share/postgresql-common/init.d-functions; then
|
||||
log_failure_msg "Probably postgresql-common does not installed."
|
||||
exit 1
|
||||
else
|
||||
. /usr/share/postgresql-common/init.d-functions
|
||||
fi
|
||||
|
||||
# Is there Patroni executable?
|
||||
if test ! -e $PATRONI; then
|
||||
log_failure_msg "Patroni executable $PATRONI does not exist."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Is there Patroni configuration file?
|
||||
if test ! -e $CONF; then
|
||||
log_failure_msg "Patroni configuration file $CONF does not exist."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Create logfile if doesn't exist
|
||||
if test ! -e $LOGFILE; then
|
||||
log_action_msg "Creating logfile for Patroni..."
|
||||
touch $LOGFILE
|
||||
chown $USER:$GROUP $LOGFILE
|
||||
fi
|
||||
|
||||
prepare_pgpath() {
|
||||
if [ "$POSTGRES_VERSION" != "" ]; then
|
||||
if [ -x /usr/lib/postgresql/$POSTGRES_VERSION/bin/pg_ctl ]; then
|
||||
PGPATH="/usr/lib/postgresql/$POSTGRES_VERSION/bin"
|
||||
else
|
||||
log_failure_msg "Postgres version incorrect, check POSTGRES_VERSION variable."
|
||||
exit 0
|
||||
fi
|
||||
else
|
||||
get_versions
|
||||
if echo $versions | grep -q -e "\s"; then
|
||||
log_warning_msg "You have several Postgres versions installed. Please, use POSTGRES_VERSION to define correct environment."
|
||||
else
|
||||
versions=`echo $versions | sed -e 's/^[ \t]*//'`
|
||||
PGPATH="/usr/lib/postgresql/$versions/bin"
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
get_pid() {
|
||||
if test -e $PIDFILE; then
|
||||
PID=`cat $PIDFILE`
|
||||
CHILDPID=`ps --ppid $PID -o %p --no-headers`
|
||||
else
|
||||
log_failure_msg "Could not find PID file. Patroni probably down."
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
|
||||
case "$1" in
|
||||
start)
|
||||
prepare_pgpath
|
||||
PGPATH=$PATH:$PGPATH
|
||||
log_success_msg "Starting Patroni\n"
|
||||
exec start-stop-daemon --start --quiet \
|
||||
--background \
|
||||
--pidfile $PIDFILE --make-pidfile \
|
||||
--chuid $USER:$GROUP \
|
||||
--chdir `eval echo ~$USER` \
|
||||
--exec $PATRONI \
|
||||
--startas /bin/sh -- \
|
||||
-c "/usr/bin/env PATH=$PGPATH /usr/bin/python $PATRONI $CONF >> $LOGFILE 2>&1"
|
||||
;;
|
||||
|
||||
stop)
|
||||
log_success_msg "Stopping Patroni"
|
||||
get_pid
|
||||
start-stop-daemon --stop --pid $CHILDPID
|
||||
start-stop-daemon --stop --pidfile $PIDFILE --remove-pidfile --quiet
|
||||
;;
|
||||
|
||||
reload)
|
||||
log_success_msg "Reloading Patroni configuration"
|
||||
get_pid
|
||||
kill -HUP $CHILDPID
|
||||
;;
|
||||
|
||||
status)
|
||||
get_pid
|
||||
if start-stop-daemon -T --pid $CHILDPID; then
|
||||
log_success_msg "Patroni is running\n"
|
||||
exit 0
|
||||
else
|
||||
log_warning_msg "Patroni in not running\n"
|
||||
fi
|
||||
;;
|
||||
|
||||
restart)
|
||||
$0 stop
|
||||
$0 start
|
||||
;;
|
||||
|
||||
*)
|
||||
echo "Usage: /etc/init.d/$NAME {start|stop|restart|reload|status}"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
||||
if [ $? -eq 0 ]; then
|
||||
echo .
|
||||
exit 0
|
||||
else
|
||||
echo " failed"
|
||||
exit 1
|
||||
fi
|
||||
@@ -11,17 +11,32 @@ Type=simple
|
||||
User=postgres
|
||||
Group=postgres
|
||||
|
||||
# Read in configuration file if it exists, otherwise proceed
|
||||
EnvironmentFile=-/etc/patroni_env.conf
|
||||
|
||||
# the default is the user's home directory, and if you want to change it, you must provide an absolute path.
|
||||
# WorkingDirectory=/home/sameuser
|
||||
|
||||
# Where to send early-startup messages from the server
|
||||
# This is normally controlled by the global default set by systemd
|
||||
# StandardOutput=syslog
|
||||
#StandardOutput=syslog
|
||||
|
||||
# Pre-commands to start watchdog device
|
||||
# Uncomment if watchdog is part of your patroni setup
|
||||
#ExecStartPre=-/usr/bin/sudo /sbin/modprobe softdog
|
||||
#ExecStartPre=-/usr/bin/sudo /bin/chown postgres /dev/watchdog
|
||||
|
||||
# Start the patroni process
|
||||
ExecStart=/bin/patroni /etc/patroni.yml
|
||||
|
||||
# Send HUP to reload from patroni.yml
|
||||
ExecReload=/bin/kill -s HUP $MAINPID
|
||||
|
||||
# only kill the patroni process, not it's children, so it will gracefully stop postgres
|
||||
KillMode=process
|
||||
|
||||
# Give a reasonable amount of time for the server to start up/shut down
|
||||
TimeoutSec=10
|
||||
TimeoutSec=30
|
||||
|
||||
# Do not restart the service if it crashes, we want to manually inspect database on failure
|
||||
Restart=no
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
#!/usr/bin/env python
|
||||
import os
|
||||
import argparse
|
||||
import shutil
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--dirname", required=True)
|
||||
parser.add_argument("--pathname", required=True)
|
||||
parser.add_argument("--filename", required=True)
|
||||
parser.add_argument("--mode", required=True, choices=("archive", "restore"))
|
||||
args, _ = parser.parse_known_args()
|
||||
|
||||
full_filename = os.path.join(args.dirname, args.filename)
|
||||
if args.mode == "archive":
|
||||
if not os.path.isdir(args.dirname):
|
||||
os.makedirs(args.dirname)
|
||||
if not os.path.exists(full_filename):
|
||||
shutil.copy(args.pathname, full_filename)
|
||||
else:
|
||||
shutil.copy(full_filename, args.pathname)
|
||||
Executable
+14
@@ -0,0 +1,14 @@
|
||||
#!/usr/bin/env python
|
||||
import argparse
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--datadir", required=True)
|
||||
parser.add_argument("--dbname", required=True)
|
||||
parser.add_argument("--walmethod", required=True, choices=("fetch", "stream", "none"))
|
||||
args, _ = parser.parse_known_args()
|
||||
|
||||
walmethod = ["-X", args.walmethod] if args.walmethod != "none" else []
|
||||
sys.exit(subprocess.call(["pg_basebackup", "-D", args.datadir, "-c", "fast", "-d", args.dbname] + walmethod))
|
||||
Executable
+11
@@ -0,0 +1,11 @@
|
||||
#!/usr/bin/env python
|
||||
import argparse
|
||||
import shutil
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--datadir", required=True)
|
||||
parser.add_argument("--sourcedir", required=True)
|
||||
args, _ = parser.parse_known_args()
|
||||
|
||||
shutil.copytree(args.sourcedir, args.datadir)
|
||||
@@ -3,15 +3,94 @@ Feature: basic replication
|
||||
|
||||
Scenario: check replication of a single table
|
||||
Given I start postgres0
|
||||
And postgres0 is a leader after 10 seconds
|
||||
And I start postgres1
|
||||
When I add the table foo to postgres0
|
||||
Then postgres0 is a leader after 10 seconds
|
||||
And there is a non empty initialize key in DCS after 15 seconds
|
||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 20, "loop_wait": 2, "synchronous_mode": true}
|
||||
Then I receive a response code 200
|
||||
When I start postgres1
|
||||
And I configure and start postgres2 with a tag replicatefrom postgres0
|
||||
And "sync" key in DCS has leader=postgres0 after 20 seconds
|
||||
And I add the table foo to postgres0
|
||||
Then table foo is present on postgres1 after 20 seconds
|
||||
Then table foo is present on postgres2 after 20 seconds
|
||||
|
||||
Scenario: check the basic failover
|
||||
When I kill postgres0
|
||||
Then postgres1 role is the primary after 32 seconds
|
||||
When I start postgres0
|
||||
Scenario: check restart of sync replica
|
||||
Given I shut down postgres2
|
||||
Then "sync" key in DCS has sync_standby=postgres1 after 5 seconds
|
||||
When I start postgres2
|
||||
And I shut down postgres1
|
||||
Then "sync" key in DCS has sync_standby=postgres2 after 10 seconds
|
||||
When I start postgres1
|
||||
And "members/postgres1" key in DCS has state=running after 10 seconds
|
||||
And I sleep for 2 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8010/sync
|
||||
Then I receive a response code 200
|
||||
When I issue a GET request to http://127.0.0.1:8009/async
|
||||
Then I receive a response code 200
|
||||
|
||||
Scenario: check stuck sync replica
|
||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"pause": true, "maximum_lag_on_syncnode": 15000000, "postgresql": {"parameters": {"synchronous_commit": "remote_apply"}}}
|
||||
Then I receive a response code 200
|
||||
And I create table on postgres0
|
||||
And table mytest is present on postgres1 after 2 seconds
|
||||
And table mytest is present on postgres2 after 2 seconds
|
||||
When I pause wal replay on postgres2
|
||||
And I load data on postgres0
|
||||
Then "sync" key in DCS has sync_standby=postgres1 after 15 seconds
|
||||
And I resume wal replay on postgres2
|
||||
And I sleep for 2 seconds
|
||||
And I issue a GET request to http://127.0.0.1:8009/sync
|
||||
Then I receive a response code 200
|
||||
When I issue a GET request to http://127.0.0.1:8010/async
|
||||
Then I receive a response code 200
|
||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"pause": null, "maximum_lag_on_syncnode": -1, "postgresql": {"parameters": {"synchronous_commit": "on"}}}
|
||||
Then I receive a response code 200
|
||||
And I drop table on postgres0
|
||||
|
||||
Scenario: check multi sync replication
|
||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"synchronous_node_count": 2}
|
||||
Then I receive a response code 200
|
||||
And I sleep for 10 seconds
|
||||
Then "sync" key in DCS has sync_standby=postgres1,postgres2 after 5 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8010/sync
|
||||
Then I receive a response code 200
|
||||
When I issue a GET request to http://127.0.0.1:8009/sync
|
||||
Then I receive a response code 200
|
||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"synchronous_node_count": 1}
|
||||
Then I receive a response code 200
|
||||
And I shut down postgres1
|
||||
And I sleep for 10 seconds
|
||||
Then "sync" key in DCS has sync_standby=postgres2 after 10 seconds
|
||||
When I start postgres1
|
||||
And "members/postgres1" key in DCS has state=running after 10 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8010/sync
|
||||
Then I receive a response code 200
|
||||
When I issue a GET request to http://127.0.0.1:8009/async
|
||||
Then I receive a response code 200
|
||||
|
||||
Scenario: check the basic failover in synchronous mode
|
||||
Given I run patronictl.py pause batman
|
||||
Then I receive a response returncode 0
|
||||
When I sleep for 2 seconds
|
||||
And I shut down postgres0
|
||||
And I run patronictl.py resume batman
|
||||
Then I receive a response returncode 0
|
||||
And postgres2 role is the primary after 24 seconds
|
||||
And Response on GET http://127.0.0.1:8010/history contains recovery after 10 seconds
|
||||
When I issue a PATCH request to http://127.0.0.1:8010/config with {"synchronous_mode": null, "master_start_timeout": 0}
|
||||
Then I receive a response code 200
|
||||
When I add the table bar to postgres2
|
||||
Then table bar is present on postgres1 after 20 seconds
|
||||
And Response on GET http://127.0.0.1:8010/config contains master_start_timeout after 10 seconds
|
||||
|
||||
Scenario: check immediate failover when master_start_timeout=0
|
||||
Given I kill postmaster on postgres2
|
||||
Then postgres1 is a leader after 10 seconds
|
||||
And postgres1 role is the primary after 10 seconds
|
||||
|
||||
Scenario: check rejoin of the former master with pg_rewind
|
||||
Given I add the table splitbrain to postgres0
|
||||
And I start postgres0
|
||||
Then postgres0 role is the secondary after 20 seconds
|
||||
When I add the table bar to postgres1
|
||||
Then table bar is present on postgres0 after 20 seconds
|
||||
When I add the table buz to postgres1
|
||||
Then table buz is present on postgres0 after 20 seconds
|
||||
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env python
|
||||
|
||||
import sys
|
||||
with open("data/{0}/{0}_cb.log".format(sys.argv[1]), "a+") as log:
|
||||
log.write(" ".join(sys.argv[-3:]) + "\n")
|
||||
@@ -8,6 +8,7 @@ Scenario: check a base backup and streaming replication from a replica
|
||||
And replication works from postgres0 to postgres1 after 20 seconds
|
||||
And I create label with "postgres0" in postgres0 data directory
|
||||
And I create label with "postgres1" in postgres1 data directory
|
||||
And "members/postgres1" key in DCS has state=running after 12 seconds
|
||||
And I configure and start postgres2 with a tag replicatefrom postgres1
|
||||
Then replication works from postgres0 to postgres2 after 30 seconds
|
||||
And there is a label with "postgres1" in postgres2 data directory
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
Feature: custom bootstrap
|
||||
We should check that patroni can bootstrap a new cluster from a backup
|
||||
|
||||
Scenario: clone existing cluster using pg_basebackup
|
||||
Given I start postgres0
|
||||
Then postgres0 is a leader after 10 seconds
|
||||
When I add the table foo to postgres0
|
||||
And I start postgres1 in a cluster batman1 as a clone of postgres0
|
||||
Then postgres1 is a leader of batman1 after 10 seconds
|
||||
Then table foo is present on postgres1 after 10 seconds
|
||||
|
||||
Scenario: make a backup and do a restore into a new cluster
|
||||
Given I add the table bar to postgres1
|
||||
And I do a backup of postgres1
|
||||
When I start postgres2 in a cluster batman2 from backup
|
||||
Then postgres2 is a leader of batman2 after 30 seconds
|
||||
And table bar is present on postgres2 after 10 seconds
|
||||
+590
-98
@@ -1,22 +1,27 @@
|
||||
import abc
|
||||
import consul
|
||||
import etcd
|
||||
import kazoo.client
|
||||
import kazoo.exceptions
|
||||
import datetime
|
||||
import os
|
||||
import psycopg2
|
||||
import json
|
||||
import shutil
|
||||
import signal
|
||||
import six
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import threading
|
||||
import time
|
||||
import yaml
|
||||
|
||||
import patroni.psycopg as psycopg
|
||||
|
||||
from six.moves.BaseHTTPServer import BaseHTTPRequestHandler, HTTPServer
|
||||
|
||||
|
||||
@six.add_metaclass(abc.ABCMeta)
|
||||
class AbstractController(object):
|
||||
|
||||
def __init__(self, name, work_directory, output_dir):
|
||||
def __init__(self, context, name, work_directory, output_dir):
|
||||
self._context = context
|
||||
self._name = name
|
||||
self._work_directory = work_directory
|
||||
self._output_dir = output_dir
|
||||
@@ -46,6 +51,7 @@ class AbstractController(object):
|
||||
|
||||
assert self._has_started(), "Process {0} is not running after being started".format(self._name)
|
||||
|
||||
max_wait_limit *= self._context.timeout_multiplier
|
||||
for _ in range(max_wait_limit):
|
||||
if self._is_accessible():
|
||||
break
|
||||
@@ -54,10 +60,11 @@ class AbstractController(object):
|
||||
assert False,\
|
||||
"{0} instance is not available for queries after {1} seconds".format(self._name, max_wait_limit)
|
||||
|
||||
def stop(self, kill=False, timeout=15):
|
||||
def stop(self, kill=False, timeout=15, _=False):
|
||||
term = False
|
||||
start_time = time.time()
|
||||
|
||||
timeout *= self._context.timeout_multiplier
|
||||
while self._handle and self._is_running():
|
||||
if kill:
|
||||
self._handle.kill()
|
||||
@@ -71,18 +78,29 @@ class AbstractController(object):
|
||||
if self._log:
|
||||
self._log.close()
|
||||
|
||||
def cancel_background(self):
|
||||
pass
|
||||
|
||||
|
||||
class PatroniController(AbstractController):
|
||||
__PORT = 5440
|
||||
__PORT = 5360
|
||||
PATRONI_CONFIG = '{}.yml'
|
||||
""" starts and stops individual patronis"""
|
||||
|
||||
def __init__(self, dcs, name, work_directory, output_dir, tags=None):
|
||||
super(PatroniController, self).__init__('patroni_' + name, work_directory, output_dir)
|
||||
def __init__(self, context, name, work_directory, output_dir, custom_config=None):
|
||||
super(PatroniController, self).__init__(context, 'patroni_' + name, work_directory, output_dir)
|
||||
PatroniController.__PORT += 1
|
||||
self._data_dir = os.path.join(work_directory, 'data', name)
|
||||
self._connstring = None
|
||||
self._config = self._make_patroni_test_config(name, dcs, tags)
|
||||
if custom_config and 'watchdog' in custom_config:
|
||||
self.watchdog = WatchdogMonitor(name, work_directory, output_dir)
|
||||
custom_config['watchdog'] = {'driver': 'testing', 'device': self.watchdog.fifo_path, 'mode': 'required'}
|
||||
else:
|
||||
self.watchdog = None
|
||||
|
||||
self._scope = (custom_config or {}).get('scope', 'batman')
|
||||
self._config = self._make_patroni_test_config(name, custom_config)
|
||||
self._closables = []
|
||||
|
||||
self._conn = None
|
||||
self._curs = None
|
||||
@@ -91,63 +109,113 @@ class PatroniController(AbstractController):
|
||||
with open(os.path.join(self._data_dir, 'label'), 'w') as f:
|
||||
f.write(content)
|
||||
|
||||
def read_label(self):
|
||||
def read_label(self, label):
|
||||
try:
|
||||
with open(os.path.join(self._data_dir, 'label'), 'r') as f:
|
||||
with open(os.path.join(self._data_dir, label), 'r') as f:
|
||||
return f.read().strip()
|
||||
except IOError:
|
||||
return None
|
||||
|
||||
def add_tag_to_config(self, tag, value):
|
||||
@staticmethod
|
||||
def recursive_update(dst, src):
|
||||
for k, v in src.items():
|
||||
if k in dst and isinstance(dst[k], dict):
|
||||
PatroniController.recursive_update(dst[k], v)
|
||||
else:
|
||||
dst[k] = v
|
||||
|
||||
def update_config(self, custom_config):
|
||||
with open(self._config) as r:
|
||||
config = yaml.safe_load(r)
|
||||
config['tags']['tag'] = value
|
||||
self.recursive_update(config, custom_config)
|
||||
with open(self._config, 'w') as w:
|
||||
yaml.safe_dump(config, w, default_flow_style=False)
|
||||
self._scope = config.get('scope', 'batman')
|
||||
|
||||
def add_tag_to_config(self, tag, value):
|
||||
self.update_config({'tags': {tag: value}})
|
||||
|
||||
def _start(self):
|
||||
return subprocess.Popen(['coverage', 'run', '--source=patroni', '-p', 'patroni.py', self._config],
|
||||
if self.watchdog:
|
||||
self.watchdog.start()
|
||||
if isinstance(self._context.dcs_ctl, KubernetesController):
|
||||
self._context.dcs_ctl.create_pod(self._name[8:], self._scope)
|
||||
os.environ['PATRONI_KUBERNETES_POD_IP'] = '10.0.0.' + self._name[-1]
|
||||
return subprocess.Popen([sys.executable, '-m', 'coverage', 'run',
|
||||
'--source=patroni', '-p', 'patroni.py', self._config],
|
||||
stdout=self._log, stderr=subprocess.STDOUT, cwd=self._work_directory)
|
||||
|
||||
def _is_accessible(self):
|
||||
return self.query("SELECT 1", fail_ok=True) is not None
|
||||
def stop(self, kill=False, timeout=15, postgres=False):
|
||||
if postgres:
|
||||
return subprocess.call(['pg_ctl', '-D', self._data_dir, 'stop', '-mi', '-w'])
|
||||
super(PatroniController, self).stop(kill, timeout)
|
||||
if isinstance(self._context.dcs_ctl, KubernetesController):
|
||||
self._context.dcs_ctl.delete_pod(self._name[8:])
|
||||
if self.watchdog:
|
||||
self.watchdog.stop()
|
||||
|
||||
def _make_patroni_test_config(self, name, dcs, tags):
|
||||
def _is_accessible(self):
|
||||
cursor = self.query("SELECT 1", fail_ok=True)
|
||||
if cursor is not None:
|
||||
cursor.execute("SET synchronous_commit TO 'local'")
|
||||
return True
|
||||
|
||||
def _make_patroni_test_config(self, name, custom_config):
|
||||
patroni_config_name = self.PATRONI_CONFIG.format(name)
|
||||
patroni_config_path = os.path.join(self._output_dir, patroni_config_name)
|
||||
|
||||
with open(patroni_config_name) as f:
|
||||
config = yaml.safe_load(f)
|
||||
config.pop('etcd')
|
||||
config.pop('etcd', None)
|
||||
|
||||
raft_port = os.environ.get('RAFT_PORT')
|
||||
if raft_port:
|
||||
os.environ['RAFT_PORT'] = str(int(raft_port) + 1)
|
||||
config['raft'] = {'data_dir': self._output_dir, 'self_addr': 'localhost:' + os.environ['RAFT_PORT']}
|
||||
|
||||
host = config['postgresql']['listen'].split(':')[0]
|
||||
|
||||
config['postgresql']['listen'] = config['postgresql']['connect_address'] = '{0}:{1}'.format(host, self.__PORT)
|
||||
|
||||
user = config['postgresql'].get('authentication', config['postgresql']).get('superuser', {})
|
||||
self._connkwargs = {k: user[n] for n, k in [('username', 'user'), ('password', 'password')] if n in user}
|
||||
self._connkwargs.update({'host': host, 'port': self.__PORT, 'database': 'postgres'})
|
||||
|
||||
config['name'] = name
|
||||
config['postgresql']['data_dir'] = self._data_dir
|
||||
config['postgresql']['basebackup'] = [{'checkpoint': 'fast'}]
|
||||
config['postgresql']['use_unix_socket'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets
|
||||
config['postgresql']['use_unix_socket_repl'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets
|
||||
config['postgresql']['pgpass'] = os.path.join(tempfile.gettempdir(), 'pgpass_' + name)
|
||||
config['postgresql']['parameters'].update({
|
||||
'logging_collector': 'on', 'log_destination': 'csvlog', 'log_directory': self._output_dir,
|
||||
'log_filename': name + '.log', 'log_statement': 'all', 'log_min_messages': 'debug1'})
|
||||
'log_filename': name + '.log', 'log_statement': 'all', 'log_min_messages': 'debug1',
|
||||
'unix_socket_directories': tempfile.gettempdir()})
|
||||
|
||||
if 'bootstrap' in config and 'initdb' in config['bootstrap']:
|
||||
config['bootstrap']['initdb'].extend([{'auth': 'md5'}, {'auth-host': 'md5'}])
|
||||
if 'bootstrap' in config:
|
||||
config['bootstrap']['post_bootstrap'] = 'psql -w -c "SELECT 1"'
|
||||
if 'initdb' in config['bootstrap']:
|
||||
config['bootstrap']['initdb'].extend([{'auth': 'md5'}, {'auth-host': 'md5'}])
|
||||
|
||||
if tags:
|
||||
config['tags'] = tags
|
||||
if custom_config is not None:
|
||||
self.recursive_update(config, custom_config)
|
||||
|
||||
self.recursive_update(config, {
|
||||
'bootstrap': {'dcs': {'postgresql': {'parameters': {'wal_keep_segments': 100}}}}})
|
||||
if config['postgresql'].get('callbacks', {}).get('on_role_change'):
|
||||
config['postgresql']['callbacks']['on_role_change'] += ' ' + str(self.__PORT)
|
||||
|
||||
with open(patroni_config_path, 'w') as f:
|
||||
yaml.safe_dump(config, f, default_flow_style=False)
|
||||
|
||||
user = config['postgresql'].get('authentication', config['postgresql']).get('superuser', {})
|
||||
self._connkwargs = {k: user[n] for n, k in [('username', 'user'), ('password', 'password')] if n in user}
|
||||
self._connkwargs.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres'})
|
||||
|
||||
self._replication = config['postgresql'].get('authentication', config['postgresql']).get('replication', {})
|
||||
self._replication.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres'})
|
||||
|
||||
return patroni_config_path
|
||||
|
||||
def _connection(self):
|
||||
if not self._conn or self._conn.closed != 0:
|
||||
self._conn = psycopg2.connect(**self._connkwargs)
|
||||
self._conn = psycopg.connect(**self._connkwargs)
|
||||
self._conn.autocommit = True
|
||||
return self._conn
|
||||
|
||||
@@ -161,7 +229,7 @@ class PatroniController(AbstractController):
|
||||
cursor = self._cursor()
|
||||
cursor.execute(query)
|
||||
return cursor
|
||||
except psycopg2.Error:
|
||||
except psycopg.Error:
|
||||
if not fail_ok:
|
||||
raise
|
||||
|
||||
@@ -177,46 +245,125 @@ class PatroniController(AbstractController):
|
||||
time.sleep(1)
|
||||
return False
|
||||
|
||||
def get_watchdog(self):
|
||||
return self.watchdog
|
||||
|
||||
def _get_pid(self):
|
||||
try:
|
||||
pidfile = os.path.join(self._data_dir, 'postmaster.pid')
|
||||
if not os.path.exists(pidfile):
|
||||
return None
|
||||
return int(open(pidfile).readline().strip())
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
def patroni_hang(self, timeout):
|
||||
hang = ProcessHang(self._handle.pid, timeout)
|
||||
self._closables.append(hang)
|
||||
hang.start()
|
||||
|
||||
def cancel_background(self):
|
||||
for obj in self._closables:
|
||||
obj.close()
|
||||
self._closables = []
|
||||
|
||||
@property
|
||||
def backup_source(self):
|
||||
return 'postgres://{username}:{password}@{host}:{port}/{dbname}'.format(**self._replication)
|
||||
|
||||
def backup(self, dest=os.path.join('data', 'basebackup')):
|
||||
subprocess.call(PatroniPoolController.BACKUP_SCRIPT + ['--walmethod=none',
|
||||
'--datadir=' + os.path.join(self._work_directory, dest),
|
||||
'--dbname=' + self.backup_source])
|
||||
|
||||
|
||||
class ProcessHang(object):
|
||||
|
||||
"""A background thread implementing a cancelable process hang via SIGSTOP."""
|
||||
|
||||
def __init__(self, pid, timeout):
|
||||
self._cancelled = threading.Event()
|
||||
self._thread = threading.Thread(target=self.run)
|
||||
self.pid = pid
|
||||
self.timeout = timeout
|
||||
|
||||
def start(self):
|
||||
self._thread.start()
|
||||
|
||||
def run(self):
|
||||
os.kill(self.pid, signal.SIGSTOP)
|
||||
try:
|
||||
self._cancelled.wait(self.timeout)
|
||||
finally:
|
||||
os.kill(self.pid, signal.SIGCONT)
|
||||
|
||||
def close(self):
|
||||
self._cancelled.set()
|
||||
self._thread.join()
|
||||
|
||||
|
||||
class AbstractDcsController(AbstractController):
|
||||
|
||||
_CLUSTER_NODE = '/service/batman'
|
||||
_CLUSTER_NODE = '/service/{0}'
|
||||
|
||||
def __init__(self, context, mktemp=True):
|
||||
work_directory = mktemp and tempfile.mkdtemp() or None
|
||||
super(AbstractDcsController, self).__init__(context, self.name(), work_directory, context.pctl.output_dir)
|
||||
|
||||
def _is_accessible(self):
|
||||
return self._is_running()
|
||||
|
||||
def stop_and_remove_work_directory(self, timeout=15):
|
||||
def stop(self, kill=False, timeout=15):
|
||||
""" terminate process and wipe out the temp work directory, but only if we actually started it"""
|
||||
self.stop(timeout=timeout)
|
||||
super(AbstractDcsController, self).stop(kill=kill, timeout=timeout)
|
||||
if self._work_directory:
|
||||
shutil.rmtree(self._work_directory)
|
||||
|
||||
def path(self, key=None):
|
||||
return self._CLUSTER_NODE + (key and '/' + key or '')
|
||||
def path(self, key=None, scope='batman'):
|
||||
return self._CLUSTER_NODE.format(scope) + (key and '/' + key or '')
|
||||
|
||||
@abc.abstractmethod
|
||||
def query(self, key):
|
||||
def query(self, key, scope='batman'):
|
||||
""" query for a value of a given key """
|
||||
|
||||
@abc.abstractmethod
|
||||
def set(self, key, value):
|
||||
""" set a value to a given key """
|
||||
|
||||
@abc.abstractmethod
|
||||
def cleanup_service_tree(self):
|
||||
""" clean all contents stored in the tree used for the tests """
|
||||
|
||||
@classmethod
|
||||
def get_subclasses(cls):
|
||||
for subclass in cls.__subclasses__():
|
||||
for subsubclass in subclass.get_subclasses():
|
||||
yield subsubclass
|
||||
yield subclass
|
||||
|
||||
@classmethod
|
||||
def name(cls):
|
||||
return cls.__name__[:-10].lower()
|
||||
|
||||
|
||||
class ConsulController(AbstractDcsController):
|
||||
|
||||
def __init__(self, output_dir):
|
||||
super(ConsulController, self).__init__('consul', tempfile.mkdtemp(), output_dir)
|
||||
def __init__(self, context):
|
||||
super(ConsulController, self).__init__(context)
|
||||
os.environ['PATRONI_CONSUL_HOST'] = 'localhost:8500'
|
||||
os.environ['PATRONI_CONSUL_REGISTER_SERVICE'] = 'on'
|
||||
self._config_file = None
|
||||
|
||||
import consul
|
||||
self._client = consul.Consul()
|
||||
|
||||
def _start(self):
|
||||
return subprocess.Popen(['consul', 'agent', '-server', '-bootstrap', '-advertise=127.0.0.1',
|
||||
'-data-dir', self._work_directory], stdout=self._log, stderr=subprocess.STDOUT)
|
||||
self._config_file = self._work_directory + '.json'
|
||||
with open(self._config_file, 'wb') as f:
|
||||
f.write(b'{"session_ttl_min":"5s","server":true,"bootstrap":true,"advertise_addr":"127.0.0.1"}')
|
||||
return subprocess.Popen(['consul', 'agent', '-config-file', self._config_file, '-data-dir',
|
||||
self._work_directory], stdout=self._log, stderr=subprocess.STDOUT)
|
||||
|
||||
def stop(self, kill=False, timeout=15):
|
||||
super(ConsulController, self).stop(kill=kill, timeout=timeout)
|
||||
if self._config_file:
|
||||
os.unlink(self._config_file)
|
||||
|
||||
def _is_running(self):
|
||||
try:
|
||||
@@ -224,83 +371,185 @@ class ConsulController(AbstractDcsController):
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def path(self, key=None):
|
||||
return super(ConsulController, self).path(key)[1:]
|
||||
def path(self, key=None, scope='batman'):
|
||||
return super(ConsulController, self).path(key, scope)[1:]
|
||||
|
||||
def query(self, key):
|
||||
_, value = self._client.kv.get(self.path(key))
|
||||
def query(self, key, scope='batman'):
|
||||
_, value = self._client.kv.get(self.path(key, scope))
|
||||
return value and value['Value'].decode('utf-8')
|
||||
|
||||
def set(self, key, value):
|
||||
self._client.kv.put(self.path(key), value)
|
||||
|
||||
def cleanup_service_tree(self):
|
||||
self._client.kv.delete(self.path(), recurse=True)
|
||||
self._client.kv.delete(self.path(scope=''), recurse=True)
|
||||
|
||||
def start(self, max_wait_limit=15):
|
||||
super(ConsulController, self).start(max_wait_limit)
|
||||
|
||||
|
||||
class EtcdController(AbstractDcsController):
|
||||
class AbstractEtcdController(AbstractDcsController):
|
||||
|
||||
""" handles all etcd related tasks, used for the tests setup and cleanup """
|
||||
|
||||
def __init__(self, output_dir):
|
||||
super(EtcdController, self).__init__('etcd', tempfile.mkdtemp(), output_dir)
|
||||
os.environ['PATRONI_ETCD_HOST'] = 'localhost:4001'
|
||||
self._client = etcd.Client()
|
||||
def __init__(self, context, client_cls):
|
||||
super(AbstractEtcdController, self).__init__(context)
|
||||
self._client_cls = client_cls
|
||||
|
||||
def _start(self):
|
||||
return subprocess.Popen(["etcd", "--debug", "--data-dir", self._work_directory],
|
||||
stdout=self._log, stderr=subprocess.STDOUT)
|
||||
|
||||
def query(self, key):
|
||||
def _is_running(self):
|
||||
from patroni.dcs.etcd import DnsCachingResolver
|
||||
# if etcd is running, but we didn't start it
|
||||
try:
|
||||
return self._client.get(self.path(key)).value
|
||||
self._client = self._client_cls({'host': 'localhost', 'port': 2379, 'retry_timeout': 30,
|
||||
'patronictl': 1}, DnsCachingResolver())
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
class EtcdController(AbstractEtcdController):
|
||||
|
||||
def __init__(self, context):
|
||||
from patroni.dcs.etcd import EtcdClient
|
||||
super(EtcdController, self).__init__(context, EtcdClient)
|
||||
os.environ['PATRONI_ETCD_HOST'] = 'localhost:2379'
|
||||
|
||||
def query(self, key, scope='batman'):
|
||||
import etcd
|
||||
try:
|
||||
return self._client.get(self.path(key, scope)).value
|
||||
except etcd.EtcdKeyNotFound:
|
||||
return None
|
||||
|
||||
def set(self, key, value):
|
||||
self._client.set(self.path(key), value)
|
||||
|
||||
def cleanup_service_tree(self):
|
||||
import etcd
|
||||
try:
|
||||
self._client.delete(self.path(), recursive=True)
|
||||
self._client.delete(self.path(scope=''), recursive=True)
|
||||
except (etcd.EtcdKeyNotFound, etcd.EtcdConnectionFailed):
|
||||
return
|
||||
except Exception as e:
|
||||
assert False, "exception when cleaning up etcd contents: {0}".format(e)
|
||||
|
||||
def _is_running(self):
|
||||
# if etcd is running, but we didn't start it
|
||||
|
||||
class Etcd3Controller(AbstractEtcdController):
|
||||
|
||||
def __init__(self, context):
|
||||
from patroni.dcs.etcd3 import Etcd3Client
|
||||
super(Etcd3Controller, self).__init__(context, Etcd3Client)
|
||||
os.environ['PATRONI_ETCD3_HOST'] = 'localhost:2379'
|
||||
|
||||
def query(self, key, scope='batman'):
|
||||
import base64
|
||||
response = self._client.range(self.path(key, scope))
|
||||
for k in response.get('kvs', []):
|
||||
return base64.b64decode(k['value']).decode('utf-8') if 'value' in k else None
|
||||
|
||||
def cleanup_service_tree(self):
|
||||
try:
|
||||
return bool(self._client.machines)
|
||||
self._client.deleteprefix(self.path(scope=''))
|
||||
except Exception as e:
|
||||
assert False, "exception when cleaning up etcd contents: {0}".format(e)
|
||||
|
||||
|
||||
class KubernetesController(AbstractDcsController):
|
||||
|
||||
def __init__(self, context):
|
||||
super(KubernetesController, self).__init__(context)
|
||||
self._namespace = 'default'
|
||||
self._labels = {"application": "patroni"}
|
||||
self._label_selector = ','.join('{0}={1}'.format(k, v) for k, v in self._labels.items())
|
||||
os.environ['PATRONI_KUBERNETES_LABELS'] = json.dumps(self._labels)
|
||||
os.environ['PATRONI_KUBERNETES_USE_ENDPOINTS'] = 'true'
|
||||
os.environ['PATRONI_KUBERNETES_BYPASS_API_SERVICE'] = 'true'
|
||||
|
||||
from patroni.dcs.kubernetes import k8s_client, k8s_config
|
||||
k8s_config.load_kube_config(context='local')
|
||||
self._client = k8s_client
|
||||
self._api = self._client.CoreV1Api()
|
||||
|
||||
def _start(self):
|
||||
pass
|
||||
|
||||
def create_pod(self, name, scope):
|
||||
labels = self._labels.copy()
|
||||
labels['cluster-name'] = scope
|
||||
metadata = self._client.V1ObjectMeta(namespace=self._namespace, name=name, labels=labels)
|
||||
spec = self._client.V1PodSpec(containers=[self._client.V1Container(name=name, image='empty')])
|
||||
body = self._client.V1Pod(metadata=metadata, spec=spec)
|
||||
self._api.create_namespaced_pod(self._namespace, body)
|
||||
|
||||
def delete_pod(self, name):
|
||||
try:
|
||||
self._api.delete_namespaced_pod(name, self._namespace, body=self._client.V1DeleteOptions())
|
||||
except Exception:
|
||||
return False
|
||||
pass
|
||||
while True:
|
||||
try:
|
||||
self._api.read_namespaced_pod(name, self._namespace)
|
||||
except Exception:
|
||||
break
|
||||
|
||||
def query(self, key, scope='batman'):
|
||||
if key.startswith('members/'):
|
||||
pod = self._api.read_namespaced_pod(key[8:], self._namespace)
|
||||
return (pod.metadata.annotations or {}).get('status', '')
|
||||
else:
|
||||
try:
|
||||
ep = scope + {'leader': '', 'history': '-config', 'initialize': '-config'}.get(key, '-' + key)
|
||||
e = self._api.read_namespaced_endpoints(ep, self._namespace)
|
||||
if key != 'sync':
|
||||
return e.metadata.annotations[key]
|
||||
else:
|
||||
return json.dumps(e.metadata.annotations)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
def cleanup_service_tree(self):
|
||||
try:
|
||||
self._api.delete_collection_namespaced_pod(self._namespace, label_selector=self._label_selector)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
self._api.delete_collection_namespaced_endpoints(self._namespace, label_selector=self._label_selector)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
while True:
|
||||
result = self._api.list_namespaced_pod(self._namespace, label_selector=self._label_selector)
|
||||
if len(result.items) < 1:
|
||||
break
|
||||
|
||||
def _is_running(self):
|
||||
return True
|
||||
|
||||
|
||||
class ZooKeeperController(AbstractDcsController):
|
||||
|
||||
""" handles all zookeeper related tasks, used for the tests setup and cleanup """
|
||||
|
||||
def __init__(self, output_dir, export_env=True):
|
||||
super(ZooKeeperController, self).__init__('zookeeper', None, output_dir)
|
||||
def __init__(self, context, export_env=True):
|
||||
super(ZooKeeperController, self).__init__(context, False)
|
||||
if export_env:
|
||||
os.environ['PATRONI_ZOOKEEPER_HOSTS'] = "'localhost:2181'"
|
||||
|
||||
import kazoo.client
|
||||
self._client = kazoo.client.KazooClient()
|
||||
|
||||
def _start(self):
|
||||
pass # TODO: implement later
|
||||
|
||||
def query(self, key):
|
||||
def query(self, key, scope='batman'):
|
||||
import kazoo.exceptions
|
||||
try:
|
||||
return self._client.get(self.path(key))[0].decode('utf-8')
|
||||
return self._client.get(self.path(key, scope))[0].decode('utf-8')
|
||||
except kazoo.exceptions.NoNodeError:
|
||||
return None
|
||||
|
||||
def set(self, key, value):
|
||||
self._client.set(self.path(key), value.encode('utf-8'))
|
||||
|
||||
def cleanup_service_tree(self):
|
||||
import kazoo.exceptions
|
||||
try:
|
||||
self._client.delete(self.path(), recursive=True)
|
||||
self._client.delete(self.path(scope=''), recursive=True)
|
||||
except (kazoo.exceptions.NoNodeError):
|
||||
return
|
||||
except Exception as e:
|
||||
@@ -316,24 +565,85 @@ class ZooKeeperController(AbstractDcsController):
|
||||
return False
|
||||
|
||||
|
||||
class MockExhibitor(BaseHTTPRequestHandler):
|
||||
|
||||
def do_GET(self):
|
||||
self.send_response(200)
|
||||
self.end_headers()
|
||||
self.wfile.write(b'{"servers":["127.0.0.1"],"port":2181}')
|
||||
|
||||
def log_message(self, fmt, *args):
|
||||
pass
|
||||
|
||||
|
||||
class ExhibitorController(ZooKeeperController):
|
||||
|
||||
def __init__(self, output_dir):
|
||||
super(ExhibitorController, self).__init__(output_dir, False)
|
||||
os.environ.update({'PATRONI_EXHIBITOR_HOSTS': 'localhost', 'PATRONI_EXHIBITOR_PORT': '8181'})
|
||||
def __init__(self, context):
|
||||
super(ExhibitorController, self).__init__(context, False)
|
||||
port = 8181
|
||||
exhibitor = HTTPServer(('', port), MockExhibitor)
|
||||
exhibitor.daemon_thread = True
|
||||
exhibitor_thread = threading.Thread(target=exhibitor.serve_forever)
|
||||
exhibitor_thread.daemon = True
|
||||
exhibitor_thread.start()
|
||||
os.environ.update({'PATRONI_EXHIBITOR_HOSTS': 'localhost', 'PATRONI_EXHIBITOR_PORT': str(port)})
|
||||
|
||||
|
||||
class RaftController(AbstractDcsController):
|
||||
|
||||
CONTROLLER_ADDR = 'localhost:1234'
|
||||
PASSWORD = '12345'
|
||||
|
||||
def __init__(self, context):
|
||||
super(RaftController, self).__init__(context)
|
||||
os.environ.update(PATRONI_RAFT_PARTNER_ADDRS="'" + self.CONTROLLER_ADDR + "'",
|
||||
PATRONI_RAFT_PASSWORD=self.PASSWORD, RAFT_PORT='1234')
|
||||
self._raft = None
|
||||
|
||||
def _start(self):
|
||||
env = os.environ.copy()
|
||||
del env['PATRONI_RAFT_PARTNER_ADDRS']
|
||||
env['PATRONI_RAFT_SELF_ADDR'] = self.CONTROLLER_ADDR
|
||||
env['PATRONI_RAFT_DATA_DIR'] = self._work_directory
|
||||
return subprocess.Popen([sys.executable, '-m', 'coverage', 'run',
|
||||
'--source=patroni', '-p', 'patroni_raft_controller.py'],
|
||||
stdout=self._log, stderr=subprocess.STDOUT, env=env)
|
||||
|
||||
def query(self, key, scope='batman'):
|
||||
ret = self._raft.get(self.path(key, scope))
|
||||
return ret and ret['value']
|
||||
|
||||
def set(self, key, value):
|
||||
self._raft.set(self.path(key), value)
|
||||
|
||||
def cleanup_service_tree(self):
|
||||
from patroni.dcs.raft import KVStoreTTL
|
||||
|
||||
if self._raft:
|
||||
self._raft.destroy()
|
||||
self.stop()
|
||||
os.makedirs(self._work_directory)
|
||||
self.start()
|
||||
|
||||
ready_event = threading.Event()
|
||||
self._raft = KVStoreTTL(ready_event.set, None, None, partner_addrs=[self.CONTROLLER_ADDR], password=self.PASSWORD)
|
||||
self._raft.startAutoTick()
|
||||
ready_event.wait()
|
||||
|
||||
|
||||
class PatroniPoolController(object):
|
||||
|
||||
KNOWN_DCS = {'consul': ConsulController, 'etcd': EtcdController,
|
||||
'zookeeper': ZooKeeperController, 'exhibitor': ExhibitorController}
|
||||
BACKUP_SCRIPT = [sys.executable, 'features/backup_create.py']
|
||||
ARCHIVE_RESTORE_SCRIPT = ' '.join((sys.executable, os.path.abspath('features/archive-restore.py')))
|
||||
|
||||
def __init__(self):
|
||||
def __init__(self, context):
|
||||
self._context = context
|
||||
self._dcs = None
|
||||
self._output_dir = None
|
||||
self._patroni_path = None
|
||||
self._processes = {}
|
||||
self.create_and_set_output_directory('')
|
||||
self.known_dcs = {subclass.name(): subclass for subclass in AbstractDcsController.get_subclasses()}
|
||||
|
||||
@property
|
||||
def patroni_path(self):
|
||||
@@ -350,55 +660,235 @@ class PatroniPoolController(object):
|
||||
def output_dir(self):
|
||||
return self._output_dir
|
||||
|
||||
def start(self, pg_name, max_wait_limit=20, tags=None):
|
||||
if pg_name not in self._processes:
|
||||
self._processes[pg_name] = PatroniController(self.dcs, pg_name, self.patroni_path, self._output_dir, tags)
|
||||
self._processes[pg_name].start(max_wait_limit)
|
||||
def start(self, name, max_wait_limit=40, custom_config=None):
|
||||
if name not in self._processes:
|
||||
self._processes[name] = PatroniController(self._context, name, self.patroni_path,
|
||||
self._output_dir, custom_config)
|
||||
self._processes[name].start(max_wait_limit)
|
||||
|
||||
def __getattr__(self, func):
|
||||
if func not in ['stop', 'query', 'write_label', 'read_label', 'check_role_has_changed_to', 'add_tag_to_config']:
|
||||
if func not in ['stop', 'query', 'write_label', 'read_label', 'check_role_has_changed_to',
|
||||
'add_tag_to_config', 'get_watchdog', 'patroni_hang', 'backup']:
|
||||
raise AttributeError("PatroniPoolController instance has no attribute '{0}'".format(func))
|
||||
|
||||
def wrapper(pg_name, *args, **kwargs):
|
||||
return getattr(self._processes[pg_name], func)(*args, **kwargs)
|
||||
def wrapper(name, *args, **kwargs):
|
||||
return getattr(self._processes[name], func)(*args, **kwargs)
|
||||
return wrapper
|
||||
|
||||
def stop_all(self):
|
||||
for ctl in self._processes.values():
|
||||
ctl.cancel_background()
|
||||
ctl.stop()
|
||||
self._processes.clear()
|
||||
|
||||
def create_and_set_output_directory(self, feature_name):
|
||||
feature_dir = os.path.join(self.patroni_path, 'features/output', feature_name.replace(' ', '_'))
|
||||
feature_dir = os.path.join(self.patroni_path, 'features', 'output', feature_name.replace(' ', '_'))
|
||||
if os.path.exists(feature_dir):
|
||||
shutil.rmtree(feature_dir)
|
||||
os.makedirs(feature_dir)
|
||||
self._output_dir = feature_dir
|
||||
|
||||
def clone(self, from_name, cluster_name, to_name):
|
||||
f = self._processes[from_name]
|
||||
custom_config = {
|
||||
'scope': cluster_name,
|
||||
'bootstrap': {
|
||||
'method': 'pg_basebackup',
|
||||
'pg_basebackup': {
|
||||
'command': " ".join(self.BACKUP_SCRIPT) + ' --walmethod=stream --dbname=' + f.backup_source
|
||||
},
|
||||
'dcs': {
|
||||
'postgresql': {
|
||||
'parameters': {
|
||||
'max_connections': 101
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
'postgresql': {
|
||||
'parameters': {
|
||||
'archive_mode': 'on',
|
||||
'archive_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode archive ' +
|
||||
'--dirname {} --filename %f --pathname %p').format(
|
||||
os.path.join(self.patroni_path, 'data', 'wal_archive'))
|
||||
},
|
||||
'authentication': {
|
||||
'superuser': {'password': 'zalando1'},
|
||||
'replication': {'password': 'rep-pass1'}
|
||||
}
|
||||
}
|
||||
}
|
||||
self.start(to_name, custom_config=custom_config)
|
||||
|
||||
def bootstrap_from_backup(self, name, cluster_name):
|
||||
custom_config = {
|
||||
'scope': cluster_name,
|
||||
'bootstrap': {
|
||||
'method': 'backup_restore',
|
||||
'backup_restore': {
|
||||
'command': (sys.executable + ' features/backup_restore.py --sourcedir=' +
|
||||
os.path.join(self.patroni_path, 'data', 'basebackup')),
|
||||
'recovery_conf': {
|
||||
'recovery_target_action': 'promote',
|
||||
'recovery_target_timeline': 'latest',
|
||||
'restore_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode restore ' +
|
||||
'--dirname {} --filename %f --pathname %p').format(
|
||||
os.path.join(self.patroni_path, 'data', 'wal_archive'))
|
||||
}
|
||||
}
|
||||
},
|
||||
'postgresql': {
|
||||
'authentication': {
|
||||
'superuser': {'password': 'zalando2'},
|
||||
'replication': {'password': 'rep-pass2'}
|
||||
}
|
||||
}
|
||||
}
|
||||
self.start(name, custom_config=custom_config)
|
||||
|
||||
@property
|
||||
def dcs(self):
|
||||
if self._dcs is None:
|
||||
self._dcs = os.environ.pop('DCS', 'etcd')
|
||||
assert self._dcs in self.KNOWN_DCS, 'Unsupported dcs: ' + self._dcs
|
||||
assert self._dcs in self.known_dcs, 'Unsupported dcs: ' + self._dcs
|
||||
return self._dcs
|
||||
|
||||
|
||||
# actions to execute on start/stop of the tests and before running invidual features
|
||||
class WatchdogMonitor(object):
|
||||
"""Testing harness for emulating a watchdog device as a named pipe. Because we can't easily emulate ioctl's we
|
||||
require a custom driver on Patroni side. The device takes no action, only notes if it was pinged and/or triggered.
|
||||
"""
|
||||
def __init__(self, name, work_directory, output_dir):
|
||||
self.fifo_path = os.path.join(work_directory, 'data', 'watchdog.{0}.fifo'.format(name))
|
||||
self.fifo_file = None
|
||||
self._stop_requested = False # Relying on bool setting being atomic
|
||||
self._thread = None
|
||||
self.last_ping = None
|
||||
self.was_pinged = False
|
||||
self.was_closed = False
|
||||
self._was_triggered = False
|
||||
self.timeout = 60
|
||||
self._log_file = open(os.path.join(output_dir, 'watchdog.{0}.log'.format(name)), 'w')
|
||||
self._log("watchdog {0} initialized".format(name))
|
||||
|
||||
def _log(self, msg):
|
||||
tstamp = datetime.datetime.now().strftime("%Y-%m-%d %H:%M:%S,%f")
|
||||
self._log_file.write("{0}: {1}\n".format(tstamp, msg))
|
||||
|
||||
def start(self):
|
||||
assert self._thread is None
|
||||
self._stop_requested = False
|
||||
self._log("starting fifo {0}".format(self.fifo_path))
|
||||
fifo_dir = os.path.dirname(self.fifo_path)
|
||||
if os.path.exists(self.fifo_path):
|
||||
os.unlink(self.fifo_path)
|
||||
elif not os.path.exists(fifo_dir):
|
||||
os.mkdir(fifo_dir)
|
||||
os.mkfifo(self.fifo_path)
|
||||
self.last_ping = time.time()
|
||||
|
||||
self._thread = threading.Thread(target=self.run)
|
||||
self._thread.start()
|
||||
|
||||
def run(self):
|
||||
try:
|
||||
while not self._stop_requested:
|
||||
self._log("opening")
|
||||
self.fifo_file = os.open(self.fifo_path, os.O_RDONLY)
|
||||
try:
|
||||
self._log("Fifo {0} connected".format(self.fifo_path))
|
||||
self.was_closed = False
|
||||
while not self._stop_requested:
|
||||
c = os.read(self.fifo_file, 1)
|
||||
|
||||
if c == b'X':
|
||||
self._log("Stop requested")
|
||||
return
|
||||
elif c == b'':
|
||||
self._log("Pipe closed")
|
||||
break
|
||||
elif c == b'C':
|
||||
command = b''
|
||||
c = os.read(self.fifo_file, 1)
|
||||
while c != b'\n' and c != b'':
|
||||
command += c
|
||||
c = os.read(self.fifo_file, 1)
|
||||
command = command.decode('utf8')
|
||||
|
||||
if command.startswith('timeout='):
|
||||
self.timeout = int(command.split('=')[1])
|
||||
self._log("timeout={0}".format(self.timeout))
|
||||
elif c in [b'V', b'1']:
|
||||
cur_time = time.time()
|
||||
if cur_time - self.last_ping > self.timeout:
|
||||
self._log("Triggered")
|
||||
self._was_triggered = True
|
||||
if c == b'V':
|
||||
self._log("magic close")
|
||||
self.was_closed = True
|
||||
elif c == b'1':
|
||||
self.was_pinged = True
|
||||
self._log("ping after {0} seconds".format(cur_time - (self.last_ping or cur_time)))
|
||||
self.last_ping = cur_time
|
||||
else:
|
||||
self._log('Unknown command {0} received from fifo'.format(c))
|
||||
finally:
|
||||
self.was_closed = True
|
||||
self._log("closing")
|
||||
os.close(self.fifo_file)
|
||||
except Exception as e:
|
||||
self._log("Error {0}".format(e))
|
||||
finally:
|
||||
self._log("stopping")
|
||||
self._log_file.flush()
|
||||
if os.path.exists(self.fifo_path):
|
||||
os.unlink(self.fifo_path)
|
||||
|
||||
def stop(self):
|
||||
self._log("Monitor stop")
|
||||
self._stop_requested = True
|
||||
try:
|
||||
if os.path.exists(self.fifo_path):
|
||||
fd = os.open(self.fifo_path, os.O_WRONLY)
|
||||
os.write(fd, b'X')
|
||||
os.close(fd)
|
||||
except Exception as e:
|
||||
self._log("err while closing: {0}".format(str(e)))
|
||||
if self._thread:
|
||||
self._thread.join()
|
||||
self._thread = None
|
||||
|
||||
def reset(self):
|
||||
self._log("reset")
|
||||
self.was_pinged = self.was_closed = self._was_triggered = False
|
||||
|
||||
@property
|
||||
def was_triggered(self):
|
||||
delta = time.time() - self.last_ping
|
||||
triggered = self._was_triggered or not self.was_closed and delta > self.timeout
|
||||
self._log("triggered={0}, {1}s left".format(triggered, self.timeout - delta))
|
||||
return triggered
|
||||
|
||||
|
||||
# actions to execute on start/stop of the tests and before running individual features
|
||||
def before_all(context):
|
||||
context.pctl = PatroniPoolController()
|
||||
context.dcs_ctl = context.pctl.KNOWN_DCS[context.pctl.dcs](context.pctl.output_dir)
|
||||
os.environ.update({'PATRONI_RESTAPI_USERNAME': 'username', 'PATRONI_RESTAPI_PASSWORD': 'password'})
|
||||
context.ci = any(a in os.environ for a in ('TRAVIS_BUILD_NUMBER', 'BUILD_NUMBER', 'GITHUB_ACTIONS'))
|
||||
context.timeout_multiplier = 5 if context.ci else 1 # MacOS sometimes is VERY slow
|
||||
context.pctl = PatroniPoolController(context)
|
||||
context.dcs_ctl = context.pctl.known_dcs[context.pctl.dcs](context)
|
||||
context.dcs_ctl.start()
|
||||
try:
|
||||
context.dcs_ctl.cleanup_service_tree()
|
||||
except AssertionError: # after_all handlers won't be executed in before_all
|
||||
context.dcs_ctl.stop_and_remove_work_directory()
|
||||
context.dcs_ctl.stop()
|
||||
raise
|
||||
|
||||
|
||||
def after_all(context):
|
||||
context.dcs_ctl.stop_and_remove_work_directory()
|
||||
subprocess.call(['coverage', 'combine'])
|
||||
subprocess.call(['coverage', 'report'])
|
||||
context.dcs_ctl.stop()
|
||||
subprocess.call([sys.executable, '-m', 'coverage', 'combine'])
|
||||
subprocess.call([sys.executable, '-m', 'coverage', 'report'])
|
||||
|
||||
|
||||
def before_feature(context, feature):
|
||||
@@ -411,3 +901,5 @@ def after_feature(context, feature):
|
||||
context.pctl.stop_all()
|
||||
shutil.rmtree(os.path.join(context.pctl.patroni_path, 'data'))
|
||||
context.dcs_ctl.cleanup_service_tree()
|
||||
if feature.status == 'failed':
|
||||
shutil.copytree(context.pctl.output_dir, context.pctl.output_dir + '_failed')
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
Feature: ignored slots
|
||||
Scenario: check ignored slots aren't removed on failover/switchover
|
||||
Given I start postgres1
|
||||
Then postgres1 is a leader after 10 seconds
|
||||
And there is a non empty initialize key in DCS after 15 seconds
|
||||
When I issue a PATCH request to http://127.0.0.1:8009/config with {"loop_wait": 2, "ignore_slots": [{"name": "unmanaged_slot_0", "database": "postgres", "plugin": "test_decoding", "type": "logical"}, {"name": "unmanaged_slot_1", "database": "postgres", "plugin": "test_decoding"}, {"name": "unmanaged_slot_2", "database": "postgres"}, {"name": "unmanaged_slot_3"}], "postgresql": {"parameters": {"wal_level": "logical"}}}
|
||||
Then I receive a response code 200
|
||||
And Response on GET http://127.0.0.1:8009/config contains ignore_slots after 10 seconds
|
||||
# Make sure the wal_level has been changed.
|
||||
When I shut down postgres1
|
||||
And I start postgres1
|
||||
Then postgres1 is a leader after 10 seconds
|
||||
And "members/postgres1" key in DCS has role=master after 3 seconds
|
||||
# Make sure Patroni has finished telling Postgres it should be accepting writes.
|
||||
And postgres1 role is the primary after 20 seconds
|
||||
# 1. Create our test logical replication slot.
|
||||
# Test that ny subset of attributes in the ignore slots matcher is enough to match a slot
|
||||
# by using 3 different slots.
|
||||
When I create a logical replication slot unmanaged_slot_0 on postgres1 with the test_decoding plugin
|
||||
And I create a logical replication slot unmanaged_slot_1 on postgres1 with the test_decoding plugin
|
||||
And I create a logical replication slot unmanaged_slot_2 on postgres1 with the test_decoding plugin
|
||||
And I create a logical replication slot unmanaged_slot_3 on postgres1 with the test_decoding plugin
|
||||
And I create a logical replication slot dummy_slot on postgres1 with the test_decoding plugin
|
||||
# It seems like it'd be obvious that these slots exist since we just created them,
|
||||
# but Patroni can actually end up dropping them almost immediately, so it's helpful
|
||||
# to verify they exist before we begin testing whether they persist through failover
|
||||
# cycles.
|
||||
Then postgres1 has a logical replication slot named unmanaged_slot_0 with the test_decoding plugin
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_1 with the test_decoding plugin
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_2 with the test_decoding plugin
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_3 with the test_decoding plugin
|
||||
|
||||
When I start postgres0
|
||||
Then "members/postgres0" key in DCS has role=replica after 3 seconds
|
||||
And postgres0 role is the secondary after 20 seconds
|
||||
# Verify that the replica has advanced beyond the point in the WAL
|
||||
# where we created the replication slot so that on the next failover
|
||||
# cycle we don't accidentally rewind to before the slot creation.
|
||||
And replication works from postgres1 to postgres0 after 20 seconds
|
||||
When I shut down postgres1
|
||||
Then "members/postgres0" key in DCS has role=master after 3 seconds
|
||||
|
||||
# 2. After a failover the server (now a replica) still has the slot.
|
||||
When I start postgres1
|
||||
Then postgres1 role is the secondary after 20 seconds
|
||||
And "members/postgres1" key in DCS has role=replica after 3 seconds
|
||||
# give Patroni time to sync replication slots
|
||||
And I sleep for 2 seconds
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_0 with the test_decoding plugin
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_1 with the test_decoding plugin
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_2 with the test_decoding plugin
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_3 with the test_decoding plugin
|
||||
And postgres1 does not have a logical replication slot named dummy_slot
|
||||
|
||||
# 3. After a failover the server (now a master) still has the slot.
|
||||
When I shut down postgres0
|
||||
Then "members/postgres1" key in DCS has role=master after 3 seconds
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_0 with the test_decoding plugin
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_1 with the test_decoding plugin
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_2 with the test_decoding plugin
|
||||
And postgres1 has a logical replication slot named unmanaged_slot_3 with the test_decoding plugin
|
||||
@@ -8,72 +8,119 @@ Scenario: check API requests on a stand-alone server
|
||||
Then I receive a response code 200
|
||||
And I receive a response state running
|
||||
And I receive a response role master
|
||||
When I issue a GET request to http://127.0.0.1:8008/standby_leader
|
||||
Then I receive a response code 503
|
||||
When I issue a GET request to http://127.0.0.1:8008/health
|
||||
Then I receive a response code 200
|
||||
When I issue a GET request to http://127.0.0.1:8008/replica
|
||||
Then I receive a response code 503
|
||||
When I run patronictl.py reinit batman postgres0 --force
|
||||
Then I receive a response returncode 0
|
||||
And I receive a response output "reinitialize failed for member postgres0, status code=503, (I am the leader, can not reinitialize)"
|
||||
When I run patronictl.py failover batman --master postgres0 --force
|
||||
And I receive a response output "Failed: reinitialize for member postgres0, status code=503, (I am the leader, can not reinitialize)"
|
||||
When I run patronictl.py switchover batman --master postgres0 --force
|
||||
Then I receive a response returncode 1
|
||||
And I receive a response output "Error: No candidates found to failover to"
|
||||
When I issue a POST request to http://127.0.0.1:8008/failover with {"leader": "postgres0"}
|
||||
Then I receive a response code 500
|
||||
And I receive a response text failover is not possible: cluster does not have members except leader
|
||||
And I receive a response output "Error: No candidates found to switchover to"
|
||||
When I issue a POST request to http://127.0.0.1:8008/switchover with {"leader": "postgres0"}
|
||||
Then I receive a response code 412
|
||||
And I receive a response text switchover is not possible: cluster does not have members except leader
|
||||
When I issue an empty POST request to http://127.0.0.1:8008/failover
|
||||
Then I receive a response code 400
|
||||
When I issue a POST request to http://127.0.0.1:8008/failover with {"foo": "bar"}
|
||||
Then I receive a response code 400
|
||||
And I receive a response text "No values given for required parameters leader and candidate"
|
||||
And I receive a response text "Failover could be performed only to a specific candidate"
|
||||
|
||||
Scenario: check local configuration reload
|
||||
Given I issue an empty POST request to http://127.0.0.1:8008/reload
|
||||
Then I receive a response code 200
|
||||
And I receive a response text nothing changed
|
||||
When I add tag new_tag new_value to postgres0 config
|
||||
Given I add tag new_tag new_value to postgres0 config
|
||||
And I issue an empty POST request to http://127.0.0.1:8008/reload
|
||||
Then I receive a response code 202
|
||||
|
||||
Scenario: check dynamic configuration change via DCS
|
||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 20, "loop_wait": 1, "postgresql": {"parameters": {"max_connections": 101}}}
|
||||
Then I receive a response code 200
|
||||
And I receive a response loop_wait 1
|
||||
Given I run patronictl.py edit-config -s 'ttl=10' -s 'loop_wait=2' -p 'max_connections=101' --force batman
|
||||
Then I receive a response returncode 0
|
||||
And I receive a response output "+loop_wait: 2"
|
||||
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 11 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8008/config
|
||||
Then I receive a response code 200
|
||||
And I receive a response loop_wait 1
|
||||
And I receive a response loop_wait 2
|
||||
When I issue a GET request to http://127.0.0.1:8008/patroni
|
||||
Then I receive a response code 200
|
||||
And I receive a response tags {'tag': 'new_value'}
|
||||
And I receive a response tags {'new_tag': 'new_value'}
|
||||
And I sleep for 4 seconds
|
||||
|
||||
Scenario: check API requests for the primary-replica pair
|
||||
Scenario: check the scheduled restart
|
||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"postgresql": {"parameters": {"superuser_reserved_connections": "6"}}}
|
||||
Then I receive a response code 200
|
||||
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 5 seconds
|
||||
Given I issue a scheduled restart at http://127.0.0.1:8008 in 5 seconds with {"role": "replica"}
|
||||
Then I receive a response code 202
|
||||
And I sleep for 8 seconds
|
||||
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 10 seconds
|
||||
Given I issue a scheduled restart at http://127.0.0.1:8008 in 5 seconds with {"restart_pending": "True"}
|
||||
Then I receive a response code 202
|
||||
And Response on GET http://127.0.0.1:8008/patroni does not contain pending_restart after 10 seconds
|
||||
And postgres0 role is the primary after 10 seconds
|
||||
|
||||
Scenario: check API requests for the primary-replica pair in the pause mode
|
||||
Given I start postgres1
|
||||
And replication works from postgres0 to postgres1 after 20 seconds
|
||||
Then replication works from postgres0 to postgres1 after 20 seconds
|
||||
When I run patronictl.py pause batman
|
||||
Then I receive a response returncode 0
|
||||
When I kill postmaster on postgres1
|
||||
And I issue a GET request to http://127.0.0.1:8009/replica
|
||||
Then I receive a response code 503
|
||||
When I run patronictl.py restart batman postgres1 --force
|
||||
Then I receive a response returncode 0
|
||||
Then replication works from postgres0 to postgres1 after 20 seconds
|
||||
And I sleep for 2 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8009/replica
|
||||
Then I receive a response code 200
|
||||
And I receive a response state running
|
||||
And I receive a response role replica
|
||||
When I run patronictl.py reinit batman postgres1 --force
|
||||
Then I receive a response returncode 0
|
||||
And I receive a response output "Succesful reinitialize on member postgres1"
|
||||
And I receive a response output "Success: reinitialize for member postgres1"
|
||||
When I run patronictl.py restart batman postgres0 --force
|
||||
Then I receive a response returncode 0
|
||||
And I receive a response output "Succesful restart on member postgres0"
|
||||
And I receive a response output "Success: restart on member postgres0"
|
||||
And postgres0 role is the primary after 5 seconds
|
||||
When I sleep for 10 seconds
|
||||
Then postgres1 role is the secondary after 15 seconds
|
||||
|
||||
Scenario: check the failover via the API
|
||||
Given I run patronictl.py failover batman --master postgres0 --candidate postgres1 --force
|
||||
Then I receive a response returncode 0
|
||||
Scenario: check the switchover via the API in the pause mode
|
||||
Given I issue a POST request to http://127.0.0.1:8008/switchover with {"leader": "postgres0", "candidate": "postgres1"}
|
||||
Then I receive a response code 200
|
||||
And postgres1 is a leader after 5 seconds
|
||||
And postgres1 role is the primary after 5 seconds
|
||||
And postgres1 role is the primary after 10 seconds
|
||||
And postgres0 role is the secondary after 10 seconds
|
||||
And replication works from postgres1 to postgres0 after 20 seconds
|
||||
And "members/postgres0" key in DCS has state=running after 10 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8008/master
|
||||
Then I receive a response code 503
|
||||
When I issue a GET request to http://127.0.0.1:8008/replica
|
||||
Then I receive a response code 200
|
||||
When I issue a GET request to http://127.0.0.1:8009/master
|
||||
Then I receive a response code 200
|
||||
When I issue a GET request to http://127.0.0.1:8009/replica
|
||||
Then I receive a response code 503
|
||||
|
||||
Scenario: check the scheduled failover
|
||||
Given I issue a scheduled failover from postgres1 to postgres0 in 1 seconds
|
||||
Scenario: check the scheduled switchover
|
||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 10 seconds
|
||||
Then I receive a response returncode 1
|
||||
And I receive a response output "Can't schedule switchover in the paused state"
|
||||
When I run patronictl.py resume batman
|
||||
Then I receive a response returncode 0
|
||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 5 seconds
|
||||
Then I receive a response returncode 0
|
||||
And postgres0 is a leader after 20 seconds
|
||||
And postgres0 role is the primary after 5 seconds
|
||||
And postgres0 role is the primary after 10 seconds
|
||||
And postgres1 role is the secondary after 10 seconds
|
||||
And replication works from postgres0 to postgres1 after 25 seconds
|
||||
And "members/postgres1" key in DCS has state=running after 10 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8008/master
|
||||
Then I receive a response code 200
|
||||
When I issue a GET request to http://127.0.0.1:8008/replica
|
||||
Then I receive a response code 503
|
||||
When I issue a GET request to http://127.0.0.1:8009/master
|
||||
Then I receive a response code 503
|
||||
When I issue a GET request to http://127.0.0.1:8009/replica
|
||||
Then I receive a response code 200
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
Feature: standby cluster
|
||||
Scenario: prepare the cluster with logical slots
|
||||
Given I start postgres1
|
||||
Then postgres1 is a leader after 10 seconds
|
||||
And there is a non empty initialize key in DCS after 15 seconds
|
||||
When I issue a PATCH request to http://127.0.0.1:8009/config with {"loop_wait": 2, "slots": {"pm_1": {"type": "physical"}}, "postgresql": {"parameters": {"wal_level": "logical"}}}
|
||||
Then I receive a response code 200
|
||||
And Response on GET http://127.0.0.1:8009/config contains slots after 10 seconds
|
||||
And I sleep for 3 seconds
|
||||
When I issue a PATCH request to http://127.0.0.1:8009/config with {"slots": {"test_logical": {"type": "logical", "database": "postgres", "plugin": "test_decoding"}}}
|
||||
Then I receive a response code 200
|
||||
And I do a backup of postgres1
|
||||
When I start postgres0
|
||||
Then "members/postgres0" key in DCS has state=running after 10 seconds
|
||||
And replication works from postgres1 to postgres0 after 15 seconds
|
||||
|
||||
@skip
|
||||
Scenario: check permanent logical slots are synced to the replica
|
||||
Given I run patronictl.py restart batman postgres1 --force
|
||||
Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds
|
||||
When I add the table replicate_me to postgres1
|
||||
And I get all changes from logical slot test_logical on postgres1
|
||||
Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds
|
||||
|
||||
Scenario: Detach exiting node from the cluster
|
||||
When I shut down postgres1
|
||||
Then postgres0 is a leader after 10 seconds
|
||||
And "members/postgres0" key in DCS has role=master after 3 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8008/
|
||||
Then I receive a response code 200
|
||||
|
||||
Scenario: check replication of a single table in a standby cluster
|
||||
Given I start postgres1 in a standby cluster batman1 as a clone of postgres0
|
||||
Then postgres1 is a leader of batman1 after 10 seconds
|
||||
When I add the table foo to postgres0
|
||||
Then table foo is present on postgres1 after 20 seconds
|
||||
And I sleep for 3 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8009/master
|
||||
Then I receive a response code 503
|
||||
When I issue a GET request to http://127.0.0.1:8009/standby_leader
|
||||
Then I receive a response code 200
|
||||
And I receive a response role standby_leader
|
||||
And there is a postgres1_cb.log with "on_role_change standby_leader batman1" in postgres1 data directory
|
||||
When I start postgres2 in a cluster batman1
|
||||
Then postgres2 role is the replica after 24 seconds
|
||||
And table foo is present on postgres2 after 20 seconds
|
||||
And postgres1 does not have a logical replication slot named test_logical
|
||||
|
||||
Scenario: check failover
|
||||
When I kill postgres1
|
||||
And I kill postmaster on postgres1
|
||||
Then postgres2 is replicating from postgres0 after 32 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8010/master
|
||||
Then I receive a response code 503
|
||||
And I sleep for 3 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8010/standby_leader
|
||||
Then I receive a response code 200
|
||||
And I receive a response role standby_leader
|
||||
And replication works from postgres0 to postgres2 after 15 seconds
|
||||
And there is a postgres2_cb.log with "on_start replica batman1\non_role_change standby_leader batman1" in postgres2 data directory
|
||||
@@ -1,4 +1,4 @@
|
||||
import psycopg2 as pg
|
||||
import patroni.psycopg as pg
|
||||
|
||||
from behave import step, then
|
||||
from time import sleep, time
|
||||
@@ -11,7 +11,7 @@ def start_patroni(context, name):
|
||||
|
||||
@step('I shut down {name:w}')
|
||||
def stop_patroni(context, name):
|
||||
return context.pctl.stop(name)
|
||||
return context.pctl.stop(name, timeout=60)
|
||||
|
||||
|
||||
@step('I kill {name:w}')
|
||||
@@ -19,6 +19,11 @@ def kill_patroni(context, name):
|
||||
return context.pctl.stop(name, kill=True)
|
||||
|
||||
|
||||
@step('I kill postmaster on {name:w}')
|
||||
def stop_postgres(context, name):
|
||||
return context.pctl.stop(name, postgres=True)
|
||||
|
||||
|
||||
@step('I add the table {table_name:w} to {pg_name:w}')
|
||||
def add_table(context, table_name, pg_name):
|
||||
# parse the configuration file and get the port
|
||||
@@ -28,8 +33,40 @@ def add_table(context, table_name, pg_name):
|
||||
assert False, "Error creating table {0} on {1}: {2}".format(table_name, pg_name, e)
|
||||
|
||||
|
||||
@step('I {action:w} wal replay on {pg_name:w}')
|
||||
def toggle_wal_replay(context, action, pg_name):
|
||||
# pause or resume the wal replay process
|
||||
try:
|
||||
version = context.pctl.query(pg_name, "select pg_catalog.pg_read_file('PG_VERSION', 0, 2)").fetchone()
|
||||
wal = version and version[0] and int(version[0].split('.')[0]) < 10 and "xlog" or "wal"
|
||||
context.pctl.query(pg_name, "SELECT pg_{0}_replay_{1}()".format(wal, action))
|
||||
except pg.Error as e:
|
||||
assert False, "Error during {0} wal recovery on {1}: {2}".format(action, pg_name, e)
|
||||
|
||||
|
||||
@step('I {action:w} table on {pg_name:w}')
|
||||
def crdr_mytest(context, action, pg_name):
|
||||
try:
|
||||
if (action == "create"):
|
||||
context.pctl.query(pg_name, "create table if not exists mytest(id Numeric)")
|
||||
else:
|
||||
context.pctl.query(pg_name, "drop table if exists mytest")
|
||||
except pg.Error as e:
|
||||
assert False, "Error {0} table mytest on {1}: {2}".format(action, pg_name, e)
|
||||
|
||||
|
||||
@step('I load data on {pg_name:w}')
|
||||
def initiate_load(context, pg_name):
|
||||
# perform dummy load
|
||||
try:
|
||||
context.pctl.query(pg_name, "begin; insert into mytest select r::numeric from generate_series(1, 350000) r; commit;")
|
||||
except pg.Error as e:
|
||||
assert False, "Error loading test data on {0}: {1}".format(pg_name, e)
|
||||
|
||||
|
||||
@then('Table {table_name:w} is present on {pg_name:w} after {max_replication_delay:d} seconds')
|
||||
def table_is_present_on(context, table_name, pg_name, max_replication_delay):
|
||||
max_replication_delay *= context.timeout_multiplier
|
||||
for _ in range(int(max_replication_delay)):
|
||||
if context.pctl.query(pg_name, "SELECT 1 FROM {0}".format(table_name), fail_ok=True) is not None:
|
||||
break
|
||||
@@ -41,6 +78,7 @@ def table_is_present_on(context, table_name, pg_name, max_replication_delay):
|
||||
|
||||
@then('{pg_name:w} role is the {pg_role:w} after {max_promotion_timeout:d} seconds')
|
||||
def check_role(context, pg_name, pg_role, max_promotion_timeout):
|
||||
max_promotion_timeout *= context.timeout_multiplier
|
||||
assert context.pctl.check_role_has_changed_to(pg_name, pg_role, timeout=int(max_promotion_timeout)),\
|
||||
"{0} role didn't change to {1} after {2} seconds".format(pg_name, pg_role, max_promotion_timeout)
|
||||
|
||||
|
||||
@@ -1,17 +1,55 @@
|
||||
import json
|
||||
import time
|
||||
|
||||
from behave import step, then
|
||||
|
||||
|
||||
@step('I configure and start {name:w} with a tag {tag_name:w} {tag_value:w}')
|
||||
def start_patroni_with_a_name_value_tag(context, name, tag_name, tag_value):
|
||||
return context.pctl.start(name, tags={tag_name: tag_value})
|
||||
return context.pctl.start(name, custom_config={'tags': {tag_name: tag_value}})
|
||||
|
||||
|
||||
@then('There is a label with "{content:w}" in {name:w} data directory')
|
||||
def check_label(context, content, name):
|
||||
label = context.pctl.read_label(name)
|
||||
assert label == content, "{0} is not equal to {1}".format(label, content)
|
||||
@then('There is a {label} with "{content}" in {name:w} data directory')
|
||||
def check_label(context, label, content, name):
|
||||
label = context.pctl.read_label(name, label)
|
||||
if label is None:
|
||||
label = ""
|
||||
label = label.replace('\n', '\\n')
|
||||
assert content in label, "\"{0}\" doesn't contain {1}".format(label, content)
|
||||
|
||||
|
||||
@step('I create label with "{content:w}" in {name:w} data directory')
|
||||
def write_label(context, content, name):
|
||||
context.pctl.write_label(name, content)
|
||||
|
||||
|
||||
@step('"{name}" key in DCS has {key:w}={value} after {time_limit:d} seconds')
|
||||
def check_member(context, name, key, value, time_limit):
|
||||
time_limit *= context.timeout_multiplier
|
||||
max_time = time.time() + int(time_limit)
|
||||
dcs_value = None
|
||||
while time.time() < max_time:
|
||||
try:
|
||||
response = json.loads(context.dcs_ctl.query(name))
|
||||
dcs_value = response.get(key)
|
||||
if dcs_value == value:
|
||||
return
|
||||
except Exception:
|
||||
pass
|
||||
time.sleep(1)
|
||||
assert False, "{0} does not have {1}={2} (found {3}) in dcs after {4} seconds".format(name, key, value,
|
||||
dcs_value, time_limit)
|
||||
|
||||
|
||||
@step('there is a non empty {key:w} key in DCS after {time_limit:d} seconds')
|
||||
def check_initialize(context, key, time_limit):
|
||||
time_limit *= context.timeout_multiplier
|
||||
max_time = time.time() + int(time_limit)
|
||||
while time.time() < max_time:
|
||||
try:
|
||||
if context.dcs_ctl.query(key):
|
||||
return
|
||||
except Exception:
|
||||
pass
|
||||
time.sleep(1)
|
||||
assert False, "There is no {0} in dcs after {1} seconds".format(key, time_limit)
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
import time
|
||||
|
||||
from behave import step, then
|
||||
|
||||
|
||||
@step('I start {name:w} in a cluster {cluster_name:w} as a clone of {name2:w}')
|
||||
def start_cluster_clone(context, name, cluster_name, name2):
|
||||
context.pctl.clone(name2, cluster_name, name)
|
||||
|
||||
|
||||
@step('I start {name:w} in a cluster {cluster_name:w} from backup')
|
||||
def start_cluster_from_backup(context, name, cluster_name):
|
||||
context.pctl.bootstrap_from_backup(name, cluster_name)
|
||||
|
||||
|
||||
@then('{name:w} is a leader of {cluster_name:w} after {time_limit:d} seconds')
|
||||
def is_a_leader(context, name, cluster_name, time_limit):
|
||||
time_limit *= context.timeout_multiplier
|
||||
max_time = time.time() + int(time_limit)
|
||||
while (context.dcs_ctl.query("leader", scope=cluster_name) != name):
|
||||
time.sleep(1)
|
||||
assert time.time() < max_time, "{0} is not a leader in dcs after {1} seconds".format(name, time_limit)
|
||||
|
||||
|
||||
@step('I do a backup of {name:w}')
|
||||
def do_backup(context, name):
|
||||
context.pctl.backup(name)
|
||||
@@ -1,14 +1,19 @@
|
||||
import json
|
||||
import os
|
||||
import parse
|
||||
import pytz
|
||||
import requests
|
||||
import shlex
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import yaml
|
||||
|
||||
from behave import register_type, step, then
|
||||
from dateutil import tz
|
||||
from datetime import datetime, timedelta
|
||||
from patroni.request import PatroniRequest
|
||||
|
||||
tzutc = tz.tzutc()
|
||||
request_executor = PatroniRequest({'ctl': {'auth': 'username:password'}})
|
||||
|
||||
|
||||
@parse.with_pattern(r'https?://(?:\w|\.|:|/)+')
|
||||
@@ -27,6 +32,7 @@ register_type(url=parse_url)
|
||||
@step('{name:w} is a leader after {time_limit:d} seconds')
|
||||
@then('{name:w} is a leader after {time_limit:d} seconds')
|
||||
def is_a_leader(context, name, time_limit):
|
||||
time_limit *= context.timeout_multiplier
|
||||
max_time = time.time() + int(time_limit)
|
||||
while (context.dcs_ctl.query("leader") != name):
|
||||
time.sleep(1)
|
||||
@@ -39,9 +45,9 @@ def sleep_for_n_seconds(context, value):
|
||||
|
||||
|
||||
def _set_response(context, response):
|
||||
context.status_code = response.status_code
|
||||
data = response.content.decode('utf-8')
|
||||
ct = response.headers.get('content-type', '')
|
||||
context.status_code = response.status
|
||||
data = response.data.decode('utf-8')
|
||||
ct = response.getheader('content-type', '')
|
||||
if ct.startswith('application/json') or\
|
||||
ct.startswith('text/yaml') or\
|
||||
ct.startswith('text/x-yaml') or\
|
||||
@@ -57,13 +63,7 @@ def _set_response(context, response):
|
||||
|
||||
@step('I issue a GET request to {url:url}')
|
||||
def do_get(context, url):
|
||||
try:
|
||||
r = requests.get(url)
|
||||
except requests.exceptions.RequestException:
|
||||
context.status_code = None
|
||||
context.response = None
|
||||
else:
|
||||
_set_response(context, r)
|
||||
do_request(context, 'GET', url, None)
|
||||
|
||||
|
||||
@step('I issue an empty POST request to {url:url}')
|
||||
@@ -73,24 +73,25 @@ def do_post_empty(context, url):
|
||||
|
||||
@step('I issue a {request_method:w} request to {url:url} with {data}')
|
||||
def do_request(context, request_method, url, data):
|
||||
data = data and json.loads(data) or {}
|
||||
data = data and json.loads(data)
|
||||
try:
|
||||
if request_method == 'PATCH':
|
||||
r = requests.patch(url, json=data)
|
||||
else:
|
||||
r = requests.post(url, json=data)
|
||||
except requests.exceptions.RequestException:
|
||||
context.status_code = None
|
||||
context.response = None
|
||||
r = request_executor.request(request_method, url, data)
|
||||
if request_method == 'PATCH' and r.status == 409:
|
||||
r = request_executor.request(request_method, url, data)
|
||||
except Exception:
|
||||
context.status_code = context.response = None
|
||||
else:
|
||||
_set_response(context, r)
|
||||
|
||||
|
||||
@step('I run {cmd}')
|
||||
def do_run(context, cmd):
|
||||
cmd = ['coverage', 'run', '--source=patroni', '-p'] + shlex.split(cmd)
|
||||
cmd = [sys.executable, '-m', 'coverage', 'run', '--source=patroni', '-p'] + shlex.split(cmd)
|
||||
try:
|
||||
response = subprocess.check_output(cmd, stderr=subprocess.STDOUT)
|
||||
# XXX: Dirty hack! We need to take name/passwd from the config!
|
||||
env = os.environ.copy()
|
||||
env.update({'PATRONI_RESTAPI_USERNAME': 'username', 'PATRONI_RESTAPI_PASSWORD': 'password'})
|
||||
response = subprocess.check_output(cmd, stderr=subprocess.STDOUT, env=env)
|
||||
context.status_code = 0
|
||||
except subprocess.CalledProcessError as e:
|
||||
response = e.output
|
||||
@@ -104,7 +105,8 @@ def check_response(context, component, data):
|
||||
assert context.status_code == int(data),\
|
||||
"status code {0} != {1}, response: {2}".format(context.status_code, data, context.response)
|
||||
elif component == 'returncode':
|
||||
assert context.status_code == int(data), "return code {0} != {1}".format(context.status_code, data)
|
||||
assert context.status_code == int(data), "return code {0} != {1}, {2}".format(context.status_code,
|
||||
data, context.response)
|
||||
elif component == 'text':
|
||||
assert context.response == data.strip('"'), "response {0} does not contain {1}".format(context.response, data)
|
||||
elif component == 'output':
|
||||
@@ -114,11 +116,18 @@ def check_response(context, component, data):
|
||||
assert str(context.response[component]) == str(data), "{0} does not contain {1}".format(component, data)
|
||||
|
||||
|
||||
@step('I issue a scheduled failover from {from_host:w} to {to_host:w} in {in_seconds:d} seconds')
|
||||
def scheduled_failover(context, from_host, to_host, in_seconds):
|
||||
@step('I issue a scheduled switchover from {from_host:w} to {to_host:w} in {in_seconds:d} seconds')
|
||||
def scheduled_switchover(context, from_host, to_host, in_seconds):
|
||||
context.execute_steps(u"""
|
||||
Given I run patronictl.py failover batman --master {0} --candidate {1} --scheduled "{2}" --force
|
||||
""".format(from_host, to_host, datetime.now(pytz.utc) + timedelta(seconds=int(in_seconds))))
|
||||
Given I run patronictl.py switchover batman --master {0} --candidate {1} --scheduled "{2}" --force
|
||||
""".format(from_host, to_host, datetime.now(tzutc) + timedelta(seconds=int(in_seconds))))
|
||||
|
||||
|
||||
@step('I issue a scheduled restart at {url:url} in {in_seconds:d} seconds with {data}')
|
||||
def scheduled_restart(context, url, in_seconds, data):
|
||||
data = data and json.loads(data) or {}
|
||||
data.update(schedule='{0}'.format((datetime.now(tzutc) + timedelta(seconds=int(in_seconds))).isoformat()))
|
||||
context.execute_steps(u"""Given I issue a POST request to {0}/restart with {1}""".format(url, json.dumps(data)))
|
||||
|
||||
|
||||
@step('I add tag {tag:w} {value:w} to {pg_name:w} config')
|
||||
@@ -127,12 +136,18 @@ def add_tag_to_config(context, tag, value, pg_name):
|
||||
|
||||
|
||||
@then('Response on GET {url} contains {value} after {timeout:d} seconds')
|
||||
def check_http_response(context, url, value, timeout):
|
||||
def check_http_response(context, url, value, timeout, negate=False):
|
||||
timeout *= context.timeout_multiplier
|
||||
for _ in range(int(timeout)):
|
||||
r = requests.get(url)
|
||||
if value in r.content.decode('utf-8'):
|
||||
r = request_executor.request('GET', url)
|
||||
if (value in r.data.decode('utf-8')) != negate:
|
||||
break
|
||||
time.sleep(1)
|
||||
else:
|
||||
assert False,\
|
||||
"Value {0} is not present in response after {1} seconds".format(value, timeout)
|
||||
"Value {0} is {1} present in response after {2} seconds".format(value, "not" if not negate else "", timeout)
|
||||
|
||||
|
||||
@then('Response on GET {url} does not contain {value} after {timeout:d} seconds')
|
||||
def check_not_in_http_response(context, url, value, timeout):
|
||||
check_http_response(context, url, value, timeout, negate=True)
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
import time
|
||||
|
||||
from behave import step, then
|
||||
import patroni.psycopg as pg
|
||||
|
||||
|
||||
@step('I create a logical replication slot {slot_name} on {pg_name:w} with the {plugin:w} plugin')
|
||||
def create_logical_replication_slot(context, slot_name, pg_name, plugin):
|
||||
try:
|
||||
output = context.pctl.query(pg_name, ("SELECT pg_create_logical_replication_slot('{0}', '{1}'),"
|
||||
" current_database()").format(slot_name, plugin))
|
||||
print(output.fetchone())
|
||||
except pg.Error as e:
|
||||
print(e)
|
||||
assert False, "Error creating slot {0} on {1} with plugin {2}".format(slot_name, pg_name, plugin)
|
||||
|
||||
|
||||
@then('{pg_name:w} has a logical replication slot named {slot_name} with the {plugin:w} plugin')
|
||||
def has_logical_replication_slot(context, pg_name, slot_name, plugin):
|
||||
try:
|
||||
row = context.pctl.query(pg_name, ("SELECT slot_type, plugin FROM pg_replication_slots"
|
||||
" WHERE slot_name = '{0}'").format(slot_name)).fetchone()
|
||||
assert row, "Couldn't find replication slot named {0}".format(slot_name)
|
||||
assert row[0] == "logical", "Found replication slot named {0} but wasn't a logical slot".format(slot_name)
|
||||
assert row[1] == plugin, ("Found replication slot named {0} but was using plugin "
|
||||
"{1} rather than {2}").format(slot_name, row[1], plugin)
|
||||
except pg.Error:
|
||||
assert False, "Error looking for slot {0} on {1} with plugin {2}".format(slot_name, pg_name, plugin)
|
||||
|
||||
|
||||
@then('{pg_name:w} does not have a logical replication slot named {slot_name}')
|
||||
def does_not_have_logical_replication_slot(context, pg_name, slot_name):
|
||||
try:
|
||||
row = context.pctl.query(pg_name, ("SELECT 1 FROM pg_replication_slots"
|
||||
" WHERE slot_name = '{0}'").format(slot_name)).fetchone()
|
||||
assert not row, "Found unexpected replication slot named {0}".format(slot_name)
|
||||
except pg.Error:
|
||||
assert False, "Error looking for slot {0} on {1}".format(slot_name, pg_name)
|
||||
|
||||
|
||||
@step('Logical slot {slot_name:w} is in sync between {pg_name1:w} and {pg_name2:w} after {time_limit:d} seconds')
|
||||
def logical_slots_in_sync(context, slot_name, pg_name1, pg_name2, time_limit):
|
||||
time_limit *= context.timeout_multiplier
|
||||
max_time = time.time() + int(time_limit)
|
||||
while time.time() < max_time:
|
||||
try:
|
||||
query = "SELECT confirmed_flush_lsn FROM pg_replication_slots WHERE slot_name = '{0}'".format(slot_name)
|
||||
slot1 = context.pctl.query(pg_name1, query).fetchone()
|
||||
slot2 = context.pctl.query(pg_name2, query).fetchone()
|
||||
if slot1[0] == slot2[0]:
|
||||
return
|
||||
except Exception:
|
||||
pass
|
||||
time.sleep(1)
|
||||
assert False, "Logical slot {0} is not in sync between {1} and {2}".format(slot_name, pg_name1, pg_name2)
|
||||
|
||||
|
||||
@step('I get all changes from logical slot {slot_name:w} on {pg_name:w}')
|
||||
def logical_slot_get_changes(context, slot_name, pg_name):
|
||||
context.pctl.query(pg_name, "SELECT * FROM pg_logical_slot_get_changes('{0}', NULL, NULL)".format(slot_name))
|
||||
@@ -0,0 +1,74 @@
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
|
||||
from behave import step
|
||||
|
||||
|
||||
select_replication_query = """
|
||||
SELECT * FROM pg_catalog.pg_stat_replication
|
||||
WHERE application_name = '{0}'
|
||||
"""
|
||||
|
||||
executable = sys.executable if os.name != 'nt' else sys.executable.replace('\\', '/')
|
||||
callback = executable + " features/callback2.py "
|
||||
|
||||
|
||||
@step('I start {name:w} in a cluster {cluster_name:w}')
|
||||
def start_patroni(context, name, cluster_name):
|
||||
return context.pctl.start(name, custom_config={
|
||||
"scope": cluster_name,
|
||||
"postgresql": {
|
||||
"callbacks": {c: callback + name for c in ('on_start', 'on_stop', 'on_restart', 'on_role_change')},
|
||||
"backup_restore": {
|
||||
"command": (executable + " features/backup_restore.py --sourcedir=" +
|
||||
os.path.join(context.pctl.patroni_path, 'data', 'basebackup'))}
|
||||
}
|
||||
})
|
||||
|
||||
|
||||
@step('I start {name:w} in a standby cluster {cluster_name:w} as a clone of {name2:w}')
|
||||
def start_patroni_standby_cluster(context, name, cluster_name, name2):
|
||||
# we need to remove patroni.dynamic.json in order to "bootstrap" standby cluster with existing PGDATA
|
||||
os.unlink(os.path.join(context.pctl._processes[name]._data_dir, 'patroni.dynamic.json'))
|
||||
port = context.pctl._processes[name2]._connkwargs.get('port')
|
||||
context.pctl._processes[name].update_config({
|
||||
"scope": cluster_name,
|
||||
"bootstrap": {
|
||||
"dcs": {
|
||||
"ttl": 20,
|
||||
"loop_wait": 2,
|
||||
"retry_timeout": 5,
|
||||
"standby_cluster": {
|
||||
"host": "localhost",
|
||||
"port": port,
|
||||
"primary_slot_name": "pm_1",
|
||||
"create_replica_methods": ["backup_restore", "basebackup"]
|
||||
},
|
||||
"postgresql": {"parameters": {"wal_level": "logical"}}
|
||||
}
|
||||
},
|
||||
"postgresql": {
|
||||
"callbacks": {c: callback + name for c in ('on_start', 'on_stop', 'on_restart', 'on_role_change')}
|
||||
}
|
||||
})
|
||||
return context.pctl.start(name)
|
||||
|
||||
|
||||
@step('{pg_name1:w} is replicating from {pg_name2:w} after {timeout:d} seconds')
|
||||
def check_replication_status(context, pg_name1, pg_name2, timeout):
|
||||
bound_time = time.time() + timeout * context.timeout_multiplier
|
||||
|
||||
while time.time() < bound_time:
|
||||
cur = context.pctl.query(
|
||||
pg_name2,
|
||||
select_replication_query.format(pg_name1),
|
||||
fail_ok=True
|
||||
)
|
||||
|
||||
if cur and len(cur.fetchall()) != 0:
|
||||
break
|
||||
|
||||
time.sleep(1)
|
||||
else:
|
||||
assert False, "{0} is not replicating from {1} after {2} seconds".format(pg_name1, pg_name2, timeout)
|
||||
@@ -0,0 +1,49 @@
|
||||
from behave import step, then
|
||||
import time
|
||||
|
||||
|
||||
def polling_loop(timeout, interval=1):
|
||||
"""Returns an iterator that returns values until timeout has passed. Timeout is measured from start of iteration."""
|
||||
start_time = time.time()
|
||||
iteration = 0
|
||||
end_time = start_time + timeout
|
||||
while time.time() < end_time:
|
||||
yield iteration
|
||||
iteration += 1
|
||||
time.sleep(interval)
|
||||
|
||||
|
||||
@step('I start {name:w} with watchdog')
|
||||
def start_patroni_with_watchdog(context, name):
|
||||
return context.pctl.start(name, custom_config={'watchdog': True})
|
||||
|
||||
|
||||
@step('{name:w} watchdog has been pinged after {timeout:d} seconds')
|
||||
def watchdog_was_pinged(context, name, timeout):
|
||||
for _ in polling_loop(timeout):
|
||||
if context.pctl.get_watchdog(name).was_pinged:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
@then('{name:w} watchdog has been closed')
|
||||
def watchdog_was_closed(context, name):
|
||||
assert context.pctl.get_watchdog(name).was_closed
|
||||
|
||||
|
||||
@step('I reset {name:w} watchdog state')
|
||||
def watchdog_reset_pinged(context, name):
|
||||
context.pctl.get_watchdog(name).reset()
|
||||
|
||||
|
||||
@then('{name:w} watchdog is triggered after {timeout:d} seconds')
|
||||
def watchdog_was_triggered(context, name, timeout):
|
||||
for _ in polling_loop(timeout):
|
||||
if context.pctl.get_watchdog(name).was_triggered:
|
||||
return True
|
||||
assert False
|
||||
|
||||
|
||||
@step('{name:w} hangs for {timeout:d} seconds')
|
||||
def patroni_hang(context, name, timeout):
|
||||
return context.pctl.patroni_hang(name, timeout)
|
||||
@@ -0,0 +1,31 @@
|
||||
Feature: watchdog
|
||||
Verify that watchdog gets pinged and triggered under appropriate circumstances.
|
||||
|
||||
Scenario: watchdog is opened and pinged
|
||||
Given I start postgres0 with watchdog
|
||||
Then postgres0 is a leader after 10 seconds
|
||||
And postgres0 role is the primary after 10 seconds
|
||||
And postgres0 watchdog has been pinged after 10 seconds
|
||||
|
||||
Scenario: watchdog is disabled during pause
|
||||
Given I run patronictl.py pause batman
|
||||
Then I receive a response returncode 0
|
||||
When I sleep for 2 seconds
|
||||
Then postgres0 watchdog has been closed
|
||||
|
||||
Scenario: watchdog is opened and pinged after resume
|
||||
Given I reset postgres0 watchdog state
|
||||
And I run patronictl.py resume batman
|
||||
Then I receive a response returncode 0
|
||||
And postgres0 watchdog has been pinged after 10 seconds
|
||||
|
||||
Scenario: watchdog is disabled when shutting down
|
||||
Given I shut down postgres0
|
||||
Then postgres0 watchdog has been closed
|
||||
|
||||
Scenario: watchdog is triggered if patroni stops responding
|
||||
Given I reset postgres0 watchdog state
|
||||
And I start postgres0 with watchdog
|
||||
Then postgres0 role is the primary after 10 seconds
|
||||
When postgres0 hangs for 30 seconds
|
||||
Then postgres0 watchdog is triggered after 30 seconds
|
||||
+18
-14
@@ -1,21 +1,25 @@
|
||||
global
|
||||
maxconn 100
|
||||
maxconn 100
|
||||
|
||||
defaults
|
||||
log global
|
||||
mode tcp
|
||||
retries 2
|
||||
timeout client 30m
|
||||
timeout connect 4s
|
||||
timeout server 30m
|
||||
timeout check 5s
|
||||
log global
|
||||
mode tcp
|
||||
retries 2
|
||||
timeout client 30m
|
||||
timeout connect 4s
|
||||
timeout server 30m
|
||||
timeout check 5s
|
||||
|
||||
frontend ft_postgresql
|
||||
bind *:5000
|
||||
default_backend bk_db
|
||||
|
||||
backend bk_db
|
||||
option httpchk
|
||||
listen stats
|
||||
mode http
|
||||
bind *:7000
|
||||
stats enable
|
||||
stats uri /
|
||||
|
||||
listen batman
|
||||
bind *:5000
|
||||
option httpchk
|
||||
http-check expect status 200
|
||||
default-server inter 3s fall 3 rise 2 on-marked-down shutdown-sessions
|
||||
server postgresql_127.0.0.1_5432 127.0.0.1:5432 maxconn 100 check port 8008
|
||||
server postgresql_127.0.0.1_5433 127.0.0.1:5433 maxconn 100 check port 8009
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
FROM postgres:11
|
||||
MAINTAINER Alexander Kukushkin <[email protected]>
|
||||
|
||||
RUN export DEBIAN_FRONTEND=noninteractive \
|
||||
&& echo 'APT::Install-Recommends "0";\nAPT::Install-Suggests "0";' > /etc/apt/apt.conf.d/01norecommend \
|
||||
&& apt-get update -y \
|
||||
&& apt-get upgrade -y \
|
||||
&& apt-cache depends patroni | sed -n -e 's/.* Depends: \(python3-.\+\)$/\1/p' \
|
||||
| grep -Ev '^python3-(sphinx|etcd|consul|kazoo|kubernetes)' \
|
||||
| xargs apt-get install -y vim-tiny curl jq locales git python3-pip python3-wheel \
|
||||
## Make sure we have a en_US.UTF-8 locale available
|
||||
&& localedef -i en_US -c -f UTF-8 -A /usr/share/locale/locale.alias en_US.UTF-8 \
|
||||
&& pip3 install setuptools \
|
||||
&& pip3 install 'git+https://github.com/zalando/patroni.git#egg=patroni[kubernetes]' \
|
||||
&& PGHOME=/home/postgres \
|
||||
&& mkdir -p $PGHOME \
|
||||
&& chown postgres $PGHOME \
|
||||
&& sed -i "s|/var/lib/postgresql.*|$PGHOME:/bin/bash|" /etc/passwd \
|
||||
# Set permissions for OpenShift
|
||||
&& chmod 775 $PGHOME \
|
||||
&& chmod 664 /etc/passwd \
|
||||
# Clean up
|
||||
&& apt-get remove -y git python3-pip python3-wheel \
|
||||
&& apt-get autoremove -y \
|
||||
&& apt-get clean -y \
|
||||
&& rm -rf /var/lib/apt/lists/* /root/.cache
|
||||
|
||||
ADD entrypoint.sh /
|
||||
|
||||
EXPOSE 5432 8008
|
||||
ENV LC_ALL=en_US.UTF-8 LANG=en_US.UTF-8 EDITOR=/usr/bin/editor
|
||||
USER postgres
|
||||
WORKDIR /home/postgres
|
||||
CMD ["/bin/bash", "/entrypoint.sh"]
|
||||
Executable
+37
@@ -0,0 +1,37 @@
|
||||
#!/bin/bash
|
||||
|
||||
if [[ $UID -ge 10000 ]]; then
|
||||
GID=$(id -g)
|
||||
sed -e "s/^postgres:x:[^:]*:[^:]*:/postgres:x:$UID:$GID:/" /etc/passwd > /tmp/passwd
|
||||
cat /tmp/passwd > /etc/passwd
|
||||
rm /tmp/passwd
|
||||
fi
|
||||
|
||||
cat > /home/postgres/patroni.yml <<__EOF__
|
||||
bootstrap:
|
||||
dcs:
|
||||
postgresql:
|
||||
use_pg_rewind: true
|
||||
initdb:
|
||||
- auth-host: md5
|
||||
- auth-local: trust
|
||||
- encoding: UTF8
|
||||
- locale: en_US.UTF-8
|
||||
- data-checksums
|
||||
pg_hba:
|
||||
- host all all 0.0.0.0/0 md5
|
||||
- host replication ${PATRONI_REPLICATION_USERNAME} ${PATRONI_KUBERNETES_POD_IP}/16 md5
|
||||
restapi:
|
||||
connect_address: '${PATRONI_KUBERNETES_POD_IP}:8008'
|
||||
postgresql:
|
||||
connect_address: '${PATRONI_KUBERNETES_POD_IP}:5432'
|
||||
authentication:
|
||||
superuser:
|
||||
password: '${PATRONI_SUPERUSER_PASSWORD}'
|
||||
replication:
|
||||
password: '${PATRONI_REPLICATION_PASSWORD}'
|
||||
__EOF__
|
||||
|
||||
unset PATRONI_SUPERUSER_PASSWORD PATRONI_REPLICATION_PASSWORD
|
||||
|
||||
exec /usr/bin/python3 /usr/local/bin/patroni /home/postgres/patroni.yml
|
||||
@@ -0,0 +1,49 @@
|
||||
# Patroni OpenShift Configuration
|
||||
Patroni can be run in OpenShift. Based on the kubernetes configuration, the Dockerfile and Entrypoint has been modified to support the dynamic UID/GID configuration that is applied in OpenShift. This can be run under the standard `restricted` SCC.
|
||||
|
||||
# Examples
|
||||
|
||||
## Create test project
|
||||
|
||||
```
|
||||
oc new-project patroni-test
|
||||
```
|
||||
|
||||
## Build the image
|
||||
|
||||
Note: Update the references when merged upstream.
|
||||
Note: If deploying as a template for multiple users, the following commands should be performed in a shared namespace like `openshift`.
|
||||
|
||||
```
|
||||
oc import-image postgres:10 --confirm -n openshift
|
||||
oc new-build https://github.com/zalando/patroni --context-dir=kubernetes -n openshift
|
||||
```
|
||||
|
||||
## Deploy the Image
|
||||
Two configuration templates exist in [templates](templates) directory:
|
||||
- Patroni Ephemeral
|
||||
- Patroni Persistent
|
||||
|
||||
The only difference is whether or not the statefulset requests persistent storage.
|
||||
|
||||
## Create the Template
|
||||
Install the template into the `openshift` namespace if this should be shared across projects:
|
||||
|
||||
```
|
||||
oc create -f templates/template_patroni_ephemeral.yml -n openshift
|
||||
```
|
||||
|
||||
Then, from your own project:
|
||||
|
||||
```
|
||||
oc new-app patroni-pgsql-ephemeral
|
||||
```
|
||||
|
||||
Once the pods are running, two configmaps should be available:
|
||||
|
||||
```
|
||||
$ oc get configmap
|
||||
NAME DATA AGE
|
||||
patroniocp-config 0 1m
|
||||
patroniocp-leader 0 1m
|
||||
```
|
||||
@@ -0,0 +1,327 @@
|
||||
apiVersion: v1
|
||||
kind: Template
|
||||
metadata:
|
||||
name: patroni-pgsql-ephemeral
|
||||
annotations:
|
||||
description: |-
|
||||
Patroni Postgresql database cluster, without persistent storage.
|
||||
|
||||
WARNING: Any data stored will be lost upon pod destruction. Only use this template for testing.
|
||||
iconClass: icon-postgresql
|
||||
openshift.io/display-name: Patroni Postgresql (Ephemeral)
|
||||
openshift.io/long-description: This template deploys a a patroni postgresql HA cluster without persistent storage.
|
||||
tags: postgresql
|
||||
objects:
|
||||
- apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
name: ${PATRONI_CLUSTER_NAME}
|
||||
spec:
|
||||
ports:
|
||||
- port: 5432
|
||||
protocol: TCP
|
||||
targetPort: 5432
|
||||
sessionAffinity: None
|
||||
type: ClusterIP
|
||||
status:
|
||||
loadBalancer: {}
|
||||
- apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
name: ${PATRONI_MASTER_SERVICE_NAME}
|
||||
spec:
|
||||
ports:
|
||||
- port: 5432
|
||||
protocol: TCP
|
||||
targetPort: 5432
|
||||
selector:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
role: master
|
||||
sessionAffinity: None
|
||||
type: ClusterIP
|
||||
status:
|
||||
loadBalancer: {}
|
||||
- apiVersion: v1
|
||||
kind: Secret
|
||||
metadata:
|
||||
name: ${PATRONI_CLUSTER_NAME}
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
stringData:
|
||||
superuser-password: ${PATRONI_SUPERUSER_PASSWORD}
|
||||
replication-password: ${PATRONI_REPLICATION_PASSWORD}
|
||||
- apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
name: ${PATRONI_REPLICA_SERVICE_NAME}
|
||||
spec:
|
||||
ports:
|
||||
- port: 5432
|
||||
protocol: TCP
|
||||
targetPort: 5432
|
||||
selector:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
role: replica
|
||||
sessionAffinity: None
|
||||
type: ClusterIP
|
||||
status:
|
||||
loadBalancer: {}
|
||||
- apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
generation: 3
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
name: ${APPLICATION_NAME}
|
||||
spec:
|
||||
podManagementPolicy: OrderedReady
|
||||
replicas: 3
|
||||
revisionHistoryLimit: 10
|
||||
selector:
|
||||
matchLabels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
serviceName: ${APPLICATION_NAME}
|
||||
template:
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
spec:
|
||||
containers:
|
||||
- env:
|
||||
- name: PATRONI_KUBERNETES_POD_IP
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
apiVersion: v1
|
||||
fieldPath: status.podIP
|
||||
- name: PATRONI_KUBERNETES_NAMESPACE
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
apiVersion: v1
|
||||
fieldPath: metadata.namespace
|
||||
- name: PATRONI_KUBERNETES_BYPASS_API_SERVICE
|
||||
value: 'true'
|
||||
- name: PATRONI_KUBERNETES_LABELS
|
||||
value: '{application: ${APPLICATION_NAME}, cluster-name: ${PATRONI_CLUSTER_NAME}}'
|
||||
- name: PATRONI_SUPERUSER_USERNAME
|
||||
value: ${PATRONI_SUPERUSER_USERNAME}
|
||||
- name: PATRONI_SUPERUSER_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
key: superuser-password
|
||||
name: ${PATRONI_CLUSTER_NAME}
|
||||
- name: PATRONI_REPLICATION_USERNAME
|
||||
value: ${PATRONI_REPLICATION_USERNAME}
|
||||
- name: PATRONI_REPLICATION_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
key: replication-password
|
||||
name: ${PATRONI_CLUSTER_NAME}
|
||||
- name: PATRONI_SCOPE
|
||||
value: ${PATRONI_CLUSTER_NAME}
|
||||
- name: PATRONI_NAME
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
apiVersion: v1
|
||||
fieldPath: metadata.name
|
||||
- name: PATRONI_POSTGRESQL_DATA_DIR
|
||||
value: /home/postgres/pgdata/pgroot/data
|
||||
- name: PATRONI_POSTGRESQL_PGPASS
|
||||
value: /tmp/pgpass
|
||||
- name: PATRONI_POSTGRESQL_LISTEN
|
||||
value: 0.0.0.0:5432
|
||||
- name: PATRONI_RESTAPI_LISTEN
|
||||
value: 0.0.0.0:8008
|
||||
image: docker-registry.default.svc:5000/${NAMESPACE}/patroni:latest
|
||||
imagePullPolicy: IfNotPresent
|
||||
name: ${APPLICATION_NAME}
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
scheme: HTTP
|
||||
path: /readiness
|
||||
port: 8008
|
||||
initialDelaySeconds: 3
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
successThreshold: 1
|
||||
failureThreshold: 3
|
||||
ports:
|
||||
- containerPort: 8008
|
||||
protocol: TCP
|
||||
- containerPort: 5432
|
||||
protocol: TCP
|
||||
resources: {}
|
||||
terminationMessagePath: /dev/termination-log
|
||||
terminationMessagePolicy: File
|
||||
volumeMounts:
|
||||
- mountPath: /home/postgres/pgdata
|
||||
name: pgdata
|
||||
dnsPolicy: ClusterFirst
|
||||
restartPolicy: Always
|
||||
schedulerName: default-scheduler
|
||||
securityContext: {}
|
||||
serviceAccount: ${SERVICE_ACCOUNT}
|
||||
serviceAccountName: ${SERVICE_ACCOUNT}
|
||||
terminationGracePeriodSeconds: 0
|
||||
volumes:
|
||||
- name: pgdata
|
||||
emptyDir: {}
|
||||
updateStrategy:
|
||||
type: OnDelete
|
||||
- apiVersion: v1
|
||||
kind: Endpoints
|
||||
metadata:
|
||||
name: ${APPLICATION_NAME}
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
subsets: []
|
||||
- apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
- apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: Role
|
||||
metadata:
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- configmaps
|
||||
verbs:
|
||||
- create
|
||||
- get
|
||||
- list
|
||||
- patch
|
||||
- update
|
||||
- watch
|
||||
# delete is required only for 'patronictl remove'
|
||||
- delete
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- endpoints
|
||||
verbs:
|
||||
- get
|
||||
- patch
|
||||
- update
|
||||
# the following three privileges are necessary only when using endpoints
|
||||
- create
|
||||
- list
|
||||
- watch
|
||||
# delete is required only for for 'patronictl remove'
|
||||
- delete
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- pods
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- patch
|
||||
- update
|
||||
- watch
|
||||
- apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: RoleBinding
|
||||
metadata:
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: Role
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
# Following privileges are only required if deployed not in the "default"
|
||||
# namespace and you want Patroni to bypass kubernetes service
|
||||
# (PATRONI_KUBERNETES_BYPASS_API_SERVICE=true)
|
||||
- apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRole
|
||||
metadata:
|
||||
name: patroni-k8s-ep-access
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- endpoints
|
||||
resourceNames:
|
||||
- kubernetes
|
||||
verbs:
|
||||
- get
|
||||
- apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRoleBinding
|
||||
metadata:
|
||||
name: ${NAMESPACE}-${SERVICE_ACCOUNT}-k8s-ep-access
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: ClusterRole
|
||||
name: patroni-k8s-ep-access
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
namespace: ${NAMESPACE}
|
||||
parameters:
|
||||
- description: The name of the application for labelling all artifacts.
|
||||
displayName: Application Name
|
||||
name: APPLICATION_NAME
|
||||
value: patroni-ephemeral
|
||||
- description: The name of the patroni-pgsql cluster.
|
||||
displayName: Cluster Name
|
||||
name: PATRONI_CLUSTER_NAME
|
||||
value: patroni-ephemeral
|
||||
- description: The name of the OpenShift Service exposed for the patroni-ephemeral-master container.
|
||||
displayName: Master service name.
|
||||
name: PATRONI_MASTER_SERVICE_NAME
|
||||
value: patroni-ephemeral-master
|
||||
- description: The name of the OpenShift Service exposed for the patroni-ephemeral-replica containers.
|
||||
displayName: Replica service name.
|
||||
name: PATRONI_REPLICA_SERVICE_NAME
|
||||
value: patroni-ephemeral-replica
|
||||
- description: Maximum amount of memory the container can use.
|
||||
displayName: Memory Limit
|
||||
name: MEMORY_LIMIT
|
||||
value: 512Mi
|
||||
- description: The OpenShift Namespace where the patroni and postgresql ImageStream resides.
|
||||
displayName: ImageStream Namespace
|
||||
name: NAMESPACE
|
||||
value: openshift
|
||||
- description: Username of the superuser account for initialization.
|
||||
displayName: Superuser Username
|
||||
name: PATRONI_SUPERUSER_USERNAME
|
||||
value: postgres
|
||||
- description: Password of the superuser account for initialization.
|
||||
displayName: Superuser Passsword
|
||||
name: PATRONI_SUPERUSER_PASSWORD
|
||||
value: postgres
|
||||
- description: Username of the replication account for initialization.
|
||||
displayName: Replication Username
|
||||
name: PATRONI_REPLICATION_USERNAME
|
||||
value: postgres
|
||||
- description: Password of the replication account for initialization.
|
||||
displayName: Repication Passsword
|
||||
name: PATRONI_REPLICATION_PASSWORD
|
||||
value: postgres
|
||||
- description: Service account name used for pods and rolebindings to form a cluster in the project.
|
||||
displayName: Service Account
|
||||
name: SERVICE_ACCOUNT
|
||||
value: patroniocp
|
||||
@@ -0,0 +1,355 @@
|
||||
apiVersion: v1
|
||||
kind: Template
|
||||
metadata:
|
||||
name: patroni-pgsql-persistent
|
||||
annotations:
|
||||
description: |-
|
||||
Patroni Postgresql database cluster, with persistent storage.
|
||||
iconClass: icon-postgresql
|
||||
openshift.io/display-name: Patroni Postgresql (Persistent)
|
||||
openshift.io/long-description: This template deploys a a patroni postgresql HA cluster with persistent storage.
|
||||
tags: postgresql
|
||||
objects:
|
||||
- apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
name: ${PATRONI_CLUSTER_NAME}
|
||||
spec:
|
||||
ports:
|
||||
- port: 5432
|
||||
protocol: TCP
|
||||
targetPort: 5432
|
||||
sessionAffinity: None
|
||||
type: ClusterIP
|
||||
status:
|
||||
loadBalancer: {}
|
||||
- apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
name: ${PATRONI_MASTER_SERVICE_NAME}
|
||||
spec:
|
||||
ports:
|
||||
- port: 5432
|
||||
protocol: TCP
|
||||
targetPort: 5432
|
||||
selector:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
role: master
|
||||
sessionAffinity: None
|
||||
type: ClusterIP
|
||||
status:
|
||||
loadBalancer: {}
|
||||
- apiVersion: v1
|
||||
kind: Secret
|
||||
metadata:
|
||||
name: ${PATRONI_CLUSTER_NAME}
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
stringData:
|
||||
superuser-password: ${PATRONI_SUPERUSER_PASSWORD}
|
||||
replication-password: ${PATRONI_REPLICATION_PASSWORD}
|
||||
- apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
name: ${PATRONI_REPLICA_SERVICE_NAME}
|
||||
spec:
|
||||
ports:
|
||||
- port: 5432
|
||||
protocol: TCP
|
||||
targetPort: 5432
|
||||
selector:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
role: replica
|
||||
sessionAffinity: None
|
||||
type: ClusterIP
|
||||
status:
|
||||
loadBalancer: {}
|
||||
- apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
generation: 3
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
name: ${APPLICATION_NAME}
|
||||
spec:
|
||||
podManagementPolicy: OrderedReady
|
||||
replicas: 3
|
||||
revisionHistoryLimit: 10
|
||||
selector:
|
||||
matchLabels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
serviceName: ${APPLICATION_NAME}
|
||||
template:
|
||||
metadata:
|
||||
creationTimestamp: null
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
spec:
|
||||
initContainers:
|
||||
- command:
|
||||
- sh
|
||||
- -c
|
||||
- "mkdir -p /home/postgres/pgdata/pgroot/data && chmod 0700 /home/postgres/pgdata/pgroot/data"
|
||||
image: docker-registry.default.svc:5000/${NAMESPACE}/patroni:latest
|
||||
imagePullPolicy: IfNotPresent
|
||||
name: fix-perms
|
||||
resources: {}
|
||||
terminationMessagePath: /dev/termination-log
|
||||
terminationMessagePolicy: File
|
||||
volumeMounts:
|
||||
- mountPath: /home/postgres/pgdata
|
||||
name: ${APPLICATION_NAME}
|
||||
containers:
|
||||
- env:
|
||||
- name: PATRONI_KUBERNETES_POD_IP
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
apiVersion: v1
|
||||
fieldPath: status.podIP
|
||||
- name: PATRONI_KUBERNETES_NAMESPACE
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
apiVersion: v1
|
||||
fieldPath: metadata.namespace
|
||||
- name: PATRONI_KUBERNETES_BYPASS_API_SERVICE
|
||||
value: 'true'
|
||||
- name: PATRONI_KUBERNETES_LABELS
|
||||
value: '{application: ${APPLICATION_NAME}, cluster-name: ${PATRONI_CLUSTER_NAME}}'
|
||||
- name: PATRONI_SUPERUSER_USERNAME
|
||||
value: ${PATRONI_SUPERUSER_USERNAME}
|
||||
- name: PATRONI_SUPERUSER_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
key: superuser-password
|
||||
name: ${PATRONI_CLUSTER_NAME}
|
||||
- name: PATRONI_REPLICATION_USERNAME
|
||||
value: ${PATRONI_REPLICATION_USERNAME}
|
||||
- name: PATRONI_REPLICATION_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
key: replication-password
|
||||
name: ${PATRONI_CLUSTER_NAME}
|
||||
- name: PATRONI_SCOPE
|
||||
value: ${PATRONI_CLUSTER_NAME}
|
||||
- name: PATRONI_NAME
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
apiVersion: v1
|
||||
fieldPath: metadata.name
|
||||
- name: PATRONI_POSTGRESQL_DATA_DIR
|
||||
value: /home/postgres/pgdata/pgroot/data
|
||||
- name: PATRONI_POSTGRESQL_PGPASS
|
||||
value: /tmp/pgpass
|
||||
- name: PATRONI_POSTGRESQL_LISTEN
|
||||
value: 0.0.0.0:5432
|
||||
- name: PATRONI_RESTAPI_LISTEN
|
||||
value: 0.0.0.0:8008
|
||||
image: docker-registry.default.svc:5000/${NAMESPACE}/patroni:latest
|
||||
imagePullPolicy: IfNotPresent
|
||||
name: ${APPLICATION_NAME}
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
scheme: HTTP
|
||||
path: /readiness
|
||||
port: 8008
|
||||
initialDelaySeconds: 3
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
successThreshold: 1
|
||||
failureThreshold: 3
|
||||
ports:
|
||||
- containerPort: 8008
|
||||
protocol: TCP
|
||||
- containerPort: 5432
|
||||
protocol: TCP
|
||||
resources: {}
|
||||
terminationMessagePath: /dev/termination-log
|
||||
terminationMessagePolicy: File
|
||||
volumeMounts:
|
||||
- mountPath: /home/postgres/pgdata
|
||||
name: ${APPLICATION_NAME}
|
||||
dnsPolicy: ClusterFirst
|
||||
restartPolicy: Always
|
||||
schedulerName: default-scheduler
|
||||
securityContext: {}
|
||||
serviceAccount: ${SERVICE_ACCOUNT}
|
||||
serviceAccountName: ${SERVICE_ACCOUNT}
|
||||
terminationGracePeriodSeconds: 0
|
||||
volumes:
|
||||
- name: ${APPLICATION_NAME}
|
||||
persistentVolumeClaim:
|
||||
claimName: ${APPLICATION_NAME}
|
||||
volumeClaimTemplates:
|
||||
- metadata:
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
name: ${APPLICATION_NAME}
|
||||
spec:
|
||||
accessModes:
|
||||
- ReadWriteOnce
|
||||
resources:
|
||||
requests:
|
||||
storage: ${PVC_SIZE}
|
||||
updateStrategy:
|
||||
type: OnDelete
|
||||
- apiVersion: v1
|
||||
kind: Endpoints
|
||||
metadata:
|
||||
name: ${APPLICATION_NAME}
|
||||
labels:
|
||||
application: ${APPLICATION_NAME}
|
||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||
subsets: []
|
||||
- apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
- apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: Role
|
||||
metadata:
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- configmaps
|
||||
verbs:
|
||||
- create
|
||||
- get
|
||||
- list
|
||||
- patch
|
||||
- update
|
||||
- watch
|
||||
# delete is required only for 'patronictl remove'
|
||||
- delete
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- endpoints
|
||||
verbs:
|
||||
- get
|
||||
- patch
|
||||
- update
|
||||
# the following three privileges are necessary only when using endpoints
|
||||
- create
|
||||
- list
|
||||
- watch
|
||||
# delete is required only for for 'patronictl remove'
|
||||
- delete
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- pods
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- patch
|
||||
- update
|
||||
- watch
|
||||
- apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: RoleBinding
|
||||
metadata:
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: Role
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
# Following privileges are only required if deployed not in the "default"
|
||||
# namespace and you want Patroni to bypass kubernetes service
|
||||
# (PATRONI_KUBERNETES_BYPASS_API_SERVICE=true)
|
||||
- apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRole
|
||||
metadata:
|
||||
name: patroni-k8s-ep-access
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- endpoints
|
||||
resourceNames:
|
||||
- kubernetes
|
||||
verbs:
|
||||
- get
|
||||
- apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRoleBinding
|
||||
metadata:
|
||||
name: ${NAMESPACE}-${SERVICE_ACCOUNT}-k8s-ep-access
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: ClusterRole
|
||||
name: patroni-k8s-ep-access
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: ${SERVICE_ACCOUNT}
|
||||
namespace: ${NAMESPACE}
|
||||
parameters:
|
||||
- description: The name of the application for labelling all artifacts.
|
||||
displayName: Application Name
|
||||
name: APPLICATION_NAME
|
||||
value: patroni-persistent
|
||||
- description: The name of the patroni-pgsql cluster.
|
||||
displayName: Cluster Name
|
||||
name: PATRONI_CLUSTER_NAME
|
||||
value: patroni-persistent
|
||||
- description: The name of the OpenShift Service exposed for the patroni-persistent-master container.
|
||||
displayName: Master service name.
|
||||
name: PATRONI_MASTER_SERVICE_NAME
|
||||
value: patroni-persistent-master
|
||||
- description: The name of the OpenShift Service exposed for the patroni-persistent-replica containers.
|
||||
displayName: Replica service name.
|
||||
name: PATRONI_REPLICA_SERVICE_NAME
|
||||
value: patroni-persistent-replica
|
||||
- description: Maximum amount of memory the container can use.
|
||||
displayName: Memory Limit
|
||||
name: MEMORY_LIMIT
|
||||
value: 512Mi
|
||||
- description: The OpenShift Namespace where the patroni and postgresql ImageStream resides.
|
||||
displayName: ImageStream Namespace
|
||||
name: NAMESPACE
|
||||
value: openshift
|
||||
- description: Username of the superuser account for initialization.
|
||||
displayName: Superuser Username
|
||||
name: PATRONI_SUPERUSER_USERNAME
|
||||
value: postgres
|
||||
- description: Password of the superuser account for initialization.
|
||||
displayName: Superuser Passsword
|
||||
name: PATRONI_SUPERUSER_PASSWORD
|
||||
value: postgres
|
||||
- description: Username of the replication account for initialization.
|
||||
displayName: Replication Username
|
||||
name: PATRONI_REPLICATION_USERNAME
|
||||
value: postgres
|
||||
- description: Password of the replication account for initialization.
|
||||
displayName: Repication Passsword
|
||||
name: PATRONI_REPLICATION_PASSWORD
|
||||
value: postgres
|
||||
- description: Service account name used for pods and rolebindings to form a cluster in the project.
|
||||
displayName: Service Account
|
||||
name: SERVICE_ACCOUNT
|
||||
value: patroni-persistent
|
||||
- description: The size of the persistent volume to create.
|
||||
displayName: Persistent Volume Size
|
||||
name: PVC_SIZE
|
||||
value: 5Gi
|
||||
+43
@@ -0,0 +1,43 @@
|
||||
pipeline {
|
||||
agent any
|
||||
stages {
|
||||
stage ('Deploy test pod'){
|
||||
when {
|
||||
expression {
|
||||
openshift.withCluster() {
|
||||
openshift.withProject() {
|
||||
return !openshift.selector( "dc", "pgbench" ).exists()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
steps {
|
||||
script {
|
||||
openshift.withCluster() {
|
||||
openshift.withProject() {
|
||||
def pgbench = openshift.newApp( "https://github.com/stewartshea/docker-pgbench/", "--name=pgbench", "-e PGPASSWORD=postgres", "-e PGUSER=postgres", "-e PGHOST=patroni-persistent-master", "-e PGDATABASE=postgres", "-e TEST_CLIENT_COUNT=20", "-e TEST_DURATION=120" )
|
||||
def pgbenchdc = openshift.selector( "dc", "pgbench" )
|
||||
timeout(5) {
|
||||
pgbenchdc.rollout().status()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
stage ('Run benchmark Test'){
|
||||
steps {
|
||||
sh '''
|
||||
oc exec $(oc get pods -l app=pgbench | grep Running | awk '{print $1}') ./test.sh
|
||||
'''
|
||||
}
|
||||
}
|
||||
stage ('Clean up pgtest pod'){
|
||||
steps {
|
||||
sh '''
|
||||
oc delete all -l app=pgbench
|
||||
'''
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
# Jenkins Test
|
||||
This pipeline test will create a separate deployment config for a pgbench pod and execute a test against the patroni cluster. This is a sample and should be customized.
|
||||
@@ -0,0 +1,281 @@
|
||||
# headless service to avoid deletion of patronidemo-config endpoint
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: patronidemo-config
|
||||
labels:
|
||||
application: patroni
|
||||
cluster-name: patronidemo
|
||||
spec:
|
||||
clusterIP: None
|
||||
|
||||
---
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
name: &cluster_name patronidemo
|
||||
labels:
|
||||
application: patroni
|
||||
cluster-name: *cluster_name
|
||||
spec:
|
||||
replicas: 3
|
||||
serviceName: *cluster_name
|
||||
selector:
|
||||
matchLabels:
|
||||
application: patroni
|
||||
cluster-name: *cluster_name
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
application: patroni
|
||||
cluster-name: *cluster_name
|
||||
spec:
|
||||
serviceAccountName: patronidemo
|
||||
containers:
|
||||
- name: *cluster_name
|
||||
image: patroni # docker build -t patroni .
|
||||
imagePullPolicy: IfNotPresent
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
scheme: HTTP
|
||||
path: /readiness
|
||||
port: 8008
|
||||
initialDelaySeconds: 3
|
||||
periodSeconds: 10
|
||||
timeoutSeconds: 5
|
||||
successThreshold: 1
|
||||
failureThreshold: 3
|
||||
ports:
|
||||
- containerPort: 8008
|
||||
protocol: TCP
|
||||
- containerPort: 5432
|
||||
protocol: TCP
|
||||
volumeMounts:
|
||||
- mountPath: /home/postgres/pgdata
|
||||
name: pgdata
|
||||
env:
|
||||
- name: PATRONI_KUBERNETES_POD_IP
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
fieldPath: status.podIP
|
||||
- name: PATRONI_KUBERNETES_NAMESPACE
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
fieldPath: metadata.namespace
|
||||
- name: PATRONI_KUBERNETES_BYPASS_API_SERVICE
|
||||
value: 'true'
|
||||
- name: PATRONI_KUBERNETES_USE_ENDPOINTS
|
||||
value: 'true'
|
||||
- name: PATRONI_KUBERNETES_LABELS
|
||||
value: '{application: patroni, cluster-name: patronidemo}'
|
||||
- name: PATRONI_SUPERUSER_USERNAME
|
||||
value: postgres
|
||||
- name: PATRONI_SUPERUSER_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: *cluster_name
|
||||
key: superuser-password
|
||||
- name: PATRONI_REPLICATION_USERNAME
|
||||
value: standby
|
||||
- name: PATRONI_REPLICATION_PASSWORD
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
name: *cluster_name
|
||||
key: replication-password
|
||||
- name: PATRONI_SCOPE
|
||||
value: *cluster_name
|
||||
- name: PATRONI_NAME
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
fieldPath: metadata.name
|
||||
- name: PATRONI_POSTGRESQL_DATA_DIR
|
||||
value: /home/postgres/pgdata/pgroot/data
|
||||
- name: PATRONI_POSTGRESQL_PGPASS
|
||||
value: /tmp/pgpass
|
||||
- name: PATRONI_POSTGRESQL_LISTEN
|
||||
value: '0.0.0.0:5432'
|
||||
- name: PATRONI_RESTAPI_LISTEN
|
||||
value: '0.0.0.0:8008'
|
||||
terminationGracePeriodSeconds: 0
|
||||
volumes:
|
||||
- name: pgdata
|
||||
emptyDir: {}
|
||||
# volumeClaimTemplates:
|
||||
# - metadata:
|
||||
# labels:
|
||||
# application: spilo
|
||||
# spilo-cluster: *cluster_name
|
||||
# annotations:
|
||||
# volume.alpha.kubernetes.io/storage-class: anything
|
||||
# name: pgdata
|
||||
# spec:
|
||||
# accessModes:
|
||||
# - ReadWriteOnce
|
||||
# resources:
|
||||
# requests:
|
||||
# storage: 5Gi
|
||||
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Endpoints
|
||||
metadata:
|
||||
name: &cluster_name patronidemo
|
||||
labels:
|
||||
application: patroni
|
||||
cluster-name: *cluster_name
|
||||
subsets: []
|
||||
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: &cluster_name patronidemo
|
||||
labels:
|
||||
application: patroni
|
||||
cluster-name: *cluster_name
|
||||
spec:
|
||||
type: ClusterIP
|
||||
ports:
|
||||
- port: 5432
|
||||
targetPort: 5432
|
||||
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: patronidemo-repl
|
||||
labels:
|
||||
application: patroni
|
||||
cluster-name: &cluster_name patronidemo
|
||||
role: replica
|
||||
spec:
|
||||
type: ClusterIP
|
||||
selector:
|
||||
application: patroni
|
||||
cluster-name: *cluster_name
|
||||
role: replica
|
||||
ports:
|
||||
- port: 5432
|
||||
targetPort: 5432
|
||||
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Secret
|
||||
metadata:
|
||||
name: &cluster_name patronidemo
|
||||
labels:
|
||||
application: patroni
|
||||
cluster-name: *cluster_name
|
||||
type: Opaque
|
||||
data:
|
||||
superuser-password: emFsYW5kbw==
|
||||
replication-password: cmVwLXBhc3M=
|
||||
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: patronidemo
|
||||
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: Role
|
||||
metadata:
|
||||
name: patronidemo
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- configmaps
|
||||
verbs:
|
||||
- create
|
||||
- get
|
||||
- list
|
||||
- patch
|
||||
- update
|
||||
- watch
|
||||
# delete and deletecollection are required only for 'patronictl remove'
|
||||
- delete
|
||||
- deletecollection
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- endpoints
|
||||
verbs:
|
||||
- get
|
||||
- patch
|
||||
- update
|
||||
# the following three privileges are necessary only when using endpoints
|
||||
- create
|
||||
- list
|
||||
- watch
|
||||
# delete and deletecollection are required only for for 'patronictl remove'
|
||||
- delete
|
||||
- deletecollection
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- pods
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- patch
|
||||
- update
|
||||
- watch
|
||||
# The following privilege is only necessary for creation of headless service
|
||||
# for patronidemo-config endpoint, in order to prevent cleaning it up by the
|
||||
# k8s master. You can avoid giving this privilege by explicitly creating the
|
||||
# service like it is done in this manifest (lines 2..10)
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- services
|
||||
verbs:
|
||||
- create
|
||||
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: RoleBinding
|
||||
metadata:
|
||||
name: patronidemo
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: Role
|
||||
name: patronidemo
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: patronidemo
|
||||
|
||||
# Following privileges are only required if deployed not in the "default"
|
||||
# namespace and you want Patroni to bypass kubernetes service
|
||||
# (PATRONI_KUBERNETES_BYPASS_API_SERVICE=true)
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRole
|
||||
metadata:
|
||||
name: patroni-k8s-ep-access
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- endpoints
|
||||
resourceNames:
|
||||
- kubernetes
|
||||
verbs:
|
||||
- get
|
||||
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRoleBinding
|
||||
metadata:
|
||||
name: patroni-k8s-ep-access
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: ClusterRole
|
||||
name: patroni-k8s-ep-access
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: patronidemo
|
||||
# The namespace must be specified explicitly.
|
||||
# If deploying to the different namespace you have to change it.
|
||||
namespace: default
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/bin/sh
|
||||
set -e
|
||||
|
||||
pip install --ignore-installed setuptools==19.2 pyinstaller
|
||||
pyinstaller --clean --onefile patroni.spec
|
||||
+1
-1
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/env python
|
||||
from patroni import main
|
||||
from patroni.__main__ import main
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
# -*- mode: python -*-
|
||||
|
||||
block_cipher = None
|
||||
|
||||
|
||||
def hiddenimports():
|
||||
import sys
|
||||
sys.path.insert(0, '.')
|
||||
try:
|
||||
import patroni.dcs
|
||||
return patroni.dcs.dcs_modules()
|
||||
finally:
|
||||
sys.path.pop(0)
|
||||
|
||||
|
||||
a = Analysis(['patroni/__main__.py'],
|
||||
pathex=[],
|
||||
binaries=None,
|
||||
datas=None,
|
||||
hiddenimports=hiddenimports(),
|
||||
hookspath=[],
|
||||
runtime_hooks=[],
|
||||
excludes=[],
|
||||
win_no_prefer_redirects=False,
|
||||
win_private_assemblies=False,
|
||||
cipher=block_cipher)
|
||||
|
||||
pyz = PYZ(a.pure, a.zipped_data, cipher=block_cipher)
|
||||
|
||||
exe = EXE(pyz,
|
||||
a.scripts,
|
||||
a.binaries,
|
||||
a.zipfiles,
|
||||
a.datas,
|
||||
name='patroni',
|
||||
debug=False,
|
||||
strip=False,
|
||||
upx=True,
|
||||
console=True)
|
||||
+29
-122
@@ -1,134 +1,41 @@
|
||||
import logging
|
||||
import signal
|
||||
import sys
|
||||
import time
|
||||
|
||||
from patroni.api import RestApiServer
|
||||
from patroni.config import Config
|
||||
from patroni.dcs import get_dcs
|
||||
from patroni.exceptions import DCSError
|
||||
from patroni.ha import Ha
|
||||
from patroni.postgresql import Postgresql
|
||||
from patroni.utils import reap_children, sigchld_handler
|
||||
from patroni.version import __version__
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
PATRONI_ENV_PREFIX = 'PATRONI_'
|
||||
KUBERNETES_ENV_PREFIX = 'KUBERNETES_'
|
||||
MIN_PSYCOPG2 = (2, 5, 4)
|
||||
|
||||
|
||||
class Patroni(object):
|
||||
def fatal(string, *args):
|
||||
sys.stderr.write('FATAL: ' + string.format(*args) + '\n')
|
||||
sys.exit(1)
|
||||
|
||||
def __init__(self):
|
||||
self.setup_signal_handlers()
|
||||
|
||||
self.version = __version__
|
||||
self.config = Config()
|
||||
self.dcs = get_dcs(self.config)
|
||||
self.load_dynamic_configuration()
|
||||
|
||||
self.postgresql = Postgresql(self.config['postgresql'])
|
||||
self.api = RestApiServer(self, self.config['restapi'])
|
||||
self.ha = Ha(self)
|
||||
|
||||
self.tags = self.get_tags()
|
||||
self.nap_time = self.config['loop_wait']
|
||||
self.next_run = time.time()
|
||||
|
||||
def load_dynamic_configuration(self):
|
||||
while True:
|
||||
def parse_version(version):
|
||||
def _parse_version(version):
|
||||
for e in version.split('.'):
|
||||
try:
|
||||
cluster = self.dcs.get_cluster()
|
||||
if cluster and cluster.config:
|
||||
self.config.set_dynamic_configuration(cluster.config)
|
||||
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
||||
self.config.set_dynamic_configuration(self.config['bootstrap']['dcs'])
|
||||
yield int(e)
|
||||
except ValueError:
|
||||
break
|
||||
except DCSError:
|
||||
logger.warning('Can not get cluster from dcs')
|
||||
|
||||
def get_tags(self):
|
||||
return {tag: value for tag, value in self.config.get('tags', {}).items()
|
||||
if tag not in ('clonefrom', 'nofailover', 'noloadbalance') or value}
|
||||
|
||||
@property
|
||||
def nofailover(self):
|
||||
return self.tags.get('nofailover', False)
|
||||
|
||||
def reload_config(self):
|
||||
try:
|
||||
self.tags = self.get_tags()
|
||||
self.nap_time = self.config['loop_wait']
|
||||
self.dcs.set_ttl(self.config.get('ttl') or 30)
|
||||
self.dcs.set_retry_timeout(self.config.get('retry_timeout') or self.nap_time)
|
||||
self.api.reload_config(self.config['restapi'])
|
||||
self.postgresql.reload_config(self.config['postgresql'])
|
||||
except Exception:
|
||||
logger.exception('Failed to reload config_file=%s', self.config.config_file)
|
||||
|
||||
@property
|
||||
def replicatefrom(self):
|
||||
return self.tags.get('replicatefrom')
|
||||
|
||||
def sighup_handler(self, *args):
|
||||
self._received_sighup = True
|
||||
|
||||
def sigterm_handler(self, *args):
|
||||
if not self._received_sigterm:
|
||||
self._received_sigterm = True
|
||||
sys.exit()
|
||||
|
||||
@property
|
||||
def noloadbalance(self):
|
||||
return self.tags.get('noloadbalance', False)
|
||||
|
||||
def schedule_next_run(self):
|
||||
self.next_run += self.nap_time
|
||||
current_time = time.time()
|
||||
nap_time = self.next_run - current_time
|
||||
if nap_time <= 0:
|
||||
self.next_run = current_time
|
||||
elif self.dcs.watch(nap_time):
|
||||
self.next_run = time.time()
|
||||
|
||||
def run(self):
|
||||
self.api.start()
|
||||
self.next_run = time.time()
|
||||
|
||||
while not self._received_sigterm:
|
||||
if self._received_sighup:
|
||||
self._received_sighup = False
|
||||
if self.config.reload_local_configuration():
|
||||
self.reload_config()
|
||||
|
||||
logger.info(self.ha.run_cycle())
|
||||
|
||||
cluster = self.dcs.cluster
|
||||
if cluster and cluster.config and self.config.set_dynamic_configuration(cluster.config):
|
||||
self.reload_config()
|
||||
|
||||
if not self.postgresql.data_directory_empty():
|
||||
self.config.save_cache()
|
||||
|
||||
reap_children()
|
||||
self.schedule_next_run()
|
||||
|
||||
def setup_signal_handlers(self):
|
||||
self._received_sighup = False
|
||||
self._received_sigterm = False
|
||||
signal.signal(signal.SIGHUP, self.sighup_handler)
|
||||
signal.signal(signal.SIGTERM, self.sigterm_handler)
|
||||
signal.signal(signal.SIGCHLD, sigchld_handler)
|
||||
return tuple(_parse_version(version.split(' ')[0]))
|
||||
|
||||
|
||||
def main():
|
||||
logging.basicConfig(format='%(asctime)s %(levelname)s: %(message)s', level=logging.INFO)
|
||||
logging.getLogger('requests').setLevel(logging.WARNING)
|
||||
# We pass MIN_PSYCOPG2 and parse_version as arguments to simplify usage of check_psycopg from the setup.py
|
||||
def check_psycopg(_min_psycopg2=MIN_PSYCOPG2, _parse_version=parse_version):
|
||||
min_psycopg2_str = '.'.join(map(str, _min_psycopg2))
|
||||
|
||||
patroni = Patroni()
|
||||
try:
|
||||
patroni.run()
|
||||
except KeyboardInterrupt:
|
||||
pass
|
||||
finally:
|
||||
patroni.api.shutdown()
|
||||
patroni.postgresql.stop(checkpoint=False)
|
||||
patroni.dcs.delete_leader()
|
||||
from psycopg2 import __version__
|
||||
if _parse_version(__version__) >= _min_psycopg2:
|
||||
return
|
||||
version_str = __version__.split(' ')[0]
|
||||
except ImportError:
|
||||
version_str = None
|
||||
|
||||
try:
|
||||
from psycopg import __version__
|
||||
except ImportError:
|
||||
error = 'Patroni requires psycopg2>={0}, psycopg2-binary, or psycopg>=3.0'.format(min_psycopg2_str)
|
||||
if version_str:
|
||||
error += ', but only psycopg2=={0} is available'.format(version_str)
|
||||
fatal(error)
|
||||
|
||||
+178
-1
@@ -1,4 +1,181 @@
|
||||
from patroni import main
|
||||
import logging
|
||||
import os
|
||||
import signal
|
||||
import time
|
||||
|
||||
from .daemon import AbstractPatroniDaemon, abstract_main
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Patroni(AbstractPatroniDaemon):
|
||||
|
||||
def __init__(self, config):
|
||||
from .api import RestApiServer
|
||||
from .dcs import get_dcs
|
||||
from .ha import Ha
|
||||
from .postgresql import Postgresql
|
||||
from .request import PatroniRequest
|
||||
from .version import __version__
|
||||
from .watchdog import Watchdog
|
||||
|
||||
super(Patroni, self).__init__(config)
|
||||
|
||||
self.version = __version__
|
||||
self.dcs = get_dcs(self.config)
|
||||
self.watchdog = Watchdog(self.config)
|
||||
self.load_dynamic_configuration()
|
||||
|
||||
self.postgresql = Postgresql(self.config['postgresql'])
|
||||
self.api = RestApiServer(self, self.config['restapi'])
|
||||
self.request = PatroniRequest(self.config, True)
|
||||
self.ha = Ha(self)
|
||||
|
||||
self.tags = self.get_tags()
|
||||
self.next_run = time.time()
|
||||
self.scheduled_restart = {}
|
||||
|
||||
def load_dynamic_configuration(self):
|
||||
from patroni.exceptions import DCSError
|
||||
while True:
|
||||
try:
|
||||
cluster = self.dcs.get_cluster()
|
||||
if cluster and cluster.config and cluster.config.data:
|
||||
if self.config.set_dynamic_configuration(cluster.config):
|
||||
self.dcs.reload_config(self.config)
|
||||
self.watchdog.reload_config(self.config)
|
||||
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
||||
if self.config.set_dynamic_configuration(self.config['bootstrap']['dcs']):
|
||||
self.dcs.reload_config(self.config)
|
||||
break
|
||||
except DCSError:
|
||||
logger.warning('Can not get cluster from dcs')
|
||||
time.sleep(5)
|
||||
|
||||
def get_tags(self):
|
||||
return {tag: value for tag, value in self.config.get('tags', {}).items()
|
||||
if tag not in ('clonefrom', 'nofailover', 'noloadbalance', 'nosync') or value}
|
||||
|
||||
@property
|
||||
def nofailover(self):
|
||||
return bool(self.tags.get('nofailover', False))
|
||||
|
||||
@property
|
||||
def nosync(self):
|
||||
return bool(self.tags.get('nosync', False))
|
||||
|
||||
def reload_config(self, sighup=False, local=False):
|
||||
try:
|
||||
super(Patroni, self).reload_config(sighup, local)
|
||||
if local:
|
||||
self.tags = self.get_tags()
|
||||
self.request.reload_config(self.config)
|
||||
if local or sighup and self.api.reload_local_certificate():
|
||||
self.api.reload_config(self.config['restapi'])
|
||||
self.watchdog.reload_config(self.config)
|
||||
self.postgresql.reload_config(self.config['postgresql'], sighup)
|
||||
self.dcs.reload_config(self.config)
|
||||
except Exception:
|
||||
logger.exception('Failed to reload config_file=%s', self.config.config_file)
|
||||
|
||||
@property
|
||||
def replicatefrom(self):
|
||||
return self.tags.get('replicatefrom')
|
||||
|
||||
@property
|
||||
def noloadbalance(self):
|
||||
return bool(self.tags.get('noloadbalance', False))
|
||||
|
||||
def schedule_next_run(self):
|
||||
self.next_run += self.dcs.loop_wait
|
||||
current_time = time.time()
|
||||
nap_time = self.next_run - current_time
|
||||
if nap_time <= 0:
|
||||
self.next_run = current_time
|
||||
# Release the GIL so we don't starve anyone waiting on async_executor lock
|
||||
time.sleep(0.001)
|
||||
# Warn user that Patroni is not keeping up
|
||||
logger.warning("Loop time exceeded, rescheduling immediately.")
|
||||
elif self.ha.watch(nap_time):
|
||||
self.next_run = time.time()
|
||||
|
||||
def run(self):
|
||||
self.api.start()
|
||||
self.next_run = time.time()
|
||||
super(Patroni, self).run()
|
||||
|
||||
def _run_cycle(self):
|
||||
logger.info(self.ha.run_cycle())
|
||||
|
||||
if self.dcs.cluster and self.dcs.cluster.config and self.dcs.cluster.config.data \
|
||||
and self.config.set_dynamic_configuration(self.dcs.cluster.config):
|
||||
self.reload_config()
|
||||
|
||||
if self.postgresql.role != 'uninitialized':
|
||||
self.config.save_cache()
|
||||
|
||||
self.schedule_next_run()
|
||||
|
||||
def _shutdown(self):
|
||||
try:
|
||||
self.api.shutdown()
|
||||
except Exception:
|
||||
logger.exception('Exception during RestApi.shutdown')
|
||||
try:
|
||||
self.ha.shutdown()
|
||||
except Exception:
|
||||
logger.exception('Exception during Ha.shutdown')
|
||||
|
||||
|
||||
def patroni_main():
|
||||
from multiprocessing import freeze_support
|
||||
from patroni.validator import schema
|
||||
|
||||
freeze_support()
|
||||
abstract_main(Patroni, schema)
|
||||
|
||||
|
||||
def main():
|
||||
if os.getpid() != 1:
|
||||
from . import check_psycopg
|
||||
|
||||
check_psycopg()
|
||||
return patroni_main()
|
||||
|
||||
# Patroni started with PID=1, it looks like we are in the container
|
||||
pid = 0
|
||||
|
||||
# Looks like we are in a docker, so we will act like init
|
||||
def sigchld_handler(signo, stack_frame):
|
||||
try:
|
||||
while True:
|
||||
ret = os.waitpid(-1, os.WNOHANG)
|
||||
if ret == (0, 0):
|
||||
break
|
||||
elif ret[0] != pid:
|
||||
logger.info('Reaped pid=%s, exit status=%s', *ret)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
def passtochild(signo, stack_frame):
|
||||
if pid:
|
||||
os.kill(pid, signo)
|
||||
|
||||
if os.name != 'nt':
|
||||
signal.signal(signal.SIGCHLD, sigchld_handler)
|
||||
signal.signal(signal.SIGHUP, passtochild)
|
||||
signal.signal(signal.SIGQUIT, passtochild)
|
||||
signal.signal(signal.SIGUSR1, passtochild)
|
||||
signal.signal(signal.SIGUSR2, passtochild)
|
||||
signal.signal(signal.SIGINT, passtochild)
|
||||
signal.signal(signal.SIGABRT, passtochild)
|
||||
signal.signal(signal.SIGTERM, passtochild)
|
||||
|
||||
import multiprocessing
|
||||
patroni = multiprocessing.Process(target=patroni_main)
|
||||
patroni.start()
|
||||
pid = patroni.pid
|
||||
patroni.join()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
+682
-193
File diff suppressed because it is too large
Load Diff
+95
-12
@@ -1,27 +1,78 @@
|
||||
import logging
|
||||
from threading import Lock, Thread
|
||||
from threading import Event, Lock, RLock, Thread
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class CriticalTask(object):
|
||||
"""Represents a critical task in a background process that we either need to cancel or get the result of.
|
||||
|
||||
Fields of this object may be accessed only when holding a lock on it. To perform the critical task the background
|
||||
thread must, while holding lock on this object, check `is_cancelled` flag, run the task and mark the task as
|
||||
complete using `complete()`.
|
||||
|
||||
The main thread must hold async lock to prevent the task from completing, hold lock on critical task object,
|
||||
call cancel. If the task has completed `cancel()` will return False and `result` field will contain the result of
|
||||
the task. When cancel returns True it is guaranteed that the background task will notice the `is_cancelled` flag.
|
||||
"""
|
||||
def __init__(self):
|
||||
self._lock = Lock()
|
||||
self.is_cancelled = False
|
||||
self.result = None
|
||||
|
||||
def reset(self):
|
||||
"""Must be called every time the background task is finished.
|
||||
|
||||
Must be called from async thread. Caller must hold lock on async executor when calling."""
|
||||
self.is_cancelled = False
|
||||
self.result = None
|
||||
|
||||
def cancel(self):
|
||||
"""Tries to cancel the task, returns True if the task has already run.
|
||||
|
||||
Caller must hold lock on async executor and the task when calling."""
|
||||
if self.result is not None:
|
||||
return False
|
||||
self.is_cancelled = True
|
||||
return True
|
||||
|
||||
def complete(self, result):
|
||||
"""Mark task as completed along with a result.
|
||||
|
||||
Must be called from async thread. Caller must hold lock on task when calling."""
|
||||
self.result = result
|
||||
|
||||
def __enter__(self):
|
||||
self._lock.acquire()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_val, exc_tb):
|
||||
self._lock.release()
|
||||
|
||||
|
||||
class AsyncExecutor(object):
|
||||
|
||||
def __init__(self):
|
||||
self._busy = False
|
||||
self._thread_lock = Lock()
|
||||
def __init__(self, cancellable, ha_wakeup):
|
||||
self._cancellable = cancellable
|
||||
self._ha_wakeup = ha_wakeup
|
||||
self._thread_lock = RLock()
|
||||
self._scheduled_action = None
|
||||
self._scheduled_action_lock = Lock()
|
||||
self._scheduled_action_lock = RLock()
|
||||
self._is_cancelled = False
|
||||
self._finish_event = Event()
|
||||
self.critical_task = CriticalTask()
|
||||
|
||||
@property
|
||||
def busy(self):
|
||||
return self._busy
|
||||
return self.scheduled_action is not None
|
||||
|
||||
def schedule(self, action, immediately=False):
|
||||
def schedule(self, action):
|
||||
with self._scheduled_action_lock:
|
||||
if self._scheduled_action is not None:
|
||||
return self._scheduled_action
|
||||
self._scheduled_action = action
|
||||
self._busy = immediately
|
||||
self._is_cancelled = False
|
||||
self._finish_event.set()
|
||||
return None
|
||||
|
||||
@property
|
||||
@@ -34,19 +85,51 @@ class AsyncExecutor(object):
|
||||
self._scheduled_action = None
|
||||
|
||||
def run(self, func, args=()):
|
||||
wakeup = False
|
||||
try:
|
||||
return func(*args) if args else func()
|
||||
except:
|
||||
with self:
|
||||
if self._is_cancelled:
|
||||
return
|
||||
self._finish_event.clear()
|
||||
|
||||
self._cancellable.reset_is_cancelled()
|
||||
# if the func returned something (not None) - wake up main HA loop
|
||||
wakeup = func(*args) if args else func()
|
||||
return wakeup
|
||||
except Exception:
|
||||
logger.exception('Exception during execution of long running task %s', self.scheduled_action)
|
||||
finally:
|
||||
with self:
|
||||
self._busy = False
|
||||
self.reset_scheduled_action()
|
||||
self._finish_event.set()
|
||||
with self.critical_task:
|
||||
self.critical_task.reset()
|
||||
if wakeup is not None:
|
||||
self._ha_wakeup()
|
||||
|
||||
def run_async(self, func, args=()):
|
||||
self._busy = True
|
||||
Thread(target=self.run, args=(func, args)).start()
|
||||
|
||||
def try_run_async(self, action, func, args=()):
|
||||
prev = self.schedule(action)
|
||||
if prev is None:
|
||||
return self.run_async(func, args)
|
||||
return 'Failed to run {0}, {1} is already in progress'.format(action, prev)
|
||||
|
||||
def cancel(self):
|
||||
with self:
|
||||
with self._scheduled_action_lock:
|
||||
if self._scheduled_action is None:
|
||||
return
|
||||
logger.warning('Cancelling long running task %s', self._scheduled_action)
|
||||
self._is_cancelled = True
|
||||
|
||||
self._cancellable.cancel()
|
||||
self._finish_event.wait()
|
||||
|
||||
with self:
|
||||
self.reset_scheduled_action()
|
||||
|
||||
def __enter__(self):
|
||||
self._thread_lock.acquire()
|
||||
|
||||
|
||||
+226
-82
@@ -1,18 +1,39 @@
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
import shutil
|
||||
import tempfile
|
||||
import yaml
|
||||
|
||||
from collections import defaultdict
|
||||
from copy import deepcopy
|
||||
from patroni import PATRONI_ENV_PREFIX
|
||||
from patroni.exceptions import ConfigParseError
|
||||
from patroni.dcs import ClusterConfig
|
||||
from patroni.postgresql import Postgresql
|
||||
from patroni.utils import deep_compare, parse_int, patch_config
|
||||
from patroni.postgresql.config import CaseInsensitiveDict, ConfigHandler
|
||||
from patroni.utils import deep_compare, parse_bool, parse_int, patch_config
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_AUTH_ALLOWED_PARAMETERS = (
|
||||
'username',
|
||||
'password',
|
||||
'sslmode',
|
||||
'sslcert',
|
||||
'sslkey',
|
||||
'sslpassword',
|
||||
'sslrootcert',
|
||||
'sslcrl',
|
||||
'sslcrldir',
|
||||
'gssencmode',
|
||||
'channel_binding'
|
||||
)
|
||||
|
||||
|
||||
def default_validator(conf):
|
||||
if not conf:
|
||||
return "Config is empty."
|
||||
|
||||
|
||||
class Config(object):
|
||||
"""
|
||||
@@ -34,40 +55,59 @@ class Config(object):
|
||||
to work with it as with the old `config` object.
|
||||
"""
|
||||
|
||||
PATRONI_ENV_PREFIX = 'PATRONI_'
|
||||
PATRONI_CONFIG_VARIABLE = PATRONI_ENV_PREFIX + 'CONFIGURATION'
|
||||
|
||||
__CACHE_FILENAME = 'patroni.dynamic.json'
|
||||
__DEFAULT_CONFIG = {
|
||||
'ttl': 30, 'loop_wait': 10, 'retry_timeout': 10,
|
||||
'maximum_lag_on_failover': 1048576,
|
||||
'maximum_lag_on_syncnode': -1,
|
||||
'check_timeline': False,
|
||||
'master_start_timeout': 300,
|
||||
'master_stop_timeout': 0,
|
||||
'synchronous_mode': False,
|
||||
'synchronous_mode_strict': False,
|
||||
'synchronous_node_count': 1,
|
||||
'standby_cluster': {
|
||||
'create_replica_methods': '',
|
||||
'host': '',
|
||||
'port': '',
|
||||
'primary_slot_name': '',
|
||||
'restore_command': '',
|
||||
'archive_cleanup_command': '',
|
||||
'recovery_min_apply_delay': ''
|
||||
},
|
||||
'postgresql': {
|
||||
'bin_dir': '',
|
||||
'use_slots': True,
|
||||
'parameters': {p: v[0] for p, v in Postgresql.CMDLINE_OPTIONS.items()}
|
||||
'parameters': CaseInsensitiveDict({p: v[0] for p, v in ConfigHandler.CMDLINE_OPTIONS.items()
|
||||
if p not in ('wal_keep_segments', 'wal_keep_size')})
|
||||
},
|
||||
'watchdog': {
|
||||
'mode': 'automatic',
|
||||
}
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
def __init__(self, configfile, validator=default_validator):
|
||||
self._modify_index = -1
|
||||
self._dynamic_configuration = {}
|
||||
|
||||
self.__environment_configuration = self._build_environment_configuration()
|
||||
|
||||
# Patroni reads the configuration from the command-line argument if it exists, otherwise from the environment
|
||||
self._config_file = len(sys.argv) >= 2 and os.path.isfile(sys.argv[1]) and sys.argv[1]
|
||||
self._config_file = configfile and os.path.exists(configfile) and configfile
|
||||
if self._config_file:
|
||||
self._local_configuration = self._load_config_file()
|
||||
else:
|
||||
config_env = os.environ.pop(self.PATRONI_CONFIG_VARIABLE, None)
|
||||
self._local_configuration = config_env and yaml.safe_load(config_env) or self.__environment_configuration
|
||||
if not self._local_configuration:
|
||||
print('Usage: {0} config.yml'.format(sys.argv[0]))
|
||||
print('\tPatroni may also read the configuration from the {0} environment variable'.
|
||||
format(self.PATRONI_CONFIG_VARIABLE))
|
||||
exit(1)
|
||||
if validator:
|
||||
error = validator(self._local_configuration)
|
||||
if error:
|
||||
raise ConfigParseError(error)
|
||||
|
||||
self.__effective_configuration = self._build_effective_configuration({}, self._local_configuration)
|
||||
self._data_dir = self.__effective_configuration['postgresql']['data_dir']
|
||||
self._data_dir = self.__effective_configuration.get('postgresql', {}).get('data_dir', "")
|
||||
self._cache_file = os.path.join(self._data_dir, self.__CACHE_FILENAME)
|
||||
self._load_cache()
|
||||
self._cache_needs_saving = False
|
||||
@@ -80,12 +120,35 @@ class Config(object):
|
||||
def dynamic_configuration(self):
|
||||
return deepcopy(self._dynamic_configuration)
|
||||
|
||||
def check_mode(self, mode):
|
||||
return bool(parse_bool(self._dynamic_configuration.get(mode)))
|
||||
|
||||
def _load_config_path(self, path):
|
||||
"""
|
||||
If path is a file, loads the yml file pointed to by path.
|
||||
If path is a directory, loads all yml files in that directory in alphabetical order
|
||||
"""
|
||||
if os.path.isfile(path):
|
||||
files = [path]
|
||||
elif os.path.isdir(path):
|
||||
files = [os.path.join(path, f) for f in sorted(os.listdir(path))
|
||||
if (f.endswith('.yml') or f.endswith('.yaml')) and os.path.isfile(os.path.join(path, f))]
|
||||
else:
|
||||
logger.error('config path %s is neither directory nor file', path)
|
||||
raise ConfigParseError('invalid config path')
|
||||
|
||||
overall_config = {}
|
||||
for fname in files:
|
||||
with open(fname) as f:
|
||||
config = yaml.safe_load(f)
|
||||
patch_config(overall_config, config)
|
||||
return overall_config
|
||||
|
||||
def _load_config_file(self):
|
||||
"""Loads config.yaml from filesystem and applies some values which were set via ENV"""
|
||||
with open(self._config_file) as f:
|
||||
config = yaml.safe_load(f)
|
||||
patch_config(config, self.__environment_configuration)
|
||||
return config
|
||||
config = self._load_config_path(self._config_file)
|
||||
patch_config(config, self.__environment_configuration)
|
||||
return config
|
||||
|
||||
def _load_cache(self):
|
||||
if os.path.isfile(self._cache_file):
|
||||
@@ -103,7 +166,7 @@ class Config(object):
|
||||
with os.fdopen(fd, 'w') as f:
|
||||
fd = None
|
||||
json.dump(self.dynamic_configuration, f)
|
||||
tmpfile = os.rename(tmpfile, self._cache_file)
|
||||
tmpfile = shutil.move(tmpfile, self._cache_file)
|
||||
self._cache_needs_saving = False
|
||||
except Exception:
|
||||
logger.exception('Exception when saving file: %s', self._cache_file)
|
||||
@@ -136,29 +199,25 @@ class Config(object):
|
||||
except Exception:
|
||||
logger.exception('Exception when setting dynamic_configuration')
|
||||
|
||||
def reload_local_configuration(self, dry_run=False):
|
||||
def reload_local_configuration(self):
|
||||
if self.config_file:
|
||||
try:
|
||||
configuration = self._load_config_file()
|
||||
if not deep_compare(self._local_configuration, configuration):
|
||||
new_configuration = self._build_effective_configuration(self._dynamic_configuration, configuration)
|
||||
if dry_run:
|
||||
return not deep_compare(new_configuration, self.__effective_configuration)
|
||||
self._local_configuration = configuration
|
||||
self.__effective_configuration = new_configuration
|
||||
return True
|
||||
else:
|
||||
logger.info('No local configuration items changed.')
|
||||
except Exception:
|
||||
logger.exception('Exception when reloading local configuration from %s', self.config_file)
|
||||
if dry_run:
|
||||
raise
|
||||
|
||||
@staticmethod
|
||||
def _process_postgresql_parameters(parameters, is_local=False):
|
||||
ret = {}
|
||||
for name, value in (parameters or {}).items():
|
||||
if name not in Postgresql.CMDLINE_OPTIONS or not is_local and Postgresql.CMDLINE_OPTIONS[name][1](value):
|
||||
ret[name] = value
|
||||
return ret
|
||||
return {name: value for name, value in (parameters or {}).items()
|
||||
if name not in ConfigHandler.CMDLINE_OPTIONS or
|
||||
not is_local and ConfigHandler.CMDLINE_OPTIONS[name][1](value)}
|
||||
|
||||
def _safe_copy_dynamic_configuration(self, dynamic_configuration):
|
||||
config = deepcopy(self.__DEFAULT_CONFIG)
|
||||
@@ -170,8 +229,15 @@ class Config(object):
|
||||
config['postgresql'][name].update(self._process_postgresql_parameters(value))
|
||||
elif name not in ('connect_address', 'listen', 'data_dir', 'pgpass', 'authentication'):
|
||||
config['postgresql'][name] = deepcopy(value)
|
||||
elif name in config: # only variables present in __DEFAULT_CONFIG allowed to be overriden from DCS
|
||||
config[name] = int(value)
|
||||
elif name == 'standby_cluster':
|
||||
for name, value in (value or {}).items():
|
||||
if name in self.__DEFAULT_CONFIG['standby_cluster']:
|
||||
config['standby_cluster'][name] = deepcopy(value)
|
||||
elif name in config: # only variables present in __DEFAULT_CONFIG allowed to be overridden from DCS
|
||||
if name in ('synchronous_mode', 'synchronous_mode_strict'):
|
||||
config[name] = value
|
||||
else:
|
||||
config[name] = int(value)
|
||||
return config
|
||||
|
||||
@staticmethod
|
||||
@@ -179,44 +245,50 @@ class Config(object):
|
||||
ret = defaultdict(dict)
|
||||
|
||||
def _popenv(name):
|
||||
return os.environ.pop(Config.PATRONI_ENV_PREFIX + name.upper(), None)
|
||||
return os.environ.pop(PATRONI_ENV_PREFIX + name.upper(), None)
|
||||
|
||||
for param in ('name', 'namespace', 'scope'):
|
||||
value = _popenv(param)
|
||||
if value:
|
||||
ret[param] = value
|
||||
|
||||
def _fix_log_env(name, oldname):
|
||||
value = _popenv(oldname)
|
||||
name = PATRONI_ENV_PREFIX + 'LOG_' + name.upper()
|
||||
if value and name not in os.environ:
|
||||
os.environ[name] = value
|
||||
|
||||
for name, oldname in (('level', 'loglevel'), ('format', 'logformat'), ('dateformat', 'log_datefmt')):
|
||||
_fix_log_env(name, oldname)
|
||||
|
||||
def _set_section_values(section, params):
|
||||
for param in params:
|
||||
value = _popenv(section + '_' + param)
|
||||
if value:
|
||||
ret[section][param] = value
|
||||
|
||||
_set_section_values('restapi', ['listen', 'connect_address', 'certfile', 'keyfile'])
|
||||
_set_section_values('postgresql', ['listen', 'connect_address', 'data_dir', 'pgpass'])
|
||||
_set_section_values('restapi', ['listen', 'connect_address', 'certfile', 'keyfile', 'keyfile_password',
|
||||
'cafile', 'ciphers', 'verify_client', 'http_extra_headers',
|
||||
'https_extra_headers', 'allowlist', 'allowlist_include_members'])
|
||||
_set_section_values('ctl', ['insecure', 'cacert', 'certfile', 'keyfile', 'keyfile_password'])
|
||||
_set_section_values('postgresql', ['listen', 'connect_address', 'config_dir', 'data_dir', 'pgpass', 'bin_dir'])
|
||||
_set_section_values('log', ['level', 'traceback_level', 'format', 'dateformat', 'max_queue_size',
|
||||
'dir', 'file_size', 'file_num', 'loggers'])
|
||||
_set_section_values('raft', ['data_dir', 'self_addr', 'partner_addrs', 'password', 'bind_addr'])
|
||||
|
||||
def _get_auth(name):
|
||||
ret = {}
|
||||
for param in ('username', 'password'):
|
||||
value = _popenv(name + '_' + param)
|
||||
if value:
|
||||
ret[param] = value
|
||||
return ret
|
||||
for first, second in (('restapi', 'allowlist_include_members'), ('ctl', 'insecure')):
|
||||
value = ret.get(first, {}).pop(second, None)
|
||||
if value:
|
||||
value = parse_bool(value)
|
||||
if value is not None:
|
||||
ret[first][second] = value
|
||||
|
||||
restapi_auth = _get_auth('restapi')
|
||||
if restapi_auth:
|
||||
ret['restapi']['authentication'] = restapi_auth
|
||||
|
||||
authentication = {}
|
||||
for user_type in ('replication', 'superuser'):
|
||||
entry = _get_auth(user_type)
|
||||
if entry:
|
||||
authentication[user_type] = entry
|
||||
|
||||
if authentication:
|
||||
ret['postgresql']['authentication'] = authentication
|
||||
|
||||
users = {}
|
||||
for second in ('max_queue_size', 'file_size', 'file_num'):
|
||||
value = ret.get('log', {}).pop(second, None)
|
||||
if value:
|
||||
value = parse_int(value)
|
||||
if value is not None:
|
||||
ret['log'][second] = value
|
||||
|
||||
def _parse_list(value):
|
||||
if not (value.strip().startswith('-') or '[' in value):
|
||||
@@ -227,29 +299,89 @@ class Config(object):
|
||||
logger.exception('Exception when parsing list %s', value)
|
||||
return None
|
||||
|
||||
for first, second in (('raft', 'partner_addrs'), ('restapi', 'allowlist')):
|
||||
value = ret.get(first, {}).pop(second, None)
|
||||
if value:
|
||||
value = _parse_list(value)
|
||||
if value:
|
||||
ret[first][second] = value
|
||||
|
||||
def _parse_dict(value):
|
||||
if not value.strip().startswith('{'):
|
||||
value = '{{{0}}}'.format(value)
|
||||
try:
|
||||
return yaml.safe_load(value)
|
||||
except Exception:
|
||||
logger.exception('Exception when parsing dict %s', value)
|
||||
return None
|
||||
|
||||
for first, params in (('restapi', ('http_extra_headers', 'https_extra_headers')), ('log', ('loggers',))):
|
||||
for second in params:
|
||||
value = ret.get(first, {}).pop(second, None)
|
||||
if value:
|
||||
value = _parse_dict(value)
|
||||
if value:
|
||||
ret[first][second] = value
|
||||
|
||||
def _get_auth(name, params=None):
|
||||
ret = {}
|
||||
for param in params or _AUTH_ALLOWED_PARAMETERS[:2]:
|
||||
value = _popenv(name + '_' + param)
|
||||
if value:
|
||||
ret[param] = value
|
||||
return ret
|
||||
|
||||
restapi_auth = _get_auth('restapi')
|
||||
if restapi_auth:
|
||||
ret['restapi']['authentication'] = restapi_auth
|
||||
|
||||
authentication = {}
|
||||
for user_type in ('replication', 'superuser', 'rewind'):
|
||||
entry = _get_auth(user_type, _AUTH_ALLOWED_PARAMETERS)
|
||||
if entry:
|
||||
authentication[user_type] = entry
|
||||
|
||||
if authentication:
|
||||
ret['postgresql']['authentication'] = authentication
|
||||
|
||||
for param in list(os.environ.keys()):
|
||||
if param.startswith(Config.PATRONI_ENV_PREFIX):
|
||||
if param.startswith(PATRONI_ENV_PREFIX):
|
||||
# PATRONI_(ETCD|CONSUL|ZOOKEEPER|EXHIBITOR|...)_(HOSTS?|PORT|..)
|
||||
name, suffix = (param[8:].split('_', 1) + [''])[:2]
|
||||
if suffix in ('HOST', 'HOSTS', 'PORT', 'USE_PROXIES', 'PROTOCOL', 'SRV', 'SRV_SUFFIX', 'URL', 'PROXY',
|
||||
'CACERT', 'CERT', 'KEY', 'VERIFY', 'TOKEN', 'CHECKS', 'DC', 'CONSISTENCY',
|
||||
'REGISTER_SERVICE', 'SERVICE_CHECK_INTERVAL', 'SERVICE_CHECK_TLS_SERVER_NAME',
|
||||
'NAMESPACE', 'CONTEXT', 'USE_ENDPOINTS', 'SCOPE_LABEL', 'ROLE_LABEL', 'POD_IP',
|
||||
'PORTS', 'LABELS', 'BYPASS_API_SERVICE', 'KEY_PASSWORD', 'USE_SSL', 'SET_ACLS') and name:
|
||||
value = os.environ.pop(param)
|
||||
if suffix == 'PORT':
|
||||
value = value and parse_int(value)
|
||||
elif suffix in ('HOSTS', 'PORTS', 'CHECKS'):
|
||||
value = value and _parse_list(value)
|
||||
elif suffix in ('LABELS', 'SET_ACLS'):
|
||||
value = _parse_dict(value)
|
||||
elif suffix in ('USE_PROXIES', 'REGISTER_SERVICE', 'USE_ENDPOINTS', 'BYPASS_API_SERVICE', 'VERIFY'):
|
||||
value = parse_bool(value)
|
||||
if value:
|
||||
ret[name.lower()][suffix.lower()] = value
|
||||
for dcs in ('etcd', 'etcd3'):
|
||||
if dcs in ret:
|
||||
ret[dcs].update(_get_auth(dcs))
|
||||
|
||||
users = {}
|
||||
for param in list(os.environ.keys()):
|
||||
if param.startswith(PATRONI_ENV_PREFIX):
|
||||
name, suffix = (param[8:].rsplit('_', 1) + [''])[:2]
|
||||
if name and suffix:
|
||||
# PATRONI_(ETCD|CONSUL|ZOOKEEPER|EXHIBITOR|...)_(HOSTS?|PORT)
|
||||
if suffix in ('HOST', 'HOSTS', 'PORT') and '_' not in name:
|
||||
value = os.environ.pop(param)
|
||||
if suffix == 'PORT':
|
||||
value = value and parse_int(value)
|
||||
elif suffix == 'HOSTS':
|
||||
value = value and _parse_list(value)
|
||||
if value:
|
||||
ret[name.lower()][suffix.lower()] = value
|
||||
# PATRONI_<username>_PASSWORD=<password>, PATRONI_<username>_OPTIONS=<option1,option2,...>
|
||||
# CREATE USER "<username>" WITH <OPTIONS> PASSWORD '<password>'
|
||||
elif suffix == 'PASSWORD':
|
||||
password = os.environ.pop(param)
|
||||
if password:
|
||||
users[name] = {'password': password}
|
||||
options = os.environ.pop(param[:-9] + '_OPTIONS', None)
|
||||
options = options and _parse_list(options)
|
||||
if options:
|
||||
users[name]['options'] = options
|
||||
# PATRONI_<username>_PASSWORD=<password>, PATRONI_<username>_OPTIONS=<option1,option2,...>
|
||||
# CREATE USER "<username>" WITH <OPTIONS> PASSWORD '<password>'
|
||||
if name and suffix == 'PASSWORD':
|
||||
password = os.environ.pop(param)
|
||||
if password:
|
||||
users[name] = {'password': password}
|
||||
options = os.environ.pop(param[:-9] + '_OPTIONS', None)
|
||||
options = options and _parse_list(options)
|
||||
if options:
|
||||
users[name]['options'] = options
|
||||
if users:
|
||||
ret['bootstrap']['users'] = users
|
||||
|
||||
@@ -264,14 +396,12 @@ class Config(object):
|
||||
config['postgresql'][name].update(self._process_postgresql_parameters(value, True))
|
||||
elif name != 'use_slots': # replication slots must be enabled/disabled globally
|
||||
config['postgresql'][name] = deepcopy(value)
|
||||
elif name not in config:
|
||||
elif name not in config or name in ['watchdog']:
|
||||
config[name] = deepcopy(value) if value else {}
|
||||
|
||||
# restapi server expects to get restapi.auth = 'username:password'
|
||||
if 'authentication' in config['restapi']:
|
||||
restapi = config['restapi']
|
||||
auth = restapi['authentication']
|
||||
restapi['auth'] = '{0}:{1}'.format(auth['username'], auth['password'])
|
||||
if 'restapi' in config and 'authentication' in config['restapi']:
|
||||
config['restapi']['auth'] = '{username}:{password}'.format(**config['restapi']['authentication'])
|
||||
|
||||
# special treatment for old config
|
||||
|
||||
@@ -289,12 +419,26 @@ class Config(object):
|
||||
if 'superuser' not in pg_config['authentication'] and 'pg_rewind' in pg_config:
|
||||
pg_config['authentication']['superuser'] = pg_config['pg_rewind']
|
||||
|
||||
# handle setting additional connection parameters that may be available
|
||||
# in the configuration file, such as SSL connection parameters
|
||||
for name, value in pg_config['authentication'].items():
|
||||
pg_config['authentication'][name] = {n: v for n, v in value.items() if n in _AUTH_ALLOWED_PARAMETERS}
|
||||
|
||||
# no 'name' in config
|
||||
if 'name' not in config and 'name' in pg_config:
|
||||
config['name'] = pg_config['name']
|
||||
|
||||
pg_config.update({p: config[p] for p in ('name', 'scope', 'retry_timeout',
|
||||
'maximum_lag_on_failover') if p in config})
|
||||
updated_fields = (
|
||||
'name',
|
||||
'scope',
|
||||
'retry_timeout',
|
||||
'synchronous_mode',
|
||||
'synchronous_mode_strict',
|
||||
'synchronous_node_count',
|
||||
'maximum_lag_on_syncnode'
|
||||
)
|
||||
|
||||
pg_config.update({p: config[p] for p in updated_fields if p in config})
|
||||
|
||||
return config
|
||||
|
||||
|
||||
+939
-272
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,104 @@
|
||||
import abc
|
||||
import os
|
||||
import signal
|
||||
import six
|
||||
import sys
|
||||
|
||||
from threading import Lock
|
||||
|
||||
|
||||
@six.add_metaclass(abc.ABCMeta)
|
||||
class AbstractPatroniDaemon(object):
|
||||
|
||||
def __init__(self, config):
|
||||
from patroni.log import PatroniLogger
|
||||
|
||||
self.setup_signal_handlers()
|
||||
|
||||
self.logger = PatroniLogger()
|
||||
self.config = config
|
||||
AbstractPatroniDaemon.reload_config(self, local=True)
|
||||
|
||||
def sighup_handler(self, *args):
|
||||
self._received_sighup = True
|
||||
|
||||
def sigterm_handler(self, *args):
|
||||
with self._sigterm_lock:
|
||||
if not self._received_sigterm:
|
||||
self._received_sigterm = True
|
||||
sys.exit()
|
||||
|
||||
def setup_signal_handlers(self):
|
||||
self._received_sighup = False
|
||||
self._sigterm_lock = Lock()
|
||||
self._received_sigterm = False
|
||||
if os.name != 'nt':
|
||||
signal.signal(signal.SIGHUP, self.sighup_handler)
|
||||
signal.signal(signal.SIGTERM, self.sigterm_handler)
|
||||
|
||||
@property
|
||||
def received_sigterm(self):
|
||||
with self._sigterm_lock:
|
||||
return self._received_sigterm
|
||||
|
||||
def reload_config(self, sighup=False, local=False):
|
||||
if local:
|
||||
self.logger.reload_config(self.config.get('log', {}))
|
||||
|
||||
@abc.abstractmethod
|
||||
def _run_cycle(self):
|
||||
"""_run_cycle"""
|
||||
|
||||
def run(self):
|
||||
self.logger.start()
|
||||
while not self.received_sigterm:
|
||||
if self._received_sighup:
|
||||
self._received_sighup = False
|
||||
self.reload_config(True, self.config.reload_local_configuration())
|
||||
|
||||
self._run_cycle()
|
||||
|
||||
@abc.abstractmethod
|
||||
def _shutdown(self):
|
||||
"""_shutdown"""
|
||||
|
||||
def shutdown(self):
|
||||
with self._sigterm_lock:
|
||||
self._received_sigterm = True
|
||||
self._shutdown()
|
||||
self.logger.shutdown()
|
||||
|
||||
|
||||
def abstract_main(cls, validator=None):
|
||||
import argparse
|
||||
|
||||
from .config import Config, ConfigParseError
|
||||
from .version import __version__
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument('--version', action='version', version='%(prog)s {0}'.format(__version__))
|
||||
if validator:
|
||||
parser.add_argument('--validate-config', action='store_true', help='Run config validator and exit')
|
||||
parser.add_argument('configfile', nargs='?', default='',
|
||||
help='Patroni may also read the configuration from the {0} environment variable'
|
||||
.format(Config.PATRONI_CONFIG_VARIABLE))
|
||||
args = parser.parse_args()
|
||||
try:
|
||||
if validator and args.validate_config:
|
||||
Config(args.configfile, validator=validator)
|
||||
sys.exit()
|
||||
|
||||
config = Config(args.configfile)
|
||||
except ConfigParseError as e:
|
||||
if e.value:
|
||||
print(e.value)
|
||||
parser.print_help()
|
||||
sys.exit(1)
|
||||
|
||||
controller = cls(config)
|
||||
try:
|
||||
controller.run()
|
||||
except KeyboardInterrupt:
|
||||
pass
|
||||
finally:
|
||||
controller.shutdown()
|
||||
+616
-70
@@ -1,17 +1,44 @@
|
||||
import abc
|
||||
import dateutil
|
||||
import dateutil.parser
|
||||
import importlib
|
||||
import inspect
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import pkgutil
|
||||
import re
|
||||
import six
|
||||
import sys
|
||||
import time
|
||||
|
||||
from collections import namedtuple
|
||||
from patroni.exceptions import PatroniException
|
||||
from collections import defaultdict, namedtuple
|
||||
from copy import deepcopy
|
||||
from random import randint
|
||||
from six.moves.urllib_parse import urlparse, urlunparse, parse_qsl
|
||||
from threading import Event, Lock
|
||||
|
||||
from ..exceptions import PatroniFatalException
|
||||
from ..utils import deep_compare, parse_bool, uri
|
||||
|
||||
slot_name_re = re.compile('^[a-z0-9_]{1,63}$')
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def slot_name_from_member_name(member_name):
|
||||
"""Translate member name to valid PostgreSQL slot name.
|
||||
|
||||
PostgreSQL replication slot names must be valid PostgreSQL names. This function maps the wider space of
|
||||
member names to valid PostgreSQL names. Names are lowercased, dashes and periods common in hostnames
|
||||
are replaced with underscores, other characters are encoded as their unicode codepoint. Name is truncated
|
||||
to 64 characters. Multiple different member names may map to a single slot name."""
|
||||
|
||||
def replace_char(match):
|
||||
c = match.group(0)
|
||||
return '_' if c in '-.' else "u{:04d}".format(ord(c))
|
||||
|
||||
slot_name = re.sub('[^a-z0-9_]', replace_char, member_name.lower())
|
||||
return slot_name[0:63]
|
||||
|
||||
|
||||
def parse_connection_string(value):
|
||||
"""Original Governor stores connection strings for each cluster members if a following format:
|
||||
@@ -30,25 +57,58 @@ def parse_connection_string(value):
|
||||
return conn_url, api_url
|
||||
|
||||
|
||||
def dcs_modules():
|
||||
"""Get names of DCS modules, depending on execution environment. If being packaged with PyInstaller,
|
||||
modules aren't discoverable dynamically by scanning source directory because `FrozenImporter` doesn't
|
||||
implement `iter_modules` method. But it is still possible to find all potential DCS modules by
|
||||
iterating through `toc`, which contains list of all "frozen" resources."""
|
||||
|
||||
dcs_dirname = os.path.dirname(__file__)
|
||||
module_prefix = __package__ + '.'
|
||||
|
||||
if getattr(sys, 'frozen', False):
|
||||
toc = set()
|
||||
# dcs_dirname may contain a dot, which causes pkgutil.iter_importers()
|
||||
# to misinterpret the path as a package name. This can be avoided
|
||||
# altogether by not passing a path at all, because PyInstaller's
|
||||
# FrozenImporter is a singleton and registered as top-level finder.
|
||||
for importer in pkgutil.iter_importers():
|
||||
if hasattr(importer, 'toc'):
|
||||
toc |= importer.toc
|
||||
return [module for module in toc if module.startswith(module_prefix) and module.count('.') == 2]
|
||||
else:
|
||||
return [module_prefix + name for _, name, is_pkg in pkgutil.iter_modules([dcs_dirname]) if not is_pkg]
|
||||
|
||||
|
||||
def get_dcs(config):
|
||||
available_implementations = set()
|
||||
for module in os.listdir(os.path.dirname(__file__)):
|
||||
if module.endswith('.py') and not module.startswith('__'): # find module
|
||||
module_name = module[:-3].lower()
|
||||
module = importlib.import_module(__package__ + '.' + module[:-3])
|
||||
for name in filter(lambda name: not name.startswith('__'), dir(module)): # iterate through module content
|
||||
value = getattr(module, name)
|
||||
name = name.lower()
|
||||
# try to find implementation of AbstractDCS interface, class name must match with module_name
|
||||
if inspect.isclass(value) and issubclass(value, AbstractDCS) and name == module_name:
|
||||
available_implementations.add(name)
|
||||
if name in config: # which has configuration section in the config file
|
||||
modules = dcs_modules()
|
||||
|
||||
for module_name in modules:
|
||||
name = module_name.split('.')[-1]
|
||||
if name in config: # we will try to import only modules which have configuration section in the config file
|
||||
try:
|
||||
module = importlib.import_module(module_name)
|
||||
for key, item in module.__dict__.items(): # iterate through the module content
|
||||
# try to find implementation of AbstractDCS interface, class name must match with module_name
|
||||
if key.lower() == name and inspect.isclass(item) and issubclass(item, AbstractDCS):
|
||||
# propagate some parameters
|
||||
config[name].update({p: config[p] for p in ('namespace', 'name',
|
||||
'scope', 'ttl', 'retry_timeout') if p in config})
|
||||
return value(config[name])
|
||||
raise PatroniException("""Can not find suitable configuration of distributed configuration store
|
||||
Available implementations: """ + ', '.join(available_implementations))
|
||||
config[name].update({p: config[p] for p in ('namespace', 'name', 'scope', 'loop_wait',
|
||||
'patronictl', 'ttl', 'retry_timeout') if p in config})
|
||||
return item(config[name])
|
||||
except ImportError:
|
||||
logger.debug('Failed to import %s', module_name)
|
||||
|
||||
available_implementations = []
|
||||
for module_name in modules:
|
||||
name = module_name.split('.')[-1]
|
||||
try:
|
||||
module = importlib.import_module(module_name)
|
||||
available_implementations.extend(name for key, item in module.__dict__.items() if key.lower() == name
|
||||
and inspect.isclass(item) and issubclass(item, AbstractDCS))
|
||||
except ImportError:
|
||||
logger.info('Failed to import %s', module_name)
|
||||
raise PatroniFatalException("""Can not find suitable configuration of distributed configuration store
|
||||
Available implementations: """ + ', '.join(sorted(set(available_implementations))))
|
||||
|
||||
|
||||
class Member(namedtuple('Member', 'index,name,session,data')):
|
||||
@@ -78,13 +138,52 @@ class Member(namedtuple('Member', 'index,name,session,data')):
|
||||
else:
|
||||
try:
|
||||
data = json.loads(data)
|
||||
if not isinstance(data, dict):
|
||||
data = {}
|
||||
except (TypeError, ValueError):
|
||||
data = {}
|
||||
return Member(index, name, session, data)
|
||||
|
||||
@property
|
||||
def conn_url(self):
|
||||
return self.data.get('conn_url')
|
||||
conn_url = self.data.get('conn_url')
|
||||
if conn_url:
|
||||
return conn_url
|
||||
|
||||
conn_kwargs = self.data.get('conn_kwargs')
|
||||
if conn_kwargs:
|
||||
conn_url = uri('postgresql', (conn_kwargs.get('host'), conn_kwargs.get('port', 5432)))
|
||||
self.data['conn_url'] = conn_url
|
||||
return conn_url
|
||||
|
||||
def conn_kwargs(self, auth=None):
|
||||
defaults = {
|
||||
"host": None,
|
||||
"port": None,
|
||||
"dbname": None
|
||||
}
|
||||
ret = self.data.get('conn_kwargs')
|
||||
if ret:
|
||||
defaults.update(ret)
|
||||
ret = defaults
|
||||
else:
|
||||
conn_url = self.conn_url
|
||||
if not conn_url:
|
||||
return {} # due to the invalid conn_url we don't care about authentication parameters
|
||||
r = urlparse(conn_url)
|
||||
ret = {
|
||||
'host': r.hostname,
|
||||
'port': r.port or 5432,
|
||||
'dbname': r.path[1:]
|
||||
}
|
||||
self.data['conn_kwargs'] = ret.copy()
|
||||
|
||||
# apply any remaining authentication parameters
|
||||
if auth and isinstance(auth, dict):
|
||||
ret.update({k: v for k, v in auth.items() if v is not None})
|
||||
if 'username' in auth:
|
||||
ret['user'] = ret.pop('username')
|
||||
return ret
|
||||
|
||||
@property
|
||||
def api_url(self):
|
||||
@@ -104,7 +203,44 @@ class Member(namedtuple('Member', 'index,name,session,data')):
|
||||
|
||||
@property
|
||||
def clonefrom(self):
|
||||
return self.tags.get('clonefrom', False)
|
||||
return self.tags.get('clonefrom', False) and bool(self.conn_url)
|
||||
|
||||
@property
|
||||
def state(self):
|
||||
return self.data.get('state', 'unknown')
|
||||
|
||||
@property
|
||||
def is_running(self):
|
||||
return self.state == 'running'
|
||||
|
||||
@property
|
||||
def version(self):
|
||||
version = self.data.get('version')
|
||||
if version:
|
||||
try:
|
||||
return tuple(map(int, version.split('.')))
|
||||
except Exception:
|
||||
logger.debug('Failed to parse Patroni version %s', version)
|
||||
|
||||
|
||||
class RemoteMember(Member):
|
||||
""" Represents a remote master for a standby cluster
|
||||
"""
|
||||
def __new__(cls, name, data):
|
||||
return super(RemoteMember, cls).__new__(cls, None, name, None, data)
|
||||
|
||||
@staticmethod
|
||||
def allowed_keys():
|
||||
return ('primary_slot_name',
|
||||
'create_replica_methods',
|
||||
'restore_command',
|
||||
'archive_cleanup_command',
|
||||
'recovery_min_apply_delay',
|
||||
'no_replication_slot')
|
||||
|
||||
def __getattr__(self, name):
|
||||
if name in RemoteMember.allowed_keys():
|
||||
return self.data.get(name)
|
||||
|
||||
|
||||
class Leader(namedtuple('Leader', 'index,session,member')):
|
||||
@@ -119,50 +255,78 @@ class Leader(namedtuple('Leader', 'index,session,member')):
|
||||
def name(self):
|
||||
return self.member.name
|
||||
|
||||
def conn_kwargs(self, auth=None):
|
||||
return self.member.conn_kwargs(auth)
|
||||
|
||||
@property
|
||||
def conn_url(self):
|
||||
return self.member.conn_url
|
||||
|
||||
@property
|
||||
def data(self):
|
||||
return self.member.data
|
||||
|
||||
@property
|
||||
def timeline(self):
|
||||
return self.data.get('timeline')
|
||||
|
||||
@property
|
||||
def checkpoint_after_promote(self):
|
||||
"""
|
||||
>>> Leader(1, '', Member.from_node(1, '', '', '{"version":"z"}')).checkpoint_after_promote
|
||||
"""
|
||||
version = self.member.version
|
||||
# 1.5.6 is the last version which doesn't expose checkpoint_after_promote: false
|
||||
if version and version > (1, 5, 6):
|
||||
return self.data.get('role') == 'master' and 'checkpoint_after_promote' not in self.data
|
||||
|
||||
|
||||
class Failover(namedtuple('Failover', 'index,leader,candidate,scheduled_at')):
|
||||
|
||||
"""
|
||||
>>> 'Failover' in str(Failover.from_node(1, '{"leader": "cluster_leader"}'))
|
||||
True
|
||||
>>> 'Failover' in str(Failover.from_node(1, {"leader": "cluster_leader"}))
|
||||
True
|
||||
>>> 'Failover' in str(Failover.from_node(1, '{"leader": "cluster_leader", "member": "cluster_candidate"}'))
|
||||
True
|
||||
>>> Failover.from_node(1, 'null') is None
|
||||
True
|
||||
False
|
||||
>>> n = '{"leader": "cluster_leader", "member": "cluster_candidate", "scheduled_at": "2016-01-14T10:09:57.1394Z"}'
|
||||
>>> 'tzinfo=' in str(Failover.from_node(1, n))
|
||||
True
|
||||
>>> Failover.from_node(1, None) is None
|
||||
True
|
||||
False
|
||||
>>> Failover.from_node(1, '{}') is None
|
||||
True
|
||||
False
|
||||
>>> 'abc' in Failover.from_node(1, 'abc:def')
|
||||
True
|
||||
"""
|
||||
@staticmethod
|
||||
def from_node(index, value):
|
||||
if not value:
|
||||
return None
|
||||
|
||||
try:
|
||||
data = json.loads(value)
|
||||
if not data:
|
||||
return None
|
||||
except ValueError:
|
||||
t = [a.strip() for a in value.split(':')]
|
||||
leader = t[0]
|
||||
candidate = t[1] if len(t) > 1 else None
|
||||
return Failover(index, leader, candidate, None) if leader or candidate else None
|
||||
if isinstance(value, dict):
|
||||
data = value
|
||||
elif value:
|
||||
try:
|
||||
data = json.loads(value)
|
||||
if not isinstance(data, dict):
|
||||
data = {}
|
||||
except ValueError:
|
||||
t = [a.strip() for a in value.split(':')]
|
||||
leader = t[0]
|
||||
candidate = t[1] if len(t) > 1 else None
|
||||
return Failover(index, leader, candidate, None) if leader or candidate else None
|
||||
else:
|
||||
data = {}
|
||||
|
||||
if data.get('scheduled_at'):
|
||||
data['scheduled_at'] = dateutil.parser.parse(data['scheduled_at'])
|
||||
|
||||
return Failover(index, data.get('leader'), data.get('member'), data.get('scheduled_at'))
|
||||
|
||||
def __len__(self):
|
||||
return int(bool(self.leader)) + int(bool(self.candidate))
|
||||
|
||||
|
||||
class ClusterConfig(namedtuple('ClusterConfig', 'index,data,modify_index')):
|
||||
|
||||
@@ -170,30 +334,138 @@ class ClusterConfig(namedtuple('ClusterConfig', 'index,data,modify_index')):
|
||||
def from_node(index, data, modify_index=None):
|
||||
"""
|
||||
>>> ClusterConfig.from_node(1, '{') is None
|
||||
True
|
||||
False
|
||||
"""
|
||||
|
||||
try:
|
||||
data = json.loads(data)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return ClusterConfig(index, data, modify_index or index)
|
||||
data = None
|
||||
modify_index = 0
|
||||
if not isinstance(data, dict):
|
||||
data = {}
|
||||
return ClusterConfig(index, data, index if modify_index is None else modify_index)
|
||||
|
||||
@property
|
||||
def permanent_slots(self):
|
||||
return isinstance(self.data, dict) and (
|
||||
self.data.get('permanent_replication_slots') or
|
||||
self.data.get('permanent_slots') or self.data.get('slots')
|
||||
) or {}
|
||||
|
||||
@property
|
||||
def ignore_slots_matchers(self):
|
||||
return isinstance(self.data, dict) and self.data.get('ignore_slots') or []
|
||||
|
||||
@property
|
||||
def max_timelines_history(self):
|
||||
return self.data.get('max_timelines_history', 0)
|
||||
|
||||
|
||||
class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_leader_operation,members,failover')):
|
||||
class SyncState(namedtuple('SyncState', 'index,leader,sync_standby')):
|
||||
"""Immutable object (namedtuple) which represents last observed synhcronous replication state
|
||||
|
||||
:param index: modification index of a synchronization key in a Configuration Store
|
||||
:param leader: reference to member that was leader
|
||||
:param sync_standby: synchronous standby list (comma delimited) which are last synchronized to leader
|
||||
"""
|
||||
|
||||
@staticmethod
|
||||
def from_node(index, value):
|
||||
"""
|
||||
>>> SyncState.from_node(1, None).leader is None
|
||||
True
|
||||
>>> SyncState.from_node(1, '{}').leader is None
|
||||
True
|
||||
>>> SyncState.from_node(1, '{').leader is None
|
||||
True
|
||||
>>> SyncState.from_node(1, '[]').leader is None
|
||||
True
|
||||
>>> SyncState.from_node(1, '{"leader": "leader"}').leader == "leader"
|
||||
True
|
||||
>>> SyncState.from_node(1, {"leader": "leader"}).leader == "leader"
|
||||
True
|
||||
"""
|
||||
if isinstance(value, dict):
|
||||
data = value
|
||||
elif value:
|
||||
try:
|
||||
data = json.loads(value)
|
||||
if not isinstance(data, dict):
|
||||
data = {}
|
||||
except (TypeError, ValueError):
|
||||
data = {}
|
||||
else:
|
||||
data = {}
|
||||
return SyncState(index, data.get('leader'), data.get('sync_standby'))
|
||||
|
||||
@property
|
||||
def members(self):
|
||||
""" Returns sync_standby in list """
|
||||
return self.sync_standby and self.sync_standby.split(',') or []
|
||||
|
||||
def matches(self, name):
|
||||
"""
|
||||
Returns if a node name matches one of the nodes in the sync state
|
||||
|
||||
>>> s = SyncState(1, 'foo', 'bar,zoo')
|
||||
>>> s.matches('foo')
|
||||
True
|
||||
>>> s.matches('bar')
|
||||
True
|
||||
>>> s.matches('zoo')
|
||||
True
|
||||
>>> s.matches('baz')
|
||||
False
|
||||
>>> s.matches(None)
|
||||
False
|
||||
>>> SyncState(1, None, None).matches('foo')
|
||||
False
|
||||
"""
|
||||
return name is not None and name in [self.leader] + self.members
|
||||
|
||||
|
||||
class TimelineHistory(namedtuple('TimelineHistory', 'index,value,lines')):
|
||||
"""Object representing timeline history file"""
|
||||
|
||||
@staticmethod
|
||||
def from_node(index, value):
|
||||
"""
|
||||
>>> h = TimelineHistory.from_node(1, 2)
|
||||
>>> h.lines
|
||||
[]
|
||||
"""
|
||||
try:
|
||||
lines = json.loads(value)
|
||||
except (TypeError, ValueError):
|
||||
lines = None
|
||||
if not isinstance(lines, list):
|
||||
lines = []
|
||||
return TimelineHistory(index, value, lines)
|
||||
|
||||
|
||||
class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_lsn,members,failover,sync,history,slots')):
|
||||
|
||||
"""Immutable object (namedtuple) which represents PostgreSQL cluster.
|
||||
Consists of the following fields:
|
||||
:param initialize: shows whether this cluster has initialization key stored in DC or not.
|
||||
:param config: global dynamic configuration, reference to `ClusterConfig` object
|
||||
:param leader: `Leader` object which represents current leader of the cluster
|
||||
:param last_leader_operation: int or long object containing position of last known leader operation.
|
||||
This value is stored in `/optime/leader` key
|
||||
:param last_lsn: int or long object containing position of last known leader LSN.
|
||||
This value is stored in the `/status` key or `/optime/leader` (legacy) key
|
||||
:param members: list of Member object, all PostgreSQL cluster members including leader
|
||||
:param failover: reference to `Failover` object"""
|
||||
:param failover: reference to `Failover` object
|
||||
:param sync: reference to `SyncState` object, last observed synchronous replication state.
|
||||
:param history: reference to `TimelineHistory` object
|
||||
:param slots: state of permanent logical replication slots on the primary in the format: {"slot_name": int}
|
||||
"""
|
||||
|
||||
@property
|
||||
def leader_name(self):
|
||||
return self.leader and self.leader.name
|
||||
|
||||
def is_unlocked(self):
|
||||
return not (self.leader and self.leader.name)
|
||||
return not self.leader_name
|
||||
|
||||
def has_member(self, member_name):
|
||||
return any(m for m in self.members if m.name == member_name)
|
||||
@@ -201,10 +473,160 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_leader_operat
|
||||
def get_member(self, member_name, fallback_to_leader=True):
|
||||
return ([m for m in self.members if m.name == member_name] or [self.leader if fallback_to_leader else None])[0]
|
||||
|
||||
def get_clone_member(self):
|
||||
candidates = [m for m in self.members if m.clonefrom and (not self.leader or m.name != self.leader.name)]
|
||||
def get_clone_member(self, exclude):
|
||||
exclude = [exclude] + [self.leader.name] if self.leader else []
|
||||
candidates = [m for m in self.members if m.clonefrom and m.is_running and m.name not in exclude]
|
||||
return candidates[randint(0, len(candidates) - 1)] if candidates else self.leader
|
||||
|
||||
def check_mode(self, mode):
|
||||
return bool(self.config and parse_bool(self.config.data.get(mode)))
|
||||
|
||||
def is_paused(self):
|
||||
return self.check_mode('pause')
|
||||
|
||||
def is_synchronous_mode(self):
|
||||
return self.check_mode('synchronous_mode')
|
||||
|
||||
@property
|
||||
def __permanent_slots(self):
|
||||
return self.config and self.config.permanent_slots or {}
|
||||
|
||||
@property
|
||||
def __permanent_physical_slots(self):
|
||||
return {name: value for name, value in self.__permanent_slots.items()
|
||||
if not value or isinstance(value, dict) and value.get('type', 'physical') == 'physical'}
|
||||
|
||||
@property
|
||||
def __permanent_logical_slots(self):
|
||||
return {name: value for name, value in self.__permanent_slots.items() if isinstance(value, dict)
|
||||
and value.get('type', 'logical') == 'logical' and value.get('database') and value.get('plugin')}
|
||||
|
||||
@property
|
||||
def use_slots(self):
|
||||
return self.config and (self.config.data.get('postgresql') or {}).get('use_slots', True)
|
||||
|
||||
def get_replication_slots(self, my_name, role, nofailover, major_version, show_error=False):
|
||||
# if the replicatefrom tag is set on the member - we should not create the replication slot for it on
|
||||
# the current master, because that member would replicate from elsewhere. We still create the slot if
|
||||
# the replicatefrom destination member is currently not a member of the cluster (fallback to the
|
||||
# master), or if replicatefrom destination member happens to be the current master
|
||||
use_slots = self.use_slots
|
||||
if role in ('master', 'standby_leader'):
|
||||
slot_members = [m.name for m in self.members if use_slots and m.name != my_name and
|
||||
(m.replicatefrom is None or m.replicatefrom == my_name or
|
||||
not self.has_member(m.replicatefrom))]
|
||||
permanent_slots = self.__permanent_slots if use_slots and \
|
||||
role == 'master' else self.__permanent_physical_slots
|
||||
else:
|
||||
# only manage slots for replicas that replicate from this one, except for the leader among them
|
||||
slot_members = [m.name for m in self.members if use_slots and
|
||||
m.replicatefrom == my_name and m.name != self.leader_name]
|
||||
permanent_slots = self.__permanent_logical_slots if use_slots and not nofailover else {}
|
||||
|
||||
slots = {slot_name_from_member_name(name): {'type': 'physical'} for name in slot_members}
|
||||
|
||||
if len(slots) < len(slot_members):
|
||||
# Find which names are conflicting for a nicer error message
|
||||
slot_conflicts = defaultdict(list)
|
||||
for name in slot_members:
|
||||
slot_conflicts[slot_name_from_member_name(name)].append(name)
|
||||
logger.error("Following cluster members share a replication slot name: %s",
|
||||
"; ".join("{} map to {}".format(", ".join(v), k)
|
||||
for k, v in slot_conflicts.items() if len(v) > 1))
|
||||
|
||||
# "merge" replication slots for members with permanent_replication_slots
|
||||
disabled_permanent_logical_slots = []
|
||||
for name, value in permanent_slots.items():
|
||||
if not slot_name_re.match(name):
|
||||
logger.error("Invalid permanent replication slot name '%s'", name)
|
||||
logger.error("Slot name may only contain lower case letters, numbers, and the underscore chars")
|
||||
continue
|
||||
|
||||
value = deepcopy(value) if value else {'type': 'physical'}
|
||||
if isinstance(value, dict):
|
||||
if 'type' not in value:
|
||||
value['type'] = 'logical' if value.get('database') and value.get('plugin') else 'physical'
|
||||
|
||||
if value['type'] == 'physical':
|
||||
# Don't try to create permanent physical replication slot for yourself
|
||||
if name != slot_name_from_member_name(my_name):
|
||||
slots[name] = value
|
||||
continue
|
||||
elif value['type'] == 'logical' and value.get('database') and value.get('plugin'):
|
||||
if major_version < 110000:
|
||||
disabled_permanent_logical_slots.append(name)
|
||||
elif name in slots:
|
||||
logger.error("Permanent logical replication slot {'%s': %s} is conflicting with" +
|
||||
" physical replication slot for cluster member", name, value)
|
||||
else:
|
||||
slots[name] = value
|
||||
continue
|
||||
|
||||
logger.error("Bad value for slot '%s' in permanent_slots: %s", name, permanent_slots[name])
|
||||
|
||||
if disabled_permanent_logical_slots and show_error:
|
||||
logger.error("Permanent logical replication slots supported by Patroni only starting from PostgreSQL 11. "
|
||||
"Following slots will not be created: %s.", disabled_permanent_logical_slots)
|
||||
|
||||
return slots
|
||||
|
||||
def has_permanent_logical_slots(self, my_name, nofailover, major_version=110000):
|
||||
if major_version < 110000:
|
||||
return False
|
||||
slots = self.get_replication_slots(my_name, 'replica', nofailover, major_version).values()
|
||||
return any(v for v in slots if v.get("type") == "logical")
|
||||
|
||||
def should_enforce_hot_standby_feedback(self, my_name, nofailover, major_version):
|
||||
"""
|
||||
The hot_standby_feedback must be enabled if the current replica has logical slots
|
||||
or it is working as a cascading replica for the other node that has logical slots.
|
||||
"""
|
||||
|
||||
if major_version < 110000:
|
||||
return False
|
||||
|
||||
if self.has_permanent_logical_slots(my_name, nofailover, major_version):
|
||||
return True
|
||||
|
||||
if self.use_slots:
|
||||
members = [m for m in self.members if m.replicatefrom == my_name and m.name != self.leader_name]
|
||||
return any(self.should_enforce_hot_standby_feedback(m.name, m.nofailover, major_version) for m in members)
|
||||
return False
|
||||
|
||||
def get_my_slot_name_on_primary(self, my_name, replicatefrom):
|
||||
"""
|
||||
P <-- I <-- L
|
||||
In case of cascading replication we have to check not our physical slot,
|
||||
but slot of the replica that connects us to the primary.
|
||||
"""
|
||||
|
||||
m = self.get_member(replicatefrom, False) if replicatefrom else None
|
||||
return self.get_my_slot_name_on_primary(m.name, m.replicatefrom) if m else slot_name_from_member_name(my_name)
|
||||
|
||||
@property
|
||||
def timeline(self):
|
||||
"""
|
||||
>>> Cluster(0, 0, 0, 0, 0, 0, 0, 0, 0).timeline
|
||||
0
|
||||
>>> Cluster(0, 0, 0, 0, 0, 0, 0, TimelineHistory.from_node(1, '[]'), 0).timeline
|
||||
1
|
||||
>>> Cluster(0, 0, 0, 0, 0, 0, 0, TimelineHistory.from_node(1, '[["a"]]'), 0).timeline
|
||||
0
|
||||
"""
|
||||
if self.history:
|
||||
if self.history.lines:
|
||||
try:
|
||||
return int(self.history.lines[-1][0]) + 1
|
||||
except Exception:
|
||||
logger.error('Failed to parse cluster history from DCS: %s', self.history.lines)
|
||||
elif self.history.value == '[]':
|
||||
return 1
|
||||
return 0
|
||||
|
||||
@property
|
||||
def min_version(self):
|
||||
return next(iter(sorted(filter(lambda v: v, [m.version for m in self.members])) + [None]))
|
||||
|
||||
|
||||
@six.add_metaclass(abc.ABCMeta)
|
||||
class AbstractDCS(object):
|
||||
@@ -213,9 +635,12 @@ class AbstractDCS(object):
|
||||
_CONFIG = 'config'
|
||||
_LEADER = 'leader'
|
||||
_FAILOVER = 'failover'
|
||||
_HISTORY = 'history'
|
||||
_MEMBERS = 'members/'
|
||||
_OPTIME = 'optime'
|
||||
_LEADER_OPTIME = _OPTIME + '/' + _LEADER
|
||||
_STATUS = 'status' # JSON, contains "leader_lsn" and confirmed_flush_lsn of logical "slots" on the leader
|
||||
_LEADER_OPTIME = _OPTIME + '/' + _LEADER # legacy
|
||||
_SYNC = 'sync'
|
||||
|
||||
def __init__(self, config):
|
||||
"""
|
||||
@@ -223,11 +648,16 @@ class AbstractDCS(object):
|
||||
i.e.: `zookeeper` for zookeeper, `etcd` for etcd, etc...
|
||||
"""
|
||||
self._name = config['name']
|
||||
self._namespace = '/{0}'.format(config.get('namespace', '/service/').strip('/'))
|
||||
self._base_path = '/'.join([self._namespace, config['scope']])
|
||||
self._base_path = re.sub('/+', '/', '/'.join(['', config.get('namespace', 'service'), config['scope']]))
|
||||
self._set_loop_wait(config.get('loop_wait', 10))
|
||||
|
||||
self._ctl = bool(config.get('patronictl', False))
|
||||
self._cluster = None
|
||||
self._cluster_valid_till = 0
|
||||
self._cluster_thread_lock = Lock()
|
||||
self._last_lsn = ''
|
||||
self._last_seen = 0
|
||||
self._last_status = {}
|
||||
self.event = Event()
|
||||
|
||||
def client_path(self, path):
|
||||
@@ -257,18 +687,50 @@ class AbstractDCS(object):
|
||||
def failover_path(self):
|
||||
return self.client_path(self._FAILOVER)
|
||||
|
||||
@property
|
||||
def history_path(self):
|
||||
return self.client_path(self._HISTORY)
|
||||
|
||||
@property
|
||||
def status_path(self):
|
||||
return self.client_path(self._STATUS)
|
||||
|
||||
@property
|
||||
def leader_optime_path(self):
|
||||
return self.client_path(self._LEADER_OPTIME)
|
||||
|
||||
@property
|
||||
def sync_path(self):
|
||||
return self.client_path(self._SYNC)
|
||||
|
||||
@abc.abstractmethod
|
||||
def set_ttl(self, ttl):
|
||||
"""Set the new ttl value for leader key"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def ttl(self):
|
||||
"""Get new ttl value"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def set_retry_timeout(self, retry_timeout):
|
||||
"""Set the new value for retry_timeout"""
|
||||
|
||||
def _set_loop_wait(self, loop_wait):
|
||||
self._loop_wait = loop_wait
|
||||
|
||||
def reload_config(self, config):
|
||||
self._set_loop_wait(config['loop_wait'])
|
||||
self.set_ttl(config['ttl'])
|
||||
self.set_retry_timeout(config['retry_timeout'])
|
||||
|
||||
@property
|
||||
def loop_wait(self):
|
||||
return self._loop_wait
|
||||
|
||||
@property
|
||||
def last_seen(self):
|
||||
return self._last_seen
|
||||
|
||||
@abc.abstractmethod
|
||||
def _load_cluster(self):
|
||||
"""Internally this method should build `Cluster` object which
|
||||
@@ -279,31 +741,63 @@ class AbstractDCS(object):
|
||||
If the current node was running as a master and exception raised,
|
||||
instance would be demoted."""
|
||||
|
||||
def get_cluster(self):
|
||||
def _bypass_caches(self):
|
||||
"""Used only in zookeeper"""
|
||||
|
||||
def get_cluster(self, force=False):
|
||||
if force:
|
||||
self._bypass_caches()
|
||||
try:
|
||||
cluster = self._load_cluster()
|
||||
except Exception:
|
||||
self.reset_cluster()
|
||||
raise
|
||||
|
||||
self._last_seen = int(time.time())
|
||||
|
||||
with self._cluster_thread_lock:
|
||||
try:
|
||||
self._load_cluster()
|
||||
except:
|
||||
self._cluster = None
|
||||
raise
|
||||
return self._cluster
|
||||
self._cluster = cluster
|
||||
self._cluster_valid_till = time.time() + self.ttl
|
||||
return cluster
|
||||
|
||||
@property
|
||||
def cluster(self):
|
||||
with self._cluster_thread_lock:
|
||||
return self._cluster
|
||||
return self._cluster if self._cluster_valid_till > time.time() else None
|
||||
|
||||
def reset_cluster(self):
|
||||
with self._cluster_thread_lock:
|
||||
self._cluster = None
|
||||
self._cluster_valid_till = 0
|
||||
|
||||
@abc.abstractmethod
|
||||
def write_leader_optime(self, last_operation):
|
||||
"""write current xlog location into `/optime/leader` key in DCS
|
||||
:param last_operation: absolute xlog location in bytes"""
|
||||
def _write_leader_optime(self, last_lsn):
|
||||
"""write current WAL LSN into `/optime/leader` key in DCS
|
||||
|
||||
:param last_lsn: absolute WAL LSN in bytes
|
||||
:returns: `!True` on success."""
|
||||
|
||||
def write_leader_optime(self, last_lsn):
|
||||
self.write_status({self._OPTIME: last_lsn})
|
||||
|
||||
@abc.abstractmethod
|
||||
def update_leader(self):
|
||||
def _write_status(self, value):
|
||||
"""write current WAL LSN and confirmed_flush_lsn of permanent slots into the `/status` key in DCS
|
||||
|
||||
:param value: status serialized in JSON forman
|
||||
:returns: `!True` on success."""
|
||||
|
||||
def write_status(self, value):
|
||||
if not deep_compare(self._last_status, value) and self._write_status(json.dumps(value, separators=(',', ':'))):
|
||||
self._last_status = value
|
||||
cluster = self.cluster
|
||||
min_version = cluster and cluster.min_version
|
||||
if min_version and min_version < (2, 1, 0) and self._last_lsn != value[self._OPTIME]:
|
||||
self._last_lsn = value[self._OPTIME]
|
||||
self._write_leader_optime(str(value[self._OPTIME]))
|
||||
|
||||
@abc.abstractmethod
|
||||
def _update_leader(self):
|
||||
"""Update leader key (or session) ttl
|
||||
|
||||
:returns: `!True` if leader key (or session) has been updated successfully.
|
||||
@@ -312,10 +806,28 @@ class AbstractDCS(object):
|
||||
You have to use CAS (Compare And Swap) operation in order to update leader key,
|
||||
for example for etcd `prevValue` parameter must be used."""
|
||||
|
||||
def update_leader(self, last_lsn, slots=None):
|
||||
"""Update leader key (or session) ttl and optime/leader
|
||||
|
||||
:param last_lsn: absolute WAL LSN in bytes
|
||||
:param slots: dict with permanent slots confirmed_flush_lsn
|
||||
:returns: `!True` if leader key (or session) has been updated successfully.
|
||||
If not, `!False` must be returned and current instance would be demoted."""
|
||||
|
||||
ret = self._update_leader()
|
||||
if ret and last_lsn:
|
||||
status = {self._OPTIME: last_lsn}
|
||||
if slots:
|
||||
status['slots'] = slots
|
||||
self.write_status(status)
|
||||
return ret
|
||||
|
||||
@abc.abstractmethod
|
||||
def attempt_to_acquire_leader(self):
|
||||
def attempt_to_acquire_leader(self, permanent=False):
|
||||
"""Attempt to acquire leader lock
|
||||
This method should create `/leader` key with value=`~self._name`
|
||||
:param permanent: if set to `!True`, the leader key will never expire.
|
||||
Used in patronictl for the external master
|
||||
:returns: `!True` if key has been created successfully.
|
||||
|
||||
Key must be created atomically. In case if key already exists it should not be
|
||||
@@ -335,7 +847,6 @@ class AbstractDCS(object):
|
||||
|
||||
if scheduled_at:
|
||||
failover_value['scheduled_at'] = scheduled_at.isoformat()
|
||||
|
||||
return self.set_failover_value(json.dumps(failover_value, separators=(',', ':')), index)
|
||||
|
||||
@abc.abstractmethod
|
||||
@@ -343,13 +854,15 @@ class AbstractDCS(object):
|
||||
"""Create or update `/config` key"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def touch_member(self, data, ttl=None):
|
||||
def touch_member(self, data, permanent=False):
|
||||
"""Update member key in DCS.
|
||||
This method should create or update key with the name = '/members/' + `~self._name`
|
||||
and value = data in a given DCS.
|
||||
|
||||
:param data: json serialized information about instance (including connection strings)
|
||||
:param data: information about instance (including connection strings)
|
||||
:param ttl: ttl for member key, optional parameter. If it is None `~self.member_ttl will be used`
|
||||
:param permanent: if set to `!True`, the member key will never expire.
|
||||
Used in patronictl for the external master.
|
||||
:returns: `!True` on success otherwise `!False`
|
||||
"""
|
||||
|
||||
@@ -371,10 +884,19 @@ class AbstractDCS(object):
|
||||
otherwise it should return `!False`"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def delete_leader(self):
|
||||
"""Voluntarily remove leader key from DCS
|
||||
def _delete_leader(self):
|
||||
"""Remove leader key from DCS.
|
||||
This method should remove leader key if current instance is the leader"""
|
||||
|
||||
def delete_leader(self, last_lsn=None):
|
||||
"""Update optime/leader and voluntarily remove leader key from DCS.
|
||||
This method should remove leader key if current instance is the leader.
|
||||
:param last_lsn: latest checkpoint location in bytes"""
|
||||
|
||||
if last_lsn:
|
||||
self.write_status({self._OPTIME: last_lsn})
|
||||
return self._delete_leader()
|
||||
|
||||
@abc.abstractmethod
|
||||
def cancel_initialization(self):
|
||||
""" Removes the initialize key for a cluster """
|
||||
@@ -383,12 +905,36 @@ class AbstractDCS(object):
|
||||
def delete_cluster(self):
|
||||
"""Delete cluster from DCS"""
|
||||
|
||||
def watch(self, timeout):
|
||||
@staticmethod
|
||||
def sync_state(leader, sync_standby):
|
||||
"""Build sync_state dict
|
||||
sync_standby dictionary key being kept for backward compatibility
|
||||
"""
|
||||
return {'leader': leader, 'sync_standby': sync_standby and ','.join(sorted(sync_standby)) or None}
|
||||
|
||||
def write_sync_state(self, leader, sync_standby, index=None):
|
||||
sync_value = self.sync_state(leader, sync_standby)
|
||||
return self.set_sync_state_value(json.dumps(sync_value, separators=(',', ':')), index)
|
||||
|
||||
@abc.abstractmethod
|
||||
def set_history_value(self, value):
|
||||
""""""
|
||||
|
||||
@abc.abstractmethod
|
||||
def set_sync_state_value(self, value, index=None):
|
||||
""""""
|
||||
|
||||
@abc.abstractmethod
|
||||
def delete_sync_state(self, index=None):
|
||||
""""""
|
||||
|
||||
def watch(self, leader_index, timeout):
|
||||
"""If the current node is a master it should just sleep.
|
||||
Any other node should watch for changes of leader key with a given timeout
|
||||
|
||||
:param leader_index: index of a leader key
|
||||
:param timeout: timeout in seconds
|
||||
:returns: `!True` if you would like to reschedule the next run of ha cycle"""
|
||||
|
||||
self.event.wait(timeout)
|
||||
return self.event.isSet()
|
||||
return self.event.is_set()
|
||||
|
||||
+437
-99
@@ -1,14 +1,22 @@
|
||||
from __future__ import absolute_import
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import socket
|
||||
import ssl
|
||||
import time
|
||||
import six
|
||||
import urllib3
|
||||
|
||||
from consul import ConsulException, NotFound, base, std
|
||||
from patroni.dcs import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member
|
||||
from patroni.exceptions import DCSError
|
||||
from patroni.utils import sleep
|
||||
from requests.exceptions import RequestException
|
||||
from collections import namedtuple
|
||||
from consul import ConsulException, NotFound, base
|
||||
from urllib3.exceptions import HTTPError
|
||||
from six.moves.urllib.parse import urlencode, urlparse, quote
|
||||
from six.moves.http_client import HTTPException
|
||||
|
||||
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member, SyncState, TimelineHistory
|
||||
from ..exceptions import DCSError
|
||||
from ..utils import deep_compare, parse_bool, Retry, RetryFailedError, split_host_port, uri, USER_AGENT
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -17,70 +25,220 @@ class ConsulError(DCSError):
|
||||
pass
|
||||
|
||||
|
||||
class HTTPClient(std.HTTPClient):
|
||||
class ConsulInternalError(ConsulException):
|
||||
"""An internal Consul server error occurred"""
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super(HTTPClient, self).__init__(*args, **kwargs)
|
||||
|
||||
def patch_default_timeout(self, timeout):
|
||||
# Set a default timeout for the `request.session.request` method, that is used
|
||||
# internally by the methods request.session.get, request.session.post and
|
||||
# others. We monkey-patch here to avoid reimplementing each individual method from
|
||||
# `std.HTTPClient`. By default, the timeout is not set. It means that a new
|
||||
# session may hang almost indefinitely waiting for the server to respond,
|
||||
# which is not what we want in Patroni.
|
||||
class InvalidSessionTTL(ConsulException):
|
||||
"""Session TTL is too small or too big"""
|
||||
|
||||
request_func = getattr(self.session.request, '__func__' if six.PY3 else 'im_func')
|
||||
defaults_attr_name = '__defaults__' if six.PY3 else 'func_defaults'
|
||||
defaults = list(getattr(request_func, defaults_attr_name))
|
||||
code = request_func.__code__ if six.PY3 else request_func.func_code
|
||||
defaults[code.co_varnames[code.co_argcount - len(defaults):code.co_argcount].index('timeout')] = timeout
|
||||
setattr(request_func, defaults_attr_name, tuple(defaults)) # monkeypatching
|
||||
|
||||
def get(self, callback, path, params=None):
|
||||
# The get function is overridden to handle a special case of it being called
|
||||
# with an index and wait parameters. That form indicates that a user needs to
|
||||
# wait for the given key to change its value, with a wait timeout supplied. We
|
||||
# don't want our monkey-patched timeout to be less than the value of the wait
|
||||
# parameter, therefore, we set it to either the value of wait or a default of 5 minutes.
|
||||
class InvalidSession(ConsulException):
|
||||
"""invalid session"""
|
||||
|
||||
if isinstance(params, dict) and 'index' in params:
|
||||
timeout = (float(params['wait'][:-1]) if 'wait' in params else 300) + 1
|
||||
else:
|
||||
timeout = None
|
||||
return callback(self.response(self.session.get(self.uri(path, params), verify=self.verify, timeout=timeout)))
|
||||
|
||||
Response = namedtuple('Response', 'code,headers,body,content')
|
||||
|
||||
|
||||
class HTTPClient(object):
|
||||
|
||||
def __init__(self, host='127.0.0.1', port=8500, token=None, scheme='http', verify=True, cert=None, ca_cert=None):
|
||||
self.token = token
|
||||
self._read_timeout = 10
|
||||
self.base_uri = uri(scheme, (host, port))
|
||||
kwargs = {}
|
||||
if cert:
|
||||
if isinstance(cert, tuple):
|
||||
# Key and cert are separate
|
||||
kwargs['cert_file'] = cert[0]
|
||||
kwargs['key_file'] = cert[1]
|
||||
else:
|
||||
# combined certificate
|
||||
kwargs['cert_file'] = cert
|
||||
if ca_cert:
|
||||
kwargs['ca_certs'] = ca_cert
|
||||
kwargs['cert_reqs'] = ssl.CERT_REQUIRED if verify or ca_cert else ssl.CERT_NONE
|
||||
self.http = urllib3.PoolManager(num_pools=10, maxsize=10, **kwargs)
|
||||
self._ttl = None
|
||||
|
||||
def set_read_timeout(self, timeout):
|
||||
self._read_timeout = timeout/3.0
|
||||
|
||||
@property
|
||||
def ttl(self):
|
||||
return self._ttl
|
||||
|
||||
def set_ttl(self, ttl):
|
||||
ret = self._ttl != ttl
|
||||
self._ttl = ttl
|
||||
return ret
|
||||
|
||||
@staticmethod
|
||||
def response(response):
|
||||
content = response.data
|
||||
body = content.decode('utf-8')
|
||||
if response.status == 500:
|
||||
msg = '{0} {1}'.format(response.status, body)
|
||||
if body.startswith('Invalid Session TTL'):
|
||||
raise InvalidSessionTTL(msg)
|
||||
elif body.startswith('invalid session'):
|
||||
raise InvalidSession(msg)
|
||||
else:
|
||||
raise ConsulInternalError(msg)
|
||||
return Response(response.status, response.headers, body, content)
|
||||
|
||||
def uri(self, path, params=None):
|
||||
return '{0}{1}{2}'.format(self.base_uri, path, params and '?' + urlencode(params) or '')
|
||||
|
||||
def __getattr__(self, method):
|
||||
if method not in ('get', 'post', 'put', 'delete'):
|
||||
raise AttributeError("HTTPClient instance has no attribute '{0}'".format(method))
|
||||
|
||||
def wrapper(callback, path, params=None, data='', headers=None):
|
||||
# python-consul doesn't allow to specify ttl smaller then 10 seconds
|
||||
# because session_ttl_min defaults to 10s, so we have to do this ugly dirty hack...
|
||||
if method == 'put' and path == '/v1/session/create':
|
||||
ttl = '"ttl": "{0}s"'.format(self._ttl)
|
||||
if not data or data == '{}':
|
||||
data = '{' + ttl + '}'
|
||||
else:
|
||||
data = data[:-1] + ', ' + ttl + '}'
|
||||
if isinstance(params, list): # starting from v1.1.0 python-consul switched from `dict` to `list` for params
|
||||
params = {k: v for k, v in params}
|
||||
kwargs = {'retries': 0, 'preload_content': False, 'body': data}
|
||||
if method == 'get' and isinstance(params, dict) and 'index' in params:
|
||||
timeout = float(params['wait'][:-1]) if 'wait' in params else 300
|
||||
# According to the documentation a small random amount of additional wait time is added to the
|
||||
# supplied maximum wait time to spread out the wake up time of any concurrent requests. This adds
|
||||
# up to wait / 16 additional time to the maximum duration. Since our goal is actually getting a
|
||||
# response rather read timeout we will add to the timeout a slightly bigger value.
|
||||
kwargs['timeout'] = timeout + max(timeout/15.0, 1)
|
||||
else:
|
||||
kwargs['timeout'] = self._read_timeout
|
||||
kwargs['headers'] = (headers or {}).copy()
|
||||
kwargs['headers'].update(urllib3.make_headers(user_agent=USER_AGENT))
|
||||
token = params.pop('token', self.token) if isinstance(params, dict) else self.token
|
||||
if token:
|
||||
kwargs['headers']['X-Consul-Token'] = token
|
||||
return callback(self.response(self.http.request(method.upper(), self.uri(path, params), **kwargs)))
|
||||
return wrapper
|
||||
|
||||
|
||||
class ConsulClient(base.Consul):
|
||||
|
||||
@staticmethod
|
||||
def connect(host, port, scheme, verify=True):
|
||||
return HTTPClient(host, port, scheme, verify)
|
||||
def __init__(self, *args, **kwargs):
|
||||
self._cert = kwargs.pop('cert', None)
|
||||
self._ca_cert = kwargs.pop('ca_cert', None)
|
||||
self.token = kwargs.get('token')
|
||||
super(ConsulClient, self).__init__(*args, **kwargs)
|
||||
|
||||
def http_connect(self, *args, **kwargs):
|
||||
kwargs.update(dict(zip(['host', 'port', 'scheme', 'verify'], args)))
|
||||
if self._cert:
|
||||
kwargs['cert'] = self._cert
|
||||
if self._ca_cert:
|
||||
kwargs['ca_cert'] = self._ca_cert
|
||||
if self.token:
|
||||
kwargs['token'] = self.token
|
||||
return HTTPClient(**kwargs)
|
||||
|
||||
def connect(self, *args, **kwargs):
|
||||
return self.http_connect(*args, **kwargs)
|
||||
|
||||
def reload_config(self, config):
|
||||
self.http.token = self.token = config.get('token')
|
||||
self.consistency = config.get('consistency', 'default')
|
||||
self.dc = config.get('dc')
|
||||
|
||||
|
||||
def catch_consul_errors(func):
|
||||
def wrapper(*args, **kwargs):
|
||||
try:
|
||||
return func(*args, **kwargs)
|
||||
except (ConsulException, RequestException):
|
||||
except (RetryFailedError, ConsulException, HTTPException, HTTPError, socket.error, socket.timeout):
|
||||
return False
|
||||
return wrapper
|
||||
|
||||
|
||||
def force_if_last_failed(func):
|
||||
def wrapper(*args, **kwargs):
|
||||
if wrapper.last_result is False:
|
||||
kwargs['force'] = True
|
||||
wrapper.last_result = func(*args, **kwargs)
|
||||
return wrapper.last_result
|
||||
|
||||
wrapper.last_result = None
|
||||
return wrapper
|
||||
|
||||
|
||||
def service_name_from_scope_name(scope_name):
|
||||
"""Translate scope name to service name which can be used in dns.
|
||||
|
||||
230 = 253 - len('replica.') - len('.service.consul')
|
||||
"""
|
||||
|
||||
def replace_char(match):
|
||||
c = match.group(0)
|
||||
return '-' if c in '. _' else "u{:04d}".format(ord(c))
|
||||
|
||||
service_name = re.sub(r'[^a-z0-9\-]', replace_char, scope_name.lower())
|
||||
return service_name[0:230]
|
||||
|
||||
|
||||
class Consul(AbstractDCS):
|
||||
|
||||
def __init__(self, config):
|
||||
super(Consul, self).__init__(config)
|
||||
self._ttl = None
|
||||
self._session = None
|
||||
self._my_member_data = None
|
||||
self.set_ttl(config.get('ttl') or 30)
|
||||
host, port = config.get('host', '127.0.0.1:8500').split(':')
|
||||
self._client = ConsulClient(host=host, port=port)
|
||||
self._client.http.patch_default_timeout(config['retry_timeout']/2.0)
|
||||
self._scope = config['scope']
|
||||
self.create_session()
|
||||
self._session = None
|
||||
self.__do_not_watch = False
|
||||
self._retry = Retry(deadline=config['retry_timeout'], max_delay=1, max_tries=-1,
|
||||
retry_exceptions=(ConsulInternalError, HTTPException,
|
||||
HTTPError, socket.error, socket.timeout))
|
||||
|
||||
kwargs = {}
|
||||
if 'url' in config:
|
||||
r = urlparse(config['url'])
|
||||
config.update({'scheme': r.scheme, 'host': r.hostname, 'port': r.port or 8500})
|
||||
elif 'host' in config:
|
||||
host, port = split_host_port(config.get('host', '127.0.0.1:8500'), 8500)
|
||||
config['host'] = host
|
||||
if 'port' not in config:
|
||||
config['port'] = int(port)
|
||||
|
||||
if config.get('cacert'):
|
||||
config['ca_cert'] = config.pop('cacert')
|
||||
|
||||
if config.get('key') and config.get('cert'):
|
||||
config['cert'] = (config['cert'], config['key'])
|
||||
|
||||
config_keys = ('host', 'port', 'token', 'scheme', 'cert', 'ca_cert', 'dc', 'consistency')
|
||||
kwargs = {p: config.get(p) for p in config_keys if config.get(p)}
|
||||
|
||||
verify = config.get('verify')
|
||||
if not isinstance(verify, bool):
|
||||
verify = parse_bool(verify)
|
||||
if isinstance(verify, bool):
|
||||
kwargs['verify'] = verify
|
||||
|
||||
self._client = ConsulClient(**kwargs)
|
||||
self.set_retry_timeout(config['retry_timeout'])
|
||||
self.set_ttl(config.get('ttl') or 30)
|
||||
self._last_session_refresh = 0
|
||||
self.__session_checks = config.get('checks', [])
|
||||
self._register_service = config.get('register_service', False)
|
||||
self._previous_loop_register_service = self._register_service
|
||||
self._service_tags = sorted(config.get('service_tags', []))
|
||||
self._previous_loop_service_tags = self._service_tags
|
||||
if self._register_service:
|
||||
self._set_service_name()
|
||||
self._service_check_interval = config.get('service_check_interval', '5s')
|
||||
self._service_check_tls_server_name = config.get('service_check_tls_server_name', None)
|
||||
if not self._ctl:
|
||||
self.create_session()
|
||||
|
||||
def retry(self, *args, **kwargs):
|
||||
return self._retry.copy()(*args, **kwargs)
|
||||
|
||||
def create_session(self):
|
||||
while not self._session:
|
||||
@@ -88,34 +246,75 @@ class Consul(AbstractDCS):
|
||||
self.refresh_session()
|
||||
except ConsulError:
|
||||
logger.info('waiting on consul')
|
||||
sleep(5)
|
||||
time.sleep(5)
|
||||
|
||||
def reload_config(self, config):
|
||||
super(Consul, self).reload_config(config)
|
||||
|
||||
consul_config = config.get('consul', {})
|
||||
self._client.reload_config(consul_config)
|
||||
self._previous_loop_service_tags = self._service_tags
|
||||
self._service_tags = sorted(consul_config.get('service_tags', []))
|
||||
|
||||
should_register_service = consul_config.get('register_service', False)
|
||||
if should_register_service and not self._register_service:
|
||||
self._set_service_name()
|
||||
|
||||
self._previous_loop_register_service = self._register_service
|
||||
self._register_service = should_register_service
|
||||
|
||||
def set_ttl(self, ttl):
|
||||
ttl = ttl/2.0 # My experiments have shown that session expires after 2*ttl time
|
||||
if self._ttl != ttl:
|
||||
if self._client.http.set_ttl(ttl/2.0): # Consul multiplies the TTL by 2x
|
||||
self._session = None
|
||||
self.__do_not_watch = True
|
||||
self._ttl = ttl
|
||||
|
||||
@property
|
||||
def ttl(self):
|
||||
return self._client.http.ttl
|
||||
|
||||
def set_retry_timeout(self, retry_timeout):
|
||||
self._client.http.patch_default_timeout(retry_timeout/2.0)
|
||||
self._retry.deadline = retry_timeout
|
||||
self._client.http.set_read_timeout(retry_timeout)
|
||||
|
||||
def refresh_session(self):
|
||||
def adjust_ttl(self):
|
||||
try:
|
||||
settings = self._client.agent.self()
|
||||
min_ttl = (settings['Config']['SessionTTLMin'] or 10000000000)/1000000000.0
|
||||
logger.warning('Changing Session TTL from %s to %s', self._client.http.ttl, min_ttl)
|
||||
self._client.http.set_ttl(min_ttl)
|
||||
except Exception:
|
||||
logger.exception('adjust_ttl')
|
||||
|
||||
def _do_refresh_session(self):
|
||||
""":returns: `!True` if it had to create new session"""
|
||||
if self._session and self._last_session_refresh + self._loop_wait > time.time():
|
||||
return False
|
||||
|
||||
if self._session:
|
||||
try:
|
||||
return self._client.session.renew(self._session) is None
|
||||
self._client.session.renew(self._session)
|
||||
except NotFound:
|
||||
self._session = None
|
||||
if not self._session:
|
||||
name = self._scope + '-' + self._name
|
||||
ret = not self._session
|
||||
if ret:
|
||||
try:
|
||||
self._session = self._client.session.create(name=name, lock_delay=0, behavior='delete', ttl=self._ttl)
|
||||
except (ConsulException, RequestException):
|
||||
self._session = self._client.session.create(name=self._scope + '-' + self._name,
|
||||
checks=self.__session_checks,
|
||||
lock_delay=0.001, behavior='delete')
|
||||
except InvalidSessionTTL:
|
||||
logger.exception('session.create')
|
||||
if not self._session:
|
||||
raise ConsulError('Failed to renew/create session')
|
||||
return True
|
||||
self.adjust_ttl()
|
||||
raise
|
||||
|
||||
self._last_session_refresh = time.time()
|
||||
return ret
|
||||
|
||||
def refresh_session(self):
|
||||
try:
|
||||
return self.retry(self._do_refresh_session)
|
||||
except (ConsulException, RetryFailedError):
|
||||
logger.exception('refresh_session')
|
||||
raise ConsulError('Failed to renew/create session')
|
||||
|
||||
def client_path(self, path):
|
||||
return super(Consul, self).client_path(path)[1:]
|
||||
@@ -127,7 +326,7 @@ class Consul(AbstractDCS):
|
||||
def _load_cluster(self):
|
||||
try:
|
||||
path = self.client_path('/')
|
||||
_, results = self._client.kv.get(path, recurse=True)
|
||||
_, results = self.retry(self._client.kv.get, path, recurse=True)
|
||||
|
||||
if results is None:
|
||||
raise NotFound
|
||||
@@ -135,7 +334,7 @@ class Consul(AbstractDCS):
|
||||
nodes = {}
|
||||
for node in results:
|
||||
node['Value'] = (node['Value'] or b'').decode('utf-8')
|
||||
nodes[os.path.relpath(node['Key'], path)] = node
|
||||
nodes[node['Key'][len(path):].lstrip('/')] = node
|
||||
|
||||
# get initialize flag
|
||||
initialize = nodes.get(self._INITIALIZE)
|
||||
@@ -145,16 +344,36 @@ class Consul(AbstractDCS):
|
||||
config = nodes.get(self._CONFIG)
|
||||
config = config and ClusterConfig.from_node(config['ModifyIndex'], config['Value'])
|
||||
|
||||
# get last leader operation
|
||||
last_leader_operation = nodes.get(self._LEADER_OPTIME)
|
||||
last_leader_operation = 0 if last_leader_operation is None else int(last_leader_operation['Value'])
|
||||
# get timeline history
|
||||
history = nodes.get(self._HISTORY)
|
||||
history = history and TimelineHistory.from_node(history['ModifyIndex'], history['Value'])
|
||||
|
||||
# get last known leader lsn and slots
|
||||
status = nodes.get(self._STATUS)
|
||||
if status:
|
||||
try:
|
||||
status = json.loads(status['Value'])
|
||||
last_lsn = status.get(self._OPTIME)
|
||||
slots = status.get('slots')
|
||||
except Exception:
|
||||
slots = last_lsn = None
|
||||
else:
|
||||
last_lsn = nodes.get(self._LEADER_OPTIME)
|
||||
last_lsn = last_lsn and last_lsn['Value']
|
||||
slots = None
|
||||
|
||||
try:
|
||||
last_lsn = int(last_lsn)
|
||||
except Exception:
|
||||
last_lsn = 0
|
||||
|
||||
# get list of members
|
||||
members = [self.member(n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
||||
|
||||
# get leader
|
||||
leader = nodes.get(self._LEADER)
|
||||
if leader and leader['Value'] == self._name and self._session != leader.get('Session', 'x'):
|
||||
if not self._ctl and leader and leader['Value'] == self._name \
|
||||
and self._session != leader.get('Session', 'x'):
|
||||
logger.info('I am leader but not owner of the session. Removing leader node')
|
||||
self._client.kv.delete(self.leader_path, cas=leader['ModifyIndex'])
|
||||
leader = None
|
||||
@@ -169,40 +388,140 @@ class Consul(AbstractDCS):
|
||||
if failover:
|
||||
failover = Failover.from_node(failover['ModifyIndex'], failover['Value'])
|
||||
|
||||
self._cluster = Cluster(initialize, config, leader, last_leader_operation, members, failover)
|
||||
# get synchronization state
|
||||
sync = nodes.get(self._SYNC)
|
||||
sync = SyncState.from_node(sync and sync['ModifyIndex'], sync and sync['Value'])
|
||||
|
||||
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||
except NotFound:
|
||||
self._cluster = Cluster(False, None, None, None, [], None)
|
||||
except:
|
||||
return Cluster(None, None, None, None, [], None, None, None, None)
|
||||
except Exception:
|
||||
logger.exception('get_cluster')
|
||||
raise ConsulError('Consul is not responding properly')
|
||||
|
||||
def touch_member(self, data, **kwargs):
|
||||
@catch_consul_errors
|
||||
def touch_member(self, data, permanent=False):
|
||||
cluster = self.cluster
|
||||
member = cluster and ([m for m in cluster.members if m.name == self._name] or [None])[0]
|
||||
create_member = self.refresh_session()
|
||||
if member and (create_member or member.session != self._session):
|
||||
try:
|
||||
self._client.kv.delete(self.member_path)
|
||||
create_member = True
|
||||
except Exception:
|
||||
return False
|
||||
member = cluster and cluster.get_member(self._name, fallback_to_leader=False)
|
||||
|
||||
if not create_member and member and data == self._my_member_data:
|
||||
try:
|
||||
create_member = not permanent and self.refresh_session()
|
||||
except DCSError:
|
||||
return False
|
||||
|
||||
if member and (create_member or member.session != self._session):
|
||||
self._client.kv.delete(self.member_path)
|
||||
create_member = True
|
||||
|
||||
if self._register_service or self._previous_loop_register_service:
|
||||
try:
|
||||
self.update_service(not create_member and member and member.data or {}, data)
|
||||
except Exception:
|
||||
logger.exception('update_service')
|
||||
|
||||
if not create_member and member and deep_compare(data, member.data):
|
||||
return True
|
||||
|
||||
try:
|
||||
self._client.kv.put(self.member_path, data, acquire=self._session)
|
||||
self._my_member_data = data
|
||||
args = {} if permanent else {'acquire': self._session}
|
||||
self._client.kv.put(self.member_path, json.dumps(data, separators=(',', ':')), **args)
|
||||
return True
|
||||
except InvalidSession:
|
||||
self._session = None
|
||||
logger.error('Our session disappeared from Consul, can not "touch_member"')
|
||||
except Exception:
|
||||
logger.exception('touch_member')
|
||||
return False
|
||||
|
||||
def _set_service_name(self):
|
||||
self._service_name = service_name_from_scope_name(self._scope)
|
||||
if self._scope != self._service_name:
|
||||
logger.warning('Using %s as consul service name instead of scope name %s', self._service_name, self._scope)
|
||||
|
||||
@catch_consul_errors
|
||||
def attempt_to_acquire_leader(self):
|
||||
ret = self._client.kv.put(self.leader_path, self._name, acquire=self._session)
|
||||
def register_service(self, service_name, **kwargs):
|
||||
logger.info('Register service %s, params %s', service_name, kwargs)
|
||||
return self._client.agent.service.register(service_name, **kwargs)
|
||||
|
||||
@catch_consul_errors
|
||||
def deregister_service(self, service_id):
|
||||
logger.info('Deregister service %s', service_id)
|
||||
# service_id can contain special characters, but is used as part of uri in deregister request
|
||||
service_id = quote(service_id)
|
||||
return self._client.agent.service.deregister(service_id)
|
||||
|
||||
def _update_service(self, data):
|
||||
service_name = self._service_name
|
||||
role = data['role'].replace('_', '-')
|
||||
state = data['state']
|
||||
api_parts = urlparse(data['api_url'])
|
||||
api_parts = api_parts._replace(path='/{0}'.format(role))
|
||||
conn_parts = urlparse(data['conn_url'])
|
||||
check = base.Check.http(api_parts.geturl(), self._service_check_interval,
|
||||
deregister='{0}s'.format(self._client.http.ttl * 10))
|
||||
if self._service_check_tls_server_name is not None:
|
||||
check['TLSServerName'] = self._service_check_tls_server_name
|
||||
tags = self._service_tags[:]
|
||||
tags.append(role)
|
||||
self._previous_loop_service_tags = self._service_tags
|
||||
|
||||
params = {
|
||||
'service_id': '{0}/{1}'.format(self._scope, self._name),
|
||||
'address': conn_parts.hostname,
|
||||
'port': conn_parts.port,
|
||||
'check': check,
|
||||
'tags': tags,
|
||||
'enable_tag_override': True,
|
||||
}
|
||||
|
||||
if state == 'stopped' or (not self._register_service and self._previous_loop_register_service):
|
||||
self._previous_loop_register_service = self._register_service
|
||||
return self.deregister_service(params['service_id'])
|
||||
|
||||
self._previous_loop_register_service = self._register_service
|
||||
if role in ['master', 'replica', 'standby-leader']:
|
||||
if state != 'running':
|
||||
return
|
||||
return self.register_service(service_name, **params)
|
||||
|
||||
logger.warning('Could not register service: unknown role type %s', role)
|
||||
|
||||
@force_if_last_failed
|
||||
def update_service(self, old_data, new_data, force=False):
|
||||
update = False
|
||||
|
||||
for key in ['role', 'api_url', 'conn_url', 'state']:
|
||||
if key not in new_data:
|
||||
logger.warning('Could not register service: not enough params in member data')
|
||||
return
|
||||
if old_data.get(key) != new_data[key]:
|
||||
update = True
|
||||
|
||||
if (
|
||||
force or update or self._register_service != self._previous_loop_register_service
|
||||
or self._service_tags != self._previous_loop_service_tags
|
||||
):
|
||||
return self._update_service(new_data)
|
||||
|
||||
@catch_consul_errors
|
||||
def _do_attempt_to_acquire_leader(self, permanent):
|
||||
try:
|
||||
kwargs = {} if permanent else {'acquire': self._session}
|
||||
return self.retry(self._client.kv.put, self.leader_path, self._name, **kwargs)
|
||||
except InvalidSession:
|
||||
self._session = None
|
||||
logger.error('Our session disappeared from Consul. Will try to get a new one and retry attempt')
|
||||
self.refresh_session()
|
||||
return self.retry(self._client.kv.put, self.leader_path, self._name, acquire=self._session)
|
||||
|
||||
def attempt_to_acquire_leader(self, permanent=False):
|
||||
if not self._session and not permanent:
|
||||
self.refresh_session()
|
||||
|
||||
ret = self._do_attempt_to_acquire_leader(permanent)
|
||||
if not ret:
|
||||
logger.info('Could not take out TTL lock')
|
||||
|
||||
return ret
|
||||
|
||||
def take_leader(self):
|
||||
@@ -217,50 +536,69 @@ class Consul(AbstractDCS):
|
||||
return self._client.kv.put(self.config_path, value, cas=index)
|
||||
|
||||
@catch_consul_errors
|
||||
def write_leader_optime(self, last_operation):
|
||||
return self._client.kv.put(self.leader_optime_path, last_operation)
|
||||
def _write_leader_optime(self, last_lsn):
|
||||
return self._client.kv.put(self.leader_optime_path, last_lsn)
|
||||
|
||||
@staticmethod
|
||||
def update_leader():
|
||||
return True
|
||||
@catch_consul_errors
|
||||
def _write_status(self, value):
|
||||
return self._client.kv.put(self.status_path, value)
|
||||
|
||||
@catch_consul_errors
|
||||
def _update_leader(self):
|
||||
if self._session:
|
||||
self.retry(self._client.session.renew, self._session)
|
||||
self._last_session_refresh = time.time()
|
||||
return bool(self._session)
|
||||
|
||||
@catch_consul_errors
|
||||
def initialize(self, create_new=True, sysid=''):
|
||||
kwargs = {'cas': 0} if create_new else {}
|
||||
return self._client.kv.put(self.initialize_path, sysid, **kwargs)
|
||||
return self.retry(self._client.kv.put, self.initialize_path, sysid, **kwargs)
|
||||
|
||||
@catch_consul_errors
|
||||
def cancel_initialization(self):
|
||||
return self._client.kv.delete(self.initialize_path)
|
||||
return self.retry(self._client.kv.delete, self.initialize_path)
|
||||
|
||||
@catch_consul_errors
|
||||
def delete_cluster(self):
|
||||
return self._client.kv.delete(self.client_path(''), recurse=True)
|
||||
return self.retry(self._client.kv.delete, self.client_path(''), recurse=True)
|
||||
|
||||
@catch_consul_errors
|
||||
def delete_leader(self):
|
||||
def set_history_value(self, value):
|
||||
return self._client.kv.put(self.history_path, value)
|
||||
|
||||
@catch_consul_errors
|
||||
def _delete_leader(self):
|
||||
cluster = self.cluster
|
||||
if cluster and isinstance(cluster.leader, Leader) and cluster.leader.name == self._name:
|
||||
return self._client.kv.delete(self.leader_path, cas=cluster.leader.index)
|
||||
|
||||
def watch(self, timeout):
|
||||
@catch_consul_errors
|
||||
def set_sync_state_value(self, value, index=None):
|
||||
return self.retry(self._client.kv.put, self.sync_path, value, cas=index)
|
||||
|
||||
@catch_consul_errors
|
||||
def delete_sync_state(self, index=None):
|
||||
return self.retry(self._client.kv.delete, self.sync_path, cas=index)
|
||||
|
||||
def watch(self, leader_index, timeout):
|
||||
self._last_session_refresh = 0
|
||||
if self.__do_not_watch:
|
||||
self.__do_not_watch = False
|
||||
return True
|
||||
|
||||
cluster = self.cluster
|
||||
if cluster and cluster.leader and cluster.leader.name != self._name and cluster.leader.index:
|
||||
if leader_index:
|
||||
end_time = time.time() + timeout
|
||||
while timeout >= 1:
|
||||
try:
|
||||
idx, _ = self._client.kv.get(self.leader_path, index=cluster.leader.index, wait=str(timeout) + 's')
|
||||
return str(idx) != str(cluster.leader.index)
|
||||
except (ConsulException, RequestException):
|
||||
logging.exception('watch')
|
||||
idx, _ = self._client.kv.get(self.leader_path, index=leader_index, wait=str(timeout) + 's')
|
||||
return str(idx) != str(leader_index)
|
||||
except (ConsulException, HTTPException, HTTPError, socket.error, socket.timeout):
|
||||
logger.exception('watch')
|
||||
|
||||
timeout = end_time - time.time()
|
||||
|
||||
try:
|
||||
return super(Consul, self).watch(timeout)
|
||||
return super(Consul, self).watch(None, timeout)
|
||||
finally:
|
||||
self.event.clear()
|
||||
|
||||
+569
-168
@@ -1,160 +1,348 @@
|
||||
from __future__ import absolute_import
|
||||
import abc
|
||||
import etcd
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import urllib3.util.connection
|
||||
import random
|
||||
import requests
|
||||
import six
|
||||
import socket
|
||||
import time
|
||||
|
||||
from dns.exception import DNSException
|
||||
from dns import resolver
|
||||
from patroni.dcs import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member
|
||||
from patroni.exceptions import DCSError
|
||||
from patroni.utils import Retry, RetryFailedError, sleep
|
||||
from urllib3.exceptions import HTTPError, ReadTimeoutError
|
||||
from requests.exceptions import RequestException
|
||||
from urllib3 import Timeout
|
||||
from urllib3.exceptions import HTTPError, ReadTimeoutError, ProtocolError
|
||||
from six.moves.queue import Queue
|
||||
from six.moves.http_client import HTTPException
|
||||
from six.moves.urllib_parse import urlparse
|
||||
from threading import Thread
|
||||
|
||||
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member, SyncState, TimelineHistory
|
||||
from ..exceptions import DCSError
|
||||
from ..request import get as requests_get
|
||||
from ..utils import Retry, RetryFailedError, split_host_port, uri, USER_AGENT
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class EtcdRaftInternal(etcd.EtcdException):
|
||||
"""Raft Internal Error"""
|
||||
|
||||
|
||||
class EtcdError(DCSError):
|
||||
pass
|
||||
|
||||
|
||||
class Client(etcd.Client):
|
||||
class DnsCachingResolver(Thread):
|
||||
|
||||
def __init__(self, config):
|
||||
super(Client, self).__init__(read_timeout=config['retry_timeout'])
|
||||
def __init__(self, cache_time=600.0, cache_fail_time=30.0):
|
||||
super(DnsCachingResolver, self).__init__()
|
||||
self._cache = {}
|
||||
self._cache_time = cache_time
|
||||
self._cache_fail_time = cache_fail_time
|
||||
self._resolve_queue = Queue()
|
||||
self.daemon = True
|
||||
self.start()
|
||||
|
||||
def run(self):
|
||||
while True:
|
||||
(host, port), attempt = self._resolve_queue.get()
|
||||
response = self._do_resolve(host, port)
|
||||
if response:
|
||||
self._cache[(host, port)] = (time.time(), response)
|
||||
else:
|
||||
if attempt < 10:
|
||||
self.resolve_async(host, port, attempt + 1)
|
||||
time.sleep(1)
|
||||
|
||||
def resolve(self, host, port):
|
||||
current_time = time.time()
|
||||
cached_time, response = self._cache.get((host, port), (0, []))
|
||||
time_passed = current_time - cached_time
|
||||
if time_passed > self._cache_time or (not response and time_passed > self._cache_fail_time):
|
||||
new_response = self._do_resolve(host, port)
|
||||
if new_response:
|
||||
self._cache[(host, port)] = (current_time, new_response)
|
||||
response = new_response
|
||||
return response
|
||||
|
||||
def resolve_async(self, host, port, attempt=0):
|
||||
self._resolve_queue.put(((host, port), attempt))
|
||||
|
||||
def remove(self, host, port):
|
||||
self._cache.pop((host, port), None)
|
||||
|
||||
@staticmethod
|
||||
def _do_resolve(host, port):
|
||||
try:
|
||||
return socket.getaddrinfo(host, port, 0, socket.SOCK_STREAM, socket.IPPROTO_TCP)
|
||||
except Exception as e:
|
||||
logger.warning('failed to resolve host %s: %s', host, e)
|
||||
return []
|
||||
|
||||
|
||||
@six.add_metaclass(abc.ABCMeta)
|
||||
class AbstractEtcdClientWithFailover(etcd.Client):
|
||||
|
||||
def __init__(self, config, dns_resolver, cache_ttl=300):
|
||||
self._dns_resolver = dns_resolver
|
||||
self.set_machines_cache_ttl(cache_ttl)
|
||||
self._machines_cache_updated = 0
|
||||
args = {p: config.get(p) for p in ('host', 'port', 'protocol', 'use_proxies', 'username', 'password',
|
||||
'cert', 'ca_cert') if config.get(p)}
|
||||
super(AbstractEtcdClientWithFailover, self).__init__(read_timeout=config['retry_timeout'], **args)
|
||||
# For some reason python3-etcd on debian and ubuntu are not based on the latest version
|
||||
# Workaround for the case when https://github.com/jplana/python-etcd/pull/196 is not applied
|
||||
self.http.connection_pool_kw.pop('ssl_version', None)
|
||||
self._config = config
|
||||
self._initial_machines_cache = []
|
||||
self._load_machines_cache()
|
||||
self._allow_reconnect = True
|
||||
# allow passing retry argument to api_execute in params
|
||||
self._comparison_conditions.add('retry')
|
||||
self._read_options.add('retry')
|
||||
self._del_conditions.add('retry')
|
||||
|
||||
def _calculate_timeouts(self, etcd_nodes, timeout=None):
|
||||
"""Calculate a request timeout and number of retries per single etcd node.
|
||||
In case if the timeout per node is too small (less than one second) we will reduce the number of nodes.
|
||||
For the cluster with only one node we will try to do 2 retries.
|
||||
For clusters with 2 nodes we will try to do 1 retry for every node.
|
||||
No retries for clusters with 3 or more nodes. We better rely on switching to a different node."""
|
||||
|
||||
per_node_timeout = timeout = float(timeout or self.read_timeout)
|
||||
|
||||
max_retries = 4 - min(etcd_nodes, 3)
|
||||
per_node_retries = 1
|
||||
min_timeout = 1.0
|
||||
|
||||
while etcd_nodes > 0:
|
||||
per_node_timeout = float(timeout) / etcd_nodes
|
||||
if per_node_timeout >= min_timeout:
|
||||
# for small clusters we will try to do more than on try on every node
|
||||
while per_node_retries < max_retries and per_node_timeout / (per_node_retries + 1) >= min_timeout:
|
||||
per_node_retries += 1
|
||||
per_node_timeout /= per_node_retries
|
||||
break
|
||||
# if the timeout per one node is to small try to reduce number of nodes
|
||||
etcd_nodes -= 1
|
||||
max_retries = 1
|
||||
|
||||
return etcd_nodes, per_node_timeout, per_node_retries - 1
|
||||
|
||||
def reload_config(self, config):
|
||||
self.username = config.get('username')
|
||||
self.password = config.get('password')
|
||||
|
||||
def _get_headers(self):
|
||||
basic_auth = ':'.join((self.username, self.password)) if self.username and self.password else None
|
||||
return urllib3.make_headers(basic_auth=basic_auth, user_agent=USER_AGENT)
|
||||
|
||||
def _prepare_common_parameters(self, etcd_nodes, timeout=None):
|
||||
kwargs = {'headers': self._get_headers(), 'redirect': self.allow_redirect, 'preload_content': False}
|
||||
|
||||
if timeout is not None:
|
||||
kwargs.update(retries=0, timeout=timeout)
|
||||
else:
|
||||
_, per_node_timeout, per_node_retries = self._calculate_timeouts(etcd_nodes)
|
||||
connect_timeout = max(1, per_node_timeout/2)
|
||||
kwargs.update(timeout=Timeout(connect=connect_timeout, total=per_node_timeout), retries=per_node_retries)
|
||||
return kwargs
|
||||
|
||||
def set_machines_cache_ttl(self, cache_ttl):
|
||||
self._machines_cache_ttl = cache_ttl
|
||||
|
||||
@abc.abstractmethod
|
||||
def _prepare_get_members(self, etcd_nodes):
|
||||
"""returns: request parameters"""
|
||||
|
||||
@abc.abstractmethod
|
||||
def _get_members(self, base_uri, **kwargs):
|
||||
"""returns: list of clientURLs"""
|
||||
|
||||
@property
|
||||
def machines_cache(self):
|
||||
base_uri, cache = self._base_uri, self._machines_cache
|
||||
return ([base_uri] if base_uri in cache else []) + [machine for machine in cache if machine != base_uri]
|
||||
|
||||
@property
|
||||
def machines(self):
|
||||
"""Original `machines` method(property) of `etcd.Client` class raise exception
|
||||
when it failed to get list of etcd cluster members. This method is being called
|
||||
only when request failed on one of the etcd members during `api_execute` call.
|
||||
For us it's more important to execute original request rather then get new
|
||||
topology of etcd cluster. So we will catch this exception and return valid list
|
||||
of machines with setting flag `self._update_machines_cache` to `!True`.
|
||||
Later, during next `api_execute` call we will forcefully update machines_cache"""
|
||||
try:
|
||||
ret = super(Client, self).machines
|
||||
random.shuffle(ret)
|
||||
return ret
|
||||
except etcd.EtcdException:
|
||||
if self._update_machines_cache: # We are updating machines_cache
|
||||
raise # This exception is fatal, we should re-raise it.
|
||||
self._update_machines_cache = True
|
||||
return [self._base_uri]
|
||||
For us it's more important to execute original request rather then get new topology
|
||||
of etcd cluster. So we will catch this exception and return empty list of machines.
|
||||
Later, during next `api_execute` call we will forcefully update machines_cache.
|
||||
|
||||
Also this method implements the same timeout-retry logic as `api_execute`, because
|
||||
the original method was retrying 2 times with the `read_timeout` on each node."""
|
||||
|
||||
machines_cache = self.machines_cache
|
||||
kwargs = self._prepare_get_members(len(machines_cache))
|
||||
|
||||
for base_uri in machines_cache:
|
||||
try:
|
||||
machines = list(set(self._get_members(base_uri, **kwargs)))
|
||||
logger.debug("Retrieved list of machines: %s", machines)
|
||||
if machines:
|
||||
random.shuffle(machines)
|
||||
if not self._use_proxies:
|
||||
self._update_dns_cache(self._dns_resolver.resolve_async, machines)
|
||||
return machines
|
||||
except Exception as e:
|
||||
self.http.clear()
|
||||
logger.error("Failed to get list of machines from %s%s: %r", base_uri, self.version_prefix, e)
|
||||
|
||||
raise etcd.EtcdConnectionFailed('No more machines in the cluster')
|
||||
|
||||
def set_read_timeout(self, timeout):
|
||||
self._read_timeout = timeout
|
||||
|
||||
def _do_http_request(self, request_executor, method, url, fields=None, **kwargs):
|
||||
try:
|
||||
response = request_executor(method, url, fields=fields, **kwargs)
|
||||
response.data.decode('utf-8')
|
||||
self._check_cluster_id(response)
|
||||
except (HTTPError, HTTPException, socket.error, socket.timeout) as e:
|
||||
if (isinstance(fields, dict) and fields.get("wait") == "true" and
|
||||
isinstance(e, ReadTimeoutError)):
|
||||
logger.debug("Watch timed out.")
|
||||
raise etcd.EtcdWatchTimedOut("Watch timed out: {0}".format(e), cause=e)
|
||||
logger.error("Request to server %s failed: %r", self._base_uri, e)
|
||||
logger.info("Reconnection allowed, looking for another server.")
|
||||
self._base_uri = self._next_server(cause=e)
|
||||
response = False
|
||||
return response
|
||||
def _do_http_request(self, retry, machines_cache, request_executor, method, path, fields=None, **kwargs):
|
||||
if fields is not None:
|
||||
kwargs['fields'] = fields
|
||||
some_request_failed = False
|
||||
for i, base_uri in enumerate(machines_cache):
|
||||
if i > 0:
|
||||
logger.info("Retrying on %s", base_uri)
|
||||
try:
|
||||
response = request_executor(method, base_uri + path, **kwargs)
|
||||
response.data.decode('utf-8')
|
||||
if some_request_failed:
|
||||
self.set_base_uri(base_uri)
|
||||
self._refresh_machines_cache()
|
||||
return response
|
||||
except (HTTPError, HTTPException, socket.error, socket.timeout) as e:
|
||||
self.http.clear()
|
||||
# switch to the next etcd node because we don't know exactly what happened,
|
||||
# whether the key didn't received an update or there is a network problem.
|
||||
if not retry and i + 1 < len(machines_cache):
|
||||
self.set_base_uri(machines_cache[i + 1])
|
||||
if (isinstance(fields, dict) and fields.get("wait") == "true" and
|
||||
isinstance(e, (ReadTimeoutError, ProtocolError))):
|
||||
logger.debug("Watch timed out.")
|
||||
raise etcd.EtcdWatchTimedOut("Watch timed out: {0}".format(e), cause=e)
|
||||
logger.error("Request to server %s failed: %r", base_uri, e)
|
||||
logger.info("Reconnection allowed, looking for another server.")
|
||||
if not retry:
|
||||
raise etcd.EtcdException('{0} {1} request failed'.format(method, path))
|
||||
some_request_failed = True
|
||||
|
||||
raise etcd.EtcdConnectionFailed('No more machines in the cluster')
|
||||
|
||||
@abc.abstractmethod
|
||||
def _prepare_request(self, kwargs, params=None, method=None):
|
||||
"""returns: request_executor"""
|
||||
|
||||
def api_execute(self, path, method, params=None, timeout=None):
|
||||
if not path.startswith('/'):
|
||||
raise ValueError('Path does not start with /')
|
||||
|
||||
kwargs = {'fields': params, 'redirect': self.allow_redirect,
|
||||
'headers': self._get_headers(), 'preload_content': False}
|
||||
|
||||
if method in [self._MGET, self._MDELETE]:
|
||||
request_executor = self.http.request
|
||||
elif method in [self._MPUT, self._MPOST]:
|
||||
request_executor = self.http.request_encode_body
|
||||
kwargs['encode_multipart'] = False
|
||||
else:
|
||||
raise etcd.EtcdException('HTTP method {0} not supported'.format(method))
|
||||
retry = params.pop('retry', None) if isinstance(params, dict) else None
|
||||
|
||||
# Update machines_cache if previous attempt of update has failed
|
||||
if self._update_machines_cache:
|
||||
self._load_machines_cache()
|
||||
elif not self._use_proxies and time.time() - self._machines_cache_updated > self._machines_cache_ttl:
|
||||
self._refresh_machines_cache()
|
||||
|
||||
if timeout is None:
|
||||
# calculate the number of retries and timeout *per node*
|
||||
# actual number of retries depends on the number of nodes
|
||||
etcd_nodes = len(self._machines_cache) + 1
|
||||
kwargs['retries'] = 0 if etcd_nodes > 3 else (1 if etcd_nodes > 1 else 2)
|
||||
machines_cache = self.machines_cache
|
||||
etcd_nodes = len(machines_cache)
|
||||
|
||||
# if etcd_nodes > 3:
|
||||
# kwargs.update({'retries': 0, 'timeout': float(self.read_timeout)/etcd_nodes})
|
||||
# elif etcd_nodes > 1:
|
||||
# kwargs.update({'retries': 1, 'timeout': self.read_timeout/2.0/etcd_nodes})
|
||||
# else:
|
||||
# kwargs.update({'retries': 2, 'timeout': self.read_timeout/3.0})
|
||||
kwargs['timeout'] = self.read_timeout/float(kwargs['retries'] + 1)/etcd_nodes
|
||||
else:
|
||||
kwargs.update({'retries': 0, 'timeout': timeout})
|
||||
kwargs = self._prepare_common_parameters(etcd_nodes, timeout)
|
||||
request_executor = self._prepare_request(kwargs, params, method)
|
||||
|
||||
response = False
|
||||
|
||||
try:
|
||||
while not response:
|
||||
response = self._do_http_request(request_executor, method, self._base_uri + path, **kwargs)
|
||||
|
||||
if response is False and not self._use_proxies:
|
||||
self._machines_cache = self.machines
|
||||
self._machines_cache.remove(self._base_uri)
|
||||
return self._handle_server_response(response)
|
||||
except etcd.EtcdConnectionFailed:
|
||||
self._update_machines_cache = True
|
||||
raise
|
||||
while True:
|
||||
try:
|
||||
response = self._do_http_request(retry, machines_cache, request_executor, method, path, **kwargs)
|
||||
return self._handle_server_response(response)
|
||||
except etcd.EtcdWatchTimedOut:
|
||||
raise
|
||||
except etcd.EtcdConnectionFailed as ex:
|
||||
try:
|
||||
if self._load_machines_cache():
|
||||
machines_cache = self.machines_cache
|
||||
etcd_nodes = len(machines_cache)
|
||||
except Exception as e:
|
||||
logger.debug('Failed to update list of etcd nodes: %r', e)
|
||||
sleeptime = retry.sleeptime
|
||||
remaining_time = retry.stoptime - sleeptime - time.time()
|
||||
nodes, timeout, retries = self._calculate_timeouts(etcd_nodes, remaining_time)
|
||||
if nodes == 0:
|
||||
self._update_machines_cache = True
|
||||
self.set_base_uri(self._base_uri) # trigger Etcd3 watcher restart
|
||||
raise ex
|
||||
retry.sleep_func(sleeptime)
|
||||
retry.update_delay()
|
||||
# We still have some time left. Partially reduce `machines_cache` and retry request
|
||||
kwargs.update(timeout=Timeout(connect=max(1, timeout/2), total=timeout), retries=retries)
|
||||
machines_cache = machines_cache[:nodes]
|
||||
|
||||
@staticmethod
|
||||
def get_srv_record(host):
|
||||
try:
|
||||
return [(str(r.target).rstrip('.'), r.port) for r in resolver.query('_etcd-server._tcp.' + host, 'SRV')]
|
||||
return [(r.target.to_text(True), r.port) for r in resolver.query(host, 'SRV')]
|
||||
except DNSException:
|
||||
logger.exception('Can not resolve SRV for %s', host)
|
||||
return []
|
||||
return []
|
||||
|
||||
def _get_machines_cache_from_srv(self, discovery_srv):
|
||||
def _get_machines_cache_from_srv(self, srv, srv_suffix=None):
|
||||
"""Fetch list of etcd-cluster member by resolving _etcd-server._tcp. SRV record.
|
||||
This record should contain list of host and peer ports which could be used to run
|
||||
'GET http://{host}:{port}/members' request (peer protocol)"""
|
||||
|
||||
ret = []
|
||||
for host, port in self.get_srv_record(discovery_srv):
|
||||
url = '{0}://{1}:{2}/members'.format(self._protocol, host, port)
|
||||
try:
|
||||
response = requests.get(url, timeout=self.read_timeout)
|
||||
if response.ok:
|
||||
for member in response.json():
|
||||
ret.extend(member['clientURLs'])
|
||||
break
|
||||
except RequestException:
|
||||
logger.exception('GET %s', url)
|
||||
for r in ['-client-ssl', '-client', '-ssl', '', '-server-ssl', '-server']:
|
||||
r = '{0}-{1}'.format(r, srv_suffix) if srv_suffix else r
|
||||
protocol = 'https' if '-ssl' in r else 'http'
|
||||
endpoint = '/members' if '-server' in r else ''
|
||||
for host, port in self.get_srv_record('_etcd{0}._tcp.{1}'.format(r, srv)):
|
||||
url = uri(protocol, (host, port), endpoint)
|
||||
if endpoint:
|
||||
try:
|
||||
response = requests_get(url, timeout=self.read_timeout, verify=False)
|
||||
if response.status < 400:
|
||||
for member in json.loads(response.data.decode('utf-8')):
|
||||
ret.extend(member['clientURLs'])
|
||||
break
|
||||
except Exception:
|
||||
logger.exception('GET %s', url)
|
||||
else:
|
||||
ret.append(url)
|
||||
if ret:
|
||||
self._protocol = protocol
|
||||
break
|
||||
else:
|
||||
logger.warning('Can not resolve SRV for %s', srv)
|
||||
return list(set(ret))
|
||||
|
||||
def _get_machines_cache_from_dns(self, addr):
|
||||
def _get_machines_cache_from_dns(self, host, port):
|
||||
"""One host might be resolved into multiple ip addresses. We will make list out of it"""
|
||||
if self.protocol == 'http':
|
||||
ret = map(lambda res: uri(self.protocol, res[-1][:2]), self._dns_resolver.resolve(host, port))
|
||||
if ret:
|
||||
return list(set(ret))
|
||||
return [uri(self.protocol, (host, port))]
|
||||
|
||||
ret = []
|
||||
host, port = addr.split(':')
|
||||
try:
|
||||
for r in set(socket.getaddrinfo(host, port, socket.AF_INET, socket.SOCK_STREAM, socket.IPPROTO_TCP)):
|
||||
ret.append('{0}://{1}:{2}'.format(self._protocol, r[4][0], r[4][1]))
|
||||
except socket.error:
|
||||
logger.exception('Can not resolve %s', host)
|
||||
return list(set(ret)) if ret else ['{0}://{1}:{2}'.format(self._protocol, host, port)]
|
||||
def _get_machines_cache_from_config(self):
|
||||
if 'proxy' in self._config:
|
||||
return [uri(self.protocol, (self._config['host'], self._config['port']))]
|
||||
|
||||
machines_cache = []
|
||||
if 'srv' in self._config:
|
||||
machines_cache = self._get_machines_cache_from_srv(self._config['srv'], self._config.get('srv_suffix'))
|
||||
|
||||
if not machines_cache and 'hosts' in self._config:
|
||||
machines_cache = list(self._config['hosts'])
|
||||
|
||||
if not machines_cache and 'host' in self._config:
|
||||
machines_cache = self._get_machines_cache_from_dns(self._config['host'], self._config['port'])
|
||||
return machines_cache
|
||||
|
||||
@staticmethod
|
||||
def _update_dns_cache(func, machines):
|
||||
for url in machines:
|
||||
r = urlparse(url)
|
||||
port = r.port or (443 if r.scheme == 'https' else 80)
|
||||
func(r.hostname, port)
|
||||
|
||||
def _load_machines_cache(self):
|
||||
"""This method should fill up `_machines_cache` from scratch.
|
||||
@@ -164,87 +352,247 @@ class Client(etcd.Client):
|
||||
|
||||
self._update_machines_cache = True
|
||||
|
||||
if 'discovery_srv' not in self._config and 'host' not in self._config:
|
||||
raise Exception('Neither discovery_srv nor host are defined in etcd section of config')
|
||||
|
||||
self._machines_cache = []
|
||||
|
||||
if 'discovery_srv' in self._config:
|
||||
self._machines_cache = self._get_machines_cache_from_srv(self._config['discovery_srv'])
|
||||
|
||||
if not self._machines_cache and 'host' in self._config:
|
||||
self._machines_cache = self._get_machines_cache_from_dns(self._config['host'])
|
||||
if 'srv' not in self._config and 'host' not in self._config and 'hosts' not in self._config:
|
||||
raise Exception('Neither srv, hosts, host nor url are defined in etcd section of config')
|
||||
|
||||
machines_cache = self._get_machines_cache_from_config()
|
||||
# Can not bootstrap list of etcd-cluster members, giving up
|
||||
if not self._machines_cache:
|
||||
if not machines_cache:
|
||||
raise etcd.EtcdException
|
||||
|
||||
# After filling up initial list of machines_cache we should ask etcd-cluster about actual list
|
||||
self._base_uri = self._machines_cache.pop(0)
|
||||
self._machines_cache = self.machines
|
||||
# enforce resolving dns name,they might get new ips
|
||||
self._update_dns_cache(self._dns_resolver.remove, machines_cache)
|
||||
|
||||
if self._base_uri in self._machines_cache:
|
||||
self._machines_cache.remove(self._base_uri)
|
||||
# The etcd cluster could change its topology over time and depending on how we resolve the initial
|
||||
# topology (list of hosts in the Patroni config or DNS records, A or SRV) we might get into the situation
|
||||
# the the real topology doesn't match anymore with the topology resolved from the configuration file.
|
||||
# In case if the "initial" topology is the same as before we will not override the `_machines_cache`.
|
||||
ret = set(machines_cache) != set(self._initial_machines_cache)
|
||||
if ret:
|
||||
self._initial_machines_cache = self._machines_cache = machines_cache
|
||||
|
||||
# After filling up the initial list of machines_cache we should ask etcd-cluster about actual list
|
||||
self._refresh_machines_cache(True)
|
||||
|
||||
self._update_machines_cache = False
|
||||
return ret
|
||||
|
||||
def _refresh_machines_cache(self, updating_cache=False):
|
||||
if self._use_proxies:
|
||||
self._machines_cache = self._get_machines_cache_from_config()
|
||||
else:
|
||||
try:
|
||||
self._machines_cache = self.machines
|
||||
except etcd.EtcdConnectionFailed:
|
||||
if updating_cache:
|
||||
raise etcd.EtcdException("Could not get the list of servers, "
|
||||
"maybe you provided the wrong "
|
||||
"host(s) to connect to?")
|
||||
return
|
||||
|
||||
if self._base_uri not in self._machines_cache:
|
||||
self.set_base_uri(self._machines_cache[0])
|
||||
self._machines_cache_updated = time.time()
|
||||
|
||||
def set_base_uri(self, value):
|
||||
if self._base_uri != value:
|
||||
logger.info('Selected new etcd server %s', value)
|
||||
self._base_uri = value
|
||||
|
||||
|
||||
def catch_etcd_errors(func):
|
||||
def wrapper(*args, **kwargs):
|
||||
try:
|
||||
return func(*args, **kwargs) is not None
|
||||
except (RetryFailedError, etcd.EtcdException):
|
||||
return False
|
||||
except:
|
||||
logger.exception("")
|
||||
raise EtcdError("unexpected error")
|
||||
class EtcdClient(AbstractEtcdClientWithFailover):
|
||||
|
||||
return wrapper
|
||||
ERROR_CLS = EtcdError
|
||||
|
||||
def __del__(self):
|
||||
if self.http is not None:
|
||||
try:
|
||||
self.http.clear()
|
||||
except (ReferenceError, TypeError, AttributeError):
|
||||
pass
|
||||
|
||||
def _prepare_get_members(self, etcd_nodes):
|
||||
return self._prepare_common_parameters(etcd_nodes)
|
||||
|
||||
def _get_members(self, base_uri, **kwargs):
|
||||
response = self.http.request(self._MGET, base_uri + self.version_prefix + '/machines', **kwargs)
|
||||
data = self._handle_server_response(response).data.decode('utf-8')
|
||||
return [m.strip() for m in data.split(',') if m.strip()]
|
||||
|
||||
def _prepare_request(self, kwargs, params=None, method=None):
|
||||
kwargs['fields'] = params
|
||||
if method in (self._MPOST, self._MPUT):
|
||||
kwargs['encode_multipart'] = False
|
||||
return self.http.request
|
||||
|
||||
|
||||
class Etcd(AbstractDCS):
|
||||
class AbstractEtcd(AbstractDCS):
|
||||
|
||||
def __init__(self, config):
|
||||
super(Etcd, self).__init__(config)
|
||||
self._ttl = int(config.get('ttl') or 30)
|
||||
def __init__(self, config, client_cls, retry_errors_cls):
|
||||
super(AbstractEtcd, self).__init__(config)
|
||||
self._retry = Retry(deadline=config['retry_timeout'], max_delay=1, max_tries=-1,
|
||||
retry_exceptions=(etcd.EtcdLeaderElectionInProgress,
|
||||
etcd.EtcdWatcherCleared,
|
||||
etcd.EtcdEventIndexCleared))
|
||||
self._client = self.get_etcd_client(config)
|
||||
retry_exceptions=retry_errors_cls)
|
||||
self._ttl = int(config.get('ttl') or 30)
|
||||
self._client = self.get_etcd_client(config, client_cls)
|
||||
self.__do_not_watch = False
|
||||
self._has_failed = False
|
||||
|
||||
def reload_config(self, config):
|
||||
super(AbstractEtcd, self).reload_config(config)
|
||||
self._client.reload_config(config.get(self.__class__.__name__.lower(), {}))
|
||||
|
||||
def retry(self, *args, **kwargs):
|
||||
return self._retry.copy()(*args, **kwargs)
|
||||
retry = self._retry.copy()
|
||||
kwargs['retry'] = retry
|
||||
return retry(*args, **kwargs)
|
||||
|
||||
def _handle_exception(self, e, name='', do_sleep=False, raise_ex=None):
|
||||
if not self._has_failed:
|
||||
logger.exception(name)
|
||||
else:
|
||||
logger.error(e)
|
||||
if do_sleep:
|
||||
time.sleep(1)
|
||||
self._has_failed = True
|
||||
if isinstance(raise_ex, Exception):
|
||||
raise raise_ex
|
||||
|
||||
@staticmethod
|
||||
def get_etcd_client(config):
|
||||
def set_socket_options(sock, socket_options):
|
||||
if socket_options:
|
||||
for opt in socket_options:
|
||||
sock.setsockopt(*opt)
|
||||
|
||||
def get_etcd_client(self, config, client_cls):
|
||||
if 'proxy' in config:
|
||||
config['use_proxies'] = True
|
||||
config['url'] = config['proxy']
|
||||
|
||||
if 'url' in config:
|
||||
r = urlparse(config['url'])
|
||||
config.update({'protocol': r.scheme, 'host': r.hostname, 'port': r.port or 2379,
|
||||
'username': r.username, 'password': r.password})
|
||||
elif 'hosts' in config:
|
||||
hosts = config.pop('hosts')
|
||||
default_port = config.pop('port', 2379)
|
||||
protocol = config.get('protocol', 'http')
|
||||
|
||||
if isinstance(hosts, six.string_types):
|
||||
hosts = hosts.split(',')
|
||||
|
||||
config['hosts'] = []
|
||||
for value in hosts:
|
||||
if isinstance(value, six.string_types):
|
||||
config['hosts'].append(uri(protocol, split_host_port(value.strip(), default_port)))
|
||||
elif 'host' in config:
|
||||
host, port = split_host_port(config['host'], 2379)
|
||||
config['host'] = host
|
||||
if 'port' not in config:
|
||||
config['port'] = int(port)
|
||||
|
||||
if config.get('cacert'):
|
||||
config['ca_cert'] = config.pop('cacert')
|
||||
|
||||
if config.get('key') and config.get('cert'):
|
||||
config['cert'] = (config['cert'], config['key'])
|
||||
|
||||
for p in ('discovery_srv', 'srv_domain'):
|
||||
if p in config:
|
||||
config['srv'] = config.pop(p)
|
||||
|
||||
dns_resolver = DnsCachingResolver()
|
||||
|
||||
def create_connection_patched(address, timeout=socket._GLOBAL_DEFAULT_TIMEOUT,
|
||||
source_address=None, socket_options=None):
|
||||
host, port = address
|
||||
if host.startswith('['):
|
||||
host = host.strip('[]')
|
||||
err = None
|
||||
for af, socktype, proto, _, sa in dns_resolver.resolve(host, port):
|
||||
sock = None
|
||||
try:
|
||||
sock = socket.socket(af, socktype, proto)
|
||||
self.set_socket_options(sock, socket_options)
|
||||
if timeout is not socket._GLOBAL_DEFAULT_TIMEOUT:
|
||||
sock.settimeout(timeout)
|
||||
if source_address:
|
||||
sock.bind(source_address)
|
||||
sock.connect(sa)
|
||||
return sock
|
||||
|
||||
except socket.error as e:
|
||||
err = e
|
||||
if sock is not None:
|
||||
sock.close()
|
||||
sock = None
|
||||
|
||||
if err is not None:
|
||||
raise err
|
||||
|
||||
raise socket.error("getaddrinfo returns an empty list")
|
||||
|
||||
urllib3.util.connection.create_connection = create_connection_patched
|
||||
|
||||
client = None
|
||||
while not client:
|
||||
try:
|
||||
client = Client(config)
|
||||
client = client_cls(config, dns_resolver)
|
||||
if 'use_proxies' in config and not client.machines:
|
||||
raise etcd.EtcdException
|
||||
except etcd.EtcdException:
|
||||
logger.info('waiting on etcd')
|
||||
sleep(5)
|
||||
time.sleep(5)
|
||||
return client
|
||||
|
||||
def set_ttl(self, ttl):
|
||||
ttl = int(ttl)
|
||||
self.__do_not_watch = self._ttl != ttl
|
||||
ret = self._ttl != ttl
|
||||
self._ttl = ttl
|
||||
self._client.set_machines_cache_ttl(ttl*10)
|
||||
return ret
|
||||
|
||||
@property
|
||||
def ttl(self):
|
||||
return self._ttl
|
||||
|
||||
def set_retry_timeout(self, retry_timeout):
|
||||
self._retry.deadline = retry_timeout
|
||||
self._client.set_read_timeout(retry_timeout)
|
||||
|
||||
|
||||
def catch_etcd_errors(func):
|
||||
def wrapper(self, *args, **kwargs):
|
||||
try:
|
||||
retval = func(self, *args, **kwargs) is not None
|
||||
self._has_failed = False
|
||||
return retval
|
||||
except (RetryFailedError, etcd.EtcdException) as e:
|
||||
self._handle_exception(e)
|
||||
return False
|
||||
except Exception as e:
|
||||
self._handle_exception(e, raise_ex=self._client.ERROR_CLS('unexpected error'))
|
||||
|
||||
return wrapper
|
||||
|
||||
|
||||
class Etcd(AbstractEtcd):
|
||||
|
||||
def __init__(self, config):
|
||||
super(Etcd, self).__init__(config, EtcdClient, (etcd.EtcdLeaderElectionInProgress, EtcdRaftInternal))
|
||||
self.__do_not_watch = False
|
||||
|
||||
def set_ttl(self, ttl):
|
||||
self.__do_not_watch = super(Etcd, self).set_ttl(ttl)
|
||||
|
||||
@staticmethod
|
||||
def member(node):
|
||||
return Member.from_node(node.modifiedIndex, os.path.basename(node.key), node.ttl, node.value)
|
||||
|
||||
def _load_cluster(self):
|
||||
cluster = None
|
||||
try:
|
||||
result = self.retry(self._client.read, self.client_path(''), recursive=True)
|
||||
nodes = {os.path.relpath(node.key, result.key): node for node in result.leaves}
|
||||
nodes = {node.key[len(result.key):].lstrip('/'): node for node in result.leaves}
|
||||
|
||||
# get initialize flag
|
||||
initialize = nodes.get(self._INITIALIZE)
|
||||
@@ -254,9 +602,28 @@ class Etcd(AbstractDCS):
|
||||
config = nodes.get(self._CONFIG)
|
||||
config = config and ClusterConfig.from_node(config.modifiedIndex, config.value)
|
||||
|
||||
# get last leader operation
|
||||
last_leader_operation = nodes.get(self._LEADER_OPTIME)
|
||||
last_leader_operation = 0 if last_leader_operation is None else int(last_leader_operation.value)
|
||||
# get timeline history
|
||||
history = nodes.get(self._HISTORY)
|
||||
history = history and TimelineHistory.from_node(history.modifiedIndex, history.value)
|
||||
|
||||
# get last know leader lsn and slots
|
||||
status = nodes.get(self._STATUS)
|
||||
if status:
|
||||
try:
|
||||
status = json.loads(status.value)
|
||||
last_lsn = status.get(self._OPTIME)
|
||||
slots = status.get('slots')
|
||||
except Exception:
|
||||
slots = last_lsn = None
|
||||
else:
|
||||
last_lsn = nodes.get(self._LEADER_OPTIME)
|
||||
last_lsn = last_lsn and last_lsn.value
|
||||
slots = None
|
||||
|
||||
try:
|
||||
last_lsn = int(last_lsn)
|
||||
except Exception:
|
||||
last_lsn = 0
|
||||
|
||||
# get list of members
|
||||
members = [self.member(n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
||||
@@ -266,31 +633,42 @@ class Etcd(AbstractDCS):
|
||||
if leader:
|
||||
member = Member(-1, leader.value, None, {})
|
||||
member = ([m for m in members if m.name == leader.value] or [member])[0]
|
||||
leader = Leader(leader.modifiedIndex, leader.ttl, member)
|
||||
index = result.etcd_index if result.etcd_index > leader.modifiedIndex else leader.modifiedIndex + 1
|
||||
leader = Leader(index, leader.ttl, member)
|
||||
|
||||
# failover key
|
||||
failover = nodes.get(self._FAILOVER)
|
||||
if failover:
|
||||
failover = Failover.from_node(failover.modifiedIndex, failover.value)
|
||||
|
||||
self._cluster = Cluster(initialize, config, leader, last_leader_operation, members, failover)
|
||||
# get synchronization state
|
||||
sync = nodes.get(self._SYNC)
|
||||
sync = SyncState.from_node(sync and sync.modifiedIndex, sync and sync.value)
|
||||
|
||||
cluster = Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||
except etcd.EtcdKeyNotFound:
|
||||
self._cluster = Cluster(False, None, None, None, [], None)
|
||||
except:
|
||||
logger.exception('get_cluster')
|
||||
raise EtcdError('Etcd is not responding properly')
|
||||
cluster = Cluster(None, None, None, None, [], None, None, None, None)
|
||||
except Exception as e:
|
||||
self._handle_exception(e, 'get_cluster', raise_ex=EtcdError('Etcd is not responding properly'))
|
||||
self._has_failed = False
|
||||
return cluster
|
||||
|
||||
@catch_etcd_errors
|
||||
def touch_member(self, data, ttl=None):
|
||||
return self.retry(self._client.set, self.member_path, data, ttl or self._ttl)
|
||||
def touch_member(self, data, permanent=False):
|
||||
data = json.dumps(data, separators=(',', ':'))
|
||||
return self._client.set(self.member_path, data, None if permanent else self._ttl)
|
||||
|
||||
@catch_etcd_errors
|
||||
def take_leader(self):
|
||||
return self.retry(self._client.set, self.leader_path, self._name, self._ttl)
|
||||
return self.retry(self._client.write, self.leader_path, self._name, ttl=self._ttl)
|
||||
|
||||
def attempt_to_acquire_leader(self):
|
||||
def attempt_to_acquire_leader(self, permanent=False):
|
||||
try:
|
||||
return bool(self.retry(self._client.write, self.leader_path, self._name, ttl=self._ttl, prevExist=False))
|
||||
return bool(self.retry(self._client.write,
|
||||
self.leader_path,
|
||||
self._name,
|
||||
ttl=None if permanent else self._ttl,
|
||||
prevExist=False))
|
||||
except etcd.EtcdAlreadyExist:
|
||||
logger.info('Could not take out TTL lock')
|
||||
except (RetryFailedError, etcd.EtcdException):
|
||||
@@ -306,19 +684,23 @@ class Etcd(AbstractDCS):
|
||||
return self._client.write(self.config_path, value, prevIndex=index or 0)
|
||||
|
||||
@catch_etcd_errors
|
||||
def write_leader_optime(self, last_operation):
|
||||
return self._client.set(self.leader_optime_path, last_operation)
|
||||
def _write_leader_optime(self, last_lsn):
|
||||
return self._client.set(self.leader_optime_path, last_lsn)
|
||||
|
||||
@catch_etcd_errors
|
||||
def update_leader(self):
|
||||
return self.retry(self._client.test_and_set, self.leader_path, self._name, self._name, self._ttl)
|
||||
def _write_status(self, value):
|
||||
return self._client.set(self.status_path, value)
|
||||
|
||||
@catch_etcd_errors
|
||||
def _update_leader(self):
|
||||
return self.retry(self._client.write, self.leader_path, self._name, prevValue=self._name, ttl=self._ttl)
|
||||
|
||||
@catch_etcd_errors
|
||||
def initialize(self, create_new=True, sysid=""):
|
||||
return self.retry(self._client.write, self.initialize_path, sysid, prevExist=(not create_new))
|
||||
|
||||
@catch_etcd_errors
|
||||
def delete_leader(self):
|
||||
def _delete_leader(self):
|
||||
return self._client.delete(self.leader_path, prevValue=self._name)
|
||||
|
||||
@catch_etcd_errors
|
||||
@@ -329,31 +711,50 @@ class Etcd(AbstractDCS):
|
||||
def delete_cluster(self):
|
||||
return self.retry(self._client.delete, self.client_path(''), recursive=True)
|
||||
|
||||
def watch(self, timeout):
|
||||
@catch_etcd_errors
|
||||
def set_history_value(self, value):
|
||||
return self._client.write(self.history_path, value)
|
||||
|
||||
@catch_etcd_errors
|
||||
def set_sync_state_value(self, value, index=None):
|
||||
return self.retry(self._client.write, self.sync_path, value, prevIndex=index or 0)
|
||||
|
||||
@catch_etcd_errors
|
||||
def delete_sync_state(self, index=None):
|
||||
return self.retry(self._client.delete, self.sync_path, prevIndex=index or 0)
|
||||
|
||||
def watch(self, leader_index, timeout):
|
||||
if self.__do_not_watch:
|
||||
self.__do_not_watch = False
|
||||
return True
|
||||
|
||||
cluster = self.cluster
|
||||
# watch on leader key changes if it is defined and current node is not lock owner
|
||||
if cluster and cluster.leader and cluster.leader.name != self._name and cluster.leader.index:
|
||||
if leader_index:
|
||||
end_time = time.time() + timeout
|
||||
|
||||
while timeout >= 1: # when timeout is too small urllib3 doesn't have enough time to connect
|
||||
try:
|
||||
self._client.watch(self.leader_path, index=cluster.leader.index + 1, timeout=timeout + 0.5)
|
||||
result = self._client.watch(self.leader_path, index=leader_index, timeout=timeout + 0.5)
|
||||
self._has_failed = False
|
||||
if result.action == 'compareAndSwap':
|
||||
time.sleep(0.01)
|
||||
# Synchronous work of all cluster members with etcd is less expensive
|
||||
# than reestablishing http connection every time from every replica.
|
||||
return True
|
||||
except etcd.EtcdWatchTimedOut:
|
||||
self._client.http.clear()
|
||||
self._has_failed = False
|
||||
return False
|
||||
except etcd.EtcdException:
|
||||
logging.exception('watch')
|
||||
except (etcd.EtcdEventIndexCleared, etcd.EtcdWatcherCleared): # Watch failed
|
||||
self._has_failed = False
|
||||
return True # leave the loop, because watch with the same parameters will fail anyway
|
||||
except etcd.EtcdException as e:
|
||||
self._handle_exception(e, 'watch', True)
|
||||
|
||||
timeout = end_time - time.time()
|
||||
|
||||
try:
|
||||
return super(Etcd, self).watch(timeout)
|
||||
return super(Etcd, self).watch(None, timeout)
|
||||
finally:
|
||||
self.event.clear()
|
||||
|
||||
|
||||
etcd.EtcdError.error_exceptions[300] = EtcdRaftInternal
|
||||
|
||||
@@ -0,0 +1,824 @@
|
||||
from __future__ import absolute_import
|
||||
import base64
|
||||
import etcd
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import six
|
||||
import socket
|
||||
import sys
|
||||
import time
|
||||
import urllib3
|
||||
|
||||
from threading import Condition, Lock, Thread
|
||||
|
||||
from . import ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
||||
from .etcd import AbstractEtcdClientWithFailover, AbstractEtcd, catch_etcd_errors
|
||||
from ..exceptions import DCSError, PatroniException
|
||||
from ..utils import deep_compare, enable_keepalive, iter_response_objects, RetryFailedError, USER_AGENT
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Etcd3Error(DCSError):
|
||||
pass
|
||||
|
||||
|
||||
class UnsupportedEtcdVersion(PatroniException):
|
||||
pass
|
||||
|
||||
|
||||
# google.golang.org/grpc/codes
|
||||
GRPCCode = type('Enum', (), {'OK': 0, 'Canceled': 1, 'Unknown': 2, 'InvalidArgument': 3, 'DeadlineExceeded': 4,
|
||||
'NotFound': 5, 'AlreadyExists': 6, 'PermissionDenied': 7, 'ResourceExhausted': 8,
|
||||
'FailedPrecondition': 9, 'Aborted': 10, 'OutOfRange': 11, 'Unimplemented': 12,
|
||||
'Internal': 13, 'Unavailable': 14, 'DataLoss': 15, 'Unauthenticated': 16})
|
||||
GRPCcodeToText = {v: k for k, v in GRPCCode.__dict__.items() if not k.startswith('__') and isinstance(v, int)}
|
||||
|
||||
|
||||
class Etcd3Exception(etcd.EtcdException):
|
||||
pass
|
||||
|
||||
|
||||
class Etcd3ClientError(Etcd3Exception):
|
||||
|
||||
def __init__(self, code=None, error=None, status=None):
|
||||
if not hasattr(self, 'error'):
|
||||
self.error = error and error.strip()
|
||||
self.codeText = GRPCcodeToText.get(code)
|
||||
self.status = status
|
||||
|
||||
def __repr__(self):
|
||||
return "<{0} error: '{1}', code: {2}>".format(self.__class__.__name__, self.error, self.code)
|
||||
|
||||
__str__ = __repr__
|
||||
|
||||
def as_dict(self):
|
||||
return {'error': self.error, 'code': self.code, 'codeText': self.codeText, 'status': self.status}
|
||||
|
||||
@classmethod
|
||||
def get_subclasses(cls):
|
||||
for subclass in cls.__subclasses__():
|
||||
for subsubclass in subclass.get_subclasses():
|
||||
yield subsubclass
|
||||
yield subclass
|
||||
|
||||
|
||||
class Unknown(Etcd3ClientError):
|
||||
code = GRPCCode.Unknown
|
||||
|
||||
|
||||
class InvalidArgument(Etcd3ClientError):
|
||||
code = GRPCCode.InvalidArgument
|
||||
|
||||
|
||||
class DeadlineExceeded(Etcd3ClientError):
|
||||
code = GRPCCode.DeadlineExceeded
|
||||
error = "context deadline exceeded"
|
||||
|
||||
|
||||
class NotFound(Etcd3ClientError):
|
||||
code = GRPCCode.NotFound
|
||||
|
||||
|
||||
class FailedPrecondition(Etcd3ClientError):
|
||||
code = GRPCCode.FailedPrecondition
|
||||
|
||||
|
||||
class Unavailable(Etcd3ClientError):
|
||||
code = GRPCCode.Unavailable
|
||||
|
||||
|
||||
# https://github.com/etcd-io/etcd/blob/master/etcdserver/api/v3rpc/rpctypes/error.go
|
||||
class LeaseNotFound(NotFound):
|
||||
error = "etcdserver: requested lease not found"
|
||||
|
||||
|
||||
class UserEmpty(InvalidArgument):
|
||||
error = "etcdserver: user name is empty"
|
||||
|
||||
|
||||
class AuthFailed(InvalidArgument):
|
||||
error = "etcdserver: authentication failed, invalid user ID or password"
|
||||
|
||||
|
||||
class PermissionDenied(Etcd3ClientError):
|
||||
code = GRPCCode.PermissionDenied
|
||||
error = "etcdserver: permission denied"
|
||||
|
||||
|
||||
class AuthNotEnabled(FailedPrecondition):
|
||||
error = "etcdserver: authentication is not enabled"
|
||||
|
||||
|
||||
class InvalidAuthToken(Etcd3ClientError):
|
||||
code = GRPCCode.Unauthenticated
|
||||
error = "etcdserver: invalid auth token"
|
||||
|
||||
|
||||
errStringToClientError = {s.error: s for s in Etcd3ClientError.get_subclasses() if hasattr(s, 'error')}
|
||||
errCodeToClientError = {s.code: s for s in Etcd3ClientError.__subclasses__()}
|
||||
|
||||
|
||||
def _raise_for_data(data, status_code=None):
|
||||
try:
|
||||
error = data.get('error') or data.get('Error')
|
||||
if isinstance(error, dict): # streaming response
|
||||
status_code = error.get('http_code')
|
||||
code = error['grpc_code']
|
||||
error = error['message']
|
||||
else:
|
||||
code = data.get('code') or data.get('Code')
|
||||
except Exception:
|
||||
error = str(data)
|
||||
code = GRPCCode.Unknown
|
||||
err = errStringToClientError.get(error) or errCodeToClientError.get(code) or Unknown
|
||||
raise err(code, error, status_code)
|
||||
|
||||
|
||||
def to_bytes(v):
|
||||
return v if isinstance(v, bytes) else v.encode('utf-8')
|
||||
|
||||
|
||||
def prefix_range_end(v):
|
||||
v = bytearray(to_bytes(v))
|
||||
for i in range(len(v) - 1, -1, -1):
|
||||
if v[i] < 0xff:
|
||||
v[i] += 1
|
||||
break
|
||||
return bytes(v)
|
||||
|
||||
|
||||
def base64_encode(v):
|
||||
return base64.b64encode(to_bytes(v)).decode('utf-8')
|
||||
|
||||
|
||||
def base64_decode(v):
|
||||
return base64.b64decode(v).decode('utf-8')
|
||||
|
||||
|
||||
def build_range_request(key, range_end=None):
|
||||
fields = {'key': base64_encode(key)}
|
||||
if range_end:
|
||||
fields['range_end'] = base64_encode(range_end)
|
||||
return fields
|
||||
|
||||
|
||||
class Etcd3Client(AbstractEtcdClientWithFailover):
|
||||
|
||||
ERROR_CLS = Etcd3Error
|
||||
|
||||
def __init__(self, config, dns_resolver, cache_ttl=300):
|
||||
self._token = None
|
||||
self._cluster_version = None
|
||||
self.version_prefix = '/v3beta'
|
||||
super(Etcd3Client, self).__init__(config, dns_resolver, cache_ttl)
|
||||
|
||||
if six.PY2: # pragma: no cover
|
||||
# Old grpc-gateway sometimes sends double 'transfer-encoding: chunked' headers,
|
||||
# what breaks the old (python2.7) httplib.HTTPConnection (it closes the socket).
|
||||
def dedup_addheader(httpm, key, value):
|
||||
prev = httpm.dict.get(key)
|
||||
if prev is None:
|
||||
httpm.dict[key] = value
|
||||
elif key != 'transfer-encoding' or prev != value:
|
||||
combined = ", ".join((prev, value))
|
||||
httpm.dict[key] = combined
|
||||
|
||||
import httplib
|
||||
httplib.HTTPMessage.addheader = dedup_addheader
|
||||
|
||||
try:
|
||||
self.authenticate()
|
||||
except AuthFailed as e:
|
||||
logger.fatal('Etcd3 authentication failed: %r', e)
|
||||
sys.exit(1)
|
||||
|
||||
def _get_headers(self):
|
||||
headers = urllib3.make_headers(user_agent=USER_AGENT)
|
||||
if self._token and self._cluster_version >= (3, 3, 0):
|
||||
headers['authorization'] = self._token
|
||||
return headers
|
||||
|
||||
def _prepare_request(self, kwargs, params=None, method=None):
|
||||
if params is not None:
|
||||
kwargs['body'] = json.dumps(params)
|
||||
kwargs['headers']['Content-Type'] = 'application/json'
|
||||
return self.http.urlopen
|
||||
|
||||
@staticmethod
|
||||
def _handle_server_response(response):
|
||||
data = response.data
|
||||
try:
|
||||
data = data.decode('utf-8')
|
||||
data = json.loads(data)
|
||||
except (TypeError, ValueError, UnicodeError) as e:
|
||||
if response.status < 400:
|
||||
raise etcd.EtcdException('Server response was not valid JSON: %r' % e)
|
||||
if response.status < 400:
|
||||
return data
|
||||
_raise_for_data(data, response.status)
|
||||
|
||||
def _ensure_version_prefix(self, base_uri, **kwargs):
|
||||
if self.version_prefix != '/v3':
|
||||
response = self.http.urlopen(self._MGET, base_uri + '/version', **kwargs)
|
||||
response = self._handle_server_response(response)
|
||||
|
||||
server_version_str = response['etcdserver']
|
||||
server_version = tuple(int(x) for x in server_version_str.split('.'))
|
||||
cluster_version_str = response['etcdcluster']
|
||||
self._cluster_version = tuple(int(x) for x in cluster_version_str.split('.'))
|
||||
|
||||
if self._cluster_version < (3, 0) or server_version < (3, 0, 4):
|
||||
raise UnsupportedEtcdVersion('Detected Etcd version {0} is lower than 3.0.4'.format(server_version_str))
|
||||
|
||||
if self._cluster_version < (3, 3):
|
||||
if self.version_prefix != '/v3alpha':
|
||||
if self._cluster_version < (3, 1):
|
||||
logger.warning('Detected Etcd version %s is lower than 3.1.0, watches are not supported',
|
||||
cluster_version_str)
|
||||
if self.username and self.password:
|
||||
logger.warning('Detected Etcd version %s is lower than 3.3.0, authentication is not supported',
|
||||
cluster_version_str)
|
||||
self.version_prefix = '/v3alpha'
|
||||
elif self._cluster_version < (3, 4):
|
||||
self.version_prefix = '/v3beta'
|
||||
else:
|
||||
self.version_prefix = '/v3'
|
||||
|
||||
def _prepare_get_members(self, etcd_nodes):
|
||||
kwargs = self._prepare_common_parameters(etcd_nodes)
|
||||
self._prepare_request(kwargs, {})
|
||||
return kwargs
|
||||
|
||||
def _get_members(self, base_uri, **kwargs):
|
||||
self._ensure_version_prefix(base_uri, **kwargs)
|
||||
resp = self.http.urlopen(self._MPOST, base_uri + self.version_prefix + '/cluster/member/list', **kwargs)
|
||||
members = self._handle_server_response(resp)['members']
|
||||
return set(url for member in members for url in member.get('clientURLs', []))
|
||||
|
||||
def call_rpc(self, method, fields, retry=None):
|
||||
fields['retry'] = retry
|
||||
return self.api_execute(self.version_prefix + method, self._MPOST, fields)
|
||||
|
||||
def authenticate(self):
|
||||
if self._use_proxies and self._cluster_version is None:
|
||||
kwargs = self._prepare_common_parameters(1)
|
||||
self._ensure_version_prefix(self._base_uri, **kwargs)
|
||||
if self._cluster_version >= (3, 3) and self.username and self.password:
|
||||
logger.info('Trying to authenticate on Etcd...')
|
||||
old_token, self._token = self._token, None
|
||||
try:
|
||||
response = self.call_rpc('/auth/authenticate', {'name': self.username, 'password': self.password})
|
||||
except AuthNotEnabled:
|
||||
logger.info('Etcd authentication is not enabled')
|
||||
self._token = None
|
||||
except Exception:
|
||||
self._token = old_token
|
||||
raise
|
||||
else:
|
||||
self._token = response.get('token')
|
||||
return old_token != self._token
|
||||
|
||||
def _handle_auth_errors(func):
|
||||
def wrapper(self, *args, **kwargs):
|
||||
def retry(ex):
|
||||
if self.username and self.password:
|
||||
self.authenticate()
|
||||
return func(self, *args, **kwargs)
|
||||
else:
|
||||
logger.fatal('Username or password not set, authentication is not possible')
|
||||
raise ex
|
||||
|
||||
try:
|
||||
return func(self, *args, **kwargs)
|
||||
except (UserEmpty, PermissionDenied) as e: # no token provided
|
||||
# PermissionDenied is raised on 3.0 and 3.1
|
||||
if self._cluster_version < (3, 3) and (not isinstance(e, PermissionDenied)
|
||||
or self._cluster_version < (3, 2)):
|
||||
raise UnsupportedEtcdVersion('Authentication is required by Etcd cluster but not '
|
||||
'supported on version lower than 3.3.0. Cluster version: '
|
||||
'{0}'.format('.'.join(map(str, self._cluster_version))))
|
||||
return retry(e)
|
||||
except InvalidAuthToken as e:
|
||||
logger.error('Invalid auth token: %s', self._token)
|
||||
return retry(e)
|
||||
|
||||
return wrapper
|
||||
|
||||
@_handle_auth_errors
|
||||
def range(self, key, range_end=None, retry=None):
|
||||
params = build_range_request(key, range_end)
|
||||
params['serializable'] = True # For better performance. We can tolerate stale reads.
|
||||
return self.call_rpc('/kv/range', params, retry)
|
||||
|
||||
def prefix(self, key, retry=None):
|
||||
return self.range(key, prefix_range_end(key), retry)
|
||||
|
||||
@_handle_auth_errors
|
||||
def lease_grant(self, ttl, retry=None):
|
||||
return self.call_rpc('/lease/grant', {'TTL': ttl}, retry)['ID']
|
||||
|
||||
def lease_keepalive(self, ID, retry=None):
|
||||
return self.call_rpc('/lease/keepalive', {'ID': ID}, retry).get('result', {}).get('TTL')
|
||||
|
||||
def txn(self, compare, success, retry=None):
|
||||
return self.call_rpc('/kv/txn', {'compare': [compare], 'success': [success]}, retry).get('succeeded')
|
||||
|
||||
@_handle_auth_errors
|
||||
def put(self, key, value, lease=None, create_revision=None, mod_revision=None, retry=None):
|
||||
fields = {'key': base64_encode(key), 'value': base64_encode(value)}
|
||||
if lease:
|
||||
fields['lease'] = lease
|
||||
if create_revision is not None:
|
||||
compare = {'target': 'CREATE', 'create_revision': create_revision}
|
||||
elif mod_revision is not None:
|
||||
compare = {'target': 'MOD', 'mod_revision': mod_revision}
|
||||
else:
|
||||
return self.call_rpc('/kv/put', fields, retry)
|
||||
compare['key'] = fields['key']
|
||||
return self.txn(compare, {'request_put': fields}, retry)
|
||||
|
||||
@_handle_auth_errors
|
||||
def deleterange(self, key, range_end=None, mod_revision=None, retry=None):
|
||||
fields = build_range_request(key, range_end)
|
||||
if mod_revision is None:
|
||||
return self.call_rpc('/kv/deleterange', fields, retry)
|
||||
compare = {'target': 'MOD', 'mod_revision': mod_revision, 'key': fields['key']}
|
||||
return self.txn(compare, {'request_delete_range': fields}, retry)
|
||||
|
||||
def deleteprefix(self, key, retry=None):
|
||||
return self.deleterange(key, prefix_range_end(key), retry=retry)
|
||||
|
||||
def watchrange(self, key, range_end=None, start_revision=None, filters=None):
|
||||
"""returns: response object"""
|
||||
params = build_range_request(key, range_end)
|
||||
if start_revision is not None:
|
||||
params['start_revision'] = start_revision
|
||||
params['filters'] = filters or []
|
||||
kwargs = self._prepare_common_parameters(1, self.read_timeout)
|
||||
request_executor = self._prepare_request(kwargs, {'create_request': params})
|
||||
kwargs.update(timeout=urllib3.Timeout(connect=kwargs['timeout']), retries=0)
|
||||
return request_executor(self._MPOST, self._base_uri + self.version_prefix + '/watch', **kwargs)
|
||||
|
||||
def watchprefix(self, key, start_revision=None, filters=None):
|
||||
return self.watchrange(key, prefix_range_end(key), start_revision, filters)
|
||||
|
||||
|
||||
class KVCache(Thread):
|
||||
|
||||
def __init__(self, dcs, client):
|
||||
Thread.__init__(self)
|
||||
self.daemon = True
|
||||
self._dcs = dcs
|
||||
self._client = client
|
||||
self.condition = Condition()
|
||||
self._config_key = base64_encode(dcs.config_path)
|
||||
self._leader_key = base64_encode(dcs.leader_path)
|
||||
self._optime_key = base64_encode(dcs.leader_optime_path)
|
||||
self._status_key = base64_encode(dcs.status_path)
|
||||
self._name = base64_encode(dcs._name)
|
||||
self._is_ready = False
|
||||
self._response = None
|
||||
self._response_lock = Lock()
|
||||
self._object_cache = {}
|
||||
self._object_cache_lock = Lock()
|
||||
self.start()
|
||||
|
||||
def set(self, value, overwrite=False):
|
||||
with self._object_cache_lock:
|
||||
name = value['key']
|
||||
old_value = self._object_cache.get(name)
|
||||
ret = not old_value or int(old_value['mod_revision']) < int(value['mod_revision'])
|
||||
if ret or overwrite and old_value['mod_revision'] == value['mod_revision']:
|
||||
self._object_cache[name] = value
|
||||
return ret, old_value
|
||||
|
||||
def delete(self, name, mod_revision):
|
||||
with self._object_cache_lock:
|
||||
old_value = self._object_cache.get(name)
|
||||
ret = old_value and int(old_value['mod_revision']) < int(mod_revision)
|
||||
if ret:
|
||||
del self._object_cache[name]
|
||||
return not old_value or ret, old_value
|
||||
|
||||
def copy(self):
|
||||
with self._object_cache_lock:
|
||||
return [v.copy() for v in self._object_cache.values()]
|
||||
|
||||
def get(self, name):
|
||||
with self._object_cache_lock:
|
||||
return self._object_cache.get(name)
|
||||
|
||||
def _process_event(self, event):
|
||||
kv = event['kv']
|
||||
key = kv['key']
|
||||
if event.get('type') == 'DELETE':
|
||||
success, old_value = self.delete(key, kv['mod_revision'])
|
||||
else:
|
||||
success, old_value = self.set(kv, True)
|
||||
|
||||
if success:
|
||||
old_value = old_value and old_value.get('value')
|
||||
new_value = kv.get('value')
|
||||
|
||||
value_changed = old_value != new_value and \
|
||||
(key == self._leader_key or key in (self._optime_key, self._status_key) and new_value is not None or
|
||||
key == self._config_key and old_value is not None and new_value is not None)
|
||||
|
||||
if value_changed:
|
||||
logger.debug('%s changed from %s to %s', key, old_value, new_value)
|
||||
|
||||
# We also want to wake up HA loop on replicas if leader optime (or status key) was updated
|
||||
if value_changed and (key not in (self._optime_key, self._status_key) or
|
||||
(self.get(self._leader_key) or {}).get('value') != self._name):
|
||||
self._dcs.event.set()
|
||||
|
||||
def _process_message(self, message):
|
||||
logger.debug('Received message: %s', message)
|
||||
if 'error' in message:
|
||||
_raise_for_data(message)
|
||||
for event in message.get('result', {}).get('events', []):
|
||||
self._process_event(event)
|
||||
|
||||
@staticmethod
|
||||
def _finish_response(response):
|
||||
try:
|
||||
response.close()
|
||||
finally:
|
||||
response.release_conn()
|
||||
|
||||
def _do_watch(self, revision):
|
||||
with self._response_lock:
|
||||
self._response = None
|
||||
response = self._client.watchprefix(self._dcs.cluster_prefix, revision)
|
||||
with self._response_lock:
|
||||
if self._response is None:
|
||||
self._response = response
|
||||
|
||||
if not self._response:
|
||||
return self._finish_response(response)
|
||||
|
||||
for message in iter_response_objects(response):
|
||||
self._process_message(message)
|
||||
|
||||
def _build_cache(self):
|
||||
result = self._dcs.retry(self._client.prefix, self._dcs.cluster_prefix)
|
||||
with self._object_cache_lock:
|
||||
self._object_cache = {node['key']: node for node in result.get('kvs', [])}
|
||||
with self.condition:
|
||||
self._is_ready = True
|
||||
self.condition.notify()
|
||||
|
||||
try:
|
||||
self._do_watch(result['header']['revision'])
|
||||
except Exception as e:
|
||||
logger.error('watchprefix failed: %r', e)
|
||||
finally:
|
||||
with self.condition:
|
||||
self._is_ready = False
|
||||
with self._response_lock:
|
||||
response, self._response = self._response, None
|
||||
if response:
|
||||
self._finish_response(response)
|
||||
|
||||
def run(self):
|
||||
while True:
|
||||
try:
|
||||
self._build_cache()
|
||||
except Exception as e:
|
||||
logger.error('KVCache.run %r', e)
|
||||
time.sleep(1)
|
||||
|
||||
def kill_stream(self):
|
||||
sock = None
|
||||
with self._response_lock:
|
||||
if self._response:
|
||||
try:
|
||||
sock = self._response.connection.sock
|
||||
except Exception:
|
||||
sock = None
|
||||
else:
|
||||
self._response = False
|
||||
if sock:
|
||||
try:
|
||||
sock.shutdown(socket.SHUT_RDWR)
|
||||
sock.close()
|
||||
except Exception as e:
|
||||
logger.debug('Error on socket.shutdown: %r', e)
|
||||
|
||||
def is_ready(self):
|
||||
"""Must be called only when holding the lock on `condition`"""
|
||||
return self._is_ready
|
||||
|
||||
|
||||
class PatroniEtcd3Client(Etcd3Client):
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
self._kv_cache = None
|
||||
super(PatroniEtcd3Client, self).__init__(*args, **kwargs)
|
||||
|
||||
def configure(self, etcd3):
|
||||
self._etcd3 = etcd3
|
||||
|
||||
def start_watcher(self):
|
||||
if self._cluster_version >= (3, 1):
|
||||
self._kv_cache = KVCache(self._etcd3, self)
|
||||
|
||||
def _restart_watcher(self):
|
||||
if self._kv_cache:
|
||||
self._kv_cache.kill_stream()
|
||||
|
||||
def set_base_uri(self, value):
|
||||
super(PatroniEtcd3Client, self).set_base_uri(value)
|
||||
self._restart_watcher()
|
||||
|
||||
def authenticate(self):
|
||||
ret = super(PatroniEtcd3Client, self).authenticate()
|
||||
if ret:
|
||||
self._restart_watcher()
|
||||
return ret
|
||||
|
||||
def _wait_cache(self, timeout):
|
||||
stop_time = time.time() + timeout
|
||||
while not self._kv_cache.is_ready():
|
||||
timeout = stop_time - time.time()
|
||||
if timeout <= 0:
|
||||
raise RetryFailedError('Exceeded retry deadline')
|
||||
self._kv_cache.condition.wait(timeout)
|
||||
|
||||
def get_cluster(self):
|
||||
if self._kv_cache:
|
||||
with self._kv_cache.condition:
|
||||
self._wait_cache(self._etcd3._retry.deadline)
|
||||
return self._kv_cache.copy()
|
||||
else:
|
||||
return self._etcd3.retry(self.prefix, self._etcd3.cluster_prefix).get('kvs', [])
|
||||
|
||||
def call_rpc(self, method, fields, retry=None):
|
||||
ret = super(PatroniEtcd3Client, self).call_rpc(method, fields, retry)
|
||||
|
||||
if self._kv_cache:
|
||||
value = delete = None
|
||||
if method == '/kv/txn' and ret.get('succeeded'):
|
||||
on_success = fields['success'][0]
|
||||
value = on_success.get('request_put')
|
||||
delete = on_success.get('request_delete_range')
|
||||
elif method == '/kv/put' and ret:
|
||||
value = fields
|
||||
elif method == '/kv/deleterange' and ret:
|
||||
delete = fields
|
||||
|
||||
if value:
|
||||
value['mod_revision'] = ret['header']['revision']
|
||||
self._kv_cache.set(value)
|
||||
elif delete and 'range_end' not in delete:
|
||||
self._kv_cache.delete(delete['key'], ret['header']['revision'])
|
||||
|
||||
return ret
|
||||
|
||||
|
||||
class Etcd3(AbstractEtcd):
|
||||
|
||||
def __init__(self, config):
|
||||
super(Etcd3, self).__init__(config, PatroniEtcd3Client, (DeadlineExceeded, Unavailable, FailedPrecondition))
|
||||
self.__do_not_watch = False
|
||||
self._lease = None
|
||||
self._last_lease_refresh = 0
|
||||
|
||||
self._client.configure(self)
|
||||
if not self._ctl:
|
||||
self._client.start_watcher()
|
||||
self.create_lease()
|
||||
|
||||
def set_socket_options(self, sock, socket_options):
|
||||
enable_keepalive(sock, self.ttl, int(self.loop_wait + self._retry.deadline))
|
||||
|
||||
def set_ttl(self, ttl):
|
||||
self.__do_not_watch = super(Etcd3, self).set_ttl(ttl)
|
||||
if self.__do_not_watch:
|
||||
self._lease = None
|
||||
|
||||
def _do_refresh_lease(self, retry=None):
|
||||
if self._lease and self._last_lease_refresh + self._loop_wait > time.time():
|
||||
return False
|
||||
|
||||
if self._lease and not self._client.lease_keepalive(self._lease, retry):
|
||||
self._lease = None
|
||||
|
||||
ret = not self._lease
|
||||
if ret:
|
||||
self._lease = self._client.lease_grant(self._ttl, retry)
|
||||
|
||||
self._last_lease_refresh = time.time()
|
||||
return ret
|
||||
|
||||
def refresh_lease(self):
|
||||
try:
|
||||
return self.retry(self._do_refresh_lease)
|
||||
except (Etcd3ClientError, RetryFailedError):
|
||||
logger.exception('refresh_lease')
|
||||
raise Etcd3Error('Failed to keepalive/grant lease')
|
||||
|
||||
def create_lease(self):
|
||||
while not self._lease:
|
||||
try:
|
||||
self.refresh_lease()
|
||||
except Etcd3Error:
|
||||
logger.info('waiting on etcd')
|
||||
time.sleep(5)
|
||||
|
||||
@property
|
||||
def cluster_prefix(self):
|
||||
return self.client_path('')
|
||||
|
||||
@staticmethod
|
||||
def member(node):
|
||||
return Member.from_node(node['mod_revision'], os.path.basename(node['key']), node['lease'], node['value'])
|
||||
|
||||
def _load_cluster(self):
|
||||
cluster = None
|
||||
try:
|
||||
path_len = len(self.cluster_prefix)
|
||||
|
||||
nodes = {}
|
||||
for node in self._client.get_cluster():
|
||||
node['key'] = base64_decode(node['key'])
|
||||
node['value'] = base64_decode(node.get('value', ''))
|
||||
node['lease'] = node.get('lease')
|
||||
nodes[node['key'][path_len:].lstrip('/')] = node
|
||||
|
||||
# get initialize flag
|
||||
initialize = nodes.get(self._INITIALIZE)
|
||||
initialize = initialize and initialize['value']
|
||||
|
||||
# get global dynamic configuration
|
||||
config = nodes.get(self._CONFIG)
|
||||
config = config and ClusterConfig.from_node(config['mod_revision'], config['value'])
|
||||
|
||||
# get timeline history
|
||||
history = nodes.get(self._HISTORY)
|
||||
history = history and TimelineHistory.from_node(history['mod_revision'], history['value'])
|
||||
|
||||
# get last know leader lsn and slots
|
||||
status = nodes.get(self._STATUS)
|
||||
if status:
|
||||
try:
|
||||
status = json.loads(status['value'])
|
||||
last_lsn = status.get(self._OPTIME)
|
||||
slots = status.get('slots')
|
||||
except Exception:
|
||||
slots = last_lsn = None
|
||||
else:
|
||||
last_lsn = nodes.get(self._LEADER_OPTIME)
|
||||
last_lsn = last_lsn and last_lsn['value']
|
||||
slots = None
|
||||
|
||||
try:
|
||||
last_lsn = int(last_lsn)
|
||||
except Exception:
|
||||
last_lsn = 0
|
||||
|
||||
# get list of members
|
||||
members = [self.member(n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
||||
|
||||
# get leader
|
||||
leader = nodes.get(self._LEADER)
|
||||
if not self._ctl and leader and leader['value'] == self._name and self._lease != leader.get('lease'):
|
||||
logger.warning('I am the leader but not owner of the lease')
|
||||
|
||||
if leader:
|
||||
member = Member(-1, leader['value'], None, {})
|
||||
member = ([m for m in members if m.name == leader['value']] or [member])[0]
|
||||
leader = Leader(leader['mod_revision'], leader['lease'], member)
|
||||
|
||||
# failover key
|
||||
failover = nodes.get(self._FAILOVER)
|
||||
if failover:
|
||||
failover = Failover.from_node(failover['mod_revision'], failover['value'])
|
||||
|
||||
# get synchronization state
|
||||
sync = nodes.get(self._SYNC)
|
||||
sync = SyncState.from_node(sync and sync['mod_revision'], sync and sync['value'])
|
||||
|
||||
cluster = Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||
except UnsupportedEtcdVersion:
|
||||
raise
|
||||
except Exception as e:
|
||||
self._handle_exception(e, 'get_cluster', raise_ex=Etcd3Error('Etcd is not responding properly'))
|
||||
self._has_failed = False
|
||||
return cluster
|
||||
|
||||
@catch_etcd_errors
|
||||
def touch_member(self, data, permanent=False):
|
||||
if not permanent:
|
||||
try:
|
||||
self.refresh_lease()
|
||||
except Etcd3Error:
|
||||
return False
|
||||
|
||||
cluster = self.cluster
|
||||
member = cluster and cluster.get_member(self._name, fallback_to_leader=False)
|
||||
|
||||
if member and member.session == self._lease and deep_compare(data, member.data):
|
||||
return True
|
||||
|
||||
data = json.dumps(data, separators=(',', ':'))
|
||||
try:
|
||||
return self._client.put(self.member_path, data, None if permanent else self._lease)
|
||||
except LeaseNotFound:
|
||||
self._lease = None
|
||||
logger.error('Our lease disappeared from Etcd, can not "touch_member"')
|
||||
|
||||
@catch_etcd_errors
|
||||
def take_leader(self):
|
||||
return self.retry(self._client.put, self.leader_path, self._name, self._lease)
|
||||
|
||||
@catch_etcd_errors
|
||||
def _do_attempt_to_acquire_leader(self, permanent):
|
||||
try:
|
||||
return self.retry(self._client.put, self.leader_path, self._name, None if permanent else self._lease, 0)
|
||||
except LeaseNotFound:
|
||||
self._lease = None
|
||||
logger.error('Our lease disappeared from Etcd. Will try to get a new one and retry attempt')
|
||||
self.refresh_lease()
|
||||
return self.retry(self._client.put, self.leader_path, self._name, None if permanent else self._lease, 0)
|
||||
|
||||
def attempt_to_acquire_leader(self, permanent=False):
|
||||
if not self._lease and not permanent:
|
||||
self.refresh_lease()
|
||||
|
||||
ret = self._do_attempt_to_acquire_leader(permanent)
|
||||
if not ret:
|
||||
logger.info('Could not take out TTL lock')
|
||||
return ret
|
||||
|
||||
@catch_etcd_errors
|
||||
def set_failover_value(self, value, index=None):
|
||||
return self._client.put(self.failover_path, value, mod_revision=index)
|
||||
|
||||
@catch_etcd_errors
|
||||
def set_config_value(self, value, index=None):
|
||||
return self._client.put(self.config_path, value, mod_revision=index)
|
||||
|
||||
@catch_etcd_errors
|
||||
def _write_leader_optime(self, last_lsn):
|
||||
return self._client.put(self.leader_optime_path, last_lsn)
|
||||
|
||||
@catch_etcd_errors
|
||||
def _write_status(self, value):
|
||||
return self._client.put(self.status_path, value)
|
||||
|
||||
@catch_etcd_errors
|
||||
def _update_leader(self):
|
||||
if not self._lease:
|
||||
self.refresh_lease()
|
||||
elif self.retry(self._client.lease_keepalive, self._lease):
|
||||
self._last_lease_refresh = time.time()
|
||||
|
||||
if self._lease:
|
||||
cluster = self.cluster
|
||||
leader_lease = cluster and isinstance(cluster.leader, Leader) and cluster.leader.session
|
||||
if leader_lease != self._lease:
|
||||
self.take_leader()
|
||||
return bool(self._lease)
|
||||
|
||||
@catch_etcd_errors
|
||||
def initialize(self, create_new=True, sysid=""):
|
||||
return self.retry(self._client.put, self.initialize_path, sysid, None, 0 if create_new else None)
|
||||
|
||||
@catch_etcd_errors
|
||||
def _delete_leader(self):
|
||||
cluster = self.cluster
|
||||
if cluster and isinstance(cluster.leader, Leader) and cluster.leader.name == self._name:
|
||||
return self._client.deleterange(self.leader_path, mod_revision=cluster.leader.index)
|
||||
|
||||
@catch_etcd_errors
|
||||
def cancel_initialization(self):
|
||||
return self.retry(self._client.deleterange, self.initialize_path)
|
||||
|
||||
@catch_etcd_errors
|
||||
def delete_cluster(self):
|
||||
return self.retry(self._client.deleteprefix, self.cluster_prefix)
|
||||
|
||||
@catch_etcd_errors
|
||||
def set_history_value(self, value):
|
||||
return self._client.put(self.history_path, value)
|
||||
|
||||
@catch_etcd_errors
|
||||
def set_sync_state_value(self, value, index=None):
|
||||
return self.retry(self._client.put, self.sync_path, value, mod_revision=index)
|
||||
|
||||
@catch_etcd_errors
|
||||
def delete_sync_state(self, index=None):
|
||||
return self.retry(self._client.deleterange, self.sync_path, mod_revision=index)
|
||||
|
||||
def watch(self, leader_index, timeout):
|
||||
if self.__do_not_watch:
|
||||
self.__do_not_watch = False
|
||||
return True
|
||||
|
||||
try:
|
||||
return super(Etcd3, self).watch(None, timeout)
|
||||
finally:
|
||||
self.event.clear()
|
||||
@@ -1,11 +1,11 @@
|
||||
import json
|
||||
import logging
|
||||
import random
|
||||
import requests
|
||||
import time
|
||||
|
||||
from patroni.dcs.zookeeper import ZooKeeper
|
||||
from patroni.utils import sleep
|
||||
from requests.exceptions import RequestException
|
||||
from patroni.request import get as requests_get
|
||||
from patroni.utils import uri
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -24,7 +24,7 @@ class ExhibitorEnsembleProvider(object):
|
||||
self._next_poll = None
|
||||
while not self.poll():
|
||||
logger.info('waiting on exhibitor')
|
||||
sleep(5)
|
||||
time.sleep(5)
|
||||
|
||||
def poll(self):
|
||||
if self._next_poll and self._next_poll > time.time():
|
||||
@@ -47,12 +47,11 @@ class ExhibitorEnsembleProvider(object):
|
||||
def _query_exhibitors(self, exhibitors):
|
||||
random.shuffle(exhibitors)
|
||||
for host in exhibitors:
|
||||
uri = 'http://{0}:{1}{2}'.format(host, self._exhibitor_port, self._uri_path)
|
||||
try:
|
||||
response = requests.get(uri, timeout=self.TIMEOUT)
|
||||
return response.json()
|
||||
except RequestException:
|
||||
pass
|
||||
response = requests_get(uri('http', (host, self._exhibitor_port), self._uri_path), timeout=self.TIMEOUT)
|
||||
return json.loads(response.data.decode('utf-8'))
|
||||
except Exception:
|
||||
logging.debug('Request to %s failed', host)
|
||||
return None
|
||||
|
||||
@property
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,423 @@
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
|
||||
from pysyncobj import SyncObj, SyncObjConf, replicated, FAIL_REASON
|
||||
from pysyncobj.dns_resolver import globalDnsResolver
|
||||
from pysyncobj.node import TCPNode
|
||||
from pysyncobj.transport import TCPTransport, CONNECTION_STATE
|
||||
from pysyncobj.utility import TcpUtility
|
||||
|
||||
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
||||
from ..utils import validate_directory
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class _TCPTransport(TCPTransport):
|
||||
|
||||
def __init__(self, syncObj, selfNode, otherNodes):
|
||||
super(_TCPTransport, self).__init__(syncObj, selfNode, otherNodes)
|
||||
self.setOnUtilityMessageCallback('members', syncObj.getMembers)
|
||||
|
||||
def _connectIfNecessarySingle(self, node):
|
||||
try:
|
||||
return super(_TCPTransport, self)._connectIfNecessarySingle(node)
|
||||
except Exception as e:
|
||||
logger.debug('Connection to %s failed: %r', node, e)
|
||||
return False
|
||||
|
||||
|
||||
def resolve_host(self):
|
||||
return globalDnsResolver().resolve(self.host)
|
||||
|
||||
|
||||
setattr(TCPNode, 'ip', property(resolve_host))
|
||||
|
||||
|
||||
class SyncObjUtility(object):
|
||||
|
||||
def __init__(self, otherNodes, conf):
|
||||
self._nodes = otherNodes
|
||||
self._utility = TcpUtility(conf.password)
|
||||
|
||||
def executeCommand(self, command):
|
||||
try:
|
||||
return self._utility.executeCommand(self.__node, command)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
def getMembers(self):
|
||||
for self.__node in self._nodes:
|
||||
response = self.executeCommand(['members'])
|
||||
if response:
|
||||
return [member['addr'] for member in response]
|
||||
|
||||
|
||||
class DynMemberSyncObj(SyncObj):
|
||||
|
||||
def __init__(self, selfAddress, partnerAddrs, conf):
|
||||
self.__early_apply_local_log = selfAddress is not None
|
||||
self.applied_local_log = False
|
||||
|
||||
utility = SyncObjUtility(partnerAddrs, conf)
|
||||
members = utility.getMembers()
|
||||
add_self = members and selfAddress not in members
|
||||
|
||||
partnerAddrs = [member for member in (members or partnerAddrs) if member != selfAddress]
|
||||
|
||||
super(DynMemberSyncObj, self).__init__(selfAddress, partnerAddrs, conf, transportClass=_TCPTransport)
|
||||
|
||||
if add_self:
|
||||
thread = threading.Thread(target=utility.executeCommand, args=(['add', selfAddress],))
|
||||
thread.daemon = True
|
||||
thread.start()
|
||||
|
||||
def getMembers(self, args, callback):
|
||||
callback([{'addr': node.id, 'leader': node == self._getLeader(), 'status': CONNECTION_STATE.CONNECTED
|
||||
if self.isNodeConnected(node) else CONNECTION_STATE.DISCONNECTED} for node in self.otherNodes] +
|
||||
[{'addr': self.selfNode.id, 'leader': self._isLeader(), 'status': CONNECTION_STATE.CONNECTED}], None)
|
||||
|
||||
def _onTick(self, timeToWait=0.0):
|
||||
super(DynMemberSyncObj, self)._onTick(timeToWait)
|
||||
|
||||
# The SyncObj calls onReady callback only when cluster got the leader and is ready for writes.
|
||||
# In some cases for us it is safe to "signal" the Raft object when the local log is fully applied.
|
||||
# We are using the `applied_local_log` property for that, but not calling the callback function.
|
||||
if self.__early_apply_local_log and not self.applied_local_log and self.raftLastApplied == self.raftCommitIndex:
|
||||
self.applied_local_log = True
|
||||
|
||||
|
||||
class KVStoreTTL(DynMemberSyncObj):
|
||||
|
||||
def __init__(self, on_ready, on_set, on_delete, **config):
|
||||
self.__thread = None
|
||||
self.__on_set = on_set
|
||||
self.__on_delete = on_delete
|
||||
self.__limb = {}
|
||||
self.__retry_timeout = None
|
||||
|
||||
self_addr = config.get('self_addr')
|
||||
partner_addrs = set(config.get('partner_addrs', []))
|
||||
if config.get('patronictl'):
|
||||
if self_addr:
|
||||
partner_addrs.add(self_addr)
|
||||
self_addr = None
|
||||
|
||||
# Create raft data_dir if necessary
|
||||
raft_data_dir = config.get('data_dir', '')
|
||||
if raft_data_dir != '':
|
||||
validate_directory(raft_data_dir)
|
||||
|
||||
file_template = (self_addr or '')
|
||||
file_template = file_template.replace(':', '_') if os.name == 'nt' else file_template
|
||||
file_template = os.path.join(raft_data_dir, file_template)
|
||||
conf = SyncObjConf(password=config.get('password'), autoTick=False, appendEntriesUseBatch=False,
|
||||
bindAddress=config.get('bind_addr'), dnsFailCacheTime=(config.get('loop_wait') or 10),
|
||||
dnsCacheTime=(config.get('ttl') or 30), commandsWaitLeader=config.get('commandsWaitLeader'),
|
||||
fullDumpFile=(file_template + '.dump' if self_addr else None),
|
||||
journalFile=(file_template + '.journal' if self_addr else None),
|
||||
onReady=on_ready, dynamicMembershipChange=True)
|
||||
|
||||
super(KVStoreTTL, self).__init__(self_addr, partner_addrs, conf)
|
||||
self.__data = {}
|
||||
|
||||
@staticmethod
|
||||
def __check_requirements(old_value, **kwargs):
|
||||
return ('prevExist' not in kwargs or bool(kwargs['prevExist']) == bool(old_value)) and \
|
||||
('prevValue' not in kwargs or old_value and old_value['value'] == kwargs['prevValue']) and \
|
||||
(not kwargs.get('prevIndex') or old_value and old_value['index'] == kwargs['prevIndex'])
|
||||
|
||||
def set_retry_timeout(self, retry_timeout):
|
||||
self.__retry_timeout = retry_timeout
|
||||
|
||||
def retry(self, func, *args, **kwargs):
|
||||
event = threading.Event()
|
||||
ret = {'result': None, 'error': -1}
|
||||
|
||||
def callback(result, error):
|
||||
ret.update(result=result, error=error)
|
||||
event.set()
|
||||
|
||||
kwargs['callback'] = callback
|
||||
timeout = kwargs.pop('timeout', None) or self.__retry_timeout
|
||||
deadline = timeout and time.time() + timeout
|
||||
|
||||
while True:
|
||||
event.clear()
|
||||
func(*args, **kwargs)
|
||||
event.wait(timeout)
|
||||
if ret['error'] == FAIL_REASON.SUCCESS:
|
||||
return ret['result']
|
||||
elif ret['error'] == FAIL_REASON.REQUEST_DENIED:
|
||||
break
|
||||
elif deadline:
|
||||
timeout = deadline - time.time()
|
||||
if timeout <= 0:
|
||||
break
|
||||
time.sleep(1)
|
||||
return False
|
||||
|
||||
@replicated
|
||||
def _set(self, key, value, **kwargs):
|
||||
old_value = self.__data.get(key, {})
|
||||
if not self.__check_requirements(old_value, **kwargs):
|
||||
return False
|
||||
|
||||
if old_value and old_value['created'] != value['created']:
|
||||
value['created'] = value['updated']
|
||||
value['index'] = self.raftLastApplied + 1
|
||||
|
||||
self.__data[key] = value
|
||||
if self.__on_set:
|
||||
self.__on_set(key, value)
|
||||
return True
|
||||
|
||||
def set(self, key, value, ttl=None, **kwargs):
|
||||
old_value = self.__data.get(key, {})
|
||||
if not self.__check_requirements(old_value, **kwargs):
|
||||
return False
|
||||
|
||||
value = {'value': value, 'updated': time.time()}
|
||||
value['created'] = old_value.get('created', value['updated'])
|
||||
if ttl:
|
||||
value['expire'] = value['updated'] + ttl
|
||||
return self.retry(self._set, key, value, **kwargs)
|
||||
|
||||
def __pop(self, key):
|
||||
self.__data.pop(key)
|
||||
if self.__on_delete:
|
||||
self.__on_delete(key)
|
||||
|
||||
@replicated
|
||||
def _delete(self, key, recursive=False, **kwargs):
|
||||
if recursive:
|
||||
for k in list(self.__data.keys()):
|
||||
if k.startswith(key):
|
||||
self.__pop(k)
|
||||
elif not self.__check_requirements(self.__data.get(key, {}), **kwargs):
|
||||
return False
|
||||
else:
|
||||
self.__pop(key)
|
||||
return True
|
||||
|
||||
def delete(self, key, recursive=False, **kwargs):
|
||||
if not recursive and not self.__check_requirements(self.__data.get(key, {}), **kwargs):
|
||||
return False
|
||||
return self.retry(self._delete, key, recursive=recursive, **kwargs)
|
||||
|
||||
@staticmethod
|
||||
def __values_match(old, new):
|
||||
return all(old.get(n) == new.get(n) for n in ('created', 'updated', 'expire', 'value'))
|
||||
|
||||
@replicated
|
||||
def _expire(self, key, value, callback=None):
|
||||
current = self.__data.get(key)
|
||||
if current and self.__values_match(current, value):
|
||||
self.__pop(key)
|
||||
|
||||
def __expire_keys(self):
|
||||
for key, value in self.__data.items():
|
||||
if value and 'expire' in value and value['expire'] <= time.time() and \
|
||||
not (key in self.__limb and self.__values_match(self.__limb[key], value)):
|
||||
self.__limb[key] = value
|
||||
|
||||
def callback(*args):
|
||||
if key in self.__limb and self.__values_match(self.__limb[key], value):
|
||||
self.__limb.pop(key)
|
||||
self._expire(key, value, callback=callback)
|
||||
|
||||
def get(self, key, recursive=False):
|
||||
if not recursive:
|
||||
return self.__data.get(key)
|
||||
return {k: v for k, v in self.__data.items() if k.startswith(key)}
|
||||
|
||||
def _onTick(self, timeToWait=0.0):
|
||||
super(KVStoreTTL, self)._onTick(timeToWait)
|
||||
|
||||
if self._isLeader():
|
||||
self.__expire_keys()
|
||||
else:
|
||||
self.__limb.clear()
|
||||
|
||||
def _autoTickThread(self):
|
||||
self.__destroying = False
|
||||
while not self.__destroying:
|
||||
self.doTick(self.conf.autoTickPeriod)
|
||||
|
||||
def startAutoTick(self):
|
||||
self.__thread = threading.Thread(target=self._autoTickThread)
|
||||
self.__thread.daemon = True
|
||||
self.__thread.start()
|
||||
|
||||
def destroy(self):
|
||||
if self.__thread:
|
||||
self.__destroying = True
|
||||
self.__thread.join()
|
||||
super(KVStoreTTL, self).destroy()
|
||||
|
||||
|
||||
class Raft(AbstractDCS):
|
||||
|
||||
def __init__(self, config):
|
||||
super(Raft, self).__init__(config)
|
||||
self._ttl = int(config.get('ttl') or 30)
|
||||
|
||||
ready_event = threading.Event()
|
||||
self._sync_obj = KVStoreTTL(ready_event.set, self._on_set, self._on_delete, commandsWaitLeader=False, **config)
|
||||
self._sync_obj.startAutoTick()
|
||||
|
||||
while True:
|
||||
ready_event.wait(5)
|
||||
if ready_event.is_set() or self._sync_obj.applied_local_log:
|
||||
break
|
||||
else:
|
||||
logger.info('waiting on raft')
|
||||
self.set_retry_timeout(int(config.get('retry_timeout') or 10))
|
||||
|
||||
def _on_set(self, key, value):
|
||||
leader = (self._sync_obj.get(self.leader_path) or {}).get('value')
|
||||
if key == value['created'] == value['updated'] and \
|
||||
(key.startswith(self.members_path) or key == self.leader_path and leader != self._name) or \
|
||||
key in (self.leader_optime_path, self.status_path) and leader != self._name or \
|
||||
key in (self.config_path, self.sync_path):
|
||||
self.event.set()
|
||||
|
||||
def _on_delete(self, key):
|
||||
if key == self.leader_path:
|
||||
self.event.set()
|
||||
|
||||
def set_ttl(self, ttl):
|
||||
self._ttl = ttl
|
||||
|
||||
@property
|
||||
def ttl(self):
|
||||
return self._ttl
|
||||
|
||||
def set_retry_timeout(self, retry_timeout):
|
||||
self._sync_obj.set_retry_timeout(retry_timeout)
|
||||
|
||||
def reload_config(self, config):
|
||||
super(Raft, self).reload_config(config)
|
||||
globalDnsResolver().setTimeouts(self.ttl, self.loop_wait)
|
||||
|
||||
@staticmethod
|
||||
def member(key, value):
|
||||
return Member.from_node(value['index'], os.path.basename(key), None, value['value'])
|
||||
|
||||
def _load_cluster(self):
|
||||
prefix = self.client_path('')
|
||||
response = self._sync_obj.get(prefix, recursive=True)
|
||||
if not response:
|
||||
return Cluster(None, None, None, None, [], None, None, None, None)
|
||||
nodes = {os.path.relpath(key, prefix).replace('\\', '/'): value for key, value in response.items()}
|
||||
|
||||
# get initialize flag
|
||||
initialize = nodes.get(self._INITIALIZE)
|
||||
initialize = initialize and initialize['value']
|
||||
|
||||
# get global dynamic configuration
|
||||
config = nodes.get(self._CONFIG)
|
||||
config = config and ClusterConfig.from_node(config['index'], config['value'])
|
||||
|
||||
# get timeline history
|
||||
history = nodes.get(self._HISTORY)
|
||||
history = history and TimelineHistory.from_node(history['index'], history['value'])
|
||||
|
||||
# get last know leader lsn and slots
|
||||
status = nodes.get(self._STATUS)
|
||||
if status:
|
||||
try:
|
||||
status = json.loads(status['value'])
|
||||
last_lsn = status.get(self._OPTIME)
|
||||
slots = status.get('slots')
|
||||
except Exception:
|
||||
slots = last_lsn = None
|
||||
else:
|
||||
last_lsn = nodes.get(self._LEADER_OPTIME)
|
||||
last_lsn = last_lsn and last_lsn['value']
|
||||
slots = None
|
||||
|
||||
try:
|
||||
last_lsn = int(last_lsn)
|
||||
except Exception:
|
||||
last_lsn = 0
|
||||
|
||||
# get list of members
|
||||
members = [self.member(k, n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
||||
|
||||
# get leader
|
||||
leader = nodes.get(self._LEADER)
|
||||
if leader:
|
||||
member = Member(-1, leader['value'], None, {})
|
||||
member = ([m for m in members if m.name == leader['value']] or [member])[0]
|
||||
leader = Leader(leader['index'], None, member)
|
||||
|
||||
# failover key
|
||||
failover = nodes.get(self._FAILOVER)
|
||||
if failover:
|
||||
failover = Failover.from_node(failover['index'], failover['value'])
|
||||
|
||||
# get synchronization state
|
||||
sync = nodes.get(self._SYNC)
|
||||
sync = SyncState.from_node(sync and sync['index'], sync and sync['value'])
|
||||
|
||||
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||
|
||||
def _write_leader_optime(self, last_lsn):
|
||||
return self._sync_obj.set(self.leader_optime_path, last_lsn, timeout=1)
|
||||
|
||||
def _write_status(self, value):
|
||||
return self._sync_obj.set(self.status_path, value, timeout=1)
|
||||
|
||||
def _update_leader(self):
|
||||
ret = self._sync_obj.set(self.leader_path, self._name, ttl=self._ttl, prevValue=self._name)
|
||||
if not ret and self._sync_obj.get(self.leader_path) is None:
|
||||
ret = self.attempt_to_acquire_leader()
|
||||
return ret
|
||||
|
||||
def attempt_to_acquire_leader(self, permanent=False):
|
||||
return self._sync_obj.set(self.leader_path, self._name, prevExist=False,
|
||||
ttl=None if permanent else self._ttl)
|
||||
|
||||
def set_failover_value(self, value, index=None):
|
||||
return self._sync_obj.set(self.failover_path, value, prevIndex=index)
|
||||
|
||||
def set_config_value(self, value, index=None):
|
||||
return self._sync_obj.set(self.config_path, value, prevIndex=index)
|
||||
|
||||
def touch_member(self, data, permanent=False):
|
||||
data = json.dumps(data, separators=(',', ':'))
|
||||
return self._sync_obj.set(self.member_path, data, None if permanent else self._ttl, timeout=2)
|
||||
|
||||
def take_leader(self):
|
||||
return self._sync_obj.set(self.leader_path, self._name, ttl=self._ttl)
|
||||
|
||||
def initialize(self, create_new=True, sysid=''):
|
||||
return self._sync_obj.set(self.initialize_path, sysid, prevExist=(not create_new))
|
||||
|
||||
def _delete_leader(self):
|
||||
return self._sync_obj.delete(self.leader_path, prevValue=self._name, timeout=1)
|
||||
|
||||
def cancel_initialization(self):
|
||||
return self._sync_obj.delete(self.initialize_path)
|
||||
|
||||
def delete_cluster(self):
|
||||
return self._sync_obj.delete(self.client_path(''), recursive=True)
|
||||
|
||||
def set_history_value(self, value):
|
||||
return self._sync_obj.set(self.history_path, value)
|
||||
|
||||
def set_sync_state_value(self, value, index=None):
|
||||
return self._sync_obj.set(self.sync_path, value, prevIndex=index)
|
||||
|
||||
def delete_sync_state(self, index=None):
|
||||
return self._sync_obj.delete(self.sync_path, prevIndex=index)
|
||||
|
||||
def watch(self, leader_index, timeout):
|
||||
try:
|
||||
return super(Raft, self).watch(leader_index, timeout)
|
||||
finally:
|
||||
self.event.clear()
|
||||
+252
-91
@@ -1,10 +1,17 @@
|
||||
import json
|
||||
import logging
|
||||
import select
|
||||
import time
|
||||
|
||||
from kazoo.client import KazooClient, KazooState
|
||||
from kazoo.exceptions import NoNodeError, NodeExistsError
|
||||
from kazoo.client import KazooClient, KazooState, KazooRetry
|
||||
from kazoo.exceptions import NoNodeError, NodeExistsError, SessionExpiredError
|
||||
from kazoo.handlers.threading import SequentialThreadingHandler
|
||||
from patroni.dcs import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member
|
||||
from patroni.exceptions import DCSError
|
||||
from kazoo.protocol.states import KeeperState
|
||||
from kazoo.security import make_acl
|
||||
|
||||
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
||||
from ..exceptions import DCSError
|
||||
from ..utils import deep_compare
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -20,7 +27,7 @@ class PatroniSequentialThreadingHandler(SequentialThreadingHandler):
|
||||
self.set_connect_timeout(connect_timeout)
|
||||
|
||||
def set_connect_timeout(self, connect_timeout):
|
||||
self._connect_timeout = max(1.0, connect_timeout/4.0)
|
||||
self._connect_timeout = max(1.0, connect_timeout/2.0) # try to connect to zookeeper node during loop_wait/2
|
||||
|
||||
def create_connection(self, *args, **kwargs):
|
||||
"""This method is trying to establish connection with one of the zookeeper nodes.
|
||||
@@ -34,12 +41,35 @@ class PatroniSequentialThreadingHandler(SequentialThreadingHandler):
|
||||
`connect_timeout` (negotiated session timeout) as the second element."""
|
||||
|
||||
args = list(args)
|
||||
if len(args) == 1:
|
||||
if len(args) == 0: # kazoo 2.6.0 slightly changed the way how it calls create_connection method
|
||||
kwargs['timeout'] = max(self._connect_timeout, kwargs.get('timeout', self._connect_timeout*10)/10.0)
|
||||
elif len(args) == 1:
|
||||
args.append(self._connect_timeout)
|
||||
else:
|
||||
args[1] = max(self._connect_timeout, args[1]/10.0)
|
||||
return super(PatroniSequentialThreadingHandler, self).create_connection(*args, **kwargs)
|
||||
|
||||
def select(self, *args, **kwargs):
|
||||
"""Python3 raises `ValueError` if socket is closed, because fd == -1"""
|
||||
try:
|
||||
return super(PatroniSequentialThreadingHandler, self).select(*args, **kwargs)
|
||||
except ValueError as e:
|
||||
raise select.error(9, str(e))
|
||||
|
||||
|
||||
class PatroniKazooClient(KazooClient):
|
||||
|
||||
def _call(self, request, async_object):
|
||||
# Before kazoo==2.7.0 it wasn't possible to send requests to zookeeper if
|
||||
# the connection is in the SUSPENDED state and Patroni was strongly relying on it.
|
||||
# The https://github.com/python-zk/kazoo/pull/588 changed it, and now such requests are queued.
|
||||
# We override the `_call()` method in order to keep the old behavior.
|
||||
|
||||
if self._state == KeeperState.CONNECTING:
|
||||
async_object.set_exception(SessionExpiredError())
|
||||
return False
|
||||
return super(PatroniKazooClient, self)._call(request, async_object)
|
||||
|
||||
|
||||
class ZooKeeper(AbstractDCS):
|
||||
|
||||
@@ -50,35 +80,100 @@ class ZooKeeper(AbstractDCS):
|
||||
if isinstance(hosts, list):
|
||||
hosts = ','.join(hosts)
|
||||
|
||||
self._client = KazooClient(hosts, handler=PatroniSequentialThreadingHandler(config['retry_timeout']),
|
||||
timeout=config['ttl'], connection_retry={'max_delay': 1, 'max_tries': -1},
|
||||
command_retry={'deadline': config['retry_timeout'], 'max_delay': 1, 'max_tries': -1})
|
||||
mapping = {'use_ssl': 'use_ssl', 'verify': 'verify_certs', 'cacert': 'ca',
|
||||
'cert': 'certfile', 'key': 'keyfile', 'key_password': 'keyfile_password'}
|
||||
kwargs = {v: config[k] for k, v in mapping.items() if k in config}
|
||||
|
||||
if 'set_acls' in config:
|
||||
kwargs['default_acl'] = []
|
||||
for principal, permissions in config['set_acls'].items():
|
||||
normalizedPermissions = [p.upper() for p in permissions]
|
||||
kwargs['default_acl'].append(make_acl(scheme='x509',
|
||||
credential=principal,
|
||||
read='READ' in normalizedPermissions,
|
||||
write='WRITE' in normalizedPermissions,
|
||||
create='CREATE' in normalizedPermissions,
|
||||
delete='DELETE' in normalizedPermissions,
|
||||
admin='ADMIN' in normalizedPermissions,
|
||||
all='ALL' in normalizedPermissions))
|
||||
|
||||
self._client = PatroniKazooClient(hosts, handler=PatroniSequentialThreadingHandler(config['retry_timeout']),
|
||||
timeout=config['ttl'], connection_retry=KazooRetry(max_delay=1, max_tries=-1,
|
||||
sleep_func=time.sleep), command_retry=KazooRetry(max_delay=1, max_tries=-1,
|
||||
deadline=config['retry_timeout'], sleep_func=time.sleep), **kwargs)
|
||||
self._client.add_listener(self.session_listener)
|
||||
|
||||
self._my_member_data = None
|
||||
self._fetch_cluster = True
|
||||
self._last_leader_operation = 0
|
||||
self._fetch_status = True
|
||||
self.__last_member_data = None
|
||||
|
||||
self._orig_kazoo_connect = self._client._connection._connect
|
||||
self._client._connection._connect = self._kazoo_connect
|
||||
|
||||
self._client.start()
|
||||
|
||||
def _kazoo_connect(self, *args):
|
||||
"""Kazoo is using Ping's to determine health of connection to zookeeper. If there is no
|
||||
response on Ping after Ping interval (1/2 from read_timeout) it will consider current
|
||||
connection dead and try to connect to another node. Without this "magic" it was taking
|
||||
up to 2/3 from session timeout (ttl) to figure out that connection was dead and we had
|
||||
only small time for reconnect and retry.
|
||||
|
||||
This method is needed to return different value of read_timeout, which is not calculated
|
||||
from negotiated session timeout but from value of `loop_wait`. And it is 2 sec smaller
|
||||
than loop_wait, because we can spend up to 2 seconds when calling `touch_member()` and
|
||||
`write_leader_optime()` methods, which also may hang..."""
|
||||
|
||||
ret = self._orig_kazoo_connect(*args)
|
||||
return max(self.loop_wait - 2, 2)*1000, ret[1]
|
||||
|
||||
def session_listener(self, state):
|
||||
if state in [KazooState.SUSPENDED, KazooState.LOST]:
|
||||
self.cluster_watcher(None)
|
||||
|
||||
def cluster_watcher(self, event):
|
||||
self._fetch_cluster = True
|
||||
def status_watcher(self, event):
|
||||
self._fetch_status = True
|
||||
self.event.set()
|
||||
|
||||
def cluster_watcher(self, event):
|
||||
self._fetch_cluster = True
|
||||
self.status_watcher(event)
|
||||
|
||||
def reload_config(self, config):
|
||||
self.set_retry_timeout(config['retry_timeout'])
|
||||
|
||||
loop_wait = config['loop_wait']
|
||||
|
||||
loop_wait_changed = self._loop_wait != loop_wait
|
||||
self._loop_wait = loop_wait
|
||||
self._client.handler.set_connect_timeout(loop_wait)
|
||||
|
||||
# We need to reestablish connection to zookeeper if we want to change
|
||||
# read_timeout (and Ping interval respectively), because read_timeout
|
||||
# is calculated in `_kazoo_connect` method. If we are changing ttl at
|
||||
# the same time, set_ttl method will reestablish connection and return
|
||||
# `!True`, otherwise we will close existing connection and let kazoo
|
||||
# open the new one.
|
||||
if not self.set_ttl(config['ttl']) and loop_wait_changed:
|
||||
self._client._connection._socket.close()
|
||||
|
||||
def set_ttl(self, ttl):
|
||||
"""It is not possible to change ttl (session_timeout) in zookeeper without
|
||||
destroying old session and creating the new one. This method returns `!True`
|
||||
if session_timeout has been changed (`restart()` has been called)."""
|
||||
ttl = int(ttl * 1000)
|
||||
# I know, it's weird to access private attributes
|
||||
if self._client._session_timeout != ttl:
|
||||
self._client._session_timeout = ttl
|
||||
self._client.restart()
|
||||
return True
|
||||
|
||||
@property
|
||||
def ttl(self):
|
||||
return self._client._session_timeout / 1000.0
|
||||
|
||||
def set_retry_timeout(self, retry_timeout):
|
||||
self._client.handler.set_connect_timeout(retry_timeout)
|
||||
self._client._retry.deadline = retry_timeout
|
||||
retry = self._client.retry if isinstance(self._client.retry, KazooRetry) else self._client._retry
|
||||
retry.deadline = retry_timeout
|
||||
|
||||
def get_node(self, key, watch=None):
|
||||
try:
|
||||
@@ -87,6 +182,30 @@ class ZooKeeper(AbstractDCS):
|
||||
except NoNodeError:
|
||||
return None
|
||||
|
||||
def get_status(self, leader):
|
||||
watch = self.status_watcher if not leader or leader.name != self._name else None
|
||||
|
||||
status = self.get_node(self.status_path, watch)
|
||||
if status:
|
||||
try:
|
||||
status = json.loads(status[0])
|
||||
last_lsn = status.get(self._OPTIME)
|
||||
slots = status.get('slots')
|
||||
except Exception:
|
||||
slots = last_lsn = None
|
||||
else:
|
||||
last_lsn = self.get_node(self.leader_optime_path, watch)
|
||||
last_lsn = last_lsn and last_lsn[0]
|
||||
slots = None
|
||||
|
||||
try:
|
||||
last_lsn = int(last_lsn)
|
||||
except Exception:
|
||||
last_lsn = 0
|
||||
|
||||
self._fetch_status = False
|
||||
return last_lsn, slots
|
||||
|
||||
@staticmethod
|
||||
def member(name, value, znode):
|
||||
return Member.from_node(znode.version, name, znode.ephemeralOwner, value)
|
||||
@@ -119,6 +238,14 @@ class ZooKeeper(AbstractDCS):
|
||||
config = self.get_node(self.config_path, watch=self.cluster_watcher) if self._CONFIG in nodes else None
|
||||
config = config and ClusterConfig.from_node(config[1].version, config[0], config[1].mzxid)
|
||||
|
||||
# get timeline history
|
||||
history = self.get_node(self.history_path, watch=self.cluster_watcher) if self._HISTORY in nodes else None
|
||||
history = history and TimelineHistory.from_node(history[1].mzxid, history[0])
|
||||
|
||||
# get synchronization state
|
||||
sync = self.get_node(self.sync_path, watch=self.cluster_watcher) if self._SYNC in nodes else None
|
||||
sync = SyncState.from_node(sync and sync[1].version, sync and sync[0])
|
||||
|
||||
# get list of members
|
||||
members = self.load_members() if self._MEMBERS[:-1] in nodes else []
|
||||
|
||||
@@ -126,7 +253,8 @@ class ZooKeeper(AbstractDCS):
|
||||
leader = self.get_node(self.leader_path) if self._LEADER in nodes else None
|
||||
if leader:
|
||||
client_id = self._client.client_id
|
||||
if leader[0] == self._name and client_id is not None and client_id[0] != leader[1].ephemeralOwner:
|
||||
if not self._ctl and leader[0] == self._name and client_id is not None \
|
||||
and client_id[0] != leader[1].ephemeralOwner:
|
||||
logger.info('I am leader but not owner of the session. Removing leader node')
|
||||
self._client.delete(self.leader_path)
|
||||
leader = None
|
||||
@@ -137,120 +265,143 @@ class ZooKeeper(AbstractDCS):
|
||||
leader = Leader(leader[1].version, leader[1].ephemeralOwner, member)
|
||||
self._fetch_cluster = member.index == -1
|
||||
|
||||
# get last known leader lsn and slots
|
||||
last_lsn, slots = self.get_status(leader)
|
||||
|
||||
# failover key
|
||||
failover = self.get_node(self.failover_path, watch=self.cluster_watcher) if self._FAILOVER in nodes else None
|
||||
failover = failover and Failover.from_node(failover[1].version, failover[0])
|
||||
|
||||
# get last leader operation
|
||||
optime = self.get_node(self.leader_optime_path) if self._OPTIME in nodes and self._fetch_cluster else None
|
||||
self._last_leader_operation = 0 if optime is None else int(optime[0])
|
||||
self._cluster = Cluster(initialize, config, leader, self._last_leader_operation, members, failover)
|
||||
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||
|
||||
def _load_cluster(self):
|
||||
if self._fetch_cluster or self._cluster is None:
|
||||
cluster = self.cluster
|
||||
if self._fetch_cluster or cluster is None:
|
||||
try:
|
||||
self._client.retry(self._inner_load_cluster)
|
||||
except:
|
||||
cluster = self._client.retry(self._inner_load_cluster)
|
||||
except Exception:
|
||||
logger.exception('get_cluster')
|
||||
self.cluster_watcher(None)
|
||||
raise ZooKeeperError('ZooKeeper in not responding properly')
|
||||
# The /status ZNode was updated or doesn't exist
|
||||
elif self._fetch_status and not self._fetch_cluster or not cluster.last_lsn \
|
||||
or cluster.has_permanent_logical_slots(self._name, False) and not cluster.slots:
|
||||
# If current node is the leader just clear the event without fetching anything (we are updating the /status)
|
||||
if cluster.leader and cluster.leader.name == self._name:
|
||||
self.event.clear()
|
||||
else:
|
||||
try:
|
||||
last_lsn, slots = self.get_status(cluster.leader)
|
||||
self.event.clear()
|
||||
cluster = Cluster(cluster.initialize, cluster.config, cluster.leader, last_lsn,
|
||||
cluster.members, cluster.failover, cluster.sync, cluster.history, slots)
|
||||
except Exception:
|
||||
pass
|
||||
return cluster
|
||||
|
||||
def _create(self, path, value, **kwargs):
|
||||
def _bypass_caches(self):
|
||||
self._fetch_cluster = True
|
||||
|
||||
def _create(self, path, value, retry=False, ephemeral=False):
|
||||
try:
|
||||
self._client.retry(self._client.create, path, value.encode('utf-8'), **kwargs)
|
||||
if retry:
|
||||
self._client.retry(self._client.create, path, value, makepath=True, ephemeral=ephemeral)
|
||||
else:
|
||||
self._client.create_async(path, value, makepath=True, ephemeral=ephemeral).get(timeout=1)
|
||||
return True
|
||||
except:
|
||||
return False
|
||||
except Exception:
|
||||
logger.exception('Failed to create %s', path)
|
||||
return False
|
||||
|
||||
def attempt_to_acquire_leader(self):
|
||||
ret = self._create(self.leader_path, self._name, makepath=True, ephemeral=True)
|
||||
def attempt_to_acquire_leader(self, permanent=False):
|
||||
ret = self._create(self.leader_path, self._name.encode('utf-8'), retry=True, ephemeral=not permanent)
|
||||
if not ret:
|
||||
logger.info('Could not take out TTL lock')
|
||||
return ret
|
||||
|
||||
def set_failover_value(self, value, index=None):
|
||||
def _set_or_create(self, key, value, index=None, retry=False, do_not_create_empty=False):
|
||||
value = value.encode('utf-8')
|
||||
try:
|
||||
self._client.retry(self._client.set, self.failover_path, value.encode('utf-8'), version=index or -1)
|
||||
if retry:
|
||||
self._client.retry(self._client.set, key, value, version=index or -1)
|
||||
else:
|
||||
self._client.set_async(key, value, version=index or -1).get(timeout=1)
|
||||
return True
|
||||
except NoNodeError:
|
||||
return value == '' or (index is None and self._create(self.failover_path, value))
|
||||
except:
|
||||
logging.exception('set_failover_value')
|
||||
return False
|
||||
if do_not_create_empty and not value:
|
||||
return True
|
||||
elif index is None:
|
||||
return self._create(key, value, retry)
|
||||
else:
|
||||
return False
|
||||
except Exception:
|
||||
logger.exception('Failed to update %s', key)
|
||||
return False
|
||||
|
||||
def set_failover_value(self, value, index=None):
|
||||
return self._set_or_create(self.failover_path, value, index)
|
||||
|
||||
def set_config_value(self, value, index=None):
|
||||
try:
|
||||
self._client.retry(self._client.set, self.config_path, value.encode('utf-8'), version=index or -1)
|
||||
return True
|
||||
except NoNodeError:
|
||||
return index is None and self._create(self.config_path, value)
|
||||
except Exception:
|
||||
logging.exception('set_config_value')
|
||||
return False
|
||||
return self._set_or_create(self.config_path, value, index, retry=True)
|
||||
|
||||
def initialize(self, create_new=True, sysid=""):
|
||||
return self._create(self.initialize_path, sysid, makepath=True) if create_new \
|
||||
else self._client.retry(self._client.set, self.initialize_path, sysid.encode("utf-8"))
|
||||
sysid = sysid.encode('utf-8')
|
||||
return self._create(self.initialize_path, sysid, retry=True) if create_new \
|
||||
else self._client.retry(self._client.set, self.initialize_path, sysid)
|
||||
|
||||
def touch_member(self, data, ttl=None):
|
||||
def touch_member(self, data, permanent=False):
|
||||
cluster = self.cluster
|
||||
member = cluster and ([m for m in cluster.members if m.name == self._name] or [None])[0]
|
||||
path = self.member_path
|
||||
data = data.encode('utf-8')
|
||||
if member and self._client.client_id is not None and member.session != self._client.client_id[0]:
|
||||
member = cluster and cluster.get_member(self._name, fallback_to_leader=False)
|
||||
member_data = self.__last_member_data or member and member.data
|
||||
if member and (self._client.client_id is not None and member.session != self._client.client_id[0] or
|
||||
not (deep_compare(member_data.get('tags', {}), data.get('tags', {})) and
|
||||
member_data.get('version') == data.get('version') and
|
||||
member_data.get('checkpoint_after_promote') == data.get('checkpoint_after_promote'))):
|
||||
try:
|
||||
self._client.retry(self._client.delete, path)
|
||||
self._client.delete_async(self.member_path).get(timeout=1)
|
||||
except NoNodeError:
|
||||
pass
|
||||
except:
|
||||
except Exception:
|
||||
return False
|
||||
member = None
|
||||
|
||||
if member and data == self._my_member_data:
|
||||
return True
|
||||
|
||||
try:
|
||||
if member:
|
||||
self._client.retry(self._client.set, path, data)
|
||||
else:
|
||||
self._client.retry(self._client.create, path, data, makepath=True, ephemeral=True)
|
||||
self._my_member_data = data
|
||||
return True
|
||||
except NodeExistsError:
|
||||
try:
|
||||
self._client.retry(self._client.set, path, data)
|
||||
self._my_member_data = data
|
||||
encoded_data = json.dumps(data, separators=(',', ':')).encode('utf-8')
|
||||
if member:
|
||||
if deep_compare(data, member_data):
|
||||
return True
|
||||
except:
|
||||
logger.exception('touch_member')
|
||||
except:
|
||||
else:
|
||||
try:
|
||||
self._client.create_async(self.member_path, encoded_data, makepath=True,
|
||||
ephemeral=not permanent).get(timeout=1)
|
||||
self.__last_member_data = data
|
||||
return True
|
||||
except Exception as e:
|
||||
if not isinstance(e, NodeExistsError):
|
||||
logger.exception('touch_member')
|
||||
return False
|
||||
try:
|
||||
self._client.set_async(self.member_path, encoded_data).get(timeout=1)
|
||||
self.__last_member_data = data
|
||||
return True
|
||||
except Exception:
|
||||
logger.exception('touch_member')
|
||||
|
||||
return False
|
||||
|
||||
def take_leader(self):
|
||||
return self.attempt_to_acquire_leader()
|
||||
|
||||
def write_leader_optime(self, last_operation):
|
||||
last_operation = last_operation.encode('utf-8')
|
||||
if last_operation != self._last_leader_operation:
|
||||
self._last_leader_operation = last_operation
|
||||
path = self.leader_optime_path
|
||||
try:
|
||||
self._client.retry(self._client.set, path, last_operation)
|
||||
except NoNodeError:
|
||||
try:
|
||||
self._client.retry(self._client.create, path, last_operation, makepath=True)
|
||||
except:
|
||||
logger.exception('Failed to create %s', path)
|
||||
except:
|
||||
logger.exception('Failed to update %s', path)
|
||||
def _write_leader_optime(self, last_lsn):
|
||||
return self._set_or_create(self.leader_optime_path, last_lsn)
|
||||
|
||||
def update_leader(self):
|
||||
def _write_status(self, value):
|
||||
return self._set_or_create(self.status_path, value)
|
||||
|
||||
def _update_leader(self):
|
||||
return True
|
||||
|
||||
def delete_leader(self):
|
||||
def _delete_leader(self):
|
||||
self._client.restart()
|
||||
self._my_member_data = None
|
||||
return True
|
||||
|
||||
def _cancel_initialization(self):
|
||||
@@ -261,7 +412,7 @@ class ZooKeeper(AbstractDCS):
|
||||
def cancel_initialization(self):
|
||||
try:
|
||||
self._client.retry(self._cancel_initialization)
|
||||
except:
|
||||
except Exception:
|
||||
logger.exception("Unable to delete initialize key")
|
||||
|
||||
def delete_cluster(self):
|
||||
@@ -270,7 +421,17 @@ class ZooKeeper(AbstractDCS):
|
||||
except NoNodeError:
|
||||
return True
|
||||
|
||||
def watch(self, timeout):
|
||||
if super(ZooKeeper, self).watch(timeout):
|
||||
def set_history_value(self, value):
|
||||
return self._set_or_create(self.history_path, value)
|
||||
|
||||
def set_sync_state_value(self, value, index=None):
|
||||
return self._set_or_create(self.sync_path, value, index, retry=True, do_not_create_empty=True)
|
||||
|
||||
def delete_sync_state(self, index=None):
|
||||
return self.set_sync_state_value("{}", index)
|
||||
|
||||
def watch(self, leader_index, timeout):
|
||||
ret = super(ZooKeeper, self).watch(leader_index, timeout + 0.5)
|
||||
if ret and not self._fetch_status:
|
||||
self._fetch_cluster = True
|
||||
return self._fetch_cluster
|
||||
return ret or self._fetch_cluster
|
||||
|
||||
@@ -13,6 +13,10 @@ class PatroniException(Exception):
|
||||
return repr(self.value)
|
||||
|
||||
|
||||
class PatroniFatalException(PatroniException):
|
||||
pass
|
||||
|
||||
|
||||
class PostgresException(PatroniException):
|
||||
pass
|
||||
|
||||
@@ -23,3 +27,11 @@ class DCSError(PatroniException):
|
||||
|
||||
class PostgresConnectionException(PostgresException):
|
||||
pass
|
||||
|
||||
|
||||
class WatchdogError(PatroniException):
|
||||
pass
|
||||
|
||||
|
||||
class ConfigParseError(PatroniException):
|
||||
pass
|
||||
|
||||
+1347
-251
File diff suppressed because it is too large
Load Diff
+207
@@ -0,0 +1,207 @@
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
|
||||
from copy import deepcopy
|
||||
from logging.handlers import RotatingFileHandler
|
||||
from patroni.utils import deep_compare
|
||||
from six.moves.queue import Queue, Full
|
||||
from threading import Lock, Thread
|
||||
|
||||
_LOGGER = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def debug_exception(logger_obj, msg, *args, **kwargs):
|
||||
kwargs.pop("exc_info", False)
|
||||
if logger_obj.isEnabledFor(logging.DEBUG):
|
||||
logger_obj.debug(msg, *args, exc_info=True, **kwargs)
|
||||
else:
|
||||
msg = "{0}, DETAIL: '{1}'".format(msg, sys.exc_info()[1])
|
||||
logger_obj.error(msg, *args, exc_info=False, **kwargs)
|
||||
|
||||
|
||||
def error_exception(logger_obj, msg, *args, **kwargs):
|
||||
exc_info = kwargs.pop("exc_info", True)
|
||||
logger_obj.error(msg, *args, exc_info=exc_info, **kwargs)
|
||||
|
||||
|
||||
class QueueHandler(logging.Handler):
|
||||
|
||||
def __init__(self):
|
||||
logging.Handler.__init__(self)
|
||||
self.queue = Queue()
|
||||
self._records_lost = 0
|
||||
|
||||
def _put_record(self, record):
|
||||
self.format(record)
|
||||
record.msg = record.message
|
||||
record.args = None
|
||||
record.exc_info = None
|
||||
self.queue.put_nowait(record)
|
||||
|
||||
def _try_to_report_lost_records(self):
|
||||
if self._records_lost:
|
||||
try:
|
||||
record = _LOGGER.makeRecord(_LOGGER.name, logging.WARNING, __file__, 0,
|
||||
'QueueHandler has lost %s log records',
|
||||
(self._records_lost,), None, 'emit')
|
||||
self._put_record(record)
|
||||
self._records_lost = 0
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def emit(self, record):
|
||||
try:
|
||||
self._put_record(record)
|
||||
self._try_to_report_lost_records()
|
||||
except Exception:
|
||||
self._records_lost += 1
|
||||
|
||||
@property
|
||||
def records_lost(self):
|
||||
return self._records_lost
|
||||
|
||||
|
||||
class ProxyHandler(logging.Handler):
|
||||
|
||||
def __init__(self, patroni_logger):
|
||||
logging.Handler.__init__(self)
|
||||
self.patroni_logger = patroni_logger
|
||||
|
||||
def emit(self, record):
|
||||
self.patroni_logger.log_handler.handle(record)
|
||||
|
||||
|
||||
class PatroniLogger(Thread):
|
||||
|
||||
DEFAULT_LEVEL = 'INFO'
|
||||
DEFAULT_TRACEBACK_LEVEL = 'ERROR'
|
||||
DEFAULT_FORMAT = '%(asctime)s %(levelname)s: %(message)s'
|
||||
|
||||
NORMAL_LOG_QUEUE_SIZE = 2 # When everything goes normal Patroni writes only 2 messages per HA loop
|
||||
DEFAULT_MAX_QUEUE_SIZE = 1000
|
||||
LOGGING_BROKEN_EXIT_CODE = 5
|
||||
|
||||
def __init__(self):
|
||||
super(PatroniLogger, self).__init__()
|
||||
self._queue_handler = QueueHandler()
|
||||
self._root_logger = logging.getLogger()
|
||||
self._config = None
|
||||
self.log_handler = None
|
||||
self.log_handler_lock = Lock()
|
||||
self._old_handlers = []
|
||||
self.reload_config({'level': 'DEBUG'})
|
||||
# We will switch to the QueueHandler only when thread was started.
|
||||
# This is necessary to protect from the cases when Patroni constructor
|
||||
# failed and PatroniLogger thread remain running and prevent shutdown.
|
||||
self._proxy_handler = ProxyHandler(self)
|
||||
self._root_logger.addHandler(self._proxy_handler)
|
||||
|
||||
def update_loggers(self):
|
||||
loggers = deepcopy(self._config.get('loggers') or {})
|
||||
for name, logger in self._root_logger.manager.loggerDict.items():
|
||||
if not isinstance(logger, logging.PlaceHolder):
|
||||
level = loggers.pop(name, logging.NOTSET)
|
||||
logger.setLevel(level)
|
||||
|
||||
for name, level in loggers.items():
|
||||
logger = self._root_logger.manager.getLogger(name)
|
||||
logger.setLevel(level)
|
||||
|
||||
def reload_config(self, config):
|
||||
if self._config is None or not deep_compare(self._config, config):
|
||||
with self._queue_handler.queue.mutex:
|
||||
self._queue_handler.queue.maxsize = config.get('max_queue_size', self.DEFAULT_MAX_QUEUE_SIZE)
|
||||
|
||||
self._root_logger.setLevel(config.get('level', PatroniLogger.DEFAULT_LEVEL))
|
||||
if config.get('traceback_level', PatroniLogger.DEFAULT_TRACEBACK_LEVEL).lower() == 'debug':
|
||||
logging.Logger.exception = debug_exception
|
||||
else:
|
||||
logging.Logger.exception = error_exception
|
||||
|
||||
new_handler = None
|
||||
if 'dir' in config:
|
||||
if not isinstance(self.log_handler, RotatingFileHandler):
|
||||
new_handler = RotatingFileHandler(os.path.join(config['dir'], __name__))
|
||||
handler = new_handler or self.log_handler
|
||||
handler.maxBytes = int(config.get('file_size', 25000000))
|
||||
handler.backupCount = int(config.get('file_num', 4))
|
||||
else:
|
||||
if self.log_handler is None or isinstance(self.log_handler, RotatingFileHandler):
|
||||
new_handler = logging.StreamHandler()
|
||||
handler = new_handler or self.log_handler
|
||||
|
||||
oldlogformat = (self._config or {}).get('format', PatroniLogger.DEFAULT_FORMAT)
|
||||
logformat = config.get('format', PatroniLogger.DEFAULT_FORMAT)
|
||||
|
||||
olddateformat = (self._config or {}).get('dateformat') or None
|
||||
dateformat = config.get('dateformat') or None # Convert empty string to `None`
|
||||
|
||||
if oldlogformat != logformat or olddateformat != dateformat or new_handler:
|
||||
handler.setFormatter(logging.Formatter(logformat, dateformat))
|
||||
|
||||
if new_handler:
|
||||
with self.log_handler_lock:
|
||||
if self.log_handler:
|
||||
self._old_handlers.append(self.log_handler)
|
||||
self.log_handler = new_handler
|
||||
|
||||
self._config = config.copy()
|
||||
self.update_loggers()
|
||||
|
||||
def _close_old_handlers(self):
|
||||
while True:
|
||||
with self.log_handler_lock:
|
||||
if not self._old_handlers:
|
||||
break
|
||||
handler = self._old_handlers.pop()
|
||||
try:
|
||||
handler.close()
|
||||
except Exception:
|
||||
_LOGGER.exception('Failed to close the old log handler %s', handler)
|
||||
|
||||
def run(self):
|
||||
# switch to QueueHandler only when the thread was started
|
||||
with self.log_handler_lock:
|
||||
self._root_logger.addHandler(self._queue_handler)
|
||||
self._root_logger.removeHandler(self._proxy_handler)
|
||||
|
||||
prev_record = None
|
||||
|
||||
while True:
|
||||
self._close_old_handlers()
|
||||
|
||||
record = self._queue_handler.queue.get(True)
|
||||
if record is None:
|
||||
break
|
||||
|
||||
if self._root_logger.level == logging.INFO:
|
||||
if record.msg.startswith('Lock owner: '):
|
||||
prev_record, record = record, None
|
||||
else:
|
||||
if prev_record and prev_record.thread == record.thread:
|
||||
if not (record.msg.startswith('no action. ') or record.msg.startswith('PAUSE: no action')):
|
||||
self.log_handler.handle(prev_record)
|
||||
prev_record = None
|
||||
|
||||
if record:
|
||||
self.log_handler.handle(record)
|
||||
|
||||
self._queue_handler.queue.task_done()
|
||||
|
||||
def shutdown(self):
|
||||
try:
|
||||
self._queue_handler.queue.put_nowait(None)
|
||||
except Full: # Queue is full.
|
||||
# It seems that logging is not working, exiting with non-standard exit-code is the best we can do.
|
||||
sys.exit(self.LOGGING_BROKEN_EXIT_CODE)
|
||||
self.join()
|
||||
logging.shutdown()
|
||||
|
||||
@property
|
||||
def queue_size(self):
|
||||
return self._queue_handler.queue.qsize()
|
||||
|
||||
@property
|
||||
def records_lost(self):
|
||||
return self._queue_handler.records_lost
|
||||
@@ -1,966 +0,0 @@
|
||||
import logging
|
||||
import os
|
||||
import psycopg2
|
||||
import shlex
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
import time
|
||||
|
||||
from patroni.exceptions import PostgresConnectionException, PostgresException
|
||||
from patroni.utils import compare_values, parse_bool, parse_int, Retry, RetryFailedError
|
||||
from six import string_types
|
||||
from six.moves.urllib_parse import urlparse
|
||||
from threading import Lock
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
ACTION_ON_START = "on_start"
|
||||
ACTION_ON_STOP = "on_stop"
|
||||
ACTION_ON_RESTART = "on_restart"
|
||||
ACTION_ON_RELOAD = "on_reload"
|
||||
ACTION_ON_ROLE_CHANGE = "on_role_change"
|
||||
|
||||
|
||||
def get_conn_kwargs(url, auth=None):
|
||||
r = urlparse(url)
|
||||
ret = {
|
||||
'host': r.hostname,
|
||||
'port': r.port or 5432,
|
||||
'database': r.path[1:],
|
||||
'fallback_application_name': 'Patroni',
|
||||
'connect_timeout': 3,
|
||||
'options': '-c statement_timeout=2000',
|
||||
}
|
||||
if auth and isinstance(auth, dict):
|
||||
if 'username' in auth:
|
||||
ret['user'] = auth['username']
|
||||
if 'password' in auth:
|
||||
ret['password'] = auth['password']
|
||||
return ret
|
||||
|
||||
|
||||
class Postgresql(object):
|
||||
|
||||
# List of parameters which must be always passed to postmaster as command line options
|
||||
# to make it not possible to change them with 'ALTER SYSTEM'.
|
||||
# Some of these parameters have sane default value assigned and Patroni doesn't allow
|
||||
# to decrease this value. E.g. 'wal_level' can't be lower then 'hot_standby' and so on.
|
||||
# These parameters could be changed only globally, i.e. via DCS.
|
||||
# P.S. 'listen_addresses' and 'port' are added here just for convenience, to mark them
|
||||
# as a parameters which should always be passed through command line.
|
||||
#
|
||||
# Format:
|
||||
# key - parameter name
|
||||
# value - tuple(default_value, check_function, min_version)
|
||||
# default_value -- some sane default value
|
||||
# check_function -- if the new value is not correct must return `!False`
|
||||
# min_version -- major version of PostgreSQL when parameter was introduced
|
||||
CMDLINE_OPTIONS = {
|
||||
'listen_addresses': (None, lambda _: False, 9.1),
|
||||
'port': (None, lambda _: False, 9.1),
|
||||
'cluster_name': (None, lambda _: False, 9.5),
|
||||
'wal_level': ('hot_standby', lambda v: v.lower() in ('hot_standby', 'logical'), 9.1),
|
||||
'hot_standby': ('on', lambda _: False, 9.1),
|
||||
'max_connections': (100, lambda v: int(v) >= 100, 9.1),
|
||||
'max_wal_senders': (5, lambda v: int(v) >= 5, 9.1),
|
||||
'wal_keep_segments': (8, lambda v: int(v) >= 8, 9.1),
|
||||
'max_prepared_transactions': (0, lambda v: int(v) >= 0, 9.1),
|
||||
'max_locks_per_transaction': (64, lambda v: int(v) >= 64, 9.1),
|
||||
'track_commit_timestamp': ('off', lambda v: parse_bool(v) is not None, 9.5),
|
||||
'max_replication_slots': (5, lambda v: int(v) >= 5, 9.4),
|
||||
'max_worker_processes': (8, lambda v: int(v) >= 8, 9.4),
|
||||
'wal_log_hints': ('on', lambda _: False, 9.4)
|
||||
}
|
||||
|
||||
def __init__(self, config):
|
||||
self.config = config
|
||||
self.name = config['name']
|
||||
self.scope = config['scope']
|
||||
self._database = config.get('database', 'postgres')
|
||||
self._data_dir = config['data_dir']
|
||||
self._pending_restart = False
|
||||
self._server_parameters = self.get_server_parameters(config)
|
||||
|
||||
self._connect_address = config.get('connect_address')
|
||||
self._superuser = config['authentication'].get('superuser', {})
|
||||
self._replication = config['authentication']['replication']
|
||||
self.resolve_connection_addresses()
|
||||
|
||||
self._need_rewind = False
|
||||
self._use_slots = config.get('use_slots', True)
|
||||
self._version_file = os.path.join(self._data_dir, 'PG_VERSION')
|
||||
self._major_version = self.get_major_version()
|
||||
self._schedule_load_slots = self.use_slots
|
||||
|
||||
self._pgpass = config.get('pgpass') or os.path.join(os.path.expanduser('~'), 'pgpass')
|
||||
self.callback = config.get('callbacks') or {}
|
||||
config_base_name = config.get('config_base_name', 'postgresql')
|
||||
self._postgresql_conf = os.path.join(self._data_dir, config_base_name + '.conf')
|
||||
self._postgresql_base_conf_name = config_base_name + '.base.conf'
|
||||
self._postgresql_base_conf = os.path.join(self._data_dir, self._postgresql_base_conf_name)
|
||||
self._recovery_conf = os.path.join(self._data_dir, 'recovery.conf')
|
||||
self._configuration_to_save = (self._postgresql_conf, self._postgresql_base_conf,
|
||||
os.path.join(self._data_dir, 'pg_hba.conf'))
|
||||
self._postmaster_pid = os.path.join(self._data_dir, 'postmaster.pid')
|
||||
self._trigger_file = config.get('recovery_conf', {}).get('trigger_file') or 'promote'
|
||||
self._trigger_file = os.path.abspath(os.path.join(self._data_dir, self._trigger_file))
|
||||
|
||||
self._connection = None
|
||||
self._cursor_holder = None
|
||||
self._sysid = None
|
||||
self._replication_slots = [] # list of already existing replication slots
|
||||
self.retry = Retry(max_tries=-1, deadline=config['retry_timeout']/2.0, max_delay=1,
|
||||
retry_exceptions=PostgresConnectionException)
|
||||
|
||||
self._state_lock = Lock()
|
||||
self.set_state('stopped')
|
||||
self._role_lock = Lock()
|
||||
self.set_role(self.get_postgres_role_from_data_directory())
|
||||
|
||||
if self.is_running():
|
||||
self.set_state('running')
|
||||
self.set_role('master' if self.is_leader() else 'replica')
|
||||
self._write_postgresql_conf() # we are "joining" already running postgres
|
||||
|
||||
@property
|
||||
def use_slots(self):
|
||||
return self._use_slots and self._major_version >= 9.4
|
||||
|
||||
def _version_file_exists(self):
|
||||
return not self.data_directory_empty() and os.path.isfile(self._version_file)
|
||||
|
||||
def get_major_version(self):
|
||||
if self._version_file_exists():
|
||||
try:
|
||||
with open(self._version_file) as f:
|
||||
return float(f.read())
|
||||
except Exception:
|
||||
logger.exception('Failed to read PG_VERSION from %s', self._data_dir)
|
||||
return 0.0
|
||||
|
||||
def get_server_parameters(self, config):
|
||||
parameters = config['parameters'].copy()
|
||||
listen_addresses, port = (config['listen'] + ':5432').split(':')[:2]
|
||||
parameters.update({'cluster_name': self.scope, 'listen_addresses': listen_addresses, 'port': port})
|
||||
return parameters
|
||||
|
||||
def resolve_connection_addresses(self):
|
||||
self._local_address = self.get_local_address()
|
||||
self.connection_string = 'postgres://{connect_address}/{database}'.format(
|
||||
connect_address=self._connect_address or self._local_address, database=self._database)
|
||||
|
||||
def pg_ctl(self, cmd, *args, **kwargs):
|
||||
"""Builds and executes pg_ctl command
|
||||
|
||||
:returns: `!True` when return_code == 0, otherwise `!False`"""
|
||||
|
||||
pg_ctl = ['pg_ctl', cmd]
|
||||
if cmd in ('start', 'stop', 'restart'):
|
||||
pg_ctl += ['-w']
|
||||
timeout = self.config.get('pg_ctl_timeout')
|
||||
if timeout:
|
||||
try:
|
||||
pg_ctl += ['-t', str(int(timeout))]
|
||||
except Exception:
|
||||
logger.error('Bad value of pg_ctl_timeout: %s', timeout)
|
||||
return subprocess.call(pg_ctl + ['-D', self._data_dir] + list(args), **kwargs) == 0
|
||||
|
||||
def reload_config(self, config):
|
||||
server_parameters = self.get_server_parameters(config)
|
||||
|
||||
listen_address_changed = pending_reload = pending_restart = False
|
||||
if self.is_healthy():
|
||||
changes = {p: v for p, v in server_parameters.items() if '.' not in p}
|
||||
changes.update({p: None for p, v in self._server_parameters.items() if not ('.' in p or p in changes)})
|
||||
if changes:
|
||||
if 'wal_segment_size' not in changes:
|
||||
changes['wal_segment_size'] = '16384kB'
|
||||
# XXX: query can raise an exception
|
||||
for r in self.query("""SELECT name, setting, unit, vartype, context
|
||||
FROM pg_settings
|
||||
WHERE name IN (""" + ', '.join(['%s'] * len(changes)) + """)
|
||||
ORDER BY 1 DESC""", *(list(changes.keys()))):
|
||||
if r[4] == 'internal':
|
||||
if r[0] == 'wal_segment_size':
|
||||
server_parameters.pop(r[0], None)
|
||||
wal_segment_size = parse_int(r[2], 'kB')
|
||||
if wal_segment_size is not None:
|
||||
changes['wal_segment_size'] = '{0}kB'.format(int(r[1]) * wal_segment_size)
|
||||
elif r[0] in changes:
|
||||
unit = changes['wal_segment_size'] if r[0] in ('min_wal_size', 'max_wal_size') else r[2]
|
||||
new_value = changes.pop(r[0])
|
||||
if new_value is None or not compare_values(r[3], unit, r[1], new_value):
|
||||
if r[4] == 'postmaster':
|
||||
pending_restart = True
|
||||
if r[0] in ('listen_addresses', 'port'):
|
||||
listen_address_changed = True
|
||||
else:
|
||||
pending_reload = True
|
||||
for param in changes:
|
||||
if param in server_parameters:
|
||||
logger.warning('Removing invalid parameter `%s` from postgresql.parameters', param)
|
||||
server_parameters.pop(param)
|
||||
|
||||
# Check that user-defined-paramters have changed (parameters with period in name)
|
||||
if not pending_reload:
|
||||
for p, v in server_parameters.items():
|
||||
if '.' in p and (p not in self._server_parameters or str(v) != str(self._server_parameters[p])):
|
||||
pending_reload = True
|
||||
break
|
||||
if not pending_reload:
|
||||
for p, v in self._server_parameters.items():
|
||||
if '.' in p and (p not in server_parameters or str(v) != str(server_parameters[p])):
|
||||
pending_reload = True
|
||||
break
|
||||
|
||||
self.config = config
|
||||
self._pending_restart = pending_restart
|
||||
self._server_parameters = server_parameters
|
||||
self._connect_address = config.get('connect_address')
|
||||
|
||||
if not listen_address_changed:
|
||||
self.resolve_connection_addresses()
|
||||
|
||||
if pending_reload:
|
||||
self._write_postgresql_conf()
|
||||
self.reload()
|
||||
self.retry.deadline = config['retry_timeout']/2.0
|
||||
|
||||
@property
|
||||
def pending_restart(self):
|
||||
return self._pending_restart
|
||||
|
||||
@property
|
||||
def can_rewind(self):
|
||||
""" check if pg_rewind executable is there and that pg_controldata indicates
|
||||
we have either wal_log_hints or checksums turned on
|
||||
"""
|
||||
# low-hanging fruit: check if pg_rewind configuration is there
|
||||
if not (self.config.get('use_pg_rewind') and all(self._superuser.get(n) for n in ('username', 'password'))):
|
||||
return False
|
||||
|
||||
cmd = ['pg_rewind', '--help']
|
||||
try:
|
||||
ret = subprocess.call(cmd, stdout=open(os.devnull, 'w'), stderr=subprocess.STDOUT)
|
||||
if ret != 0: # pg_rewind is not there, close up the shop and go home
|
||||
return False
|
||||
except OSError:
|
||||
return False
|
||||
# check if the cluster's configuration permits pg_rewind
|
||||
data = self.controldata()
|
||||
return data.get('wal_log_hints setting', 'off') == 'on' or data.get('Data page checksum version', '0') != '0'
|
||||
|
||||
@property
|
||||
def sysid(self):
|
||||
if not self._sysid:
|
||||
data = self.controldata()
|
||||
self._sysid = data.get('Database system identifier', "")
|
||||
return self._sysid
|
||||
|
||||
def get_local_address(self):
|
||||
listen_addresses = self._server_parameters['listen_addresses'].split(',')
|
||||
local_address = listen_addresses[0].strip() # take first address from listen_addresses
|
||||
|
||||
for la in listen_addresses:
|
||||
if la.strip().lower() in ('*', '0.0.0.0', '127.0.0.1', 'localhost'): # we are listening on '*' or localhost
|
||||
local_address = 'localhost' # connection via localhost is preferred
|
||||
break
|
||||
return local_address + ':' + self._server_parameters['port']
|
||||
|
||||
def get_postgres_role_from_data_directory(self):
|
||||
if self.data_directory_empty():
|
||||
return 'uninitialized'
|
||||
elif os.path.exists(self._recovery_conf):
|
||||
return 'replica'
|
||||
else:
|
||||
return 'master'
|
||||
|
||||
@property
|
||||
def _connect_kwargs(self):
|
||||
return get_conn_kwargs('postgres://{0}/{1}'.format(self._local_address, self._database), self._superuser)
|
||||
|
||||
def connection(self):
|
||||
if not self._connection or self._connection.closed != 0:
|
||||
self._connection = psycopg2.connect(**self._connect_kwargs)
|
||||
self._connection.autocommit = True
|
||||
self.server_version = self._connection.server_version
|
||||
return self._connection
|
||||
|
||||
def _cursor(self):
|
||||
if not self._cursor_holder or self._cursor_holder.closed or self._cursor_holder.connection.closed != 0:
|
||||
logger.info("establishing a new patroni connection to the postgres cluster")
|
||||
self._cursor_holder = self.connection().cursor()
|
||||
return self._cursor_holder
|
||||
|
||||
def close_connection(self):
|
||||
if self._cursor_holder and self._cursor_holder.connection and self._cursor_holder.connection.closed == 0:
|
||||
self._cursor_holder.connection.close()
|
||||
logger.info("closed patroni connection to the postgresql cluster")
|
||||
|
||||
def _query(self, sql, *params):
|
||||
cursor = None
|
||||
try:
|
||||
cursor = self._cursor()
|
||||
cursor.execute(sql, params)
|
||||
return cursor
|
||||
except psycopg2.Error as e:
|
||||
if cursor and cursor.connection.closed == 0:
|
||||
raise e
|
||||
if self.state == 'restarting':
|
||||
raise RetryFailedError('cluster is being restarted')
|
||||
raise PostgresConnectionException('connection problems')
|
||||
|
||||
def query(self, sql, *params):
|
||||
try:
|
||||
return self.retry(self._query, sql, *params)
|
||||
except RetryFailedError as e:
|
||||
raise PostgresConnectionException(str(e))
|
||||
|
||||
def data_directory_empty(self):
|
||||
return not os.path.exists(self._data_dir) or os.listdir(self._data_dir) == []
|
||||
|
||||
@staticmethod
|
||||
def initdb_allowed_option(name):
|
||||
if name in ['pgdata', 'nosync', 'pwfile', 'sync-only']:
|
||||
raise Exception('{0} option for initdb is not allowed'.format(name))
|
||||
return True
|
||||
|
||||
def get_initdb_options(self, config):
|
||||
options = []
|
||||
for o in config:
|
||||
if isinstance(o, string_types) and self.initdb_allowed_option(o):
|
||||
options.append('--{0}'.format(o))
|
||||
elif isinstance(o, dict):
|
||||
keys = list(o.keys())
|
||||
if len(keys) != 1 or not isinstance(keys[0], string_types) or not self.initdb_allowed_option(keys[0]):
|
||||
raise Exception('Invalid option: {0}'.format(o))
|
||||
options.append('--{0}={1}'.format(keys[0], o[keys[0]]))
|
||||
else:
|
||||
raise Exception('Unknown type of initdb option: {0}'.format(o))
|
||||
return options
|
||||
|
||||
def _initialize(self, config):
|
||||
self.set_state('initalizing new cluster')
|
||||
options = self.get_initdb_options(config.get('initdb') or [])
|
||||
pwfile = None
|
||||
|
||||
if self._superuser:
|
||||
if 'username' in self._superuser:
|
||||
options.append('--username={0}'.format(self._superuser['username']))
|
||||
if 'password' in self._superuser:
|
||||
(fd, pwfile) = tempfile.mkstemp()
|
||||
os.write(fd, self._superuser['password'].encode('utf-8'))
|
||||
os.close(fd)
|
||||
options.append('--pwfile={0}'.format(pwfile))
|
||||
options = ['-o', ' '.join(options)] if options else []
|
||||
|
||||
ret = self.pg_ctl('initdb', *options)
|
||||
if pwfile:
|
||||
os.remove(pwfile)
|
||||
if ret:
|
||||
self.write_pg_hba(config.get('pg_hba', []))
|
||||
self._major_version = self.get_major_version()
|
||||
else:
|
||||
self.set_state('initdb failed')
|
||||
return ret
|
||||
|
||||
def delete_trigger_file(self):
|
||||
if os.path.exists(self._trigger_file):
|
||||
os.unlink(self._trigger_file)
|
||||
|
||||
def write_pgpass(self, record):
|
||||
with open(self._pgpass, 'w') as f:
|
||||
os.fchmod(f.fileno(), 0o600)
|
||||
f.write('{host}:{port}:*:{user}:{password}\n'.format(**record))
|
||||
|
||||
env = os.environ.copy()
|
||||
env['PGPASSFILE'] = self._pgpass
|
||||
return env
|
||||
|
||||
def replica_method_can_work_without_replication_connection(self, method):
|
||||
return method != 'basebackup' and self.config and self.config.get(method, {}).get('no_master')
|
||||
|
||||
def can_create_replica_without_replication_connection(self):
|
||||
""" go through the replication methods to see if there are ones
|
||||
that does not require a working replication connection.
|
||||
"""
|
||||
replica_methods = self.config.get('create_replica_method', [])
|
||||
return any(self.replica_method_can_work_without_replication_connection(method) for method in replica_methods)
|
||||
|
||||
def create_replica(self, clone_member):
|
||||
"""
|
||||
create the replica according to the replica_method
|
||||
defined by the user. this is a list, so we need to
|
||||
loop through all methods the user supplies
|
||||
"""
|
||||
|
||||
self.set_state('creating replica')
|
||||
self._sysid = None
|
||||
|
||||
# get list of replica methods from config.
|
||||
# If there is no configuration key, or no value is specified, use basebackup
|
||||
replica_methods = self.config.get('create_replica_method') or ['basebackup']
|
||||
|
||||
if clone_member:
|
||||
r = get_conn_kwargs(clone_member.conn_url, self._replication)
|
||||
connstring = 'postgres://{user}@{host}:{port}/{database}'.format(**r)
|
||||
# add the credentials to connect to the replica origin to pgpass.
|
||||
env = self.write_pgpass(r)
|
||||
else:
|
||||
connstring = ''
|
||||
env = os.environ.copy()
|
||||
# if we don't have any source, leave only replica methods that work without it
|
||||
replica_methods = \
|
||||
[r for r in replica_methods if self.replica_method_can_work_without_replication_connection(r)]
|
||||
|
||||
# go through them in priority order
|
||||
ret = 1
|
||||
for replica_method in replica_methods:
|
||||
# if the method is basebackup, then use the built-in
|
||||
if replica_method == "basebackup":
|
||||
ret = self.basebackup(connstring, env)
|
||||
if ret == 0:
|
||||
logger.info("replica has been created using basebackup")
|
||||
# if basebackup succeeds, exit with success
|
||||
break
|
||||
else:
|
||||
cmd = replica_method
|
||||
method_config = {}
|
||||
# user-defined method; check for configuration
|
||||
# not required, actually
|
||||
if replica_method in self.config:
|
||||
method_config = self.config[replica_method].copy()
|
||||
# look to see if the user has supplied a full command path
|
||||
# if not, use the method name as the command
|
||||
cmd = method_config.pop('command', cmd)
|
||||
# add the default parameters
|
||||
try:
|
||||
method_config.update({"scope": self.scope,
|
||||
"role": "replica",
|
||||
"datadir": self._data_dir,
|
||||
"connstring": connstring})
|
||||
params = ["--{0}={1}".format(arg, val) for arg, val in method_config.items()]
|
||||
# call script with the full set of parameters
|
||||
ret = subprocess.call(shlex.split(cmd) + params, env=env)
|
||||
# if we succeeded, stop
|
||||
if ret == 0:
|
||||
logger.info('replica has been created using %s', replica_method)
|
||||
break
|
||||
except Exception:
|
||||
logger.exception('Error creating replica using method %s', replica_method)
|
||||
ret = 1
|
||||
|
||||
self.set_state('stopped')
|
||||
return ret
|
||||
|
||||
def is_leader(self):
|
||||
return not self.query('SELECT pg_is_in_recovery()').fetchone()[0]
|
||||
|
||||
def is_running(self):
|
||||
if not (self._version_file_exists() and os.path.isfile(self._postmaster_pid)):
|
||||
return False
|
||||
try:
|
||||
with open(self._postmaster_pid) as f:
|
||||
pid = int(f.readline())
|
||||
if pid < 0:
|
||||
pid = -pid
|
||||
return pid > 0 and pid != os.getpid() and pid != os.getppid() and (os.kill(pid, 0) or True)
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def call_nowait(self, cb_name):
|
||||
""" pick a callback command and call it without waiting for it to finish """
|
||||
if not self.callback or cb_name not in self.callback:
|
||||
return False
|
||||
cmd = self.callback[cb_name]
|
||||
try:
|
||||
subprocess.Popen(shlex.split(cmd) + [cb_name, self.role, self.scope])
|
||||
except OSError:
|
||||
logger.exception('callback %s %s %s %s failed', cmd, cb_name, self.role, self.scope)
|
||||
return False
|
||||
return True
|
||||
|
||||
@property
|
||||
def role(self):
|
||||
with self._role_lock:
|
||||
return self._role
|
||||
|
||||
def set_role(self, value):
|
||||
with self._role_lock:
|
||||
self._role = value
|
||||
|
||||
@property
|
||||
def state(self):
|
||||
with self._state_lock:
|
||||
return self._state
|
||||
|
||||
def set_state(self, value):
|
||||
with self._state_lock:
|
||||
self._state = value
|
||||
|
||||
def start(self, block_callbacks=False):
|
||||
if self.is_running():
|
||||
logger.error('Cannot start PostgreSQL because one is already running.')
|
||||
return True
|
||||
|
||||
self.set_role(self.get_postgres_role_from_data_directory())
|
||||
if os.path.exists(self._postmaster_pid):
|
||||
os.remove(self._postmaster_pid)
|
||||
logger.info('Removed %s', self._postmaster_pid)
|
||||
|
||||
if not block_callbacks:
|
||||
self.set_state('starting')
|
||||
|
||||
env = {'PATH': os.environ.get('PATH')}
|
||||
# pg_ctl will write a FATAL if the username is incorrect. exporting PGUSER if necessary
|
||||
if 'username' in self._superuser and self._superuser['username'] != os.environ.get('USER'):
|
||||
env['PGUSER'] = self._superuser['username']
|
||||
|
||||
self._write_postgresql_conf()
|
||||
self.resolve_connection_addresses()
|
||||
|
||||
options = ' '.join("--{0}='{1}'".format(p, self._server_parameters[p]) for p, v in self.CMDLINE_OPTIONS.items()
|
||||
if self._major_version >= v[2])
|
||||
|
||||
ret = self.pg_ctl('start', '-o', options, env=env, preexec_fn=os.setsid)
|
||||
self._pending_restart = False
|
||||
|
||||
self.set_state('running' if ret else 'start failed')
|
||||
|
||||
self._schedule_load_slots = ret and self.use_slots
|
||||
self.save_configuration_files()
|
||||
# block_callbacks is used during restart to avoid
|
||||
# running start/stop callbacks in addition to restart ones
|
||||
if ret and not block_callbacks:
|
||||
self.call_nowait(ACTION_ON_START)
|
||||
return ret
|
||||
|
||||
def checkpoint(self, connect_kwargs=None):
|
||||
check_not_is_in_recovery = connect_kwargs is not None
|
||||
connect_kwargs = connect_kwargs or self._connect_kwargs
|
||||
for p in ['connect_timeout', 'options']:
|
||||
connect_kwargs.pop(p, None)
|
||||
try:
|
||||
with psycopg2.connect(**connect_kwargs) as conn:
|
||||
conn.autocommit = True
|
||||
with conn.cursor() as cur:
|
||||
cur.execute("SET statement_timeout = 0")
|
||||
if check_not_is_in_recovery:
|
||||
cur.execute('SELECT pg_is_in_recovery()')
|
||||
if cur.fetchone()[0]:
|
||||
return 'is_in_recovery=true'
|
||||
return cur.execute('CHECKPOINT')
|
||||
except psycopg2.Error:
|
||||
logging.exception('Exception during CHECKPOINT')
|
||||
return 'not accessible or not healty'
|
||||
|
||||
def stop(self, mode='fast', block_callbacks=False, checkpoint=True):
|
||||
# make sure we close all connections established against
|
||||
# the former node, otherwise, we might get a stalled one
|
||||
# after kill -9, which would report incorrect data to
|
||||
# patroni.
|
||||
|
||||
self.close_connection()
|
||||
if not self.is_running():
|
||||
if not block_callbacks:
|
||||
self.set_state('stopped')
|
||||
return True
|
||||
|
||||
if checkpoint:
|
||||
self.checkpoint()
|
||||
|
||||
if not block_callbacks:
|
||||
self.set_state('stopping')
|
||||
|
||||
ret = self.pg_ctl('stop', '-m', mode)
|
||||
# block_callbacks is used during restart to avoid
|
||||
# running start/stop callbacks in addition to restart ones
|
||||
if not ret:
|
||||
logger.warning('pg_ctl stop failed')
|
||||
self.set_state('stop failed')
|
||||
elif not block_callbacks:
|
||||
self.set_state('stopped')
|
||||
self.call_nowait(ACTION_ON_STOP)
|
||||
return ret
|
||||
|
||||
def reload(self):
|
||||
ret = self.pg_ctl('reload')
|
||||
if ret:
|
||||
self.call_nowait(ACTION_ON_RELOAD)
|
||||
return ret
|
||||
|
||||
def restart(self):
|
||||
self.set_state('restarting')
|
||||
ret = self.stop(block_callbacks=True) and self.start(block_callbacks=True)
|
||||
if ret:
|
||||
self.call_nowait(ACTION_ON_RESTART)
|
||||
else:
|
||||
self.set_state('restart failed ({0})'.format(self.state))
|
||||
return ret
|
||||
|
||||
def _write_postgresql_conf(self):
|
||||
# rename the original configuration if it is necessary
|
||||
if not os.path.exists(self._postgresql_base_conf):
|
||||
os.rename(self._postgresql_conf, self._postgresql_base_conf)
|
||||
|
||||
with open(self._postgresql_conf, 'w') as f:
|
||||
f.write('# Do not edit this file manually!\n# It will be overwritten by Patroni!\n')
|
||||
f.write("include '{0}'\n\n".format(self._postgresql_base_conf_name))
|
||||
for name, value in sorted(self._server_parameters.items()):
|
||||
if name not in self.CMDLINE_OPTIONS:
|
||||
f.write("{0} = '{1}'\n".format(name, value))
|
||||
|
||||
def is_healthy(self):
|
||||
if not self.is_running():
|
||||
logger.warning('Postgresql is not running.')
|
||||
return False
|
||||
return True
|
||||
|
||||
def check_replication_lag(self, last_leader_operation):
|
||||
return (last_leader_operation or 0) - self.xlog_position() <= self.config.get('maximum_lag_on_failover', 0)
|
||||
|
||||
def write_pg_hba(self, config):
|
||||
with open(os.path.join(self._data_dir, 'pg_hba.conf'), 'a') as f:
|
||||
f.write('\n{}\n'.format('\n'.join(config)))
|
||||
|
||||
def primary_conninfo(self, node_to_follow_url):
|
||||
r = get_conn_kwargs(node_to_follow_url, self._replication)
|
||||
r.update({'application_name': self.name, 'sslmode': 'prefer', 'sslcompression': '1'})
|
||||
keywords = 'user password host port sslmode sslcompression application_name'.split()
|
||||
return ' '.join('{0}={{{0}}}'.format(kw) for kw in keywords).format(**r)
|
||||
|
||||
def check_recovery_conf(self, node_to_follow):
|
||||
if not os.path.isfile(self._recovery_conf):
|
||||
return False
|
||||
|
||||
pattern = node_to_follow and node_to_follow.conn_url and self.primary_conninfo(node_to_follow.conn_url)
|
||||
|
||||
with open(self._recovery_conf, 'r') as f:
|
||||
for line in f:
|
||||
if line.startswith('primary_conninfo'):
|
||||
return pattern and (pattern in line)
|
||||
return not pattern
|
||||
|
||||
def write_recovery_conf(self, node_to_follow):
|
||||
with open(self._recovery_conf, 'w') as f:
|
||||
f.write("standby_mode = 'on'\nrecovery_target_timeline = 'latest'\n")
|
||||
if node_to_follow and node_to_follow.conn_url:
|
||||
f.write("primary_conninfo = '{0}'\n".format(self.primary_conninfo(node_to_follow.conn_url)))
|
||||
if self.use_slots:
|
||||
f.write("primary_slot_name = '{0}'\n".format(self.name))
|
||||
for name, value in self.config.get('recovery_conf', {}).items():
|
||||
if name not in ('standby_mode', 'recovery_target_timeline', 'primary_conninfo', 'primary_slot_name'):
|
||||
f.write("{0} = '{1}'\n".format(name, value))
|
||||
|
||||
def rewind(self, r):
|
||||
# prepare pg_rewind connection
|
||||
env = self.write_pgpass(r)
|
||||
dsn = 'user={user} host={host} port={port} dbname={database} sslmode=prefer sslcompression=1'.format(**r)
|
||||
logger.info('running pg_rewind from %s', dsn)
|
||||
try:
|
||||
return subprocess.call(['pg_rewind', '-D', self._data_dir, '--source-server', dsn], env=env) == 0
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
def controldata(self):
|
||||
""" return the contents of pg_controldata, or non-True value if pg_controldata call failed """
|
||||
result = {}
|
||||
# Don't try to call pg_controldata during backup restore
|
||||
if self._version_file_exists() and self.state != 'creating replica':
|
||||
try:
|
||||
data = subprocess.check_output(['pg_controldata', self._data_dir])
|
||||
if data:
|
||||
data = data.decode('utf-8').splitlines()
|
||||
result = {l.split(':')[0].replace('Current ', '', 1): l.split(':')[1].strip() for l in data if l}
|
||||
except subprocess.CalledProcessError:
|
||||
logger.exception("Error when calling pg_controldata")
|
||||
return result
|
||||
|
||||
def read_postmaster_opts(self):
|
||||
""" returns the list of option names/values from postgres.opts, Empty dict if read failed or no file """
|
||||
result = {}
|
||||
try:
|
||||
with open(os.path.join(self._data_dir, "postmaster.opts")) as f:
|
||||
data = f.read()
|
||||
opts = [opt.strip('"\n') for opt in data.split(' "')]
|
||||
for opt in opts:
|
||||
if '=' in opt and opt.startswith('--'):
|
||||
name, val = opt.split('=', 1)
|
||||
name = name.strip('-')
|
||||
result[name] = val
|
||||
except IOError:
|
||||
logger.exception('Error when reading postmaster.opts')
|
||||
return result
|
||||
|
||||
def single_user_mode(self, command=None, options=None):
|
||||
""" run a given command in a single-user mode. If the command is empty - then just start and stop """
|
||||
cmd = ['postgres', '--single', '-D', self._data_dir]
|
||||
for opt, val in sorted((options or {}).items()):
|
||||
cmd.extend(['-c', '{0}={1}'.format(opt, val)])
|
||||
# need a database name to connect
|
||||
cmd.append(self._database)
|
||||
p = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=open(os.devnull, 'w'), stderr=subprocess.STDOUT)
|
||||
if p:
|
||||
if command:
|
||||
p.communicate('{0}\n'.format(command))
|
||||
p.stdin.close()
|
||||
return p.wait()
|
||||
return 1
|
||||
|
||||
def cleanup_archive_status(self):
|
||||
status_dir = os.path.join(self._data_dir, 'pg_xlog', 'archive_status')
|
||||
try:
|
||||
for f in os.listdir(status_dir):
|
||||
path = os.path.join(status_dir, f)
|
||||
try:
|
||||
if os.path.islink(path):
|
||||
os.unlink(path)
|
||||
elif os.path.isfile(path):
|
||||
os.remove(path)
|
||||
except OSError:
|
||||
logger.exception("Unable to remove %s", path)
|
||||
except OSError:
|
||||
logger.exception("Unable to list %s", status_dir)
|
||||
|
||||
def follow(self, member, leader, recovery=False):
|
||||
if self.check_recovery_conf(member) and not recovery:
|
||||
return True
|
||||
|
||||
change_role = self.role == 'master'
|
||||
|
||||
if change_role:
|
||||
if leader:
|
||||
if leader.name == self.name:
|
||||
self._need_rewind = False
|
||||
member = None
|
||||
if self.is_running():
|
||||
return
|
||||
else:
|
||||
self._need_rewind = bool(leader.conn_url) and self.can_rewind
|
||||
else:
|
||||
self._need_rewind = False
|
||||
member = None
|
||||
|
||||
if self._need_rewind:
|
||||
logger.info("set the rewind flag after demote")
|
||||
|
||||
self.set_role('unknown')
|
||||
if self.is_running() and not self.stop():
|
||||
return logger.warning('Can not run pg_rewind because postgres is still running')
|
||||
|
||||
if not (leader and leader.conn_url):
|
||||
return logger.info('Leader unknown, can not rewind')
|
||||
|
||||
# prepare pg_rewind connection
|
||||
r = get_conn_kwargs(leader.conn_url, self._superuser)
|
||||
|
||||
# first make sure that we are really trying to rewind
|
||||
# from the master and run a checkpoint on a t in order to
|
||||
# make it store the new timeline ([email protected])
|
||||
leader_status = self.checkpoint(r)
|
||||
if leader_status:
|
||||
return logger.warning('Can not use %s for rewind: %s', leader.name, leader_status)
|
||||
|
||||
# at present, pg_rewind only runs when the cluster is shut down cleanly
|
||||
# and not shutdown in recovery. We have to remove the recovery.conf if present
|
||||
# and start/shutdown in a single user mode to emulate this.
|
||||
# XXX: if recovery.conf is linked, it will be written anew as a normal file.
|
||||
if os.path.isfile(self._recovery_conf) or os.path.islink(self._recovery_conf):
|
||||
os.unlink(self._recovery_conf)
|
||||
|
||||
# Archived segments might be useful to pg_rewind,
|
||||
# clean the flags that tell we should remove them.
|
||||
self.cleanup_archive_status()
|
||||
|
||||
# Start in a single user mode and stop to produce a clean shutdown
|
||||
opts = self.read_postmaster_opts()
|
||||
opts.update({'archive_mode': 'on', 'archive_command': 'false'})
|
||||
self.single_user_mode(options=opts)
|
||||
|
||||
if self.rewind(r) or not self.config.get('remove_data_directory_on_rewind_failure', False):
|
||||
self.write_recovery_conf(member)
|
||||
ret = self.start()
|
||||
else:
|
||||
logger.error('unable to rewind the former master')
|
||||
self.remove_data_directory()
|
||||
self.set_role('uninitialized')
|
||||
ret = True
|
||||
self._need_rewind = False
|
||||
else:
|
||||
self.write_recovery_conf(member)
|
||||
ret = self.restart()
|
||||
self.set_role('replica')
|
||||
|
||||
if change_role:
|
||||
self.call_nowait(ACTION_ON_ROLE_CHANGE)
|
||||
return ret
|
||||
|
||||
def save_configuration_files(self):
|
||||
"""
|
||||
copy postgresql.conf to postgresql.conf.backup to be able to retrive configuration files
|
||||
- originally stored as symlinks, those are normally skipped by pg_basebackup
|
||||
- in case of WAL-E basebackup (see http://comments.gmane.org/gmane.comp.db.postgresql.wal-e/239)
|
||||
"""
|
||||
try:
|
||||
for f in self._configuration_to_save:
|
||||
if os.path.isfile(f):
|
||||
shutil.copy(f, f + '.backup')
|
||||
except IOError:
|
||||
logger.exception('unable to create backup copies of configuration files')
|
||||
|
||||
def restore_configuration_files(self):
|
||||
""" restore a previously saved postgresql.conf """
|
||||
try:
|
||||
for f in self._configuration_to_save:
|
||||
if not os.path.isfile(f) and os.path.isfile(f + '.backup'):
|
||||
shutil.copy(f + '.backup', f)
|
||||
except IOError:
|
||||
logger.exception('unable to restore configuration files from backup')
|
||||
|
||||
def promote(self):
|
||||
if self.role == 'master':
|
||||
return True
|
||||
ret = self.pg_ctl('promote')
|
||||
if ret:
|
||||
self.set_role('master')
|
||||
logger.info("cleared rewind flag after becoming the leader")
|
||||
self._need_rewind = False
|
||||
self.call_nowait(ACTION_ON_ROLE_CHANGE)
|
||||
return ret
|
||||
|
||||
def create_or_update_role(self, name, password, options):
|
||||
options = list(map(str.upper, options))
|
||||
if 'NOLOGIN' not in options and 'LOGIN' not in options:
|
||||
options.append('LOGIN')
|
||||
|
||||
self.query("""DO $$
|
||||
BEGIN
|
||||
SET local synchronous_commit = 'local';
|
||||
PERFORM * FROM pg_authid WHERE rolname = %s;
|
||||
IF FOUND THEN
|
||||
ALTER ROLE "{0}" WITH {1} PASSWORD %s;
|
||||
ELSE
|
||||
CREATE ROLE "{0}" WITH {1} PASSWORD %s;
|
||||
END IF;
|
||||
END;
|
||||
$$""".format(name, ' '.join(options)), name, password, password)
|
||||
|
||||
def xlog_position(self):
|
||||
return self.query("""SELECT pg_xlog_location_diff(CASE WHEN pg_is_in_recovery()
|
||||
THEN pg_last_xlog_replay_location()
|
||||
ELSE pg_current_xlog_location()
|
||||
END, '0/0')::bigint""").fetchone()[0]
|
||||
|
||||
def load_replication_slots(self):
|
||||
if self.use_slots and self._schedule_load_slots:
|
||||
cursor = self.query("SELECT slot_name FROM pg_replication_slots WHERE slot_type='physical'")
|
||||
self._replication_slots = [r[0] for r in cursor]
|
||||
self._schedule_load_slots = False
|
||||
|
||||
def sync_replication_slots(self, cluster):
|
||||
if self.use_slots:
|
||||
try:
|
||||
self.load_replication_slots()
|
||||
# if the replicatefrom tag is set on the member - we should not create the replication slot for it on
|
||||
# the current master, because that member would replicate from elsewhere. We still create the slot if
|
||||
# the replicatefrom destination member is currently not a member of the cluster (fallback to the
|
||||
# master), or if replicatefrom destination member happens to be the current master
|
||||
if self.role == 'master':
|
||||
slots = [m.name for m in cluster.members if m.name != self.name and
|
||||
(m.replicatefrom is None or m.replicatefrom == self.name or
|
||||
not cluster.has_member(m.replicatefrom))]
|
||||
else:
|
||||
# only manage slots for replicas that replicate from this one, except for the leader among them
|
||||
slots = [m.name for m in cluster.members if m.replicatefrom == self.name and
|
||||
m.name != cluster.leader.name]
|
||||
# drop unused slots
|
||||
for slot in set(self._replication_slots) - set(slots):
|
||||
self.query("""SELECT pg_drop_replication_slot(%s)
|
||||
WHERE EXISTS(SELECT 1 FROM pg_replication_slots
|
||||
WHERE slot_name = %s)""", slot, slot)
|
||||
|
||||
# create new slots
|
||||
for slot in set(slots) - set(self._replication_slots):
|
||||
self.query("""SELECT pg_create_physical_replication_slot(%s)
|
||||
WHERE NOT EXISTS (SELECT 1 FROM pg_replication_slots
|
||||
WHERE slot_name = %s)""", slot, slot)
|
||||
|
||||
self._replication_slots = slots
|
||||
except psycopg2.Error:
|
||||
logger.exception('Exception when changing replication slots')
|
||||
|
||||
def last_operation(self):
|
||||
return str(self.xlog_position())
|
||||
|
||||
def clone(self, clone_member):
|
||||
"""
|
||||
- initialize the replica from an existing member (master or replica)
|
||||
- initialize the replica using the replica creation method that
|
||||
works without the replication connection (i.e. restore from on-disk
|
||||
base backup)
|
||||
"""
|
||||
|
||||
ret = self.create_replica(clone_member) == 0
|
||||
if ret:
|
||||
self._major_version = self.get_major_version()
|
||||
self.delete_trigger_file()
|
||||
self.restore_configuration_files()
|
||||
return ret
|
||||
|
||||
def bootstrap(self, config):
|
||||
""" Initialize a new node from scratch and start it. """
|
||||
if self._initialize(config) and self.start():
|
||||
for name, value in config['users'].items():
|
||||
if name not in (self._superuser.get('username'), self._replication['username']):
|
||||
self.create_or_update_role(name, value['password'], value.get('options', []))
|
||||
self.create_or_update_role(self._replication['username'], self._replication['password'], ['REPLICATION'])
|
||||
else:
|
||||
raise PostgresException("Could not bootstrap master PostgreSQL")
|
||||
|
||||
def move_data_directory(self):
|
||||
if os.path.isdir(self._data_dir) and not self.is_running():
|
||||
try:
|
||||
new_name = '{0}_{1}'.format(self._data_dir, time.strftime('%Y-%m-%d-%H-%M-%S'))
|
||||
logger.info('renaming data directory to %s', new_name)
|
||||
os.rename(self._data_dir, new_name)
|
||||
except OSError:
|
||||
logger.exception("Could not rename data directory %s", self._data_dir)
|
||||
|
||||
def remove_data_directory(self):
|
||||
logger.info('Removing data directory: %s', self._data_dir)
|
||||
try:
|
||||
if os.path.islink(self._data_dir):
|
||||
os.unlink(self._data_dir)
|
||||
elif not os.path.exists(self._data_dir):
|
||||
return
|
||||
elif os.path.isfile(self._data_dir):
|
||||
os.remove(self._data_dir)
|
||||
elif os.path.isdir(self._data_dir):
|
||||
shutil.rmtree(self._data_dir)
|
||||
except (IOError, OSError):
|
||||
logger.exception('Could not remove data directory %s', self._data_dir)
|
||||
self.move_data_directory()
|
||||
|
||||
def basebackup(self, conn_url, env):
|
||||
# creates a replica data dir using pg_basebackup.
|
||||
# this is the default, built-in create_replica_method
|
||||
# tries twice, then returns failure (as 1)
|
||||
# uses "stream" as the xlog-method to avoid sync issues
|
||||
maxfailures = 2
|
||||
ret = 1
|
||||
for bbfailures in range(0, maxfailures):
|
||||
try:
|
||||
ret = subprocess.call(['pg_basebackup', '--pgdata=' + self._data_dir,
|
||||
'--xlog-method=stream', "--dbname=" + conn_url], env=env)
|
||||
if ret == 0:
|
||||
break
|
||||
|
||||
except Exception as e:
|
||||
logger.error('Error when fetching backup with pg_basebackup: {0}'.format(e))
|
||||
|
||||
if bbfailures < maxfailures - 1:
|
||||
logger.error('Trying again in 5 seconds')
|
||||
time.sleep(5)
|
||||
|
||||
return ret
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,381 @@
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
import tempfile
|
||||
import time
|
||||
|
||||
from six import string_types
|
||||
|
||||
from ..dcs import RemoteMember
|
||||
from ..psycopg import quote_ident, quote_literal
|
||||
from ..utils import deep_compare
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Bootstrap(object):
|
||||
|
||||
def __init__(self, postgresql):
|
||||
self._postgresql = postgresql
|
||||
self._running_custom_bootstrap = False
|
||||
|
||||
@property
|
||||
def running_custom_bootstrap(self):
|
||||
return self._running_custom_bootstrap
|
||||
|
||||
@property
|
||||
def keep_existing_recovery_conf(self):
|
||||
return self._running_custom_bootstrap and self._keep_existing_recovery_conf
|
||||
|
||||
@staticmethod
|
||||
def process_user_options(tool, options, not_allowed_options, error_handler):
|
||||
user_options = []
|
||||
|
||||
def option_is_allowed(name):
|
||||
ret = name not in not_allowed_options
|
||||
if not ret:
|
||||
error_handler('{0} option for {1} is not allowed'.format(name, tool))
|
||||
return ret
|
||||
|
||||
if isinstance(options, dict):
|
||||
for k, v in options.items():
|
||||
if k and v:
|
||||
user_options.append('--{0}={1}'.format(k, v))
|
||||
elif isinstance(options, list):
|
||||
for opt in options:
|
||||
if isinstance(opt, string_types) and option_is_allowed(opt):
|
||||
user_options.append('--{0}'.format(opt))
|
||||
elif isinstance(opt, dict):
|
||||
keys = list(opt.keys())
|
||||
if len(keys) != 1 or not isinstance(opt[keys[0]], string_types) or not option_is_allowed(keys[0]):
|
||||
error_handler('Error when parsing {0} key-value option {1}: only one key-value is allowed'
|
||||
' and value should be a string'.format(tool, opt[keys[0]]))
|
||||
user_options.append('--{0}={1}'.format(keys[0], opt[keys[0]]))
|
||||
else:
|
||||
error_handler('Error when parsing {0} option {1}: value should be string value'
|
||||
' or a single key-value pair'.format(tool, opt))
|
||||
else:
|
||||
error_handler('{0} options must be list or dict'.format(tool))
|
||||
return user_options
|
||||
|
||||
def _initdb(self, config):
|
||||
self._postgresql.set_state('initalizing new cluster')
|
||||
not_allowed_options = ('pgdata', 'nosync', 'pwfile', 'sync-only', 'version')
|
||||
|
||||
def error_handler(e):
|
||||
raise Exception(e)
|
||||
|
||||
options = self.process_user_options('initdb', config or [], not_allowed_options, error_handler)
|
||||
pwfile = None
|
||||
|
||||
if self._postgresql.config.superuser:
|
||||
if 'username' in self._postgresql.config.superuser:
|
||||
options.append('--username={0}'.format(self._postgresql.config.superuser['username']))
|
||||
if 'password' in self._postgresql.config.superuser:
|
||||
(fd, pwfile) = tempfile.mkstemp()
|
||||
os.write(fd, self._postgresql.config.superuser['password'].encode('utf-8'))
|
||||
os.close(fd)
|
||||
options.append('--pwfile={0}'.format(pwfile))
|
||||
options = ['-o', ' '.join(options)] if options else []
|
||||
|
||||
ret = self._postgresql.pg_ctl('initdb', *options)
|
||||
if pwfile:
|
||||
os.remove(pwfile)
|
||||
if ret:
|
||||
self._postgresql.configure_server_parameters()
|
||||
else:
|
||||
self._postgresql.set_state('initdb failed')
|
||||
return ret
|
||||
|
||||
def _post_restore(self):
|
||||
self._postgresql.config.restore_configuration_files()
|
||||
self._postgresql.configure_server_parameters()
|
||||
|
||||
# make sure there is no trigger file or postgres will be automatically promoted
|
||||
trigger_file = self._postgresql.config.triggerfile_good_name
|
||||
trigger_file = (self._postgresql.config.get('recovery_conf') or {}).get(trigger_file) or 'promote'
|
||||
trigger_file = os.path.abspath(os.path.join(self._postgresql.data_dir, trigger_file))
|
||||
if os.path.exists(trigger_file):
|
||||
os.unlink(trigger_file)
|
||||
|
||||
def _custom_bootstrap(self, config):
|
||||
self._postgresql.set_state('running custom bootstrap script')
|
||||
params = [] if config.get('no_params') else ['--scope=' + self._postgresql.scope,
|
||||
'--datadir=' + self._postgresql.data_dir]
|
||||
try:
|
||||
logger.info('Running custom bootstrap script: %s', config['command'])
|
||||
if self._postgresql.cancellable.call(shlex.split(config['command']) + params) != 0:
|
||||
self._postgresql.set_state('custom bootstrap failed')
|
||||
return False
|
||||
except Exception:
|
||||
logger.exception('Exception during custom bootstrap')
|
||||
return False
|
||||
self._post_restore()
|
||||
|
||||
if 'recovery_conf' in config:
|
||||
self._postgresql.config.write_recovery_conf(config['recovery_conf'])
|
||||
elif not self.keep_existing_recovery_conf:
|
||||
self._postgresql.config.remove_recovery_conf()
|
||||
return True
|
||||
|
||||
def call_post_bootstrap(self, config):
|
||||
"""
|
||||
runs a script after initdb or custom bootstrap script is called and waits until completion.
|
||||
"""
|
||||
cmd = config.get('post_bootstrap') or config.get('post_init')
|
||||
if cmd:
|
||||
r = self._postgresql.config.local_connect_kwargs
|
||||
connstring = self._postgresql.config.format_dsn(r, True)
|
||||
if 'host' not in r:
|
||||
# https://www.postgresql.org/docs/current/static/libpq-pgpass.html
|
||||
# A host name of localhost matches both TCP (host name localhost) and Unix domain socket
|
||||
# (pghost empty or the default socket directory) connections coming from the local machine.
|
||||
r['host'] = 'localhost' # set it to localhost to write into pgpass
|
||||
|
||||
env = self._postgresql.config.write_pgpass(r)
|
||||
env['PGOPTIONS'] = '-c synchronous_commit=local'
|
||||
|
||||
try:
|
||||
ret = self._postgresql.cancellable.call(shlex.split(cmd) + [connstring], env=env)
|
||||
except OSError:
|
||||
logger.error('post_init script %s failed', cmd)
|
||||
return False
|
||||
if ret != 0:
|
||||
logger.error('post_init script %s returned non-zero code %d', cmd, ret)
|
||||
return False
|
||||
return True
|
||||
|
||||
def create_replica(self, clone_member):
|
||||
"""
|
||||
create the replica according to the replica_method
|
||||
defined by the user. this is a list, so we need to
|
||||
loop through all methods the user supplies
|
||||
"""
|
||||
|
||||
self._postgresql.set_state('creating replica')
|
||||
self._postgresql.schedule_sanity_checks_after_pause()
|
||||
|
||||
is_remote_master = isinstance(clone_member, RemoteMember)
|
||||
|
||||
# get list of replica methods either from clone member or from
|
||||
# the config. If there is no configuration key, or no value is
|
||||
# specified, use basebackup
|
||||
replica_methods = (clone_member.create_replica_methods if is_remote_master
|
||||
else self._postgresql.create_replica_methods) or ['basebackup']
|
||||
|
||||
if clone_member and clone_member.conn_url:
|
||||
r = clone_member.conn_kwargs(self._postgresql.config.replication)
|
||||
# add the credentials to connect to the replica origin to pgpass.
|
||||
env = self._postgresql.config.write_pgpass(r)
|
||||
connstring = self._postgresql.config.format_dsn(r, True)
|
||||
else:
|
||||
connstring = ''
|
||||
env = os.environ.copy()
|
||||
# if we don't have any source, leave only replica methods that work without it
|
||||
replica_methods = [r for r in replica_methods
|
||||
if self._postgresql.replica_method_can_work_without_replication_connection(r)]
|
||||
|
||||
# go through them in priority order
|
||||
ret = 1
|
||||
for replica_method in replica_methods:
|
||||
if self._postgresql.cancellable.is_cancelled:
|
||||
break
|
||||
|
||||
method_config = self._postgresql.replica_method_options(replica_method)
|
||||
|
||||
# if the method is basebackup, then use the built-in
|
||||
if replica_method == "basebackup":
|
||||
ret = self.basebackup(connstring, env, method_config)
|
||||
if ret == 0:
|
||||
logger.info("replica has been created using basebackup")
|
||||
# if basebackup succeeds, exit with success
|
||||
break
|
||||
else:
|
||||
if not self._postgresql.data_directory_empty():
|
||||
if method_config.get('keep_data', False):
|
||||
logger.info('Leaving data directory uncleaned')
|
||||
else:
|
||||
self._postgresql.remove_data_directory()
|
||||
|
||||
cmd = replica_method
|
||||
# user-defined method; check for configuration
|
||||
# not required, actually
|
||||
if method_config:
|
||||
# look to see if the user has supplied a full command path
|
||||
# if not, use the method name as the command
|
||||
cmd = method_config.pop('command', cmd)
|
||||
|
||||
# add the default parameters
|
||||
if not method_config.get('no_params', False):
|
||||
method_config.update({"scope": self._postgresql.scope,
|
||||
"role": "replica",
|
||||
"datadir": self._postgresql.data_dir,
|
||||
"connstring": connstring})
|
||||
else:
|
||||
for param in ('no_params', 'no_master', 'keep_data'):
|
||||
method_config.pop(param, None)
|
||||
params = ["--{0}={1}".format(arg, val) for arg, val in method_config.items()]
|
||||
try:
|
||||
# call script with the full set of parameters
|
||||
ret = self._postgresql.cancellable.call(shlex.split(cmd) + params, env=env)
|
||||
# if we succeeded, stop
|
||||
if ret == 0:
|
||||
logger.info('replica has been created using %s', replica_method)
|
||||
break
|
||||
else:
|
||||
logger.error('Error creating replica using method %s: %s exited with code=%s',
|
||||
replica_method, cmd, ret)
|
||||
except Exception:
|
||||
logger.exception('Error creating replica using method %s', replica_method)
|
||||
ret = 1
|
||||
|
||||
self._postgresql.set_state('stopped')
|
||||
return ret
|
||||
|
||||
def basebackup(self, conn_url, env, options):
|
||||
# creates a replica data dir using pg_basebackup.
|
||||
# this is the default, built-in create_replica_methods
|
||||
# tries twice, then returns failure (as 1)
|
||||
# uses "stream" as the xlog-method to avoid sync issues
|
||||
# supports additional user-supplied options, those are not validated
|
||||
maxfailures = 2
|
||||
ret = 1
|
||||
not_allowed_options = ('pgdata', 'format', 'wal-method', 'xlog-method', 'gzip',
|
||||
'version', 'compress', 'dbname', 'host', 'port', 'username', 'password')
|
||||
user_options = self.process_user_options('basebackup', options, not_allowed_options, logger.error)
|
||||
|
||||
for bbfailures in range(0, maxfailures):
|
||||
if self._postgresql.cancellable.is_cancelled:
|
||||
break
|
||||
if not self._postgresql.data_directory_empty():
|
||||
self._postgresql.remove_data_directory()
|
||||
try:
|
||||
ret = self._postgresql.cancellable.call([self._postgresql.pgcommand('pg_basebackup'),
|
||||
'--pgdata=' + self._postgresql.data_dir, '-X', 'stream',
|
||||
'--dbname=' + conn_url] + user_options, env=env)
|
||||
if ret == 0:
|
||||
break
|
||||
else:
|
||||
logger.error('Error when fetching backup: pg_basebackup exited with code=%s', ret)
|
||||
|
||||
except Exception as e:
|
||||
logger.error('Error when fetching backup with pg_basebackup: %s', e)
|
||||
|
||||
if bbfailures < maxfailures - 1:
|
||||
logger.warning('Trying again in 5 seconds')
|
||||
time.sleep(5)
|
||||
|
||||
return ret
|
||||
|
||||
def clone(self, clone_member):
|
||||
"""
|
||||
- initialize the replica from an existing member (master or replica)
|
||||
- initialize the replica using the replica creation method that
|
||||
works without the replication connection (i.e. restore from on-disk
|
||||
base backup)
|
||||
"""
|
||||
|
||||
ret = self.create_replica(clone_member) == 0
|
||||
if ret:
|
||||
self._post_restore()
|
||||
return ret
|
||||
|
||||
def bootstrap(self, config):
|
||||
""" Initialize a new node from scratch and start it. """
|
||||
pg_hba = config.get('pg_hba', [])
|
||||
method = config.get('method') or 'initdb'
|
||||
if method != 'initdb' and method in config and 'command' in config[method]:
|
||||
self._keep_existing_recovery_conf = config[method].get('keep_existing_recovery_conf')
|
||||
self._running_custom_bootstrap = True
|
||||
do_initialize = self._custom_bootstrap
|
||||
else:
|
||||
method = 'initdb'
|
||||
do_initialize = self._initdb
|
||||
return do_initialize(config.get(method)) and self._postgresql.config.append_pg_hba(pg_hba) \
|
||||
and self._postgresql.config.save_configuration_files() and self._postgresql.start()
|
||||
|
||||
def create_or_update_role(self, name, password, options):
|
||||
options = list(map(str.upper, options))
|
||||
if 'NOLOGIN' not in options and 'LOGIN' not in options:
|
||||
options.append('LOGIN')
|
||||
|
||||
if password:
|
||||
options.extend(['PASSWORD', quote_literal(password)])
|
||||
|
||||
sql = """DO $$
|
||||
BEGIN
|
||||
SET local synchronous_commit = 'local';
|
||||
PERFORM * FROM pg_catalog.pg_authid WHERE rolname = {0};
|
||||
IF FOUND THEN
|
||||
ALTER ROLE {1} WITH {2};
|
||||
ELSE
|
||||
CREATE ROLE {1} WITH {2};
|
||||
END IF;
|
||||
END;$$""".format(quote_literal(name), quote_ident(name, self._postgresql.connection()), ' '.join(options))
|
||||
self._postgresql.query('SET log_statement TO none')
|
||||
self._postgresql.query('SET log_min_duration_statement TO -1')
|
||||
self._postgresql.query("SET log_min_error_statement TO 'log'")
|
||||
try:
|
||||
self._postgresql.query(sql)
|
||||
finally:
|
||||
self._postgresql.query('RESET log_min_error_statement')
|
||||
self._postgresql.query('RESET log_min_duration_statement')
|
||||
self._postgresql.query('RESET log_statement')
|
||||
|
||||
def post_bootstrap(self, config, task):
|
||||
try:
|
||||
postgresql = self._postgresql
|
||||
superuser = postgresql.config.superuser
|
||||
if 'username' in superuser and 'password' in superuser:
|
||||
self.create_or_update_role(superuser['username'], superuser['password'], ['SUPERUSER'])
|
||||
|
||||
task.complete(self.call_post_bootstrap(config))
|
||||
if task.result:
|
||||
replication = postgresql.config.replication
|
||||
self.create_or_update_role(replication['username'], replication.get('password'), ['REPLICATION'])
|
||||
|
||||
rewind = postgresql.config.rewind_credentials
|
||||
if not deep_compare(rewind, superuser):
|
||||
self.create_or_update_role(rewind['username'], rewind.get('password'), [])
|
||||
for f in ('pg_ls_dir(text, boolean, boolean)', 'pg_stat_file(text, boolean)',
|
||||
'pg_read_binary_file(text)', 'pg_read_binary_file(text, bigint, bigint, boolean)'):
|
||||
sql = """DO $$
|
||||
BEGIN
|
||||
SET local synchronous_commit = 'local';
|
||||
GRANT EXECUTE ON function pg_catalog.{0} TO {1};
|
||||
END;$$""".format(f, quote_ident(rewind['username'], self._postgresql.connection()))
|
||||
postgresql.query(sql)
|
||||
|
||||
for name, value in (config.get('users') or {}).items():
|
||||
if all(name != a.get('username') for a in (superuser, replication, rewind)):
|
||||
self.create_or_update_role(name, value.get('password'), value.get('options', []))
|
||||
|
||||
# We were doing a custom bootstrap instead of running initdb, therefore we opened trust
|
||||
# access from certain addresses to be able to reach cluster and change password
|
||||
if self._running_custom_bootstrap:
|
||||
self._running_custom_bootstrap = False
|
||||
# If we don't have custom configuration for pg_hba.conf we need to restore original file
|
||||
if not postgresql.config.get('pg_hba'):
|
||||
if os.path.exists(postgresql.config.pg_hba_conf):
|
||||
os.unlink(postgresql.config.pg_hba_conf)
|
||||
postgresql.config.restore_configuration_files()
|
||||
postgresql.config.write_postgresql_conf()
|
||||
postgresql.config.replace_pg_ident()
|
||||
|
||||
# at this point there should be no recovery.conf
|
||||
postgresql.config.remove_recovery_conf()
|
||||
|
||||
if postgresql.config.hba_file:
|
||||
postgresql.restart()
|
||||
else:
|
||||
postgresql.config.replace_pg_hba()
|
||||
if postgresql.pending_restart:
|
||||
postgresql.restart()
|
||||
else:
|
||||
postgresql.reload()
|
||||
time.sleep(1) # give a time to postgres to "reload" configuration files
|
||||
postgresql.connection().close() # close connection to reconnect with a new password
|
||||
except Exception:
|
||||
logger.exception('post_bootstrap')
|
||||
task.complete(False)
|
||||
return task.result
|
||||
@@ -0,0 +1,36 @@
|
||||
import logging
|
||||
|
||||
from patroni.postgresql.cancellable import CancellableExecutor
|
||||
from threading import Condition, Thread
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class CallbackExecutor(CancellableExecutor, Thread):
|
||||
|
||||
def __init__(self):
|
||||
CancellableExecutor.__init__(self)
|
||||
Thread.__init__(self)
|
||||
self.daemon = True
|
||||
self._cmd = None
|
||||
self._condition = Condition()
|
||||
self.start()
|
||||
|
||||
def call(self, cmd):
|
||||
self._kill_process()
|
||||
with self._condition:
|
||||
self._cmd = cmd
|
||||
self._condition.notify()
|
||||
|
||||
def run(self):
|
||||
while True:
|
||||
with self._condition:
|
||||
if self._cmd is None:
|
||||
self._condition.wait()
|
||||
cmd, self._cmd = self._cmd, None
|
||||
|
||||
with self._lock:
|
||||
if not self._start_process(cmd, close_fds=True):
|
||||
continue
|
||||
self._process.wait()
|
||||
self._kill_children()
|
||||
@@ -0,0 +1,133 @@
|
||||
import logging
|
||||
import psutil
|
||||
import subprocess
|
||||
|
||||
from patroni.exceptions import PostgresException
|
||||
from patroni.utils import polling_loop
|
||||
from threading import Lock
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class CancellableExecutor(object):
|
||||
|
||||
"""
|
||||
There must be only one such process so that AsyncExecutor can easily cancel it.
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
self._process = None
|
||||
self._process_cmd = None
|
||||
self._process_children = []
|
||||
self._lock = Lock()
|
||||
|
||||
def _start_process(self, cmd, *args, **kwargs):
|
||||
"""This method must be executed only when the `_lock` is acquired"""
|
||||
|
||||
try:
|
||||
self._process_children = []
|
||||
self._process_cmd = cmd
|
||||
self._process = psutil.Popen(cmd, *args, **kwargs)
|
||||
except Exception:
|
||||
return logger.exception('Failed to execute %s', cmd)
|
||||
return True
|
||||
|
||||
def _kill_process(self):
|
||||
with self._lock:
|
||||
if self._process is not None and self._process.is_running() and not self._process_children:
|
||||
try:
|
||||
self._process.suspend() # Suspend the process before getting list of children
|
||||
except psutil.Error as e:
|
||||
logger.info('Failed to suspend the process: %s', e.msg)
|
||||
|
||||
try:
|
||||
self._process_children = self._process.children(recursive=True)
|
||||
except psutil.Error:
|
||||
pass
|
||||
|
||||
try:
|
||||
self._process.kill()
|
||||
logger.warning('Killed %s because it was still running', self._process_cmd)
|
||||
except psutil.NoSuchProcess:
|
||||
pass
|
||||
except psutil.AccessDenied as e:
|
||||
logger.warning('Failed to kill the process: %s', e.msg)
|
||||
|
||||
def _kill_children(self):
|
||||
waitlist = []
|
||||
with self._lock:
|
||||
for child in self._process_children:
|
||||
try:
|
||||
child.kill()
|
||||
except psutil.NoSuchProcess:
|
||||
continue
|
||||
except psutil.AccessDenied as e:
|
||||
logger.info('Failed to kill child process: %s', e.msg)
|
||||
waitlist.append(child)
|
||||
psutil.wait_procs(waitlist)
|
||||
|
||||
|
||||
class CancellableSubprocess(CancellableExecutor):
|
||||
|
||||
def __init__(self):
|
||||
super(CancellableSubprocess, self).__init__()
|
||||
self._is_cancelled = False
|
||||
|
||||
def call(self, *args, **kwargs):
|
||||
for s in ('stdin', 'stdout', 'stderr'):
|
||||
kwargs.pop(s, None)
|
||||
|
||||
communicate = kwargs.pop('communicate', None)
|
||||
if isinstance(communicate, dict):
|
||||
input_data = communicate.get('input')
|
||||
if input_data:
|
||||
if input_data[-1] != '\n':
|
||||
input_data += '\n'
|
||||
input_data = input_data.encode('utf-8')
|
||||
kwargs['stdin'] = subprocess.PIPE
|
||||
kwargs['stdout'] = subprocess.PIPE
|
||||
kwargs['stderr'] = subprocess.PIPE
|
||||
|
||||
try:
|
||||
with self._lock:
|
||||
if self._is_cancelled:
|
||||
raise PostgresException('cancelled')
|
||||
|
||||
self._is_cancelled = False
|
||||
started = self._start_process(*args, **kwargs)
|
||||
|
||||
if started:
|
||||
if isinstance(communicate, dict):
|
||||
communicate['stdout'], communicate['stderr'] = self._process.communicate(input_data)
|
||||
return self._process.wait()
|
||||
finally:
|
||||
with self._lock:
|
||||
self._process = None
|
||||
self._kill_children()
|
||||
|
||||
def reset_is_cancelled(self):
|
||||
with self._lock:
|
||||
self._is_cancelled = False
|
||||
|
||||
@property
|
||||
def is_cancelled(self):
|
||||
with self._lock:
|
||||
return self._is_cancelled
|
||||
|
||||
def cancel(self, kill=False):
|
||||
with self._lock:
|
||||
self._is_cancelled = True
|
||||
if self._process is None or not self._process.is_running():
|
||||
return
|
||||
|
||||
logger.info('Terminating %s', self._process_cmd)
|
||||
self._process.terminate()
|
||||
|
||||
for _ in polling_loop(10):
|
||||
with self._lock:
|
||||
if self._process is None or not self._process.is_running():
|
||||
return
|
||||
if kill:
|
||||
break
|
||||
|
||||
self._kill_process()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,48 @@
|
||||
import logging
|
||||
|
||||
from contextlib import contextmanager
|
||||
from threading import Lock
|
||||
|
||||
from .. import psycopg
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Connection(object):
|
||||
|
||||
def __init__(self):
|
||||
self._lock = Lock()
|
||||
self._connection = None
|
||||
self._cursor_holder = None
|
||||
|
||||
def set_conn_kwargs(self, conn_kwargs):
|
||||
self._conn_kwargs = conn_kwargs
|
||||
|
||||
def get(self):
|
||||
with self._lock:
|
||||
if not self._connection or self._connection.closed != 0:
|
||||
self._connection = psycopg.connect(**self._conn_kwargs)
|
||||
self._connection.autocommit = True
|
||||
self.server_version = self._connection.server_version
|
||||
return self._connection
|
||||
|
||||
def cursor(self):
|
||||
if not self._cursor_holder or self._cursor_holder.closed or self._cursor_holder.connection.closed != 0:
|
||||
logger.info("establishing a new patroni connection to the postgres cluster")
|
||||
self._cursor_holder = self.get().cursor()
|
||||
return self._cursor_holder
|
||||
|
||||
def close(self):
|
||||
if self._connection and self._connection.closed == 0:
|
||||
self._connection.close()
|
||||
logger.info("closed patroni connection to the postgresql cluster")
|
||||
self._cursor_holder = self._connection = None
|
||||
|
||||
|
||||
@contextmanager
|
||||
def get_connection_cursor(**kwargs):
|
||||
conn = psycopg.connect(**kwargs)
|
||||
conn.autocommit = True
|
||||
with conn.cursor() as cur:
|
||||
yield cur
|
||||
conn.close()
|
||||
@@ -0,0 +1,75 @@
|
||||
import logging
|
||||
|
||||
from patroni.exceptions import PostgresException
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def postgres_version_to_int(pg_version):
|
||||
"""Convert the server_version to integer
|
||||
|
||||
>>> postgres_version_to_int('9.5.3')
|
||||
90503
|
||||
>>> postgres_version_to_int('9.3.13')
|
||||
90313
|
||||
>>> postgres_version_to_int('10.1')
|
||||
100001
|
||||
>>> postgres_version_to_int('10') # doctest: +IGNORE_EXCEPTION_DETAIL
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
PostgresException: 'Invalid PostgreSQL version format: X.Y or X.Y.Z is accepted: 10'
|
||||
>>> postgres_version_to_int('9.6') # doctest: +IGNORE_EXCEPTION_DETAIL
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
PostgresException: 'Invalid PostgreSQL version format: X.Y or X.Y.Z is accepted: 9.6'
|
||||
>>> postgres_version_to_int('a.b.c') # doctest: +IGNORE_EXCEPTION_DETAIL
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
PostgresException: 'Invalid PostgreSQL version: a.b.c'
|
||||
"""
|
||||
|
||||
try:
|
||||
components = list(map(int, pg_version.split('.')))
|
||||
except ValueError:
|
||||
raise PostgresException('Invalid PostgreSQL version: {0}'.format(pg_version))
|
||||
|
||||
if len(components) < 2 or len(components) == 2 and components[0] < 10 or len(components) > 3:
|
||||
raise PostgresException('Invalid PostgreSQL version format: X.Y or X.Y.Z is accepted: {0}'.format(pg_version))
|
||||
|
||||
if len(components) == 2:
|
||||
# new style version numbers, i.e. 10.1 becomes 100001
|
||||
components.insert(1, 0)
|
||||
|
||||
return int(''.join('{0:02d}'.format(c) for c in components))
|
||||
|
||||
|
||||
def postgres_major_version_to_int(pg_version):
|
||||
"""
|
||||
>>> postgres_major_version_to_int('10')
|
||||
100000
|
||||
>>> postgres_major_version_to_int('9.6')
|
||||
90600
|
||||
"""
|
||||
return postgres_version_to_int(pg_version + '.0')
|
||||
|
||||
|
||||
def parse_lsn(lsn):
|
||||
t = lsn.split('/')
|
||||
return int(t[0], 16) * 0x100000000 + int(t[1], 16)
|
||||
|
||||
|
||||
def parse_history(data):
|
||||
for line in data.split('\n'):
|
||||
values = line.strip().split('\t')
|
||||
if len(values) == 3:
|
||||
try:
|
||||
values[0] = int(values[0])
|
||||
values[1] = parse_lsn(values[1])
|
||||
yield values
|
||||
except (IndexError, ValueError):
|
||||
logger.exception('Exception when parsing timeline history line "%s"', values)
|
||||
|
||||
|
||||
def format_lsn(lsn, full=False):
|
||||
template = '{0:X}/{1:08X}' if full else '{0:X}/{1:X}'
|
||||
return template.format(lsn >> 32, lsn & 0xFFFFFFFF)
|
||||
@@ -0,0 +1,246 @@
|
||||
import logging
|
||||
import multiprocessing
|
||||
import os
|
||||
import psutil
|
||||
import re
|
||||
import signal
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
from patroni import PATRONI_ENV_PREFIX, KUBERNETES_ENV_PREFIX
|
||||
|
||||
# avoid spawning the resource tracker process
|
||||
if sys.version_info >= (3, 8): # pragma: no cover
|
||||
import multiprocessing.resource_tracker
|
||||
multiprocessing.resource_tracker.getfd = lambda: 0
|
||||
elif sys.version_info >= (3, 4): # pragma: no cover
|
||||
import multiprocessing.semaphore_tracker
|
||||
multiprocessing.semaphore_tracker.getfd = lambda: 0
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
STOP_SIGNALS = {
|
||||
'smart': 'TERM',
|
||||
'fast': 'INT',
|
||||
'immediate': 'QUIT',
|
||||
}
|
||||
|
||||
|
||||
def pg_ctl_start(conn, cmdline, env):
|
||||
if os.name != 'nt':
|
||||
os.setsid()
|
||||
try:
|
||||
postmaster = subprocess.Popen(cmdline, close_fds=True, env=env)
|
||||
conn.send(postmaster.pid)
|
||||
except Exception:
|
||||
logger.exception('Failed to execute %s', cmdline)
|
||||
conn.send(None)
|
||||
conn.close()
|
||||
|
||||
|
||||
class PostmasterProcess(psutil.Process):
|
||||
|
||||
def __init__(self, pid):
|
||||
self.is_single_user = False
|
||||
if pid < 0:
|
||||
pid = -pid
|
||||
self.is_single_user = True
|
||||
super(PostmasterProcess, self).__init__(pid)
|
||||
|
||||
@staticmethod
|
||||
def _read_postmaster_pidfile(data_dir):
|
||||
"""Reads and parses postmaster.pid from the data directory
|
||||
|
||||
:returns dictionary of values if successful, empty dictionary otherwise
|
||||
"""
|
||||
pid_line_names = ['pid', 'data_dir', 'start_time', 'port', 'socket_dir', 'listen_addr', 'shmem_key']
|
||||
try:
|
||||
with open(os.path.join(data_dir, 'postmaster.pid')) as f:
|
||||
return {name: line.rstrip('\n') for name, line in zip(pid_line_names, f)}
|
||||
except IOError:
|
||||
return {}
|
||||
|
||||
def _is_postmaster_process(self):
|
||||
try:
|
||||
start_time = int(self._postmaster_pid.get('start_time', 0))
|
||||
if start_time and abs(self.create_time() - start_time) > 3:
|
||||
logger.info('Process %s is not postmaster, too much difference between PID file start time %s and '
|
||||
'process start time %s', self.pid, self.create_time(), start_time)
|
||||
return False
|
||||
except ValueError:
|
||||
logger.warning('Garbage start time value in pid file: %r', self._postmaster_pid.get('start_time'))
|
||||
|
||||
# Extra safety check. The process can't be ourselves, our parent or our direct child.
|
||||
if self.pid == os.getpid() or self.pid == os.getppid() or self.ppid() == os.getpid():
|
||||
logger.info('Patroni (pid=%s, ppid=%s), "fake postmaster" (pid=%s, ppid=%s)',
|
||||
os.getpid(), os.getppid(), self.pid, self.ppid())
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def _from_pidfile(cls, data_dir):
|
||||
postmaster_pid = PostmasterProcess._read_postmaster_pidfile(data_dir)
|
||||
try:
|
||||
pid = int(postmaster_pid.get('pid', 0))
|
||||
if pid:
|
||||
proc = cls(pid)
|
||||
proc._postmaster_pid = postmaster_pid
|
||||
return proc
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
@staticmethod
|
||||
def from_pidfile(data_dir):
|
||||
try:
|
||||
proc = PostmasterProcess._from_pidfile(data_dir)
|
||||
return proc if proc and proc._is_postmaster_process() else None
|
||||
except psutil.NoSuchProcess:
|
||||
return None
|
||||
|
||||
@classmethod
|
||||
def from_pid(cls, pid):
|
||||
try:
|
||||
return cls(pid)
|
||||
except psutil.NoSuchProcess:
|
||||
return None
|
||||
|
||||
def signal_kill(self):
|
||||
"""to suspend and kill postmaster and all children
|
||||
|
||||
:returns True if postmaster and children are killed, False if error
|
||||
"""
|
||||
try:
|
||||
self.suspend()
|
||||
except psutil.NoSuchProcess:
|
||||
return True
|
||||
except psutil.Error as e:
|
||||
logger.warning('Failed to suspend postmaster: %s', e)
|
||||
|
||||
try:
|
||||
children = self.children(recursive=True)
|
||||
except psutil.NoSuchProcess:
|
||||
return True
|
||||
except psutil.Error as e:
|
||||
logger.warning('Failed to get a list of postmaster children: %s', e)
|
||||
children = []
|
||||
|
||||
try:
|
||||
self.kill()
|
||||
except psutil.NoSuchProcess:
|
||||
return True
|
||||
except psutil.Error as e:
|
||||
logger.warning('Could not kill postmaster: %s', e)
|
||||
return False
|
||||
|
||||
for child in children:
|
||||
try:
|
||||
child.kill()
|
||||
except psutil.Error:
|
||||
pass
|
||||
psutil.wait_procs(children + [self])
|
||||
return True
|
||||
|
||||
def signal_stop(self, mode, pg_ctl='pg_ctl'):
|
||||
"""Signal postmaster process to stop
|
||||
|
||||
:returns None if signaled, True if process is already gone, False if error
|
||||
"""
|
||||
if self.is_single_user:
|
||||
logger.warning("Cannot stop server; single-user server is running (PID: {0})".format(self.pid))
|
||||
return False
|
||||
if os.name != 'posix':
|
||||
return self.pg_ctl_kill(mode, pg_ctl)
|
||||
try:
|
||||
self.send_signal(getattr(signal, 'SIG' + STOP_SIGNALS[mode]))
|
||||
except psutil.NoSuchProcess:
|
||||
return True
|
||||
except psutil.AccessDenied as e:
|
||||
logger.warning("Could not send stop signal to PostgreSQL (error: {0})".format(e))
|
||||
return False
|
||||
|
||||
return None
|
||||
|
||||
def pg_ctl_kill(self, mode, pg_ctl):
|
||||
try:
|
||||
status = subprocess.call([pg_ctl, "kill", STOP_SIGNALS[mode], str(self.pid)])
|
||||
except OSError:
|
||||
return False
|
||||
if status == 0:
|
||||
return None
|
||||
else:
|
||||
return not self.is_running()
|
||||
|
||||
def wait_for_user_backends_to_close(self):
|
||||
# These regexps are cross checked against versions PostgreSQL 9.1 .. 11
|
||||
aux_proc_re = re.compile("(?:postgres:)( .*:)? (?:(?:archiver|startup|autovacuum launcher|autovacuum worker|"
|
||||
"checkpointer|logger|stats collector|wal receiver|wal writer|writer)(?: process )?|"
|
||||
"walreceiver|wal sender process|walsender|walwriter|background writer|"
|
||||
"logical replication launcher|logical replication worker for|bgworker:) ")
|
||||
|
||||
try:
|
||||
children = self.children()
|
||||
except psutil.Error:
|
||||
return logger.debug('Failed to get list of postmaster children')
|
||||
|
||||
user_backends = []
|
||||
user_backends_cmdlines = []
|
||||
for child in children:
|
||||
try:
|
||||
cmdline = child.cmdline()
|
||||
if cmdline and not aux_proc_re.match(cmdline[0]):
|
||||
user_backends.append(child)
|
||||
user_backends_cmdlines.append(cmdline[0])
|
||||
except psutil.NoSuchProcess:
|
||||
pass
|
||||
if user_backends:
|
||||
logger.debug('Waiting for user backends %s to close', ', '.join(user_backends_cmdlines))
|
||||
psutil.wait_procs(user_backends)
|
||||
logger.debug("Backends closed")
|
||||
|
||||
@staticmethod
|
||||
def start(pgcommand, data_dir, conf, options):
|
||||
# Unfortunately `pg_ctl start` does not return postmaster pid to us. Without this information
|
||||
# it is hard to know the current state of postgres startup, so we had to reimplement pg_ctl start
|
||||
# in python. It will start postgres, wait for port to be open and wait until postgres will start
|
||||
# accepting connections.
|
||||
# Important!!! We can't just start postgres using subprocess.Popen, because in this case it
|
||||
# will be our child for the rest of our live and we will have to take care of it (`waitpid`).
|
||||
# So we will use the same approach as pg_ctl uses: start a new process, which will start postgres.
|
||||
# This process will write postmaster pid to stdout and exit immediately. Now it's responsibility
|
||||
# of init process to take care about postmaster.
|
||||
# In order to make everything portable we can't use fork&exec approach here, so we will call
|
||||
# ourselves and pass list of arguments which must be used to start postgres.
|
||||
# On Windows, in order to run a side-by-side assembly the specified env must include a valid SYSTEMROOT.
|
||||
env = {p: os.environ[p] for p in os.environ if not p.startswith(
|
||||
PATRONI_ENV_PREFIX) and not p.startswith(KUBERNETES_ENV_PREFIX)}
|
||||
try:
|
||||
proc = PostmasterProcess._from_pidfile(data_dir)
|
||||
if proc and not proc._is_postmaster_process():
|
||||
# Upon start postmaster process performs various safety checks if there is a postmaster.pid
|
||||
# file in the data directory. Although Patroni already detected that the running process
|
||||
# corresponding to the postmaster.pid is not a postmaster, the new postmaster might fail
|
||||
# to start, because it thinks that postmaster.pid is already locked.
|
||||
# Important!!! Unlink of postmaster.pid isn't an option, because it has a lot of nasty race conditions.
|
||||
# Luckily there is a workaround to this problem, we can pass the pid from postmaster.pid
|
||||
# in the `PG_GRANDPARENT_PID` environment variable and postmaster will ignore it.
|
||||
logger.info("Telling pg_ctl that it is safe to ignore postmaster.pid for process %s", proc.pid)
|
||||
env['PG_GRANDPARENT_PID'] = str(proc.pid)
|
||||
except psutil.NoSuchProcess:
|
||||
pass
|
||||
cmdline = [pgcommand, '-D', data_dir, '--config-file={}'.format(conf)] + options
|
||||
logger.debug("Starting postgres: %s", " ".join(cmdline))
|
||||
ctx = multiprocessing.get_context('spawn') if sys.version_info >= (3, 4) else multiprocessing
|
||||
parent_conn, child_conn = ctx.Pipe(False)
|
||||
proc = ctx.Process(target=pg_ctl_start, args=(child_conn, cmdline, env))
|
||||
proc.start()
|
||||
pid = parent_conn.recv()
|
||||
proc.join()
|
||||
if pid is None:
|
||||
return
|
||||
logger.info('postmaster pid=%s', pid)
|
||||
|
||||
# TODO: In an extremely unlikely case, the process could have exited and the pid reassigned. The start
|
||||
# initiation time is not accurate enough to compare to create time as start time would also likely
|
||||
# be relatively close. We need the subprocess extract pid+start_time in a race free manner.
|
||||
return PostmasterProcess.from_pid(pid)
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user