mirror of
https://github.com/outbackdingo/patroni.git
synced 2026-08-26 15:40:21 +00:00
Compare commits
141
Commits
ci/macos-ssl
...
master
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3a9befc1a2 | ||
|
|
8f22fd255e | ||
|
|
6938c21ff7 | ||
|
|
32934b205f | ||
|
|
1ba9e68b4b | ||
|
|
deb9cc6b73 | ||
|
|
a3c772dfc9 | ||
|
|
9977850b56 | ||
|
|
7543e64000 | ||
|
|
5ed4d33f7d | ||
|
|
c6943dc415 | ||
|
|
36011e936a | ||
|
|
a316105412 | ||
|
|
92c4f9fbb5 | ||
|
|
1c5d9f5653 | ||
|
|
66cf21767d | ||
|
|
b573bd4c9d | ||
|
|
33600976b1 | ||
|
|
e9ba775959 | ||
|
|
cf427e8b0b | ||
|
|
7531d41587 | ||
|
|
5dbfc9401b | ||
|
|
ce79152088 | ||
|
|
6920b3af0e | ||
|
|
0d87270897 | ||
|
|
1a31ea6e20 | ||
|
|
8de904e556 | ||
|
|
c97ad83396 | ||
|
|
0bb12473fb | ||
|
|
302757b71a | ||
|
|
3d932e1e73 | ||
|
|
38aef484e8 | ||
|
|
34b2a77294 | ||
|
|
6caa2fa99c | ||
|
|
b4eab48971 | ||
|
|
2bc25a32e4 | ||
|
|
7db7dfd3c5 | ||
|
|
3938bb9a16 | ||
|
|
26ae38960a | ||
|
|
836e527e6d | ||
|
|
e73f2044c8 | ||
|
|
39f5de2e77 | ||
|
|
46e20edbc2 | ||
|
|
578dc39291 | ||
|
|
9d1609e0eb | ||
|
|
fb0fcc859a | ||
|
|
a903438a5a | ||
|
|
19f75b407e | ||
|
|
3f00b7a6c7 | ||
|
|
4ce0f99cfb | ||
|
|
efba02f52e | ||
|
|
e1faa38e90 | ||
|
|
177101a1cc | ||
|
|
7dcb9b9840 | ||
|
|
e8a8bfe42f | ||
|
|
72be036c99 | ||
|
|
969d7ec4ab | ||
|
|
75ff8b3256 | ||
|
|
4853b3b430 | ||
|
|
ba970d8c63 | ||
|
|
ff278705d6 | ||
|
|
74c0acf36d | ||
|
|
58ee52b401 | ||
|
|
877acf2a55 | ||
|
|
8e46086335 | ||
|
|
e91e6b5484 | ||
|
|
6b685036d0 | ||
|
|
78a46b9ebc | ||
|
|
d7e172c20a | ||
|
|
87cb7481ae | ||
|
|
bfa9b0ca4b | ||
|
|
416a0f7c8b | ||
|
|
94a592d275 | ||
|
|
74a72e4f78 | ||
|
|
4c951a2937 | ||
|
|
b3ae8652c9 | ||
|
|
66f98c80e8 | ||
|
|
57ed40f66c | ||
|
|
d5d6a51e2c | ||
|
|
2f800173a5 | ||
|
|
db82a83eb4 | ||
|
|
3ecdf01b50 | ||
|
|
c9322df095 | ||
|
|
b470ade20e | ||
|
|
835d93951d | ||
|
|
8cdb0c25d9 | ||
|
|
6d65aa311a | ||
|
|
8c5ab4c07d | ||
|
|
31cf951b69 | ||
|
|
7659ccd50b | ||
|
|
a03dba04e3 | ||
|
|
93eb4edbe6 | ||
|
|
c931da1eb3 | ||
|
|
fc5a8ed01c | ||
|
|
0fa41502f1 | ||
|
|
384705ad97 | ||
|
|
56dba93c55 | ||
|
|
b458bd992a | ||
|
|
5eb431b719 | ||
|
|
ab9faf9471 | ||
|
|
cd3f52b029 | ||
|
|
4456e267eb | ||
|
|
c6339234c6 | ||
|
|
b1d442e7a4 | ||
|
|
b8b5518e8c | ||
|
|
a5796a03f1 | ||
|
|
fbbd32a537 | ||
|
|
c687838074 | ||
|
|
622d41c83c | ||
|
|
f00826d6e6 | ||
|
|
a4a6dc0299 | ||
|
|
3c07410695 | ||
|
|
7067d744c6 | ||
|
|
6b7ec49282 | ||
|
|
d4fd782038 | ||
|
|
6e1f9f7a6e | ||
|
|
a5d095e316 | ||
|
|
2a003a36bb | ||
|
|
af03c619ec | ||
|
|
14a44e14ba | ||
|
|
dc7ba3fe15 | ||
|
|
1ed207cbf0 | ||
|
|
b6c5a12017 | ||
|
|
1b7b8e60fb | ||
|
|
0a91948a49 | ||
|
|
0a6c09e252 | ||
|
|
ff31f45226 | ||
|
|
ff99d29e6d | ||
|
|
03bb9125cb | ||
|
|
634b44ee05 | ||
|
|
290d05c642 | ||
|
|
9d231aeecd | ||
|
|
48fbf64ea9 | ||
|
|
3fd7c98d2b | ||
|
|
d7454f7bcd | ||
|
|
ceb2965ab8 | ||
|
|
ae53260030 | ||
|
|
9b237b332e | ||
|
|
b09af642e6 | ||
|
|
014777b20a | ||
|
|
a8cfd46801 |
@@ -9,6 +9,11 @@ import zipfile
|
|||||||
|
|
||||||
|
|
||||||
def install_requirements(what):
|
def install_requirements(what):
|
||||||
|
subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'pip'])
|
||||||
|
s = subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'wheel', 'setuptools'])
|
||||||
|
if s != 0:
|
||||||
|
return s
|
||||||
|
|
||||||
old_path = sys.path[:]
|
old_path = sys.path[:]
|
||||||
w = os.path.join(os.getcwd(), os.path.dirname(inspect.getfile(inspect.currentframe())))
|
w = os.path.join(os.getcwd(), os.path.dirname(inspect.getfile(inspect.currentframe())))
|
||||||
sys.path.insert(0, os.path.dirname(os.path.dirname(w)))
|
sys.path.insert(0, os.path.dirname(os.path.dirname(w)))
|
||||||
@@ -16,23 +21,25 @@ def install_requirements(what):
|
|||||||
from setup import EXTRAS_REQUIRE, read
|
from setup import EXTRAS_REQUIRE, read
|
||||||
finally:
|
finally:
|
||||||
sys.path = old_path
|
sys.path = old_path
|
||||||
requirements = ['mock>=2.0.0', 'flake8', 'pytest', 'pytest-cov'] if what == 'all' else ['behave']
|
requirements = ['flake8', 'pytest', 'pytest-cov'] if what == 'all' else ['behave']
|
||||||
requirements += ['coverage']
|
requirements += ['coverage']
|
||||||
# try to split tests between psycopg2 and psycopg3
|
# try to split tests between psycopg2 and psycopg3
|
||||||
requirements += ['psycopg[binary]'] if sys.version_info >= (3, 8, 0) and\
|
requirements += ['psycopg[binary]'] if sys.version_info >= (3, 8, 0) and\
|
||||||
(sys.platform != 'darwin' or what == 'etcd3') else ['psycopg2-binary']
|
(sys.platform != 'darwin' or what == 'etcd3') else ['psycopg2-binary==2.9.9'
|
||||||
for r in read('requirements.txt').split('\n'):
|
if sys.platform == 'darwin' else 'psycopg2-binary']
|
||||||
r = r.strip()
|
|
||||||
if r != '':
|
|
||||||
extras = {e for e, v in EXTRAS_REQUIRE.items() if v and any(r.startswith(x) for x in v)}
|
|
||||||
if not extras or what == 'all' or what in extras:
|
|
||||||
requirements.append(r)
|
|
||||||
|
|
||||||
subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'pip'])
|
from pip._vendor.distlib.markers import evaluator, DEFAULT_CONTEXT
|
||||||
subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'wheel'])
|
from pip._vendor.distlib.util import parse_requirement
|
||||||
r = subprocess.call([sys.executable, '-m', 'pip', 'install'] + requirements)
|
|
||||||
s = subprocess.call([sys.executable, '-m', 'pip', 'install', '--upgrade', 'setuptools'])
|
for r in read('requirements.txt').split('\n'):
|
||||||
return s | r
|
r = parse_requirement(r)
|
||||||
|
if not r or r.marker and not evaluator.evaluate(r.marker, DEFAULT_CONTEXT):
|
||||||
|
continue
|
||||||
|
extras = {e for e, v in EXTRAS_REQUIRE.items() if v and any(r.requirement.startswith(x) for x in v)}
|
||||||
|
if not extras or what == 'all' or what in extras:
|
||||||
|
requirements.append(r.requirement)
|
||||||
|
|
||||||
|
return subprocess.call([sys.executable, '-m', 'pip', 'install'] + requirements)
|
||||||
|
|
||||||
|
|
||||||
def install_packages(what):
|
def install_packages(what):
|
||||||
@@ -45,8 +52,8 @@ def install_packages(what):
|
|||||||
packages['exhibitor'] = packages['zookeeper']
|
packages['exhibitor'] = packages['zookeeper']
|
||||||
packages = packages.get(what, [])
|
packages = packages.get(what, [])
|
||||||
ver = versions.get(what)
|
ver = versions.get(what)
|
||||||
if float(ver) >= 15:
|
if 15 <= float(ver) < 18:
|
||||||
packages += ['postgresql-{0}-citus-12.1'.format(ver)]
|
packages += ['postgresql-{0}-citus-13.0'.format(ver)]
|
||||||
subprocess.call(['sudo', 'apt-get', 'update', '-y'])
|
subprocess.call(['sudo', 'apt-get', 'update', '-y'])
|
||||||
return subprocess.call(['sudo', 'apt-get', 'install', '-y', 'postgresql-' + ver, 'expect-dev'] + packages)
|
return subprocess.call(['sudo', 'apt-get', 'install', '-y', 'postgresql-' + ver, 'expect-dev'] + packages)
|
||||||
|
|
||||||
|
|||||||
@@ -1 +1 @@
|
|||||||
versions = {'etcd': '9.6', 'etcd3': '16', 'consul': '13', 'exhibitor': '12', 'raft': '14', 'kubernetes': '15'}
|
versions = {'etcd': '9.6', 'etcd3': '16', 'consul': '17', 'exhibitor': '12', 'raft': '14', 'kubernetes': '15'}
|
||||||
|
|||||||
@@ -10,13 +10,15 @@ jobs:
|
|||||||
build-n-publish:
|
build-n-publish:
|
||||||
name: Build and publish Patroni distributions to PyPI and TestPyPI
|
name: Build and publish Patroni distributions to PyPI and TestPyPI
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
|
permissions:
|
||||||
|
id-token: write
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@master
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
- name: Set up Python 3.9
|
- name: Set up Python 3.11
|
||||||
uses: actions/setup-python@v4
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: 3.9
|
python-version: 3.11
|
||||||
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: python .github/workflows/install_deps.py
|
run: python .github/workflows/install_deps.py
|
||||||
@@ -32,13 +34,10 @@ jobs:
|
|||||||
|
|
||||||
- name: Publish distribution to Test PyPI
|
- name: Publish distribution to Test PyPI
|
||||||
if: github.event_name == 'push'
|
if: github.event_name == 'push'
|
||||||
uses: pypa/gh-action-pypi-publish@v1.5.1
|
uses: pypa/gh-action-pypi-publish@v1.9.0
|
||||||
with:
|
with:
|
||||||
password: ${{ secrets.TEST_PYPI_API_TOKEN }}
|
|
||||||
repository_url: https://test.pypi.org/legacy/
|
repository_url: https://test.pypi.org/legacy/
|
||||||
|
|
||||||
- name: Publish distribution to PyPI
|
- name: Publish distribution to PyPI
|
||||||
if: github.event_name == 'release'
|
if: github.event_name == 'release'
|
||||||
uses: pypa/gh-action-pypi-publish@v1.5.1
|
uses: pypa/gh-action-pypi-publish@v1.9.0
|
||||||
with:
|
|
||||||
password: ${{ secrets.PYPI_API_TOKEN }}
|
|
||||||
|
|||||||
@@ -27,11 +27,11 @@ def main():
|
|||||||
|
|
||||||
version = versions.get(what)
|
version = versions.get(what)
|
||||||
path = '/usr/lib/postgresql/{0}/bin:.'.format(version)
|
path = '/usr/lib/postgresql/{0}/bin:.'.format(version)
|
||||||
unbuffer = ['timeout', '900', 'unbuffer']
|
unbuffer = ['timeout', '1200', 'unbuffer']
|
||||||
else:
|
else:
|
||||||
if sys.platform == 'darwin':
|
if sys.platform == 'darwin':
|
||||||
version = os.environ.get('PGVERSION', '16.1-1')
|
version = os.environ.get('PGVERSION', '16.1-1')
|
||||||
path = '/usr/local/opt/postgresql@{0}/bin:.'.format(version.split('.')[0])
|
path = '/opt/homebrew/opt/postgresql@{0}/bin:.'.format(version.split('.')[0])
|
||||||
unbuffer = ['unbuffer']
|
unbuffer = ['unbuffer']
|
||||||
else:
|
else:
|
||||||
path = os.path.abspath(os.path.join('pgsql', 'bin'))
|
path = os.path.abspath(os.path.join('pgsql', 'bin'))
|
||||||
|
|||||||
@@ -13,26 +13,29 @@ env:
|
|||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
unit:
|
unit:
|
||||||
runs-on: ${{ matrix.os }}-latest
|
runs-on: ${{ fromJson('{"ubuntu":"ubuntu-22.04","windows":"windows-latest","macos":"macos-latest"}')[matrix.os] }}
|
||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
os: [ubuntu, windows, macos]
|
os: [ubuntu, windows, macos]
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
- name: Set up Python 3.7
|
- name: Set up Python 3.7
|
||||||
uses: actions/setup-python@v4
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: 3.7
|
python-version: 3.7
|
||||||
|
if: matrix.os != 'macos'
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: python .github/workflows/install_deps.py
|
run: python .github/workflows/install_deps.py
|
||||||
|
if: matrix.os != 'macos'
|
||||||
- name: Run tests and flake8
|
- name: Run tests and flake8
|
||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
if: matrix.os != 'macos'
|
||||||
|
|
||||||
- name: Set up Python 3.8
|
- name: Set up Python 3.8
|
||||||
uses: actions/setup-python@v4
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: 3.8
|
python-version: 3.8
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
@@ -41,7 +44,7 @@ jobs:
|
|||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
- name: Set up Python 3.9
|
- name: Set up Python 3.9
|
||||||
uses: actions/setup-python@v4
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: 3.9
|
python-version: 3.9
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
@@ -50,7 +53,7 @@ jobs:
|
|||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
- name: Set up Python 3.10
|
- name: Set up Python 3.10
|
||||||
uses: actions/setup-python@v4
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: '3.10'
|
python-version: '3.10'
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
@@ -59,7 +62,7 @@ jobs:
|
|||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
- name: Set up Python 3.11
|
- name: Set up Python 3.11
|
||||||
uses: actions/setup-python@v4
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: 3.11
|
python-version: 3.11
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
@@ -67,6 +70,27 @@ jobs:
|
|||||||
- name: Run tests and flake8
|
- name: Run tests and flake8
|
||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
|
- name: Set up Python 3.12
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: 3.12
|
||||||
|
- name: Install dependencies
|
||||||
|
run: python .github/workflows/install_deps.py
|
||||||
|
- name: Run tests and flake8
|
||||||
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
|
- name: Set up Python 3.13
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: 3.13
|
||||||
|
if: matrix.os != 'macos'
|
||||||
|
- name: Install dependencies
|
||||||
|
run: python .github/workflows/install_deps.py
|
||||||
|
if: matrix.os != 'macos'
|
||||||
|
- name: Run tests and flake8
|
||||||
|
run: python .github/workflows/run_tests.py
|
||||||
|
if: matrix.os != 'macos'
|
||||||
|
|
||||||
- name: Combine coverage
|
- name: Combine coverage
|
||||||
run: python .github/workflows/run_tests.py combine
|
run: python .github/workflows/run_tests.py combine
|
||||||
|
|
||||||
@@ -81,7 +105,7 @@ jobs:
|
|||||||
run: python -m coveralls --service=github
|
run: python -m coveralls --service=github
|
||||||
|
|
||||||
behave:
|
behave:
|
||||||
runs-on: ${{ matrix.os }}-latest
|
runs-on: ${{ fromJson('{"ubuntu":"ubuntu-22.04","windows":"windows-latest","macos":"macos-latest"}')[matrix.os] }}
|
||||||
env:
|
env:
|
||||||
DCS: ${{ matrix.dcs }}
|
DCS: ${{ matrix.dcs }}
|
||||||
ETCDVERSION: 3.4.23
|
ETCDVERSION: 3.4.23
|
||||||
@@ -90,7 +114,7 @@ jobs:
|
|||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
os: [ubuntu]
|
os: [ubuntu]
|
||||||
python-version: [3.7, '3.10']
|
python-version: [3.7, 3.13]
|
||||||
dcs: [etcd, etcd3, consul, exhibitor, kubernetes, raft]
|
dcs: [etcd, etcd3, consul, exhibitor, kubernetes, raft]
|
||||||
include:
|
include:
|
||||||
- os: macos
|
- os: macos
|
||||||
@@ -104,9 +128,9 @@ jobs:
|
|||||||
dcs: etcd3
|
dcs: etcd3
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v4
|
||||||
- name: Set up Python
|
- name: Set up Python
|
||||||
uses: actions/setup-python@v4
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python-version }}
|
python-version: ${{ matrix.python-version }}
|
||||||
- uses: nolar/setup-k3d-k3s@v1
|
- uses: nolar/setup-k3d-k3s@v1
|
||||||
@@ -125,12 +149,12 @@ jobs:
|
|||||||
- name: Run behave tests
|
- name: Run behave tests
|
||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
- name: Upload logs if behave failed
|
- name: Upload logs if behave failed
|
||||||
uses: actions/upload-artifact@v3
|
uses: actions/upload-artifact@v4
|
||||||
if: failure()
|
if: failure()
|
||||||
with:
|
with:
|
||||||
name: behave-${{ matrix.os }}-${{ matrix.dcs }}-${{ matrix.python-version }}-logs
|
name: behave-${{ matrix.os }}-${{ matrix.dcs }}-${{ matrix.python-version }}-logs
|
||||||
path: |
|
path: |
|
||||||
features/output/*_failed/*postgres?.*
|
features/output/*_failed/*postgres-?.*
|
||||||
features/output/*.log
|
features/output/*.log
|
||||||
if-no-files-found: error
|
if-no-files-found: error
|
||||||
retention-days: 5
|
retention-days: 5
|
||||||
@@ -145,7 +169,7 @@ jobs:
|
|||||||
needs: unit
|
needs: unit
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/setup-python@v4
|
- uses: actions/setup-python@v5
|
||||||
- run: python -m pip install coveralls
|
- run: python -m pip install coveralls
|
||||||
- run: python -m coveralls --service=github --finish
|
- run: python -m coveralls --service=github --finish
|
||||||
env:
|
env:
|
||||||
@@ -162,27 +186,44 @@ jobs:
|
|||||||
pyright:
|
pyright:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
- name: Set up Python 3.11
|
- name: Set up Python 3.13
|
||||||
uses: actions/setup-python@v4
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: 3.11
|
python-version: 3.13
|
||||||
|
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: python -m pip install -r requirements.txt psycopg2-binary psycopg
|
run: python -m pip install -r requirements.txt psycopg2-binary psycopg
|
||||||
|
|
||||||
- uses: jakebailey/pyright-action@v1
|
- uses: jakebailey/pyright-action@v2
|
||||||
with:
|
with:
|
||||||
version: 1.1.347
|
version: 1.1.394
|
||||||
|
|
||||||
|
ydiff:
|
||||||
|
name: Test compatibility with the latest version of ydiff
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up Python 3.13
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: 3.13
|
||||||
|
- name: Install dependencies
|
||||||
|
run: python .github/workflows/install_deps.py
|
||||||
|
- name: Update ydiff
|
||||||
|
run: python -m pip install -U ydiff
|
||||||
|
- name: Run tests
|
||||||
|
run: python -m pytest tests/test_ctl.py -v
|
||||||
|
|
||||||
docs:
|
docs:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
- name: Set up Python 3.11
|
- name: Set up Python 3.11
|
||||||
uses: actions/setup-python@v4
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: 3.11
|
python-version: 3.11
|
||||||
cache: pip
|
cache: pip
|
||||||
@@ -199,3 +240,20 @@ jobs:
|
|||||||
|
|
||||||
- name: Generate documentation
|
- name: Generate documentation
|
||||||
run: tox -m docs
|
run: tox -m docs
|
||||||
|
|
||||||
|
isort:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Set up Python 3.12
|
||||||
|
uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: 3.12
|
||||||
|
cache: pip
|
||||||
|
|
||||||
|
- name: isort
|
||||||
|
uses: isort/isort-action@master
|
||||||
|
with:
|
||||||
|
requirementsFiles: "requirements.txt requirements.dev.txt requirements.docs.txt"
|
||||||
|
sort-paths: "patroni tests features setup.py"
|
||||||
|
|||||||
+2
-2
@@ -7,7 +7,7 @@ ARG PGDATA=$PGHOME/data
|
|||||||
ARG LC_ALL=C.UTF-8
|
ARG LC_ALL=C.UTF-8
|
||||||
ARG LANG=C.UTF-8
|
ARG LANG=C.UTF-8
|
||||||
|
|
||||||
FROM postgres:$PG_MAJOR as builder
|
FROM postgres:$PG_MAJOR AS builder
|
||||||
|
|
||||||
ARG PGHOME
|
ARG PGHOME
|
||||||
ARG PGDATA
|
ARG PGDATA
|
||||||
@@ -180,7 +180,7 @@ RUN sed -i 's/env python/&3/' /patroni*.py \
|
|||||||
&& sed -i "s|^\( data_dir: \).*|\1$PGDATA|" postgres?.yml \
|
&& sed -i "s|^\( data_dir: \).*|\1$PGDATA|" postgres?.yml \
|
||||||
&& sed -i "s|^#\( bin_dir: \).*|\1$PGBIN|" postgres?.yml \
|
&& sed -i "s|^#\( bin_dir: \).*|\1$PGBIN|" postgres?.yml \
|
||||||
&& sed -i 's/^ - encoding: UTF8/ - locale: en_US.UTF-8\n&/' postgres?.yml \
|
&& sed -i 's/^ - encoding: UTF8/ - locale: en_US.UTF-8\n&/' postgres?.yml \
|
||||||
&& sed -i 's/^scope:/log:\n loggers:\n patroni.postgresql.citus: DEBUG\n#&/' postgres?.yml \
|
&& sed -i 's/^scope:/log:\n loggers:\n patroni.postgresql.mpp.citus: DEBUG\n#&/' postgres?.yml \
|
||||||
&& sed -i 's/^\(name\|etcd\| host\| authentication\| connect_address\| parameters\):/#&/' postgres?.yml \
|
&& sed -i 's/^\(name\|etcd\| host\| authentication\| connect_address\| parameters\):/#&/' postgres?.yml \
|
||||||
&& sed -i 's/^ \(replication\|superuser\|rewind\|unix_socket_directories\|\(\( \)\{0,1\}\(username\|password\)\)\):/#&/' postgres?.yml \
|
&& sed -i 's/^ \(replication\|superuser\|rewind\|unix_socket_directories\|\(\( \)\{0,1\}\(username\|password\)\)\):/#&/' postgres?.yml \
|
||||||
&& sed -i 's/^postgresql:/&\n basebackup:\n checkpoint: fast/' postgres?.yml \
|
&& sed -i 's/^postgresql:/&\n basebackup:\n checkpoint: fast/' postgres?.yml \
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
The MIT License (MIT)
|
The MIT License (MIT)
|
||||||
|
|
||||||
Copyright (c) 2015 Compose, Zalando SE
|
Copyright (c) 2025 Compose, Zalando SE, Patroni Contributors
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
of this software and associated documentation files (the "Software"), to deal
|
of this software and associated documentation files (the "Software"), to deal
|
||||||
|
|||||||
+18
-18
@@ -12,11 +12,11 @@ Patroni is a template for high availability (HA) PostgreSQL solutions using Pyth
|
|||||||
|
|
||||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely.
|
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely.
|
||||||
|
|
||||||
Currently supported PostgreSQL versions: 9.3 to 16.
|
Currently supported PostgreSQL versions: 9.3 to 17.
|
||||||
|
|
||||||
**Note to Citus users**: Starting from 3.0 Patroni nicely integrates with the `Citus <https://github.com/citusdata/citus>`__ database extension to Postgres. Please check the `Citus support page <https://github.com/zalando/patroni/blob/master/docs/citus.rst>`__ in the Patroni documentation for more info about how to use Patroni high availability together with a Citus distributed cluster.
|
**Note to Citus users**: Starting from 3.0 Patroni nicely integrates with the `Citus <https://github.com/citusdata/citus>`__ database extension to Postgres. Please check the `Citus support page <https://github.com/patroni/patroni/blob/master/docs/citus.rst>`__ in the Patroni documentation for more info about how to use Patroni high availability together with a Citus distributed cluster.
|
||||||
|
|
||||||
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the `Kubernetes <https://github.com/zalando/patroni/blob/master/docs/kubernetes.rst>`__ chapter of the Patroni documentation.
|
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the `Kubernetes <https://github.com/patroni/patroni/blob/master/docs/kubernetes.rst>`__ chapter of the Patroni documentation.
|
||||||
|
|
||||||
.. contents::
|
.. contents::
|
||||||
:local:
|
:local:
|
||||||
@@ -27,29 +27,27 @@ Currently supported PostgreSQL versions: 9.3 to 16.
|
|||||||
How Patroni Works
|
How Patroni Works
|
||||||
=================
|
=================
|
||||||
|
|
||||||
Patroni originated as a fork of `Governor <https://github.com/compose/governor>`__, the project from Compose. It includes plenty of new features.
|
Patroni (formerly known as Zalando's Patroni) originated as a fork of `Governor <https://github.com/compose/governor>`__, the project from Compose. It includes plenty of new features.
|
||||||
|
|
||||||
For an example of a Docker-based deployment with Patroni, see `Spilo <https://github.com/zalando/spilo>`__, currently in use at Zalando.
|
|
||||||
|
|
||||||
For additional background info, see:
|
For additional background info, see:
|
||||||
|
|
||||||
* `Elephants on Automatic: HA Clustered PostgreSQL with Helm <https://www.youtube.com/watch?v=CftcVhFMGSY>`_, talk by Josh Berkus and Oleksii Kliukin at KubeCon Berlin 2017
|
* `Elephants on Automatic: HA Clustered PostgreSQL with Helm <https://www.youtube.com/watch?v=CftcVhFMGSY>`_, talk by Josh Berkus and Oleksii Kliukin at KubeCon Berlin 2017
|
||||||
* `PostgreSQL HA with Kubernetes and Patroni <https://www.youtube.com/watch?v=iruaCgeG7qs>`__, talk by Josh Berkus at KubeCon 2016 (video)
|
* `PostgreSQL HA with Kubernetes and Patroni <https://www.youtube.com/watch?v=iruaCgeG7qs>`__, talk by Josh Berkus at KubeCon 2016 (video)
|
||||||
* `Feb. 2016 Zalando Tech blog post <https://tech.zalando.de/blog/zalandos-patroni-a-template-for-high-availability-postgresql/>`__
|
* `Feb. 2016 Zalando Tech blog post <https://engineering.zalando.com/posts/2016/02/zalandos-patroni-a-template-for-high-availability-postgresql.html>`__
|
||||||
|
|
||||||
==================
|
==================
|
||||||
Development Status
|
Development Status
|
||||||
==================
|
==================
|
||||||
|
|
||||||
Patroni is in active development and accepts contributions. See our `Contributing <https://github.com/zalando/patroni/blob/master/docs/CONTRIBUTING.rst>`__ section below for more details.
|
Patroni is in active development and accepts contributions. See our `Contributing <https://github.com/patroni/patroni/blob/master/docs/contributing_guidelines.rst>`__ section below for more details.
|
||||||
|
|
||||||
We report new releases information `here <https://github.com/zalando/patroni/releases>`__.
|
We report new releases information `here <https://github.com/patroni/patroni/releases>`__.
|
||||||
|
|
||||||
=========
|
=========
|
||||||
Community
|
Community
|
||||||
=========
|
=========
|
||||||
|
|
||||||
There are two places to connect with the Patroni community: `on github <https://github.com/zalando/patroni>`__, via Issues and PRs, and on channel `#patroni <https://postgresteam.slack.com/archives/C9XPYG92A>`__ in the `PostgreSQL Slack <https://pgtreats.info/slack-invite>`__. If you're using Patroni, or just interested, please join us.
|
There are two places to connect with the Patroni community: `on github <https://github.com/patroni/patroni>`__, via Issues and PRs, and on channel `#patroni <https://postgresteam.slack.com/archives/C9XPYG92A>`__ in the `PostgreSQL Slack <https://pgtreats.info/slack-invite>`__. If you're using Patroni, or just interested, please join us.
|
||||||
|
|
||||||
===================================
|
===================================
|
||||||
Technical Requirements/Installation
|
Technical Requirements/Installation
|
||||||
@@ -93,7 +91,7 @@ where dependencies can be either empty, or consist of one or more of the followi
|
|||||||
etcd or etcd3
|
etcd or etcd3
|
||||||
`python-etcd` module in order to use Etcd as DCS
|
`python-etcd` module in order to use Etcd as DCS
|
||||||
consul
|
consul
|
||||||
`python-consul` module in order to use Consul as DCS
|
`py-consul` module in order to use Consul as DCS
|
||||||
zookeeper
|
zookeeper
|
||||||
`kazoo` module in order to use Zookeeper as DCS
|
`kazoo` module in order to use Zookeeper as DCS
|
||||||
exhibitor
|
exhibitor
|
||||||
@@ -104,6 +102,8 @@ raft
|
|||||||
`pysyncobj` module in order to use python Raft implementation as DCS
|
`pysyncobj` module in order to use python Raft implementation as DCS
|
||||||
aws
|
aws
|
||||||
`boto3` in order to use AWS callbacks
|
`boto3` in order to use AWS callbacks
|
||||||
|
systemd
|
||||||
|
`systemd-python` in order to use sd_notify integration
|
||||||
all
|
all
|
||||||
all of the above (except psycopg family)
|
all of the above (except psycopg family)
|
||||||
psycopg3
|
psycopg3
|
||||||
@@ -151,19 +151,19 @@ run:
|
|||||||
YAML Configuration
|
YAML Configuration
|
||||||
==================
|
==================
|
||||||
|
|
||||||
Go `here <https://github.com/zalando/patroni/blob/master/docs/dynamic_configuration.rst>`__ for comprehensive information about settings for etcd, consul, and ZooKeeper. And for an example, see `postgres0.yml <https://github.com/zalando/patroni/blob/master/postgres0.yml>`__.
|
Go `here <https://github.com/patroni/patroni/blob/master/docs/dynamic_configuration.rst>`__ for comprehensive information about settings for etcd, consul, and ZooKeeper. And for an example, see `postgres0.yml <https://github.com/patroni/patroni/blob/master/postgres0.yml>`__.
|
||||||
|
|
||||||
=========================
|
=========================
|
||||||
Environment Configuration
|
Environment Configuration
|
||||||
=========================
|
=========================
|
||||||
|
|
||||||
Go `here <https://github.com/zalando/patroni/blob/master/docs/ENVIRONMENT.rst>`__ for comprehensive information about configuring(overriding) settings via environment variables.
|
Go `here <https://github.com/patroni/patroni/blob/master/docs/ENVIRONMENT.rst>`__ for comprehensive information about configuring(overriding) settings via environment variables.
|
||||||
|
|
||||||
===================
|
===================
|
||||||
Replication Choices
|
Replication Choices
|
||||||
===================
|
===================
|
||||||
|
|
||||||
Patroni uses Postgres' streaming replication, which is asynchronous by default. Patroni's asynchronous replication configuration allows for ``maximum_lag_on_failover`` settings. This setting ensures failover will not occur if a follower is more than a certain number of bytes behind the leader. This setting should be increased or decreased based on business requirements. It's also possible to use synchronous replication for better durability guarantees. See `replication modes documentation <https://github.com/zalando/patroni/blob/master/docs/replication_modes.rst>`__ for details.
|
Patroni uses Postgres' streaming replication, which is asynchronous by default. Patroni's asynchronous replication configuration allows for ``maximum_lag_on_failover`` settings. This setting ensures failover will not occur if a follower is more than a certain number of bytes behind the leader. This setting should be increased or decreased based on business requirements. It's also possible to use synchronous replication for better durability guarantees. See `replication modes documentation <https://github.com/patroni/patroni/blob/master/docs/replication_modes.rst>`__ for details.
|
||||||
|
|
||||||
======================================
|
======================================
|
||||||
Applications Should Not Use Superusers
|
Applications Should Not Use Superusers
|
||||||
@@ -171,7 +171,7 @@ Applications Should Not Use Superusers
|
|||||||
|
|
||||||
When connecting from an application, always use a non-superuser. Patroni requires access to the database to function properly. By using a superuser from an application, you can potentially use the entire connection pool, including the connections reserved for superusers, with the ``superuser_reserved_connections`` setting. If Patroni cannot access the Primary because the connection pool is full, behavior will be undesirable.
|
When connecting from an application, always use a non-superuser. Patroni requires access to the database to function properly. By using a superuser from an application, you can potentially use the entire connection pool, including the connections reserved for superusers, with the ``superuser_reserved_connections`` setting. If Patroni cannot access the Primary because the connection pool is full, behavior will be undesirable.
|
||||||
|
|
||||||
.. |Tests Status| image:: https://github.com/zalando/patroni/actions/workflows/tests.yaml/badge.svg
|
.. |Tests Status| image:: https://github.com/patroni/patroni/actions/workflows/tests.yaml/badge.svg
|
||||||
:target: https://github.com/zalando/patroni/actions/workflows/tests.yaml?query=branch%3Amaster
|
:target: https://github.com/patroni/patroni/actions/workflows/tests.yaml?query=branch%3Amaster
|
||||||
.. |Coverage Status| image:: https://coveralls.io/repos/zalando/patroni/badge.svg?branch=master
|
.. |Coverage Status| image:: https://coveralls.io/repos/patroni/patroni/badge.svg?branch=master
|
||||||
:target: https://coveralls.io/github/zalando/patroni?branch=master
|
:target: https://coveralls.io/github/patroni/patroni?branch=master
|
||||||
|
|||||||
+33
-10
@@ -9,15 +9,18 @@
|
|||||||
# $ docker-compose -f docker-compose-citus.yml up -d
|
# $ docker-compose -f docker-compose-citus.yml up -d
|
||||||
# You can read more about it in the:
|
# You can read more about it in the:
|
||||||
# https://github.com/zalando/patroni/blob/master/docker/README.md#citus-cluster
|
# https://github.com/zalando/patroni/blob/master/docker/README.md#citus-cluster
|
||||||
version: "2"
|
version: "3"
|
||||||
|
|
||||||
networks:
|
networks:
|
||||||
demo:
|
demo:
|
||||||
|
|
||||||
services:
|
services:
|
||||||
etcd1: &etcd
|
etcd1: &etcd
|
||||||
image: ${PATRONI_TEST_IMAGE:-patroni-citus}
|
image: harbor.optimcloud.com/library/optim/patroni-citus:latest
|
||||||
networks: [ demo ]
|
networks: [ demo ]
|
||||||
|
ports:
|
||||||
|
- 2379
|
||||||
|
- 2380
|
||||||
environment:
|
environment:
|
||||||
ETCD_LISTEN_PEER_URLS: http://0.0.0.0:2380
|
ETCD_LISTEN_PEER_URLS: http://0.0.0.0:2380
|
||||||
ETCD_LISTEN_CLIENT_URLS: http://0.0.0.0:2379
|
ETCD_LISTEN_CLIENT_URLS: http://0.0.0.0:2379
|
||||||
@@ -32,17 +35,23 @@ services:
|
|||||||
etcd2:
|
etcd2:
|
||||||
<<: *etcd
|
<<: *etcd
|
||||||
container_name: demo-etcd2
|
container_name: demo-etcd2
|
||||||
|
ports:
|
||||||
|
- 2379
|
||||||
|
- 2380
|
||||||
hostname: etcd2
|
hostname: etcd2
|
||||||
command: etcd --name etcd2 --initial-advertise-peer-urls http://etcd2:2380
|
command: etcd --name etcd2 --initial-advertise-peer-urls http://etcd2:2380
|
||||||
|
|
||||||
etcd3:
|
etcd3:
|
||||||
<<: *etcd
|
<<: *etcd
|
||||||
container_name: demo-etcd3
|
container_name: demo-etcd3
|
||||||
|
ports:
|
||||||
|
- 2379
|
||||||
|
- 2380
|
||||||
hostname: etcd3
|
hostname: etcd3
|
||||||
command: etcd --name etcd3 --initial-advertise-peer-urls http://etcd3:2380
|
command: etcd --name etcd3 --initial-advertise-peer-urls http://etcd3:2380
|
||||||
|
|
||||||
haproxy:
|
haproxy:
|
||||||
image: ${PATRONI_TEST_IMAGE:-patroni-citus}
|
image: harbor.optimcloud.com/library/optim/patroni-citus:latest
|
||||||
networks: [ demo ]
|
networks: [ demo ]
|
||||||
env_file: docker/patroni.env
|
env_file: docker/patroni.env
|
||||||
hostname: haproxy
|
hostname: haproxy
|
||||||
@@ -63,8 +72,10 @@ services:
|
|||||||
PGSSLROOTCERT: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
PGSSLROOTCERT: /etc/ssl/certs/ssl-cert-snakeoil.pem
|
||||||
|
|
||||||
coord1:
|
coord1:
|
||||||
image: ${PATRONI_TEST_IMAGE:-patroni-citus}
|
image: harbor.optimcloud.com/library/optim/patroni-citus:latest
|
||||||
networks: [ demo ]
|
networks: [ demo ]
|
||||||
|
ports:
|
||||||
|
- "5432"
|
||||||
env_file: docker/patroni.env
|
env_file: docker/patroni.env
|
||||||
hostname: coord1
|
hostname: coord1
|
||||||
container_name: demo-coord1
|
container_name: demo-coord1
|
||||||
@@ -74,8 +85,10 @@ services:
|
|||||||
PATRONI_CITUS_GROUP: 0
|
PATRONI_CITUS_GROUP: 0
|
||||||
|
|
||||||
coord2:
|
coord2:
|
||||||
image: ${PATRONI_TEST_IMAGE:-patroni-citus}
|
image: harbor.optimcloud.com/library/optim/patroni-citus:latest
|
||||||
networks: [ demo ]
|
networks: [ demo ]
|
||||||
|
ports:
|
||||||
|
- "5432"
|
||||||
env_file: docker/patroni.env
|
env_file: docker/patroni.env
|
||||||
hostname: coord2
|
hostname: coord2
|
||||||
container_name: demo-coord2
|
container_name: demo-coord2
|
||||||
@@ -84,8 +97,10 @@ services:
|
|||||||
PATRONI_NAME: coord2
|
PATRONI_NAME: coord2
|
||||||
|
|
||||||
coord3:
|
coord3:
|
||||||
image: ${PATRONI_TEST_IMAGE:-patroni-citus}
|
image: harbor.optimcloud.com/library/optim/patroni-citus:latest
|
||||||
networks: [ demo ]
|
networks: [ demo ]
|
||||||
|
ports:
|
||||||
|
- "5432"
|
||||||
env_file: docker/patroni.env
|
env_file: docker/patroni.env
|
||||||
hostname: coord3
|
hostname: coord3
|
||||||
container_name: demo-coord3
|
container_name: demo-coord3
|
||||||
@@ -95,8 +110,10 @@ services:
|
|||||||
|
|
||||||
|
|
||||||
work1-1:
|
work1-1:
|
||||||
image: ${PATRONI_TEST_IMAGE:-patroni-citus}
|
image: harbor.optimcloud.com/library/optim/patroni-citus:latest
|
||||||
networks: [ demo ]
|
networks: [ demo ]
|
||||||
|
ports:
|
||||||
|
- "5432"
|
||||||
env_file: docker/patroni.env
|
env_file: docker/patroni.env
|
||||||
hostname: work1-1
|
hostname: work1-1
|
||||||
container_name: demo-work1-1
|
container_name: demo-work1-1
|
||||||
@@ -106,8 +123,10 @@ services:
|
|||||||
PATRONI_CITUS_GROUP: 1
|
PATRONI_CITUS_GROUP: 1
|
||||||
|
|
||||||
work1-2:
|
work1-2:
|
||||||
image: ${PATRONI_TEST_IMAGE:-patroni-citus}
|
image: harbor.optimcloud.com/library/optim/patroni-citus:latest
|
||||||
networks: [ demo ]
|
networks: [ demo ]
|
||||||
|
ports:
|
||||||
|
- "5432"
|
||||||
env_file: docker/patroni.env
|
env_file: docker/patroni.env
|
||||||
hostname: work1-2
|
hostname: work1-2
|
||||||
container_name: demo-work1-2
|
container_name: demo-work1-2
|
||||||
@@ -117,8 +136,10 @@ services:
|
|||||||
|
|
||||||
|
|
||||||
work2-1:
|
work2-1:
|
||||||
image: ${PATRONI_TEST_IMAGE:-patroni-citus}
|
image: harbor.optimcloud.com/library/optim/patroni-citus:latest
|
||||||
networks: [ demo ]
|
networks: [ demo ]
|
||||||
|
ports:
|
||||||
|
- "5432"
|
||||||
env_file: docker/patroni.env
|
env_file: docker/patroni.env
|
||||||
hostname: work2-1
|
hostname: work2-1
|
||||||
container_name: demo-work2-1
|
container_name: demo-work2-1
|
||||||
@@ -128,8 +149,10 @@ services:
|
|||||||
PATRONI_CITUS_GROUP: 2
|
PATRONI_CITUS_GROUP: 2
|
||||||
|
|
||||||
work2-2:
|
work2-2:
|
||||||
image: ${PATRONI_TEST_IMAGE:-patroni-citus}
|
image: harbor.optimcloud.com/library/optim/patroni-citus:latest
|
||||||
networks: [ demo ]
|
networks: [ demo ]
|
||||||
|
ports:
|
||||||
|
- "5432"
|
||||||
env_file: docker/patroni.env
|
env_file: docker/patroni.env
|
||||||
hostname: work2-2
|
hostname: work2-2
|
||||||
container_name: demo-work2-2
|
container_name: demo-work2-2
|
||||||
|
|||||||
+1
-1
@@ -6,7 +6,7 @@
|
|||||||
# The cluster could be started as:
|
# The cluster could be started as:
|
||||||
# $ docker-compose up -d
|
# $ docker-compose up -d
|
||||||
# You can read more about it in the:
|
# You can read more about it in the:
|
||||||
# https://github.com/zalando/patroni/blob/master/docker/README.md
|
# https://github.com/patroni/patroni/blob/master/docker/README.md
|
||||||
version: "2"
|
version: "2"
|
||||||
|
|
||||||
networks:
|
networks:
|
||||||
|
|||||||
+129
-107
@@ -40,27 +40,26 @@ Example session:
|
|||||||
aef8bf3ee91f patroni "/bin/sh /entrypoint…" 15 minutes ago Up 15 minutes demo-etcd1
|
aef8bf3ee91f patroni "/bin/sh /entrypoint…" 15 minutes ago Up 15 minutes demo-etcd1
|
||||||
|
|
||||||
$ docker logs demo-patroni1
|
$ docker logs demo-patroni1
|
||||||
2023-11-21 09:04:33,547 INFO: Selected new etcd server http://172.29.0.3:2379
|
2024-08-26 09:04:33,547 INFO: Selected new etcd server http://172.29.0.3:2379
|
||||||
2023-11-21 09:04:33,605 INFO: Lock owner: None; I am patroni1
|
2024-08-26 09:04:33,605 INFO: Lock owner: None; I am patroni1
|
||||||
2023-11-21 09:04:33,693 INFO: trying to bootstrap a new cluster
|
2024-08-26 09:04:33,693 INFO: trying to bootstrap a new cluster
|
||||||
...
|
...
|
||||||
2023-11-21 09:04:34.920 UTC [43] LOG: starting PostgreSQL 15.5 (Debian 15.5-1.pgdg120+1) on x86_64-pc-linux-gnu, compiled by gcc (Debian 12.2.0-14) 12.2.0, 64-bit
|
2024-08-26 09:04:34.920 UTC [43] LOG: starting PostgreSQL 16.4 (Debian 16.4-1.pgdg120+1) on x86_64-pc-linux-gnu, compiled by gcc (Debian 12.2.0-14) 12.2.0, 64-bit
|
||||||
2023-11-21 09:04:34.921 UTC [43] LOG: listening on IPv4 address "0.0.0.0", port 5432
|
2024-08-26 09:04:34.921 UTC [43] LOG: listening on IPv4 address "0.0.0.0", port 5432
|
||||||
2023-11-21 09:04:34,922 INFO: postmaster pid=43
|
2024-08-26 09:04:34,922 INFO: postmaster pid=43
|
||||||
2023-11-21 09:04:34.922 UTC [43] LOG: listening on Unix socket "/var/run/postgresql/.s.PGSQL.5432"
|
2024-08-26 09:04:34.922 UTC [43] LOG: listening on Unix socket "/var/run/postgresql/.s.PGSQL.5432"
|
||||||
2023-11-21 09:04:34.925 UTC [47] LOG: database system was shut down at 2023-11-21 09:04:34 UTC
|
2024-08-26 09:04:34.925 UTC [47] LOG: database system was shut down at 2024-08-26 09:04:34 UTC
|
||||||
2023-11-21 09:04:34.928 UTC [43] LOG: database system is ready to accept connections
|
2024-08-26 09:04:34.928 UTC [43] LOG: database system is ready to accept connections
|
||||||
localhost:5432 - accepting connections
|
localhost:5432 - accepting connections
|
||||||
localhost:5432 - accepting connections
|
localhost:5432 - accepting connections
|
||||||
2023-11-21 09:04:34,938 INFO: establishing a new patroni heartbeat connection to postgres
|
2024-08-26 09:04:34,938 INFO: establishing a new patroni heartbeat connection to postgres
|
||||||
2023-11-21 09:04:34,992 INFO: running post_bootstrap
|
2024-08-26 09:04:34,992 INFO: running post_bootstrap
|
||||||
2023-11-21 09:04:35,004 WARNING: User creation via "bootstrap.users" will be removed in v4.0.0
|
2024-08-26 09:04:35,004 WARNING: User creation via "bootstrap.users" will be removed in v4.0.0
|
||||||
2023-11-21 09:04:35,009 WARNING: Could not activate Linux watchdog device: Can't open watchdog device: [Errno 2] No such file or directory: '/dev/watchdog'
|
2024-08-26 09:04:35,189 INFO: initialized a new cluster
|
||||||
2023-11-21 09:04:35,189 INFO: initialized a new cluster
|
2024-08-26 09:04:35,328 INFO: no action. I am (patroni1), the leader with the lock
|
||||||
2023-11-21 09:04:35,328 INFO: no action. I am (patroni1), the leader with the lock
|
2024-08-26 09:04:43,824 INFO: establishing a new patroni restapi connection to postgres
|
||||||
2023-11-21 09:04:43,824 INFO: establishing a new patroni restapi connection to postgres
|
2024-08-26 09:04:45,322 INFO: no action. I am (patroni1), the leader with the lock
|
||||||
2023-11-21 09:04:45,322 INFO: no action. I am (patroni1), the leader with the lock
|
2024-08-26 09:04:55,320 INFO: no action. I am (patroni1), the leader with the lock
|
||||||
2023-11-21 09:04:55,320 INFO: no action. I am (patroni1), the leader with the lock
|
|
||||||
...
|
...
|
||||||
|
|
||||||
$ docker exec -ti demo-patroni1 bash
|
$ docker exec -ti demo-patroni1 bash
|
||||||
@@ -93,7 +92,7 @@ Example session:
|
|||||||
$ docker exec -ti demo-haproxy bash
|
$ docker exec -ti demo-haproxy bash
|
||||||
postgres@haproxy:~$ psql -h localhost -p 5000 -U postgres -W
|
postgres@haproxy:~$ psql -h localhost -p 5000 -U postgres -W
|
||||||
Password: postgres
|
Password: postgres
|
||||||
psql (15.5 (Debian 15.5-1.pgdg120+1))
|
psql (16.4 (Debian 16.4-1.pgdg120+1))
|
||||||
Type "help" for help.
|
Type "help" for help.
|
||||||
|
|
||||||
postgres=# SELECT pg_is_in_recovery();
|
postgres=# SELECT pg_is_in_recovery();
|
||||||
@@ -106,7 +105,7 @@ Example session:
|
|||||||
|
|
||||||
postgres@haproxy:~$ psql -h localhost -p 5001 -U postgres -W
|
postgres@haproxy:~$ psql -h localhost -p 5001 -U postgres -W
|
||||||
Password: postgres
|
Password: postgres
|
||||||
psql (15.5 (Debian 15.5-1.pgdg120+1))
|
psql (16.4 (Debian 16.4-1.pgdg120+1))
|
||||||
Type "help" for help.
|
Type "help" for help.
|
||||||
|
|
||||||
postgres=# SELECT pg_is_in_recovery();
|
postgres=# SELECT pg_is_in_recovery();
|
||||||
@@ -122,7 +121,7 @@ The haproxy listens on ports 5000 (connects to the coordinator primary) and 5001
|
|||||||
|
|
||||||
Example session:
|
Example session:
|
||||||
|
|
||||||
$ docker compose -f docker-compose-citus.yml up -d
|
$ docker-compose -f docker-compose-citus.yml up -d
|
||||||
✔ Network patroni_demo Created
|
✔ Network patroni_demo Created
|
||||||
✔ Container demo-coord2 Started
|
✔ Container demo-coord2 Started
|
||||||
✔ Container demo-work2-2 Started
|
✔ Container demo-work2-2 Started
|
||||||
@@ -153,48 +152,62 @@ Example session:
|
|||||||
|
|
||||||
|
|
||||||
$ docker logs demo-coord1
|
$ docker logs demo-coord1
|
||||||
2023-11-21 09:36:14,293 INFO: Selected new etcd server http://172.30.0.4:2379
|
2024-08-26 08:21:05,323 INFO: Selected new etcd server http://172.19.0.5:2379
|
||||||
2023-11-21 09:36:14,390 INFO: Lock owner: None; I am coord1
|
2024-08-26 08:21:05,339 INFO: No PostgreSQL configuration items changed, nothing to reload.
|
||||||
2023-11-21 09:36:14,478 INFO: trying to bootstrap a new cluster
|
2024-08-26 08:21:05,388 INFO: Lock owner: None; I am coord1
|
||||||
|
2024-08-26 08:21:05,480 INFO: trying to bootstrap a new cluster
|
||||||
...
|
...
|
||||||
2023-11-21 09:36:16,475 INFO: postmaster pid=52
|
2024-08-26 08:21:17,115 INFO: postmaster pid=35
|
||||||
localhost:5432 - no response
|
localhost:5432 - no response
|
||||||
2023-11-21 09:36:16.495 UTC [52] LOG: starting PostgreSQL 15.5 (Debian 15.5-1.pgdg120+1) on x86_64-pc-linux-gnu, compiled by gcc (Debian 12.2.0-14) 12.2.0, 64-bit
|
2024-08-26 08:21:17.127 UTC [35] LOG: starting PostgreSQL 16.4 (Debian 16.4-1.pgdg120+1) on x86_64-pc-linux-gnu, compiled by gcc (Debian 12.2.0-14) 12.2.0, 64-bit
|
||||||
2023-11-21 09:36:16.495 UTC [52] LOG: listening on IPv4 address "0.0.0.0", port 5432
|
2024-08-26 08:21:17.127 UTC [35] LOG: listening on IPv4 address "0.0.0.0", port 5432
|
||||||
2023-11-21 09:36:16.496 UTC [52] LOG: listening on Unix socket "/var/run/postgresql/.s.PGSQL.5432"
|
2024-08-26 08:21:17.141 UTC [35] LOG: listening on Unix socket "/var/run/postgresql/.s.PGSQL.5432"
|
||||||
2023-11-21 09:36:16.498 UTC [56] LOG: database system was shut down at 2023-11-21 09:36:15 UTC
|
2024-08-26 08:21:17.155 UTC [39] LOG: database system was shut down at 2024-08-26 08:21:05 UTC
|
||||||
2023-11-21 09:36:16.501 UTC [52] LOG: database system is ready to accept connections
|
2024-08-26 08:21:17.182 UTC [35] LOG: database system is ready to accept connections
|
||||||
|
2024-08-26 08:21:17,683 INFO: establishing a new patroni heartbeat connection to postgres
|
||||||
|
2024-08-26 08:21:17,704 INFO: establishing a new patroni restapi connection to postgres
|
||||||
localhost:5432 - accepting connections
|
localhost:5432 - accepting connections
|
||||||
localhost:5432 - accepting connections
|
localhost:5432 - accepting connections
|
||||||
2023-11-21 09:36:17,509 INFO: establishing a new patroni heartbeat connection to postgres
|
2024-08-26 08:21:18,202 INFO: running post_bootstrap
|
||||||
2023-11-21 09:36:17,569 INFO: running post_bootstrap
|
2024-08-26 08:21:19.048 UTC [53] LOG: starting maintenance daemon on database 16385 user 10
|
||||||
2023-11-21 09:36:17,593 WARNING: User creation via "bootstrap.users" will be removed in v4.0.0
|
2024-08-26 08:21:19.048 UTC [53] CONTEXT: Citus maintenance daemon for database 16385 user 10
|
||||||
2023-11-21 09:36:17,783 INFO: establishing a new patroni restapi connection to postgres
|
2024-08-26 08:21:19,058 DEBUG: Could not activate Linux watchdog device: Can't open watchdog device: [Errno 2] No such file or directory: '/dev/watchdog'
|
||||||
2023-11-21 09:36:17,969 WARNING: Could not activate Linux watchdog device: Can't open watchdog device: [Errno 2] No such file or directory: '/dev/watchdog'
|
2024-08-26 08:21:19.250 UTC [37] LOG: checkpoint starting: immediate force wait
|
||||||
2023-11-21 09:36:17.969 UTC [70] LOG: starting maintenance daemon on database 16386 user 10
|
2024-08-26 08:21:19,275 INFO: initialized a new cluster
|
||||||
2023-11-21 09:36:17.969 UTC [70] CONTEXT: Citus maintenance daemon for database 16386 user 10
|
2024-08-26 08:21:22.946 UTC [37] LOG: checkpoint starting: immediate force wait
|
||||||
2023-11-21 09:36:18.159 UTC [54] LOG: checkpoint starting: immediate force wait
|
2024-08-26 08:21:29,059 INFO: Lock owner: coord1; I am coord1
|
||||||
2023-11-21 09:36:18,162 INFO: initialized a new cluster
|
2024-08-26 08:21:29,205 INFO: Enabled synchronous replication
|
||||||
2023-11-21 09:36:18,164 INFO: Lock owner: coord1; I am coord1
|
2024-08-26 08:21:29,206 DEBUG: query(SELECT groupid, nodename, nodeport, noderole, nodeid FROM pg_catalog.pg_dist_node, ())
|
||||||
2023-11-21 09:36:18,297 INFO: Enabled synchronous replication
|
2024-08-26 08:21:29,206 INFO: establishing a new patroni citus connection to postgres
|
||||||
2023-11-21 09:36:18,298 DEBUG: Adding the new task: PgDistNode(nodeid=None,group=0,host=172.30.0.3,port=5432,event=after_promote)
|
2024-08-26 08:21:29,206 DEBUG: Adding the new task: PgDistTask({PgDistNode(nodeid=None,host=172.19.0.8,port=5432,role=primary)})
|
||||||
2023-11-21 09:36:18,298 DEBUG: Adding the new task: PgDistNode(nodeid=None,group=1,host=172.30.0.7,port=5432,event=after_promote)
|
2024-08-26 08:21:29,206 DEBUG: Adding the new task: PgDistTask({PgDistNode(nodeid=None,host=172.19.0.2,port=5432,role=primary)})
|
||||||
2023-11-21 09:36:18,298 DEBUG: Adding the new task: PgDistNode(nodeid=None,group=2,host=172.30.0.8,port=5432,event=after_promote)
|
2024-08-26 08:21:29,206 DEBUG: Adding the new task: PgDistTask({PgDistNode(nodeid=None,host=172.19.0.9,port=5432,role=primary)})
|
||||||
2023-11-21 09:36:18,299 DEBUG: query(SELECT nodeid, groupid, nodename, nodeport, noderole FROM pg_catalog.pg_dist_node WHERE noderole = 'primary', ())
|
2024-08-26 08:21:29,219 DEBUG: query(SELECT pg_catalog.citus_add_node(%s, %s, %s, %s, 'default'), ('172.19.0.2', 5432, 1, 'primary'))
|
||||||
2023-11-21 09:36:18,299 INFO: establishing a new patroni citus connection to postgres
|
2024-08-26 08:21:29,256 DEBUG: query(SELECT pg_catalog.citus_add_node(%s, %s, %s, %s, 'default'), ('172.19.0.9', 5432, 2, 'primary'))
|
||||||
2023-11-21 09:36:18,323 DEBUG: query(SELECT pg_catalog.citus_add_node(%s, %s, %s, 'primary', 'default'), ('172.30.0.7', 5432, 1))
|
2024-08-26 08:21:29,474 INFO: no action. I am (coord1), the leader with the lock
|
||||||
2023-11-21 09:36:18,361 INFO: no action. I am (coord1), the leader with the lock
|
2024-08-26 08:21:39,060 INFO: Lock owner: coord1; I am coord1
|
||||||
2023-11-21 09:36:18,393 DEBUG: query(SELECT pg_catalog.citus_add_node(%s, %s, %s, 'primary', 'default'), ('172.30.0.8', 5432, 2))
|
2024-08-26 08:21:39,159 DEBUG: Adding the new task: PgDistTask({PgDistNode(nodeid=None,host=172.19.0.8,port=5432,role=primary), PgDistNode(nodeid=None,host=172.19.0.11,port=5432,role=secondary), PgDistNode(nodeid=None,host=172.19.0.7,port=5432,role=secondary)})
|
||||||
2023-11-21 09:36:28,164 INFO: Lock owner: coord1; I am coord1
|
2024-08-26 08:21:39,159 DEBUG: Adding the new task: PgDistTask({PgDistNode(nodeid=None,host=172.19.0.2,port=5432,role=primary), PgDistNode(nodeid=None,host=172.19.0.12,port=5432,role=secondary)})
|
||||||
2023-11-21 09:36:28,251 INFO: Assigning synchronous standby status to ['coord3']
|
2024-08-26 08:21:39,159 DEBUG: Adding the new task: PgDistTask({PgDistNode(nodeid=None,host=172.19.0.6,port=5432,role=secondary), PgDistNode(nodeid=None,host=172.19.0.9,port=5432,role=primary)})
|
||||||
|
2024-08-26 08:21:39,160 DEBUG: query(BEGIN, ())
|
||||||
|
2024-08-26 08:21:39,160 DEBUG: query(SELECT pg_catalog.citus_add_node(%s, %s, %s, %s, 'default'), ('172.19.0.11', 5432, 0, 'secondary'))
|
||||||
|
2024-08-26 08:21:39,164 DEBUG: query(SELECT pg_catalog.citus_add_node(%s, %s, %s, %s, 'default'), ('172.19.0.7', 5432, 0, 'secondary'))
|
||||||
|
2024-08-26 08:21:39,166 DEBUG: query(COMMIT, ())
|
||||||
|
2024-08-26 08:21:39,176 DEBUG: query(SELECT pg_catalog.citus_add_node(%s, %s, %s, %s, 'default'), ('172.19.0.12', 5432, 1, 'secondary'))
|
||||||
|
2024-08-26 08:21:39,191 DEBUG: query(SELECT pg_catalog.citus_add_node(%s, %s, %s, %s, 'default'), ('172.19.0.6', 5432, 2, 'secondary'))
|
||||||
|
2024-08-26 08:21:39,211 INFO: no action. I am (coord1), the leader with the lock
|
||||||
|
2024-08-26 08:21:49,060 INFO: Lock owner: coord1; I am coord1
|
||||||
|
2024-08-26 08:21:49,166 INFO: Setting synchronous replication to 1 of 2 (coord2, coord3)
|
||||||
server signaled
|
server signaled
|
||||||
2023-11-21 09:36:28.435 UTC [52] LOG: received SIGHUP, reloading configuration files
|
2024-08-26 08:21:49.170 UTC [35] LOG: received SIGHUP, reloading configuration files
|
||||||
2023-11-21 09:36:28.436 UTC [52] LOG: parameter "synchronous_standby_names" changed to "coord3"
|
2024-08-26 08:21:49.171 UTC [35] LOG: parameter "synchronous_standby_names" changed to "ANY 1 (coord2,coord3)"
|
||||||
2023-11-21 09:36:28.641 UTC [83] LOG: standby "coord3" is now a synchronous standby with priority 1
|
2024-08-26 08:21:49.377 UTC [68] LOG: standby "coord2" is now a candidate for quorum synchronous standby
|
||||||
2023-11-21 09:36:28.641 UTC [83] STATEMENT: START_REPLICATION SLOT "coord3" 0/3000000 TIMELINE 1
|
2024-08-26 08:21:49.377 UTC [68] STATEMENT: START_REPLICATION SLOT "coord2" 0/3000000 TIMELINE 1
|
||||||
2023-11-21 09:36:30,582 INFO: Synchronous standby status assigned to ['coord3']
|
2024-08-26 08:21:49.377 UTC [69] LOG: standby "coord3" is now a candidate for quorum synchronous standby
|
||||||
2023-11-21 09:36:30,626 INFO: no action. I am (coord1), the leader with the lock
|
2024-08-26 08:21:49.377 UTC [69] STATEMENT: START_REPLICATION SLOT "coord3" 0/4000000 TIMELINE 1
|
||||||
2023-11-21 09:36:38,250 INFO: no action. I am (coord1), the leader with the lock
|
2024-08-26 08:21:50,278 INFO: Setting leader to coord1, quorum to 1 of 2 (coord2, coord3)
|
||||||
|
2024-08-26 08:21:50,390 INFO: no action. I am (coord1), the leader with the lock
|
||||||
|
2024-08-26 08:21:59,159 INFO: no action. I am (coord1), the leader with the lock
|
||||||
...
|
...
|
||||||
|
|
||||||
$ docker exec -ti demo-haproxy bash
|
$ docker exec -ti demo-haproxy bash
|
||||||
@@ -229,7 +242,7 @@ Example session:
|
|||||||
|
|
||||||
postgres@haproxy:~$ psql -h localhost -p 5000 -U postgres -d citus
|
postgres@haproxy:~$ psql -h localhost -p 5000 -U postgres -d citus
|
||||||
Password for user postgres: postgres
|
Password for user postgres: postgres
|
||||||
psql (15.5 (Debian 15.5-1.pgdg120+1))
|
psql (16.4 (Debian 16.4-1.pgdg120+1))
|
||||||
SSL connection (protocol: TLSv1.3, cipher: TLS_AES_256_GCM_SHA384, compression: off)
|
SSL connection (protocol: TLSv1.3, cipher: TLS_AES_256_GCM_SHA384, compression: off)
|
||||||
Type "help" for help.
|
Type "help" for help.
|
||||||
|
|
||||||
@@ -240,67 +253,76 @@ Example session:
|
|||||||
(1 row)
|
(1 row)
|
||||||
|
|
||||||
citus=# table pg_dist_node;
|
citus=# table pg_dist_node;
|
||||||
nodeid | groupid | nodename | nodeport | noderack | hasmetadata | isactive | noderole | nodecluster | metadatasynced | shouldhaveshards
|
nodeid | groupid | nodename | nodeport | noderack | hasmetadata | isactive | noderole | nodecluster | metadatasynced | shouldhaveshards
|
||||||
--------+---------+------------+----------+----------+-------------+----------+----------+-------------+----------------+------------------
|
--------+---------+-------------+----------+----------+-------------+----------+-----------+-------------+----------------+------------------
|
||||||
1 | 0 | 172.30.0.3 | 5432 | default | t | t | primary | default | t | f
|
1 | 0 | 172.19.0.8 | 5432 | default | t | t | primary | default | t | f
|
||||||
2 | 1 | 172.30.0.7 | 5432 | default | t | t | primary | default | t | t
|
2 | 1 | 172.19.0.2 | 5432 | default | t | t | primary | default | t | t
|
||||||
3 | 2 | 172.30.0.8 | 5432 | default | t | t | primary | default | t | t
|
3 | 2 | 172.19.0.9 | 5432 | default | t | t | primary | default | t | t
|
||||||
(3 rows)
|
4 | 0 | 172.19.0.11 | 5432 | default | t | t | secondary | default | t | f
|
||||||
|
5 | 0 | 172.19.0.7 | 5432 | default | t | t | secondary | default | t | f
|
||||||
|
6 | 1 | 172.19.0.12 | 5432 | default | f | t | secondary | default | f | t
|
||||||
|
7 | 2 | 172.19.0.6 | 5432 | default | f | t | secondary | default | f | t
|
||||||
|
(7 rows)
|
||||||
|
|
||||||
citus=# \q
|
citus=# \q
|
||||||
|
|
||||||
postgres@haproxy:~$ patronictl list
|
postgres@haproxy:~$ patronictl list
|
||||||
+ Citus cluster: demo ----------+--------------+-----------+----+-----------+
|
+ Citus cluster: demo ----------+----------------+-----------+----+-----------+
|
||||||
| Group | Member | Host | Role | State | TL | Lag in MB |
|
| Group | Member | Host | Role | State | TL | Lag in MB |
|
||||||
+-------+---------+-------------+--------------+-----------+----+-----------+
|
+-------+---------+-------------+----------------+-----------+----+-----------+
|
||||||
| 0 | coord1 | 172.30.0.3 | Leader | running | 1 | |
|
| 0 | coord1 | 172.19.0.8 | Leader | running | 1 | |
|
||||||
| 0 | coord2 | 172.30.0.12 | Replica | streaming | 1 | 0 |
|
| 0 | coord2 | 172.19.0.7 | Quorum Standby | streaming | 1 | 0 |
|
||||||
| 0 | coord3 | 172.30.0.2 | Sync Standby | streaming | 1 | 0 |
|
| 0 | coord3 | 172.19.0.11 | Quorum Standby | streaming | 1 | 0 |
|
||||||
| 1 | work1-1 | 172.30.0.7 | Leader | running | 1 | |
|
| 1 | work1-1 | 172.19.0.12 | Quorum Standby | streaming | 1 | 0 |
|
||||||
| 1 | work1-2 | 172.30.0.10 | Sync Standby | streaming | 1 | 0 |
|
| 1 | work1-2 | 172.19.0.2 | Leader | running | 1 | |
|
||||||
| 2 | work2-1 | 172.30.0.8 | Leader | running | 1 | |
|
| 2 | work2-1 | 172.19.0.6 | Quorum Standby | streaming | 1 | 0 |
|
||||||
| 2 | work2-2 | 172.30.0.11 | Sync Standby | streaming | 1 | 0 |
|
| 2 | work2-2 | 172.19.0.9 | Leader | running | 1 | |
|
||||||
+-------+---------+-------------+--------------+-----------+----+-----------+
|
+-------+---------+-------------+----------------+-----------+----+-----------+
|
||||||
|
|
||||||
|
|
||||||
postgres@haproxy:~$ patronictl switchover --group 2 --force
|
postgres@haproxy:~$ patronictl switchover --group 2 --force
|
||||||
Current cluster topology
|
Current cluster topology
|
||||||
+ Citus cluster: demo (group: 2, 7303846899271086103) --+-----------+
|
+ Citus cluster: demo (group: 2, 7407360296219029527) ---+-----------+
|
||||||
| Member | Host | Role | State | TL | Lag in MB |
|
| Member | Host | Role | State | TL | Lag in MB |
|
||||||
+---------+-------------+--------------+-----------+----+-----------+
|
+---------+------------+----------------+-----------+----+-----------+
|
||||||
| work2-1 | 172.30.0.8 | Leader | running | 1 | |
|
| work2-1 | 172.19.0.6 | Quorum Standby | streaming | 1 | 0 |
|
||||||
| work2-2 | 172.30.0.11 | Sync Standby | streaming | 1 | 0 |
|
| work2-2 | 172.19.0.9 | Leader | running | 1 | |
|
||||||
+---------+-------------+--------------+-----------+----+-----------+
|
+---------+------------+----------------+-----------+----+-----------+
|
||||||
2023-11-21 09:44:15.83849 Successfully switched over to "work2-2"
|
2024-08-26 08:31:45.92277 Successfully switched over to "work2-1"
|
||||||
+ Citus cluster: demo (group: 2, 7303846899271086103) -------+
|
+ Citus cluster: demo (group: 2, 7407360296219029527) ------+
|
||||||
| Member | Host | Role | State | TL | Lag in MB |
|
| Member | Host | Role | State | TL | Lag in MB |
|
||||||
+---------+-------------+---------+---------+----+-----------+
|
+---------+------------+---------+---------+----+-----------+
|
||||||
| work2-1 | 172.30.0.8 | Replica | stopped | | unknown |
|
| work2-1 | 172.19.0.6 | Leader | running | 1 | |
|
||||||
| work2-2 | 172.30.0.11 | Leader | running | 1 | |
|
| work2-2 | 172.19.0.9 | Replica | stopped | | unknown |
|
||||||
+---------+-------------+---------+---------+----+-----------+
|
+---------+------------+---------+---------+----+-----------+
|
||||||
|
|
||||||
postgres@haproxy:~$ patronictl list
|
postgres@haproxy:~$ patronictl list
|
||||||
+ Citus cluster: demo ----------+--------------+-----------+----+-----------+
|
+ Citus cluster: demo ----------+----------------+-----------+----+-----------+
|
||||||
| Group | Member | Host | Role | State | TL | Lag in MB |
|
| Group | Member | Host | Role | State | TL | Lag in MB |
|
||||||
+-------+---------+-------------+--------------+-----------+----+-----------+
|
+-------+---------+-------------+----------------+-----------+----+-----------+
|
||||||
| 0 | coord1 | 172.30.0.3 | Leader | running | 1 | |
|
| 0 | coord1 | 172.19.0.8 | Leader | running | 1 | |
|
||||||
| 0 | coord2 | 172.30.0.12 | Replica | streaming | 1 | 0 |
|
| 0 | coord2 | 172.19.0.7 | Quorum Standby | streaming | 1 | 0 |
|
||||||
| 0 | coord3 | 172.30.0.2 | Sync Standby | streaming | 1 | 0 |
|
| 0 | coord3 | 172.19.0.11 | Quorum Standby | streaming | 1 | 0 |
|
||||||
| 1 | work1-1 | 172.30.0.7 | Leader | running | 1 | |
|
| 1 | work1-1 | 172.19.0.12 | Quorum Standby | streaming | 1 | 0 |
|
||||||
| 1 | work1-2 | 172.30.0.10 | Sync Standby | streaming | 1 | 0 |
|
| 1 | work1-2 | 172.19.0.2 | Leader | running | 1 | |
|
||||||
| 2 | work2-1 | 172.30.0.8 | Sync Standby | streaming | 2 | 0 |
|
| 2 | work2-1 | 172.19.0.6 | Leader | running | 2 | |
|
||||||
| 2 | work2-2 | 172.30.0.11 | Leader | running | 2 | |
|
| 2 | work2-2 | 172.19.0.9 | Quorum Standby | streaming | 2 | 0 |
|
||||||
+-------+---------+-------------+--------------+-----------+----+-----------+
|
+-------+---------+-------------+----------------+-----------+----+-----------+
|
||||||
|
|
||||||
postgres@haproxy:~$ psql -h localhost -p 5000 -U postgres -d citus
|
postgres@haproxy:~$ psql -h localhost -p 5000 -U postgres -d citus
|
||||||
psql (15.5 (Debian 15.5-1.pgdg120+1))
|
Password for user postgres: postgres
|
||||||
|
psql (16.4 (Debian 16.4-1.pgdg120+1))
|
||||||
SSL connection (protocol: TLSv1.3, cipher: TLS_AES_256_GCM_SHA384, compression: off)
|
SSL connection (protocol: TLSv1.3, cipher: TLS_AES_256_GCM_SHA384, compression: off)
|
||||||
Type "help" for help.
|
Type "help" for help.
|
||||||
|
|
||||||
citus=# table pg_dist_node;
|
citus=# table pg_dist_node;
|
||||||
nodeid | groupid | nodename | nodeport | noderack | hasmetadata | isactive | noderole | nodecluster | metadatasynced | shouldhaveshards
|
nodeid | groupid | nodename | nodeport | noderack | hasmetadata | isactive | noderole | nodecluster | metadatasynced | shouldhaveshards
|
||||||
--------+---------+-------------+----------+----------+-------------+----------+----------+-------------+----------------+------------------
|
--------+---------+-------------+----------+----------+-------------+----------+-----------+-------------+----------------+------------------
|
||||||
1 | 0 | 172.30.0.3 | 5432 | default | t | t | primary | default | t | f
|
1 | 0 | 172.19.0.8 | 5432 | default | t | t | primary | default | t | f
|
||||||
3 | 2 | 172.30.0.11 | 5432 | default | t | t | primary | default | t | t
|
4 | 0 | 172.19.0.11 | 5432 | default | t | t | secondary | default | t | f
|
||||||
2 | 1 | 172.30.0.7 | 5432 | default | t | t | primary | default | t | t
|
5 | 0 | 172.19.0.7 | 5432 | default | t | t | secondary | default | t | f
|
||||||
(3 rows)
|
6 | 1 | 172.19.0.12 | 5432 | default | f | t | secondary | default | f | t
|
||||||
|
3 | 2 | 172.19.0.6 | 5432 | default | t | t | primary | default | t | t
|
||||||
|
2 | 1 | 172.19.0.2 | 5432 | default | t | t | primary | default | t | t
|
||||||
|
8 | 2 | 172.19.0.9 | 5432 | default | f | t | secondary | default | f | t
|
||||||
|
(7 rows)
|
||||||
|
|||||||
@@ -38,7 +38,7 @@ EOT
|
|||||||
exec dumb-init "$@"
|
exec dumb-init "$@"
|
||||||
;;
|
;;
|
||||||
etcd)
|
etcd)
|
||||||
exec "$@" -advertise-client-urls "http://$DOCKER_IP:2379"
|
exec "$@" --auto-compaction-retention=1 -advertise-client-urls "http://$DOCKER_IP:2379"
|
||||||
;;
|
;;
|
||||||
zookeeper)
|
zookeeper)
|
||||||
exec /usr/share/zookeeper/bin/zkServer.sh start-foreground
|
exec /usr/share/zookeeper/bin/zkServer.sh start-foreground
|
||||||
|
|||||||
+19
-10
@@ -28,9 +28,14 @@ Log
|
|||||||
- **PATRONI\_LOG\_STATIC\_FIELDS**: add additional fields to the log. This option is only available when the log type is set to **json**. Example ``PATRONI_LOG_STATIC_FIELDS="{app: patroni}"``
|
- **PATRONI\_LOG\_STATIC\_FIELDS**: add additional fields to the log. This option is only available when the log type is set to **json**. Example ``PATRONI_LOG_STATIC_FIELDS="{app: patroni}"``
|
||||||
- **PATRONI\_LOG\_MAX\_QUEUE\_SIZE**: Patroni is using two-step logging. Log records are written into the in-memory queue and there is a separate thread which pulls them from the queue and writes to stderr or file. The maximum size of the internal queue is limited by default by **1000** records, which is enough to keep logs for the past 1h20m.
|
- **PATRONI\_LOG\_MAX\_QUEUE\_SIZE**: Patroni is using two-step logging. Log records are written into the in-memory queue and there is a separate thread which pulls them from the queue and writes to stderr or file. The maximum size of the internal queue is limited by default by **1000** records, which is enough to keep logs for the past 1h20m.
|
||||||
- **PATRONI\_LOG\_DIR**: Directory to write application logs to. The directory must exist and be writable by the user executing Patroni. If you set this env variable, the application will retain 4 25MB logs by default. You can tune those retention values with `PATRONI_LOG_FILE_NUM` and `PATRONI_LOG_FILE_SIZE` (see below).
|
- **PATRONI\_LOG\_DIR**: Directory to write application logs to. The directory must exist and be writable by the user executing Patroni. If you set this env variable, the application will retain 4 25MB logs by default. You can tune those retention values with `PATRONI_LOG_FILE_NUM` and `PATRONI_LOG_FILE_SIZE` (see below).
|
||||||
|
- **PATRONI\_LOG\_MODE**: Permissions for log files (for example, ``0644``). If not specified, permissions will be set based on the current umask value.
|
||||||
- **PATRONI\_LOG\_FILE\_NUM**: The number of application logs to retain.
|
- **PATRONI\_LOG\_FILE\_NUM**: The number of application logs to retain.
|
||||||
- **PATRONI\_LOG\_FILE\_SIZE**: Size of patroni.log file (in bytes) that triggers a log rolling.
|
- **PATRONI\_LOG\_FILE\_SIZE**: Size of patroni.log file (in bytes) that triggers a log rolling.
|
||||||
- **PATRONI\_LOG\_LOGGERS**: Redefine logging level per python module. Example ``PATRONI_LOG_LOGGERS="{patroni.postmaster: WARNING, urllib3: DEBUG}"``
|
- **PATRONI\_LOG\_LOGGERS**: Redefine logging level per python module. Example ``PATRONI_LOG_LOGGERS="{patroni.postmaster: WARNING, urllib3: DEBUG}"``
|
||||||
|
- **PATRONI\_LOG\_DEDUPLICATE\_HEARTBEAT\_LOGS**: If set to ``true``, successive heartbeat logs that are identical shall not be output. Default value is ``false``.
|
||||||
|
|
||||||
|
.. warning::
|
||||||
|
The time the HA loop executes at can be very valuable information in diagnosing failovers due to resource exhaustion and similar problems. When ``PATRONI_LOG_DEDUPLICATE_HEARTBEAT_LOGS`` is set to ``true`` there will be no log generated for the HA loop execution (unless the leader changes) and hence this potentially useful information will not be available from the logs.
|
||||||
|
|
||||||
Citus
|
Citus
|
||||||
-----
|
-----
|
||||||
@@ -54,9 +59,9 @@ Consul
|
|||||||
- **PATRONI\_CONSUL\_CONSISTENCY**: (optional) Select consul consistency mode. Possible values are ``default``, ``consistent``, or ``stale`` (more details in `consul API reference <https://www.consul.io/api/features/consistency.html/>`__)
|
- **PATRONI\_CONSUL\_CONSISTENCY**: (optional) Select consul consistency mode. Possible values are ``default``, ``consistent``, or ``stale`` (more details in `consul API reference <https://www.consul.io/api/features/consistency.html/>`__)
|
||||||
- **PATRONI\_CONSUL\_CHECKS**: (optional) list of Consul health checks used for the session. By default an empty list is used.
|
- **PATRONI\_CONSUL\_CHECKS**: (optional) list of Consul health checks used for the session. By default an empty list is used.
|
||||||
- **PATRONI\_CONSUL\_REGISTER\_SERVICE**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, primary, replica, or standby-leader depending on the node's role. Defaults to **false**
|
- **PATRONI\_CONSUL\_REGISTER\_SERVICE**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, primary, replica, or standby-leader depending on the node's role. Defaults to **false**
|
||||||
- **PATRONI\_CONSUL\_SERVICE\_TAGS**: (optional) additional static tags to add to the Consul service apart from the role (``master``/``primary``/``replica``/``standby-leader``). By default an empty list is used.
|
- **PATRONI\_CONSUL\_SERVICE\_TAGS**: (optional) additional static tags to add to the Consul service apart from the role (``primary``/``replica``/``standby-leader``). By default an empty list is used.
|
||||||
- **PATRONI\_CONSUL\_SERVICE\_CHECK\_INTERVAL**: (optional) how often to perform health check against registered url
|
- **PATRONI\_CONSUL\_SERVICE\_CHECK\_INTERVAL**: (optional) how often to perform health check against registered url
|
||||||
- **PATRONI\_CONSUL\_SERVICE\_CHECK\_TLS\_SERVER\_NAME**: (optional) overide SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
|
- **PATRONI\_CONSUL\_SERVICE\_CHECK\_TLS\_SERVER\_NAME**: (optional) override SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
|
||||||
|
|
||||||
Etcd
|
Etcd
|
||||||
----
|
----
|
||||||
@@ -80,7 +85,7 @@ Etcdv3
|
|||||||
Environment names for Etcdv3 are similar as for Etcd, you just need to use ``ETCD3`` instead of ``ETCD`` in the variable name. Example: ``PATRONI_ETCD3_HOST``, ``PATRONI_ETCD3_CACERT``, and so on.
|
Environment names for Etcdv3 are similar as for Etcd, you just need to use ``ETCD3`` instead of ``ETCD`` in the variable name. Example: ``PATRONI_ETCD3_HOST``, ``PATRONI_ETCD3_CACERT``, and so on.
|
||||||
|
|
||||||
.. warning::
|
.. warning::
|
||||||
Keys created with protocol version 2 are not visible with protocol version 3 and the other way around, therefore it is not possible to switch from Etcd to Etcdv3 just by updating Patroni configuration.
|
Keys created with protocol version 2 are not visible with protocol version 3 and the other way around, therefore it is not possible to switch from Etcd to Etcdv3 just by updating Patroni configuration. In addition, Patroni uses Etcd's gRPC-gateway (proxy) to communicate with the V3 API, which means that TLS common name authentication is not possible.
|
||||||
|
|
||||||
|
|
||||||
ZooKeeper
|
ZooKeeper
|
||||||
@@ -112,11 +117,12 @@ Kubernetes
|
|||||||
- **PATRONI\_KUBERNETES\_NAMESPACE**: (optional) Kubernetes namespace where the Patroni pod is running. Default value is `default`.
|
- **PATRONI\_KUBERNETES\_NAMESPACE**: (optional) Kubernetes namespace where the Patroni pod is running. Default value is `default`.
|
||||||
- **PATRONI\_KUBERNETES\_LABELS**: Labels in format ``{label1: value1, label2: value2}``. These labels will be used to find existing objects (Pods and either Endpoints or ConfigMaps) associated with the current cluster. Also Patroni will set them on every object (Endpoint or ConfigMap) it creates.
|
- **PATRONI\_KUBERNETES\_LABELS**: Labels in format ``{label1: value1, label2: value2}``. These labels will be used to find existing objects (Pods and either Endpoints or ConfigMaps) associated with the current cluster. Also Patroni will set them on every object (Endpoint or ConfigMap) it creates.
|
||||||
- **PATRONI\_KUBERNETES\_SCOPE\_LABEL**: (optional) name of the label containing cluster name. Default value is `cluster-name`.
|
- **PATRONI\_KUBERNETES\_SCOPE\_LABEL**: (optional) name of the label containing cluster name. Default value is `cluster-name`.
|
||||||
- **PATRONI\_KUBERNETES\_ROLE\_LABEL**: (optional) name of the label containing role (master or replica or other custom value). Patroni will set this label on the pod it runs in. Default value is ``role``.
|
- **PATRONI\_KUBERNETES\_BOOTSTRAP\_LABELS**: (optional) Labels in format ``{label1: value1, label2: value2}``. These labels will be assigned to a Patroni pod when its state is either ``initializing new cluster``, ``running custom bootstrap script``, ``starting after custom bootstrap`` or ``creating replica``.
|
||||||
- **PATRONI\_KUBERNETES\_LEADER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is `master`. Default value is `master`.
|
- **PATRONI\_KUBERNETES\_ROLE\_LABEL**: (optional) name of the label containing role (`primary`, `replica` or other custom value). Patroni will set this label on the pod it runs in. Default value is ``role``.
|
||||||
|
- **PATRONI\_KUBERNETES\_LEADER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is `primary`. Default value is `primary`.
|
||||||
- **PATRONI\_KUBERNETES\_FOLLOWER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is `replica`. Default value is `replica`.
|
- **PATRONI\_KUBERNETES\_FOLLOWER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is `replica`. Default value is `replica`.
|
||||||
- **PATRONI\_KUBERNETES\_STANDBY\_LEADER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is ``standby_leader``. Default value is ``master``.
|
- **PATRONI\_KUBERNETES\_STANDBY\_LEADER\_LABEL\_VALUE**: (optional) value of the pod label when Postgres role is ``standby_leader``. Default value is ``primary``.
|
||||||
- **PATRONI\_KUBERNETES\_TMP\_ROLE\_LABEL**: (optional) name of the temporary label containing role (master or replica). Value of this label will always use the default of corresponding role. Set only when necessary.
|
- **PATRONI\_KUBERNETES\_TMP\_ROLE\_LABEL**: (optional) name of the temporary label containing role (`primary` or `replica`). Value of this label will always use the default of corresponding role. Set only when necessary.
|
||||||
- **PATRONI\_KUBERNETES\_USE\_ENDPOINTS**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
- **PATRONI\_KUBERNETES\_USE\_ENDPOINTS**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||||
- **PATRONI\_KUBERNETES\_POD\_IP**: (optional) IP address of the pod Patroni is running in. This value is required when `PATRONI_KUBERNETES_USE_ENDPOINTS` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
- **PATRONI\_KUBERNETES\_POD\_IP**: (optional) IP address of the pod Patroni is running in. This value is required when `PATRONI_KUBERNETES_USE_ENDPOINTS` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||||
- **PATRONI\_KUBERNETES\_PORTS**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``PATRONI_KUBERNETES_PORTS='[{"name": "postgresql", "port": 5432}]'`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `PATRONI_KUBERNETES_USE_ENDPOINTS` is set.
|
- **PATRONI\_KUBERNETES\_PORTS**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``PATRONI_KUBERNETES_PORTS='[{"name": "postgresql", "port": 5432}]'`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `PATRONI_KUBERNETES_USE_ENDPOINTS` is set.
|
||||||
@@ -154,9 +160,10 @@ PostgreSQL
|
|||||||
- **PATRONI\_REPLICATION\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
- **PATRONI\_REPLICATION\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||||
- **PATRONI\_REPLICATION\_SSLPASSWORD**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``PATRONI_REPLICATION_SSLKEY``.
|
- **PATRONI\_REPLICATION\_SSLPASSWORD**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``PATRONI_REPLICATION_SSLKEY``.
|
||||||
- **PATRONI\_REPLICATION\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **PATRONI\_REPLICATION\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **PATRONI\_REPLICATION\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **PATRONI\_REPLICATION\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one or more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **PATRONI\_REPLICATION\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **PATRONI\_REPLICATION\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **PATRONI\_REPLICATION\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **PATRONI\_REPLICATION\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **PATRONI\_REPLICATION\_SSLNEGOTIATION**: (optional) maps to the `sslnegotiation <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLNEGOTIATION>`__ connection parameter, which controls how SSL encryption is negotiated with the server, if SSL is used.
|
||||||
- **PATRONI\_REPLICATION\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **PATRONI\_REPLICATION\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **PATRONI\_REPLICATION\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **PATRONI\_REPLICATION\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
- **PATRONI\_SUPERUSER\_USERNAME**: name for the superuser, set during initialization (initdb) and later used by Patroni to connect to the postgres. Also this user is used by pg_rewind.
|
- **PATRONI\_SUPERUSER\_USERNAME**: name for the superuser, set during initialization (initdb) and later used by Patroni to connect to the postgres. Also this user is used by pg_rewind.
|
||||||
@@ -165,9 +172,10 @@ PostgreSQL
|
|||||||
- **PATRONI\_SUPERUSER\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
- **PATRONI\_SUPERUSER\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||||
- **PATRONI\_SUPERUSER\_SSLPASSWORD**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``PATRONI_SUPERUSER_SSLKEY``.
|
- **PATRONI\_SUPERUSER\_SSLPASSWORD**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``PATRONI_SUPERUSER_SSLKEY``.
|
||||||
- **PATRONI\_SUPERUSER\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **PATRONI\_SUPERUSER\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **PATRONI\_SUPERUSER\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **PATRONI\_SUPERUSER\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one or more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **PATRONI\_SUPERUSER\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **PATRONI\_SUPERUSER\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **PATRONI\_SUPERUSER\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **PATRONI\_SUPERUSER\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **PATRONI\_SUPERUSER\_SSLNEGOTIATION**: (optional) maps to the `sslnegotiation <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLNEGOTIATION>`__ connection parameter, which controls how SSL encryption is negotiated with the server, if SSL is used.
|
||||||
- **PATRONI\_SUPERUSER\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **PATRONI\_SUPERUSER\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **PATRONI\_SUPERUSER\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **PATRONI\_SUPERUSER\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
- **PATRONI\_REWIND\_USERNAME**: (optional) name for the user for ``pg_rewind``; the user will be created during initialization of postgres 11+ and all necessary `permissions <https://www.postgresql.org/docs/11/app-pgrewind.html#id-1.9.5.8.8>`__ will be granted.
|
- **PATRONI\_REWIND\_USERNAME**: (optional) name for the user for ``pg_rewind``; the user will be created during initialization of postgres 11+ and all necessary `permissions <https://www.postgresql.org/docs/11/app-pgrewind.html#id-1.9.5.8.8>`__ will be granted.
|
||||||
@@ -176,9 +184,10 @@ PostgreSQL
|
|||||||
- **PATRONI\_REWIND\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
- **PATRONI\_REWIND\_SSLKEY**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||||
- **PATRONI\_REWIND\_SSLPASSWORD**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``PATRONI_REWIND_SSLKEY``.
|
- **PATRONI\_REWIND\_SSLPASSWORD**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``PATRONI_REWIND_SSLKEY``.
|
||||||
- **PATRONI\_REWIND\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **PATRONI\_REWIND\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **PATRONI\_REWIND\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **PATRONI\_REWIND\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one or more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **PATRONI\_REWIND\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **PATRONI\_REWIND\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **PATRONI\_REWIND\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **PATRONI\_REWIND\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **PATRONI\_REWIND\_SSLNEGOTIATION**: (optional) maps to the `sslnegotiation <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLNEGOTIATION>`__ connection parameter, which controls how SSL encryption is negotiated with the server, if SSL is used.
|
||||||
- **PATRONI\_REWIND\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **PATRONI\_REWIND\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **PATRONI\_REWIND\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **PATRONI\_REWIND\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
|
|
||||||
|
|||||||
+3
-5
@@ -6,12 +6,10 @@ Introduction
|
|||||||
|
|
||||||
Patroni is a template for high availability (HA) PostgreSQL solutions using Python. Patroni originated as a fork of `Governor <https://github.com/compose/governor>`__, the project from Compose. It includes plenty of new features.
|
Patroni is a template for high availability (HA) PostgreSQL solutions using Python. Patroni originated as a fork of `Governor <https://github.com/compose/governor>`__, the project from Compose. It includes plenty of new features.
|
||||||
|
|
||||||
For an example of a Docker-based deployment with Patroni, see `Spilo <https://github.com/zalando/spilo>`__, currently in use at Zalando.
|
|
||||||
|
|
||||||
For additional background info, see:
|
For additional background info, see:
|
||||||
|
|
||||||
* `PostgreSQL HA with Kubernetes and Patroni <https://www.youtube.com/watch?v=iruaCgeG7qs>`__, talk by Josh Berkus at KubeCon 2016 (video)
|
* `PostgreSQL HA with Kubernetes and Patroni <https://www.youtube.com/watch?v=iruaCgeG7qs>`__, talk by Josh Berkus at KubeCon 2016 (video)
|
||||||
* `Feb. 2016 Zalando Tech blog post <https://tech.zalando.de/blog/zalandos-patroni-a-template-for-high-availability-postgresql/>`__
|
* `Feb. 2016 Zalando Tech blog post <https://engineering.zalando.com/posts/2016/02/zalandos-patroni-a-template-for-high-availability-postgresql.html>`__
|
||||||
|
|
||||||
|
|
||||||
Development Status
|
Development Status
|
||||||
@@ -39,7 +37,7 @@ perfectly fine. You can add more standby nodes later.
|
|||||||
Running and Configuring
|
Running and Configuring
|
||||||
-----------------------
|
-----------------------
|
||||||
|
|
||||||
The following section assumes Patroni repository as being cloned from https://github.com/zalando/patroni. Namely, you
|
The following section assumes Patroni repository as being cloned from https://github.com/patroni/patroni. Namely, you
|
||||||
will need example configuration files `postgres0.yml` and `postgres1.yml`. If you installed Patroni with pip, you can
|
will need example configuration files `postgres0.yml` and `postgres1.yml`. If you installed Patroni with pip, you can
|
||||||
obtain those files from the git repository and replace `./patroni.py` below with `patroni` command.
|
obtain those files from the git repository and replace `./patroni.py` below with `patroni` command.
|
||||||
|
|
||||||
@@ -69,7 +67,7 @@ run:
|
|||||||
YAML Configuration
|
YAML Configuration
|
||||||
------------------
|
------------------
|
||||||
|
|
||||||
Go :ref:`here <yaml_configuration>` for comprehensive information about settings for etcd, consul, and ZooKeeper. And for an example, see `postgres0.yml <https://github.com/zalando/patroni/blob/master/postgres0.yml>`__.
|
Go :ref:`here <yaml_configuration>` for comprehensive information about settings for etcd, consul, and ZooKeeper. And for an example, see `postgres0.yml <https://github.com/patroni/patroni/blob/master/postgres0.yml>`__.
|
||||||
|
|
||||||
|
|
||||||
Environment Configuration
|
Environment Configuration
|
||||||
|
|||||||
+71
-60
@@ -34,6 +34,8 @@ There are only a few simple rules you need to follow:
|
|||||||
|
|
||||||
After that you just need to start Patroni and it will handle the rest:
|
After that you just need to start Patroni and it will handle the rest:
|
||||||
|
|
||||||
|
0. Patroni will set ``bootstrap.dcs.synchronous_mode`` to :ref:`quorum <quorum_mode>`
|
||||||
|
if it is not explicitly set to any other value.
|
||||||
1. ``citus`` extension will be automatically added to ``shared_preload_libraries``.
|
1. ``citus`` extension will be automatically added to ``shared_preload_libraries``.
|
||||||
2. If ``max_prepared_transactions`` isn't explicitly set in the global
|
2. If ``max_prepared_transactions`` isn't explicitly set in the global
|
||||||
:ref:`dynamic configuration <dynamic_configuration>` Patroni will
|
:ref:`dynamic configuration <dynamic_configuration>` Patroni will
|
||||||
@@ -56,7 +58,7 @@ patronictl
|
|||||||
----------
|
----------
|
||||||
|
|
||||||
Coordinator and worker clusters are physically different PostgreSQL/Patroni
|
Coordinator and worker clusters are physically different PostgreSQL/Patroni
|
||||||
clusters that are just logically groupped together using the
|
clusters that are just logically grouped together using the
|
||||||
`Citus <https://github.com/citusdata/citus>`__ database extension to
|
`Citus <https://github.com/citusdata/citus>`__ database extension to
|
||||||
PostgreSQL. Therefore in most cases it is not possible to manage them as a
|
PostgreSQL. Therefore in most cases it is not possible to manage them as a
|
||||||
single entity.
|
single entity.
|
||||||
@@ -77,36 +79,36 @@ It results in two major differences in :ref:`patronictl` behaviour when
|
|||||||
An example of :ref:`patronictl_list` output for the Citus cluster::
|
An example of :ref:`patronictl_list` output for the Citus cluster::
|
||||||
|
|
||||||
postgres@coord1:~$ patronictl list demo
|
postgres@coord1:~$ patronictl list demo
|
||||||
+ Citus cluster: demo ----------+--------------+---------+----+-----------+
|
+ Citus cluster: demo ----------+----------------+---------+----+-----------+
|
||||||
| Group | Member | Host | Role | State | TL | Lag in MB |
|
| Group | Member | Host | Role | State | TL | Lag in MB |
|
||||||
+-------+---------+-------------+--------------+---------+----+-----------+
|
+-------+---------+-------------+----------------+---------+----+-----------+
|
||||||
| 0 | coord1 | 172.27.0.10 | Replica | running | 1 | 0 |
|
| 0 | coord1 | 172.27.0.10 | Replica | running | 1 | 0 |
|
||||||
| 0 | coord2 | 172.27.0.6 | Sync Standby | running | 1 | 0 |
|
| 0 | coord2 | 172.27.0.6 | Quorum Standby | running | 1 | 0 |
|
||||||
| 0 | coord3 | 172.27.0.4 | Leader | running | 1 | |
|
| 0 | coord3 | 172.27.0.4 | Leader | running | 1 | |
|
||||||
| 1 | work1-1 | 172.27.0.8 | Sync Standby | running | 1 | 0 |
|
| 1 | work1-1 | 172.27.0.8 | Quorum Standby | running | 1 | 0 |
|
||||||
| 1 | work1-2 | 172.27.0.2 | Leader | running | 1 | |
|
| 1 | work1-2 | 172.27.0.2 | Leader | running | 1 | |
|
||||||
| 2 | work2-1 | 172.27.0.5 | Sync Standby | running | 1 | 0 |
|
| 2 | work2-1 | 172.27.0.5 | Quorum Standby | running | 1 | 0 |
|
||||||
| 2 | work2-2 | 172.27.0.7 | Leader | running | 1 | |
|
| 2 | work2-2 | 172.27.0.7 | Leader | running | 1 | |
|
||||||
+-------+---------+-------------+--------------+---------+----+-----------+
|
+-------+---------+-------------+----------------+---------+----+-----------+
|
||||||
|
|
||||||
If we add the ``--group`` option, the output will change to::
|
If we add the ``--group`` option, the output will change to::
|
||||||
|
|
||||||
postgres@coord1:~$ patronictl list demo --group 0
|
postgres@coord1:~$ patronictl list demo --group 0
|
||||||
+ Citus cluster: demo (group: 0, 7179854923829112860) -----------+
|
+ Citus cluster: demo (group: 0, 7179854923829112860) -+-----------+
|
||||||
| Member | Host | Role | State | TL | Lag in MB |
|
| Member | Host | Role | State | TL | Lag in MB |
|
||||||
+--------+-------------+--------------+---------+----+-----------+
|
+--------+-------------+----------------+---------+----+-----------+
|
||||||
| coord1 | 172.27.0.10 | Replica | running | 1 | 0 |
|
| coord1 | 172.27.0.10 | Replica | running | 1 | 0 |
|
||||||
| coord2 | 172.27.0.6 | Sync Standby | running | 1 | 0 |
|
| coord2 | 172.27.0.6 | Quorum Standby | running | 1 | 0 |
|
||||||
| coord3 | 172.27.0.4 | Leader | running | 1 | |
|
| coord3 | 172.27.0.4 | Leader | running | 1 | |
|
||||||
+--------+-------------+--------------+---------+----+-----------+
|
+--------+-------------+----------------+---------+----+-----------+
|
||||||
|
|
||||||
postgres@coord1:~$ patronictl list demo --group 1
|
postgres@coord1:~$ patronictl list demo --group 1
|
||||||
+ Citus cluster: demo (group: 1, 7179854923881963547) -----------+
|
+ Citus cluster: demo (group: 1, 7179854923881963547) -+-----------+
|
||||||
| Member | Host | Role | State | TL | Lag in MB |
|
| Member | Host | Role | State | TL | Lag in MB |
|
||||||
+---------+------------+--------------+---------+----+-----------+
|
+---------+------------+----------------+---------+----+-----------+
|
||||||
| work1-1 | 172.27.0.8 | Sync Standby | running | 1 | 0 |
|
| work1-1 | 172.27.0.8 | Quorum Standby | running | 1 | 0 |
|
||||||
| work1-2 | 172.27.0.2 | Leader | running | 1 | |
|
| work1-2 | 172.27.0.2 | Leader | running | 1 | |
|
||||||
+---------+------------+--------------+---------+----+-----------+
|
+---------+------------+----------------+---------+----+-----------+
|
||||||
|
|
||||||
Citus worker switchover
|
Citus worker switchover
|
||||||
-----------------------
|
-----------------------
|
||||||
@@ -122,30 +124,30 @@ new primary worker node is ready to accept read-write queries.
|
|||||||
An example of :ref:`patronictl_switchover` on the worker cluster::
|
An example of :ref:`patronictl_switchover` on the worker cluster::
|
||||||
|
|
||||||
postgres@coord1:~$ patronictl switchover demo
|
postgres@coord1:~$ patronictl switchover demo
|
||||||
+ Citus cluster: demo ----------+--------------+---------+----+-----------+
|
+ Citus cluster: demo ----------+----------------+---------+----+-----------+
|
||||||
| Group | Member | Host | Role | State | TL | Lag in MB |
|
| Group | Member | Host | Role | State | TL | Lag in MB |
|
||||||
+-------+---------+-------------+--------------+---------+----+-----------+
|
+-------+---------+-------------+----------------+---------+----+-----------+
|
||||||
| 0 | coord1 | 172.27.0.10 | Replica | running | 1 | 0 |
|
| 0 | coord1 | 172.27.0.10 | Replica | running | 1 | 0 |
|
||||||
| 0 | coord2 | 172.27.0.6 | Sync Standby | running | 1 | 0 |
|
| 0 | coord2 | 172.27.0.6 | Quorum Standby | running | 1 | 0 |
|
||||||
| 0 | coord3 | 172.27.0.4 | Leader | running | 1 | |
|
| 0 | coord3 | 172.27.0.4 | Leader | running | 1 | |
|
||||||
| 1 | work1-1 | 172.27.0.8 | Leader | running | 1 | |
|
| 1 | work1-1 | 172.27.0.8 | Leader | running | 1 | |
|
||||||
| 1 | work1-2 | 172.27.0.2 | Sync Standby | running | 1 | 0 |
|
| 1 | work1-2 | 172.27.0.2 | Quorum Standby | running | 1 | 0 |
|
||||||
| 2 | work2-1 | 172.27.0.5 | Sync Standby | running | 1 | 0 |
|
| 2 | work2-1 | 172.27.0.5 | Quorum Standby | running | 1 | 0 |
|
||||||
| 2 | work2-2 | 172.27.0.7 | Leader | running | 1 | |
|
| 2 | work2-2 | 172.27.0.7 | Leader | running | 1 | |
|
||||||
+-------+---------+-------------+--------------+---------+----+-----------+
|
+-------+---------+-------------+----------------+---------+----+-----------+
|
||||||
Citus group: 2
|
Citus group: 2
|
||||||
Primary [work2-2]:
|
Primary [work2-2]:
|
||||||
Candidate ['work2-1'] []:
|
Candidate ['work2-1'] []:
|
||||||
When should the switchover take place (e.g. 2022-12-22T08:02 ) [now]:
|
When should the switchover take place (e.g. 2024-08-26T08:02 ) [now]:
|
||||||
Current cluster topology
|
Current cluster topology
|
||||||
+ Citus cluster: demo (group: 2, 7179854924063375386) -----------+
|
+ Citus cluster: demo (group: 2, 7179854924063375386) -+-----------+
|
||||||
| Member | Host | Role | State | TL | Lag in MB |
|
| Member | Host | Role | State | TL | Lag in MB |
|
||||||
+---------+------------+--------------+---------+----+-----------+
|
+---------+------------+----------------+---------+----+-----------+
|
||||||
| work2-1 | 172.27.0.5 | Sync Standby | running | 1 | 0 |
|
| work2-1 | 172.27.0.5 | Quorum Standby | running | 1 | 0 |
|
||||||
| work2-2 | 172.27.0.7 | Leader | running | 1 | |
|
| work2-2 | 172.27.0.7 | Leader | running | 1 | |
|
||||||
+---------+------------+--------------+---------+----+-----------+
|
+---------+------------+----------------+---------+----+-----------+
|
||||||
Are you sure you want to switchover cluster demo, demoting current primary work2-2? [y/N]: y
|
Are you sure you want to switchover cluster demo, demoting current primary work2-2? [y/N]: y
|
||||||
2022-12-22 07:02:40.33003 Successfully switched over to "work2-1"
|
2024-08-26 07:02:40.33003 Successfully switched over to "work2-1"
|
||||||
+ Citus cluster: demo (group: 2, 7179854924063375386) ------+
|
+ Citus cluster: demo (group: 2, 7179854924063375386) ------+
|
||||||
| Member | Host | Role | State | TL | Lag in MB |
|
| Member | Host | Role | State | TL | Lag in MB |
|
||||||
+---------+------------+---------+---------+----+-----------+
|
+---------+------------+---------+---------+----+-----------+
|
||||||
@@ -154,32 +156,41 @@ An example of :ref:`patronictl_switchover` on the worker cluster::
|
|||||||
+---------+------------+---------+---------+----+-----------+
|
+---------+------------+---------+---------+----+-----------+
|
||||||
|
|
||||||
postgres@coord1:~$ patronictl list demo
|
postgres@coord1:~$ patronictl list demo
|
||||||
+ Citus cluster: demo ----------+--------------+---------+----+-----------+
|
+ Citus cluster: demo ----------+----------------+---------+----+-----------+
|
||||||
| Group | Member | Host | Role | State | TL | Lag in MB |
|
| Group | Member | Host | Role | State | TL | Lag in MB |
|
||||||
+-------+---------+-------------+--------------+---------+----+-----------+
|
+-------+---------+-------------+----------------+---------+----+-----------+
|
||||||
| 0 | coord1 | 172.27.0.10 | Replica | running | 1 | 0 |
|
| 0 | coord1 | 172.27.0.10 | Replica | running | 1 | 0 |
|
||||||
| 0 | coord2 | 172.27.0.6 | Sync Standby | running | 1 | 0 |
|
| 0 | coord2 | 172.27.0.6 | Quorum Standby | running | 1 | 0 |
|
||||||
| 0 | coord3 | 172.27.0.4 | Leader | running | 1 | |
|
| 0 | coord3 | 172.27.0.4 | Leader | running | 1 | |
|
||||||
| 1 | work1-1 | 172.27.0.8 | Leader | running | 1 | |
|
| 1 | work1-1 | 172.27.0.8 | Leader | running | 1 | |
|
||||||
| 1 | work1-2 | 172.27.0.2 | Sync Standby | running | 1 | 0 |
|
| 1 | work1-2 | 172.27.0.2 | Quorum Standby | running | 1 | 0 |
|
||||||
| 2 | work2-1 | 172.27.0.5 | Leader | running | 2 | |
|
| 2 | work2-1 | 172.27.0.5 | Leader | running | 2 | |
|
||||||
| 2 | work2-2 | 172.27.0.7 | Sync Standby | running | 2 | 0 |
|
| 2 | work2-2 | 172.27.0.7 | Quorum Standby | running | 2 | 0 |
|
||||||
+-------+---------+-------------+--------------+---------+----+-----------+
|
+-------+---------+-------------+----------------+---------+----+-----------+
|
||||||
|
|
||||||
And this is how it looks on the coordinator side::
|
And this is how it looks on the coordinator side::
|
||||||
|
|
||||||
# The worker primary notifies the coordinator that it is going to execute "pg_ctl stop".
|
# The worker primary notifies the coordinator that it is going to execute "pg_ctl stop".
|
||||||
2022-12-22 07:02:38,636 DEBUG: query("BEGIN")
|
2024-08-26 07:02:38,636 DEBUG: query(BEGIN, ())
|
||||||
2022-12-22 07:02:38,636 DEBUG: query("SELECT pg_catalog.citus_update_node(3, '172.27.0.7-demoted', 5432, true, 10000)")
|
2024-08-26 07:02:38,636 DEBUG: query(SELECT pg_catalog.citus_update_node(%s, %s, %s, true, %s), (3, '172.19.0.7-demoted', 5432, 10000))
|
||||||
# From this moment all application traffic on the coordinator to the worker group 2 is paused.
|
# From this moment all application traffic on the coordinator to the worker group 2 is paused.
|
||||||
|
|
||||||
|
# The old worker primary is assigned as a secondary.
|
||||||
|
2024-08-26 07:02:40,084 DEBUG: query(SELECT pg_catalog.citus_update_node(%s, %s, %s, true, %s), (7, '172.19.0.7', 5432, 10000))
|
||||||
|
|
||||||
# The future worker primary notifies the coordinator that it acquired the leader lock in DCS and about to run "pg_ctl promote".
|
# The future worker primary notifies the coordinator that it acquired the leader lock in DCS and about to run "pg_ctl promote".
|
||||||
2022-12-22 07:02:40,085 DEBUG: query("SELECT pg_catalog.citus_update_node(3, '172.27.0.5', 5432)")
|
2024-08-26 07:02:40,085 DEBUG: query(SELECT pg_catalog.citus_update_node(%s, %s, %s, true, %s), (3, '172.19.0.5', 5432, 10000))
|
||||||
|
|
||||||
# The new worker primary just finished promote and notifies coordinator that it is ready to accept read-write traffic.
|
# The new worker primary just finished promote and notifies coordinator that it is ready to accept read-write traffic.
|
||||||
2022-12-22 07:02:41,485 DEBUG: query("COMMIT")
|
2024-08-26 07:02:41,485 DEBUG: query(COMMIT, ())
|
||||||
# From this moment the application traffic on the coordinator to the worker group 2 is unblocked.
|
# From this moment the application traffic on the coordinator to the worker group 2 is unblocked.
|
||||||
|
|
||||||
|
Secondary nodes
|
||||||
|
---------------
|
||||||
|
|
||||||
|
Starting from Patroni v4.0.0 Citus secondary nodes without ``noloadbalance`` :ref:`tag <tags_settings>` are also registered in ``pg_dist_node``.
|
||||||
|
However, to use secondary nodes for read-only queries applications need to change `citus.use_secondary_nodes <https://docs.citusdata.com/en/latest/develop/api_guc.html#citus-use-secondary-nodes-enum>`__ GUC.
|
||||||
|
|
||||||
Peek into DCS
|
Peek into DCS
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
@@ -335,7 +346,7 @@ new Kubernetes objects ConfigMaps or Endpoints, it automatically puts the
|
|||||||
You can find a complete example of Patroni deployment on Kubernetes with Citus
|
You can find a complete example of Patroni deployment on Kubernetes with Citus
|
||||||
support in the `kubernetes`__ folder of the Patroni repository.
|
support in the `kubernetes`__ folder of the Patroni repository.
|
||||||
|
|
||||||
__ https://github.com/zalando/patroni/tree/master/kubernetes
|
__ https://github.com/patroni/patroni/tree/master/kubernetes
|
||||||
|
|
||||||
There are two important files for you:
|
There are two important files for you:
|
||||||
|
|
||||||
|
|||||||
+38
-14
@@ -21,6 +21,8 @@ import os
|
|||||||
|
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
|
from sphinx.application import ENV_PICKLE_FILENAME
|
||||||
|
|
||||||
sys.path.insert(0, os.path.abspath('..'))
|
sys.path.insert(0, os.path.abspath('..'))
|
||||||
|
|
||||||
from patroni.version import __version__
|
from patroni.version import __version__
|
||||||
@@ -75,8 +77,8 @@ master_doc = 'index'
|
|||||||
|
|
||||||
# General information about the project.
|
# General information about the project.
|
||||||
project = 'Patroni'
|
project = 'Patroni'
|
||||||
copyright = '2015 Compose, Zalando SE'
|
copyright = '2025 Compose, Zalando SE, Patroni Contributors'
|
||||||
author = 'Zalando SE'
|
author = 'Patroni Contributors'
|
||||||
|
|
||||||
# The version info for the project you're documenting, acts as replacement for
|
# The version info for the project you're documenting, acts as replacement for
|
||||||
# |version| and |release|, also used in various other places throughout the
|
# |version| and |release|, also used in various other places throughout the
|
||||||
@@ -114,9 +116,6 @@ todo_include_todos = True
|
|||||||
|
|
||||||
html_theme = 'sphinx_rtd_theme'
|
html_theme = 'sphinx_rtd_theme'
|
||||||
on_rtd = os.environ.get('READTHEDOCS', None) == 'True'
|
on_rtd = os.environ.get('READTHEDOCS', None) == 'True'
|
||||||
if not on_rtd: # only import and set the theme if we're building docs locally
|
|
||||||
import sphinx_rtd_theme
|
|
||||||
html_theme_path = [sphinx_rtd_theme.get_html_theme_path()]
|
|
||||||
|
|
||||||
# Theme options are theme-specific and customize the look and feel of a theme
|
# Theme options are theme-specific and customize the look and feel of a theme
|
||||||
# further. For a list of options available for each theme, see the
|
# further. For a list of options available for each theme, see the
|
||||||
@@ -132,7 +131,7 @@ html_static_path = ['_static']
|
|||||||
# Replace "source" links with "edit on GitHub" when using rtd theme
|
# Replace "source" links with "edit on GitHub" when using rtd theme
|
||||||
html_context = {
|
html_context = {
|
||||||
'display_github': True,
|
'display_github': True,
|
||||||
'github_user': 'zalando',
|
'github_user': 'patroni',
|
||||||
'github_repo': 'patroni',
|
'github_repo': 'patroni',
|
||||||
'github_version': 'master',
|
'github_version': 'master',
|
||||||
'conf_py_path': '/docs/',
|
'conf_py_path': '/docs/',
|
||||||
@@ -189,7 +188,7 @@ latex_elements = {
|
|||||||
# author, documentclass [howto, manual, or own class]).
|
# author, documentclass [howto, manual, or own class]).
|
||||||
latex_documents = [
|
latex_documents = [
|
||||||
(master_doc, 'Patroni.tex', 'Patroni Documentation',
|
(master_doc, 'Patroni.tex', 'Patroni Documentation',
|
||||||
'Zalando SE', 'manual'),
|
'Patroni Contributors', 'manual'),
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
@@ -243,13 +242,24 @@ intersphinx_mapping = {'python': ('https://docs.python.org/', None)}
|
|||||||
# Remove these pages from index, references, toc trees, etc.
|
# Remove these pages from index, references, toc trees, etc.
|
||||||
# If the builder is not 'html' then add the API docs modules index to pages to be removed.
|
# If the builder is not 'html' then add the API docs modules index to pages to be removed.
|
||||||
exclude_from_builder = {
|
exclude_from_builder = {
|
||||||
'latex': ['modules/modules'],
|
'latex': ['modules/'],
|
||||||
'epub': ['modules/modules'],
|
'epub': ['modules/'],
|
||||||
}
|
}
|
||||||
# Internal holding list, anything added here will always be excluded
|
# Internal holding list, anything added here will always be excluded
|
||||||
_docs_to_remove = []
|
_docs_to_remove = []
|
||||||
|
|
||||||
|
|
||||||
|
def config_inited(app, config):
|
||||||
|
"""Run during Sphinx `config-inited` phase.
|
||||||
|
|
||||||
|
rtd reuses the environment, and there is no way to customize this behavior.
|
||||||
|
Thus we remove the saved env.
|
||||||
|
"""
|
||||||
|
pickle_file = os.path.join(app.doctreedir, ENV_PICKLE_FILENAME)
|
||||||
|
if on_rtd and os.path.exists(pickle_file):
|
||||||
|
os.remove(pickle_file)
|
||||||
|
|
||||||
|
|
||||||
def builder_inited(app):
|
def builder_inited(app):
|
||||||
"""Run during Sphinx `builder-inited` phase.
|
"""Run during Sphinx `builder-inited` phase.
|
||||||
|
|
||||||
@@ -263,14 +273,28 @@ def builder_inited(app):
|
|||||||
_docs_to_remove.extend(exclude_from_builder[app.builder.name])
|
_docs_to_remove.extend(exclude_from_builder[app.builder.name])
|
||||||
|
|
||||||
|
|
||||||
|
def _to_be_removed(doc):
|
||||||
|
for remove in _docs_to_remove:
|
||||||
|
if doc.startswith(remove):
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
def env_get_outdated(app, env, added, changed, removed):
|
def env_get_outdated(app, env, added, changed, removed):
|
||||||
"""Run during Sphinx `env-get-outdated` phase.
|
"""Run during Sphinx `env-get-outdated` phase.
|
||||||
|
|
||||||
Remove the items listed in `docs_to_remove` from known pages.
|
Remove the items listed in `docs_to_remove` from known pages.
|
||||||
"""
|
"""
|
||||||
added.difference_update(_docs_to_remove)
|
to_remove = set()
|
||||||
changed.difference_update(_docs_to_remove)
|
if hasattr(env, 'found_docs'):
|
||||||
removed.update(_docs_to_remove)
|
for doc in env.found_docs:
|
||||||
|
if _to_be_removed(doc):
|
||||||
|
to_remove.add(doc)
|
||||||
|
added.difference_update(to_remove)
|
||||||
|
changed.difference_update(to_remove)
|
||||||
|
removed.update(to_remove)
|
||||||
|
if hasattr(env, 'project'):
|
||||||
|
env.project.docnames.difference_update(to_remove)
|
||||||
return []
|
return []
|
||||||
|
|
||||||
|
|
||||||
@@ -282,8 +306,7 @@ def doctree_read(app, doctree):
|
|||||||
from sphinx import addnodes
|
from sphinx import addnodes
|
||||||
for toc_tree_node in doctree.traverse(addnodes.toctree):
|
for toc_tree_node in doctree.traverse(addnodes.toctree):
|
||||||
for e in toc_tree_node['entries']:
|
for e in toc_tree_node['entries']:
|
||||||
ref = str(e[1])
|
if _to_be_removed(str(e[1])):
|
||||||
if ref in _docs_to_remove:
|
|
||||||
toc_tree_node['entries'].remove(e)
|
toc_tree_node['entries'].remove(e)
|
||||||
|
|
||||||
|
|
||||||
@@ -304,6 +327,7 @@ def setup(app):
|
|||||||
app.add_stylesheet('custom.css')
|
app.add_stylesheet('custom.css')
|
||||||
|
|
||||||
# Run extra steps to remove module docs when running with a non-html builder
|
# Run extra steps to remove module docs when running with a non-html builder
|
||||||
|
app.connect('config-inited', config_inited)
|
||||||
app.connect('builder-inited', builder_inited)
|
app.connect('builder-inited', builder_inited)
|
||||||
app.connect('env-get-outdated', env_get_outdated)
|
app.connect('env-get-outdated', env_get_outdated)
|
||||||
app.connect('doctree-read', doctree_read)
|
app.connect('doctree-read', doctree_read)
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ Reporting bugs
|
|||||||
--------------
|
--------------
|
||||||
|
|
||||||
Before reporting a bug please make sure to **reproduce it with the latest Patroni version**!
|
Before reporting a bug please make sure to **reproduce it with the latest Patroni version**!
|
||||||
Also please double check if the issue already exists in our `Issues Tracker <https://github.com/zalando/patroni/issues>`__.
|
Also please double check if the issue already exists in our `Issues Tracker <https://github.com/patroni/patroni/issues>`__.
|
||||||
|
|
||||||
Running tests
|
Running tests
|
||||||
-------------
|
-------------
|
||||||
|
|||||||
@@ -21,19 +21,20 @@ In order to change the dynamic configuration you can use either :ref:`patronictl
|
|||||||
|
|
||||||
|
|
||||||
- **maximum\_lag\_on\_failover**: the maximum bytes a follower may lag to be able to participate in leader election.
|
- **maximum\_lag\_on\_failover**: the maximum bytes a follower may lag to be able to participate in leader election.
|
||||||
- **maximum\_lag\_on\_syncnode**: the maximum bytes a synchronous follower may lag before it is considered as an unhealthy candidate and swapped by healthy asynchronous follower. Patroni utilize the max replica lsn if there is more than one follower, otherwise it will use leader's current wal lsn. Default is -1, Patroni will not take action to swap synchronous unhealthy follower when the value is set to 0 or below. Please set the value high enough so Patroni won't swap synchrounous follower fequently during high transaction volume.
|
- **maximum\_lag\_on\_syncnode**: the maximum bytes a synchronous follower may lag before it is considered as an unhealthy candidate and swapped by healthy asynchronous follower. Patroni utilize the max replica lsn if there is more than one follower, otherwise it will use leader's current wal lsn. Default is -1, Patroni will not take action to swap synchronous unhealthy follower when the value is set to 0 or below. Please set the value high enough so Patroni won't swap synchrounous follower frequently during high transaction volume.
|
||||||
- **max\_timelines\_history**: maximum number of timeline history items kept in DCS. Default value: 0. When set to 0, it keeps the full history in DCS.
|
- **max\_timelines\_history**: maximum number of timeline history items kept in DCS. Default value: 0. When set to 0, it keeps the full history in DCS.
|
||||||
- **primary\_start\_timeout**: the amount of time a primary is allowed to recover from failures before failover is triggered (in seconds). Default is 300 seconds. When set to 0 failover is done immediately after a crash is detected if possible. When using asynchronous replication a failover can cause lost transactions. Worst case failover time for primary failure is: loop\_wait + primary\_start\_timeout + loop\_wait, unless primary\_start\_timeout is zero, in which case it's just loop\_wait. Set the value according to your durability/availability tradeoff.
|
- **primary\_start\_timeout**: the amount of time a primary is allowed to recover from failures before failover is triggered (in seconds). Default is 300 seconds. When set to 0 failover is done immediately after a crash is detected if possible. When using asynchronous replication a failover can cause lost transactions. Worst case failover time for primary failure is: loop\_wait + primary\_start\_timeout + loop\_wait, unless primary\_start\_timeout is zero, in which case it's just loop\_wait. Set the value according to your durability/availability tradeoff.
|
||||||
- **primary\_stop\_timeout**: The number of seconds Patroni is allowed to wait when stopping Postgres and effective only when synchronous_mode is enabled. When set to > 0 and the synchronous_mode is enabled, Patroni sends SIGKILL to the postmaster if the stop operation is running for more than the value set by primary\_stop\_timeout. Set the value according to your durability/availability tradeoff. If the parameter is not set or set <= 0, primary\_stop\_timeout does not apply.
|
- **primary\_stop\_timeout**: The number of seconds Patroni is allowed to wait when stopping Postgres and effective only when synchronous_mode is enabled. When set to > 0 and the synchronous_mode is enabled, Patroni sends SIGKILL to the postmaster if the stop operation is running for more than the value set by primary\_stop\_timeout. Set the value according to your durability/availability tradeoff. If the parameter is not set or set <= 0, primary\_stop\_timeout does not apply.
|
||||||
- **synchronous\_mode**: turns on synchronous replication mode. In this mode a replica will be chosen as synchronous and only the latest leader and synchronous replica are able to participate in leader election. Synchronous mode makes sure that successfully committed transactions will not be lost at failover, at the cost of losing availability for writes when Patroni cannot ensure transaction durability. See :ref:`replication modes documentation <replication_modes>` for details.
|
- **synchronous\_mode**: turns on synchronous replication mode. Possible values: ``off``, ``on``, ``quorum``. In this mode the leader takes care of management of ``synchronous_standby_names``, and only the last known leader, or one of synchronous replicas, are allowed to participate in leader race. Synchronous mode makes sure that successfully committed transactions will not be lost at failover, at the cost of losing availability for writes when Patroni cannot ensure transaction durability. See :ref:`replication modes documentation <replication_modes>` for details.
|
||||||
- **synchronous\_mode\_strict**: prevents disabling synchronous replication if no synchronous replicas are available, blocking all client writes to the primary. See :ref:`replication modes documentation <replication_modes>` for details.
|
- **synchronous\_mode\_strict**: prevents disabling synchronous replication if no synchronous replicas are available, blocking all client writes to the primary. See :ref:`replication modes documentation <replication_modes>` for details.
|
||||||
|
- **synchronous\_node\_count**: if ``synchronous_mode`` is enabled, this parameter is used by Patroni to manage the precise number of synchronous standby instances and adjusts the state in DCS and the ``synchronous_standby_names`` parameter in PostgreSQL as members join and leave. If the parameter is set to a value higher than the number of eligible nodes, it will be automatically adjusted. Defaults to ``1``.
|
||||||
- **failsafe\_mode**: Enables :ref:`DCS Failsafe Mode <dcs_failsafe_mode>`. Defaults to `false`.
|
- **failsafe\_mode**: Enables :ref:`DCS Failsafe Mode <dcs_failsafe_mode>`. Defaults to `false`.
|
||||||
- **postgresql**:
|
- **postgresql**:
|
||||||
|
|
||||||
- **use\_pg\_rewind**: whether or not to use pg_rewind. Defaults to `false`.
|
- **use\_pg\_rewind**: whether or not to use pg_rewind. Defaults to `false`. Note that either the cluster must be initialized with ``data page checksums`` (``--data-checksums`` option for ``initdb``) and/or ``wal_log_hints`` must be set to ``on``, or ``pg_rewind`` will not work.
|
||||||
- **use\_slots**: whether or not to use replication slots. Defaults to `true` on PostgreSQL 9.4+.
|
- **use\_slots**: whether or not to use replication slots. Defaults to `true` on PostgreSQL 9.4+.
|
||||||
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower. There is no recovery.conf anymore in PostgreSQL 12, but you may continue using this section, because Patroni handles it transparently.
|
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower. There is no recovery.conf anymore in PostgreSQL 12, but you may continue using this section, because Patroni handles it transparently.
|
||||||
- **parameters**: list of configuration settings for Postgres.
|
- **parameters**: configuration parameters (GUCs) for Postgres in format ``{max_connections: 100, wal_level: "replica", max_wal_senders: 10, wal_log_hints: "on"}``. Many of these are required for replication to work.
|
||||||
|
|
||||||
- **pg\_hba**: list of lines that Patroni will use to generate ``pg_hba.conf``. Patroni ignores this parameter if ``hba_file`` PostgreSQL parameter is set to a non-default value.
|
- **pg\_hba**: list of lines that Patroni will use to generate ``pg_hba.conf``. Patroni ignores this parameter if ``hba_file`` PostgreSQL parameter is set to a non-default value.
|
||||||
|
|
||||||
@@ -55,13 +56,15 @@ In order to change the dynamic configuration you can use either :ref:`patronictl
|
|||||||
- **archive\_cleanup\_command**: cleanup command for standby leader
|
- **archive\_cleanup\_command**: cleanup command for standby leader
|
||||||
- **recovery\_min\_apply\_delay**: how long to wait before actually apply WAL records on a standby leader
|
- **recovery\_min\_apply\_delay**: how long to wait before actually apply WAL records on a standby leader
|
||||||
|
|
||||||
|
- **member_slots_ttl**: retention time of physical replication slots for replicas when they are shut down. Default value: `30min`. Set it to `0` if you want to keep the old behavior (when the member key expires from DCS, the slot is immediately removed). The feature works only starting from PostgreSQL 11.
|
||||||
- **slots**: define permanent replication slots. These slots will be preserved during switchover/failover. Permanent slots that don't exist will be created by Patroni. With PostgreSQL 11 onwards permanent physical slots are created on all nodes and their position is advanced every **loop_wait** seconds. For PostgreSQL versions older than 11 permanent physical replication slots are maintained only on the current primary. The logical slots are copied from the primary to a standby with restart, and after that their position advanced every **loop_wait** seconds (if necessary). Copying logical slot files performed via ``libpq`` connection and using either rewind or superuser credentials (see **postgresql.authentication** section). There is always a chance that the logical slot position on the replica is a bit behind the former primary, therefore application should be prepared that some messages could be received the second time after the failover. The easiest way of doing so - tracking ``confirmed_flush_lsn``. Enabling permanent replication slots requires **postgresql.use_slots** to be set to ``true``. If there are permanent logical replication slots defined Patroni will automatically enable the ``hot_standby_feedback``. Since the failover of logical replication slots is unsafe on PostgreSQL 9.6 and older and PostgreSQL version 10 is missing some important functions, the feature only works with PostgreSQL 11+.
|
- **slots**: define permanent replication slots. These slots will be preserved during switchover/failover. Permanent slots that don't exist will be created by Patroni. With PostgreSQL 11 onwards permanent physical slots are created on all nodes and their position is advanced every **loop_wait** seconds. For PostgreSQL versions older than 11 permanent physical replication slots are maintained only on the current primary. The logical slots are copied from the primary to a standby with restart, and after that their position advanced every **loop_wait** seconds (if necessary). Copying logical slot files performed via ``libpq`` connection and using either rewind or superuser credentials (see **postgresql.authentication** section). There is always a chance that the logical slot position on the replica is a bit behind the former primary, therefore application should be prepared that some messages could be received the second time after the failover. The easiest way of doing so - tracking ``confirmed_flush_lsn``. Enabling permanent replication slots requires **postgresql.use_slots** to be set to ``true``. If there are permanent logical replication slots defined Patroni will automatically enable the ``hot_standby_feedback``. Since the failover of logical replication slots is unsafe on PostgreSQL 9.6 and older and PostgreSQL version 10 is missing some important functions, the feature only works with PostgreSQL 11+.
|
||||||
|
|
||||||
- **my\_slot\_name**: the name of the permanent replication slot. If the permanent slot name matches with the name of the current node it will not be created on this node. If you add a permanent physical replication slot which name matches the name of a Patroni member, Patroni will ensure that the slot that was created is not removed even if the corresponding member becomes unresponsive, situation which would normally result in the slot's removal by Patroni. Although this can be useful in some situations, such as when you want replication slots used by members to persist during temporary failures or when importing existing members to a new Patroni cluster (see :ref:`Convert a Standalone to a Patroni Cluster <existing_data>` for details), caution should be exercised by the operator that these clashes in names are not persisted in the DCS, when the slot is no longer required, due to its effect on normal functioning of Patroni.
|
- **my\_slot\_name**: the name of the permanent replication slot. If the permanent slot name matches with the name of the current node it will not be created on this node. If you add a permanent physical replication slot which name matches the name of a Patroni member, Patroni will ensure that the slot that was created is not removed even if the corresponding member becomes unresponsive, situation which would normally result in the slot's removal by Patroni. Although this can be useful in some situations, such as when you want replication slots used by members to persist during temporary failures or when importing existing members to a new Patroni cluster (see :ref:`Convert a Standalone to a Patroni Cluster <existing_data>` for details), caution should be exercised by the operator that these clashes in names are not persisted in the DCS, when the slot is no longer required, due to its effect on normal functioning of Patroni.
|
||||||
|
|
||||||
- **type**: slot type. Could be ``physical`` or ``logical``. If the slot is logical, you have to additionally define ``database`` and ``plugin``.
|
- **type**: slot type. Could be ``physical`` or ``logical``. If the slot is logical, you have to additionally define ``database`` and ``plugin``. If the slot is physical, you can optionally define ``cluster_type``.
|
||||||
- **database**: the database name where logical slots should be created.
|
- **database**: the database name where logical slots should be created.
|
||||||
- **plugin**: the plugin name for the logical slot.
|
- **plugin**: the plugin name for the logical slot.
|
||||||
|
- **cluster_type**: the type of cluster (``primary`` or ``standby``) the slot should only be created on, otherwise it will not be created or an already existing slot will be dropped.
|
||||||
|
|
||||||
- **ignore\_slots**: list of sets of replication slot properties for which Patroni should ignore matching slots. This configuration/feature/etc. is useful when some replication slots are managed outside of Patroni. Any subset of matching properties will cause a slot to be ignored.
|
- **ignore\_slots**: list of sets of replication slot properties for which Patroni should ignore matching slots. This configuration/feature/etc. is useful when some replication slots are managed outside of Patroni. Any subset of matching properties will cause a slot to be ignored.
|
||||||
|
|
||||||
@@ -91,7 +94,7 @@ Note: **slots** is a hashmap while **ignore_slots** is an array. For example:
|
|||||||
type: physical
|
type: physical
|
||||||
...
|
...
|
||||||
|
|
||||||
Note: if cluster topology is static (fixed number of nodes that never change their names) you can configure permanent physical replication slots with names corresponding to names of nodes to avoid recycling of WAL files while replica is temporary down:
|
Note: When running PostgreSQL v11 or newer Patroni maintains physical replication slots on all nodes that could potentially become a leader, so that replica nodes keep WAL segments reserved if they are potentially required by other nodes. In case the node is absent and its member key in DCS gets expired, the corresponding replication slot is dropped after ``member_slots_ttl`` (default value is `30min`). You can increase or decrease retention based on your needs. Alternatively, if your cluster topology is static (fixed number of nodes that never change their names) you can configure permanent physical replication slots with names corresponding to the names of the nodes to avoid slots removal and recycling of WAL files while replica is temporarily down:
|
||||||
|
|
||||||
.. code:: YAML
|
.. code:: YAML
|
||||||
|
|
||||||
@@ -107,4 +110,8 @@ Note: if cluster topology is static (fixed number of nodes that never change the
|
|||||||
|
|
||||||
.. warning::
|
.. warning::
|
||||||
Permanent replication slots are synchronized only from the ``primary``/``standby_leader`` to replica nodes. That means, applications are supposed to be using them only from the leader node. Using them on replica nodes will cause indefinite growth of ``pg_wal`` on all other nodes in the cluster.
|
Permanent replication slots are synchronized only from the ``primary``/``standby_leader`` to replica nodes. That means, applications are supposed to be using them only from the leader node. Using them on replica nodes will cause indefinite growth of ``pg_wal`` on all other nodes in the cluster.
|
||||||
An exception to that rule are permanent physical slots that match the Patroni member names, if you happen to configure any. Those will be synchronized among all nodes as they are used for replication among them.
|
An exception to that rule are physical slots that match the Patroni member names (created and maintained by Patroni). Those will be synchronized among all nodes as they are used for replication among them.
|
||||||
|
|
||||||
|
|
||||||
|
.. warning::
|
||||||
|
Setting ``nostream`` tag on standby disables copying and synchronization of permanent logical replication slots on the node itself and all its cascading replicas if any.
|
||||||
|
|||||||
+5
-2
@@ -181,13 +181,16 @@ What is the difference between ``etcd`` and ``etcd3`` in Patroni configuration?
|
|||||||
* API version 2 will be completely removed on Etcd v3.6.
|
* API version 2 will be completely removed on Etcd v3.6.
|
||||||
|
|
||||||
I have ``use_slots`` enabled in my Patroni configuration, but when a cluster member goes offline for some time, the replication slot used by that member is dropped on the upstream node. What can I do to avoid that issue?
|
I have ``use_slots`` enabled in my Patroni configuration, but when a cluster member goes offline for some time, the replication slot used by that member is dropped on the upstream node. What can I do to avoid that issue?
|
||||||
You can configure a permanent physical replication slot for the members.
|
There are two options:
|
||||||
|
|
||||||
|
1. You can tune ``member_slots_ttl`` (default value ``30min``, available since Patroni ``4.0.0`` and PostgreSQL 11 onwards) and replication slots for absent members will not be removed when the members downtime is shorter than the configured threshold.
|
||||||
|
2. You can configure permanent physical replication slots for the members.
|
||||||
|
|
||||||
Since Patroni ``3.2.0`` it is now possible to have member slots as permanent slots managed by Patroni.
|
Since Patroni ``3.2.0`` it is now possible to have member slots as permanent slots managed by Patroni.
|
||||||
|
|
||||||
Patroni will create the permanent physical slots on all nodes, and make sure to not remove the slots, as well as to advance the slots' LSN on all nodes according to the LSN that has been consumed by the member.
|
Patroni will create the permanent physical slots on all nodes, and make sure to not remove the slots, as well as to advance the slots' LSN on all nodes according to the LSN that has been consumed by the member.
|
||||||
|
|
||||||
Later, if you decide to remove the corresponding member, it's **your responsability** to adjust the permanent slots configuration, otherwise Patroni will keep the slots around forever.
|
Later, if you decide to remove the corresponding member, it's **your responsibility** to adjust the permanent slots configuration, otherwise Patroni will keep the slots around forever.
|
||||||
|
|
||||||
**Note:** on Patroni older than ``3.2.0`` you could still have member slots configured as permanent physical slots, however they would be managed only on the current leader. That is, in case of failover/switchover these slots would be created on the new leader, but that wouldn't guarantee that it had all WAL segments for the absent node.
|
**Note:** on Patroni older than ``3.2.0`` you could still have member slots configured as permanent physical slots, however they would be managed only on the current leader. That is, in case of failover/switchover these slots would be created on the new leader, but that wouldn't guarantee that it had all WAL segments for the absent node.
|
||||||
|
|
||||||
|
|||||||
@@ -109,7 +109,7 @@ digraph G {
|
|||||||
subgraph cluster_process_healthy_cluster {
|
subgraph cluster_process_healthy_cluster {
|
||||||
label = "process_healthy_cluster"
|
label = "process_healthy_cluster"
|
||||||
"healthy_has_lock" [label="Am I the owner of the leader lock?", shape=diamond]
|
"healthy_has_lock" [label="Am I the owner of the leader lock?", shape=diamond]
|
||||||
"healthy_is_leader" [label="Is Postgres running as master?", shape=diamond]
|
"healthy_is_leader" [label="Is Postgres running as primary?", shape=diamond]
|
||||||
"healthy_no_lock" [label="Follow the leader (async,\ncreate/update recovery.conf and restart if necessary)"]
|
"healthy_no_lock" [label="Follow the leader (async,\ncreate/update recovery.conf and restart if necessary)"]
|
||||||
"healthy_has_lock" -> "healthy_no_lock" [label="no" color="red"]
|
"healthy_has_lock" -> "healthy_no_lock" [label="no" color="red"]
|
||||||
"healthy_has_lock" -> "healthy_update_leader_lock" [label="yes" color="green"]
|
"healthy_has_lock" -> "healthy_update_leader_lock" [label="yes" color="green"]
|
||||||
@@ -119,7 +119,7 @@ digraph G {
|
|||||||
"healthy_update_success" -> "healthy_is_leader" [label="yes" color="green"]
|
"healthy_update_success" -> "healthy_is_leader" [label="yes" color="green"]
|
||||||
"healthy_update_success" -> "healthy_demote" [label="no" color="red"]
|
"healthy_update_success" -> "healthy_demote" [label="no" color="red"]
|
||||||
"healthy_demote" [label="Demote (async,\nrestart in read-only)"]
|
"healthy_demote" [label="Demote (async,\nrestart in read-only)"]
|
||||||
"healthy_failover" [label="Promote Postgres to master"]
|
"healthy_failover" [label="Promote Postgres to primary"]
|
||||||
"healthy_is_leader" -> "healthy_failover" [label="no" color="red"]
|
"healthy_is_leader" -> "healthy_failover" [label="no" color="red"]
|
||||||
}
|
}
|
||||||
"healthy_demote" -> "update_member"
|
"healthy_demote" -> "update_member"
|
||||||
@@ -134,10 +134,10 @@ digraph G {
|
|||||||
"unhealthy_leader_race" [label="Try to create leader key"]
|
"unhealthy_leader_race" [label="Try to create leader key"]
|
||||||
"unhealthy_leader_race" -> "unhealthy_acquire_lock"
|
"unhealthy_leader_race" -> "unhealthy_acquire_lock"
|
||||||
"unhealthy_acquire_lock" [label="Was I able to get the lock?", shape="diamond"]
|
"unhealthy_acquire_lock" [label="Was I able to get the lock?", shape="diamond"]
|
||||||
"unhealthy_is_leader" [label="Is Postgres running as master?", shape=diamond]
|
"unhealthy_is_leader" [label="Is Postgres running as primary?", shape=diamond]
|
||||||
"unhealthy_acquire_lock" -> "unhealthy_is_leader" [label="yes" color="green"]
|
"unhealthy_acquire_lock" -> "unhealthy_is_leader" [label="yes" color="green"]
|
||||||
"unhealthy_is_leader" -> "unhealthy_promote" [label="no" color="red"]
|
"unhealthy_is_leader" -> "unhealthy_promote" [label="no" color="red"]
|
||||||
"unhealthy_promote" [label="Promote to master"]
|
"unhealthy_promote" [label="Promote to primary"]
|
||||||
"unhealthy_is_healthiest" -> "unhealthy_follow" [label="no" color="red"]
|
"unhealthy_is_healthiest" -> "unhealthy_follow" [label="no" color="red"]
|
||||||
"unhealthy_follow" [label="try to follow somebody else()"]
|
"unhealthy_follow" [label="try to follow somebody else()"]
|
||||||
"unhealthy_acquire_lock" -> "unhealthy_follow" [label="no" color="red"]
|
"unhealthy_acquire_lock" -> "unhealthy_follow" [label="no" color="red"]
|
||||||
|
|||||||
Binary file not shown.
|
Before Width: | Height: | Size: 507 KiB After Width: | Height: | Size: 524 KiB |
@@ -44,7 +44,7 @@ You should not use ``pg_ctl promote`` in this scenario, you need "manually promo
|
|||||||
|
|
||||||
In case you want to return to the "initial" state, there are only two ways of resolving it:
|
In case you want to return to the "initial" state, there are only two ways of resolving it:
|
||||||
|
|
||||||
- Add the standby_cluster section back and it will trigger pg_rewind, but there are chances that pg_rewind will fail.
|
- Add the standby_cluster section back and it will trigger ``pg_rewind``; however, for ``pg_rewind`` to function properly, either the cluster must be initialized with ``data page checksums`` (``--data-checksums`` option for ``initdb``) and/or ``wal_log_hints`` must be set to ``on``, but there are still chances that ``pg_rewind`` might fail due to other factors.
|
||||||
- Rebuild the standby cluster from scratch.
|
- Rebuild the standby cluster from scratch.
|
||||||
|
|
||||||
Before promoting standby cluster one have to manually ensure that the source cluster is down (STONITH). When DC1 recovers, the cluster has to be converted to a standby cluster.
|
Before promoting standby cluster one have to manually ensure that the source cluster is down (STONITH). When DC1 recovers, the cluster has to be converted to a standby cluster.
|
||||||
|
|||||||
+3
-1
@@ -10,7 +10,7 @@ Patroni is a template for high availability (HA) PostgreSQL solutions using Pyth
|
|||||||
|
|
||||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
|
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
|
||||||
|
|
||||||
Currently supported PostgreSQL versions: 9.3 to 16.
|
Currently supported PostgreSQL versions: 9.3 to 17.
|
||||||
|
|
||||||
**Note to Citus users**: Starting from 3.0 Patroni nicely integrates with the `Citus <https://github.com/citusdata/citus>`__ database extension to Postgres. Please check the :ref:`Citus support page <citus>` in the Patroni documentation for more info about how to use Patroni high availability together with a Citus distributed cluster.
|
**Note to Citus users**: Starting from 3.0 Patroni nicely integrates with the `Citus <https://github.com/citusdata/citus>`__ database extension to Postgres. Please check the :ref:`Citus support page <citus>` in the Patroni documentation for more info about how to use Patroni high availability together with a Citus distributed cluster.
|
||||||
|
|
||||||
@@ -28,12 +28,14 @@ Currently supported PostgreSQL versions: 9.3 to 16.
|
|||||||
patronictl
|
patronictl
|
||||||
replica_bootstrap
|
replica_bootstrap
|
||||||
replication_modes
|
replication_modes
|
||||||
|
standby_cluster
|
||||||
watchdog
|
watchdog
|
||||||
pause
|
pause
|
||||||
dcs_failsafe_mode
|
dcs_failsafe_mode
|
||||||
kubernetes
|
kubernetes
|
||||||
citus
|
citus
|
||||||
existing_data
|
existing_data
|
||||||
|
tools_integration
|
||||||
security
|
security
|
||||||
ha_multi_dc
|
ha_multi_dc
|
||||||
faq
|
faq
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ where ``dependencies`` can be either empty, or consist of one or more of the fol
|
|||||||
etcd or etcd3
|
etcd or etcd3
|
||||||
`python-etcd` module in order to use Etcd as Distributed Configuration Store (DCS)
|
`python-etcd` module in order to use Etcd as Distributed Configuration Store (DCS)
|
||||||
consul
|
consul
|
||||||
`python-consul` module in order to use Consul as DCS
|
`py-consul` module in order to use Consul as DCS
|
||||||
zookeeper
|
zookeeper
|
||||||
`kazoo` module in order to use Zookeeper as DCS
|
`kazoo` module in order to use Zookeeper as DCS
|
||||||
exhibitor
|
exhibitor
|
||||||
@@ -62,6 +62,8 @@ aws
|
|||||||
`boto3` in order to use AWS callbacks
|
`boto3` in order to use AWS callbacks
|
||||||
jsonlogger
|
jsonlogger
|
||||||
`python-json-logger` module in order to enable :ref:`logging <log_settings>` in json format
|
`python-json-logger` module in order to enable :ref:`logging <log_settings>` in json format
|
||||||
|
systemd
|
||||||
|
`systemd-python` in order to use sd_notify integration
|
||||||
all
|
all
|
||||||
all of the above (except psycopg family)
|
all of the above (except psycopg family)
|
||||||
psycopg
|
psycopg
|
||||||
|
|||||||
+7
-7
@@ -37,7 +37,7 @@ Patroni Kubernetes :ref:`settings <kubernetes_settings>` and :ref:`environment v
|
|||||||
Customize role label
|
Customize role label
|
||||||
^^^^^^^^^^^^^^^^^^^^
|
^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
By default, Patroni will set corresponding labels on the pod it runs in based on node's role, such as ``role=master``.
|
By default, Patroni will set corresponding labels on the pod it runs in based on node's role, such as ``role=primary``.
|
||||||
The key and value of label can be customized by `kubernetes.role_label`, `kubernetes.leader_label_value`, `kubernetes.follower_label_value` and `kubernetes.standby_leader_label_value`.
|
The key and value of label can be customized by `kubernetes.role_label`, `kubernetes.leader_label_value`, `kubernetes.follower_label_value` and `kubernetes.standby_leader_label_value`.
|
||||||
|
|
||||||
Note that if you migrate from default role labels to custom ones, you can reduce downtime by following migration steps:
|
Note that if you migrate from default role labels to custom ones, you can reduce downtime by following migration steps:
|
||||||
@@ -48,8 +48,8 @@ Note that if you migrate from default role labels to custom ones, you can reduce
|
|||||||
|
|
||||||
labels:
|
labels:
|
||||||
cluster-name: foo
|
cluster-name: foo
|
||||||
role: master
|
role: primary
|
||||||
tmp_role: master
|
tmp_role: primary
|
||||||
|
|
||||||
2. After all pods have been updated, modify the service selector to select the temporary label.
|
2. After all pods have been updated, modify the service selector to select the temporary label.
|
||||||
|
|
||||||
@@ -57,7 +57,7 @@ Note that if you migrate from default role labels to custom ones, you can reduce
|
|||||||
|
|
||||||
selector:
|
selector:
|
||||||
cluster-name: foo
|
cluster-name: foo
|
||||||
tmp_role: master
|
tmp_role: primary
|
||||||
|
|
||||||
3. Add your custom role label (e.g., set `kubernetes.leader_label_value=primary`). Once pods are restarted they will get following new labels set by Patroni:
|
3. Add your custom role label (e.g., set `kubernetes.leader_label_value=primary`). Once pods are restarted they will get following new labels set by Patroni:
|
||||||
|
|
||||||
@@ -66,7 +66,7 @@ Note that if you migrate from default role labels to custom ones, you can reduce
|
|||||||
labels:
|
labels:
|
||||||
cluster-name: foo
|
cluster-name: foo
|
||||||
role: primary
|
role: primary
|
||||||
tmp_role: master
|
tmp_role: primary
|
||||||
|
|
||||||
4. After all pods have been updated again, modify the service selector to use new role value.
|
4. After all pods have been updated again, modify the service selector to use new role value.
|
||||||
|
|
||||||
@@ -87,7 +87,7 @@ Note that if you migrate from default role labels to custom ones, you can reduce
|
|||||||
Examples
|
Examples
|
||||||
--------
|
--------
|
||||||
|
|
||||||
- The `kubernetes <https://github.com/zalando/patroni/tree/master/kubernetes>`__ folder of the Patroni repository contains
|
- The `kubernetes <https://github.com/patroni/patroni/tree/master/kubernetes>`__ folder of the Patroni repository contains
|
||||||
examples of the Docker image, and the Kubernetes manifest to test Patroni Kubernetes setup.
|
examples of the Docker image, and the Kubernetes manifest to test Patroni Kubernetes setup.
|
||||||
Note that in the current state it will not be able to use PersistentVolumes because of permission issues.
|
Note that in the current state it will not be able to use PersistentVolumes because of permission issues.
|
||||||
|
|
||||||
@@ -98,5 +98,5 @@ Examples
|
|||||||
to deploy the Spilo image configured with Patroni running using Kubernetes.
|
to deploy the Spilo image configured with Patroni running using Kubernetes.
|
||||||
|
|
||||||
- In order to run your database clusters at scale using Patroni and Spilo, take a look at the
|
- In order to run your database clusters at scale using Patroni and Spilo, take a look at the
|
||||||
`postgres-operator <https://github.com/zalando-incubator/postgres-operator>`_ project. It implements the operator pattern
|
`postgres-operator <https://github.com/zalando/postgres-operator>`_ project. It implements the operator pattern
|
||||||
to manage Spilo clusters.
|
to manage Spilo clusters.
|
||||||
|
|||||||
@@ -49,10 +49,11 @@ Some of the PostgreSQL parameters **must hold the same values on the primary and
|
|||||||
|
|
||||||
For the parameters below, PostgreSQL does not require equal values among the primary and all the replicas. However, considering the possibility of a replica to become the primary at any time, it doesn't really make sense to set them differently; therefore, **Patroni restricts setting their values to the** :ref:`dynamic configuration <dynamic_configuration>`.
|
For the parameters below, PostgreSQL does not require equal values among the primary and all the replicas. However, considering the possibility of a replica to become the primary at any time, it doesn't really make sense to set them differently; therefore, **Patroni restricts setting their values to the** :ref:`dynamic configuration <dynamic_configuration>`.
|
||||||
|
|
||||||
- **max_wal_senders**: 5
|
- **max_wal_senders**: 10
|
||||||
- **max_replication_slots**: 5
|
- **max_replication_slots**: 10
|
||||||
- **wal_keep_segments**: 8
|
- **wal_keep_segments**: 8
|
||||||
- **wal_keep_size**: 128MB
|
- **wal_keep_size**: 128MB
|
||||||
|
- **wal_log_hints**: on
|
||||||
|
|
||||||
These parameters are validated to ensure they are sane, or meet a minimum value.
|
These parameters are validated to ensure they are sane, or meet a minimum value.
|
||||||
|
|
||||||
@@ -62,9 +63,8 @@ There are some other Postgres parameters controlled by Patroni:
|
|||||||
- **port** - is set either from ``postgresql.listen`` or from ``PATRONI_POSTGRESQL_LISTEN`` environment variable
|
- **port** - is set either from ``postgresql.listen`` or from ``PATRONI_POSTGRESQL_LISTEN`` environment variable
|
||||||
- **cluster_name** - is set either from ``scope`` or from ``PATRONI_SCOPE`` environment variable
|
- **cluster_name** - is set either from ``scope`` or from ``PATRONI_SCOPE`` environment variable
|
||||||
- **hot_standby: on**
|
- **hot_standby: on**
|
||||||
- **wal_log_hints: on** - for Postgres 9.4 and newer.
|
|
||||||
|
|
||||||
To be on the safe side parameters from the above lists are not written into ``postgresql.conf``, but passed as a list of arguments to the ``pg_ctl start`` which gives them the highest precedence, even above `ALTER SYSTEM <https://www.postgresql.org/docs/current/static/sql-altersystem.html>`__
|
To be on the safe side parameters from the above lists are written into ``postgresql.conf``, and passed as a list of arguments to the ``postgres`` which gives them the highest precedence (except ``wal_keep_segments`` and ``wal_keep_size``), even above `ALTER SYSTEM <https://www.postgresql.org/docs/current/static/sql-altersystem.html>`__
|
||||||
|
|
||||||
There also are some parameters like **postgresql.listen**, **postgresql.data_dir** that **can be set only locally**, i.e. in the Patroni :ref:`config file <yaml_configuration>` or via :ref:`configuration <environment>` variable. In most cases the local configuration will override the dynamic configuration.
|
There also are some parameters like **postgresql.listen**, **postgresql.data_dir** that **can be set only locally**, i.e. in the Patroni :ref:`config file <yaml_configuration>` or via :ref:`configuration <environment>` variable. In most cases the local configuration will override the dynamic configuration.
|
||||||
|
|
||||||
@@ -239,7 +239,7 @@ Validate Patroni configuration
|
|||||||
|
|
||||||
.. code:: text
|
.. code:: text
|
||||||
|
|
||||||
patroni --validate-config [configfile]
|
patroni --validate-config [configfile] [--ignore-listen-port | -i]
|
||||||
|
|
||||||
Description
|
Description
|
||||||
"""""""""""
|
"""""""""""
|
||||||
@@ -251,3 +251,9 @@ Parameters
|
|||||||
|
|
||||||
``configfile``
|
``configfile``
|
||||||
Full path to the configuration file to check. If not given or file does not exist, will try to read from the ``PATRONI_CONFIG_VARIABLE`` environment variable or, if not set, from the :ref:`Patroni environment variables <environment>`.
|
Full path to the configuration file to check. If not given or file does not exist, will try to read from the ``PATRONI_CONFIG_VARIABLE`` environment variable or, if not set, from the :ref:`Patroni environment variables <environment>`.
|
||||||
|
|
||||||
|
``--ignore-listen-port | -i``
|
||||||
|
Optional flag to ignore bind failures for ``listen`` ports that are already in use when validating the ``configfile``.
|
||||||
|
|
||||||
|
``--print | -p``
|
||||||
|
Optional flag to print out local configuration (including environment configuration overrides) after it has been successfully validated.
|
||||||
|
|||||||
+13
-24
@@ -78,7 +78,7 @@ This is the synopsis for running a command from the ``patronictl``:
|
|||||||
- Things written in uppercase represent a literal that should be given a value to.
|
- Things written in uppercase represent a literal that should be given a value to.
|
||||||
|
|
||||||
We will use this same syntax when describing ``patronictl`` sub-commands in the following sub-sections.
|
We will use this same syntax when describing ``patronictl`` sub-commands in the following sub-sections.
|
||||||
Also, when describing sub-commands in the following sub-sections, the commands' synposis should be seen as a replacement for the ``SUBCOMMAND`` in the above synopsis.
|
Also, when describing sub-commands in the following sub-sections, the commands' synopsis should be seen as a replacement for the ``SUBCOMMAND`` in the above synopsis.
|
||||||
|
|
||||||
In the following sub-sections you can find a description of each command implemented by ``patronictl``. For sake of example, we will use the configuration files present in the GitHub repository of Patroni (files ``postgres0.yml``, ``postgres1.yml`` and ``postgres2.yml``).
|
In the following sub-sections you can find a description of each command implemented by ``patronictl``. For sake of example, we will use the configuration files present in the GitHub repository of Patroni (files ``postgres0.yml``, ``postgres1.yml`` and ``postgres2.yml``).
|
||||||
|
|
||||||
@@ -224,7 +224,7 @@ Parameters
|
|||||||
|
|
||||||
``PG_CONFIG`` is the name of the Postgres configuration to be set.
|
``PG_CONFIG`` is the name of the Postgres configuration to be set.
|
||||||
|
|
||||||
``PG_VALUE`` is the value for ``PG_CONFIG``. If it is ``nulll``, then ``PG_CONFIG`` will be removed from the dynamic configuration.
|
``PG_VALUE`` is the value for ``PG_CONFIG``. If it is ``null``, then ``PG_CONFIG`` will be removed from the dynamic configuration.
|
||||||
|
|
||||||
``--apply``
|
``--apply``
|
||||||
Apply dynamic configuration from the given file.
|
Apply dynamic configuration from the given file.
|
||||||
@@ -320,7 +320,6 @@ Synopsis
|
|||||||
failover
|
failover
|
||||||
[ CLUSTER_NAME ]
|
[ CLUSTER_NAME ]
|
||||||
[ --group CITUS_GROUP ]
|
[ --group CITUS_GROUP ]
|
||||||
[ { --leader | --primary } LEADER_NAME ]
|
|
||||||
--candidate CANDIDATE_NAME
|
--candidate CANDIDATE_NAME
|
||||||
[ --force ]
|
[ --force ]
|
||||||
|
|
||||||
@@ -359,16 +358,6 @@ Parameters
|
|||||||
|
|
||||||
``CITUS_GROUP`` is the ID of the Citus group.
|
``CITUS_GROUP`` is the ID of the Citus group.
|
||||||
|
|
||||||
``--leader`` / ``--primary``
|
|
||||||
Indicate who is the expected leader at failover time.
|
|
||||||
|
|
||||||
If given, a switchover is performed instead of a failover.
|
|
||||||
|
|
||||||
``LEADER_NAME`` should match the name of the current leader in the cluster.
|
|
||||||
|
|
||||||
.. warning::
|
|
||||||
This argument is deprecated and will be removed in a future release.
|
|
||||||
|
|
||||||
``--candidate``
|
``--candidate``
|
||||||
The node to be promoted on failover.
|
The node to be promoted on failover.
|
||||||
|
|
||||||
@@ -1065,7 +1054,7 @@ Run a SQL command as ``postgres`` user, and take password from ``libpq`` environ
|
|||||||
|
|
||||||
.. code:: bash
|
.. code:: bash
|
||||||
|
|
||||||
$ PGPASSWORD=zalando patronictl -c postgres0.yml query batman -U postgres -c "SELECT now()"
|
$ PGPASSWORD=patroni patronictl -c postgres0.yml query batman -U postgres -c "SELECT now()"
|
||||||
now
|
now
|
||||||
2023-09-12 18:11:37.639500+00:00
|
2023-09-12 18:11:37.639500+00:00
|
||||||
|
|
||||||
@@ -1438,7 +1427,7 @@ Parameters
|
|||||||
``--scheduled``
|
``--scheduled``
|
||||||
Schedule a restart to occur at the given timestamp.
|
Schedule a restart to occur at the given timestamp.
|
||||||
|
|
||||||
``TIMESTAMP`` is the timestamp when the restart should occur. Specify it in unambiguous format, preferrably with time zone. You can also use the literal ``now`` for the restart to be executed immediately.
|
``TIMESTAMP`` is the timestamp when the restart should occur. Specify it in unambiguous format, preferably with time zone. You can also use the literal ``now`` for the restart to be executed immediately.
|
||||||
|
|
||||||
``--force``
|
``--force``
|
||||||
Flag to skip confirmation prompts when requesting the restart operations.
|
Flag to skip confirmation prompts when requesting the restart operations.
|
||||||
@@ -1676,7 +1665,7 @@ Parameters
|
|||||||
``--scheduled``
|
``--scheduled``
|
||||||
Schedule a switchover to occur at the given timestamp.
|
Schedule a switchover to occur at the given timestamp.
|
||||||
|
|
||||||
``TIMESTAMP`` is the timestamp when the switchover should occur. Specify it in unambiguous format, preferrably with time zone. You can also use the literal ``now`` for the switchover to be executed immediately.
|
``TIMESTAMP`` is the timestamp when the switchover should occur. Specify it in unambiguous format, preferably with time zone. You can also use the literal ``now`` for the switchover to be executed immediately.
|
||||||
|
|
||||||
``--force``
|
``--force``
|
||||||
Flag to skip confirmation prompts when performing the switchover.
|
Flag to skip confirmation prompts when performing the switchover.
|
||||||
@@ -1951,25 +1940,25 @@ Get version of ``patronictl`` only:
|
|||||||
.. code:: bash
|
.. code:: bash
|
||||||
|
|
||||||
$ patronictl -c postgres0.yml version
|
$ patronictl -c postgres0.yml version
|
||||||
patronictl version 3.1.0
|
patronictl version 4.0.0
|
||||||
|
|
||||||
Get version of ``patronictl`` and of all members of cluster ``batman``:
|
Get version of ``patronictl`` and of all members of cluster ``batman``:
|
||||||
|
|
||||||
.. code:: bash
|
.. code:: bash
|
||||||
|
|
||||||
$ patronictl -c postgres0.yml version batman
|
$ patronictl -c postgres0.yml version batman
|
||||||
patronictl version 3.1.0
|
patronictl version 4.0.0
|
||||||
|
|
||||||
postgresql0: Patroni 3.1.0 PostgreSQL 15.2
|
postgresql0: Patroni 4.0.0 PostgreSQL 16.4
|
||||||
postgresql1: Patroni 3.1.0 PostgreSQL 15.2
|
postgresql1: Patroni 4.0.0 PostgreSQL 16.4
|
||||||
postgresql2: Patroni 3.1.0 PostgreSQL 15.2
|
postgresql2: Patroni 4.0.0 PostgreSQL 16.4
|
||||||
|
|
||||||
Get version of ``patronictl`` and of members ``postgresql1`` and ``postgresql2`` of cluster ``batman``:
|
Get version of ``patronictl`` and of members ``postgresql1`` and ``postgresql2`` of cluster ``batman``:
|
||||||
|
|
||||||
.. code:: bash
|
.. code:: bash
|
||||||
|
|
||||||
$ patronictl -c postgres0.yml version batman postgresql1 postgresql2
|
$ patronictl -c postgres0.yml version batman postgresql1 postgresql2
|
||||||
patronictl version 3.1.0
|
patronictl version 4.0.0
|
||||||
|
|
||||||
postgresql1: Patroni 3.1.0 PostgreSQL 15.2
|
postgresql1: Patroni 4.0.0 PostgreSQL 16.4
|
||||||
postgresql2: Patroni 3.1.0 PostgreSQL 15.2
|
postgresql2: Patroni 4.0.0 PostgreSQL 16.4
|
||||||
|
|||||||
+508
-12
@@ -3,9 +3,399 @@
|
|||||||
Release notes
|
Release notes
|
||||||
=============
|
=============
|
||||||
|
|
||||||
|
Version 4.0.5
|
||||||
|
-------------
|
||||||
|
|
||||||
|
Released 2025-02-20
|
||||||
|
|
||||||
|
**Stability improvements**
|
||||||
|
|
||||||
|
- Compatibility with ``python-json-logger>=3.1`` (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Get rid of the warnings produced by the old API usage.
|
||||||
|
|
||||||
|
- Compatibility with Python 3.13 (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Run tests against Python 3.13.
|
||||||
|
|
||||||
|
- Compatibility with ``pyinstaller>=4.4`` (Joe Jensen)
|
||||||
|
|
||||||
|
Fall back to the default ``iter_modules`` if ``pyinstaller`` ``toc`` attribute is not present.
|
||||||
|
|
||||||
|
- Fix issues with PostgreSQL 9.5 support (Alexander Kukushkin)
|
||||||
|
|
||||||
|
- Properly handle ``pg_rewind`` output format.
|
||||||
|
- Consider ``synchronous_standby_names`` format not supporting "num" specification.
|
||||||
|
|
||||||
|
- Compatibility with the latest changes in ``urlparse`` (Alexander Kukushkin)
|
||||||
|
|
||||||
|
``urlparse`` doesn't accept multiple hosts with ``[]`` character in URL anymore. To mitigate the problem, switch to the native wrappers of ``PQconninfoParse()`` from ``libpq``, when it is possible, and use our implementation only for older ``psycopg2`` versions that are linked with an outdated version of ``libpq``.
|
||||||
|
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Show only the members to be restarted upon restart confirmation (András Váczi)
|
||||||
|
|
||||||
|
Previously, when doing ``patronictl restart <clustername> --pending``, the confirmation listed all members, regardless of whether their restart is pending.
|
||||||
|
|
||||||
|
- Cancel long-running jobs on Patroni stop and remove data directory on replica bootstrap failure (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Previously, Patroni could be doing replica bootstrap, while ``pg_basebackup`` / ``wal-g`` / ``pgBackRest`` / ``barman`` or similar keep running.
|
||||||
|
|
||||||
|
- Properly handle cluster names with a slash in ``patronictl edit-config`` (Antoni Mur)
|
||||||
|
|
||||||
|
Replace a forward slash in ``cluster_name`` with an underscore.
|
||||||
|
|
||||||
|
- Avoid dropping physical slots too early (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Postpone removal of physical replication slots containing ``xmin`` after a failover: on the new primary -- until this member is promoted, on replicas -- until there is a leader in the cluster.
|
||||||
|
|
||||||
|
- Handle all exceptions raised by subprocess in ``controldata()`` (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Patroni was not properly handling all exceptions possibly raised when calling ``pg_controldata`` utility.
|
||||||
|
|
||||||
|
- Fix bug with a slot for a former leader not retained on failover (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Avoid falsely relying on members being present in DCS, while on failover ``/member`` key for the former leader is expiring exactly at the same time.
|
||||||
|
|
||||||
|
- Fix a couple of bugs in the quorum state machine (Alexander Kukushkin)
|
||||||
|
|
||||||
|
- When evaluating whether there are healthy nodes for a leader race, before demoting we need to take into account quorum requirements. Without it, the former leader may end up in recovery surrounded by asynchronous nodes.
|
||||||
|
- ``QuorumStateResolver`` wasn't correctly handling the case when a replica node quickly joined and disconnected.
|
||||||
|
|
||||||
|
|
||||||
|
**Improvements**
|
||||||
|
|
||||||
|
- Improve error on am empty or non-dictionary configuration file (Julian)
|
||||||
|
|
||||||
|
Throw a more explicit exception when validating if Patroni configuration file contains a valid ``Mapping`` object.
|
||||||
|
|
||||||
|
|
||||||
|
Version 4.0.4
|
||||||
|
-------------
|
||||||
|
|
||||||
|
Released 2024-11-22
|
||||||
|
|
||||||
|
**Stability improvements**
|
||||||
|
|
||||||
|
- Add compatibility with the ``py-consul`` module (Alexander Kukushkin)
|
||||||
|
|
||||||
|
``python-consul`` module is unmaintained for a long time, while ``py-consul`` is the official replacement. Backward compatibility with python-consul is retained.
|
||||||
|
|
||||||
|
- Add compatibility with the ``prettytable>=3.12.0`` module (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Address deprecation warnings.
|
||||||
|
|
||||||
|
- Compatibility with the ``ydiff==1.4.2`` module (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Fix compatibility issues for the latest version, constrain version in ``requirements.txt``, and introduce latest version compatibility test.
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Run ``on_role_change`` callback after a failed primary recovery (Polina Bungina, Alexander Kukushkin)
|
||||||
|
|
||||||
|
Additionally run ``on_role_change`` callback for a primary that failed to start after a crash to increase chances the callback is executed, even if the further start as a replica fails.
|
||||||
|
|
||||||
|
- Fix a thread leak in ``patronictl list -W`` (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Cache DCS instance object to avoid thread leak.
|
||||||
|
|
||||||
|
- Ensure only supported parameters are written to the connection string (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Patroni used to pass parameters introduced in newer versions to the connection string, which had been leading to connection errors.
|
||||||
|
|
||||||
|
|
||||||
|
Version 4.0.3
|
||||||
|
-------------
|
||||||
|
|
||||||
|
Released 2024-10-18
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Disable ``pgaudit`` when creating users not to expose password (kviset)
|
||||||
|
|
||||||
|
Patroni was logging ``superuser``, ``replication``, and ``rewind`` passwords on their creation when ``pgaudit`` extension was enabled.
|
||||||
|
|
||||||
|
- Fix issue with mixed setups: primary on pre-Patroni v4 and replicas on v4+ (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Use ``xlog_location`` extracted from ``/members`` key instead of trying to get a member's slot position from ``/status`` key if Patroni version running on the leader is pre-4.0.0. Not doing so has been causing WALs accumulation on replicas.
|
||||||
|
|
||||||
|
- Do not ignore valid PostgreSQL GUCs that don't have Patroni validator (Polina Bungina)
|
||||||
|
|
||||||
|
Still check against ``postgres --describe-config`` if a GUC does not have a Patroni validator but is, in fact, a valid GUC.
|
||||||
|
|
||||||
|
|
||||||
|
**Improvements**
|
||||||
|
|
||||||
|
- Recheck annotations on 409 status code when reading leader object in K8s (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Avoid an additional update if ``PATCH`` request was canceled by Patroni, while the request successfully updated the target.
|
||||||
|
|
||||||
|
- Add support of ``sslnegotiation`` client-side connection option (Alexander Kukushkin)
|
||||||
|
|
||||||
|
``sslnegotiation`` was added to the final PostgreSQL 17 release.
|
||||||
|
|
||||||
|
|
||||||
|
Version 4.0.2
|
||||||
|
-------------
|
||||||
|
|
||||||
|
Released 2024-09-17
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Handle exceptions while discovering configuration validation files (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Skip directories for which Patroni does not have sufficient permissions to perform list operations.
|
||||||
|
|
||||||
|
- Make sure inactive hot physical replication slots don't hold ``xmin`` (Alexander Kukushkin, Polina Bungina)
|
||||||
|
|
||||||
|
Since version 3.2.0 Patroni creates physical replication slots for all members on replicas and periodically moves them forward using ``pg_replication_slot_advance()`` function. However if for any reason ``hot_standby_feedback`` is enabled and the primary is demoted to replica, the now inactive slots have ``NOT NULL`` ``xmin`` value propagated back to the new primary. This results in ``xmin`` horizon not being moved forward and vacuum not being able to clean up dead tuples. With this fix, Patroni recreates the physical replication slots that are supposed to be inactive but have ``NOT NULL`` ``xmin`` value.
|
||||||
|
|
||||||
|
- Fix unhandled ``DCSError`` during the startup phase (Waynerv)
|
||||||
|
|
||||||
|
Ensure DCS connectivity before trying to check the uniqueness of the node name.
|
||||||
|
|
||||||
|
- Explicitly include ``CMDLINE_OPTIONS`` GUCs when querying ``pg_settings`` (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Make sure all GUCs that are passed to postmaster as command line parameters are restored when Patroni is joining a running standby. This is a follow-up for the bug fixed in Patroni 3.2.2.
|
||||||
|
|
||||||
|
- Fix bug in ``synchronous_standby_names`` quotting logic (Alexander Kukushkin)
|
||||||
|
|
||||||
|
According to PostgreSQL documentation, ``ANY`` and ``FIRST`` keywords are supposed to be double-quoted, which Patroni did not do before.
|
||||||
|
|
||||||
|
- Fix keepalive connection out-of-range issue (hadizamani021)
|
||||||
|
|
||||||
|
Ensure that ``keepalive`` option value calculated based on the ``ttl`` set does not exceed the maximum allowed value for the current platform.
|
||||||
|
|
||||||
|
|
||||||
|
Version 4.0.1
|
||||||
|
-------------
|
||||||
|
|
||||||
|
Released 2024-08-30
|
||||||
|
|
||||||
|
**Bugfix**
|
||||||
|
|
||||||
|
- Patroni was creating unnecessary replication slots for itself (Alexander Kukushkin)
|
||||||
|
|
||||||
|
It was happening if ``name`` contains upper-case or special characters.
|
||||||
|
|
||||||
|
|
||||||
|
Version 4.0.0
|
||||||
|
-------------
|
||||||
|
|
||||||
|
Released 2024-08-29
|
||||||
|
|
||||||
|
.. warning::
|
||||||
|
- This version completes work on getting rid of the "master" term, in favor of "primary". This means a couple of breaking changes, please read the release notes carefully. Upgrading to the Patroni 4+ will work reliably only if you run Patroni 3.1.0 or newer. Upgrading from an older version directly to 4+ is possible but may lead to unexpected behavior if the primary fails while the rest of the nodes are running on other Patroni versions.
|
||||||
|
|
||||||
|
|
||||||
|
**Breaking changes**
|
||||||
|
|
||||||
|
- The following breaking changes were introduced when getting rid of the non-inclusive "master" term in the Patroni code:
|
||||||
|
|
||||||
|
- On Kubernetes, Patroni by default will set ``role`` label to ``primary``. In case if you want to keep the old behavior and avoid downtime or lengthy complex migrations, you can configure parameters ``kubernetes.leader_label_value`` and ``kubernetes.standby_leader_label_value`` to ``master``. Read more :ref:`here <kubernetes_role_values>`.
|
||||||
|
- Patroni role is written to DCS as ``primary`` instead of ``master``.
|
||||||
|
- Patroni role returned by Patroni REST API has been changed from ``master`` to ``primary``.
|
||||||
|
- Patroni REST API no longer accepts ``role=master`` in requests to ``/switchover``, ``/failover``, ``/restart`` endpoints.
|
||||||
|
- ``/metrics`` REST API endpoint will no longer report ``patroni_master`` metric.
|
||||||
|
- ``patronictl`` no longer accepts ``--master`` option for any command. ``--leader`` or ``--primary`` options should be used instead.
|
||||||
|
- ``no_master`` option in the declarative configuration of custom replica creation methods is no longer treated as a special option, please use ``no_leader`` instead.
|
||||||
|
- ``patroni_wale_restore`` script doesn't accept ``--no_master`` option anymore.
|
||||||
|
- ``patroni_barman`` script doesn't accept ``--role=master`` option anymore.
|
||||||
|
- All callback scripts are executed with ``role=primary`` option passed instead of ``role=master``.
|
||||||
|
|
||||||
|
- ``patronictl failover`` does not accept ``--leader`` option that was deprecated since Patroni 3.2.0.
|
||||||
|
|
||||||
|
- User creation functionality (``bootstrap.users`` configuration section) deprecated since Patroni 3.2.0 has been removed.
|
||||||
|
|
||||||
|
|
||||||
|
**New features**
|
||||||
|
|
||||||
|
- Quorum-based failover (Ants Aasma, Alexander Kukushkin)
|
||||||
|
|
||||||
|
The feature implements quorum-based synchronous replication (available from PostgreSQL v10) which helps to reduce worst-case latencies, even during normal operation, as a higher latency of replicating to one standby can be compensated by other standbys. Patroni implements additional safeguards to prevent any user-visible data loss by choosing a failover candidate based on the latest transaction received.
|
||||||
|
|
||||||
|
- Register Citus secondaries in ``pg_dist_node`` (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Patroni now maintains the list of nodes with ``role==replica``, ``state==running`` and without ``noloadbalance`` :ref:`tag <tags_settings>` in ``pg_dist_node``.
|
||||||
|
|
||||||
|
- Configurable retention of members' replication slots (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Implements support of ``member_slots_ttl`` global configuration parameter that controls for how long member replication slots should be kept around when the member key is absent.
|
||||||
|
|
||||||
|
- Make permissions of log files created by Patroni configurable (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Allows to set specific permissions for log files created by Patroni. If not specified, permissions are set based on the current ``umask`` value.
|
||||||
|
|
||||||
|
- Compatibility with PostgreSQL 17 beta3 (Alexander Kukushkin)
|
||||||
|
|
||||||
|
GUC's validator rules were extended. Patroni handles all the new auxiliary backends during shutdown and sets ``dbname`` in ``primary_conninfo``, as it is required for logical replication slots synchronization.
|
||||||
|
|
||||||
|
- Implement ``--ignore-listen-port`` option for Patroni config validation (Sahil Naphade)
|
||||||
|
|
||||||
|
Make it possible to ignore already bound ports when running ``patroni --validate-config``.
|
||||||
|
|
||||||
|
|
||||||
|
**Improvements**
|
||||||
|
|
||||||
|
- Make ``wal_log_hints`` configurable (Paul_Kim)
|
||||||
|
|
||||||
|
Allows to avoid the overhead of ``wal_log_hints`` configuration being enabled in case ``use_pg_rewind`` is set to ``off``.
|
||||||
|
|
||||||
|
- Log ``pg_basebackup`` command in ``DEBUG`` level (Waynerv)
|
||||||
|
|
||||||
|
Facilitates failed initialization debugging.
|
||||||
|
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Advance permanent slots for cascading nodes while in failsafe (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Ensure that slots for cascading replicas are properly advanced on the primary when failsafe mode is activated. It is done by extending replicas response on ``POST /failsafe`` REST API request with their ``xlog_location``.
|
||||||
|
|
||||||
|
- Don't let the current node be chosen as synchronous (Alexander Kukushkin)
|
||||||
|
|
||||||
|
There may be "something" streaming from the current primary node with ``application_name`` that matches the name of the current primary. Patroni was not properly handling this situation, which could end up in the primary being declared as a synchronous node and consequently was blocking switchovers.
|
||||||
|
|
||||||
|
- Ignore ``restapi.allowlist_include_members`` for POST /failsafe (Alexander Kukushkin)
|
||||||
|
|
||||||
|
- Improve GUCs validation (Polina Bungina)
|
||||||
|
|
||||||
|
Due to additional validation through running ``postgres --describe-config`` command, it was previously not possible to set GUCs not listed there through Patroni configuration. This limitation is now removed.
|
||||||
|
|
||||||
|
- Add line with ``localhost`` to ``.pgpass`` file when unix sockets are detected (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Patroni will add an additional line to ``.pgpass`` file if ``host`` parameter specified starts with ``/`` character. This allows to cover a corner case when ``host`` matches the default socket directory path.
|
||||||
|
|
||||||
|
- Fix logging issues (Waynerv)
|
||||||
|
|
||||||
|
Defined proper request URL in failsafe handling logs and fixed the order of timestamps in postmaster check log.
|
||||||
|
|
||||||
|
|
||||||
|
Version 3.3.2
|
||||||
|
-------------
|
||||||
|
|
||||||
|
Released 2024-07-11
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Fix plain Postgres synchronous replication mode (Israel Barth Rubio)
|
||||||
|
|
||||||
|
Since ``synchronous_mode`` was introduced to Patroni, the plain Postgres synchronous replication was not working. With this bugfix, Patroni sets the value of ``synchronous_standby_names`` as configured by the user, if that is the case, when ``synchronous_mode`` is disabled.
|
||||||
|
|
||||||
|
- Handle logical slots invalidation on a standby (Polina Bungina)
|
||||||
|
|
||||||
|
Since PG16 logical replication slots on a standby can be invalidated due to horizon: from now on, Patroni forces copy (i.e., recreation) of invalidated slots.
|
||||||
|
|
||||||
|
- Fix race condition with logical slot advance and copy (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Due to this bug, it was a possible situation when an invalidated logical replication slot was copied with PostgreSQL restart more than once.
|
||||||
|
|
||||||
|
|
||||||
|
Version 3.3.1
|
||||||
|
-------------
|
||||||
|
|
||||||
|
Released 2024-06-17
|
||||||
|
|
||||||
|
**Stability improvements**
|
||||||
|
|
||||||
|
- Compatibility with Python 3.12 (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Handle a new attribute added to ``logging.LogRecord``.
|
||||||
|
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Fix infinite recursion in ``replicatefrom`` tags handling (Alexander Kukushkin)
|
||||||
|
|
||||||
|
As a part of this fix, also improve ``is_physical_slot()`` check and adjust documentation.
|
||||||
|
|
||||||
|
- Fix wrong role reporting in standby clusters (Alexander Kukushkin)
|
||||||
|
|
||||||
|
``synchronous_standby_names`` and synchronous replication only work on a real primary node and in the case of cascading replication are simply ignored by Postgres. Before this fix, ``patronictl list`` and ``GET /cluster`` were falsely reporting some nodes as synchronous.
|
||||||
|
|
||||||
|
- Fix availability of the ``allow_in_place_tablespaces`` GUC (Polina Bungina)
|
||||||
|
|
||||||
|
``allow_in_place_tablespaces`` was not only added to PostgreSQL 15 but also backpatched to PostgreSQL 10-14.
|
||||||
|
|
||||||
|
|
||||||
|
Version 3.3.0
|
||||||
|
-------------
|
||||||
|
|
||||||
|
Released 2024-04-04
|
||||||
|
|
||||||
|
.. warning::
|
||||||
|
All older Partoni versions are not compatible with ``ydiff>=1.3``.
|
||||||
|
|
||||||
|
There are the following options available to "fix" the problem:
|
||||||
|
|
||||||
|
1. upgrade Patroni to the latest version
|
||||||
|
2. install ``ydiff<1.3`` after installing Patroni
|
||||||
|
3. install ``cdiff`` module
|
||||||
|
|
||||||
|
|
||||||
|
**New features**
|
||||||
|
|
||||||
|
- Add ability to pass ``auth_data`` to Zookeeper client (Aras Mumcuyan)
|
||||||
|
|
||||||
|
It allows to specify the authentication credentials to use for the connection.
|
||||||
|
|
||||||
|
- Add a contrib script for ``Barman`` integration (Israel Barth Rubio)
|
||||||
|
|
||||||
|
Provide an application ``patroni_barman`` that allows to perform ``Barman`` operations remotely and can be used as a custom bootstrap/custom replica method or as an ``on_role_change`` callback. Please check :ref:`here <tools_integration>` for more information.
|
||||||
|
|
||||||
|
- Support ``JSON`` log format (alisalemmi)
|
||||||
|
|
||||||
|
Apart from ``plain`` (default), Patroni now also supports ``json`` log format. Requires ``python-json-logger>=2.0.2`` library to be installed.
|
||||||
|
|
||||||
|
- Show ``pending_restart_reason`` information (Polina Bungina)
|
||||||
|
|
||||||
|
Provide extended information about the PostgreSQL parameters that caused ``pending_restart`` flag to be set. Both ``patronictl list`` and ``/patroni`` REST API endpoint now show the parameters names and their "diff" as ``pending_restart_reason``.
|
||||||
|
|
||||||
|
- Implement ``nostream`` tag (Grigory Smolkin)
|
||||||
|
|
||||||
|
If ``nostream`` tag is set to ``true``, the node will not use replication protocol to stream WAL but instead rely on archive recovery (if ``restore_command`` is configured). It also disables copying and synchronization of permanent logical replication slots on the node itself and all its cascading replicas.
|
||||||
|
|
||||||
|
|
||||||
|
**Improvements**
|
||||||
|
|
||||||
|
- Implement validation of the ``log`` section (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Until now validator was not checking the correctness of the logging configuration provided.
|
||||||
|
|
||||||
|
- Improve logging for PostgreSQL parameters change (Polina Bungina)
|
||||||
|
|
||||||
|
Convert old values to a human-readable format and log information about the ``pg_controldata`` vs Patroni global configuration mismatch.
|
||||||
|
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Properly filter out not allowed ``pg_basebackup`` options (Israel Barth Rubio)
|
||||||
|
|
||||||
|
Due to a bug, Patroni was not properly filtering out the not allowed options configured for the ``basebackup`` replica bootstrap method, when provided in the ``- setting: value`` format.
|
||||||
|
|
||||||
|
- Fix ``etcd3`` authentication error handling (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Always retry one time on ``etcd3`` authentication error if authentication was not done right before executing the request. Also, do not restart watchers on reauthentication.
|
||||||
|
|
||||||
|
- Improve logic of the validator files discovery (Waynerv)
|
||||||
|
|
||||||
|
Use ``importlib`` library to discover the files with available configuration parameters when possible (for Python 3.9+). This implementation is more stable and doesn't break the Patroni distributions based on ``zip`` archives.
|
||||||
|
|
||||||
|
- Use ``target_session_attrs`` only when multiple hosts are specified in the ``standby_cluster`` section (Alexander Kukushkin)
|
||||||
|
|
||||||
|
``target_session_attrs=read-write`` is now added to the ``primary_conninfo`` on the standby leader node only when ``standby_cluster.host`` section contains multiple hosts separated by commas.
|
||||||
|
|
||||||
|
- Add compatibility code for ``ydiff`` library version 1.3+ (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Patroni is relying on some API from ``ydiff`` that is not public because it is supposed to be just a terminal tool rather than a python module. Unfortunately, the API change in 1.3 broke old Patroni versions.
|
||||||
|
|
||||||
|
|
||||||
Version 3.2.2
|
Version 3.2.2
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2024-01-17
|
||||||
|
|
||||||
**Bugfixes**
|
**Bugfixes**
|
||||||
|
|
||||||
- Don't let replica restore initialize key when DCS was wiped (Alexander Kukushkin)
|
- Don't let replica restore initialize key when DCS was wiped (Alexander Kukushkin)
|
||||||
@@ -56,6 +446,8 @@ Version 3.2.2
|
|||||||
Version 3.2.1
|
Version 3.2.1
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-11-30
|
||||||
|
|
||||||
**Bugfixes**
|
**Bugfixes**
|
||||||
|
|
||||||
- Limit accepted values for ``--format`` argument in ``patronictl`` (Alexander Kukushkin)
|
- Limit accepted values for ``--format`` argument in ``patronictl`` (Alexander Kukushkin)
|
||||||
@@ -94,6 +486,8 @@ Version 3.2.1
|
|||||||
Version 3.2.0
|
Version 3.2.0
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-10-25
|
||||||
|
|
||||||
**Deprecation notice**
|
**Deprecation notice**
|
||||||
|
|
||||||
- The ``bootstrap.users`` support will be removed in version 4.0.0. If you need to create users after deploying a new cluster please use the ``bootstrap.post_bootstrap`` hook for that.
|
- The ``bootstrap.users`` support will be removed in version 4.0.0. If you need to create users after deploying a new cluster please use the ``bootstrap.post_bootstrap`` hook for that.
|
||||||
@@ -166,6 +560,8 @@ Version 3.2.0
|
|||||||
Version 3.1.2
|
Version 3.1.2
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-09-26
|
||||||
|
|
||||||
**Bugfixes**
|
**Bugfixes**
|
||||||
|
|
||||||
- Fixed bug with ``wal_keep_size`` checks (Alexander Kukushkin)
|
- Fixed bug with ``wal_keep_size`` checks (Alexander Kukushkin)
|
||||||
@@ -188,6 +584,8 @@ Version 3.1.2
|
|||||||
Version 3.1.1
|
Version 3.1.1
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-09-20
|
||||||
|
|
||||||
**Bugfixes**
|
**Bugfixes**
|
||||||
|
|
||||||
- Reset failsafe state on promote (ChenChangAo)
|
- Reset failsafe state on promote (ChenChangAo)
|
||||||
@@ -240,12 +638,14 @@ Version 3.1.1
|
|||||||
|
|
||||||
- Don't rely on ``pg_stat_wal_receiver`` when deciding on ``pg_rewind`` (Alexander Kukushkin)
|
- Don't rely on ``pg_stat_wal_receiver`` when deciding on ``pg_rewind`` (Alexander Kukushkin)
|
||||||
|
|
||||||
It could happen that ``received_tli`` reported by ``pg_stat_wal_recevier`` is ahead of the actual replayed timeline, while the timeline reported by ``DENTIFY_SYSTEM`` via replication connection is always correct.
|
It could happen that ``received_tli`` reported by ``pg_stat_wal_receiver`` is ahead of the actual replayed timeline, while the timeline reported by ``DENTIFY_SYSTEM`` via replication connection is always correct.
|
||||||
|
|
||||||
|
|
||||||
Version 3.1.0
|
Version 3.1.0
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-08-03
|
||||||
|
|
||||||
**Breaking changes**
|
**Breaking changes**
|
||||||
|
|
||||||
- Changed semantic of ``restapi.keyfile`` and ``restapi.certfile`` (Alexander Kukushkin)
|
- Changed semantic of ``restapi.keyfile`` and ``restapi.certfile`` (Alexander Kukushkin)
|
||||||
@@ -324,6 +724,8 @@ Version 3.1.0
|
|||||||
Version 3.0.4
|
Version 3.0.4
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-07-13
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Make the replication status of standby nodes visible (Alexander Kukushkin)
|
- Make the replication status of standby nodes visible (Alexander Kukushkin)
|
||||||
@@ -368,6 +770,8 @@ Version 3.0.4
|
|||||||
Version 3.0.3
|
Version 3.0.3
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-06-22
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Compatibility with PostgreSQL 16 beta1 (Alexander Kukushkin)
|
- Compatibility with PostgreSQL 16 beta1 (Alexander Kukushkin)
|
||||||
@@ -422,6 +826,8 @@ Version 3.0.3
|
|||||||
Version 3.0.2
|
Version 3.0.2
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-03-24
|
||||||
|
|
||||||
.. warning::
|
.. warning::
|
||||||
Version 3.0.2 dropped support of Python older than 3.6.
|
Version 3.0.2 dropped support of Python older than 3.6.
|
||||||
|
|
||||||
@@ -478,6 +884,8 @@ Version 3.0.2
|
|||||||
Version 3.0.1
|
Version 3.0.1
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-02-16
|
||||||
|
|
||||||
**Bugfixes**
|
**Bugfixes**
|
||||||
|
|
||||||
- Pass proper role name to an ``on_role_change`` callback script'. (Alexander Kukushkin, Polina Bungina)
|
- Pass proper role name to an ``on_role_change`` callback script'. (Alexander Kukushkin, Polina Bungina)
|
||||||
@@ -488,6 +896,8 @@ Version 3.0.1
|
|||||||
Version 3.0.0
|
Version 3.0.0
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-01-30
|
||||||
|
|
||||||
This version adds integration with `Citus <https://www.citusdata.com>`__ and makes it possible to survive temporary DCS outages without demoting primary.
|
This version adds integration with `Citus <https://www.citusdata.com>`__ and makes it possible to survive temporary DCS outages without demoting primary.
|
||||||
|
|
||||||
.. warning::
|
.. warning::
|
||||||
@@ -538,6 +948,8 @@ This version adds integration with `Citus <https://www.citusdata.com>`__ and mak
|
|||||||
Version 2.1.7
|
Version 2.1.7
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2023-01-04
|
||||||
|
|
||||||
**Bugfixes**
|
**Bugfixes**
|
||||||
|
|
||||||
- Fixed little incompatibilities with legacy python modules (Alexander Kukushkin)
|
- Fixed little incompatibilities with legacy python modules (Alexander Kukushkin)
|
||||||
@@ -548,6 +960,8 @@ Version 2.1.7
|
|||||||
Version 2.1.6
|
Version 2.1.6
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2022-12-30
|
||||||
|
|
||||||
**Improvements**
|
**Improvements**
|
||||||
|
|
||||||
- Fix annoying exceptions on ssl socket shutdown (Alexander Kukushkin)
|
- Fix annoying exceptions on ssl socket shutdown (Alexander Kukushkin)
|
||||||
@@ -603,6 +1017,8 @@ Version 2.1.6
|
|||||||
Version 2.1.5
|
Version 2.1.5
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2022-11-28
|
||||||
|
|
||||||
This version enhances compatibility with PostgreSQL 15 and declares Etcd v3 support as production ready. The Patroni on Raft remains in Beta.
|
This version enhances compatibility with PostgreSQL 15 and declares Etcd v3 support as production ready. The Patroni on Raft remains in Beta.
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
@@ -701,6 +1117,8 @@ This version enhances compatibility with PostgreSQL 15 and declares Etcd v3 supp
|
|||||||
Version 2.1.4
|
Version 2.1.4
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2022-06-01
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Improve ``pg_rewind`` behavior on typical Debian/Ubuntu systems (Gunnar "Nick" Bluth)
|
- Improve ``pg_rewind`` behavior on typical Debian/Ubuntu systems (Gunnar "Nick" Bluth)
|
||||||
@@ -773,6 +1191,8 @@ Version 2.1.4
|
|||||||
Version 2.1.3
|
Version 2.1.3
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2022-02-18
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Added support for encrypted TLS keys for ``patronictl`` (Alexander Kukushkin)
|
- Added support for encrypted TLS keys for ``patronictl`` (Alexander Kukushkin)
|
||||||
@@ -839,6 +1259,8 @@ Version 2.1.3
|
|||||||
Version 2.1.2
|
Version 2.1.2
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2021-12-03
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Compatibility with ``psycopg>=3.0`` (Alexander Kukushkin)
|
- Compatibility with ``psycopg>=3.0`` (Alexander Kukushkin)
|
||||||
@@ -943,6 +1365,8 @@ Version 2.1.2
|
|||||||
Version 2.1.1
|
Version 2.1.1
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2021-08-19
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Support for ETCD SRV name suffix (David Pavlicek)
|
- Support for ETCD SRV name suffix (David Pavlicek)
|
||||||
@@ -979,6 +1403,8 @@ Version 2.1.1
|
|||||||
Version 2.1.0
|
Version 2.1.0
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2021-07-06
|
||||||
|
|
||||||
This version adds compatibility with PostgreSQL v14, makes logical replication slots to survive failover/switchover, implements support of allowlist for REST API, and also reducing the number of logs to one line per heart-beat.
|
This version adds compatibility with PostgreSQL v14, makes logical replication slots to survive failover/switchover, implements support of allowlist for REST API, and also reducing the number of logs to one line per heart-beat.
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
@@ -1071,6 +1497,8 @@ This version adds compatibility with PostgreSQL v14, makes logical replication s
|
|||||||
Version 2.0.2
|
Version 2.0.2
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2021-02-22
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Ability to ignore externally managed replication slots (James Coleman)
|
- Ability to ignore externally managed replication slots (James Coleman)
|
||||||
@@ -1167,6 +1595,8 @@ Version 2.0.2
|
|||||||
Version 2.0.1
|
Version 2.0.1
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2020-10-01
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Use ``more`` as pager in ``patronictl edit-config`` if ``less`` is not available (Pavel Golub)
|
- Use ``more`` as pager in ``patronictl edit-config`` if ``less`` is not available (Pavel Golub)
|
||||||
@@ -1215,6 +1645,8 @@ Version 2.0.1
|
|||||||
Version 2.0.0
|
Version 2.0.0
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2020-09-02
|
||||||
|
|
||||||
This version enhances compatibility with PostgreSQL 13, adds support of multiple synchronous standbys, has significant improvements in handling of ``pg_rewind``, adds support of Etcd v3 and Patroni on pure RAFT (without Etcd, Consul, or Zookeeper), and makes it possible to optionally call the ``pre_promote`` (fencing) script.
|
This version enhances compatibility with PostgreSQL 13, adds support of multiple synchronous standbys, has significant improvements in handling of ``pg_rewind``, adds support of Etcd v3 and Patroni on pure RAFT (without Etcd, Consul, or Zookeeper), and makes it possible to optionally call the ``pre_promote`` (fencing) script.
|
||||||
|
|
||||||
**PostgreSQL 13 support**
|
**PostgreSQL 13 support**
|
||||||
@@ -1266,7 +1698,7 @@ This version enhances compatibility with PostgreSQL 13, adds support of multiple
|
|||||||
|
|
||||||
Replicas are waiting for checkpoint indication via member key of the leader in DCS. The key is normally updated only once per HA loop. Without waking the main thread up, replicas will have to wait up to ``loop_wait`` seconds longer than necessary.
|
Replicas are waiting for checkpoint indication via member key of the leader in DCS. The key is normally updated only once per HA loop. Without waking the main thread up, replicas will have to wait up to ``loop_wait`` seconds longer than necessary.
|
||||||
|
|
||||||
- Use of ``pg_stat_wal_recevier`` view on 9.6+ (Alexander Kukushkin)
|
- Use of ``pg_stat_wal_receiver`` view on 9.6+ (Alexander Kukushkin)
|
||||||
|
|
||||||
The view contains up-to-date values of ``primary_conninfo`` and ``primary_slot_name``, while the contents of ``recovery.conf`` could be stale.
|
The view contains up-to-date values of ``primary_conninfo`` and ``primary_slot_name``, while the contents of ``recovery.conf`` could be stale.
|
||||||
|
|
||||||
@@ -1437,6 +1869,8 @@ This version enhances compatibility with PostgreSQL 13, adds support of multiple
|
|||||||
Version 1.6.5
|
Version 1.6.5
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2020-08-23
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Master stop timeout (Krishna Sarabu)
|
- Master stop timeout (Krishna Sarabu)
|
||||||
@@ -1553,6 +1987,8 @@ Version 1.6.5
|
|||||||
Version 1.6.4
|
Version 1.6.4
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2020-01-27
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Implemented ``--wait`` option for ``patronictl reinit`` (Igor Yanchenko)
|
- Implemented ``--wait`` option for ``patronictl reinit`` (Igor Yanchenko)
|
||||||
@@ -1615,11 +2051,13 @@ Version 1.6.4
|
|||||||
Version 1.6.3
|
Version 1.6.3
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2019-12-05
|
||||||
|
|
||||||
**Bugfixes**
|
**Bugfixes**
|
||||||
|
|
||||||
- Don't expose password when running ``pg_rewind`` (Alexander Kukushkin)
|
- Don't expose password when running ``pg_rewind`` (Alexander Kukushkin)
|
||||||
|
|
||||||
Bug was introduced in the `#1301 <https://github.com/zalando/patroni/pull/1301>`__
|
Bug was introduced in the `#1301 <https://github.com/patroni/patroni/pull/1301>`__
|
||||||
|
|
||||||
- Apply connection parameters specified in the ``postgresql.authentication`` to ``pg_basebackup`` and custom replica creation methods (Alexander Kukushkin)
|
- Apply connection parameters specified in the ``postgresql.authentication`` to ``pg_basebackup`` and custom replica creation methods (Alexander Kukushkin)
|
||||||
|
|
||||||
@@ -1629,6 +2067,8 @@ Version 1.6.3
|
|||||||
Version 1.6.2
|
Version 1.6.2
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2019-12-05
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Implemented ``patroni --version`` (Igor Yanchenko)
|
- Implemented ``patroni --version`` (Igor Yanchenko)
|
||||||
@@ -1684,6 +2124,8 @@ Version 1.6.2
|
|||||||
Version 1.6.1
|
Version 1.6.1
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2019-11-15
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Added ``PATRONICTL_CONFIG_FILE`` environment variable (msvechla)
|
- Added ``PATRONICTL_CONFIG_FILE`` environment variable (msvechla)
|
||||||
@@ -1813,6 +2255,8 @@ Version 1.6.1
|
|||||||
Version 1.6.0
|
Version 1.6.0
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2019-08-05
|
||||||
|
|
||||||
This version adds compatibility with PostgreSQL 12, makes is possible to run pg_rewind without superuser on PostgreSQL 11 and newer, and enables IPv6 support.
|
This version adds compatibility with PostgreSQL 12, makes is possible to run pg_rewind without superuser on PostgreSQL 11 and newer, and enables IPv6 support.
|
||||||
|
|
||||||
|
|
||||||
@@ -1925,6 +2369,8 @@ This version adds compatibility with PostgreSQL 12, makes is possible to run pg_
|
|||||||
Version 1.5.6
|
Version 1.5.6
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2019-08-03
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Support work with etcd cluster via set of proxies (Alexander Kukushkin)
|
- Support work with etcd cluster via set of proxies (Alexander Kukushkin)
|
||||||
@@ -1968,6 +2414,8 @@ Version 1.5.6
|
|||||||
Version 1.5.5
|
Version 1.5.5
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2019-02-15
|
||||||
|
|
||||||
This version introduces the possibility of automatic reinit of the former master, improves patronictl list output and fixes a number of bugs.
|
This version introduces the possibility of automatic reinit of the former master, improves patronictl list output and fixes a number of bugs.
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
@@ -2006,6 +2454,8 @@ This version introduces the possibility of automatic reinit of the former master
|
|||||||
Version 1.5.4
|
Version 1.5.4
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2019-01-15
|
||||||
|
|
||||||
This version implements flexible logging and fixes a number of bugs.
|
This version implements flexible logging and fixes a number of bugs.
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
@@ -2065,6 +2515,8 @@ This version implements flexible logging and fixes a number of bugs.
|
|||||||
Version 1.5.3
|
Version 1.5.3
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2018-12-03
|
||||||
|
|
||||||
Compatibility and bugfix release.
|
Compatibility and bugfix release.
|
||||||
|
|
||||||
- Improve stability when running with python3 against zookeeper (Alexander Kukushkin)
|
- Improve stability when running with python3 against zookeeper (Alexander Kukushkin)
|
||||||
@@ -2082,6 +2534,8 @@ Compatibility and bugfix release.
|
|||||||
Version 1.5.2
|
Version 1.5.2
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2018-11-26
|
||||||
|
|
||||||
Compatibility and bugfix release.
|
Compatibility and bugfix release.
|
||||||
|
|
||||||
- Compatibility with kazoo-2.6.0 (Alexander Kukushkin)
|
- Compatibility with kazoo-2.6.0 (Alexander Kukushkin)
|
||||||
@@ -2095,6 +2549,8 @@ Compatibility and bugfix release.
|
|||||||
Version 1.5.1
|
Version 1.5.1
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2018-11-01
|
||||||
|
|
||||||
This version implements support of permanent replication slots, adds support of pgBackRest and fixes number of bugs.
|
This version implements support of permanent replication slots, adds support of pgBackRest and fixes number of bugs.
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
@@ -2111,15 +2567,17 @@ This version implements support of permanent replication slots, adds support of
|
|||||||
|
|
||||||
- A few bugfixes in the "standby cluster" workflow (Alexander Kukushkin)
|
- A few bugfixes in the "standby cluster" workflow (Alexander Kukushkin)
|
||||||
|
|
||||||
Please see https://github.com/zalando/patroni/pull/823 for more details.
|
Please see https://github.com/patroni/patroni/pull/823 for more details.
|
||||||
|
|
||||||
- Fix REST API health check when cluster management is paused and DCS is not accessible (Alexander Kukushkin)
|
- Fix REST API health check when cluster management is paused and DCS is not accessible (Alexander Kukushkin)
|
||||||
|
|
||||||
Regression was introduced in https://github.com/zalando/patroni/commit/90cf930036a9d5249265af15d2b787ec7517cf57
|
Regression was introduced in https://github.com/patroni/patroni/commit/90cf930036a9d5249265af15d2b787ec7517cf57
|
||||||
|
|
||||||
Version 1.5.0
|
Version 1.5.0
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2018-09-20
|
||||||
|
|
||||||
This version enables Patroni HA cluster to operate in a standby mode, introduces experimental support for running on Windows, and provides a new configuration parameter to register PostgreSQL service in Consul.
|
This version enables Patroni HA cluster to operate in a standby mode, introduces experimental support for running on Windows, and provides a new configuration parameter to register PostgreSQL service in Consul.
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
@@ -2168,6 +2626,8 @@ This version enables Patroni HA cluster to operate in a standby mode, introduces
|
|||||||
Version 1.4.6
|
Version 1.4.6
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2018-08-14
|
||||||
|
|
||||||
**Bug fixes and stability improvements**
|
**Bug fixes and stability improvements**
|
||||||
|
|
||||||
This release fixes a critical issue with Patroni API /master endpoint returning 200 for the non-master node. This is a
|
This release fixes a critical issue with Patroni API /master endpoint returning 200 for the non-master node. This is a
|
||||||
@@ -2185,6 +2645,8 @@ reporting issue, no actual split-brain, but under certain circumstances clients
|
|||||||
Version 1.4.5
|
Version 1.4.5
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2018-08-03
|
||||||
|
|
||||||
**New features**
|
**New features**
|
||||||
|
|
||||||
- Improve logging when applying new postgres configuration (Don Seiler)
|
- Improve logging when applying new postgres configuration (Don Seiler)
|
||||||
@@ -2249,6 +2711,8 @@ Version 1.4.5
|
|||||||
Version 1.4.4
|
Version 1.4.4
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2018-05-22
|
||||||
|
|
||||||
**Stability improvements**
|
**Stability improvements**
|
||||||
|
|
||||||
- Fix race condition in poll_failover_result (Alexander Kukushkin)
|
- Fix race condition in poll_failover_result (Alexander Kukushkin)
|
||||||
@@ -2311,6 +2775,8 @@ Version 1.4.4
|
|||||||
Version 1.4.3
|
Version 1.4.3
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2018-03-05
|
||||||
|
|
||||||
**Improvements in logging**
|
**Improvements in logging**
|
||||||
|
|
||||||
- Make log level configurable from environment variables (Andy Newton, Keyvan Hedayati)
|
- Make log level configurable from environment variables (Andy Newton, Keyvan Hedayati)
|
||||||
@@ -2331,12 +2797,14 @@ Version 1.4.3
|
|||||||
|
|
||||||
- Single user mode was waiting for user input and never finish (Alexander Kukushkin)
|
- Single user mode was waiting for user input and never finish (Alexander Kukushkin)
|
||||||
|
|
||||||
Regression was introduced in https://github.com/zalando/patroni/pull/576
|
Regression was introduced in https://github.com/patroni/patroni/pull/576
|
||||||
|
|
||||||
|
|
||||||
Version 1.4.2
|
Version 1.4.2
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2018-01-30
|
||||||
|
|
||||||
**Improvements in patronictl**
|
**Improvements in patronictl**
|
||||||
|
|
||||||
- Rename scheduled failover to scheduled switchover (Alexander Kukushkin)
|
- Rename scheduled failover to scheduled switchover (Alexander Kukushkin)
|
||||||
@@ -2380,6 +2848,8 @@ Version 1.4.2
|
|||||||
Version 1.4.1
|
Version 1.4.1
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2018-01-17
|
||||||
|
|
||||||
**Fixes in patronictl**
|
**Fixes in patronictl**
|
||||||
|
|
||||||
- Don't show current leader in suggested list of members to failover to. (Alexander Kukushkin)
|
- Don't show current leader in suggested list of members to failover to. (Alexander Kukushkin)
|
||||||
@@ -2394,6 +2864,8 @@ Version 1.4.1
|
|||||||
Version 1.4
|
Version 1.4
|
||||||
-----------
|
-----------
|
||||||
|
|
||||||
|
Released 2018-01-10
|
||||||
|
|
||||||
This version adds support for using Kubernetes as a DCS, allowing to run Patroni as a cloud-native agent in Kubernetes without any additional deployments of Etcd, Zookeeper or Consul.
|
This version adds support for using Kubernetes as a DCS, allowing to run Patroni as a cloud-native agent in Kubernetes without any additional deployments of Etcd, Zookeeper or Consul.
|
||||||
|
|
||||||
**Upgrade notice**
|
**Upgrade notice**
|
||||||
@@ -2411,7 +2883,7 @@ In addition to using Endpoints, Patroni supports ConfigMaps. You can find more i
|
|||||||
|
|
||||||
This object identifies a running postmaster process via pid and start time and simplifies detection (and resolution) of situations when the postmaster was restarted behind our back or when postgres directory disappeared from the file system.
|
This object identifies a running postmaster process via pid and start time and simplifies detection (and resolution) of situations when the postmaster was restarted behind our back or when postgres directory disappeared from the file system.
|
||||||
|
|
||||||
- Minimize the amount of SELECT's issued by Patroni on every loop of HA cylce (Alexander Kukushkin)
|
- Minimize the amount of SELECT's issued by Patroni on every loop of HA cycle (Alexander Kukushkin)
|
||||||
|
|
||||||
On every iteration of HA loop Patroni needs to know recovery status and absolute wal position. From now on Patroni will run only single SELECT to get this information instead of two on the replica and three on the master.
|
On every iteration of HA loop Patroni needs to know recovery status and absolute wal position. From now on Patroni will run only single SELECT to get this information instead of two on the replica and three on the master.
|
||||||
|
|
||||||
@@ -2435,7 +2907,7 @@ In addition to using Endpoints, Patroni supports ConfigMaps. You can find more i
|
|||||||
|
|
||||||
- Improve ``patronictl reinit`` (Alexander Kukushkin)
|
- Improve ``patronictl reinit`` (Alexander Kukushkin)
|
||||||
|
|
||||||
Sometimes ``patronictl reinit`` refused to proceed when Patroni was busy with other actions, namely trying to start postgres. `patronictl` didn't provide any commands to cancel such long running actions and the only (dangerous) workarond was removing a data directory manually. The new implementation of `reinit` forcefully cancells other long-running actions before proceeding with reinit.
|
Sometimes ``patronictl reinit`` refused to proceed when Patroni was busy with other actions, namely trying to start postgres. `patronictl` didn't provide any commands to cancel such long running actions and the only (dangerous) workarond was removing a data directory manually. The new implementation of `reinit` forcefully cancels other long-running actions before proceeding with reinit.
|
||||||
|
|
||||||
- Implement ``--wait`` flag in ``patronictl pause`` and ``patronictl resume`` (Alexander Kukushkin)
|
- Implement ``--wait`` flag in ``patronictl pause`` and ``patronictl resume`` (Alexander Kukushkin)
|
||||||
|
|
||||||
@@ -2464,7 +2936,7 @@ In addition to using Endpoints, Patroni supports ConfigMaps. You can find more i
|
|||||||
|
|
||||||
- Add new /sync and /async endpoints (Alexander Kukushkin, Oleksii Kliukin)
|
- Add new /sync and /async endpoints (Alexander Kukushkin, Oleksii Kliukin)
|
||||||
|
|
||||||
Those endpoints (also accessible as /synchronous and /asynchronous) return 200 only for synchronous and asynchronous replicas correspondingly (exclusing those marked as `noloadbalance`).
|
Those endpoints (also accessible as /synchronous and /asynchronous) return 200 only for synchronous and asynchronous replicas correspondingly (excluding those marked as `noloadbalance`).
|
||||||
|
|
||||||
**Allow multiple hosts for Etcd**
|
**Allow multiple hosts for Etcd**
|
||||||
|
|
||||||
@@ -2476,6 +2948,8 @@ In addition to using Endpoints, Patroni supports ConfigMaps. You can find more i
|
|||||||
Version 1.3.6
|
Version 1.3.6
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2017-11-10
|
||||||
|
|
||||||
**Stability improvements**
|
**Stability improvements**
|
||||||
|
|
||||||
- Verify process start time when checking if postgres is running. (Ants Aasma)
|
- Verify process start time when checking if postgres is running. (Ants Aasma)
|
||||||
@@ -2517,6 +2991,8 @@ Version 1.3.6
|
|||||||
Version 1.3.5
|
Version 1.3.5
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2017-10-12
|
||||||
|
|
||||||
**Bugfix**
|
**Bugfix**
|
||||||
|
|
||||||
- Set role to 'uninitialized' if data directory was removed (Alexander Kukushkin)
|
- Set role to 'uninitialized' if data directory was removed (Alexander Kukushkin)
|
||||||
@@ -2548,6 +3024,8 @@ Version 1.3.5
|
|||||||
Version 1.3.4
|
Version 1.3.4
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2017-09-08
|
||||||
|
|
||||||
**Different Consul improvements**
|
**Different Consul improvements**
|
||||||
|
|
||||||
- Pass the consul token as a header (Andrew Colin Kissa)
|
- Pass the consul token as a header (Andrew Colin Kissa)
|
||||||
@@ -2586,6 +3064,8 @@ Version 1.3.4
|
|||||||
Version 1.3.3
|
Version 1.3.3
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2017-08-04
|
||||||
|
|
||||||
**Bugfixes**
|
**Bugfixes**
|
||||||
|
|
||||||
- synchronous replication was disabled shortly after promotion even when synchronous_mode_strict was turned on (Alexander Kukushkin)
|
- synchronous replication was disabled shortly after promotion even when synchronous_mode_strict was turned on (Alexander Kukushkin)
|
||||||
@@ -2596,6 +3076,8 @@ Version 1.3.3
|
|||||||
Version 1.3.2
|
Version 1.3.2
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2017-07-31
|
||||||
|
|
||||||
**Bugfix**
|
**Bugfix**
|
||||||
|
|
||||||
- patronictl edit-config didn't work with ZooKeeper (Alexander Kukushkin)
|
- patronictl edit-config didn't work with ZooKeeper (Alexander Kukushkin)
|
||||||
@@ -2604,6 +3086,8 @@ Version 1.3.2
|
|||||||
Version 1.3.1
|
Version 1.3.1
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
|
Released 2017-07-28
|
||||||
|
|
||||||
**Bugfix**
|
**Bugfix**
|
||||||
|
|
||||||
- failover via API was broken due to change in ``_MemberStatus`` (Alexander Kukushkin)
|
- failover via API was broken due to change in ``_MemberStatus`` (Alexander Kukushkin)
|
||||||
@@ -2612,6 +3096,8 @@ Version 1.3.1
|
|||||||
Version 1.3
|
Version 1.3
|
||||||
-----------
|
-----------
|
||||||
|
|
||||||
|
Released 2017-07-27
|
||||||
|
|
||||||
Version 1.3 adds custom bootstrap possibility, significantly improves support for pg_rewind, enhances the
|
Version 1.3 adds custom bootstrap possibility, significantly improves support for pg_rewind, enhances the
|
||||||
synchronous mode support, adds configuration editing to patronictl and implements watchdog support on Linux.
|
synchronous mode support, adds configuration editing to patronictl and implements watchdog support on Linux.
|
||||||
In addition, this is the first version to work correctly with PostgreSQL 10.
|
In addition, this is the first version to work correctly with PostgreSQL 10.
|
||||||
@@ -2631,7 +3117,7 @@ at the end.
|
|||||||
Allow custom bootstrap scripts instead of ``initdb`` when initializing the very first node in the cluster.
|
Allow custom bootstrap scripts instead of ``initdb`` when initializing the very first node in the cluster.
|
||||||
The bootstrap command receives the name of the cluster and the path to the data directory. The resulting cluster can
|
The bootstrap command receives the name of the cluster and the path to the data directory. The resulting cluster can
|
||||||
be configured to perform recovery, making it possible to bootstrap from a backup and do point in time recovery. Refer
|
be configured to perform recovery, making it possible to bootstrap from a backup and do point in time recovery. Refer
|
||||||
to the :ref:`documentaton page <custom_bootstrap>` for more detailed description of this feature.
|
to the :ref:`documentation page <custom_bootstrap>` for more detailed description of this feature.
|
||||||
|
|
||||||
**Smarter pg_rewind support**
|
**Smarter pg_rewind support**
|
||||||
|
|
||||||
@@ -2740,6 +3226,8 @@ at the end.
|
|||||||
Version 1.2
|
Version 1.2
|
||||||
-----------
|
-----------
|
||||||
|
|
||||||
|
Released 2016-12-13
|
||||||
|
|
||||||
This version introduces significant improvements over the handling of synchronous replication, makes the startup process and failover more reliable, adds PostgreSQL 9.6 support and fixes plenty of bugs.
|
This version introduces significant improvements over the handling of synchronous replication, makes the startup process and failover more reliable, adds PostgreSQL 9.6 support and fixes plenty of bugs.
|
||||||
In addition, the documentation, including these release notes, has been moved to https://patroni.readthedocs.io.
|
In addition, the documentation, including these release notes, has been moved to https://patroni.readthedocs.io.
|
||||||
|
|
||||||
@@ -2819,7 +3307,7 @@ In addition, the documentation, including these release notes, has been moved to
|
|||||||
|
|
||||||
**Documentation improvements**
|
**Documentation improvements**
|
||||||
|
|
||||||
- Add a Patroni main `loop workflow diagram <https://raw.githubusercontent.com/zalando/patroni/master/docs/ha_loop_diagram.png>`__. (Alejandro Martínez, Alexander Kukushkin)
|
- Add a Patroni main `loop workflow diagram <https://raw.githubusercontent.com/patroni/patroni/master/docs/ha_loop_diagram.png>`__. (Alejandro Martínez, Alexander Kukushkin)
|
||||||
|
|
||||||
- Improve README, adding the Helm chart and links to release notes. (Lauri Apple)
|
- Improve README, adding the Helm chart and links to release notes. (Lauri Apple)
|
||||||
|
|
||||||
@@ -2835,6 +3323,8 @@ In addition, the documentation, including these release notes, has been moved to
|
|||||||
Version 1.1
|
Version 1.1
|
||||||
-----------
|
-----------
|
||||||
|
|
||||||
|
Released 2016-09-07
|
||||||
|
|
||||||
This release improves management of Patroni cluster by bring in pause mode, improves maintenance with scheduled and conditional restarts, makes Patroni interaction with Etcd or Zookeeper more resilient and greatly enhances patronictl.
|
This release improves management of Patroni cluster by bring in pause mode, improves maintenance with scheduled and conditional restarts, makes Patroni interaction with Etcd or Zookeeper more resilient and greatly enhances patronictl.
|
||||||
|
|
||||||
**Upgrade notice**
|
**Upgrade notice**
|
||||||
@@ -2932,6 +3422,8 @@ Previously, there was no reliable way to query Patroni about PostgreSQL instance
|
|||||||
Version 1.0
|
Version 1.0
|
||||||
-----------
|
-----------
|
||||||
|
|
||||||
|
Released 2016-07-05
|
||||||
|
|
||||||
This release introduces the global dynamic configuration that allows dynamic changes of the PostgreSQL and Patroni configuration parameters for the entire HA cluster. It also delivers numerous bugfixes.
|
This release introduces the global dynamic configuration that allows dynamic changes of the PostgreSQL and Patroni configuration parameters for the entire HA cluster. It also delivers numerous bugfixes.
|
||||||
|
|
||||||
**Upgrade notice**
|
**Upgrade notice**
|
||||||
@@ -3023,6 +3515,8 @@ When upgrading from v0.90 or below, always upgrade all replicas before the maste
|
|||||||
Version 0.90
|
Version 0.90
|
||||||
------------
|
------------
|
||||||
|
|
||||||
|
Released 2016-04-27
|
||||||
|
|
||||||
This releases adds support for Consul, includes a new *noloadbalance* tag, changes the behavior of the *clonefrom* tag, improves *pg_rewind* handling and improves *patronictl* control program.
|
This releases adds support for Consul, includes a new *noloadbalance* tag, changes the behavior of the *clonefrom* tag, improves *pg_rewind* handling and improves *patronictl* control program.
|
||||||
|
|
||||||
**Consul support**
|
**Consul support**
|
||||||
@@ -3085,6 +3579,8 @@ This releases adds support for Consul, includes a new *noloadbalance* tag, chang
|
|||||||
Version 0.80
|
Version 0.80
|
||||||
------------
|
------------
|
||||||
|
|
||||||
|
Released 2016-03-14
|
||||||
|
|
||||||
This release adds support for *cascading replication* and simplifies Patroni management by providing *scheduled failovers*. One may use older versions of Patroni (in particular, 0.78) combined with this one in order to migrate to the new release. Note that the scheduled failover and cascading replication related features will only work with Patroni 0.80 and above.
|
This release adds support for *cascading replication* and simplifies Patroni management by providing *scheduled failovers*. One may use older versions of Patroni (in particular, 0.78) combined with this one in order to migrate to the new release. Note that the scheduled failover and cascading replication related features will only work with Patroni 0.80 and above.
|
||||||
|
|
||||||
**Cascading replication**
|
**Cascading replication**
|
||||||
@@ -3125,4 +3621,4 @@ This release adds support for *cascading replication* and simplifies Patroni man
|
|||||||
|
|
||||||
The tests can be launched manually using the *behave* command. They are also launched automatically for pull requests and after commits.
|
The tests can be launched manually using the *behave* command. They are also launched automatically for pull requests and after commits.
|
||||||
|
|
||||||
Release notes for some older versions can be found on `project's github page <https://github.com/zalando/patroni/releases>`__.
|
Release notes for some older versions can be found on `project's github page <https://github.com/patroni/patroni/releases>`__.
|
||||||
|
|||||||
@@ -1,3 +1,5 @@
|
|||||||
|
.. _replica_imaging_and_bootstrap:
|
||||||
|
|
||||||
Replica imaging and bootstrap
|
Replica imaging and bootstrap
|
||||||
=============================
|
=============================
|
||||||
|
|
||||||
@@ -50,7 +52,7 @@ cleans up after itself and releases the initialize lock to give another node the
|
|||||||
If a ``recovery_conf`` block is defined in the same section as the custom bootstrap method, Patroni will generate a
|
If a ``recovery_conf`` block is defined in the same section as the custom bootstrap method, Patroni will generate a
|
||||||
``recovery.conf`` before starting the newly bootstrapped instance (or set the recovery settings on Postgres configuration if
|
``recovery.conf`` before starting the newly bootstrapped instance (or set the recovery settings on Postgres configuration if
|
||||||
running PostgreSQL >= 12).
|
running PostgreSQL >= 12).
|
||||||
Typically, such recovery configuration should contain at least one of the ``recovery_target_*`` parameters, together with the ``recovery_target_timeline`` set to ``promote``.
|
Typically, such recovery configuration should contain at least one of the ``recovery_target_*`` parameters, together with the ``recovery_target_action`` set to ``promote``.
|
||||||
|
|
||||||
If ``keep_existing_recovery_conf`` is defined and set to ``True``, Patroni will not remove the existing ``recovery.conf`` file if it exists (PostgreSQL <= 11).
|
If ``keep_existing_recovery_conf`` is defined and set to ``True``, Patroni will not remove the existing ``recovery.conf`` file if it exists (PostgreSQL <= 11).
|
||||||
Similarly, in that case Patroni will not remove the existing ``recovery.signal`` or ``standby.signal`` if either exists, nor will it override the configured recovery settings (PostgreSQL >= 12).
|
Similarly, in that case Patroni will not remove the existing ``recovery.signal`` or ``standby.signal`` if either exists, nor will it override the configured recovery settings (PostgreSQL >= 12).
|
||||||
@@ -79,14 +81,13 @@ As an example, you are able to bootstrap a fresh Patroni cluster from a Barman b
|
|||||||
method: barman
|
method: barman
|
||||||
barman:
|
barman:
|
||||||
keep_existing_recovery_conf: true
|
keep_existing_recovery_conf: true
|
||||||
command: patroni_barman_recover
|
command: patroni_barman --api-url https://barman-host:7480 recover
|
||||||
api-url: https://barman-host:7480
|
|
||||||
barman-server: my_server
|
barman-server: my_server
|
||||||
ssh-command: ssh postgres@patroni-host
|
ssh-command: ssh postgres@patroni-host
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
``patroni_barman_recover`` requires that you have both Barman and ``pg-backup-api`` configured in the Barman host, so it can execute a remote ``barman recover`` through the backup API.
|
``patroni_barman recover`` requires that you have both Barman and ``pg-backup-api`` configured in the Barman host, so it can execute a remote ``barman recover`` through the backup API.
|
||||||
The above example uses a subset of the available parameters. You can get more information running ``patroni_barman_recover --help``.
|
The above example uses a subset of the available parameters. You can get more information running ``patroni_barman recover --help``.
|
||||||
|
|
||||||
.. _custom_replica_creation:
|
.. _custom_replica_creation:
|
||||||
|
|
||||||
@@ -150,16 +151,15 @@ example: Barman
|
|||||||
- barman
|
- barman
|
||||||
- basebackup
|
- basebackup
|
||||||
barman:
|
barman:
|
||||||
command: patroni_barman_recover
|
command: patroni_barman --api-url https://barman-host:7480 recover
|
||||||
api-url: https://barman-host:7480
|
|
||||||
barman-server: my_server
|
barman-server: my_server
|
||||||
ssh-command: ssh postgres@patroni-host
|
ssh-command: ssh postgres@patroni-host
|
||||||
basebackup:
|
basebackup:
|
||||||
max-rate: '100M'
|
max-rate: '100M'
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
``patroni_barman_recover`` requires that you have both Barman and ``pg-backup-api`` configured in the Barman host, so it can execute a remote ``barman recover`` through the backup API.
|
``patroni_barman recover`` requires that you have both Barman and ``pg-backup-api`` configured in the Barman host, so it can execute a remote ``barman recover`` through the backup API.
|
||||||
The above example uses a subset of the available parameters. You can get more information running ``patroni_barman_recover --help``.
|
The above example uses a subset of the available parameters. You can get more information running ``patroni_barman recover --help``.
|
||||||
|
|
||||||
The ``create_replica_methods`` defines available replica creation methods and the order of executing them. Patroni will
|
The ``create_replica_methods`` defines available replica creation methods and the order of executing them. Patroni will
|
||||||
stop on the first one that returns 0. Each method should define a separate section in the configuration file, listing the command
|
stop on the first one that returns 0. Each method should define a separate section in the configuration file, listing the command
|
||||||
@@ -219,59 +219,3 @@ and
|
|||||||
- waldir: /pg-wal-mount/external-waldir
|
- waldir: /pg-wal-mount/external-waldir
|
||||||
|
|
||||||
If all replica creation methods fail, Patroni will try again all methods in order during the next event loop cycle.
|
If all replica creation methods fail, Patroni will try again all methods in order during the next event loop cycle.
|
||||||
|
|
||||||
.. _standby_cluster:
|
|
||||||
|
|
||||||
Standby cluster
|
|
||||||
---------------
|
|
||||||
|
|
||||||
Another available option is to run a "standby cluster", that contains only of
|
|
||||||
standby nodes replicating from some remote node. This type of clusters has:
|
|
||||||
|
|
||||||
* "standby leader", that behaves pretty much like a regular cluster leader,
|
|
||||||
except it replicates from a remote node.
|
|
||||||
|
|
||||||
* cascade replicas, that are replicating from standby leader.
|
|
||||||
|
|
||||||
Standby leader holds and updates a leader lock in DCS. If the leader lock
|
|
||||||
expires, cascade replicas will perform an election to choose another leader
|
|
||||||
from the standbys.
|
|
||||||
|
|
||||||
There is no further relationship between the standby cluster and the primary
|
|
||||||
cluster it replicates from, in particular, they must not share the same DCS
|
|
||||||
scope if they use the same DCS. They do not know anything else from each other
|
|
||||||
apart from replication information. Also, the standby cluster is not being
|
|
||||||
displayed in :ref:`patronictl_list` or :ref:`patronictl_topology` output on the
|
|
||||||
primary cluster.
|
|
||||||
|
|
||||||
For the sake of flexibility, you can specify methods of creating a replica and
|
|
||||||
recovery WAL records when a cluster is in the "standby mode" by providing
|
|
||||||
`create_replica_methods` key in `standby_cluster` section. It is distinct from
|
|
||||||
creating replicas, when cluster is detached and functions as a normal cluster,
|
|
||||||
which is controlled by `create_replica_methods` in `postgresql` section. Both
|
|
||||||
"standby" and "normal" `create_replica_methods` reference keys in `postgresql`
|
|
||||||
section.
|
|
||||||
|
|
||||||
To configure such cluster you need to specify the section ``standby_cluster``
|
|
||||||
in a patroni configuration:
|
|
||||||
|
|
||||||
.. code:: YAML
|
|
||||||
|
|
||||||
bootstrap:
|
|
||||||
dcs:
|
|
||||||
standby_cluster:
|
|
||||||
host: 1.2.3.4
|
|
||||||
port: 5432
|
|
||||||
primary_slot_name: patroni
|
|
||||||
create_replica_methods:
|
|
||||||
- basebackup
|
|
||||||
|
|
||||||
Note, that these options will be applied only once during cluster bootstrap,
|
|
||||||
and the only way to change them afterwards is through DCS.
|
|
||||||
|
|
||||||
Patroni expects to find `postgresql.conf` or `postgresql.conf.backup` in PGDATA
|
|
||||||
of the remote primary and will not start if it does not find it after a
|
|
||||||
basebackup. If the remote primary keeps its `postgresql.conf` elsewhere, it is
|
|
||||||
your responsibility to copy it to PGDATA.
|
|
||||||
|
|
||||||
If you use replication slots on the standby cluster, you must also create the corresponding replication slot on the primary cluster. It will not be done automatically by the standby cluster implementation. You can use Patroni's permanent replication slots feature on the primary cluster to maintain a replication slot with the same name as ``primary_slot_name``, or its default value if ``primary_slot_name`` is not provided.
|
|
||||||
|
|||||||
+111
-14
@@ -6,8 +6,9 @@ Replication modes
|
|||||||
|
|
||||||
Patroni uses PostgreSQL streaming replication. For more information about streaming replication, see the `Postgres documentation <http://www.postgresql.org/docs/current/static/warm-standby.html#STREAMING-REPLICATION>`__. By default Patroni configures PostgreSQL for asynchronous replication. Choosing your replication schema is dependent on your business considerations. Investigate both async and sync replication, as well as other HA solutions, to determine which solution is best for you.
|
Patroni uses PostgreSQL streaming replication. For more information about streaming replication, see the `Postgres documentation <http://www.postgresql.org/docs/current/static/warm-standby.html#STREAMING-REPLICATION>`__. By default Patroni configures PostgreSQL for asynchronous replication. Choosing your replication schema is dependent on your business considerations. Investigate both async and sync replication, as well as other HA solutions, to determine which solution is best for you.
|
||||||
|
|
||||||
|
|
||||||
Asynchronous mode durability
|
Asynchronous mode durability
|
||||||
----------------------------
|
============================
|
||||||
|
|
||||||
In asynchronous mode the cluster is allowed to lose some committed transactions to ensure availability. When the primary server fails or becomes unavailable for any other reason Patroni will automatically promote a sufficiently healthy standby to primary. Any transactions that have not been replicated to that standby remain in a "forked timeline" on the primary, and are effectively unrecoverable [1]_.
|
In asynchronous mode the cluster is allowed to lose some committed transactions to ensure availability. When the primary server fails or becomes unavailable for any other reason Patroni will automatically promote a sufficiently healthy standby to primary. Any transactions that have not been replicated to that standby remain in a "forked timeline" on the primary, and are effectively unrecoverable [1]_.
|
||||||
|
|
||||||
@@ -15,10 +16,11 @@ The amount of transactions that can be lost is controlled via ``maximum_lag_on_f
|
|||||||
|
|
||||||
By default, when running leader elections, Patroni does not take into account the current timeline of replicas, what in some cases could be undesirable behavior. You can prevent the node not having the same timeline as a former primary become the new leader by changing the value of ``check_timeline`` parameter to ``true``.
|
By default, when running leader elections, Patroni does not take into account the current timeline of replicas, what in some cases could be undesirable behavior. You can prevent the node not having the same timeline as a former primary become the new leader by changing the value of ``check_timeline`` parameter to ``true``.
|
||||||
|
|
||||||
PostgreSQL synchronous replication
|
|
||||||
----------------------------------
|
|
||||||
|
|
||||||
You can use Postgres's `synchronous replication <http://www.postgresql.org/docs/current/static/warm-standby.html#SYNCHRONOUS-REPLICATION>`__ with Patroni. Synchronous replication ensures consistency across a cluster by confirming that writes are written to a secondary before returning to the connecting client with a success. The cost of synchronous replication: reduced throughput on writes. This throughput will be entirely based on network performance.
|
PostgreSQL synchronous replication
|
||||||
|
==================================
|
||||||
|
|
||||||
|
You can use Postgres's `synchronous replication <http://www.postgresql.org/docs/current/static/warm-standby.html#SYNCHRONOUS-REPLICATION>`__ with Patroni. Synchronous replication ensures consistency across a cluster by confirming that writes are written to a secondary before returning to the connecting client with a success. The cost of synchronous replication: increased latency and reduced throughput on writes. This throughput will be entirely based on network performance.
|
||||||
|
|
||||||
In hosted datacenter environments (like AWS, Rackspace, or any network you do not control), synchronous replication significantly increases the variability of write performance. If followers become inaccessible from the leader, the leader effectively becomes read-only.
|
In hosted datacenter environments (like AWS, Rackspace, or any network you do not control), synchronous replication significantly increases the variability of write performance. If followers become inaccessible from the leader, the leader effectively becomes read-only.
|
||||||
|
|
||||||
@@ -33,10 +35,11 @@ When using PostgreSQL synchronous replication, use at least three Postgres data
|
|||||||
|
|
||||||
Using PostgreSQL synchronous replication does not guarantee zero lost transactions under all circumstances. When the primary and the secondary that is currently acting as a synchronous replica fail simultaneously a third node that might not contain all transactions will be promoted.
|
Using PostgreSQL synchronous replication does not guarantee zero lost transactions under all circumstances. When the primary and the secondary that is currently acting as a synchronous replica fail simultaneously a third node that might not contain all transactions will be promoted.
|
||||||
|
|
||||||
|
|
||||||
.. _synchronous_mode:
|
.. _synchronous_mode:
|
||||||
|
|
||||||
Synchronous mode
|
Synchronous mode
|
||||||
----------------
|
================
|
||||||
|
|
||||||
For use cases where losing committed transactions is not permissible you can turn on Patroni's ``synchronous_mode``. When ``synchronous_mode`` is turned on Patroni will not promote a standby unless it is certain that the standby contains all transactions that may have returned a successful commit status to client [2]_. This means that the system may be unavailable for writes even though some servers are available. System administrators can still use manual failover commands to promote a standby even if it results in transaction loss.
|
For use cases where losing committed transactions is not permissible you can turn on Patroni's ``synchronous_mode``. When ``synchronous_mode`` is turned on Patroni will not promote a standby unless it is certain that the standby contains all transactions that may have returned a successful commit status to client [2]_. This means that the system may be unavailable for writes even though some servers are available. System administrators can still use manual failover commands to promote a standby even if it results in transaction loss.
|
||||||
|
|
||||||
@@ -53,32 +56,126 @@ are available. As a downside, the primary is not be available for writes
|
|||||||
blocking all client write requests until at least one synchronous replica comes
|
blocking all client write requests until at least one synchronous replica comes
|
||||||
up.
|
up.
|
||||||
|
|
||||||
You can ensure that a standby never becomes the synchronous standby by setting ``nosync`` tag to true. This is recommended to set for standbys that are behind slow network connections and would cause performance degradation when becoming a synchronous standby.
|
You can ensure that a standby never becomes the synchronous standby by setting ``nosync`` tag to true. This is recommended to set for standbys that are behind slow network connections and would cause performance degradation when becoming a synchronous standby. Setting tag ``nostream`` to true will also have the same effect.
|
||||||
|
|
||||||
Synchronous mode can be switched on and off via Patroni REST interface. See :ref:`dynamic configuration <dynamic_configuration>` for instructions.
|
Synchronous mode can be switched on and off using ``patronictl edit-config`` command or via Patroni REST interface. See :ref:`dynamic configuration <dynamic_configuration>` for instructions.
|
||||||
|
|
||||||
Note: Because of the way synchronous replication is implemented in PostgreSQL it is still possible to lose transactions even when using ``synchronous_mode_strict``. If the PostgreSQL backend is cancelled while waiting to acknowledge replication (as a result of packet cancellation due to client timeout or backend failure) transaction changes become visible for other backends. Such changes are not yet replicated and may be lost in case of standby promotion.
|
Note: Because of the way synchronous replication is implemented in PostgreSQL it is still possible to lose transactions even when using ``synchronous_mode_strict``. If the PostgreSQL backend is cancelled while waiting to acknowledge replication (as a result of packet cancellation due to client timeout or backend failure) transaction changes become visible for other backends. Such changes are not yet replicated and may be lost in case of standby promotion.
|
||||||
|
|
||||||
|
|
||||||
Synchronous Replication Factor
|
Synchronous Replication Factor
|
||||||
------------------------------
|
==============================
|
||||||
The parameter ``synchronous_node_count`` is used by Patroni to manage number of synchronous standby databases. It is set to 1 by default. It has no effect when ``synchronous_mode`` is set to off. When enabled, Patroni manages precise number of synchronous standby databases based on parameter ``synchronous_node_count`` and adjusts the state in DCS & synchronous_standby_names as members join and leave.
|
|
||||||
|
The parameter ``synchronous_node_count`` is used by Patroni to manage the number of synchronous standby databases. It is set to ``1`` by default. It has no effect when ``synchronous_mode`` is set to ``off``. When enabled, Patroni manages the precise number of synchronous standby databases based on parameter ``synchronous_node_count`` and adjusts the state in DCS & ``synchronous_standby_names`` in PostgreSQL as members join and leave. If the parameter is set to a value higher than the number of eligible nodes it will be automatically reduced by Patroni.
|
||||||
|
|
||||||
|
|
||||||
|
Maximum lag on synchronous node
|
||||||
|
===============================
|
||||||
|
|
||||||
|
By default Patroni sticks to nodes that are declared as ``synchronous``, according to the ``pg_stat_replication`` view, even when there are other nodes ahead of it. This is done to minimize the number of changes of ``synchronous_standby_names``. To change this behavior one may use ``maximum_lag_on_syncnode`` parameter. It controls how much lag the replica can have to still be considered as "synchronous".
|
||||||
|
|
||||||
|
Patroni utilizes the max replica LSN if there is more than one standby, otherwise it will use leader's current wal LSN. The default is ``-1``, and Patroni will not take action to swap a synchronous unhealthy standby when the value is set to ``0`` or less. Please set the value high enough so that Patroni won't swap synchronous standbys frequently during high transaction volume.
|
||||||
|
|
||||||
|
|
||||||
Synchronous mode implementation
|
Synchronous mode implementation
|
||||||
-------------------------------
|
===============================
|
||||||
|
|
||||||
When in synchronous mode Patroni maintains synchronization state in the DCS, containing the latest primary and current synchronous standby databases. This state is updated with strict ordering constraints to ensure the following invariants:
|
When in synchronous mode Patroni maintains synchronization state in the DCS (``/sync`` key), containing the latest primary and current synchronous standby databases. This state is updated with strict ordering constraints to ensure the following invariants:
|
||||||
|
|
||||||
- A node must be marked as the latest leader whenever it can accept write transactions. Patroni crashing or PostgreSQL not shutting down can cause violations of this invariant.
|
- A node must be marked as the latest leader whenever it can accept write transactions. Patroni crashing or PostgreSQL not shutting down can cause violations of this invariant.
|
||||||
|
|
||||||
- A node must be set as the synchronous standby in PostgreSQL as long as it is published as the synchronous standby.
|
- A node must be set as the synchronous standby in PostgreSQL as long as it is published as the synchronous standby in the ``/sync`` key in DCS..
|
||||||
|
|
||||||
- A node that is not the leader or current synchronous standby is not allowed to promote itself automatically.
|
- A node that is not the leader or current synchronous standby is not allowed to promote itself automatically.
|
||||||
|
|
||||||
Patroni will only assign one or more synchronous standby nodes based on ``synchronous_node_count`` parameter to ``synchronous_standby_names``.
|
Patroni will only assign one or more synchronous standby nodes based on ``synchronous_node_count`` parameter to ``synchronous_standby_names``.
|
||||||
|
|
||||||
On each HA loop iteration Patroni re-evaluates synchronous standby nodes choice. If the current list of synchronous standby nodes are connected and has not requested its synchronous status to be removed it remains picked. Otherwise the cluster member available for sync that is furthest ahead in replication is picked.
|
On each HA loop iteration Patroni re-evaluates synchronous standby nodes choice. If the current list of synchronous standby nodes are connected and has not requested its synchronous status to be removed it remains picked. Otherwise the cluster members available for sync that are furthest ahead in replication are picked.
|
||||||
|
|
||||||
|
Example:
|
||||||
|
---------
|
||||||
|
|
||||||
|
``/config`` key in DCS
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
.. code-block:: YAML
|
||||||
|
|
||||||
|
synchronous_mode: on
|
||||||
|
synchronous_node_count: 2
|
||||||
|
...
|
||||||
|
|
||||||
|
``/sync`` key in DCS
|
||||||
|
^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
.. code-block:: JSON
|
||||||
|
|
||||||
|
{
|
||||||
|
"leader": "node0",
|
||||||
|
"sync_standby": "node1,node2"
|
||||||
|
}
|
||||||
|
|
||||||
|
postgresql.conf
|
||||||
|
^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
.. code-block:: INI
|
||||||
|
|
||||||
|
synchronous_standby_names = 'FIRST 2 (node1,node2)'
|
||||||
|
|
||||||
|
|
||||||
.. [1] The data is still there, but recovering it requires a manual recovery effort by data recovery specialists. When Patroni is allowed to rewind with ``use_pg_rewind`` the forked timeline will be automatically erased to rejoin the failed primary with the cluster.
|
In the above examples only nodes ``node1`` and ``node2`` are known to be synchronous and allowed to be automatically promoted if the primary (``node0``) fails.
|
||||||
|
|
||||||
|
|
||||||
|
.. _quorum_mode:
|
||||||
|
|
||||||
|
Quorum commit mode
|
||||||
|
==================
|
||||||
|
|
||||||
|
Starting from PostgreSQL v10 Patroni supports quorum-based synchronous replication.
|
||||||
|
|
||||||
|
In this mode, Patroni maintains synchronization state in the DCS, containing the latest known primary, the number of nodes required for quorum, and the nodes currently eligible to vote on quorum. In steady state, the nodes voting on quorum are the leader and all synchronous standbys. This state is updated with strict ordering constraints, with regards to node promotion and ``synchronous_standby_names``, to ensure that at all times any subset of voters that can achieve quorum includes at least one node with the latest successful commit.
|
||||||
|
|
||||||
|
On each iteration of HA loop, Patroni re-evaluates synchronous standby choices and quorum, based on node availability and requested cluster configuration. In PostgreSQL versions above 9.6 all eligible nodes are added as synchronous standbys as soon as their replication catches up to leader.
|
||||||
|
|
||||||
|
Quorum commit helps to reduce worst case latencies, even during normal operation, as a higher latency of replicating to one standby can be compensated by other standbys.
|
||||||
|
|
||||||
|
The quorum-based synchronous mode could be enabled by setting ``synchronous_mode`` to ``quorum`` using ``patronictl edit-config`` command or via Patroni REST interface. See :ref:`dynamic configuration <dynamic_configuration>` for instructions.
|
||||||
|
|
||||||
|
Other parameters, like ``synchronous_node_count``, ``maximum_lag_on_syncnode``, and ``synchronous_mode_strict`` continue to work the same way as with ``synchronous_mode=on``.
|
||||||
|
|
||||||
|
Example:
|
||||||
|
---------
|
||||||
|
|
||||||
|
``/config`` key in DCS
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
.. code-block:: YAML
|
||||||
|
|
||||||
|
synchronous_mode: quorum
|
||||||
|
synchronous_node_count: 2
|
||||||
|
...
|
||||||
|
|
||||||
|
``/sync`` key in DCS
|
||||||
|
^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
.. code-block:: JSON
|
||||||
|
|
||||||
|
{
|
||||||
|
"leader": "node0",
|
||||||
|
"sync_standby": "node1,node2,node3",
|
||||||
|
"quorum": 1
|
||||||
|
}
|
||||||
|
|
||||||
|
postgresql.conf
|
||||||
|
^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
.. code-block:: INI
|
||||||
|
|
||||||
|
synchronous_standby_names = 'ANY 2 (node1,node2,node3)'
|
||||||
|
|
||||||
|
|
||||||
|
If the primary (``node0``) failed, in the above example two of the ``node1``, ``node2``, ``node3`` will have the latest transaction received, but we don't know which ones. To figure out whether the node ``node1`` has received the latest transaction, we need to compare its LSN with the LSN on **at least** one node (``quorum=1`` in the ``/sync`` key) among ``node2`` and ``node3``. If ``node1`` isn't behind of at least one of them, we can guarantee that there will be no user visible data loss if ``node1`` is promoted.
|
||||||
|
|
||||||
|
|
||||||
|
.. [1] The data is still there, but recovering it requires a manual recovery effort by data recovery specialists. When Patroni is allowed to rewind with ``use_pg_rewind`` the forked timeline will be automatically erased to rejoin the failed primary with the cluster. However, for ``use_pg_rewind`` to function properly, either the cluster must be initialized with ``data page checksums`` (``--data-checksums`` option for ``initdb``) and/or ``wal_log_hints`` must be set to ``on``.
|
||||||
|
|
||||||
.. [2] Clients can change the behavior per transaction using PostgreSQL's ``synchronous_commit`` setting. Transactions with ``synchronous_commit`` values of ``off`` and ``local`` may be lost on fail over, but will not be blocked by replication delays.
|
.. [2] Clients can change the behavior per transaction using PostgreSQL's ``synchronous_commit`` setting. Transactions with ``synchronous_commit`` values of ``off`` and ``local`` may be lost on fail over, but will not be blocked by replication delays.
|
||||||
|
|||||||
+39
-26
@@ -45,6 +45,10 @@ For all health check ``GET`` requests Patroni returns a JSON document with the s
|
|||||||
|
|
||||||
- ``GET /read-only-sync``: like the above endpoint, but also includes the primary.
|
- ``GET /read-only-sync``: like the above endpoint, but also includes the primary.
|
||||||
|
|
||||||
|
- ``GET /quorum``: returns HTTP status code **200** only when this Patroni node is listed as a quorum node in ``synchronous_standby_names`` on the primary.
|
||||||
|
|
||||||
|
- ``GET /read-only-quorum``: like the above endpoint, but also includes the primary.
|
||||||
|
|
||||||
- ``GET /asynchronous`` or ``GET /async``: returns HTTP status code **200** only when the Patroni node is running as an asynchronous standby.
|
- ``GET /asynchronous`` or ``GET /async``: returns HTTP status code **200** only when the Patroni node is running as an asynchronous standby.
|
||||||
|
|
||||||
|
|
||||||
@@ -99,9 +103,9 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
|||||||
$ curl -s http://localhost:8008/patroni | jq .
|
$ curl -s http://localhost:8008/patroni | jq .
|
||||||
{
|
{
|
||||||
"state": "running",
|
"state": "running",
|
||||||
"postmaster_start_time": "2023-08-18 11:03:37.966359+00:00",
|
"postmaster_start_time": "2024-08-28 19:39:26.352526+00:00",
|
||||||
"role": "master",
|
"role": "primary",
|
||||||
"server_version": 150004,
|
"server_version": 160004,
|
||||||
"xlog": {
|
"xlog": {
|
||||||
"location": 67395656
|
"location": 67395656
|
||||||
},
|
},
|
||||||
@@ -130,7 +134,7 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
|||||||
},
|
},
|
||||||
"database_system_identifier": "7268616322854375442",
|
"database_system_identifier": "7268616322854375442",
|
||||||
"patroni": {
|
"patroni": {
|
||||||
"version": "3.1.0",
|
"version": "4.0.0",
|
||||||
"scope": "demo",
|
"scope": "demo",
|
||||||
"name": "patroni1"
|
"name": "patroni1"
|
||||||
}
|
}
|
||||||
@@ -143,9 +147,9 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
|||||||
$ curl -s http://localhost:8008/patroni | jq .
|
$ curl -s http://localhost:8008/patroni | jq .
|
||||||
{
|
{
|
||||||
"state": "running",
|
"state": "running",
|
||||||
"postmaster_start_time": "2023-08-18 11:09:08.615242+00:00",
|
"postmaster_start_time": "2024-08-28 19:39:26.352526+00:00",
|
||||||
"role": "replica",
|
"role": "replica",
|
||||||
"server_version": 150004,
|
"server_version": 160004,
|
||||||
"xlog": {
|
"xlog": {
|
||||||
"received_location": 67419744,
|
"received_location": 67419744,
|
||||||
"replayed_location": 67419744,
|
"replayed_location": 67419744,
|
||||||
@@ -178,7 +182,7 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
|||||||
},
|
},
|
||||||
"database_system_identifier": "7268616322854375442",
|
"database_system_identifier": "7268616322854375442",
|
||||||
"patroni": {
|
"patroni": {
|
||||||
"version": "3.1.0",
|
"version": "4.0.0",
|
||||||
"scope": "demo",
|
"scope": "demo",
|
||||||
"name": "patroni1"
|
"name": "patroni1"
|
||||||
}
|
}
|
||||||
@@ -191,9 +195,9 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
|||||||
$ curl -s http://localhost:8008/patroni | jq .
|
$ curl -s http://localhost:8008/patroni | jq .
|
||||||
{
|
{
|
||||||
"state": "running",
|
"state": "running",
|
||||||
"postmaster_start_time": "2023-08-18 11:09:08.615242+00:00",
|
"postmaster_start_time": "2024-08-28 19:39:26.352526+00:00",
|
||||||
"role": "replica",
|
"role": "replica",
|
||||||
"server_version": 150004,
|
"server_version": 160004,
|
||||||
"xlog": {
|
"xlog": {
|
||||||
"location": 67420024
|
"location": 67420024
|
||||||
},
|
},
|
||||||
@@ -224,7 +228,7 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
|||||||
},
|
},
|
||||||
"database_system_identifier": "7268616322854375442",
|
"database_system_identifier": "7268616322854375442",
|
||||||
"patroni": {
|
"patroni": {
|
||||||
"version": "3.1.0",
|
"version": "4.0.0",
|
||||||
"scope": "demo",
|
"scope": "demo",
|
||||||
"name": "patroni1"
|
"name": "patroni1"
|
||||||
}
|
}
|
||||||
@@ -237,9 +241,9 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
|||||||
$ curl -s http://localhost:8008/patroni | jq .
|
$ curl -s http://localhost:8008/patroni | jq .
|
||||||
{
|
{
|
||||||
"state": "running",
|
"state": "running",
|
||||||
"postmaster_start_time": "2023-08-18 11:09:08.615242+00:00",
|
"postmaster_start_time": "2024-08-28 19:39:26.352526+00:00",
|
||||||
"role": "replica",
|
"role": "replica",
|
||||||
"server_version": 150004,
|
"server_version": 160004,
|
||||||
"xlog": {
|
"xlog": {
|
||||||
"location": 67420024
|
"location": 67420024
|
||||||
},
|
},
|
||||||
@@ -263,13 +267,13 @@ The ``GET /patroni`` is used by Patroni during the leader race. It also could be
|
|||||||
}
|
}
|
||||||
],
|
],
|
||||||
"pause": true,
|
"pause": true,
|
||||||
"dcs_last_seen": 1692356928,
|
"dcs_last_seen": 1724874295,
|
||||||
"tags": {
|
"tags": {
|
||||||
"clonefrom": true
|
"clonefrom": true
|
||||||
},
|
},
|
||||||
"database_system_identifier": "7268616322854375442",
|
"database_system_identifier": "7268616322854375442",
|
||||||
"patroni": {
|
"patroni": {
|
||||||
"version": "3.1.0",
|
"version": "4.0.0",
|
||||||
"scope": "demo",
|
"scope": "demo",
|
||||||
"name": "patroni1"
|
"name": "patroni1"
|
||||||
}
|
}
|
||||||
@@ -283,16 +287,13 @@ Retrieve the Patroni metrics in Prometheus format through the ``GET /metrics`` e
|
|||||||
|
|
||||||
# HELP patroni_version Patroni semver without periods. \
|
# HELP patroni_version Patroni semver without periods. \
|
||||||
# TYPE patroni_version gauge
|
# TYPE patroni_version gauge
|
||||||
patroni_version{scope="batman",name="patroni1"} 020103
|
patroni_version{scope="batman",name="patroni1"} 040000
|
||||||
# HELP patroni_postgres_running Value is 1 if Postgres is running, 0 otherwise.
|
# HELP patroni_postgres_running Value is 1 if Postgres is running, 0 otherwise.
|
||||||
# TYPE patroni_postgres_running gauge
|
# TYPE patroni_postgres_running gauge
|
||||||
patroni_postgres_running{scope="batman",name="patroni1"} 1
|
patroni_postgres_running{scope="batman",name="patroni1"} 1
|
||||||
# HELP patroni_postmaster_start_time Epoch seconds since Postgres started.
|
# HELP patroni_postmaster_start_time Epoch seconds since Postgres started.
|
||||||
# TYPE patroni_postmaster_start_time gauge
|
# TYPE patroni_postmaster_start_time gauge
|
||||||
patroni_postmaster_start_time{scope="batman",name="patroni1"} 1657656955.179243
|
patroni_postmaster_start_time{scope="batman",name="patroni1"} 1724873966.352526
|
||||||
# HELP patroni_master Value is 1 if this node is the leader, 0 otherwise.
|
|
||||||
# TYPE patroni_master gauge
|
|
||||||
patroni_master{scope="batman",name="patroni1"} 1
|
|
||||||
# HELP patroni_primary Value is 1 if this node is the leader, 0 otherwise.
|
# HELP patroni_primary Value is 1 if this node is the leader, 0 otherwise.
|
||||||
# TYPE patroni_primary gauge
|
# TYPE patroni_primary gauge
|
||||||
patroni_primary{scope="batman",name="patroni1"} 1
|
patroni_primary{scope="batman",name="patroni1"} 1
|
||||||
@@ -308,6 +309,9 @@ Retrieve the Patroni metrics in Prometheus format through the ``GET /metrics`` e
|
|||||||
# HELP patroni_sync_standby Value is 1 if this node is a sync standby replica, 0 otherwise.
|
# HELP patroni_sync_standby Value is 1 if this node is a sync standby replica, 0 otherwise.
|
||||||
# TYPE patroni_sync_standby gauge
|
# TYPE patroni_sync_standby gauge
|
||||||
patroni_sync_standby{scope="batman",name="patroni1"} 0
|
patroni_sync_standby{scope="batman",name="patroni1"} 0
|
||||||
|
# HELP patroni_quorum_standby Value is 1 if this node is a quorum standby replica, 0 otherwise.
|
||||||
|
# TYPE patroni_quorum_standby gauge
|
||||||
|
patroni_quorum_standby{scope="batman",name="patroni1"} 0
|
||||||
# HELP patroni_xlog_received_location Current location of the received Postgres transaction log, 0 if this node is not a replica.
|
# HELP patroni_xlog_received_location Current location of the received Postgres transaction log, 0 if this node is not a replica.
|
||||||
# TYPE patroni_xlog_received_location counter
|
# TYPE patroni_xlog_received_location counter
|
||||||
patroni_xlog_received_location{scope="batman",name="patroni1"} 0
|
patroni_xlog_received_location{scope="batman",name="patroni1"} 0
|
||||||
@@ -328,7 +332,7 @@ Retrieve the Patroni metrics in Prometheus format through the ``GET /metrics`` e
|
|||||||
patroni_postgres_in_archive_recovery{scope="batman",name="patroni1"} 0
|
patroni_postgres_in_archive_recovery{scope="batman",name="patroni1"} 0
|
||||||
# HELP patroni_postgres_server_version Version of Postgres (if running), 0 otherwise.
|
# HELP patroni_postgres_server_version Version of Postgres (if running), 0 otherwise.
|
||||||
# TYPE patroni_postgres_server_version gauge
|
# TYPE patroni_postgres_server_version gauge
|
||||||
patroni_postgres_server_version{scope="batman",name="patroni1"} 140004
|
patroni_postgres_server_version{scope="batman",name="patroni1"} 160004
|
||||||
# HELP patroni_cluster_unlocked Value is 1 if the cluster is unlocked, 0 if locked.
|
# HELP patroni_cluster_unlocked Value is 1 if the cluster is unlocked, 0 if locked.
|
||||||
# TYPE patroni_cluster_unlocked gauge
|
# TYPE patroni_cluster_unlocked gauge
|
||||||
patroni_cluster_unlocked{scope="batman",name="patroni1"} 0
|
patroni_cluster_unlocked{scope="batman",name="patroni1"} 0
|
||||||
@@ -340,7 +344,7 @@ Retrieve the Patroni metrics in Prometheus format through the ``GET /metrics`` e
|
|||||||
patroni_postgres_timeline{scope="batman",name="patroni1"} 24
|
patroni_postgres_timeline{scope="batman",name="patroni1"} 24
|
||||||
# HELP patroni_dcs_last_seen Epoch timestamp when DCS was last contacted successfully by Patroni.
|
# HELP patroni_dcs_last_seen Epoch timestamp when DCS was last contacted successfully by Patroni.
|
||||||
# TYPE patroni_dcs_last_seen gauge
|
# TYPE patroni_dcs_last_seen gauge
|
||||||
patroni_dcs_last_seen{scope="batman",name="patroni1"} 1677658321
|
patroni_dcs_last_seen{scope="batman",name="patroni1"} 1724874235
|
||||||
# HELP patroni_pending_restart Value is 1 if the node needs a restart, 0 otherwise.
|
# HELP patroni_pending_restart Value is 1 if the node needs a restart, 0 otherwise.
|
||||||
# TYPE patroni_pending_restart gauge
|
# TYPE patroni_pending_restart gauge
|
||||||
patroni_pending_restart{scope="batman",name="patroni1"} 1
|
patroni_pending_restart{scope="batman",name="patroni1"} 1
|
||||||
@@ -488,20 +492,29 @@ Let's check that the node processed this configuration. First of all it should s
|
|||||||
|
|
||||||
$ curl -s http://localhost:8008/patroni | jq .
|
$ curl -s http://localhost:8008/patroni | jq .
|
||||||
{
|
{
|
||||||
"pending_restart": true,
|
|
||||||
"database_system_identifier": "6287881213849985952",
|
"database_system_identifier": "6287881213849985952",
|
||||||
"postmaster_start_time": "2016-06-13 13:13:05.211 CEST",
|
"postmaster_start_time": "2024-08-28 19:39:26.352526+00:00",
|
||||||
"xlog": {
|
"xlog": {
|
||||||
"location": 2197818976
|
"location": 2197818976
|
||||||
},
|
},
|
||||||
|
"timeline": 1,
|
||||||
|
"dcs_last_seen": 1724874545,
|
||||||
|
"database_system_identifier": "7408277255830290455",
|
||||||
|
"pending_restart": true,
|
||||||
|
"pending_restart_reason": {
|
||||||
|
"max_connections": {
|
||||||
|
"old_value": "100",
|
||||||
|
"new_value": "101"
|
||||||
|
}
|
||||||
|
},
|
||||||
"patroni": {
|
"patroni": {
|
||||||
"version": "1.0",
|
"version": "4.0.0",
|
||||||
"scope": "batman",
|
"scope": "batman",
|
||||||
"name": "patroni1"
|
"name": "patroni1"
|
||||||
},
|
},
|
||||||
"state": "running",
|
"state": "running",
|
||||||
"role": "master",
|
"role": "primary",
|
||||||
"server_version": 90503
|
"server_version": 160004
|
||||||
}
|
}
|
||||||
|
|
||||||
Removing parameters:
|
Removing parameters:
|
||||||
|
|||||||
@@ -0,0 +1,82 @@
|
|||||||
|
.. _standby_cluster:
|
||||||
|
|
||||||
|
Standby cluster
|
||||||
|
---------------
|
||||||
|
|
||||||
|
Patroni also support running cascading replication to a remote datacenter
|
||||||
|
(region) using a feature that is called "standby cluster". This type of
|
||||||
|
clusters has:
|
||||||
|
|
||||||
|
* "standby leader", that behaves pretty much like a regular cluster leader,
|
||||||
|
except it replicates from a remote node.
|
||||||
|
|
||||||
|
* cascade replicas, that are replicating from standby leader.
|
||||||
|
|
||||||
|
Standby leader holds and updates a leader lock in DCS. If the leader lock
|
||||||
|
expires, cascade replicas will perform an election to choose another leader
|
||||||
|
from the standbys.
|
||||||
|
|
||||||
|
There is no further relationship between the standby cluster and the primary
|
||||||
|
cluster it replicates from, in particular, they must not share the same DCS
|
||||||
|
scope if they use the same DCS. They do not know anything else from each other
|
||||||
|
apart from replication information. Also, the standby cluster is not being
|
||||||
|
displayed in :ref:`patronictl_list` or :ref:`patronictl_topology` output on the
|
||||||
|
primary cluster.
|
||||||
|
|
||||||
|
For the sake of flexibility, you can specify methods of creating a replica and
|
||||||
|
recovery WAL records when a cluster is in the "standby mode" by providing
|
||||||
|
:ref:`create_replica_methods <custom_replica_creation>` key in
|
||||||
|
`standby_cluster` section. It is distinct from creating replicas, when cluster
|
||||||
|
is detached and functions as a normal cluster, which is controlled by
|
||||||
|
`create_replica_methods` in `postgresql` section. Both "standby" and "normal"
|
||||||
|
`create_replica_methods` reference keys in `postgresql` section.
|
||||||
|
|
||||||
|
To configure such cluster you need to specify the section ``standby_cluster``
|
||||||
|
in a patroni configuration:
|
||||||
|
|
||||||
|
.. code:: YAML
|
||||||
|
|
||||||
|
bootstrap:
|
||||||
|
dcs:
|
||||||
|
standby_cluster:
|
||||||
|
host: 1.2.3.4
|
||||||
|
port: 5432
|
||||||
|
primary_slot_name: patroni
|
||||||
|
create_replica_methods:
|
||||||
|
- basebackup
|
||||||
|
|
||||||
|
Note, that these options will be applied only once during cluster bootstrap,
|
||||||
|
and the only way to change them afterwards is through DCS.
|
||||||
|
|
||||||
|
Patroni expects to find `postgresql.conf` or `postgresql.conf.backup` in PGDATA
|
||||||
|
of the remote primary and will not start if it does not find it after a
|
||||||
|
basebackup. If the remote primary keeps its `postgresql.conf` elsewhere, it is
|
||||||
|
your responsibility to copy it to PGDATA.
|
||||||
|
|
||||||
|
If you use replication slots on the standby cluster, you must also create the
|
||||||
|
corresponding replication slot on the primary cluster. It will not be done
|
||||||
|
automatically by the standby cluster implementation. You can use Patroni's
|
||||||
|
permanent replication slots feature on the primary cluster to maintain a
|
||||||
|
replication slot with the same name as ``primary_slot_name``, or its default
|
||||||
|
value if ``primary_slot_name`` is not provided.
|
||||||
|
|
||||||
|
In case the remote site doesn't provide a single endpoint that connects to a
|
||||||
|
primary, one could list all hosts of the source cluster in the
|
||||||
|
``standby_cluster.host`` section. When ``standby_cluster.host`` contains
|
||||||
|
multiple hosts separated by commas, Patroni will:
|
||||||
|
|
||||||
|
* add ``target_session_attrs=read-write`` to the ``primary_conninfo`` on the
|
||||||
|
standby leader node.
|
||||||
|
* use ``target_session_attrs=read-write`` when trying to determine whether we
|
||||||
|
need to run ``pg_rewind`` or when executing ``pg_rewind`` on all nodes of the
|
||||||
|
standby cluster.
|
||||||
|
* It is important to note that for ``pg_rewind`` to operate successfully,
|
||||||
|
either the cluster must be initialized with ``data page checksums``
|
||||||
|
(``--data-checksums`` option for ``initdb``) and/or ``wal_log_hints`` must be set to ``on``.
|
||||||
|
Otherwise, ``pg_rewind`` will not function properly.
|
||||||
|
|
||||||
|
There is also a possibility to replicate the standby cluster from another
|
||||||
|
standby cluster or from a standby member of the primary cluster: for that, you
|
||||||
|
need to define a single host in the ``standby_cluster.host`` section. However,
|
||||||
|
you need to beware that in this case ``pg_rewind`` will fail to execute on the
|
||||||
|
standby cluster.
|
||||||
@@ -0,0 +1,64 @@
|
|||||||
|
.. _tools_integration:
|
||||||
|
|
||||||
|
Integration with other tools
|
||||||
|
============================
|
||||||
|
|
||||||
|
Patroni is able to integrate with other tools in your stack. In this section you
|
||||||
|
will find a list of examples, which although not an exhaustive list, might
|
||||||
|
provide you with ideas on how Patroni can integrate with other tools.
|
||||||
|
|
||||||
|
Barman
|
||||||
|
------
|
||||||
|
|
||||||
|
Patroni delivers an application named ``patroni_barman`` which has logic to
|
||||||
|
communicate with ``pg-backup-api``, so you are able to perform Barman operations
|
||||||
|
remotely.
|
||||||
|
|
||||||
|
This application currently has a couple of sub-commands: ``recover`` and
|
||||||
|
``config-switch``.
|
||||||
|
|
||||||
|
patroni_barman recover
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
The ``recover`` sub-command can be used as a custom bootstrap or custom replica
|
||||||
|
creation method. You can find more information about that in
|
||||||
|
:ref:`replica_imaging_and_bootstrap`.
|
||||||
|
|
||||||
|
patroni_barman config-switch
|
||||||
|
^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
||||||
|
|
||||||
|
The ``config-switch`` sub-command is designed to be used as an ``on_role_change``
|
||||||
|
callback in Patroni. As an example, assume you are streaming WALs from your
|
||||||
|
current primary to your Barman host. In the event of a failover in the cluster
|
||||||
|
you might want to start streaming WALs from the new primary. You can accomplish
|
||||||
|
this by using ``patroni_barman config-switch`` as the ``on_role_change`` callback.
|
||||||
|
|
||||||
|
.. note::
|
||||||
|
That sub-command relies on the ``barman config-switch`` command, which is in
|
||||||
|
charge of overriding the configuration of a Barman server by applying a
|
||||||
|
pre-defined model on top of it. This command is available since Barman 3.10.
|
||||||
|
Please consult the Barman documentation for more details.
|
||||||
|
|
||||||
|
This is an example of how you can configure Patroni to apply a configuration
|
||||||
|
model in case this Patroni node is promoted to primary:
|
||||||
|
|
||||||
|
.. code:: YAML
|
||||||
|
|
||||||
|
postgresql:
|
||||||
|
callbacks:
|
||||||
|
on_role_change: >
|
||||||
|
patroni_barman
|
||||||
|
--api-url YOUR_API_URL
|
||||||
|
config-switch
|
||||||
|
--barman-server YOUR_BARMAN_SERVER_NAME
|
||||||
|
--barman-model YOUR_BARMAN_MODEL_NAME
|
||||||
|
--switch-when promoted
|
||||||
|
|
||||||
|
.. note::
|
||||||
|
``patroni_barman config-switch`` requires that you have both Barman and
|
||||||
|
``pg-backup-api`` configured in the Barman host, so it can execute a remote
|
||||||
|
``barman config-switch`` through the backup API. Also, it requires that you
|
||||||
|
have pre-configured Barman models to be applied. The above example uses a
|
||||||
|
subset of the available parameters. You can get more information running
|
||||||
|
``patroni_barman config-switch --help``, and by consulting the Barman
|
||||||
|
documentation.
|
||||||
+25
-14
@@ -29,12 +29,17 @@ Log
|
|||||||
- **static_fields**: add additional fields to the log. This option is only available when the log type is set to **json**.
|
- **static_fields**: add additional fields to the log. This option is only available when the log type is set to **json**.
|
||||||
- **max\_queue\_size**: Patroni is using two-step logging. Log records are written into the in-memory queue and there is a separate thread which pulls them from the queue and writes to stderr or file. The maximum size of the internal queue is limited by default by **1000** records, which is enough to keep logs for the past 1h20m.
|
- **max\_queue\_size**: Patroni is using two-step logging. Log records are written into the in-memory queue and there is a separate thread which pulls them from the queue and writes to stderr or file. The maximum size of the internal queue is limited by default by **1000** records, which is enough to keep logs for the past 1h20m.
|
||||||
- **dir**: Directory to write application logs to. The directory must exist and be writable by the user executing Patroni. If you set this value, the application will retain 4 25MB logs by default. You can tune those retention values with `file_num` and `file_size` (see below).
|
- **dir**: Directory to write application logs to. The directory must exist and be writable by the user executing Patroni. If you set this value, the application will retain 4 25MB logs by default. You can tune those retention values with `file_num` and `file_size` (see below).
|
||||||
|
- **mode**: Permissions for log files (for example, ``0644``). If not specified, permissions will be set based on the current umask value.
|
||||||
- **file\_num**: The number of application logs to retain.
|
- **file\_num**: The number of application logs to retain.
|
||||||
- **file\_size**: Size of patroni.log file (in bytes) that triggers a log rolling.
|
- **file\_size**: Size of patroni.log file (in bytes) that triggers a log rolling.
|
||||||
- **loggers**: This section allows redefining logging level per python module
|
- **loggers**: This section allows redefining logging level per python module
|
||||||
|
|
||||||
- **patroni.postmaster: WARNING**
|
- **patroni.postmaster: WARNING**
|
||||||
- **urllib3: DEBUG**
|
- **urllib3: DEBUG**
|
||||||
|
- **deduplicate_heartbeat_logs**: If set to ``true``, successive heartbeat logs that are identical shall not be output. Default value is ``false``.
|
||||||
|
|
||||||
|
.. warning::
|
||||||
|
The time the HA loop executes at can be very valuable information in diagnosing failovers due to resource exhaustion and similar problems. When ``deduplicate_heartbeat_logs`` is set to ``true`` there will be no log generated for the HA loop execution (unless the leader changes) and hence this potentially useful information will not be available from the logs.
|
||||||
|
|
||||||
Here is an example of how to config patroni to log in json format.
|
Here is an example of how to config patroni to log in json format.
|
||||||
|
|
||||||
@@ -103,9 +108,9 @@ Most of the parameters are optional, but you have to specify one of the **host**
|
|||||||
- **consistency**: (optional) Select consul consistency mode. Possible values are ``default``, ``consistent``, or ``stale`` (more details in `consul API reference <https://www.consul.io/api/features/consistency.html/>`__)
|
- **consistency**: (optional) Select consul consistency mode. Possible values are ``default``, ``consistent``, or ``stale`` (more details in `consul API reference <https://www.consul.io/api/features/consistency.html/>`__)
|
||||||
- **checks**: (optional) list of Consul health checks used for the session. By default an empty list is used.
|
- **checks**: (optional) list of Consul health checks used for the session. By default an empty list is used.
|
||||||
- **register\_service**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, primary, replica, or standby-leader depending on the node's role. Defaults to **false**.
|
- **register\_service**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, primary, replica, or standby-leader depending on the node's role. Defaults to **false**.
|
||||||
- **service\_tags**: (optional) additional static tags to add to the Consul service apart from the role (``master``/``primary``/``replica``/``standby-leader``). By default an empty list is used.
|
- **service\_tags**: (optional) additional static tags to add to the Consul service apart from the role (``primary``/``replica``/``standby-leader``). By default an empty list is used.
|
||||||
- **service\_check\_interval**: (optional) how often to perform health check against registered url. Defaults to '5s'.
|
- **service\_check\_interval**: (optional) how often to perform health check against registered url. Defaults to '5s'.
|
||||||
- **service\_check\_tls\_server\_name**: (optional) overide SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
|
- **service\_check\_tls\_server\_name**: (optional) override SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
|
||||||
|
|
||||||
The ``token`` needs to have the following ACL permissions:
|
The ``token`` needs to have the following ACL permissions:
|
||||||
|
|
||||||
@@ -144,7 +149,7 @@ Etcdv3
|
|||||||
If you want that Patroni works with Etcd cluster via protocol version 3, you need to use the ``etcd3`` section in the Patroni configuration file. All configuration parameters are the same as for ``etcd``.
|
If you want that Patroni works with Etcd cluster via protocol version 3, you need to use the ``etcd3`` section in the Patroni configuration file. All configuration parameters are the same as for ``etcd``.
|
||||||
|
|
||||||
.. warning::
|
.. warning::
|
||||||
Keys created with protocol version 2 are not visible with protocol version 3 and the other way around, therefore it is not possible to switch from ``etcd`` to ``etcd3`` just by updating Patroni config file.
|
Keys created with protocol version 2 are not visible with protocol version 3 and the other way around, therefore it is not possible to switch from ``etcd`` to ``etcd3`` just by updating Patroni config file. In addition, Patroni uses Etcd's gRPC-gateway (proxy) to communicate with the V3 API, which means that TLS common name authentication is not possible.
|
||||||
|
|
||||||
|
|
||||||
ZooKeeper
|
ZooKeeper
|
||||||
@@ -177,11 +182,12 @@ Kubernetes
|
|||||||
- **namespace**: (optional) Kubernetes namespace where Patroni pod is running. Default value is `default`.
|
- **namespace**: (optional) Kubernetes namespace where Patroni pod is running. Default value is `default`.
|
||||||
- **labels**: Labels in format ``{label1: value1, label2: value2}``. These labels will be used to find existing objects (Pods and either Endpoints or ConfigMaps) associated with the current cluster. Also Patroni will set them on every object (Endpoint or ConfigMap) it creates.
|
- **labels**: Labels in format ``{label1: value1, label2: value2}``. These labels will be used to find existing objects (Pods and either Endpoints or ConfigMaps) associated with the current cluster. Also Patroni will set them on every object (Endpoint or ConfigMap) it creates.
|
||||||
- **scope\_label**: (optional) name of the label containing cluster name. Default value is `cluster-name`.
|
- **scope\_label**: (optional) name of the label containing cluster name. Default value is `cluster-name`.
|
||||||
- **role\_label**: (optional) name of the label containing role (master or replica or other custom value). Patroni will set this label on the pod it runs in. Default value is ``role``.
|
- **bootstrap\_labels**: (optional) Labels in format ``{label1: value1, label2: value2}``. These labels will be assigned to a Patroni pod when its state is either ``initializing new cluster``, ``running custom bootstrap script``, ``starting after custom bootstrap`` or ``creating replica``.
|
||||||
- **leader\_label\_value**: (optional) value of the pod label when Postgres role is ``master``. Default value is ``master``.
|
- **role\_label**: (optional) name of the label containing role (`primary`, `replica`, or other custom value). Patroni will set this label on the pod it runs in. Default value is ``role``.
|
||||||
|
- **leader\_label\_value**: (optional) value of the pod label when Postgres role is ``primary``. Default value is ``primary``.
|
||||||
- **follower\_label\_value**: (optional) value of the pod label when Postgres role is ``replica``. Default value is ``replica``.
|
- **follower\_label\_value**: (optional) value of the pod label when Postgres role is ``replica``. Default value is ``replica``.
|
||||||
- **standby\_leader\_label\_value**: (optional) value of the pod label when Postgres role is ``standby_leader``. Default value is ``master``.
|
- **standby\_leader\_label\_value**: (optional) value of the pod label when Postgres role is ``standby_leader``. Default value is ``primary``.
|
||||||
- **tmp_\role\_label**: (optional) name of the temporary label containing role (master or replica). Value of this label will always use the default of corresponding role. Set only when necessary.
|
- **tmp\_role\_label**: (optional) name of the temporary label containing role (`primary` or `replica`). Value of this label will always use the default of corresponding role. Set only when necessary.
|
||||||
- **use\_endpoints**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
- **use\_endpoints**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||||
- **pod\_ip**: (optional) IP address of the pod Patroni is running in. This value is required when `use_endpoints` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
- **pod\_ip**: (optional) IP address of the pod Patroni is running in. This value is required when `use_endpoints` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||||
- **ports**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``kubernetes.ports: [{"name": "postgresql", "port": 5432}]`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `kubernetes.use_endpoints` is set.
|
- **ports**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``kubernetes.ports: [{"name": "postgresql", "port": 5432}]`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `kubernetes.use_endpoints` is set.
|
||||||
@@ -238,9 +244,10 @@ PostgreSQL
|
|||||||
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||||
- **sslpassword**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``sslkey``.
|
- **sslpassword**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``sslkey``.
|
||||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one or more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **sslnegotiation**: (optional) maps to the `sslnegotiation <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLNEGOTIATION>`__ connection parameter, which controls how SSL encryption is negotiated with the server, if SSL is used.
|
||||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
- **replication**:
|
- **replication**:
|
||||||
@@ -251,9 +258,10 @@ PostgreSQL
|
|||||||
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||||
- **sslpassword**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``sslkey``.
|
- **sslpassword**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``sslkey``.
|
||||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one or more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **sslnegotiation**: (optional) maps to the `sslnegotiation <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLNEGOTIATION>`__ connection parameter, which controls how SSL encryption is negotiated with the server, if SSL is used.
|
||||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
- **rewind**:
|
- **rewind**:
|
||||||
@@ -264,9 +272,10 @@ PostgreSQL
|
|||||||
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
- **sslkey**: (optional) maps to the `sslkey <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLKEY>`__ connection parameter, which specifies the location of the secret key used with the client's certificate.
|
||||||
- **sslpassword**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``sslkey``.
|
- **sslpassword**: (optional) maps to the `sslpassword <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLPASSWORD>`__ connection parameter, which specifies the password for the secret key specified in ``sslkey``.
|
||||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one or more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **sslnegotiation**: (optional) maps to the `sslnegotiation <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLNEGOTIATION>`__ connection parameter, which controls how SSL encryption is negotiated with the server, if SSL is used.
|
||||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
|
|
||||||
@@ -300,7 +309,7 @@ PostgreSQL
|
|||||||
- **pgpass**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup, the post_init script and under some other circumstances. The location must be writable by Patroni.
|
- **pgpass**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup, the post_init script and under some other circumstances. The location must be writable by Patroni.
|
||||||
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower.
|
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower.
|
||||||
- **custom\_conf** : path to an optional custom ``postgresql.conf`` file, that will be used in place of ``postgresql.base.conf``. The file must exist on all cluster nodes, be readable by PostgreSQL and will be included from its location on the real ``postgresql.conf``. Note that Patroni will not monitor this file for changes, nor backup it. However, its settings can still be overridden by Patroni's own configuration facilities - see :ref:`dynamic configuration <patroni_configuration>` for details.
|
- **custom\_conf** : path to an optional custom ``postgresql.conf`` file, that will be used in place of ``postgresql.base.conf``. The file must exist on all cluster nodes, be readable by PostgreSQL and will be included from its location on the real ``postgresql.conf``. Note that Patroni will not monitor this file for changes, nor backup it. However, its settings can still be overridden by Patroni's own configuration facilities - see :ref:`dynamic configuration <patroni_configuration>` for details.
|
||||||
- **parameters**: list of configuration settings for Postgres. Many of these are required for replication to work.
|
- **parameters**: configuration parameters (GUCs) for Postgres in format ``{ssl: "on", ssl_cert_file: "cert_file"}``.
|
||||||
- **pg\_hba**: list of lines that Patroni will use to generate ``pg_hba.conf``. Patroni ignores this parameter if ``hba_file`` PostgreSQL parameter is set to a non-default value. Together with :ref:`dynamic configuration <dynamic_configuration>` this parameter simplifies management of ``pg_hba.conf``.
|
- **pg\_hba**: list of lines that Patroni will use to generate ``pg_hba.conf``. Patroni ignores this parameter if ``hba_file`` PostgreSQL parameter is set to a non-default value. Together with :ref:`dynamic configuration <dynamic_configuration>` this parameter simplifies management of ``pg_hba.conf``.
|
||||||
|
|
||||||
- **- host all all 0.0.0.0/0 md5**
|
- **- host all all 0.0.0.0/0 md5**
|
||||||
@@ -310,7 +319,7 @@ PostgreSQL
|
|||||||
- **- mapname1 systemname1 pguser1**
|
- **- mapname1 systemname1 pguser1**
|
||||||
- **- mapname1 systemname2 pguser2**
|
- **- mapname1 systemname2 pguser2**
|
||||||
- **pg\_ctl\_timeout**: How long should pg_ctl wait when doing ``start``, ``stop`` or ``restart``. Default value is 60 seconds.
|
- **pg\_ctl\_timeout**: How long should pg_ctl wait when doing ``start``, ``stop`` or ``restart``. Default value is 60 seconds.
|
||||||
- **use\_pg\_rewind**: try to use pg\_rewind on the former leader when it joins cluster as a replica.
|
- **use\_pg\_rewind**: try to use pg\_rewind on the former leader when it joins cluster as a replica. Either the cluster must be initialized with ``data page checksums`` (``--data-checksums`` option for ``initdb``) and/or ``wal_log_hints`` must be set to ``on``, or ``pg_rewind`` will not work.
|
||||||
- **remove\_data\_directory\_on\_rewind\_failure**: If this option is enabled, Patroni will remove the PostgreSQL data directory and recreate the replica. Otherwise it will try to follow the new leader. Default value is **false**.
|
- **remove\_data\_directory\_on\_rewind\_failure**: If this option is enabled, Patroni will remove the PostgreSQL data directory and recreate the replica. Otherwise it will try to follow the new leader. Default value is **false**.
|
||||||
- **remove\_data\_directory\_on\_diverged\_timelines**: Patroni will remove the PostgreSQL data directory and recreate the replica if it notices that timelines are diverging and the former primary can not start streaming from the new primary. This option is useful when ``pg_rewind`` can not be used. While performing timelines divergence check on PostgreSQL v10 and older Patroni will try to connect with replication credential to the "postgres" database. Hence, such access should be allowed in the pg_hba.conf. Default value is **false**.
|
- **remove\_data\_directory\_on\_diverged\_timelines**: Patroni will remove the PostgreSQL data directory and recreate the replica if it notices that timelines are diverging and the former primary can not start streaming from the new primary. This option is useful when ``pg_rewind`` can not be used. While performing timelines divergence check on PostgreSQL v10 and older Patroni will try to connect with replication credential to the "postgres" database. Hence, such access should be allowed in the pg_hba.conf. Default value is **false**.
|
||||||
- **replica\_method**: for each create_replica_methods other than basebackup, you would add a configuration section of the same name. At a minimum, this should include "command" with a full path to the actual script to be executed. Other configuration parameters will be passed along to the script in the form "parameter=value".
|
- **replica\_method**: for each create_replica_methods other than basebackup, you would add a configuration section of the same name. At a minimum, this should include "command" with a full path to the actual script to be executed. Other configuration parameters will be passed along to the script in the form "parameter=value".
|
||||||
@@ -395,10 +404,12 @@ Tags
|
|||||||
----
|
----
|
||||||
- **clonefrom**: ``true`` or ``false``. If set to ``true`` other nodes might prefer to use this node for bootstrap (take ``pg_basebackup`` from). If there are several nodes with ``clonefrom`` tag set to ``true`` the node to bootstrap from will be chosen randomly. The default value is ``false``.
|
- **clonefrom**: ``true`` or ``false``. If set to ``true`` other nodes might prefer to use this node for bootstrap (take ``pg_basebackup`` from). If there are several nodes with ``clonefrom`` tag set to ``true`` the node to bootstrap from will be chosen randomly. The default value is ``false``.
|
||||||
- **noloadbalance**: ``true`` or ``false``. If set to ``true`` the node will return HTTP Status Code 503 for the ``GET /replica`` REST API health-check and therefore will be excluded from the load-balancing. Defaults to ``false``.
|
- **noloadbalance**: ``true`` or ``false``. If set to ``true`` the node will return HTTP Status Code 503 for the ``GET /replica`` REST API health-check and therefore will be excluded from the load-balancing. Defaults to ``false``.
|
||||||
- **replicatefrom**: The IP address/hostname of another replica. Used to support cascading replication.
|
- **replicatefrom**: The name of another replica to replicate from. Used to support cascading replication.
|
||||||
- **nosync**: ``true`` or ``false``. If set to ``true`` the node will never be selected as a synchronous replica.
|
- **nosync**: ``true`` or ``false``. If set to ``true`` the node will never be selected as a synchronous replica.
|
||||||
|
- **sync_priority**: integer, controls the priority this node should have during synchronous replica selection when ``synchronous_mode`` is set to ``on``. Nodes with higher priority will be preferred over lower-priority nodes. If the ``sync_priority`` is 0 or negative - such node is not allowed to be written to ``synchronous_standby_names`` PostgreSQL parameter (similar to ``nosync: true``). Keep in mind, that this parameter has the opposite meaning to ``sync_priority`` value reported in ``pg_stat_replication`` view.
|
||||||
- **nofailover**: ``true`` or ``false``, controls whether this node is allowed to participate in the leader race and become a leader. Defaults to ``false``, meaning this node _can_ participate in leader races.
|
- **nofailover**: ``true`` or ``false``, controls whether this node is allowed to participate in the leader race and become a leader. Defaults to ``false``, meaning this node _can_ participate in leader races.
|
||||||
- **failover_priority**: integer, controls the priority that this node should have during failover. Nodes with higher priority will be preferred over lower priority nodes if they received/replayed the same amount of WAL. However, nodes with higher values of receive/replay LSN are preferred regardless of their priority. If the ``failover_priority`` is 0 or negative - such node is not allowed to participate in the leader race and to become a leader (similar to ``nofailover: true``).
|
- **failover_priority**: integer, controls the priority this node should have during failover. Nodes with higher priority will be preferred over lower-priority nodes if they received/replayed the same amount of WAL. However, nodes with higher values of receive/replay LSN are preferred regardless of their priority. If the ``failover_priority`` is 0 or negative - such node is not allowed to participate in the leader race and to become a leader (similar to ``nofailover: true``).
|
||||||
|
- **nostream**: ``true`` or ``false``. If set to ``true`` the node will not use replication protocol to stream WAL. It will rely instead on archive recovery (if ``restore_command`` is configured) and ``pg_wal``/``pg_xlog`` polling. It also disables copying and synchronization of permanent logical replication slots on the node itself and all its cascading replicas. Setting this tag on primary node has no effect.
|
||||||
|
|
||||||
.. warning::
|
.. warning::
|
||||||
Provide only one of ``nofailover`` or ``failover_priority``. Providing ``nofailover: true`` is the same as ``failover_priority: 0``, and providing ``nofailover: false`` will give the node priority 1.
|
Provide only one of ``nofailover`` or ``failover_priority``. Providing ``nofailover: true`` is the same as ``failover_priority: 0``, and providing ``nofailover: false`` will give the node priority 1.
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ Description=Runners to orchestrate a high-availability PostgreSQL
|
|||||||
After=syslog.target network.target
|
After=syslog.target network.target
|
||||||
|
|
||||||
[Service]
|
[Service]
|
||||||
Type=simple
|
Type=notify
|
||||||
|
|
||||||
User=postgres
|
User=postgres
|
||||||
Group=postgres
|
Group=postgres
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
#!/usr/bin/env python
|
#!/usr/bin/env python
|
||||||
import os
|
|
||||||
import argparse
|
import argparse
|
||||||
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
@@ -3,12 +3,18 @@ import argparse
|
|||||||
import subprocess
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
|
from time import sleep
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
parser = argparse.ArgumentParser()
|
parser = argparse.ArgumentParser()
|
||||||
parser.add_argument("--datadir", required=True)
|
parser.add_argument("--datadir", required=True)
|
||||||
parser.add_argument("--dbname", required=True)
|
parser.add_argument("--dbname", required=True)
|
||||||
parser.add_argument("--walmethod", required=True, choices=("fetch", "stream", "none"))
|
parser.add_argument("--walmethod", required=True, choices=("fetch", "stream", "none"))
|
||||||
|
parser.add_argument("--sleep", required=False, type=int)
|
||||||
args, _ = parser.parse_known_args()
|
args, _ = parser.parse_known_args()
|
||||||
|
|
||||||
|
if args.sleep:
|
||||||
|
sleep(args.sleep)
|
||||||
|
|
||||||
walmethod = ["-X", args.walmethod] if args.walmethod != "none" else []
|
walmethod = ["-X", args.walmethod] if args.walmethod != "none" else []
|
||||||
sys.exit(subprocess.call(["pg_basebackup", "-D", args.datadir, "-c", "fast", "-d", args.dbname] + walmethod))
|
sys.exit(subprocess.call(["pg_basebackup", "-D", args.datadir, "-c", "fast", "-d", args.dbname] + walmethod))
|
||||||
|
|||||||
@@ -2,11 +2,17 @@
|
|||||||
import argparse
|
import argparse
|
||||||
import shutil
|
import shutil
|
||||||
|
|
||||||
|
from time import sleep
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
parser = argparse.ArgumentParser()
|
parser = argparse.ArgumentParser()
|
||||||
parser.add_argument("--datadir", required=True)
|
parser.add_argument("--datadir", required=True)
|
||||||
parser.add_argument("--sourcedir", required=True)
|
parser.add_argument("--sourcedir", required=True)
|
||||||
parser.add_argument("--test-argument", required=True)
|
parser.add_argument("--test-argument", required=True)
|
||||||
|
parser.add_argument("--sleep", required=False, type=int)
|
||||||
args, _ = parser.parse_known_args()
|
args, _ = parser.parse_known_args()
|
||||||
|
|
||||||
|
if args.sleep:
|
||||||
|
sleep(args.sleep)
|
||||||
|
|
||||||
shutil.copytree(args.sourcedir, args.datadir)
|
shutil.copytree(args.sourcedir, args.datadir)
|
||||||
|
|||||||
@@ -2,57 +2,57 @@ Feature: basic replication
|
|||||||
We should check that the basic bootstrapping, replication and failover works.
|
We should check that the basic bootstrapping, replication and failover works.
|
||||||
|
|
||||||
Scenario: check replication of a single table
|
Scenario: check replication of a single table
|
||||||
Given I start postgres0
|
Given I start postgres-0
|
||||||
Then postgres0 is a leader after 10 seconds
|
Then postgres-0 is a leader after 10 seconds
|
||||||
And there is a non empty initialize key in DCS after 15 seconds
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 20, "synchronous_mode": true}
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 20, "synchronous_mode": true}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
When I start postgres1
|
When I start postgres-1
|
||||||
And I configure and start postgres2 with a tag replicatefrom postgres0
|
And I configure and start postgres-2 with a tag replicatefrom postgres-0
|
||||||
And "sync" key in DCS has leader=postgres0 after 20 seconds
|
And "sync" key in DCS has leader=postgres-0 after 20 seconds
|
||||||
And I add the table foo to postgres0
|
And I add the table foo to postgres-0
|
||||||
Then table foo is present on postgres1 after 20 seconds
|
Then table foo is present on postgres-1 after 20 seconds
|
||||||
Then table foo is present on postgres2 after 20 seconds
|
Then table foo is present on postgres-2 after 20 seconds
|
||||||
|
|
||||||
Scenario: check restart of sync replica
|
Scenario: check restart of sync replica
|
||||||
Given I shut down postgres2
|
Given I shut down postgres-2
|
||||||
Then "sync" key in DCS has sync_standby=postgres1 after 5 seconds
|
Then "sync" key in DCS has sync_standby=postgres-1 after 5 seconds
|
||||||
When I start postgres2
|
When I start postgres-2
|
||||||
And I shut down postgres1
|
And I shut down postgres-1
|
||||||
Then "sync" key in DCS has sync_standby=postgres2 after 10 seconds
|
Then "sync" key in DCS has sync_standby=postgres-2 after 10 seconds
|
||||||
When I start postgres1
|
When I start postgres-1
|
||||||
Then "members/postgres1" key in DCS has state=running after 10 seconds
|
Then "members/postgres-1" key in DCS has state=running after 10 seconds
|
||||||
And Status code on GET http://127.0.0.1:8010/sync is 200 after 3 seconds
|
And Status code on GET http://127.0.0.1:8010/sync is 200 after 3 seconds
|
||||||
And Status code on GET http://127.0.0.1:8009/async is 200 after 3 seconds
|
And Status code on GET http://127.0.0.1:8009/async is 200 after 3 seconds
|
||||||
|
|
||||||
Scenario: check stuck sync replica
|
Scenario: check stuck sync replica
|
||||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"pause": true, "maximum_lag_on_syncnode": 15000000, "postgresql": {"parameters": {"synchronous_commit": "remote_apply"}}}
|
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"pause": true, "maximum_lag_on_syncnode": 15000000, "postgresql": {"parameters": {"synchronous_commit": "remote_apply"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I create table on postgres0
|
And I create table on postgres-0
|
||||||
And table mytest is present on postgres1 after 2 seconds
|
And table mytest is present on postgres-1 after 2 seconds
|
||||||
And table mytest is present on postgres2 after 2 seconds
|
And table mytest is present on postgres-2 after 2 seconds
|
||||||
When I pause wal replay on postgres2
|
When I pause wal replay on postgres-2
|
||||||
And I load data on postgres0
|
And I load data on postgres-0
|
||||||
Then "sync" key in DCS has sync_standby=postgres1 after 15 seconds
|
Then "sync" key in DCS has sync_standby=postgres-1 after 15 seconds
|
||||||
And I resume wal replay on postgres2
|
And I resume wal replay on postgres-2
|
||||||
And Status code on GET http://127.0.0.1:8009/sync is 200 after 3 seconds
|
And Status code on GET http://127.0.0.1:8009/sync is 200 after 3 seconds
|
||||||
And Status code on GET http://127.0.0.1:8010/async is 200 after 3 seconds
|
And Status code on GET http://127.0.0.1:8010/async is 200 after 3 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"pause": null, "maximum_lag_on_syncnode": -1, "postgresql": {"parameters": {"synchronous_commit": "on"}}}
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"pause": null, "maximum_lag_on_syncnode": -1, "postgresql": {"parameters": {"synchronous_commit": "on"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I drop table on postgres0
|
And I drop table on postgres-0
|
||||||
|
|
||||||
Scenario: check multi sync replication
|
Scenario: check multi sync replication
|
||||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"synchronous_node_count": 2}
|
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"synchronous_node_count": 2}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
Then "sync" key in DCS has sync_standby=postgres1,postgres2 after 10 seconds
|
Then "sync" key in DCS has sync_standby=postgres-1,postgres-2 after 10 seconds
|
||||||
And Status code on GET http://127.0.0.1:8010/sync is 200 after 3 seconds
|
And Status code on GET http://127.0.0.1:8010/sync is 200 after 3 seconds
|
||||||
And Status code on GET http://127.0.0.1:8009/sync is 200 after 3 seconds
|
And Status code on GET http://127.0.0.1:8009/sync is 200 after 3 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"synchronous_node_count": 1}
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"synchronous_node_count": 1}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I shut down postgres1
|
And I shut down postgres-1
|
||||||
Then "sync" key in DCS has sync_standby=postgres2 after 10 seconds
|
Then "sync" key in DCS has sync_standby=postgres-2 after 10 seconds
|
||||||
When I start postgres1
|
When I start postgres-1
|
||||||
Then "members/postgres1" key in DCS has state=running after 10 seconds
|
Then "members/postgres-1" key in DCS has state=running after 10 seconds
|
||||||
And Status code on GET http://127.0.0.1:8010/sync is 200 after 3 seconds
|
And Status code on GET http://127.0.0.1:8010/sync is 200 after 3 seconds
|
||||||
And Status code on GET http://127.0.0.1:8009/async is 200 after 3 seconds
|
And Status code on GET http://127.0.0.1:8009/async is 200 after 3 seconds
|
||||||
|
|
||||||
@@ -60,26 +60,26 @@ Feature: basic replication
|
|||||||
Given I run patronictl.py pause batman
|
Given I run patronictl.py pause batman
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
When I sleep for 2 seconds
|
When I sleep for 2 seconds
|
||||||
And I shut down postgres0
|
And I shut down postgres-0
|
||||||
And I run patronictl.py resume batman
|
And I run patronictl.py resume batman
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And postgres2 role is the primary after 24 seconds
|
And postgres-2 role is the primary after 24 seconds
|
||||||
And Response on GET http://127.0.0.1:8010/history contains recovery after 10 seconds
|
And Response on GET http://127.0.0.1:8010/history contains recovery after 10 seconds
|
||||||
And there is a postgres2_cb.log with "on_role_change master batman" in postgres2 data directory
|
And there is a postgres-2_cb.log with "on_role_change primary batman" in postgres-2 data directory
|
||||||
When I issue a PATCH request to http://127.0.0.1:8010/config with {"synchronous_mode": null, "master_start_timeout": 0}
|
When I issue a PATCH request to http://127.0.0.1:8010/config with {"synchronous_mode": null, "master_start_timeout": 0}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
When I add the table bar to postgres2
|
When I add the table bar to postgres-2
|
||||||
Then table bar is present on postgres1 after 20 seconds
|
Then table bar is present on postgres-1 after 20 seconds
|
||||||
And Response on GET http://127.0.0.1:8010/config contains master_start_timeout after 10 seconds
|
And Response on GET http://127.0.0.1:8010/config contains master_start_timeout after 10 seconds
|
||||||
|
|
||||||
Scenario: check rejoin of the former primary with pg_rewind
|
Scenario: check rejoin of the former primary with pg_rewind
|
||||||
Given I add the table splitbrain to postgres0
|
Given I add the table splitbrain to postgres-0
|
||||||
And I start postgres0
|
And I start postgres-0
|
||||||
Then postgres0 role is the secondary after 20 seconds
|
Then postgres-0 role is the secondary after 20 seconds
|
||||||
When I add the table buz to postgres2
|
When I add the table buz to postgres-2
|
||||||
Then table buz is present on postgres0 after 20 seconds
|
Then table buz is present on postgres-0 after 20 seconds
|
||||||
|
|
||||||
@reject-duplicate-name
|
@reject-duplicate-name
|
||||||
Scenario: check graceful rejection when two nodes have the same name
|
Scenario: check graceful rejection when two nodes have the same name
|
||||||
Given I start duplicate postgres0 on port 8011
|
Given I start duplicate postgres-0 on port 8011
|
||||||
Then there is one of ["Can't start; there is already a node named 'postgres0' running"] CRITICAL in the dup-postgres0 patroni log after 5 seconds
|
Then there is one of ["Can't start; there is already a node named 'postgres-0' running"] CRITICAL in the dup-postgres-0 patroni log after 5 seconds
|
||||||
|
|||||||
@@ -0,0 +1,22 @@
|
|||||||
|
Feature: bootstrap labels
|
||||||
|
Check that user-configurable bootstrap labels are set and removed with state change
|
||||||
|
|
||||||
|
Scenario: check label for cluster bootstrap
|
||||||
|
When I start postgres-0
|
||||||
|
Then postgres-0 is a leader after 10 seconds
|
||||||
|
When I start postgres-1 in a cluster batman1 as a long-running clone of postgres-0
|
||||||
|
Then "members/postgres-1" key in DCS has state=running custom bootstrap script after 20 seconds
|
||||||
|
And postgres-1 is labeled with "foo"
|
||||||
|
And postgres-1 is a leader of batman1 after 20 seconds
|
||||||
|
|
||||||
|
Scenario: check label for replica bootstrap
|
||||||
|
When I do a backup of postgres-1
|
||||||
|
And I start postgres-2 in cluster batman1 using long-running backup_restore
|
||||||
|
Then "members/postgres-2" key in DCS has state=creating replica after 20 seconds
|
||||||
|
And postgres-2 is labeled with "foo"
|
||||||
|
|
||||||
|
Scenario: check bootstrap label is removed
|
||||||
|
Given "members/postgres-1" key in DCS has state=running after 2 seconds
|
||||||
|
And "members/postgres-2" key in DCS has state=running after 20 seconds
|
||||||
|
Then postgres-1 is not labeled with "foo"
|
||||||
|
And postgres-2 is not labeled with "foo"
|
||||||
@@ -1,5 +1,6 @@
|
|||||||
#!/usr/bin/env python
|
#!/usr/bin/env python
|
||||||
|
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
with open("data/{0}/{0}_cb.log".format(sys.argv[1]), "a+") as log:
|
with open("data/{0}/{0}_cb.log".format(sys.argv[1]), "a+") as log:
|
||||||
log.write(" ".join(sys.argv[-3:]) + "\n")
|
log.write(" ".join(sys.argv[-3:]) + "\n")
|
||||||
|
|||||||
@@ -2,13 +2,13 @@ Feature: cascading replication
|
|||||||
We should check that patroni can do base backup and streaming from the replica
|
We should check that patroni can do base backup and streaming from the replica
|
||||||
|
|
||||||
Scenario: check a base backup and streaming replication from a replica
|
Scenario: check a base backup and streaming replication from a replica
|
||||||
Given I start postgres0
|
Given I start postgres-0
|
||||||
And postgres0 is a leader after 10 seconds
|
And postgres-0 is a leader after 10 seconds
|
||||||
And I configure and start postgres1 with a tag clonefrom true
|
And I configure and start postgres-1 with a tag clonefrom true
|
||||||
And replication works from postgres0 to postgres1 after 20 seconds
|
And replication works from postgres-0 to postgres-1 after 20 seconds
|
||||||
And I create label with "postgres0" in postgres0 data directory
|
And I create label with "postgres-0" in postgres-0 data directory
|
||||||
And I create label with "postgres1" in postgres1 data directory
|
And I create label with "postgres-1" in postgres-1 data directory
|
||||||
And "members/postgres1" key in DCS has state=running after 12 seconds
|
And "members/postgres-1" key in DCS has state=running after 12 seconds
|
||||||
And I configure and start postgres2 with a tag replicatefrom postgres1
|
And I configure and start postgres-2 with a tag replicatefrom postgres-1
|
||||||
Then replication works from postgres0 to postgres2 after 30 seconds
|
Then replication works from postgres-0 to postgres-2 after 30 seconds
|
||||||
And there is a label with "postgres1" in postgres2 data directory
|
And there is a label with "postgres-1" in postgres-2 data directory
|
||||||
|
|||||||
+54
-47
@@ -2,72 +2,79 @@ Feature: citus
|
|||||||
We should check that coordinator discovers and registers workers and clients don't have errors when worker cluster switches over
|
We should check that coordinator discovers and registers workers and clients don't have errors when worker cluster switches over
|
||||||
|
|
||||||
Scenario: check that worker cluster is registered in the coordinator
|
Scenario: check that worker cluster is registered in the coordinator
|
||||||
Given I start postgres0 in citus group 0
|
Given I start postgres-0 in citus group 0
|
||||||
And I start postgres2 in citus group 1
|
And I start postgres-2 in citus group 1
|
||||||
Then postgres0 is a leader in a group 0 after 10 seconds
|
Then postgres-0 is a leader in a group 0 after 10 seconds
|
||||||
And postgres2 is a leader in a group 1 after 10 seconds
|
And postgres-2 is a leader in a group 1 after 10 seconds
|
||||||
When I start postgres1 in citus group 0
|
When I start postgres-1 in citus group 0
|
||||||
And I start postgres3 in citus group 1
|
And I start postgres-3 in citus group 1
|
||||||
Then replication works from postgres0 to postgres1 after 15 seconds
|
Then replication works from postgres-0 to postgres-1 after 15 seconds
|
||||||
Then replication works from postgres2 to postgres3 after 15 seconds
|
Then replication works from postgres-2 to postgres-3 after 15 seconds
|
||||||
And postgres0 is registered in the postgres0 as the primary in group 0 after 5 seconds
|
And postgres-0 is registered in the postgres-0 as the primary in group 0 after 5 seconds
|
||||||
And postgres2 is registered in the postgres0 as the primary in group 1 after 5 seconds
|
And postgres-1 is registered in the postgres-0 as the secondary in group 0 after 5 seconds
|
||||||
|
And postgres-2 is registered in the postgres-0 as the primary in group 1 after 5 seconds
|
||||||
|
And postgres-3 is registered in the postgres-0 as the secondary in group 1 after 5 seconds
|
||||||
|
|
||||||
Scenario: coordinator failover updates pg_dist_node
|
Scenario: coordinator failover updates pg_dist_node
|
||||||
Given I run patronictl.py failover batman --group 0 --candidate postgres1 --force
|
Given I run patronictl.py failover batman --group 0 --candidate postgres-1 --force
|
||||||
Then postgres1 role is the primary after 10 seconds
|
Then postgres-1 role is the primary after 10 seconds
|
||||||
And "members/postgres0" key in a group 0 in DCS has state=running after 15 seconds
|
And "members/postgres-0" key in a group 0 in DCS has state=running after 15 seconds
|
||||||
And replication works from postgres1 to postgres0 after 15 seconds
|
And replication works from postgres-1 to postgres-0 after 15 seconds
|
||||||
And postgres1 is registered in the postgres2 as the primary in group 0 after 5 seconds
|
And postgres-1 is registered in the postgres-2 as the primary in group 0 after 5 seconds
|
||||||
And "sync" key in a group 0 in DCS has sync_standby=postgres0 after 15 seconds
|
And postgres-0 is registered in the postgres-2 as the secondary in group 0 after 15 seconds
|
||||||
When I run patronictl.py switchover batman --group 0 --candidate postgres0 --force
|
And "sync" key in a group 0 in DCS has sync_standby=postgres-0 after 15 seconds
|
||||||
Then postgres0 role is the primary after 10 seconds
|
When I run patronictl.py switchover batman --group 0 --candidate postgres-0 --force
|
||||||
And replication works from postgres0 to postgres1 after 15 seconds
|
Then postgres-0 role is the primary after 10 seconds
|
||||||
And postgres0 is registered in the postgres2 as the primary in group 0 after 5 seconds
|
And replication works from postgres-0 to postgres-1 after 15 seconds
|
||||||
And "sync" key in a group 0 in DCS has sync_standby=postgres1 after 15 seconds
|
And postgres-0 is registered in the postgres-2 as the primary in group 0 after 5 seconds
|
||||||
|
And postgres-1 is registered in the postgres-2 as the secondary in group 0 after 15 seconds
|
||||||
|
And "sync" key in a group 0 in DCS has sync_standby=postgres-1 after 15 seconds
|
||||||
|
|
||||||
Scenario: worker switchover doesn't break client queries on the coordinator
|
Scenario: worker switchover doesn't break client queries on the coordinator
|
||||||
Given I create a distributed table on postgres0
|
Given I create a distributed table on postgres-0
|
||||||
And I start a thread inserting data on postgres0
|
And I start a thread inserting data on postgres-0
|
||||||
When I run patronictl.py switchover batman --group 1 --force
|
When I run patronictl.py switchover batman --group 1 --force
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And postgres3 role is the primary after 10 seconds
|
And postgres-3 role is the primary after 10 seconds
|
||||||
And "members/postgres2" key in a group 1 in DCS has state=running after 15 seconds
|
And "members/postgres-2" key in a group 1 in DCS has state=running after 15 seconds
|
||||||
And replication works from postgres3 to postgres2 after 15 seconds
|
And replication works from postgres-3 to postgres-2 after 15 seconds
|
||||||
And postgres3 is registered in the postgres0 as the primary in group 1 after 5 seconds
|
And postgres-3 is registered in the postgres-0 as the primary in group 1 after 5 seconds
|
||||||
And "sync" key in a group 1 in DCS has sync_standby=postgres2 after 15 seconds
|
And postgres-2 is registered in the postgres-0 as the secondary in group 1 after 15 seconds
|
||||||
|
And "sync" key in a group 1 in DCS has sync_standby=postgres-2 after 15 seconds
|
||||||
And a thread is still alive
|
And a thread is still alive
|
||||||
When I run patronictl.py switchover batman --group 1 --force
|
When I run patronictl.py switchover batman --group 1 --force
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And postgres2 role is the primary after 10 seconds
|
And postgres-2 role is the primary after 10 seconds
|
||||||
And replication works from postgres2 to postgres3 after 15 seconds
|
And replication works from postgres-2 to postgres-3 after 15 seconds
|
||||||
And postgres2 is registered in the postgres0 as the primary in group 1 after 5 seconds
|
And postgres-2 is registered in the postgres-0 as the primary in group 1 after 5 seconds
|
||||||
And "sync" key in a group 1 in DCS has sync_standby=postgres3 after 15 seconds
|
And postgres-3 is registered in the postgres-0 as the secondary in group 1 after 15 seconds
|
||||||
|
And "sync" key in a group 1 in DCS has sync_standby=postgres-3 after 15 seconds
|
||||||
And a thread is still alive
|
And a thread is still alive
|
||||||
When I stop a thread
|
When I stop a thread
|
||||||
Then a distributed table on postgres0 has expected rows
|
Then a distributed table on postgres-0 has expected rows
|
||||||
|
|
||||||
Scenario: worker primary restart doesn't break client queries on the coordinator
|
Scenario: worker primary restart doesn't break client queries on the coordinator
|
||||||
Given I cleanup a distributed table on postgres0
|
Given I cleanup a distributed table on postgres-0
|
||||||
And I start a thread inserting data on postgres0
|
And I start a thread inserting data on postgres-0
|
||||||
When I run patronictl.py restart batman postgres2 --group 1 --force
|
When I run patronictl.py restart batman postgres-2 --group 1 --force
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And postgres2 role is the primary after 10 seconds
|
And postgres-2 role is the primary after 10 seconds
|
||||||
And replication works from postgres2 to postgres3 after 15 seconds
|
And replication works from postgres-2 to postgres-3 after 15 seconds
|
||||||
And postgres2 is registered in the postgres0 as the primary in group 1 after 5 seconds
|
And postgres-2 is registered in the postgres-0 as the primary in group 1 after 5 seconds
|
||||||
|
And postgres-3 is registered in the postgres-0 as the secondary in group 1 after 15 seconds
|
||||||
And a thread is still alive
|
And a thread is still alive
|
||||||
When I stop a thread
|
When I stop a thread
|
||||||
Then a distributed table on postgres0 has expected rows
|
Then a distributed table on postgres-0 has expected rows
|
||||||
|
|
||||||
Scenario: check that in-flight transaction is rolled back after timeout when other workers need to change pg_dist_node
|
Scenario: check that in-flight transaction is rolled back after timeout when other workers need to change pg_dist_node
|
||||||
Given I start postgres4 in citus group 2
|
Given I start postgres-4 in citus group 2
|
||||||
Then postgres4 is a leader in a group 2 after 10 seconds
|
Then postgres-4 is a leader in a group 2 after 10 seconds
|
||||||
And "members/postgres4" key in a group 2 in DCS has role=master after 3 seconds
|
And "members/postgres-4" key in a group 2 in DCS has role=primary after 3 seconds
|
||||||
When I run patronictl.py edit-config batman --group 2 -s ttl=20 --force
|
When I run patronictl.py edit-config batman --group 2 -s ttl=20 --force
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And I receive a response output "+ttl: 20"
|
And I receive a response output "+ttl: 20"
|
||||||
Then postgres4 is registered in the postgres2 as the primary in group 2 after 5 seconds
|
Then postgres-4 is registered in the postgres-2 as the primary in group 2 after 5 seconds
|
||||||
When I shut down postgres4
|
When I shut down postgres-4
|
||||||
Then there is a transaction in progress on postgres0 changing pg_dist_node after 5 seconds
|
Then there is a transaction in progress on postgres-0 changing pg_dist_node after 5 seconds
|
||||||
When I run patronictl.py restart batman postgres2 --group 1 --force
|
When I run patronictl.py restart batman postgres-2 --group 1 --force
|
||||||
Then a transaction finishes in 20 seconds
|
Then a transaction finishes in 20 seconds
|
||||||
|
|||||||
@@ -2,16 +2,16 @@ Feature: custom bootstrap
|
|||||||
We should check that patroni can bootstrap a new cluster from a backup
|
We should check that patroni can bootstrap a new cluster from a backup
|
||||||
|
|
||||||
Scenario: clone existing cluster using pg_basebackup
|
Scenario: clone existing cluster using pg_basebackup
|
||||||
Given I start postgres0
|
Given I start postgres-0
|
||||||
Then postgres0 is a leader after 10 seconds
|
Then postgres-0 is a leader after 10 seconds
|
||||||
When I add the table foo to postgres0
|
When I add the table foo to postgres-0
|
||||||
And I start postgres1 in a cluster batman1 as a clone of postgres0
|
And I start postgres-1 in a cluster batman1 as a clone of postgres-0
|
||||||
Then postgres1 is a leader of batman1 after 10 seconds
|
Then postgres-1 is a leader of batman1 after 10 seconds
|
||||||
Then table foo is present on postgres1 after 10 seconds
|
Then table foo is present on postgres-1 after 10 seconds
|
||||||
|
|
||||||
Scenario: make a backup and do a restore into a new cluster
|
Scenario: make a backup and do a restore into a new cluster
|
||||||
Given I add the table bar to postgres1
|
Given I add the table bar to postgres-1
|
||||||
And I do a backup of postgres1
|
And I do a backup of postgres-1
|
||||||
When I start postgres2 in a cluster batman2 from backup
|
When I start postgres-2 in a cluster batman2 from backup
|
||||||
Then postgres2 is a leader of batman2 after 30 seconds
|
Then postgres-2 is a leader of batman2 after 30 seconds
|
||||||
And table bar is present on postgres2 after 10 seconds
|
And table bar is present on postgres-2 after 10 seconds
|
||||||
|
|||||||
@@ -2,16 +2,16 @@ Feature: dcs failsafe mode
|
|||||||
We should check the basic dcs failsafe mode functioning
|
We should check the basic dcs failsafe mode functioning
|
||||||
|
|
||||||
Scenario: check failsafe mode can be successfully enabled
|
Scenario: check failsafe mode can be successfully enabled
|
||||||
Given I start postgres0
|
Given I start postgres-0
|
||||||
And postgres0 is a leader after 10 seconds
|
And postgres-0 is a leader after 10 seconds
|
||||||
Then "config" key in DCS has ttl=30 after 10 seconds
|
Then "config" key in DCS has ttl=30 after 10 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"loop_wait": 2, "ttl": 20, "retry_timeout": 3, "failsafe_mode": true}
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"loop_wait": 2, "ttl": 20, "retry_timeout": 3, "failsafe_mode": true}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And Response on GET http://127.0.0.1:8008/failsafe contains postgres0 after 10 seconds
|
And Response on GET http://127.0.0.1:8008/failsafe contains postgres-0 after 10 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8008/failsafe
|
When I issue a GET request to http://127.0.0.1:8008/failsafe
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I receive a response postgres0 http://127.0.0.1:8008/patroni
|
And I receive a response postgres-0 http://127.0.0.1:8008/patroni
|
||||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"postgresql": {"parameters": {"wal_level": "logical"}},"slots":{"dcs_slot_1": null,"postgres0":null}}
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"postgresql": {"parameters": {"wal_level": "logical"}},"slots":{"dcs_slot_1": null,"postgres_0":null}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"slots": {"dcs_slot_0": {"type": "logical", "database": "postgres", "plugin": "test_decoding"}}}
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"slots": {"dcs_slot_0": {"type": "logical", "database": "postgres", "plugin": "test_decoding"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
@@ -20,97 +20,99 @@ Feature: dcs failsafe mode
|
|||||||
Scenario: check one-node cluster is functioning while DCS is down
|
Scenario: check one-node cluster is functioning while DCS is down
|
||||||
Given DCS is down
|
Given DCS is down
|
||||||
Then Response on GET http://127.0.0.1:8008/primary contains failsafe_mode_is_active after 12 seconds
|
Then Response on GET http://127.0.0.1:8008/primary contains failsafe_mode_is_active after 12 seconds
|
||||||
And postgres0 role is the primary after 10 seconds
|
And postgres-0 role is the primary after 10 seconds
|
||||||
|
|
||||||
@dcs-failsafe
|
@dcs-failsafe
|
||||||
Scenario: check new replica isn't promoted when leader is down and DCS is up
|
Scenario: check new replica isn't promoted when leader is down and DCS is up
|
||||||
Given DCS is up
|
Given DCS is up
|
||||||
When I do a backup of postgres0
|
When I do a backup of postgres-0
|
||||||
And I shut down postgres0
|
And I shut down postgres-0
|
||||||
When I start postgres1 in a cluster batman from backup with no_leader
|
When I start postgres-1 in a cluster batman from backup with no_leader
|
||||||
Then postgres1 role is the replica after 12 seconds
|
Then postgres-1 role is the replica after 12 seconds
|
||||||
|
|
||||||
Scenario: check leader and replica are both in /failsafe key after leader is back
|
Scenario: check leader and replica are both in /failsafe key after leader is back
|
||||||
Given I start postgres0
|
Given I start postgres-0
|
||||||
And I start postgres1
|
And I start postgres-1
|
||||||
Then "members/postgres0" key in DCS has state=running after 10 seconds
|
Then "members/postgres-0" key in DCS has state=running after 10 seconds
|
||||||
And "members/postgres1" key in DCS has state=running after 2 seconds
|
And "members/postgres-1" key in DCS has state=running after 2 seconds
|
||||||
And Response on GET http://127.0.0.1:8009/failsafe contains postgres1 after 10 seconds
|
And Response on GET http://127.0.0.1:8009/failsafe contains postgres-1 after 10 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8009/failsafe
|
When I issue a GET request to http://127.0.0.1:8009/failsafe
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I receive a response postgres0 http://127.0.0.1:8008/patroni
|
And I receive a response postgres-0 http://127.0.0.1:8008/patroni
|
||||||
And I receive a response postgres1 http://127.0.0.1:8009/patroni
|
And I receive a response postgres-1 http://127.0.0.1:8009/patroni
|
||||||
|
|
||||||
@dcs-failsafe
|
@dcs-failsafe
|
||||||
@slot-advance
|
@pg110000
|
||||||
Scenario: check leader and replica are functioning while DCS is down
|
Scenario: check leader and replica are functioning while DCS is down
|
||||||
Given I get all changes from physical slot dcs_slot_1 on postgres0
|
Given I get all changes from physical slot dcs_slot_1 on postgres-0
|
||||||
Then physical slot dcs_slot_1 is in sync between postgres0 and postgres1 after 10 seconds
|
Then physical slot dcs_slot_1 is in sync between postgres-0 and postgres-1 after 10 seconds
|
||||||
And logical slot dcs_slot_0 is in sync between postgres0 and postgres1 after 10 seconds
|
And logical slot dcs_slot_0 is in sync between postgres-0 and postgres-1 after 10 seconds
|
||||||
And DCS is down
|
And DCS is down
|
||||||
Then Response on GET http://127.0.0.1:8008/primary contains failsafe_mode_is_active after 12 seconds
|
Then Response on GET http://127.0.0.1:8008/primary contains failsafe_mode_is_active after 12 seconds
|
||||||
Then postgres0 role is the primary after 10 seconds
|
Then postgres-0 role is the primary after 10 seconds
|
||||||
And postgres1 role is the replica after 2 seconds
|
And postgres-1 role is the replica after 2 seconds
|
||||||
And replication works from postgres0 to postgres1 after 10 seconds
|
And replication works from postgres-0 to postgres-1 after 10 seconds
|
||||||
When I get all changes from logical slot dcs_slot_0 on postgres0
|
When I get all changes from logical slot dcs_slot_0 on postgres-0
|
||||||
And I get all changes from physical slot dcs_slot_1 on postgres0
|
And I get all changes from physical slot dcs_slot_1 on postgres-0
|
||||||
Then logical slot dcs_slot_0 is in sync between postgres0 and postgres1 after 20 seconds
|
Then logical slot dcs_slot_0 is in sync between postgres-0 and postgres-1 after 20 seconds
|
||||||
And physical slot dcs_slot_1 is in sync between postgres0 and postgres1 after 10 seconds
|
And physical slot dcs_slot_1 is in sync between postgres-0 and postgres-1 after 10 seconds
|
||||||
|
|
||||||
@dcs-failsafe
|
@dcs-failsafe
|
||||||
Scenario: check primary is demoted when one replica is shut down and DCS is down
|
Scenario: check primary is demoted when one replica is shut down and DCS is down
|
||||||
Given DCS is down
|
Given DCS is down
|
||||||
And I kill postgres1
|
And I kill postgres-1
|
||||||
And I kill postmaster on postgres1
|
And I kill postmaster on postgres-1
|
||||||
Then postgres0 role is the replica after 12 seconds
|
Then postgres-0 role is the replica after 12 seconds
|
||||||
|
|
||||||
@dcs-failsafe
|
@dcs-failsafe
|
||||||
Scenario: check known replica is promoted when leader is down and DCS is up
|
Scenario: check known replica is promoted when leader is down and DCS is up
|
||||||
Given I kill postgres0
|
Given I kill postgres-0
|
||||||
And I shut down postmaster on postgres0
|
And I shut down postmaster on postgres-0
|
||||||
And DCS is up
|
And DCS is up
|
||||||
When I start postgres1
|
When I start postgres-1
|
||||||
Then "members/postgres1" key in DCS has state=running after 10 seconds
|
Then "members/postgres-1" key in DCS has state=running after 10 seconds
|
||||||
And postgres1 role is the primary after 25 seconds
|
And postgres-1 role is the primary after 25 seconds
|
||||||
|
|
||||||
@dcs-failsafe
|
@dcs-failsafe
|
||||||
Scenario: scale to three-node cluster
|
Scenario: scale to three-node cluster
|
||||||
Given I start postgres0
|
Given I start postgres-0
|
||||||
And I start postgres2
|
And I configure and start postgres-2 with a tag replicatefrom postgres-0
|
||||||
Then "members/postgres2" key in DCS has state=running after 10 seconds
|
Then "members/postgres-2" key in DCS has state=running after 10 seconds
|
||||||
And "members/postgres0" key in DCS has state=running after 20 seconds
|
And "members/postgres-0" key in DCS has state=running after 20 seconds
|
||||||
And Response on GET http://127.0.0.1:8008/failsafe contains postgres2 after 10 seconds
|
And Response on GET http://127.0.0.1:8008/failsafe contains postgres-2 after 10 seconds
|
||||||
And replication works from postgres1 to postgres0 after 10 seconds
|
And replication works from postgres-1 to postgres-0 after 10 seconds
|
||||||
And replication works from postgres1 to postgres2 after 10 seconds
|
And replication works from postgres-1 to postgres-2 after 10 seconds
|
||||||
|
|
||||||
@dcs-failsafe
|
@dcs-failsafe
|
||||||
@slot-advance
|
@pg110000
|
||||||
Scenario: make sure permanent slots exist on replicas
|
Scenario: make sure permanent slots exist on replicas
|
||||||
Given I issue a PATCH request to http://127.0.0.1:8009/config with {"slots":{"dcs_slot_0":null,"dcs_slot_2":{"type":"logical","database":"postgres","plugin":"test_decoding"}}}
|
Given I issue a PATCH request to http://127.0.0.1:8009/config with {"slots":{"dcs_slot_0":null,"dcs_slot_2":{"type":"logical","database":"postgres","plugin":"test_decoding"}}}
|
||||||
Then logical slot dcs_slot_2 is in sync between postgres1 and postgres0 after 20 seconds
|
Then logical slot dcs_slot_2 is in sync between postgres-1 and postgres-0 after 20 seconds
|
||||||
And logical slot dcs_slot_2 is in sync between postgres1 and postgres2 after 20 seconds
|
And logical slot dcs_slot_2 is in sync between postgres-1 and postgres-2 after 20 seconds
|
||||||
When I get all changes from physical slot dcs_slot_1 on postgres1
|
When I get all changes from physical slot dcs_slot_1 on postgres-1
|
||||||
Then physical slot dcs_slot_1 is in sync between postgres1 and postgres0 after 10 seconds
|
Then physical slot dcs_slot_1 is in sync between postgres-1 and postgres-0 after 10 seconds
|
||||||
And physical slot dcs_slot_1 is in sync between postgres1 and postgres2 after 10 seconds
|
And physical slot dcs_slot_1 is in sync between postgres-1 and postgres-2 after 10 seconds
|
||||||
And physical slot postgres0 is in sync between postgres1 and postgres2 after 10 seconds
|
And physical slot postgres_0 is in sync between postgres-1 and postgres-2 after 10 seconds
|
||||||
|
And physical slot postgres_2 is in sync between postgres-0 and postgres-1 after 10 seconds
|
||||||
|
|
||||||
@dcs-failsafe
|
@dcs-failsafe
|
||||||
Scenario: check three-node cluster is functioning while DCS is down
|
Scenario: check three-node cluster is functioning while DCS is down
|
||||||
Given DCS is down
|
Given DCS is down
|
||||||
Then Response on GET http://127.0.0.1:8009/primary contains failsafe_mode_is_active after 12 seconds
|
Then Response on GET http://127.0.0.1:8009/primary contains failsafe_mode_is_active after 12 seconds
|
||||||
Then postgres1 role is the primary after 10 seconds
|
Then postgres-1 role is the primary after 10 seconds
|
||||||
And postgres0 role is the replica after 2 seconds
|
And postgres-0 role is the replica after 2 seconds
|
||||||
And postgres2 role is the replica after 2 seconds
|
And postgres-2 role is the replica after 2 seconds
|
||||||
|
|
||||||
@dcs-failsafe
|
@dcs-failsafe
|
||||||
@slot-advance
|
@pg110000
|
||||||
Scenario: check that permanent slots are in sync between nodes while DCS is down
|
Scenario: check that permanent slots are in sync between nodes while DCS is down
|
||||||
Given replication works from postgres1 to postgres0 after 10 seconds
|
Given replication works from postgres-1 to postgres-0 after 10 seconds
|
||||||
And replication works from postgres1 to postgres2 after 10 seconds
|
And replication works from postgres-1 to postgres-2 after 10 seconds
|
||||||
When I get all changes from logical slot dcs_slot_2 on postgres1
|
When I get all changes from logical slot dcs_slot_2 on postgres-1
|
||||||
And I get all changes from physical slot dcs_slot_1 on postgres1
|
And I get all changes from physical slot dcs_slot_1 on postgres-1
|
||||||
Then logical slot dcs_slot_2 is in sync between postgres1 and postgres0 after 20 seconds
|
Then logical slot dcs_slot_2 is in sync between postgres-1 and postgres-0 after 20 seconds
|
||||||
And logical slot dcs_slot_2 is in sync between postgres1 and postgres2 after 20 seconds
|
And logical slot dcs_slot_2 is in sync between postgres-1 and postgres-2 after 20 seconds
|
||||||
And physical slot dcs_slot_1 is in sync between postgres1 and postgres0 after 10 seconds
|
And physical slot dcs_slot_1 is in sync between postgres-1 and postgres-0 after 10 seconds
|
||||||
And physical slot dcs_slot_1 is in sync between postgres1 and postgres2 after 10 seconds
|
And physical slot dcs_slot_1 is in sync between postgres-1 and postgres-2 after 10 seconds
|
||||||
And physical slot postgres0 is in sync between postgres1 and postgres2 after 10 seconds
|
And physical slot postgres_0 is in sync between postgres-1 and postgres-2 after 10 seconds
|
||||||
|
And physical slot postgres_2 is in sync between postgres-0 and postgres-1 after 10 seconds
|
||||||
|
|||||||
+54
-28
@@ -1,9 +1,8 @@
|
|||||||
import abc
|
import abc
|
||||||
import datetime
|
import datetime
|
||||||
import glob
|
import glob
|
||||||
import os
|
|
||||||
import json
|
import json
|
||||||
import psutil
|
import os
|
||||||
import re
|
import re
|
||||||
import shutil
|
import shutil
|
||||||
import signal
|
import signal
|
||||||
@@ -13,11 +12,14 @@ import sys
|
|||||||
import tempfile
|
import tempfile
|
||||||
import threading
|
import threading
|
||||||
import time
|
import time
|
||||||
|
|
||||||
|
from http.server import BaseHTTPRequestHandler, HTTPServer
|
||||||
|
|
||||||
|
import psutil
|
||||||
import yaml
|
import yaml
|
||||||
|
|
||||||
import patroni.psycopg as psycopg
|
import patroni.psycopg as psycopg
|
||||||
|
|
||||||
from http.server import BaseHTTPRequestHandler, HTTPServer
|
|
||||||
from patroni.request import PatroniRequest
|
from patroni.request import PatroniRequest
|
||||||
|
|
||||||
|
|
||||||
@@ -52,6 +54,8 @@ class AbstractController(abc.ABC):
|
|||||||
self._log = open(os.path.join(self._output_dir, self._name + '.log'), 'a')
|
self._log = open(os.path.join(self._output_dir, self._name + '.log'), 'a')
|
||||||
self._handle = self._start()
|
self._handle = self._start()
|
||||||
|
|
||||||
|
if max_wait_limit < 0:
|
||||||
|
return
|
||||||
max_wait_limit *= self._context.timeout_multiplier
|
max_wait_limit *= self._context.timeout_multiplier
|
||||||
for _ in range(max_wait_limit):
|
for _ in range(max_wait_limit):
|
||||||
assert self._has_started(), "Process {0} is not running after being started".format(self._name)
|
assert self._has_started(), "Process {0} is not running after being started".format(self._name)
|
||||||
@@ -216,6 +220,8 @@ class PatroniController(AbstractController):
|
|||||||
'host replication replicator all md5',
|
'host replication replicator all md5',
|
||||||
'host all all all md5'
|
'host all all all md5'
|
||||||
]
|
]
|
||||||
|
if isinstance(self._context.dcs_ctl, KubernetesController):
|
||||||
|
config['kubernetes'] = {'bootstrap_labels': {'foo': 'bar'}}
|
||||||
|
|
||||||
if self._context.postgres_supports_ssl and self._context.certfile:
|
if self._context.postgres_supports_ssl and self._context.certfile:
|
||||||
config['postgresql']['parameters'].update({
|
config['postgresql']['parameters'].update({
|
||||||
@@ -256,10 +262,18 @@ class PatroniController(AbstractController):
|
|||||||
'parameters': {
|
'parameters': {
|
||||||
'wal_keep_segments': 100,
|
'wal_keep_segments': 100,
|
||||||
'archive_mode': 'on',
|
'archive_mode': 'on',
|
||||||
'archive_command': (PatroniPoolController.ARCHIVE_RESTORE_SCRIPT
|
'archive_command':
|
||||||
+ ' --mode archive '
|
(PatroniPoolController.ARCHIVE_RESTORE_SCRIPT
|
||||||
+ '--dirname {} --filename %f --pathname %p').format(
|
+ ' --mode archive '
|
||||||
os.path.join(self._work_directory, 'data', 'wal_archive'))
|
+ '--dirname {} --filename %f --pathname %p').format(
|
||||||
|
os.path.join(self._work_directory, 'data',
|
||||||
|
f'wal_archive{str(self._citus_group or "")}')).replace('\\', '/'),
|
||||||
|
'restore_command':
|
||||||
|
(PatroniPoolController.ARCHIVE_RESTORE_SCRIPT
|
||||||
|
+ ' --mode restore '
|
||||||
|
+ '--dirname {} --filename %f --pathname %p').format(
|
||||||
|
os.path.join(self._work_directory, 'data',
|
||||||
|
f'wal_archive{str(self._citus_group or "")}')).replace('\\', '/')
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -490,6 +504,7 @@ class AbstractEtcdController(AbstractDcsController):
|
|||||||
|
|
||||||
def _is_running(self):
|
def _is_running(self):
|
||||||
from patroni.dcs.etcd import DnsCachingResolver
|
from patroni.dcs.etcd import DnsCachingResolver
|
||||||
|
|
||||||
# if etcd is running, but we didn't start it
|
# if etcd is running, but we didn't start it
|
||||||
try:
|
try:
|
||||||
self._client = self._client_cls({'host': 'localhost', 'port': 2379, 'retry_timeout': 30,
|
self._client = self._client_cls({'host': 'localhost', 'port': 2379, 'retry_timeout': 30,
|
||||||
@@ -646,6 +661,10 @@ class KubernetesController(AbstractExternalDcsController):
|
|||||||
except Exception:
|
except Exception:
|
||||||
break
|
break
|
||||||
|
|
||||||
|
def pod_labels(self, name):
|
||||||
|
pod = self._api.read_namespaced_pod(name, self._namespace)
|
||||||
|
return pod.metadata.labels or {}
|
||||||
|
|
||||||
def query(self, key, scope='batman', group=None):
|
def query(self, key, scope='batman', group=None):
|
||||||
if key.startswith('members/'):
|
if key.startswith('members/'):
|
||||||
pod = self._api.read_namespaced_pod(key[8:], self._namespace)
|
pod = self._api.read_namespaced_pod(key[8:], self._namespace)
|
||||||
@@ -859,7 +878,7 @@ class PatroniPoolController(object):
|
|||||||
os.makedirs(feature_dir)
|
os.makedirs(feature_dir)
|
||||||
self._output_dir = feature_dir
|
self._output_dir = feature_dir
|
||||||
|
|
||||||
def clone(self, from_name, cluster_name, to_name):
|
def clone(self, from_name, cluster_name, to_name, long_running=False):
|
||||||
f = self._processes[from_name]
|
f = self._processes[from_name]
|
||||||
custom_config = {
|
custom_config = {
|
||||||
'scope': cluster_name,
|
'scope': cluster_name,
|
||||||
@@ -867,7 +886,8 @@ class PatroniPoolController(object):
|
|||||||
'method': 'pg_basebackup',
|
'method': 'pg_basebackup',
|
||||||
'pg_basebackup': {
|
'pg_basebackup': {
|
||||||
'command': " ".join(self.BACKUP_SCRIPT
|
'command': " ".join(self.BACKUP_SCRIPT
|
||||||
+ ['--walmethod=stream', '--dbname="{0}"'.format(f.backup_source)])
|
+ ['--walmethod=stream', f'--dbname="{f.backup_source}"',
|
||||||
|
f'--sleep {5 if long_running else 0}'])
|
||||||
},
|
},
|
||||||
'dcs': {
|
'dcs': {
|
||||||
'postgresql': {
|
'postgresql': {
|
||||||
@@ -885,17 +905,21 @@ class PatroniPoolController(object):
|
|||||||
.format(os.path.join(self.patroni_path, 'data', 'wal_archive_clone').replace('\\', '/'))
|
.format(os.path.join(self.patroni_path, 'data', 'wal_archive_clone').replace('\\', '/'))
|
||||||
},
|
},
|
||||||
'authentication': {
|
'authentication': {
|
||||||
'superuser': {'password': 'zalando1'},
|
'superuser': {'password': 'patroni1'},
|
||||||
'replication': {'password': 'rep-pass1'}
|
'replication': {'password': 'rep-pass1'}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
self.start(to_name, custom_config=custom_config)
|
kwargs = {'custom_config': custom_config}
|
||||||
|
if long_running:
|
||||||
|
kwargs['max_wait_limit'] = -1
|
||||||
|
self.start(to_name, **kwargs)
|
||||||
|
|
||||||
def backup_restore_config(self, params=None):
|
def backup_restore_config(self, params=None, long_running=False):
|
||||||
return {
|
return {
|
||||||
'command': (self.BACKUP_RESTORE_SCRIPT
|
'command': (self.BACKUP_RESTORE_SCRIPT
|
||||||
+ ' --sourcedir=' + os.path.join(self.patroni_path, 'data', 'basebackup')).replace('\\', '/'),
|
+ ' --sourcedir=' + os.path.join(self.patroni_path, 'data', 'basebackup')
|
||||||
|
+ f' --sleep {5 if long_running else 0}').replace('\\', '/'),
|
||||||
'test-argument': 'test-value', # test config mapping approach on custom bootstrap/replica creation
|
'test-argument': 'test-value', # test config mapping approach on custom bootstrap/replica creation
|
||||||
**(params or {}),
|
**(params or {}),
|
||||||
}
|
}
|
||||||
@@ -917,7 +941,7 @@ class PatroniPoolController(object):
|
|||||||
},
|
},
|
||||||
'postgresql': {
|
'postgresql': {
|
||||||
'authentication': {
|
'authentication': {
|
||||||
'superuser': {'password': 'zalando2'},
|
'superuser': {'password': 'patroni2'},
|
||||||
'replication': {'password': 'rep-pass2'}
|
'replication': {'password': 'rep-pass2'}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -928,11 +952,6 @@ class PatroniPoolController(object):
|
|||||||
custom_config = {
|
custom_config = {
|
||||||
'scope': cluster_name,
|
'scope': cluster_name,
|
||||||
'postgresql': {
|
'postgresql': {
|
||||||
'recovery_conf': {
|
|
||||||
'restore_command': (self.ARCHIVE_RESTORE_SCRIPT + ' --mode restore '
|
|
||||||
+ '--dirname {} --filename %f --pathname %p')
|
|
||||||
.format(os.path.join(self.patroni_path, 'data', 'wal_archive').replace('\\', '/'))
|
|
||||||
},
|
|
||||||
'create_replica_methods': ['no_leader_bootstrap'],
|
'create_replica_methods': ['no_leader_bootstrap'],
|
||||||
'no_leader_bootstrap': self.backup_restore_config({'no_leader': '1'})
|
'no_leader_bootstrap': self.backup_restore_config({'no_leader': '1'})
|
||||||
}
|
}
|
||||||
@@ -1073,8 +1092,6 @@ def before_all(context):
|
|||||||
context.keyfile = os.path.join(context.pctl.output_dir, 'patroni.key')
|
context.keyfile = os.path.join(context.pctl.output_dir, 'patroni.key')
|
||||||
context.certfile = os.path.join(context.pctl.output_dir, 'patroni.crt')
|
context.certfile = os.path.join(context.pctl.output_dir, 'patroni.crt')
|
||||||
try:
|
try:
|
||||||
if sys.platform == 'darwin' and 'GITHUB_ACTIONS' in os.environ:
|
|
||||||
raise Exception
|
|
||||||
with open(os.devnull, 'w') as null:
|
with open(os.devnull, 'w') as null:
|
||||||
ret = subprocess.call(['openssl', 'req', '-nodes', '-new', '-x509', '-subj', '/CN=batman.patroni',
|
ret = subprocess.call(['openssl', 'req', '-nodes', '-new', '-x509', '-subj', '/CN=batman.patroni',
|
||||||
'-addext', 'subjectAltName=IP:127.0.0.1', '-keyout', context.keyfile,
|
'-addext', 'subjectAltName=IP:127.0.0.1', '-keyout', context.keyfile,
|
||||||
@@ -1119,12 +1136,14 @@ def before_feature(context, feature):
|
|||||||
elif feature.name == 'citus':
|
elif feature.name == 'citus':
|
||||||
lib = subprocess.check_output(['pg_config', '--pkglibdir']).decode('utf-8').strip()
|
lib = subprocess.check_output(['pg_config', '--pkglibdir']).decode('utf-8').strip()
|
||||||
if not os.path.exists(os.path.join(lib, 'citus.so')):
|
if not os.path.exists(os.path.join(lib, 'citus.so')):
|
||||||
return feature.skip("Citus extenstion isn't available")
|
return feature.skip("Citus extension isn't available")
|
||||||
|
elif feature.name == 'bootstrap labels' and context.dcs_ctl.name() != 'kubernetes':
|
||||||
|
feature.skip("Tested only on Kubernetes")
|
||||||
context.pctl.create_and_set_output_directory(feature.name)
|
context.pctl.create_and_set_output_directory(feature.name)
|
||||||
|
|
||||||
|
|
||||||
def after_feature(context, feature):
|
def after_feature(context, feature):
|
||||||
""" send SIGCONT to a dcs if neccessary,
|
""" send SIGCONT to a dcs if necessary,
|
||||||
stop all Patronis remove their data directory and cleanup the keys in etcd """
|
stop all Patronis remove their data directory and cleanup the keys in etcd """
|
||||||
context.dcs_ctl.stop_outage()
|
context.dcs_ctl.stop_outage()
|
||||||
context.pctl.stop_all()
|
context.pctl.stop_all()
|
||||||
@@ -1149,11 +1168,18 @@ def after_feature(context, feature):
|
|||||||
|
|
||||||
|
|
||||||
def before_scenario(context, scenario):
|
def before_scenario(context, scenario):
|
||||||
if 'slot-advance' in scenario.effective_tags:
|
for tag in scenario.effective_tags:
|
||||||
for p in context.pctl._processes.values():
|
if tag.startswith('pg') and 6 < len(tag) < 9:
|
||||||
if p._conn and p._conn.server_version < 110000:
|
try:
|
||||||
scenario.skip('pg_replication_slot_advance() is not supported on {0}'.format(p._conn.server_version))
|
ver = int(tag[2:])
|
||||||
break
|
except Exception:
|
||||||
|
ver = 0
|
||||||
|
if not ver:
|
||||||
|
continue
|
||||||
|
for p in context.pctl._processes.values():
|
||||||
|
if p._conn and p._conn.server_version < ver:
|
||||||
|
scenario.skip('not supported on {0}'.format(p._conn.server_version))
|
||||||
|
break
|
||||||
if 'dcs-failsafe' in scenario.effective_tags and not context.dcs_ctl._handle:
|
if 'dcs-failsafe' in scenario.effective_tags and not context.dcs_ctl._handle:
|
||||||
scenario.skip('it is not possible to control state of {0} from tests'.format(context.dcs_ctl.name()))
|
scenario.skip('it is not possible to control state of {0} from tests'.format(context.dcs_ctl.name()))
|
||||||
if 'reject-duplicate-name' in scenario.effective_tags and context.dcs_ctl.name() == 'raft':
|
if 'reject-duplicate-name' in scenario.effective_tags and context.dcs_ctl.name() == 'raft':
|
||||||
|
|||||||
@@ -1,61 +1,66 @@
|
|||||||
Feature: ignored slots
|
Feature: ignored slots
|
||||||
Scenario: check ignored slots aren't removed on failover/switchover
|
Scenario: check ignored slots aren't removed on failover/switchover
|
||||||
Given I start postgres1
|
Given I start postgres-1
|
||||||
Then postgres1 is a leader after 10 seconds
|
Then postgres-1 is a leader after 10 seconds
|
||||||
And there is a non empty initialize key in DCS after 15 seconds
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8009/config with {"ignore_slots": [{"name": "unmanaged_slot_0", "database": "postgres", "plugin": "test_decoding", "type": "logical"}, {"name": "unmanaged_slot_1", "database": "postgres", "plugin": "test_decoding"}, {"name": "unmanaged_slot_2", "database": "postgres"}, {"name": "unmanaged_slot_3"}], "postgresql": {"parameters": {"wal_level": "logical"}}}
|
When I issue a PATCH request to http://127.0.0.1:8009/config with {"ignore_slots": [{"name": "unmanaged_slot_0", "database": "postgres", "plugin": "test_decoding", "type": "logical"}, {"name": "unmanaged_slot_1", "database": "postgres", "plugin": "test_decoding"}, {"name": "unmanaged_slot_2", "database": "postgres"}, {"name": "unmanaged_slot_3"}], "postgresql": {"parameters": {"wal_level": "logical"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And Response on GET http://127.0.0.1:8009/config contains ignore_slots after 10 seconds
|
And Response on GET http://127.0.0.1:8009/config contains ignore_slots after 10 seconds
|
||||||
|
And Response on GET http://127.0.0.1:8009/patroni contains pending_restart after 10 seconds
|
||||||
# Make sure the wal_level has been changed.
|
# Make sure the wal_level has been changed.
|
||||||
When I shut down postgres1
|
When I run patronictl.py restart batman postgres-1 --force
|
||||||
And I start postgres1
|
Then "members/postgres-1" key in DCS has role=primary after 10 seconds
|
||||||
Then postgres1 is a leader after 10 seconds
|
|
||||||
And "members/postgres1" key in DCS has role=master after 10 seconds
|
|
||||||
# Make sure Patroni has finished telling Postgres it should be accepting writes.
|
# Make sure Patroni has finished telling Postgres it should be accepting writes.
|
||||||
And postgres1 role is the primary after 20 seconds
|
And postgres-1 role is the primary after 20 seconds
|
||||||
# 1. Create our test logical replication slot.
|
# 1. Create our test logical replication slot.
|
||||||
# Test that ny subset of attributes in the ignore slots matcher is enough to match a slot
|
# Test that ny subset of attributes in the ignore slots matcher is enough to match a slot
|
||||||
# by using 3 different slots.
|
# by using 3 different slots.
|
||||||
When I create a logical replication slot unmanaged_slot_0 on postgres1 with the test_decoding plugin
|
When I create a logical replication slot unmanaged_slot_0 on postgres-1 with the test_decoding plugin
|
||||||
And I create a logical replication slot unmanaged_slot_1 on postgres1 with the test_decoding plugin
|
And I create a logical replication slot unmanaged_slot_1 on postgres-1 with the test_decoding plugin
|
||||||
And I create a logical replication slot unmanaged_slot_2 on postgres1 with the test_decoding plugin
|
And I create a logical replication slot unmanaged_slot_2 on postgres-1 with the test_decoding plugin
|
||||||
And I create a logical replication slot unmanaged_slot_3 on postgres1 with the test_decoding plugin
|
And I create a logical replication slot unmanaged_slot_3 on postgres-1 with the test_decoding plugin
|
||||||
And I create a logical replication slot dummy_slot on postgres1 with the test_decoding plugin
|
And I create a logical replication slot dummy_slot on postgres-1 with the test_decoding plugin
|
||||||
# It seems like it'd be obvious that these slots exist since we just created them,
|
# It seems like it'd be obvious that these slots exist since we just created them,
|
||||||
# but Patroni can actually end up dropping them almost immediately, so it's helpful
|
# but Patroni can actually end up dropping them almost immediately, so it's helpful
|
||||||
# to verify they exist before we begin testing whether they persist through failover
|
# to verify they exist before we begin testing whether they persist through failover
|
||||||
# cycles.
|
# cycles.
|
||||||
Then postgres1 has a logical replication slot named unmanaged_slot_0 with the test_decoding plugin after 2 seconds
|
Then postgres-1 has a logical replication slot named unmanaged_slot_0 with the test_decoding plugin after 2 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_1 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_1 with the test_decoding plugin after 2 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_2 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_2 with the test_decoding plugin after 2 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_3 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_3 with the test_decoding plugin after 2 seconds
|
||||||
|
|
||||||
When I start postgres0
|
When I start postgres-0
|
||||||
Then "members/postgres0" key in DCS has role=replica after 10 seconds
|
Then "members/postgres-0" key in DCS has role=replica after 10 seconds
|
||||||
And postgres0 role is the secondary after 20 seconds
|
And postgres-0 role is the secondary after 20 seconds
|
||||||
# Verify that the replica has advanced beyond the point in the WAL
|
# Verify that the replica has advanced beyond the point in the WAL
|
||||||
# where we created the replication slot so that on the next failover
|
# where we created the replication slot so that on the next failover
|
||||||
# cycle we don't accidentally rewind to before the slot creation.
|
# cycle we don't accidentally rewind to before the slot creation.
|
||||||
And replication works from postgres1 to postgres0 after 20 seconds
|
And replication works from postgres-1 to postgres-0 after 20 seconds
|
||||||
When I shut down postgres1
|
When I shut down postgres-1
|
||||||
Then "members/postgres0" key in DCS has role=master after 10 seconds
|
Then "members/postgres-0" key in DCS has role=primary after 10 seconds
|
||||||
|
|
||||||
# 2. After a failover the server (now a replica) still has the slot.
|
# 2. After a failover the server (now a replica) still has the slot.
|
||||||
When I start postgres1
|
When I start postgres-1
|
||||||
Then postgres1 role is the secondary after 20 seconds
|
Then postgres-1 role is the secondary after 20 seconds
|
||||||
And "members/postgres1" key in DCS has role=replica after 10 seconds
|
And "members/postgres-1" key in DCS has role=replica after 10 seconds
|
||||||
# give Patroni time to sync replication slots
|
# give Patroni time to sync replication slots
|
||||||
And I sleep for 2 seconds
|
And I sleep for 2 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_0 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_0 with the test_decoding plugin after 2 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_1 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_1 with the test_decoding plugin after 2 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_2 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_2 with the test_decoding plugin after 2 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_3 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_3 with the test_decoding plugin after 2 seconds
|
||||||
And postgres1 does not have a replication slot named dummy_slot
|
And postgres-1 does not have a replication slot named dummy_slot
|
||||||
|
|
||||||
# 3. After a failover the server (now a primary) still has the slot.
|
# 3. After a failover the server (now a primary) still has the slot.
|
||||||
When I shut down postgres0
|
When I shut down postgres-0
|
||||||
Then "members/postgres1" key in DCS has role=master after 10 seconds
|
Then "members/postgres-1" key in DCS has role=primary after 10 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_0 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_0 with the test_decoding plugin after 2 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_1 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_1 with the test_decoding plugin after 2 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_2 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_2 with the test_decoding plugin after 2 seconds
|
||||||
And postgres1 has a logical replication slot named unmanaged_slot_3 with the test_decoding plugin after 2 seconds
|
And postgres-1 has a logical replication slot named unmanaged_slot_3 with the test_decoding plugin after 2 seconds
|
||||||
|
|
||||||
|
@pg170000
|
||||||
|
Scenario: check that logical slots with failover are not removed by Patroni
|
||||||
|
Given I create a logical failover slot test17 on postgres-1 with the pgoutput plugin
|
||||||
|
When I run patronictl.py restart batman postgres-1 --force
|
||||||
|
Then postgres-1 has a logical replication slot named test17 with the pgoutput plugin after 2 seconds
|
||||||
|
|||||||
@@ -0,0 +1,26 @@
|
|||||||
|
Feature: nostream node
|
||||||
|
|
||||||
|
Scenario: check nostream node is recovering from archive
|
||||||
|
When I start postgres-0
|
||||||
|
And I configure and start postgres-1 with a tag nostream true
|
||||||
|
Then "members/postgres-1" key in DCS has replication_state=in archive recovery after 10 seconds
|
||||||
|
And replication works from postgres-0 to postgres-1 after 30 seconds
|
||||||
|
|
||||||
|
@pg110000
|
||||||
|
Scenario: check permanent logical replication slots are not copied
|
||||||
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"postgresql": {"parameters": {"wal_level": "logical"}}, "slots":{"test_logical":{"type":"logical","database":"postgres","plugin":"test_decoding"}}}
|
||||||
|
Then I receive a response code 200
|
||||||
|
When I run patronictl.py restart batman postgres-0 --force
|
||||||
|
Then postgres-0 has a logical replication slot named test_logical with the test_decoding plugin after 10 seconds
|
||||||
|
When I configure and start postgres-2 with a tag replicatefrom postgres-1
|
||||||
|
Then "members/postgres-2" key in DCS has replication_state=streaming after 10 seconds
|
||||||
|
And postgres-1 does not have a replication slot named test_logical
|
||||||
|
And postgres-2 does not have a replication slot named test_logical
|
||||||
|
|
||||||
|
@pg110000
|
||||||
|
Scenario: check that slots are written to the /status key
|
||||||
|
Given "status" key in DCS has postgres_0 in slots
|
||||||
|
And "status" key in DCS has postgres_2 in slots
|
||||||
|
And "status" key in DCS has test_logical in slots
|
||||||
|
And "status" key in DCS has test_logical in slots
|
||||||
|
And "status" key in DCS does not have postgres_1 in slots
|
||||||
@@ -2,12 +2,12 @@ Feature: patroni api
|
|||||||
We should check that patroni correctly responds to valid and not-valid API requests.
|
We should check that patroni correctly responds to valid and not-valid API requests.
|
||||||
|
|
||||||
Scenario: check API requests on a stand-alone server
|
Scenario: check API requests on a stand-alone server
|
||||||
Given I start postgres0
|
Given I start postgres-0
|
||||||
And postgres0 is a leader after 10 seconds
|
And postgres-0 is a leader after 10 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8008/
|
When I issue a GET request to http://127.0.0.1:8008/
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I receive a response state running
|
And I receive a response state running
|
||||||
And I receive a response role master
|
And I receive a response role primary
|
||||||
When I issue a GET request to http://127.0.0.1:8008/standby_leader
|
When I issue a GET request to http://127.0.0.1:8008/standby_leader
|
||||||
Then I receive a response code 503
|
Then I receive a response code 503
|
||||||
When I issue a GET request to http://127.0.0.1:8008/health
|
When I issue a GET request to http://127.0.0.1:8008/health
|
||||||
@@ -17,10 +17,10 @@ Scenario: check API requests on a stand-alone server
|
|||||||
When I issue a POST request to http://127.0.0.1:8008/reinitialize with {"force": true}
|
When I issue a POST request to http://127.0.0.1:8008/reinitialize with {"force": true}
|
||||||
Then I receive a response code 503
|
Then I receive a response code 503
|
||||||
And I receive a response text I am the leader, can not reinitialize
|
And I receive a response text I am the leader, can not reinitialize
|
||||||
When I run patronictl.py switchover batman --master postgres0 --force
|
When I run patronictl.py switchover batman --primary postgres-0 --force
|
||||||
Then I receive a response returncode 1
|
Then I receive a response returncode 1
|
||||||
And I receive a response output "Error: No candidates found to switchover to"
|
And I receive a response output "Error: No candidates found to switchover to"
|
||||||
When I issue a POST request to http://127.0.0.1:8008/switchover with {"leader": "postgres0"}
|
When I issue a POST request to http://127.0.0.1:8008/switchover with {"leader": "postgres-0"}
|
||||||
Then I receive a response code 412
|
Then I receive a response code 412
|
||||||
And I receive a response text switchover is not possible: cluster does not have members except leader
|
And I receive a response text switchover is not possible: cluster does not have members except leader
|
||||||
When I issue an empty POST request to http://127.0.0.1:8008/failover
|
When I issue an empty POST request to http://127.0.0.1:8008/failover
|
||||||
@@ -30,7 +30,7 @@ Scenario: check API requests on a stand-alone server
|
|||||||
And I receive a response text "Failover could be performed only to a specific candidate"
|
And I receive a response text "Failover could be performed only to a specific candidate"
|
||||||
|
|
||||||
Scenario: check local configuration reload
|
Scenario: check local configuration reload
|
||||||
Given I add tag new_tag new_value to postgres0 config
|
Given I add tag new_tag new_value to postgres-0 config
|
||||||
And I issue an empty POST request to http://127.0.0.1:8008/reload
|
And I issue an empty POST request to http://127.0.0.1:8008/reload
|
||||||
Then I receive a response code 202
|
Then I receive a response code 202
|
||||||
|
|
||||||
@@ -58,43 +58,43 @@ Scenario: check the scheduled restart
|
|||||||
Given I issue a scheduled restart at http://127.0.0.1:8008 in 5 seconds with {"restart_pending": "True"}
|
Given I issue a scheduled restart at http://127.0.0.1:8008 in 5 seconds with {"restart_pending": "True"}
|
||||||
Then I receive a response code 202
|
Then I receive a response code 202
|
||||||
And Response on GET http://127.0.0.1:8008/patroni does not contain pending_restart after 10 seconds
|
And Response on GET http://127.0.0.1:8008/patroni does not contain pending_restart after 10 seconds
|
||||||
And postgres0 role is the primary after 10 seconds
|
And postgres-0 role is the primary after 10 seconds
|
||||||
|
|
||||||
Scenario: check API requests for the primary-replica pair in the pause mode
|
Scenario: check API requests for the primary-replica pair in the pause mode
|
||||||
Given I start postgres1
|
Given I start postgres-1
|
||||||
Then replication works from postgres0 to postgres1 after 20 seconds
|
Then replication works from postgres-0 to postgres-1 after 20 seconds
|
||||||
When I run patronictl.py pause batman
|
When I run patronictl.py pause batman
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
When I kill postmaster on postgres1
|
When I kill postmaster on postgres-1
|
||||||
And I issue a GET request to http://127.0.0.1:8009/replica
|
And I issue a GET request to http://127.0.0.1:8009/replica
|
||||||
Then I receive a response code 503
|
Then I receive a response code 503
|
||||||
And "members/postgres1" key in DCS has state=stopped after 10 seconds
|
And "members/postgres-1" key in DCS has state=stopped after 10 seconds
|
||||||
When I run patronictl.py restart batman postgres1 --force
|
When I run patronictl.py restart batman postgres-1 --force
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
Then replication works from postgres0 to postgres1 after 20 seconds
|
Then replication works from postgres-0 to postgres-1 after 20 seconds
|
||||||
And I sleep for 2 seconds
|
And I sleep for 2 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8009/replica
|
When I issue a GET request to http://127.0.0.1:8009/replica
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I receive a response state running
|
And I receive a response state running
|
||||||
And I receive a response role replica
|
And I receive a response role replica
|
||||||
When I run patronictl.py reinit batman postgres1 --force --wait
|
When I run patronictl.py reinit batman postgres-1 --force --wait
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And I receive a response output "Success: reinitialize for member postgres1"
|
And I receive a response output "Success: reinitialize for member postgres-1"
|
||||||
And postgres1 role is the secondary after 30 seconds
|
And postgres-1 role is the secondary after 30 seconds
|
||||||
And replication works from postgres0 to postgres1 after 20 seconds
|
And replication works from postgres-0 to postgres-1 after 20 seconds
|
||||||
When I run patronictl.py restart batman postgres0 --force
|
When I run patronictl.py restart batman postgres-0 --force
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And I receive a response output "Success: restart on member postgres0"
|
And I receive a response output "Success: restart on member postgres-0"
|
||||||
And postgres0 role is the primary after 5 seconds
|
And postgres-0 role is the primary after 5 seconds
|
||||||
|
|
||||||
Scenario: check the switchover via the API in the pause mode
|
Scenario: check the switchover via the API in the pause mode
|
||||||
Given I issue a POST request to http://127.0.0.1:8008/switchover with {"leader": "postgres0", "candidate": "postgres1"}
|
Given I issue a POST request to http://127.0.0.1:8008/switchover with {"leader": "postgres-0", "candidate": "postgres-1"}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And postgres1 is a leader after 5 seconds
|
And postgres-1 is a leader after 5 seconds
|
||||||
And postgres1 role is the primary after 10 seconds
|
And postgres-1 role is the primary after 10 seconds
|
||||||
And postgres0 role is the secondary after 10 seconds
|
And postgres-0 role is the secondary after 10 seconds
|
||||||
And replication works from postgres1 to postgres0 after 20 seconds
|
And replication works from postgres-1 to postgres-0 after 20 seconds
|
||||||
And "members/postgres0" key in DCS has state=running after 10 seconds
|
And "members/postgres-0" key in DCS has state=running after 10 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8008/primary
|
When I issue a GET request to http://127.0.0.1:8008/primary
|
||||||
Then I receive a response code 503
|
Then I receive a response code 503
|
||||||
When I issue a GET request to http://127.0.0.1:8008/replica
|
When I issue a GET request to http://127.0.0.1:8008/replica
|
||||||
@@ -105,18 +105,18 @@ Scenario: check the switchover via the API in the pause mode
|
|||||||
Then I receive a response code 503
|
Then I receive a response code 503
|
||||||
|
|
||||||
Scenario: check the scheduled switchover
|
Scenario: check the scheduled switchover
|
||||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 10 seconds
|
Given I issue a scheduled switchover from postgres-1 to postgres-0 in 10 seconds
|
||||||
Then I receive a response returncode 1
|
Then I receive a response returncode 1
|
||||||
And I receive a response output "Can't schedule switchover in the paused state"
|
And I receive a response output "Can't schedule switchover in the paused state"
|
||||||
When I run patronictl.py resume batman
|
When I run patronictl.py resume batman
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 10 seconds
|
Given I issue a scheduled switchover from postgres-1 to postgres-0 in 10 seconds
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And postgres0 is a leader after 20 seconds
|
And postgres-0 is a leader after 20 seconds
|
||||||
And postgres0 role is the primary after 10 seconds
|
And postgres-0 role is the primary after 10 seconds
|
||||||
And postgres1 role is the secondary after 10 seconds
|
And postgres-1 role is the secondary after 10 seconds
|
||||||
And replication works from postgres0 to postgres1 after 25 seconds
|
And replication works from postgres-0 to postgres-1 after 25 seconds
|
||||||
And "members/postgres1" key in DCS has state=running after 10 seconds
|
And "members/postgres-1" key in DCS has state=running after 10 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8008/primary
|
When I issue a GET request to http://127.0.0.1:8008/primary
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
When I issue a GET request to http://127.0.0.1:8008/replica
|
When I issue a GET request to http://127.0.0.1:8008/replica
|
||||||
|
|||||||
@@ -1,75 +1,86 @@
|
|||||||
Feature: permanent slots
|
Feature: permanent slots
|
||||||
Scenario: check that physical permanent slots are created
|
Scenario: check that physical permanent slots are created
|
||||||
Given I start postgres0
|
Given I start postgres-0
|
||||||
Then postgres0 is a leader after 10 seconds
|
Then postgres-0 is a leader after 10 seconds
|
||||||
And there is a non empty initialize key in DCS after 15 seconds
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"slots":{"test_physical":0,"postgres0":0,"postgres1":0,"postgres3":0},"postgresql":{"parameters":{"wal_level":"logical"}}}
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"slots":{"test_physical":0,"postgres_3":0},"postgresql":{"parameters":{"wal_level":"logical"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And Response on GET http://127.0.0.1:8008/config contains slots after 10 seconds
|
And Response on GET http://127.0.0.1:8008/config contains slots after 10 seconds
|
||||||
When I start postgres1
|
When I start postgres-1
|
||||||
And I start postgres2
|
And I configure and start postgres-2 with a tag nofailover true
|
||||||
And I configure and start postgres3 with a tag replicatefrom postgres2
|
And I configure and start postgres-3 with a tag replicatefrom postgres-2
|
||||||
Then postgres0 has a physical replication slot named test_physical after 10 seconds
|
Then postgres-0 has a physical replication slot named test_physical after 10 seconds
|
||||||
And postgres0 has a physical replication slot named postgres1 after 10 seconds
|
And postgres-0 has a physical replication slot named postgres_1 after 10 seconds
|
||||||
And postgres0 has a physical replication slot named postgres2 after 10 seconds
|
And postgres-0 has a physical replication slot named postgres_2 after 10 seconds
|
||||||
And postgres2 has a physical replication slot named postgres3 after 10 seconds
|
And postgres-2 has a physical replication slot named postgres_3 after 10 seconds
|
||||||
|
And postgres-2 does not have a replication slot named test_physical
|
||||||
|
|
||||||
@slot-advance
|
@pg110000
|
||||||
Scenario: check that logical permanent slots are created
|
Scenario: check that logical permanent slots are created
|
||||||
Given I run patronictl.py restart batman postgres0 --force
|
Given I run patronictl.py restart batman postgres-0 --force
|
||||||
And I issue a PATCH request to http://127.0.0.1:8008/config with {"slots":{"test_logical":{"type":"logical","database":"postgres","plugin":"test_decoding"}}}
|
And I issue a PATCH request to http://127.0.0.1:8008/config with {"slots":{"test_logical":{"type":"logical","database":"postgres","plugin":"test_decoding"}}}
|
||||||
Then postgres0 has a logical replication slot named test_logical with the test_decoding plugin after 10 seconds
|
Then postgres-0 has a logical replication slot named test_logical with the test_decoding plugin after 10 seconds
|
||||||
|
|
||||||
@slot-advance
|
@pg110000
|
||||||
Scenario: check that permanent slots are created on replicas
|
Scenario: check that permanent slots are created on replicas
|
||||||
Given postgres1 has a logical replication slot named test_logical with the test_decoding plugin after 10 seconds
|
Given postgres-1 has a logical replication slot named test_logical with the test_decoding plugin after 10 seconds
|
||||||
Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds
|
Then Logical slot test_logical is in sync between postgres-0 and postgres-1 after 10 seconds
|
||||||
And Logical slot test_logical is in sync between postgres0 and postgres2 after 10 seconds
|
And Logical slot test_logical is in sync between postgres-0 and postgres-3 after 10 seconds
|
||||||
And Logical slot test_logical is in sync between postgres0 and postgres3 after 10 seconds
|
And postgres-1 has a physical replication slot named test_physical after 2 seconds
|
||||||
And postgres1 has a physical replication slot named test_physical after 2 seconds
|
And postgres-2 does not have a replication slot named test_logical
|
||||||
And postgres2 has a physical replication slot named test_physical after 2 seconds
|
And postgres-3 has a physical replication slot named test_physical after 2 seconds
|
||||||
And postgres3 has a physical replication slot named test_physical after 2 seconds
|
|
||||||
|
|
||||||
@slot-advance
|
@pg110000
|
||||||
Scenario: check permanent physical slots that match with member names
|
Scenario: check permanent physical slots that match with member names
|
||||||
Given postgres0 has a physical replication slot named postgres3 after 2 seconds
|
Given postgres-0 has a physical replication slot named postgres_3 after 2 seconds
|
||||||
And postgres1 has a physical replication slot named postgres0 after 2 seconds
|
And postgres-1 has a physical replication slot named postgres_0 after 2 seconds
|
||||||
And postgres1 has a physical replication slot named postgres3 after 2 seconds
|
And postgres-1 has a physical replication slot named postgres_2 after 2 seconds
|
||||||
And postgres2 has a physical replication slot named postgres0 after 2 seconds
|
And postgres-1 has a physical replication slot named postgres_3 after 2 seconds
|
||||||
And postgres2 has a physical replication slot named postgres3 after 2 seconds
|
And postgres-2 does not have a replication slot named postgres_0
|
||||||
And postgres2 has a physical replication slot named postgres1 after 2 seconds
|
And postgres-2 does not have a replication slot named postgres_1
|
||||||
And postgres1 does not have a replication slot named postgres2
|
And postgres-2 has a physical replication slot named postgres_3 after 2 seconds
|
||||||
And postgres3 does not have a replication slot named postgres2
|
And postgres-3 has a physical replication slot named postgres_0 after 2 seconds
|
||||||
|
And postgres-3 has a physical replication slot named postgres_1 after 2 seconds
|
||||||
|
And postgres-3 has a physical replication slot named postgres_2 after 2 seconds
|
||||||
|
|
||||||
@slot-advance
|
@pg110000
|
||||||
Scenario: check that permanent slots are advanced on replicas
|
Scenario: check that permanent slots are advanced on replicas
|
||||||
Given I add the table replicate_me to postgres0
|
Given I add the table replicate_me to postgres-0
|
||||||
When I get all changes from logical slot test_logical on postgres0
|
When I get all changes from logical slot test_logical on postgres-0
|
||||||
And I get all changes from physical slot test_physical on postgres0
|
And I get all changes from physical slot test_physical on postgres-0
|
||||||
Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds
|
Then Logical slot test_logical is in sync between postgres-0 and postgres-1 after 10 seconds
|
||||||
And Physical slot test_physical is in sync between postgres0 and postgres1 after 10 seconds
|
And Physical slot test_physical is in sync between postgres-0 and postgres-1 after 10 seconds
|
||||||
And Logical slot test_logical is in sync between postgres0 and postgres2 after 10 seconds
|
And Logical slot test_logical is in sync between postgres-0 and postgres-3 after 10 seconds
|
||||||
And Physical slot test_physical is in sync between postgres0 and postgres2 after 10 seconds
|
And Physical slot test_physical is in sync between postgres-0 and postgres-3 after 10 seconds
|
||||||
And Logical slot test_logical is in sync between postgres0 and postgres3 after 10 seconds
|
And Physical slot postgres_1 is in sync between postgres-0 and postgres-3 after 10 seconds
|
||||||
And Physical slot test_physical is in sync between postgres0 and postgres3 after 10 seconds
|
And Physical slot postgres_3 is in sync between postgres-2 and postgres-0 after 20 seconds
|
||||||
And Physical slot postgres1 is in sync between postgres0 and postgres2 after 10 seconds
|
And Physical slot postgres_3 is in sync between postgres-2 and postgres-1 after 10 seconds
|
||||||
And Physical slot postgres3 is in sync between postgres2 and postgres0 after 20 seconds
|
|
||||||
And Physical slot postgres3 is in sync between postgres2 and postgres1 after 10 seconds
|
|
||||||
And postgres1 does not have a replication slot named postgres2
|
|
||||||
And postgres3 does not have a replication slot named postgres2
|
|
||||||
|
|
||||||
@slot-advance
|
@pg110000
|
||||||
Scenario: check that only permanent slots are written to the /status key
|
Scenario: check that permanent slots and member slots are written to the /status key
|
||||||
Given "status" key in DCS has test_physical in slots
|
Given "status" key in DCS has test_physical in slots
|
||||||
And "status" key in DCS has postgres0 in slots
|
And "status" key in DCS has postgres_0 in slots
|
||||||
And "status" key in DCS has postgres1 in slots
|
And "status" key in DCS has postgres_1 in slots
|
||||||
And "status" key in DCS does not have postgres2 in slots
|
And "status" key in DCS has postgres_2 in slots
|
||||||
And "status" key in DCS has postgres3 in slots
|
And "status" key in DCS has postgres_3 in slots
|
||||||
|
|
||||||
|
@pg110000
|
||||||
|
Scenario: check that only non-permanent member slots are written to the retain_slots in /status key
|
||||||
|
Given "status" key in DCS has postgres_0 in retain_slots
|
||||||
|
And "status" key in DCS has postgres_1 in retain_slots
|
||||||
|
And "status" key in DCS has postgres_2 in retain_slots
|
||||||
|
And "status" key in DCS does not have postgres_3 in retain_slots
|
||||||
|
|
||||||
Scenario: check permanent physical replication slot after failover
|
Scenario: check permanent physical replication slot after failover
|
||||||
Given I shut down postgres3
|
Given I shut down postgres-3
|
||||||
And I shut down postgres2
|
And I shut down postgres-2
|
||||||
And I shut down postgres0
|
And I shut down postgres-0
|
||||||
Then postgres1 has a physical replication slot named test_physical after 10 seconds
|
Then postgres-1 has a physical replication slot named test_physical after 10 seconds
|
||||||
And postgres1 has a physical replication slot named postgres0 after 10 seconds
|
And postgres-1 has a physical replication slot named postgres_0 after 10 seconds
|
||||||
And postgres1 has a physical replication slot named postgres3 after 10 seconds
|
And postgres-1 has a physical replication slot named postgres_3 after 10 seconds
|
||||||
|
When I start postgres-0
|
||||||
|
Then postgres-0 role is the replica after 20 seconds
|
||||||
|
And physical replication slot named postgres_1 on postgres-0 has no xmin value after 10 seconds
|
||||||
|
# postgres_2 and postgres_3 slots are retained, but postgres_2 will still have xmin value :(
|
||||||
|
And postgres-0 has a physical replication slot named postgres_2 after 10 seconds
|
||||||
|
And postgres-0 has a physical replication slot named postgres_3 after 10 seconds
|
||||||
|
|||||||
@@ -2,38 +2,38 @@ Feature: priority replication
|
|||||||
We should check that we can give nodes priority during failover
|
We should check that we can give nodes priority during failover
|
||||||
|
|
||||||
Scenario: check failover priority 0 prevents leaderships
|
Scenario: check failover priority 0 prevents leaderships
|
||||||
Given I configure and start postgres0 with a tag failover_priority 1
|
Given I configure and start postgres-0 with a tag failover_priority 1
|
||||||
And I configure and start postgres1 with a tag failover_priority 0
|
And I configure and start postgres-1 with a tag failover_priority 0
|
||||||
Then replication works from postgres0 to postgres1 after 20 seconds
|
Then replication works from postgres-0 to postgres-1 after 20 seconds
|
||||||
When I shut down postgres0
|
When I shut down postgres-0
|
||||||
And there is one of ["following a different leader because I am not allowed to promote"] INFO in the postgres1 patroni log after 5 seconds
|
And there is one of ["following a different leader because I am not allowed to promote"] INFO in the postgres-1 patroni log after 5 seconds
|
||||||
Then postgres1 role is the secondary after 10 seconds
|
Then postgres-1 role is the secondary after 10 seconds
|
||||||
When I start postgres0
|
When I start postgres-0
|
||||||
Then postgres0 role is the primary after 10 seconds
|
Then postgres-0 role is the primary after 10 seconds
|
||||||
|
|
||||||
Scenario: check higher failover priority is respected
|
Scenario: check higher failover priority is respected
|
||||||
Given I configure and start postgres2 with a tag failover_priority 1
|
Given I configure and start postgres-2 with a tag failover_priority 1
|
||||||
And I configure and start postgres3 with a tag failover_priority 2
|
And I configure and start postgres-3 with a tag failover_priority 2
|
||||||
Then replication works from postgres0 to postgres2 after 20 seconds
|
Then replication works from postgres-0 to postgres-2 after 20 seconds
|
||||||
And replication works from postgres0 to postgres3 after 20 seconds
|
And replication works from postgres-0 to postgres-3 after 20 seconds
|
||||||
When I shut down postgres0
|
When I shut down postgres-0
|
||||||
Then postgres3 role is the primary after 10 seconds
|
Then postgres-3 role is the primary after 10 seconds
|
||||||
And there is one of ["postgres3 has equally tolerable WAL position and priority 2, while this node has priority 1","Wal position of postgres3 is ahead of my wal position"] INFO in the postgres2 patroni log after 5 seconds
|
And there is one of ["postgres-3 has equally tolerable WAL position and priority 2, while this node has priority 1","Wal position of postgres-3 is ahead of my wal position"] INFO in the postgres-2 patroni log after 5 seconds
|
||||||
|
|
||||||
Scenario: check conflicting configuration handling
|
Scenario: check conflicting configuration handling
|
||||||
When I set nofailover tag in postgres2 config
|
When I set nofailover tag in postgres-2 config
|
||||||
And I issue an empty POST request to http://127.0.0.1:8010/reload
|
And I issue an empty POST request to http://127.0.0.1:8010/reload
|
||||||
Then I receive a response code 202
|
Then I receive a response code 202
|
||||||
And there is one of ["Conflicting configuration between nofailover: True and failover_priority: 1. Defaulting to nofailover: True"] WARNING in the postgres2 patroni log after 5 seconds
|
And there is one of ["Conflicting configuration between nofailover: True and failover_priority: 1. Defaulting to nofailover: True"] WARNING in the postgres-2 patroni log after 5 seconds
|
||||||
And "members/postgres2" key in DCS has tags={'failover_priority': '1', 'nofailover': True} after 10 seconds
|
And "members/postgres-2" key in DCS has tags={'failover_priority': '1', 'nofailover': True} after 10 seconds
|
||||||
When I issue a POST request to http://127.0.0.1:8010/failover with {"candidate": "postgres2"}
|
When I issue a POST request to http://127.0.0.1:8010/failover with {"candidate": "postgres-2"}
|
||||||
Then I receive a response code 412
|
Then I receive a response code 412
|
||||||
And I receive a response text "failover is not possible: no good candidates have been found"
|
And I receive a response text "failover is not possible: no good candidates have been found"
|
||||||
When I reset nofailover tag in postgres1 config
|
When I reset nofailover tag in postgres-1 config
|
||||||
And I issue an empty POST request to http://127.0.0.1:8009/reload
|
And I issue an empty POST request to http://127.0.0.1:8009/reload
|
||||||
Then I receive a response code 202
|
Then I receive a response code 202
|
||||||
And there is one of ["Conflicting configuration between nofailover: False and failover_priority: 0. Defaulting to nofailover: False"] WARNING in the postgres1 patroni log after 5 seconds
|
And there is one of ["Conflicting configuration between nofailover: False and failover_priority: 0. Defaulting to nofailover: False"] WARNING in the postgres-1 patroni log after 5 seconds
|
||||||
And "members/postgres1" key in DCS has tags={'failover_priority': '0', 'nofailover': False} after 10 seconds
|
And "members/postgres-1" key in DCS has tags={'failover_priority': '0', 'nofailover': False} after 10 seconds
|
||||||
And I issue a POST request to http://127.0.0.1:8009/failover with {"candidate": "postgres1"}
|
And I issue a POST request to http://127.0.0.1:8009/failover with {"candidate": "postgres-1"}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And postgres1 role is the primary after 10 seconds
|
And postgres-1 role is the primary after 10 seconds
|
||||||
|
|||||||
@@ -0,0 +1,38 @@
|
|||||||
|
Feature: synchronous replicas priority
|
||||||
|
We should check that we can give nodes priority for becoming synchronous replicas
|
||||||
|
|
||||||
|
Scenario: check replica with sync_priority=0 does not become a synchronous replica
|
||||||
|
Given I start postgres-0
|
||||||
|
Then postgres-0 is a leader after 10 seconds
|
||||||
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 20, "synchronous_mode": true}
|
||||||
|
Then I receive a response code 200
|
||||||
|
When I configure and start postgres-1 with a tag sync_priority 0
|
||||||
|
Then sync key in DCS has leader=postgres-0 after 20 seconds
|
||||||
|
And sync key in DCS has sync_standby=None after 5 seconds
|
||||||
|
|
||||||
|
Scenario: check higher synchronous replicas priority is respected
|
||||||
|
Given I configure and start postgres-2 with a tag sync_priority 1
|
||||||
|
And I configure and start postgres-3 with a tag sync_priority 2
|
||||||
|
Then replication works from postgres-0 to postgres-2 after 20 seconds
|
||||||
|
And replication works from postgres-0 to postgres-3 after 20 seconds
|
||||||
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"synchronous_node_count": 1}
|
||||||
|
Then I receive a response code 200
|
||||||
|
And sync key in DCS has sync_standby=postgres-3 after 10 seconds
|
||||||
|
|
||||||
|
|
||||||
|
Scenario: check conflicting configuration handling
|
||||||
|
When I set nosync tag in postgres-3 config
|
||||||
|
And I issue an empty POST request to http://127.0.0.1:8011/reload
|
||||||
|
Then I receive a response code 202
|
||||||
|
And there is one of ["Conflicting configuration between nosync: True and sync_priority: 2. Defaulting to nosync: True"] WARNING in the postgres-3 patroni log after 5 seconds
|
||||||
|
And "members/postgres-3" key in DCS has tags={'nosync': True, 'sync_priority': '2'} after 10 seconds
|
||||||
|
And "sync" key in DCS has sync_standby=postgres-2 after 10 seconds
|
||||||
|
When I reset nosync tag in postgres-1 config
|
||||||
|
And I issue an empty POST request to http://127.0.0.1:8009/reload
|
||||||
|
Then I receive a response code 202
|
||||||
|
And there is one of ["Conflicting configuration between nosync: False and sync_priority: 0. Defaulting to nosync: False"] WARNING in the postgres-1 patroni log after 5 seconds
|
||||||
|
And "members/postgres-1" key in DCS has tags={'nosync': False, 'sync_priority': '0'} after 10 seconds
|
||||||
|
When I shut down postgres-2
|
||||||
|
And "sync" key in DCS has sync_standby=postgres-1 after 3 seconds
|
||||||
|
|
||||||
@@ -0,0 +1,68 @@
|
|||||||
|
Feature: quorum commit
|
||||||
|
Check basic workfrlows when quorum commit is enabled
|
||||||
|
|
||||||
|
Scenario: check enable quorum commit and that the only leader promotes after restart
|
||||||
|
Given I start postgres-0
|
||||||
|
Then postgres-0 is a leader after 10 seconds
|
||||||
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"ttl": 20, "synchronous_mode": "quorum"}
|
||||||
|
Then I receive a response code 200
|
||||||
|
And sync key in DCS has leader=postgres-0 after 20 seconds
|
||||||
|
And sync key in DCS has quorum=0 after 2 seconds
|
||||||
|
And synchronous_standby_names on postgres-0 is set to '_empty_str_' after 2 seconds
|
||||||
|
When I shut down postgres-0
|
||||||
|
And sync key in DCS has leader=postgres-0 after 2 seconds
|
||||||
|
When I start postgres-0
|
||||||
|
Then postgres-0 role is the primary after 10 seconds
|
||||||
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"synchronous_mode_strict": true}
|
||||||
|
Then synchronous_standby_names on postgres-0 is set to 'ANY 1 (*)' after 10 seconds
|
||||||
|
|
||||||
|
Scenario: check failover with one quorum standby
|
||||||
|
Given I start postgres-1
|
||||||
|
Then sync key in DCS has sync_standby=postgres-1 after 10 seconds
|
||||||
|
And synchronous_standby_names on postgres-0 is set to 'ANY 1 ("postgres-1")' after 2 seconds
|
||||||
|
When I shut down postgres-0
|
||||||
|
Then postgres-1 role is the primary after 10 seconds
|
||||||
|
And sync key in DCS has quorum=0 after 10 seconds
|
||||||
|
Then synchronous_standby_names on postgres-1 is set to 'ANY 1 (*)' after 10 seconds
|
||||||
|
When I start postgres-0
|
||||||
|
Then sync key in DCS has leader=postgres-1 after 10 seconds
|
||||||
|
Then sync key in DCS has sync_standby=postgres-0 after 10 seconds
|
||||||
|
And synchronous_standby_names on postgres-1 is set to 'ANY 1 ("postgres-0")' after 2 seconds
|
||||||
|
|
||||||
|
Scenario: check behavior with three nodes and different replication factor
|
||||||
|
Given I start postgres-2
|
||||||
|
Then sync key in DCS has sync_standby=postgres-0,postgres-2 after 10 seconds
|
||||||
|
And sync key in DCS has quorum=1 after 2 seconds
|
||||||
|
And synchronous_standby_names on postgres-1 is set to 'ANY 1 ("postgres-0","postgres-2")' after 2 seconds
|
||||||
|
When I issue a PATCH request to http://127.0.0.1:8009/config with {"synchronous_node_count": 2}
|
||||||
|
Then sync key in DCS has quorum=0 after 10 seconds
|
||||||
|
And synchronous_standby_names on postgres-1 is set to 'ANY 2 ("postgres-0","postgres-2")' after 2 seconds
|
||||||
|
|
||||||
|
Scenario: switch from quorum replication to good old multisync and back
|
||||||
|
Given I issue a PATCH request to http://127.0.0.1:8009/config with {"synchronous_mode": true, "synchronous_node_count": 1}
|
||||||
|
And I shut down postgres-0
|
||||||
|
Then synchronous_standby_names on postgres-1 is set to '"postgres-2"' after 10 seconds
|
||||||
|
And sync key in DCS has sync_standby=postgres-2 after 10 seconds
|
||||||
|
Then sync key in DCS has quorum=0 after 2 seconds
|
||||||
|
When I issue a PATCH request to http://127.0.0.1:8009/config with {"synchronous_mode": "quorum"}
|
||||||
|
And I start postgres-0
|
||||||
|
Then synchronous_standby_names on postgres-1 is set to 'ANY 1 ("postgres-0","postgres-2")' after 10 seconds
|
||||||
|
And sync key in DCS has sync_standby=postgres-0,postgres-2 after 10 seconds
|
||||||
|
Then sync key in DCS has quorum=1 after 2 seconds
|
||||||
|
|
||||||
|
Scenario: REST API and patronictl
|
||||||
|
Given I run patronictl.py list batman
|
||||||
|
Then I receive a response returncode 0
|
||||||
|
And I receive a response output "Quorum Standby"
|
||||||
|
And Status code on GET http://127.0.0.1:8008/quorum is 200 after 3 seconds
|
||||||
|
And Status code on GET http://127.0.0.1:8010/quorum is 200 after 3 seconds
|
||||||
|
|
||||||
|
Scenario: nosync node is removed from voters and synchronous_standby_names
|
||||||
|
Given I add tag nosync true to postgres-2 config
|
||||||
|
When I issue an empty POST request to http://127.0.0.1:8010/reload
|
||||||
|
Then I receive a response code 202
|
||||||
|
And sync key in DCS has quorum=0 after 10 seconds
|
||||||
|
And sync key in DCS has sync_standby=postgres-0 after 10 seconds
|
||||||
|
And synchronous_standby_names on postgres-1 is set to 'ANY 1 ("postgres-0")' after 2 seconds
|
||||||
|
And Status code on GET http://127.0.0.1:8010/quorum is 503 after 10 seconds
|
||||||
+22
-13
@@ -2,25 +2,34 @@ Feature: recovery
|
|||||||
We want to check that crashed postgres is started back
|
We want to check that crashed postgres is started back
|
||||||
|
|
||||||
Scenario: check that timeline is not incremented when primary is started after crash
|
Scenario: check that timeline is not incremented when primary is started after crash
|
||||||
Given I start postgres0
|
Given I start postgres-0
|
||||||
Then postgres0 is a leader after 10 seconds
|
Then postgres-0 is a leader after 10 seconds
|
||||||
And there is a non empty initialize key in DCS after 15 seconds
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
When I start postgres1
|
When I start postgres-1
|
||||||
And I add the table foo to postgres0
|
And I add the table foo to postgres-0
|
||||||
Then table foo is present on postgres1 after 20 seconds
|
Then table foo is present on postgres-1 after 20 seconds
|
||||||
When I kill postmaster on postgres0
|
When I kill postmaster on postgres-0
|
||||||
Then postgres0 role is the primary after 10 seconds
|
Then postgres-0 role is the primary after 10 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8008/
|
When I issue a GET request to http://127.0.0.1:8008/
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I receive a response role master
|
And I receive a response role primary
|
||||||
And I receive a response timeline 1
|
And I receive a response timeline 1
|
||||||
And "members/postgres0" key in DCS has state=running after 12 seconds
|
And "members/postgres-0" key in DCS has state=running after 12 seconds
|
||||||
And replication works from postgres0 to postgres1 after 15 seconds
|
And replication works from postgres-0 to postgres-1 after 15 seconds
|
||||||
|
|
||||||
Scenario: check immediate failover when master_start_timeout=0
|
Scenario: check immediate failover when master_start_timeout=0
|
||||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"master_start_timeout": 0}
|
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"master_start_timeout": 0}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And Response on GET http://127.0.0.1:8008/config contains master_start_timeout after 10 seconds
|
And Response on GET http://127.0.0.1:8008/config contains master_start_timeout after 10 seconds
|
||||||
When I kill postmaster on postgres0
|
When I kill postmaster on postgres-0
|
||||||
Then postgres1 is a leader after 10 seconds
|
Then postgres-1 is a leader after 10 seconds
|
||||||
And postgres1 role is the primary after 10 seconds
|
And postgres-1 role is the primary after 10 seconds
|
||||||
|
|
||||||
|
Scenario: check crashed primary demotes after failed attempt to start
|
||||||
|
Given I issue a PATCH request to http://127.0.0.1:8009/config with {"master_start_timeout": null}
|
||||||
|
Then I receive a response code 200
|
||||||
|
And postgres-0 role is the replica after 10 seconds
|
||||||
|
When I ensure postgres-1 fails to start after a failure
|
||||||
|
When I kill postmaster on postgres-1
|
||||||
|
Then postgres-0 is a leader after 10 seconds
|
||||||
|
And there is a postgres-1_cb.log with "on_role_change demoted batman" in postgres-1 data directory
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
Feature: standby cluster
|
Feature: standby cluster
|
||||||
Scenario: prepare the cluster with logical slots
|
Scenario: prepare the cluster with logical slots
|
||||||
Given I start postgres1
|
Given I start postgres-1
|
||||||
Then postgres1 is a leader after 10 seconds
|
Then postgres-1 is a leader after 10 seconds
|
||||||
And there is a non empty initialize key in DCS after 15 seconds
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8009/config with {"slots": {"pm_1": {"type": "physical"}}, "postgresql": {"parameters": {"wal_level": "logical"}}}
|
When I issue a PATCH request to http://127.0.0.1:8009/config with {"slots": {"pm_1": {"type": "physical"}}, "postgresql": {"parameters": {"wal_level": "logical"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
@@ -9,63 +9,58 @@ Feature: standby cluster
|
|||||||
And I sleep for 3 seconds
|
And I sleep for 3 seconds
|
||||||
When I issue a PATCH request to http://127.0.0.1:8009/config with {"slots": {"test_logical": {"type": "logical", "database": "postgres", "plugin": "test_decoding"}}}
|
When I issue a PATCH request to http://127.0.0.1:8009/config with {"slots": {"test_logical": {"type": "logical", "database": "postgres", "plugin": "test_decoding"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I do a backup of postgres1
|
And I do a backup of postgres-1
|
||||||
When I start postgres0
|
When I start postgres-0
|
||||||
Then "members/postgres0" key in DCS has state=running after 10 seconds
|
Then "members/postgres-0" key in DCS has state=running after 10 seconds
|
||||||
And replication works from postgres1 to postgres0 after 15 seconds
|
And replication works from postgres-1 to postgres-0 after 15 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8008/patroni
|
And Response on GET http://127.0.0.1:8008/patroni contains replication_state=streaming after 10 seconds
|
||||||
Then I receive a response code 200
|
And "members/postgres-0" key in DCS has replication_state=streaming after 10 seconds
|
||||||
And I receive a response replication_state streaming
|
|
||||||
And "members/postgres0" key in DCS has replication_state=streaming after 10 seconds
|
|
||||||
|
|
||||||
@slot-advance
|
@pg110000
|
||||||
Scenario: check permanent logical slots are synced to the replica
|
Scenario: check permanent logical slots are synced to the replica
|
||||||
Given I run patronictl.py restart batman postgres1 --force
|
Given I run patronictl.py restart batman postgres-1 --force
|
||||||
Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds
|
Then Logical slot test_logical is in sync between postgres-0 and postgres-1 after 10 seconds
|
||||||
|
|
||||||
Scenario: Detach exiting node from the cluster
|
Scenario: Detach exiting node from the cluster
|
||||||
When I shut down postgres1
|
When I shut down postgres-1
|
||||||
Then postgres0 is a leader after 10 seconds
|
Then postgres-0 is a leader after 10 seconds
|
||||||
And "members/postgres0" key in DCS has role=master after 3 seconds
|
And "members/postgres-0" key in DCS has role=primary after 5 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8008/
|
When I issue a GET request to http://127.0.0.1:8008/
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
|
|
||||||
Scenario: check replication of a single table in a standby cluster
|
Scenario: check replication of a single table in a standby cluster
|
||||||
Given I start postgres1 in a standby cluster batman1 as a clone of postgres0
|
Given I start postgres-1 in a standby cluster batman1 as a clone of postgres-0
|
||||||
Then postgres1 is a leader of batman1 after 10 seconds
|
Then postgres-1 is a leader of batman1 after 10 seconds
|
||||||
When I add the table foo to postgres0
|
When I add the table foo to postgres-0
|
||||||
Then table foo is present on postgres1 after 20 seconds
|
Then table foo is present on postgres-1 after 20 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8009/patroni
|
And Response on GET http://127.0.0.1:8009/patroni contains replication_state=streaming after 10 seconds
|
||||||
Then I receive a response code 200
|
|
||||||
And I receive a response replication_state streaming
|
|
||||||
And I sleep for 3 seconds
|
And I sleep for 3 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8009/primary
|
When I issue a GET request to http://127.0.0.1:8009/primary
|
||||||
Then I receive a response code 503
|
Then I receive a response code 503
|
||||||
When I issue a GET request to http://127.0.0.1:8009/standby_leader
|
When I issue a GET request to http://127.0.0.1:8009/standby_leader
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I receive a response role standby_leader
|
And I receive a response role standby_leader
|
||||||
And there is a postgres1_cb.log with "on_role_change standby_leader batman1" in postgres1 data directory
|
And there is a postgres-1_cb.log with "on_role_change standby_leader batman1" in postgres-1 data directory
|
||||||
When I start postgres2 in a cluster batman1
|
When I start postgres-2 in a cluster batman1
|
||||||
Then postgres2 role is the replica after 24 seconds
|
Then postgres-2 role is the replica after 24 seconds
|
||||||
And table foo is present on postgres2 after 20 seconds
|
And postgres-2 is replicating from postgres-1 after 10 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8010/patroni
|
And table foo is present on postgres-2 after 20 seconds
|
||||||
Then I receive a response code 200
|
And Response on GET http://127.0.0.1:8010/patroni contains replication_state=streaming after 10 seconds
|
||||||
And I receive a response replication_state streaming
|
And postgres-1 does not have a replication slot named test_logical
|
||||||
And postgres1 does not have a replication slot named test_logical
|
|
||||||
|
|
||||||
Scenario: check switchover
|
Scenario: check switchover
|
||||||
Given I run patronictl.py switchover batman1 --force
|
Given I run patronictl.py switchover batman1 --force
|
||||||
Then Status code on GET http://127.0.0.1:8010/standby_leader is 200 after 10 seconds
|
Then Status code on GET http://127.0.0.1:8010/standby_leader is 200 after 10 seconds
|
||||||
And postgres1 is replicating from postgres2 after 32 seconds
|
And postgres-1 is replicating from postgres-2 after 32 seconds
|
||||||
And there is a postgres2_cb.log with "on_start replica batman1\non_role_change standby_leader batman1" in postgres2 data directory
|
And there is a postgres-2_cb.log with "on_start replica batman1\non_role_change standby_leader batman1" in postgres-2 data directory
|
||||||
|
|
||||||
Scenario: check failover
|
Scenario: check failover
|
||||||
When I kill postgres2
|
When I kill postgres-2
|
||||||
And I kill postmaster on postgres2
|
And I kill postmaster on postgres-2
|
||||||
Then postgres1 is replicating from postgres0 after 32 seconds
|
Then postgres-1 is replicating from postgres-0 after 32 seconds
|
||||||
And Status code on GET http://127.0.0.1:8009/standby_leader is 200 after 10 seconds
|
And Status code on GET http://127.0.0.1:8009/standby_leader is 200 after 10 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8009/primary
|
When I issue a GET request to http://127.0.0.1:8009/primary
|
||||||
Then I receive a response code 503
|
Then I receive a response code 503
|
||||||
And I receive a response role standby_leader
|
And I receive a response role standby_leader
|
||||||
And replication works from postgres0 to postgres1 after 15 seconds
|
And replication works from postgres-0 to postgres-1 after 15 seconds
|
||||||
And there is a postgres1_cb.log with "on_role_change replica batman1\non_role_change standby_leader batman1" in postgres1 data directory
|
And there is a postgres-1_cb.log with "on_role_change replica batman1\non_role_change standby_leader batman1" in postgres-1 data directory
|
||||||
|
|||||||
@@ -1,16 +1,28 @@
|
|||||||
import json
|
import json
|
||||||
import patroni.psycopg as pg
|
|
||||||
|
|
||||||
from behave import step, then
|
|
||||||
from time import sleep, time
|
from time import sleep, time
|
||||||
|
|
||||||
|
import parse
|
||||||
|
|
||||||
@step('I start {name:w}')
|
from behave import register_type, step, then
|
||||||
|
|
||||||
|
import patroni.psycopg as pg
|
||||||
|
|
||||||
|
|
||||||
|
@parse.with_pattern(r'[a-z][a-z0-9_\-]*[a-z0-9]')
|
||||||
|
def parse_name(text):
|
||||||
|
return text
|
||||||
|
|
||||||
|
|
||||||
|
register_type(name=parse_name)
|
||||||
|
|
||||||
|
|
||||||
|
@step('I start {name:name}')
|
||||||
def start_patroni(context, name):
|
def start_patroni(context, name):
|
||||||
return context.pctl.start(name)
|
return context.pctl.start(name)
|
||||||
|
|
||||||
|
|
||||||
@step('I start duplicate {name:w} on port {port:d}')
|
@step('I start duplicate {name:name} on port {port:d}')
|
||||||
def start_duplicate_patroni(context, name, port):
|
def start_duplicate_patroni(context, name, port):
|
||||||
config = {
|
config = {
|
||||||
"name": name,
|
"name": name,
|
||||||
@@ -26,47 +38,51 @@ def start_duplicate_patroni(context, name, port):
|
|||||||
"No error was raised by duplicate start of {0} ".format(name)
|
"No error was raised by duplicate start of {0} ".format(name)
|
||||||
|
|
||||||
|
|
||||||
@step('I shut down {name:w}')
|
@step('I shut down {name:name}')
|
||||||
def stop_patroni(context, name):
|
def stop_patroni(context, name):
|
||||||
return context.pctl.stop(name, timeout=60)
|
return context.pctl.stop(name, timeout=60)
|
||||||
|
|
||||||
|
|
||||||
@step('I kill {name:w}')
|
@step('I kill {name:name}')
|
||||||
def kill_patroni(context, name):
|
def kill_patroni(context, name):
|
||||||
return context.pctl.stop(name, kill=True)
|
return context.pctl.stop(name, kill=True)
|
||||||
|
|
||||||
|
|
||||||
@step('I shut down postmaster on {name:w}')
|
@step('I shut down postmaster on {name:name}')
|
||||||
def stop_postgres(context, name):
|
def stop_postgres(context, name):
|
||||||
return context.pctl.stop(name, postgres=True)
|
return context.pctl.stop(name, postgres=True)
|
||||||
|
|
||||||
|
|
||||||
@step('I kill postmaster on {name:w}')
|
@step('I kill postmaster on {name:name}')
|
||||||
def kill_postgres(context, name):
|
def kill_postgres(context, name):
|
||||||
return context.pctl.stop(name, kill=True, postgres=True)
|
return context.pctl.stop(name, kill=True, postgres=True)
|
||||||
|
|
||||||
|
|
||||||
@step('I add the table {table_name:w} to {pg_name:w}')
|
def get_wal_name(context, pg_name):
|
||||||
|
version = context.pctl.query(pg_name, "SHOW server_version_num").fetchone()[0]
|
||||||
|
return 'xlog' if int(version) / 10000 < 10 else 'wal'
|
||||||
|
|
||||||
|
|
||||||
|
@step('I add the table {table_name:w} to {pg_name:name}')
|
||||||
def add_table(context, table_name, pg_name):
|
def add_table(context, table_name, pg_name):
|
||||||
# parse the configuration file and get the port
|
# parse the configuration file and get the port
|
||||||
try:
|
try:
|
||||||
context.pctl.query(pg_name, "CREATE TABLE public.{0}()".format(table_name))
|
context.pctl.query(pg_name, "CREATE TABLE public.{0}()".format(table_name))
|
||||||
|
context.pctl.query(pg_name, "SELECT pg_switch_{0}()".format(get_wal_name(context, pg_name)))
|
||||||
except pg.Error as e:
|
except pg.Error as e:
|
||||||
assert False, "Error creating table {0} on {1}: {2}".format(table_name, pg_name, e)
|
assert False, "Error creating table {0} on {1}: {2}".format(table_name, pg_name, e)
|
||||||
|
|
||||||
|
|
||||||
@step('I {action:w} wal replay on {pg_name:w}')
|
@step('I {action:w} wal replay on {pg_name:name}')
|
||||||
def toggle_wal_replay(context, action, pg_name):
|
def toggle_wal_replay(context, action, pg_name):
|
||||||
# pause or resume the wal replay process
|
# pause or resume the wal replay process
|
||||||
try:
|
try:
|
||||||
version = context.pctl.query(pg_name, "SHOW server_version_num").fetchone()[0]
|
context.pctl.query(pg_name, "SELECT pg_{0}_replay_{1}()".format(get_wal_name(context, pg_name), action))
|
||||||
wal_name = 'xlog' if int(version) / 10000 < 10 else 'wal'
|
|
||||||
context.pctl.query(pg_name, "SELECT pg_{0}_replay_{1}()".format(wal_name, action))
|
|
||||||
except pg.Error as e:
|
except pg.Error as e:
|
||||||
assert False, "Error during {0} wal recovery on {1}: {2}".format(action, pg_name, e)
|
assert False, "Error during {0} wal recovery on {1}: {2}".format(action, pg_name, e)
|
||||||
|
|
||||||
|
|
||||||
@step('I {action:w} table on {pg_name:w}')
|
@step('I {action:w} table on {pg_name:name}')
|
||||||
def crdr_mytest(context, action, pg_name):
|
def crdr_mytest(context, action, pg_name):
|
||||||
try:
|
try:
|
||||||
if (action == "create"):
|
if (action == "create"):
|
||||||
@@ -77,7 +93,7 @@ def crdr_mytest(context, action, pg_name):
|
|||||||
assert False, "Error {0} table mytest on {1}: {2}".format(action, pg_name, e)
|
assert False, "Error {0} table mytest on {1}: {2}".format(action, pg_name, e)
|
||||||
|
|
||||||
|
|
||||||
@step('I load data on {pg_name:w}')
|
@step('I load data on {pg_name:name}')
|
||||||
def initiate_load(context, pg_name):
|
def initiate_load(context, pg_name):
|
||||||
# perform dummy load
|
# perform dummy load
|
||||||
try:
|
try:
|
||||||
@@ -86,7 +102,7 @@ def initiate_load(context, pg_name):
|
|||||||
assert False, "Error loading test data on {0}: {1}".format(pg_name, e)
|
assert False, "Error loading test data on {0}: {1}".format(pg_name, e)
|
||||||
|
|
||||||
|
|
||||||
@then('Table {table_name:w} is present on {pg_name:w} after {max_replication_delay:d} seconds')
|
@then('Table {table_name:w} is present on {pg_name:name} after {max_replication_delay:d} seconds')
|
||||||
def table_is_present_on(context, table_name, pg_name, max_replication_delay):
|
def table_is_present_on(context, table_name, pg_name, max_replication_delay):
|
||||||
max_replication_delay *= context.timeout_multiplier
|
max_replication_delay *= context.timeout_multiplier
|
||||||
for _ in range(int(max_replication_delay)):
|
for _ in range(int(max_replication_delay)):
|
||||||
@@ -98,15 +114,15 @@ def table_is_present_on(context, table_name, pg_name, max_replication_delay):
|
|||||||
"Table {0} is not present on {1} after {2} seconds".format(table_name, pg_name, max_replication_delay)
|
"Table {0} is not present on {1} after {2} seconds".format(table_name, pg_name, max_replication_delay)
|
||||||
|
|
||||||
|
|
||||||
@then('{pg_name:w} role is the {pg_role:w} after {max_promotion_timeout:d} seconds')
|
@then('{pg_name:name} role is the {pg_role:w} after {max_promotion_timeout:d} seconds')
|
||||||
def check_role(context, pg_name, pg_role, max_promotion_timeout):
|
def check_role(context, pg_name, pg_role, max_promotion_timeout):
|
||||||
max_promotion_timeout *= context.timeout_multiplier
|
max_promotion_timeout *= context.timeout_multiplier
|
||||||
assert context.pctl.check_role_has_changed_to(pg_name, pg_role, timeout=int(max_promotion_timeout)), \
|
assert context.pctl.check_role_has_changed_to(pg_name, pg_role, timeout=int(max_promotion_timeout)), \
|
||||||
"{0} role didn't change to {1} after {2} seconds".format(pg_name, pg_role, max_promotion_timeout)
|
"{0} role didn't change to {1} after {2} seconds".format(pg_name, pg_role, max_promotion_timeout)
|
||||||
|
|
||||||
|
|
||||||
@step('replication works from {primary:w} to {replica:w} after {time_limit:d} seconds')
|
@step('replication works from {primary:name} to {replica:name} after {time_limit:d} seconds')
|
||||||
@then('replication works from {primary:w} to {replica:w} after {time_limit:d} seconds')
|
@then('replication works from {primary:name} to {replica:name} after {time_limit:d} seconds')
|
||||||
def replication_works(context, primary, replica, time_limit):
|
def replication_works(context, primary, replica, time_limit):
|
||||||
context.execute_steps(u"""
|
context.execute_steps(u"""
|
||||||
When I add the table test_{0} to {1}
|
When I add the table test_{0} to {1}
|
||||||
@@ -120,8 +136,8 @@ def check_patroni_log(context, message_list, level, node, timeout):
|
|||||||
message_list = json.loads(message_list)
|
message_list = json.loads(message_list)
|
||||||
|
|
||||||
for _ in range(int(timeout)):
|
for _ in range(int(timeout)):
|
||||||
messsages_of_level = context.pctl.read_patroni_log(node, level)
|
messages_of_level = context.pctl.read_patroni_log(node, level)
|
||||||
if any(any(message in line for line in messsages_of_level) for message in message_list):
|
if any(any(message in line for line in messages_of_level) for message in message_list):
|
||||||
break
|
break
|
||||||
sleep(1)
|
sleep(1)
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
from behave import step, then
|
||||||
|
|
||||||
|
|
||||||
|
@step('I start {name:name} in a cluster {cluster_name:w} as a long-running clone of {name2:name}')
|
||||||
|
def start_cluster_clone(context, name, cluster_name, name2):
|
||||||
|
context.pctl.clone(name2, cluster_name, name, True)
|
||||||
|
|
||||||
|
|
||||||
|
@step('I start {name:name} in cluster {cluster_name:w} using long-running backup_restore')
|
||||||
|
def start_patroni(context, name, cluster_name):
|
||||||
|
return context.pctl.start(name, custom_config={
|
||||||
|
"scope": cluster_name,
|
||||||
|
"postgresql": {
|
||||||
|
'create_replica_methods': ['backup_restore'],
|
||||||
|
"backup_restore": context.pctl.backup_restore_config(long_running=True),
|
||||||
|
'authentication': {
|
||||||
|
'superuser': {'password': 'patroni1'},
|
||||||
|
'replication': {'password': 'rep-pass1'}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}, max_wait_limit=-1)
|
||||||
|
|
||||||
|
|
||||||
|
@then('{name:name} is labeled with "{label:w}"')
|
||||||
|
def pod_labeled(context, name, label):
|
||||||
|
assert label in context.dcs_ctl.pod_labels(name), f'pod {name} is not labeled with {label}'
|
||||||
|
|
||||||
|
|
||||||
|
@then('{name:name} is not labeled with "{label:w}"')
|
||||||
|
def pod_not_labeled(context, name, label):
|
||||||
|
assert label not in context.dcs_ctl.pod_labels(name), f'pod {name} is still labeled with {label}'
|
||||||
@@ -4,18 +4,18 @@ import time
|
|||||||
from behave import step, then
|
from behave import step, then
|
||||||
|
|
||||||
|
|
||||||
@step('I configure and start {name:w} with a tag {tag_name:w} {tag_value:w}')
|
@step('I configure and start {name:name} with a tag {tag_name:w} {tag_value}')
|
||||||
def start_patroni_with_a_name_value_tag(context, name, tag_name, tag_value):
|
def start_patroni_with_a_name_value_tag(context, name, tag_name, tag_value):
|
||||||
return context.pctl.start(name, custom_config={'tags': {tag_name: tag_value}})
|
return context.pctl.start(name, custom_config={'tags': {tag_name: tag_value}})
|
||||||
|
|
||||||
|
|
||||||
@then('There is a {label} with "{content}" in {name:w} data directory')
|
@then('There is a {label} with "{content}" in {name:name} data directory')
|
||||||
def check_label(context, label, content, name):
|
def check_label(context, label, content, name):
|
||||||
value = (context.pctl.read_label(name, label) or '').replace('\n', '\\n')
|
value = (context.pctl.read_label(name, label) or '').replace('\n', '\\n')
|
||||||
assert content in value, "\"{0}\" in {1} doesn't contain {2}".format(value, label, content)
|
assert content in value, "\"{0}\" in {1} doesn't contain {2}".format(value, label, content)
|
||||||
|
|
||||||
|
|
||||||
@step('I create label with "{content:w}" in {name:w} data directory')
|
@step('I create label with "{content}" in {name:name} data directory')
|
||||||
def write_label(context, content, name):
|
def write_label(context, content, name):
|
||||||
context.pctl.write_label(name, content)
|
context.pctl.write_label(name, content)
|
||||||
|
|
||||||
|
|||||||
+13
-12
@@ -1,17 +1,18 @@
|
|||||||
import json
|
import json
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from behave import step, then
|
|
||||||
from dateutil import tz
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from functools import partial
|
from functools import partial
|
||||||
from threading import Thread, Event
|
from threading import Event, Thread
|
||||||
|
|
||||||
|
from behave import step, then
|
||||||
|
from dateutil import tz
|
||||||
|
|
||||||
tzutc = tz.tzutc()
|
tzutc = tz.tzutc()
|
||||||
|
|
||||||
|
|
||||||
@step('{name:w} is a leader in a group {group:d} after {time_limit:d} seconds')
|
@step('{name:name} is a leader in a group {group:d} after {time_limit:d} seconds')
|
||||||
@then('{name:w} is a leader in a group {group:d} after {time_limit:d} seconds')
|
@then('{name:name} is a leader in a group {group:d} after {time_limit:d} seconds')
|
||||||
def is_a_group_leader(context, name, group, time_limit):
|
def is_a_group_leader(context, name, group, time_limit):
|
||||||
time_limit *= context.timeout_multiplier
|
time_limit *= context.timeout_multiplier
|
||||||
max_time = time.time() + int(time_limit)
|
max_time = time.time() + int(time_limit)
|
||||||
@@ -39,12 +40,12 @@ def check_group_member(context, name, group, key, value, time_limit):
|
|||||||
" after {5} seconds").format(name, group, key, value, response, time_limit)
|
" after {5} seconds").format(name, group, key, value, response, time_limit)
|
||||||
|
|
||||||
|
|
||||||
@step('I start {name:w} in citus group {group:d}')
|
@step('I start {name:name} in citus group {group:d}')
|
||||||
def start_citus(context, name, group):
|
def start_citus(context, name, group):
|
||||||
return context.pctl.start(name, custom_config={"citus": {"database": "postgres", "group": int(group)}})
|
return context.pctl.start(name, custom_config={"citus": {"database": "postgres", "group": int(group)}})
|
||||||
|
|
||||||
|
|
||||||
@step('{name1:w} is registered in the {name2:w} as the {role:w} in group {group:d} after {time_limit:d} seconds')
|
@step('{name1:name} is registered in the {name2:name} as the {role:w} in group {group:d} after {time_limit:d} seconds')
|
||||||
def check_registration(context, name1, name2, role, group, time_limit):
|
def check_registration(context, name1, name2, role, group, time_limit):
|
||||||
time_limit *= context.timeout_multiplier
|
time_limit *= context.timeout_multiplier
|
||||||
max_time = time.time() + int(time_limit)
|
max_time = time.time() + int(time_limit)
|
||||||
@@ -64,13 +65,13 @@ def check_registration(context, name1, name2, role, group, time_limit):
|
|||||||
assert False, "Node {0} is not registered in pg_dist_node on the node {1}".format(name1, name2)
|
assert False, "Node {0} is not registered in pg_dist_node on the node {1}".format(name1, name2)
|
||||||
|
|
||||||
|
|
||||||
@step('I create a distributed table on {name:w}')
|
@step('I create a distributed table on {name:name}')
|
||||||
def create_distributed_table(context, name):
|
def create_distributed_table(context, name):
|
||||||
context.pctl.query(name, 'CREATE TABLE public.d(id int not null)')
|
context.pctl.query(name, 'CREATE TABLE public.d(id int not null)')
|
||||||
context.pctl.query(name, "SELECT create_distributed_table('public.d', 'id')")
|
context.pctl.query(name, "SELECT create_distributed_table('public.d', 'id')")
|
||||||
|
|
||||||
|
|
||||||
@step('I cleanup a distributed table on {name:w}')
|
@step('I cleanup a distributed table on {name:name}')
|
||||||
def cleanup_distributed_table(context, name):
|
def cleanup_distributed_table(context, name):
|
||||||
context.pctl.query(name, 'TRUNCATE public.d')
|
context.pctl.query(name, 'TRUNCATE public.d')
|
||||||
|
|
||||||
@@ -86,7 +87,7 @@ def insert_thread(query_func, context):
|
|||||||
context.thread_stop_event.wait(0.01)
|
context.thread_stop_event.wait(0.01)
|
||||||
|
|
||||||
|
|
||||||
@step('I start a thread inserting data on {name:w}')
|
@step('I start a thread inserting data on {name:name}')
|
||||||
def start_insert_thread(context, name):
|
def start_insert_thread(context, name):
|
||||||
context.thread_stop_event = Event()
|
context.thread_stop_event = Event()
|
||||||
context.insert_counter = 0
|
context.insert_counter = 0
|
||||||
@@ -109,13 +110,13 @@ def stop_insert_thread(context):
|
|||||||
assert not context.thread.is_alive(), "Thread is still alive"
|
assert not context.thread.is_alive(), "Thread is still alive"
|
||||||
|
|
||||||
|
|
||||||
@step("a distributed table on {name:w} has expected rows")
|
@step("a distributed table on {name:name} has expected rows")
|
||||||
def count_rows(context, name):
|
def count_rows(context, name):
|
||||||
rows = context.pctl.query(name, "SELECT COUNT(*) FROM public.d").fetchone()[0]
|
rows = context.pctl.query(name, "SELECT COUNT(*) FROM public.d").fetchone()[0]
|
||||||
assert rows == context.insert_counter, "Distributed table doesn't have expected amount of rows"
|
assert rows == context.insert_counter, "Distributed table doesn't have expected amount of rows"
|
||||||
|
|
||||||
|
|
||||||
@step("there is a transaction in progress on {name:w} changing pg_dist_node after {time_limit:d} seconds")
|
@step("there is a transaction in progress on {name:name} changing pg_dist_node after {time_limit:d} seconds")
|
||||||
def check_transaction(context, name, time_limit):
|
def check_transaction(context, name, time_limit):
|
||||||
time_limit *= context.timeout_multiplier
|
time_limit *= context.timeout_multiplier
|
||||||
max_time = time.time() + int(time_limit)
|
max_time = time.time() + int(time_limit)
|
||||||
|
|||||||
@@ -3,17 +3,17 @@ import time
|
|||||||
from behave import step, then
|
from behave import step, then
|
||||||
|
|
||||||
|
|
||||||
@step('I start {name:w} in a cluster {cluster_name:w} as a clone of {name2:w}')
|
@step('I start {name:name} in a cluster {cluster_name:w} as a clone of {name2:name}')
|
||||||
def start_cluster_clone(context, name, cluster_name, name2):
|
def start_cluster_clone(context, name, cluster_name, name2):
|
||||||
context.pctl.clone(name2, cluster_name, name)
|
context.pctl.clone(name2, cluster_name, name)
|
||||||
|
|
||||||
|
|
||||||
@step('I start {name:w} in a cluster {cluster_name:w} from backup')
|
@step('I start {name:name} in a cluster {cluster_name:w} from backup')
|
||||||
def start_cluster_from_backup(context, name, cluster_name):
|
def start_cluster_from_backup(context, name, cluster_name):
|
||||||
context.pctl.bootstrap_from_backup(name, cluster_name)
|
context.pctl.bootstrap_from_backup(name, cluster_name)
|
||||||
|
|
||||||
|
|
||||||
@then('{name:w} is a leader of {cluster_name:w} after {time_limit:d} seconds')
|
@then('{name:name} is a leader of {cluster_name:w} after {time_limit:d} seconds')
|
||||||
def is_a_leader(context, name, cluster_name, time_limit):
|
def is_a_leader(context, name, cluster_name, time_limit):
|
||||||
time_limit *= context.timeout_multiplier
|
time_limit *= context.timeout_multiplier
|
||||||
max_time = time.time() + int(time_limit)
|
max_time = time.time() + int(time_limit)
|
||||||
@@ -22,6 +22,6 @@ def is_a_leader(context, name, cluster_name, time_limit):
|
|||||||
assert time.time() < max_time, "{0} is not a leader in dcs after {1} seconds".format(name, time_limit)
|
assert time.time() < max_time, "{0} is not a leader in dcs after {1} seconds".format(name, time_limit)
|
||||||
|
|
||||||
|
|
||||||
@step('I do a backup of {name:w}')
|
@step('I do a backup of {name:name}')
|
||||||
def do_backup(context, name):
|
def do_backup(context, name):
|
||||||
context.pctl.backup(name)
|
context.pctl.backup(name)
|
||||||
|
|||||||
@@ -11,6 +11,6 @@ def stop_dcs_outage(context):
|
|||||||
context.dcs_ctl.stop_outage()
|
context.dcs_ctl.stop_outage()
|
||||||
|
|
||||||
|
|
||||||
@step('I start {name:w} in a cluster {cluster_name:w} from backup with no_leader')
|
@step('I start {name:name} in a cluster {cluster_name:w} from backup with no_leader')
|
||||||
def start_cluster_from_backup_no_leader(context, name, cluster_name):
|
def start_cluster_from_backup_no_leader(context, name, cluster_name):
|
||||||
context.pctl.bootstrap_from_backup_no_leader(name, cluster_name)
|
context.pctl.bootstrap_from_backup_no_leader(name, cluster_name)
|
||||||
|
|||||||
@@ -1,14 +1,16 @@
|
|||||||
import json
|
import json
|
||||||
import parse
|
|
||||||
import shlex
|
import shlex
|
||||||
import subprocess
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
import time
|
import time
|
||||||
|
|
||||||
|
from datetime import datetime, timedelta
|
||||||
|
|
||||||
|
import parse
|
||||||
import yaml
|
import yaml
|
||||||
|
|
||||||
from behave import register_type, step, then
|
from behave import register_type, step, then
|
||||||
from dateutil import tz
|
from dateutil import tz
|
||||||
from datetime import datetime, timedelta
|
|
||||||
|
|
||||||
tzutc = tz.tzutc()
|
tzutc = tz.tzutc()
|
||||||
|
|
||||||
@@ -26,8 +28,8 @@ register_type(url=parse_url)
|
|||||||
# just rely on the database availability, since there is
|
# just rely on the database availability, since there is
|
||||||
# a short gap between the time PostgreSQL becomes available
|
# a short gap between the time PostgreSQL becomes available
|
||||||
# and Patroni assuming the leader role.
|
# and Patroni assuming the leader role.
|
||||||
@step('{name:w} is a leader after {time_limit:d} seconds')
|
@step('{name:name} is a leader after {time_limit:d} seconds')
|
||||||
@then('{name:w} is a leader after {time_limit:d} seconds')
|
@then('{name:name} is a leader after {time_limit:d} seconds')
|
||||||
def is_a_leader(context, name, time_limit):
|
def is_a_leader(context, name, time_limit):
|
||||||
time_limit *= context.timeout_multiplier
|
time_limit *= context.timeout_multiplier
|
||||||
max_time = time.time() + int(time_limit)
|
max_time = time.time() + int(time_limit)
|
||||||
@@ -95,7 +97,7 @@ def do_run(context, cmd):
|
|||||||
context.response = response.decode('utf-8').strip()
|
context.response = response.decode('utf-8').strip()
|
||||||
|
|
||||||
|
|
||||||
@then('I receive a response {component:w} {data}')
|
@then('I receive a response {component:name} {data}')
|
||||||
def check_response(context, component, data):
|
def check_response(context, component, data):
|
||||||
if component == 'code':
|
if component == 'code':
|
||||||
assert context.status_code == int(data), \
|
assert context.status_code == int(data), \
|
||||||
@@ -114,10 +116,10 @@ def check_response(context, component, data):
|
|||||||
assert str(context.response[component]) == str(data), "{0} does not contain {1}".format(component, data)
|
assert str(context.response[component]) == str(data), "{0} does not contain {1}".format(component, data)
|
||||||
|
|
||||||
|
|
||||||
@step('I issue a scheduled switchover from {from_host:w} to {to_host:w} in {in_seconds:d} seconds')
|
@step('I issue a scheduled switchover from {from_host:name} to {to_host:name} in {in_seconds:d} seconds')
|
||||||
def scheduled_switchover(context, from_host, to_host, in_seconds):
|
def scheduled_switchover(context, from_host, to_host, in_seconds):
|
||||||
context.execute_steps(u"""
|
context.execute_steps(u"""
|
||||||
Given I run patronictl.py switchover batman --master {0} --candidate {1} --scheduled "{2}" --force
|
Given I run patronictl.py switchover batman --primary {0} --candidate {1} --scheduled "{2}" --force
|
||||||
""".format(from_host, to_host, datetime.now(tzutc) + timedelta(seconds=int(in_seconds))))
|
""".format(from_host, to_host, datetime.now(tzutc) + timedelta(seconds=int(in_seconds))))
|
||||||
|
|
||||||
|
|
||||||
@@ -128,13 +130,13 @@ def scheduled_restart(context, url, in_seconds, data):
|
|||||||
context.execute_steps(u"""Given I issue a POST request to {0}/restart with {1}""".format(url, json.dumps(data)))
|
context.execute_steps(u"""Given I issue a POST request to {0}/restart with {1}""".format(url, json.dumps(data)))
|
||||||
|
|
||||||
|
|
||||||
@step('I {action:w} {tag:w} tag in {pg_name:w} config')
|
@step('I {action:w} {tag:w} tag in {pg_name:name} config')
|
||||||
def add_bool_tag_to_config(context, action, tag, pg_name):
|
def add_bool_tag_to_config(context, action, tag, pg_name):
|
||||||
value = action == 'set'
|
value = action == 'set'
|
||||||
context.pctl.add_tag_to_config(pg_name, tag, value)
|
context.pctl.add_tag_to_config(pg_name, tag, value)
|
||||||
|
|
||||||
|
|
||||||
@step('I add tag {tag:w} {value:w} to {pg_name:w} config')
|
@step('I add tag {tag:w} {value:w} to {pg_name:name} config')
|
||||||
def add_tag_to_config(context, tag, value, pg_name):
|
def add_tag_to_config(context, tag, value, pg_name):
|
||||||
context.pctl.add_tag_to_config(pg_name, tag, value)
|
context.pctl.add_tag_to_config(pg_name, tag, value)
|
||||||
|
|
||||||
@@ -158,9 +160,24 @@ def check_http_response(context, url, value, timeout, negate=False):
|
|||||||
if context.certfile:
|
if context.certfile:
|
||||||
url = url.replace('http://', 'https://')
|
url = url.replace('http://', 'https://')
|
||||||
timeout *= context.timeout_multiplier
|
timeout *= context.timeout_multiplier
|
||||||
|
if '=' in value:
|
||||||
|
key, val = value.split('=', 1)
|
||||||
|
else:
|
||||||
|
key, val = value, None
|
||||||
for _ in range(int(timeout)):
|
for _ in range(int(timeout)):
|
||||||
r = context.request_executor.request('GET', url)
|
r = context.request_executor.request('GET', url)
|
||||||
if (value in r.data.decode('utf-8')) != negate:
|
data = r.data.decode('utf-8')
|
||||||
|
if val is not None:
|
||||||
|
try:
|
||||||
|
data = json.loads(data)
|
||||||
|
if negate:
|
||||||
|
if key not in data or data[key] != val:
|
||||||
|
break
|
||||||
|
elif key in data and data[key] == val:
|
||||||
|
break
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
elif (value in r.data.decode('utf-8')) != negate:
|
||||||
break
|
break
|
||||||
time.sleep(1)
|
time.sleep(1)
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -0,0 +1,60 @@
|
|||||||
|
import json
|
||||||
|
import re
|
||||||
|
import time
|
||||||
|
|
||||||
|
from behave import step, then
|
||||||
|
|
||||||
|
|
||||||
|
@step('sync key in DCS has {key:w}={value} after {time_limit:d} seconds')
|
||||||
|
def check_sync(context, key, value, time_limit):
|
||||||
|
time_limit *= context.timeout_multiplier
|
||||||
|
max_time = time.time() + int(time_limit)
|
||||||
|
dcs_value = None
|
||||||
|
while time.time() < max_time:
|
||||||
|
try:
|
||||||
|
response = json.loads(context.dcs_ctl.query('sync'))
|
||||||
|
dcs_value = response.get(key)
|
||||||
|
if key == 'sync_standby' and set((dcs_value or '').split(',')) == set(value.split(',')):
|
||||||
|
return
|
||||||
|
elif str(dcs_value) == value:
|
||||||
|
return
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
time.sleep(1)
|
||||||
|
assert False, "sync does not have {0}={1} (found {2}) in dcs after {3} seconds".format(key, value,
|
||||||
|
dcs_value, time_limit)
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_synchronous_standby_names(value):
|
||||||
|
if '(' in value:
|
||||||
|
m = re.match(r'.*(\d+) \(([^)]+)\)', value)
|
||||||
|
expected_value = set(m.group(2).split())
|
||||||
|
expected_num = m.group(1)
|
||||||
|
else:
|
||||||
|
expected_value = set([value])
|
||||||
|
expected_num = '1'
|
||||||
|
return expected_num, expected_value
|
||||||
|
|
||||||
|
|
||||||
|
@then("synchronous_standby_names on {name:2} is set to '{value}' after {time_limit:d} seconds")
|
||||||
|
def check_synchronous_standby_names(context, name, value, time_limit):
|
||||||
|
time_limit *= context.timeout_multiplier
|
||||||
|
max_time = time.time() + int(time_limit)
|
||||||
|
|
||||||
|
if value == '_empty_str_':
|
||||||
|
value = ''
|
||||||
|
|
||||||
|
expected_num, expected_value = _parse_synchronous_standby_names(value)
|
||||||
|
|
||||||
|
ssn = None
|
||||||
|
while time.time() < max_time:
|
||||||
|
try:
|
||||||
|
ssn = context.pctl.query(name, "SHOW synchronous_standby_names").fetchone()[0]
|
||||||
|
db_num, db_value = _parse_synchronous_standby_names(ssn)
|
||||||
|
if expected_value == db_value and expected_num == db_num:
|
||||||
|
return
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
time.sleep(1)
|
||||||
|
assert False, "synchronous_standby_names is not set to '{0}' (found '{1}') after {2} seconds".format(value, ssn,
|
||||||
|
time_limit)
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
import os
|
||||||
|
|
||||||
|
from behave import step
|
||||||
|
|
||||||
|
|
||||||
|
@step('I ensure {name:name} fails to start after a failure')
|
||||||
|
def spoil_autoconf(context, name):
|
||||||
|
with open(os.path.join(context.pctl._processes[name]._data_dir, 'postgresql.auto.conf'), 'w') as f:
|
||||||
|
f.write('foo=bar')
|
||||||
+37
-17
@@ -2,23 +2,22 @@ import json
|
|||||||
import time
|
import time
|
||||||
|
|
||||||
from behave import step, then
|
from behave import step, then
|
||||||
|
|
||||||
import patroni.psycopg as pg
|
import patroni.psycopg as pg
|
||||||
|
|
||||||
|
|
||||||
@step('I create a logical replication slot {slot_name} on {pg_name:w} with the {plugin:w} plugin')
|
@step('I create a logical {slot_type} slot {slot_name} on {pg_name:name} with the {plugin:w} plugin')
|
||||||
def create_logical_replication_slot(context, slot_name, pg_name, plugin):
|
def create_logical_replication_slot(context, slot_type, slot_name, pg_name, plugin):
|
||||||
|
failover = ', failover=>true' if slot_type == 'failover' else ''
|
||||||
try:
|
try:
|
||||||
output = context.pctl.query(pg_name, ("SELECT pg_create_logical_replication_slot('{0}', '{1}'),"
|
context.pctl.query(pg_name, f"SELECT pg_create_logical_replication_slot('{slot_name}', '{plugin}'{failover})")
|
||||||
" current_database()").format(slot_name, plugin))
|
|
||||||
print(output.fetchone())
|
|
||||||
except pg.Error as e:
|
except pg.Error as e:
|
||||||
print(e)
|
assert False, "Error creating slot {0} on {1} with plugin {2}: {3}".format(slot_name, pg_name, plugin, e)
|
||||||
assert False, "Error creating slot {0} on {1} with plugin {2}".format(slot_name, pg_name, plugin)
|
|
||||||
|
|
||||||
|
|
||||||
@step('{pg_name:w} has a logical replication slot named {slot_name}'
|
@step('{pg_name:name} has a logical replication slot named {slot_name}'
|
||||||
' with the {plugin:w} plugin after {time_limit:d} seconds')
|
' with the {plugin:w} plugin after {time_limit:d} seconds')
|
||||||
@then('{pg_name:w} has a logical replication slot named {slot_name}'
|
@then('{pg_name:name} has a logical replication slot named {slot_name}'
|
||||||
' with the {plugin:w} plugin after {time_limit:d} seconds')
|
' with the {plugin:w} plugin after {time_limit:d} seconds')
|
||||||
def has_logical_replication_slot(context, pg_name, slot_name, plugin, time_limit):
|
def has_logical_replication_slot(context, pg_name, slot_name, plugin, time_limit):
|
||||||
time_limit *= context.timeout_multiplier
|
time_limit *= context.timeout_multiplier
|
||||||
@@ -37,8 +36,8 @@ def has_logical_replication_slot(context, pg_name, slot_name, plugin, time_limit
|
|||||||
assert False, f"Error looking for slot {slot_name} on {pg_name} with plugin {plugin}"
|
assert False, f"Error looking for slot {slot_name} on {pg_name} with plugin {plugin}"
|
||||||
|
|
||||||
|
|
||||||
@step('{pg_name:w} does not have a replication slot named {slot_name:w}')
|
@step('{pg_name:name} does not have a replication slot named {slot_name:w}')
|
||||||
@then('{pg_name:w} does not have a replication slot named {slot_name:w}')
|
@then('{pg_name:name} does not have a replication slot named {slot_name:w}')
|
||||||
def does_not_have_replication_slot(context, pg_name, slot_name):
|
def does_not_have_replication_slot(context, pg_name, slot_name):
|
||||||
try:
|
try:
|
||||||
row = context.pctl.query(pg_name, ("SELECT 1 FROM pg_replication_slots"
|
row = context.pctl.query(pg_name, ("SELECT 1 FROM pg_replication_slots"
|
||||||
@@ -48,7 +47,8 @@ def does_not_have_replication_slot(context, pg_name, slot_name):
|
|||||||
assert False, "Error looking for slot {0} on {1}".format(slot_name, pg_name)
|
assert False, "Error looking for slot {0} on {1}".format(slot_name, pg_name)
|
||||||
|
|
||||||
|
|
||||||
@step('{slot_type:w} slot {slot_name:w} is in sync between {pg_name1:w} and {pg_name2:w} after {time_limit:d} seconds')
|
@step('{slot_type:w} slot {slot_name:w} is in sync between '
|
||||||
|
'{pg_name1:name} and {pg_name2:name} after {time_limit:d} seconds')
|
||||||
def slots_in_sync(context, slot_type, slot_name, pg_name1, pg_name2, time_limit):
|
def slots_in_sync(context, slot_type, slot_name, pg_name1, pg_name2, time_limit):
|
||||||
time_limit *= context.timeout_multiplier
|
time_limit *= context.timeout_multiplier
|
||||||
max_time = time.time() + int(time_limit)
|
max_time = time.time() + int(time_limit)
|
||||||
@@ -67,17 +67,17 @@ def slots_in_sync(context, slot_type, slot_name, pg_name1, pg_name2, time_limit)
|
|||||||
f"{slot_type} slot {slot_name} is not in sync between {pg_name1} and {pg_name2} after {time_limit} seconds"
|
f"{slot_type} slot {slot_name} is not in sync between {pg_name1} and {pg_name2} after {time_limit} seconds"
|
||||||
|
|
||||||
|
|
||||||
@step('I get all changes from logical slot {slot_name:w} on {pg_name:w}')
|
@step('I get all changes from logical slot {slot_name:w} on {pg_name:name}')
|
||||||
def logical_slot_get_changes(context, slot_name, pg_name):
|
def logical_slot_get_changes(context, slot_name, pg_name):
|
||||||
context.pctl.query(pg_name, "SELECT * FROM pg_logical_slot_get_changes('{0}', NULL, NULL)".format(slot_name))
|
context.pctl.query(pg_name, "SELECT * FROM pg_logical_slot_get_changes('{0}', NULL, NULL)".format(slot_name))
|
||||||
|
|
||||||
|
|
||||||
@step('I get all changes from physical slot {slot_name:w} on {pg_name:w}')
|
@step('I get all changes from physical slot {slot_name:w} on {pg_name:name}')
|
||||||
def physical_slot_get_changes(context, slot_name, pg_name):
|
def physical_slot_get_changes(context, slot_name, pg_name):
|
||||||
context.pctl.query(pg_name, f"SELECT * FROM pg_replication_slot_advance('{slot_name}', pg_current_wal_lsn())")
|
context.pctl.query(pg_name, f"SELECT * FROM pg_replication_slot_advance('{slot_name}', pg_current_wal_lsn())")
|
||||||
|
|
||||||
|
|
||||||
@step('{pg_name:w} has a physical replication slot named {slot_name} after {time_limit:d} seconds')
|
@step('{pg_name:name} has a physical replication slot named {slot_name} after {time_limit:d} seconds')
|
||||||
def has_physical_replication_slot(context, pg_name, slot_name, time_limit):
|
def has_physical_replication_slot(context, pg_name, slot_name, time_limit):
|
||||||
time_limit *= context.timeout_multiplier
|
time_limit *= context.timeout_multiplier
|
||||||
max_time = time.time() + int(time_limit)
|
max_time = time.time() + int(time_limit)
|
||||||
@@ -93,13 +93,33 @@ def has_physical_replication_slot(context, pg_name, slot_name, time_limit):
|
|||||||
assert False, f"Physical slot {slot_name} doesn't exist after {time_limit} seconds"
|
assert False, f"Physical slot {slot_name} doesn't exist after {time_limit} seconds"
|
||||||
|
|
||||||
|
|
||||||
@step('"{name}" key in DCS has {subkey:w} in {key:w}')
|
@step('physical replication slot named {slot_name} on {pg_name:name} has no xmin value after {time_limit:d} seconds')
|
||||||
|
def physical_slot_no_xmin(context, pg_name, slot_name, time_limit):
|
||||||
|
time_limit *= context.timeout_multiplier
|
||||||
|
max_time = time.time() + int(time_limit)
|
||||||
|
query = "SELECT xmin FROM pg_catalog.pg_replication_slots WHERE slot_type = 'physical'" +\
|
||||||
|
f" AND slot_name = '{slot_name}'"
|
||||||
|
exists = False
|
||||||
|
while time.time() < max_time:
|
||||||
|
try:
|
||||||
|
row = context.pctl.query(pg_name, query).fetchone()
|
||||||
|
exists = bool(row)
|
||||||
|
if exists and row[0] is None:
|
||||||
|
return
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
time.sleep(1)
|
||||||
|
assert False, f"Physical slot {slot_name} doesn't exist after {time_limit} seconds" if not exists \
|
||||||
|
else f"Physical slot {slot_name} has xmin value after {time_limit} seconds"
|
||||||
|
|
||||||
|
|
||||||
|
@step('"{name}" key in DCS has {subkey} in {key:w}')
|
||||||
def dcs_key_contains(context, name, subkey, key):
|
def dcs_key_contains(context, name, subkey, key):
|
||||||
response = json.loads(context.dcs_ctl.query(name))
|
response = json.loads(context.dcs_ctl.query(name))
|
||||||
assert key in response and subkey in response[key], f"{name} key in DCS doesn't have {subkey} in {key}"
|
assert key in response and subkey in response[key], f"{name} key in DCS doesn't have {subkey} in {key}"
|
||||||
|
|
||||||
|
|
||||||
@step('"{name}" key in DCS does not have {subkey:w} in {key:w}')
|
@step('"{name}" key in DCS does not have {subkey} in {key:w}')
|
||||||
def dcs_key_does_not_contain(context, name, subkey, key):
|
def dcs_key_does_not_contain(context, name, subkey, key):
|
||||||
response = json.loads(context.dcs_ctl.query(name))
|
response = json.loads(context.dcs_ctl.query(name))
|
||||||
assert key not in response or subkey not in response[key], f"{name} key in DCS has {subkey} in {key}"
|
assert key not in response or subkey not in response[key], f"{name} key in DCS has {subkey} in {key}"
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ def callbacks(context, name):
|
|||||||
for c in ('on_start', 'on_stop', 'on_restart', 'on_role_change')}
|
for c in ('on_start', 'on_stop', 'on_restart', 'on_role_change')}
|
||||||
|
|
||||||
|
|
||||||
@step('I start {name:w} in a cluster {cluster_name:w}')
|
@step('I start {name:name} in a cluster {cluster_name:w}')
|
||||||
def start_patroni(context, name, cluster_name):
|
def start_patroni(context, name, cluster_name):
|
||||||
return context.pctl.start(name, custom_config={
|
return context.pctl.start(name, custom_config={
|
||||||
"scope": cluster_name,
|
"scope": cluster_name,
|
||||||
@@ -20,7 +20,7 @@ def start_patroni(context, name, cluster_name):
|
|||||||
})
|
})
|
||||||
|
|
||||||
|
|
||||||
@step('I start {name:w} in a standby cluster {cluster_name:w} as a clone of {name2:w}')
|
@step('I start {name:name} in a standby cluster {cluster_name:w} as a clone of {name2:name}')
|
||||||
def start_patroni_standby_cluster(context, name, cluster_name, name2):
|
def start_patroni_standby_cluster(context, name, cluster_name, name2):
|
||||||
# we need to remove patroni.dynamic.json in order to "bootstrap" standby cluster with existing PGDATA
|
# we need to remove patroni.dynamic.json in order to "bootstrap" standby cluster with existing PGDATA
|
||||||
os.unlink(os.path.join(context.pctl._processes[name]._data_dir, 'patroni.dynamic.json'))
|
os.unlink(os.path.join(context.pctl._processes[name]._data_dir, 'patroni.dynamic.json'))
|
||||||
@@ -49,7 +49,7 @@ def start_patroni_standby_cluster(context, name, cluster_name, name2):
|
|||||||
return context.pctl.start(name)
|
return context.pctl.start(name)
|
||||||
|
|
||||||
|
|
||||||
@step('{pg_name1:w} is replicating from {pg_name2:w} after {timeout:d} seconds')
|
@step('{pg_name1:name} is replicating from {pg_name2:name} after {timeout:d} seconds')
|
||||||
def check_replication_status(context, pg_name1, pg_name2, timeout):
|
def check_replication_status(context, pg_name1, pg_name2, timeout):
|
||||||
bound_time = time.time() + timeout * context.timeout_multiplier
|
bound_time = time.time() + timeout * context.timeout_multiplier
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
from behave import step, then
|
|
||||||
import time
|
import time
|
||||||
|
|
||||||
|
from behave import step, then
|
||||||
|
|
||||||
|
|
||||||
def polling_loop(timeout, interval=1):
|
def polling_loop(timeout, interval=1):
|
||||||
"""Returns an iterator that returns values until timeout has passed. Timeout is measured from start of iteration."""
|
"""Returns an iterator that returns values until timeout has passed. Timeout is measured from start of iteration."""
|
||||||
@@ -13,12 +14,12 @@ def polling_loop(timeout, interval=1):
|
|||||||
time.sleep(interval)
|
time.sleep(interval)
|
||||||
|
|
||||||
|
|
||||||
@step('I start {name:w} with watchdog')
|
@step('I start {name:name} with watchdog')
|
||||||
def start_patroni_with_watchdog(context, name):
|
def start_patroni_with_watchdog(context, name):
|
||||||
return context.pctl.start(name, custom_config={'watchdog': True, 'bootstrap': {'dcs': {'ttl': 20}}})
|
return context.pctl.start(name, custom_config={'watchdog': True, 'bootstrap': {'dcs': {'ttl': 20}}})
|
||||||
|
|
||||||
|
|
||||||
@step('{name:w} watchdog has been pinged after {timeout:d} seconds')
|
@step('{name:name} watchdog has been pinged after {timeout:d} seconds')
|
||||||
def watchdog_was_pinged(context, name, timeout):
|
def watchdog_was_pinged(context, name, timeout):
|
||||||
for _ in polling_loop(timeout):
|
for _ in polling_loop(timeout):
|
||||||
if context.pctl.get_watchdog(name).was_pinged:
|
if context.pctl.get_watchdog(name).was_pinged:
|
||||||
@@ -26,22 +27,22 @@ def watchdog_was_pinged(context, name, timeout):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
@then('{name:w} watchdog has been closed')
|
@then('{name:name} watchdog has been closed')
|
||||||
def watchdog_was_closed(context, name):
|
def watchdog_was_closed(context, name):
|
||||||
assert context.pctl.get_watchdog(name).was_closed
|
assert context.pctl.get_watchdog(name).was_closed
|
||||||
|
|
||||||
|
|
||||||
@step('{name:w} watchdog has a {timeout:d} second timeout')
|
@step('{name:name} watchdog has a {timeout:d} second timeout')
|
||||||
def watchdog_has_timeout(context, name, timeout):
|
def watchdog_has_timeout(context, name, timeout):
|
||||||
assert context.pctl.get_watchdog(name).timeout == timeout
|
assert context.pctl.get_watchdog(name).timeout == timeout
|
||||||
|
|
||||||
|
|
||||||
@step('I reset {name:w} watchdog state')
|
@step('I reset {name:name} watchdog state')
|
||||||
def watchdog_reset_pinged(context, name):
|
def watchdog_reset_pinged(context, name):
|
||||||
context.pctl.get_watchdog(name).reset()
|
context.pctl.get_watchdog(name).reset()
|
||||||
|
|
||||||
|
|
||||||
@then('{name:w} watchdog is triggered after {timeout:d} seconds')
|
@then('{name:name} watchdog is triggered after {timeout:d} seconds')
|
||||||
def watchdog_was_triggered(context, name, timeout):
|
def watchdog_was_triggered(context, name, timeout):
|
||||||
for _ in polling_loop(timeout):
|
for _ in polling_loop(timeout):
|
||||||
if context.pctl.get_watchdog(name).was_triggered:
|
if context.pctl.get_watchdog(name).was_triggered:
|
||||||
@@ -49,6 +50,6 @@ def watchdog_was_triggered(context, name, timeout):
|
|||||||
assert False
|
assert False
|
||||||
|
|
||||||
|
|
||||||
@step('{name:w} hangs for {timeout:d} seconds')
|
@step('{name:name} hangs for {timeout:d} seconds')
|
||||||
def patroni_hang(context, name, timeout):
|
def patroni_hang(context, name, timeout):
|
||||||
return context.pctl.patroni_hang(name, timeout)
|
return context.pctl.patroni_hang(name, timeout)
|
||||||
|
|||||||
+16
-16
@@ -2,38 +2,38 @@ Feature: watchdog
|
|||||||
Verify that watchdog gets pinged and triggered under appropriate circumstances.
|
Verify that watchdog gets pinged and triggered under appropriate circumstances.
|
||||||
|
|
||||||
Scenario: watchdog is opened and pinged
|
Scenario: watchdog is opened and pinged
|
||||||
Given I start postgres0 with watchdog
|
Given I start postgres-0 with watchdog
|
||||||
Then postgres0 is a leader after 10 seconds
|
Then postgres-0 is a leader after 10 seconds
|
||||||
And postgres0 role is the primary after 10 seconds
|
And postgres-0 role is the primary after 10 seconds
|
||||||
And postgres0 watchdog has been pinged after 10 seconds
|
And postgres-0 watchdog has been pinged after 10 seconds
|
||||||
And postgres0 watchdog has a 15 second timeout
|
And postgres-0 watchdog has a 15 second timeout
|
||||||
|
|
||||||
Scenario: watchdog is reconfigured after global ttl changed
|
Scenario: watchdog is reconfigured after global ttl changed
|
||||||
Given I run patronictl.py edit-config batman -s ttl=30 --force
|
Given I run patronictl.py edit-config batman -s ttl=30 --force
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And I receive a response output "+ttl: 30"
|
And I receive a response output "+ttl: 30"
|
||||||
When I sleep for 4 seconds
|
When I sleep for 4 seconds
|
||||||
Then postgres0 watchdog has a 25 second timeout
|
Then postgres-0 watchdog has a 25 second timeout
|
||||||
|
|
||||||
Scenario: watchdog is disabled during pause
|
Scenario: watchdog is disabled during pause
|
||||||
Given I run patronictl.py pause batman
|
Given I run patronictl.py pause batman
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
When I sleep for 2 seconds
|
When I sleep for 2 seconds
|
||||||
Then postgres0 watchdog has been closed
|
Then postgres-0 watchdog has been closed
|
||||||
|
|
||||||
Scenario: watchdog is opened and pinged after resume
|
Scenario: watchdog is opened and pinged after resume
|
||||||
Given I reset postgres0 watchdog state
|
Given I reset postgres-0 watchdog state
|
||||||
And I run patronictl.py resume batman
|
And I run patronictl.py resume batman
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And postgres0 watchdog has been pinged after 10 seconds
|
And postgres-0 watchdog has been pinged after 10 seconds
|
||||||
|
|
||||||
Scenario: watchdog is disabled when shutting down
|
Scenario: watchdog is disabled when shutting down
|
||||||
Given I shut down postgres0
|
Given I shut down postgres-0
|
||||||
Then postgres0 watchdog has been closed
|
Then postgres-0 watchdog has been closed
|
||||||
|
|
||||||
Scenario: watchdog is triggered if patroni stops responding
|
Scenario: watchdog is triggered if patroni stops responding
|
||||||
Given I reset postgres0 watchdog state
|
Given I reset postgres-0 watchdog state
|
||||||
And I start postgres0 with watchdog
|
And I start postgres-0 with watchdog
|
||||||
Then postgres0 role is the primary after 10 seconds
|
Then postgres-0 role is the primary after 10 seconds
|
||||||
When postgres0 hangs for 30 seconds
|
When postgres-0 hangs for 30 seconds
|
||||||
Then postgres0 watchdog is triggered after 30 seconds
|
Then postgres-0 watchdog is triggered after 30 seconds
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ RUN export DEBIAN_FRONTEND=noninteractive \
|
|||||||
## Make sure we have a en_US.UTF-8 locale available
|
## Make sure we have a en_US.UTF-8 locale available
|
||||||
&& localedef -i en_US -c -f UTF-8 -A /usr/share/locale/locale.alias en_US.UTF-8 \
|
&& localedef -i en_US -c -f UTF-8 -A /usr/share/locale/locale.alias en_US.UTF-8 \
|
||||||
&& pip3 install --break-system-packages setuptools \
|
&& pip3 install --break-system-packages setuptools \
|
||||||
&& pip3 install --break-system-packages 'git+https://github.com/zalando/patroni.git#egg=patroni[kubernetes]' \
|
&& pip3 install --break-system-packages 'git+https://github.com/patroni/patroni.git#egg=patroni[kubernetes]' \
|
||||||
&& PGHOME=/home/postgres \
|
&& PGHOME=/home/postgres \
|
||||||
&& mkdir -p $PGHOME \
|
&& mkdir -p $PGHOME \
|
||||||
&& chown postgres $PGHOME \
|
&& chown postgres $PGHOME \
|
||||||
|
|||||||
@@ -27,7 +27,7 @@ RUN export DEBIAN_FRONTEND=noninteractive \
|
|||||||
&& apt-get -y install postgresql-16-citus-12.1; \
|
&& apt-get -y install postgresql-16-citus-12.1; \
|
||||||
fi \
|
fi \
|
||||||
&& pip3 install --break-system-packages setuptools \
|
&& pip3 install --break-system-packages setuptools \
|
||||||
&& pip3 install --break-system-packages 'git+https://github.com/zalando/patroni.git#egg=patroni[kubernetes]' \
|
&& pip3 install --break-system-packages 'git+https://github.com/patroni/patroni.git#egg=patroni[kubernetes]' \
|
||||||
&& PGHOME=/home/postgres \
|
&& PGHOME=/home/postgres \
|
||||||
&& mkdir -p $PGHOME \
|
&& mkdir -p $PGHOME \
|
||||||
&& chown postgres $PGHOME \
|
&& chown postgres $PGHOME \
|
||||||
|
|||||||
+28
-24
@@ -23,7 +23,7 @@ Example session:
|
|||||||
|
|
||||||
$ docker build -t patroni .
|
$ docker build -t patroni .
|
||||||
Sending build context to Docker daemon 138.8kB
|
Sending build context to Docker daemon 138.8kB
|
||||||
Step 1/9 : FROM postgres:15
|
Step 1/9 : FROM postgres:16
|
||||||
...
|
...
|
||||||
Successfully built e9bfe69c5d2b
|
Successfully built e9bfe69c5d2b
|
||||||
Successfully tagged patroni:latest
|
Successfully tagged patroni:latest
|
||||||
@@ -46,7 +46,7 @@ Example session:
|
|||||||
|
|
||||||
$ kubectl get pods -L role
|
$ kubectl get pods -L role
|
||||||
NAME READY STATUS RESTARTS AGE ROLE
|
NAME READY STATUS RESTARTS AGE ROLE
|
||||||
patronidemo-0 1/1 Running 0 34s master
|
patronidemo-0 1/1 Running 0 34s primary
|
||||||
patronidemo-1 1/1 Running 0 30s replica
|
patronidemo-1 1/1 Running 0 30s replica
|
||||||
patronidemo-2 1/1 Running 0 26s replica
|
patronidemo-2 1/1 Running 0 26s replica
|
||||||
|
|
||||||
@@ -82,7 +82,7 @@ Example session:
|
|||||||
|
|
||||||
demo@localhost:~/git/patroni/kubernetes$ docker build -f Dockerfile.citus -t patroni-citus-k8s .
|
demo@localhost:~/git/patroni/kubernetes$ docker build -f Dockerfile.citus -t patroni-citus-k8s .
|
||||||
Sending build context to Docker daemon 138.8kB
|
Sending build context to Docker daemon 138.8kB
|
||||||
Step 1/11 : FROM postgres:15
|
Step 1/11 : FROM postgres:16
|
||||||
...
|
...
|
||||||
Successfully built 8cd73e325028
|
Successfully built 8cd73e325028
|
||||||
Successfully tagged patroni-citus-k8s:latest
|
Successfully tagged patroni-citus-k8s:latest
|
||||||
@@ -119,36 +119,40 @@ Example session:
|
|||||||
|
|
||||||
$ kubectl get pods -l cluster-name=citusdemo -L role
|
$ kubectl get pods -l cluster-name=citusdemo -L role
|
||||||
NAME READY STATUS RESTARTS AGE ROLE
|
NAME READY STATUS RESTARTS AGE ROLE
|
||||||
citusdemo-0-0 1/1 Running 0 105s master
|
citusdemo-0-0 1/1 Running 0 105s primary
|
||||||
citusdemo-0-1 1/1 Running 0 101s replica
|
citusdemo-0-1 1/1 Running 0 101s replica
|
||||||
citusdemo-0-2 1/1 Running 0 96s replica
|
citusdemo-0-2 1/1 Running 0 96s replica
|
||||||
citusdemo-1-0 1/1 Running 0 105s master
|
citusdemo-1-0 1/1 Running 0 105s primary
|
||||||
citusdemo-1-1 1/1 Running 0 101s replica
|
citusdemo-1-1 1/1 Running 0 101s replica
|
||||||
citusdemo-2-0 1/1 Running 0 105s master
|
citusdemo-2-0 1/1 Running 0 105s primary
|
||||||
citusdemo-2-1 1/1 Running 0 101s replica
|
citusdemo-2-1 1/1 Running 0 101s replica
|
||||||
|
|
||||||
$ kubectl exec -ti citusdemo-0-0 -- bash
|
$ kubectl exec -ti citusdemo-0-0 -- bash
|
||||||
postgres@citusdemo-0-0:~$ patronictl list
|
postgres@citusdemo-0-0:~$ patronictl list
|
||||||
+ Citus cluster: citusdemo -----------+--------------+---------+----+-----------+
|
+ Citus cluster: citusdemo -----------+----------------+---------+----+-----------+
|
||||||
| Group | Member | Host | Role | State | TL | Lag in MB |
|
| Group | Member | Host | Role | State | TL | Lag in MB |
|
||||||
+-------+---------------+-------------+--------------+---------+----+-----------+
|
+-------+---------------+-------------+----------------+---------+----+-----------+
|
||||||
| 0 | citusdemo-0-0 | 10.244.0.10 | Leader | running | 1 | |
|
| 0 | citusdemo-0-0 | 10.244.0.10 | Leader | running | 1 | |
|
||||||
| 0 | citusdemo-0-1 | 10.244.0.12 | Replica | running | 1 | 0 |
|
| 0 | citusdemo-0-1 | 10.244.0.12 | Replica | running | 1 | 0 |
|
||||||
| 0 | citusdemo-0-2 | 10.244.0.14 | Sync Standby | running | 1 | 0 |
|
| 0 | citusdemo-0-2 | 10.244.0.14 | Quorum Standby | running | 1 | 0 |
|
||||||
| 1 | citusdemo-1-0 | 10.244.0.8 | Leader | running | 1 | |
|
| 1 | citusdemo-1-0 | 10.244.0.8 | Leader | running | 1 | |
|
||||||
| 1 | citusdemo-1-1 | 10.244.0.11 | Sync Standby | running | 1 | 0 |
|
| 1 | citusdemo-1-1 | 10.244.0.11 | Quorum Standby | running | 1 | 0 |
|
||||||
| 2 | citusdemo-2-0 | 10.244.0.9 | Leader | running | 1 | |
|
| 2 | citusdemo-2-0 | 10.244.0.9 | Leader | running | 1 | |
|
||||||
| 2 | citusdemo-2-1 | 10.244.0.13 | Sync Standby | running | 1 | 0 |
|
| 2 | citusdemo-2-1 | 10.244.0.13 | Quorum Standby | running | 1 | 0 |
|
||||||
+-------+---------------+-------------+--------------+---------+----+-----------+
|
+-------+---------------+-------------+----------------+---------+----+-----------+
|
||||||
|
|
||||||
postgres@citusdemo-0-0:~$ psql citus
|
postgres@citusdemo-0-0:~$ psql citus
|
||||||
psql (15.1 (Debian 15.1-1.pgdg110+1))
|
psql (16.4 (Debian 16.4-1.pgdg120+1))
|
||||||
Type "help" for help.
|
Type "help" for help.
|
||||||
|
|
||||||
citus=# table pg_dist_node;
|
citus=# table pg_dist_node;
|
||||||
nodeid | groupid | nodename | nodeport | noderack | hasmetadata | isactive | noderole | nodecluster | metadatasynced | shouldhaveshards
|
nodeid | groupid | nodename | nodeport | noderack | hasmetadata | isactive | noderole | nodecluster | metadatasynced | shouldhaveshards
|
||||||
--------+---------+-------------+----------+----------+-------------+----------+----------+-------------+----------------+------------------
|
--------+---------+-------------+----------+----------+-------------+----------+-----------+-------------+----------------+------------------
|
||||||
1 | 0 | 10.244.0.10 | 5432 | default | t | t | primary | default | t | f
|
1 | 0 | 10.244.0.10 | 5432 | default | t | t | primary | default | t | f
|
||||||
2 | 1 | 10.244.0.8 | 5432 | default | t | t | primary | default | t | t
|
2 | 1 | 10.244.0.8 | 5432 | default | t | t | primary | default | t | t
|
||||||
3 | 2 | 10.244.0.9 | 5432 | default | t | t | primary | default | t | t
|
3 | 2 | 10.244.0.9 | 5432 | default | t | t | primary | default | t | t
|
||||||
(3 rows)
|
4 | 0 | 10.244.0.14 | 5432 | default | t | t | secondary | default | t | f
|
||||||
|
5 | 0 | 10.244.0.12 | 5432 | default | t | t | secondary | default | t | f
|
||||||
|
6 | 1 | 10.244.0.11 | 5432 | default | t | t | secondary | default | t | t
|
||||||
|
7 | 2 | 10.244.0.13 | 5432 | default | t | t | secondary | default | t | t
|
||||||
|
(7 rows)
|
||||||
|
|||||||
@@ -59,7 +59,7 @@ spec:
|
|||||||
serviceAccountName: citusdemo
|
serviceAccountName: citusdemo
|
||||||
containers:
|
containers:
|
||||||
- name: *cluster_name
|
- name: *cluster_name
|
||||||
image: patroni-citus-k8s # docker build -f Dockerfile.citus -t patroni-citus-k8s .
|
image: harbor.optimcloud.com/library/optim/patroni-citus-k8s # docker build -f Dockerfile.citus -t patroni-citus-k8s .
|
||||||
imagePullPolicy: IfNotPresent
|
imagePullPolicy: IfNotPresent
|
||||||
readinessProbe:
|
readinessProbe:
|
||||||
httpGet:
|
httpGet:
|
||||||
@@ -169,7 +169,7 @@ spec:
|
|||||||
serviceAccountName: citusdemo
|
serviceAccountName: citusdemo
|
||||||
containers:
|
containers:
|
||||||
- name: *cluster_name
|
- name: *cluster_name
|
||||||
image: patroni-citus-k8s # docker build -f Dockerfile.citus -t patroni-citus-k8s .
|
image: harbor.optimcloud.com/library/optim/patroni-citus-k8s # docker build -f Dockerfile.citus -t patroni-citus-k8s .
|
||||||
imagePullPolicy: IfNotPresent
|
imagePullPolicy: IfNotPresent
|
||||||
readinessProbe:
|
readinessProbe:
|
||||||
httpGet:
|
httpGet:
|
||||||
@@ -279,7 +279,7 @@ spec:
|
|||||||
serviceAccountName: citusdemo
|
serviceAccountName: citusdemo
|
||||||
containers:
|
containers:
|
||||||
- name: *cluster_name
|
- name: *cluster_name
|
||||||
image: patroni-citus-k8s # docker build -f Dockerfile.citus -t patroni-citus-k8s .
|
image: harbor.optimcloud.com/library/optim/patroni-citus-k8s # docker build -f Dockerfile.citus -t patroni-citus-k8s .
|
||||||
imagePullPolicy: IfNotPresent
|
imagePullPolicy: IfNotPresent
|
||||||
readinessProbe:
|
readinessProbe:
|
||||||
httpGet:
|
httpGet:
|
||||||
@@ -458,7 +458,7 @@ metadata:
|
|||||||
application: patroni
|
application: patroni
|
||||||
cluster-name: citusdemo
|
cluster-name: citusdemo
|
||||||
citus-type: worker
|
citus-type: worker
|
||||||
role: master
|
role: primary
|
||||||
spec:
|
spec:
|
||||||
type: ClusterIP
|
type: ClusterIP
|
||||||
selector:
|
selector:
|
||||||
|
|||||||
@@ -15,6 +15,7 @@ bootstrap:
|
|||||||
pg_hba:
|
pg_hba:
|
||||||
- host all all 0.0.0.0/0 md5
|
- host all all 0.0.0.0/0 md5
|
||||||
- host replication ${PATRONI_REPLICATION_USERNAME} ${PATRONI_KUBERNETES_POD_IP}/16 md5
|
- host replication ${PATRONI_REPLICATION_USERNAME} ${PATRONI_KUBERNETES_POD_IP}/16 md5
|
||||||
|
- host replication ${PATRONI_REPLICATION_USERNAME} 127.0.0.1/32 md5
|
||||||
initdb:
|
initdb:
|
||||||
- auth-host: md5
|
- auth-host: md5
|
||||||
- auth-local: trust
|
- auth-local: trust
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ Note: If deploying as a template for multiple users, the following commands shou
|
|||||||
|
|
||||||
```
|
```
|
||||||
oc import-image postgres:10 --confirm -n openshift
|
oc import-image postgres:10 --confirm -n openshift
|
||||||
oc new-build https://github.com/zalando/patroni --context-dir=kubernetes -n openshift
|
oc new-build https://github.com/patroni/patroni --context-dir=kubernetes -n openshift
|
||||||
```
|
```
|
||||||
|
|
||||||
## Deploy the Image
|
## Deploy the Image
|
||||||
|
|||||||
@@ -36,7 +36,7 @@ objects:
|
|||||||
labels:
|
labels:
|
||||||
application: ${APPLICATION_NAME}
|
application: ${APPLICATION_NAME}
|
||||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||||
name: ${PATRONI_MASTER_SERVICE_NAME}
|
name: ${PATRONI_PRIMARY_SERVICE_NAME}
|
||||||
spec:
|
spec:
|
||||||
ports:
|
ports:
|
||||||
- port: 5432
|
- port: 5432
|
||||||
@@ -45,7 +45,7 @@ objects:
|
|||||||
selector:
|
selector:
|
||||||
application: ${APPLICATION_NAME}
|
application: ${APPLICATION_NAME}
|
||||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||||
role: master
|
role: primary
|
||||||
sessionAffinity: None
|
sessionAffinity: None
|
||||||
type: ClusterIP
|
type: ClusterIP
|
||||||
status:
|
status:
|
||||||
@@ -289,12 +289,12 @@ parameters:
|
|||||||
displayName: Cluster Name
|
displayName: Cluster Name
|
||||||
name: PATRONI_CLUSTER_NAME
|
name: PATRONI_CLUSTER_NAME
|
||||||
value: patroni-ephemeral
|
value: patroni-ephemeral
|
||||||
- description: The name of the OpenShift Service exposed for the patroni-ephemeral-master container.
|
- description: The name of the OpenShift Service exposed for the patroni-ephemeral-primary container.
|
||||||
displayName: Master service name.
|
displayName: Primary service name.
|
||||||
name: PATRONI_MASTER_SERVICE_NAME
|
name: PATRONI_PRIMARY_SERVICE_NAME
|
||||||
value: patroni-ephemeral-master
|
value: patroni-ephemeral-primary
|
||||||
- description: The name of the OpenShift Service exposed for the patroni-ephemeral-replica containers.
|
- description: The name of the OpenShift Service exposed for the patroni-ephemeral-replica containers.
|
||||||
displayName: Replica service name.
|
displayName: Replica service name.
|
||||||
name: PATRONI_REPLICA_SERVICE_NAME
|
name: PATRONI_REPLICA_SERVICE_NAME
|
||||||
value: patroni-ephemeral-replica
|
value: patroni-ephemeral-replica
|
||||||
- description: Maximum amount of memory the container can use.
|
- description: Maximum amount of memory the container can use.
|
||||||
@@ -310,7 +310,7 @@ parameters:
|
|||||||
name: PATRONI_SUPERUSER_USERNAME
|
name: PATRONI_SUPERUSER_USERNAME
|
||||||
value: postgres
|
value: postgres
|
||||||
- description: Password of the superuser account for initialization.
|
- description: Password of the superuser account for initialization.
|
||||||
displayName: Superuser Passsword
|
displayName: Superuser Password
|
||||||
name: PATRONI_SUPERUSER_PASSWORD
|
name: PATRONI_SUPERUSER_PASSWORD
|
||||||
value: postgres
|
value: postgres
|
||||||
- description: Username of the replication account for initialization.
|
- description: Username of the replication account for initialization.
|
||||||
@@ -318,10 +318,10 @@ parameters:
|
|||||||
name: PATRONI_REPLICATION_USERNAME
|
name: PATRONI_REPLICATION_USERNAME
|
||||||
value: postgres
|
value: postgres
|
||||||
- description: Password of the replication account for initialization.
|
- description: Password of the replication account for initialization.
|
||||||
displayName: Repication Passsword
|
displayName: Repication Password
|
||||||
name: PATRONI_REPLICATION_PASSWORD
|
name: PATRONI_REPLICATION_PASSWORD
|
||||||
value: postgres
|
value: postgres
|
||||||
- description: Service account name used for pods and rolebindings to form a cluster in the project.
|
- description: Service account name used for pods and rolebindings to form a cluster in the project.
|
||||||
displayName: Service Account
|
displayName: Service Account
|
||||||
name: SERVICE_ACCOUNT
|
name: SERVICE_ACCOUNT
|
||||||
value: patroniocp
|
value: patroniocp
|
||||||
|
|||||||
@@ -34,7 +34,7 @@ objects:
|
|||||||
labels:
|
labels:
|
||||||
application: ${APPLICATION_NAME}
|
application: ${APPLICATION_NAME}
|
||||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||||
name: ${PATRONI_MASTER_SERVICE_NAME}
|
name: ${PATRONI_PRIMARY_SERVICE_NAME}
|
||||||
spec:
|
spec:
|
||||||
ports:
|
ports:
|
||||||
- port: 5432
|
- port: 5432
|
||||||
@@ -43,7 +43,7 @@ objects:
|
|||||||
selector:
|
selector:
|
||||||
application: ${APPLICATION_NAME}
|
application: ${APPLICATION_NAME}
|
||||||
cluster-name: ${PATRONI_CLUSTER_NAME}
|
cluster-name: ${PATRONI_CLUSTER_NAME}
|
||||||
role: master
|
role: primary
|
||||||
sessionAffinity: None
|
sessionAffinity: None
|
||||||
type: ClusterIP
|
type: ClusterIP
|
||||||
status:
|
status:
|
||||||
@@ -107,7 +107,7 @@ objects:
|
|||||||
initContainers:
|
initContainers:
|
||||||
- command:
|
- command:
|
||||||
- sh
|
- sh
|
||||||
- -c
|
- -c
|
||||||
- "mkdir -p /home/postgres/pgdata/pgroot/data && chmod 0700 /home/postgres/pgdata/pgroot/data"
|
- "mkdir -p /home/postgres/pgdata/pgroot/data && chmod 0700 /home/postgres/pgdata/pgroot/data"
|
||||||
image: docker-registry.default.svc:5000/${NAMESPACE}/patroni:latest
|
image: docker-registry.default.svc:5000/${NAMESPACE}/patroni:latest
|
||||||
imagePullPolicy: IfNotPresent
|
imagePullPolicy: IfNotPresent
|
||||||
@@ -196,7 +196,7 @@ objects:
|
|||||||
terminationGracePeriodSeconds: 0
|
terminationGracePeriodSeconds: 0
|
||||||
volumes:
|
volumes:
|
||||||
- name: ${APPLICATION_NAME}
|
- name: ${APPLICATION_NAME}
|
||||||
persistentVolumeClaim:
|
persistentVolumeClaim:
|
||||||
claimName: ${APPLICATION_NAME}
|
claimName: ${APPLICATION_NAME}
|
||||||
volumeClaimTemplates:
|
volumeClaimTemplates:
|
||||||
- metadata:
|
- metadata:
|
||||||
@@ -313,12 +313,12 @@ parameters:
|
|||||||
displayName: Cluster Name
|
displayName: Cluster Name
|
||||||
name: PATRONI_CLUSTER_NAME
|
name: PATRONI_CLUSTER_NAME
|
||||||
value: patroni-persistent
|
value: patroni-persistent
|
||||||
- description: The name of the OpenShift Service exposed for the patroni-persistent-master container.
|
- description: The name of the OpenShift Service exposed for the patroni-persistent-primary container.
|
||||||
displayName: Master service name.
|
displayName: Primary service name.
|
||||||
name: PATRONI_MASTER_SERVICE_NAME
|
name: PATRONI_PRIMARY_SERVICE_NAME
|
||||||
value: patroni-persistent-master
|
value: patroni-persistent-primary
|
||||||
- description: The name of the OpenShift Service exposed for the patroni-persistent-replica containers.
|
- description: The name of the OpenShift Service exposed for the patroni-persistent-replica containers.
|
||||||
displayName: Replica service name.
|
displayName: Replica service name.
|
||||||
name: PATRONI_REPLICA_SERVICE_NAME
|
name: PATRONI_REPLICA_SERVICE_NAME
|
||||||
value: patroni-persistent-replica
|
value: patroni-persistent-replica
|
||||||
- description: Maximum amount of memory the container can use.
|
- description: Maximum amount of memory the container can use.
|
||||||
@@ -334,7 +334,7 @@ parameters:
|
|||||||
name: PATRONI_SUPERUSER_USERNAME
|
name: PATRONI_SUPERUSER_USERNAME
|
||||||
value: postgres
|
value: postgres
|
||||||
- description: Password of the superuser account for initialization.
|
- description: Password of the superuser account for initialization.
|
||||||
displayName: Superuser Passsword
|
displayName: Superuser Password
|
||||||
name: PATRONI_SUPERUSER_PASSWORD
|
name: PATRONI_SUPERUSER_PASSWORD
|
||||||
value: postgres
|
value: postgres
|
||||||
- description: Username of the replication account for initialization.
|
- description: Username of the replication account for initialization.
|
||||||
@@ -342,14 +342,14 @@ parameters:
|
|||||||
name: PATRONI_REPLICATION_USERNAME
|
name: PATRONI_REPLICATION_USERNAME
|
||||||
value: postgres
|
value: postgres
|
||||||
- description: Password of the replication account for initialization.
|
- description: Password of the replication account for initialization.
|
||||||
displayName: Repication Passsword
|
displayName: Repication Password
|
||||||
name: PATRONI_REPLICATION_PASSWORD
|
name: PATRONI_REPLICATION_PASSWORD
|
||||||
value: postgres
|
value: postgres
|
||||||
- description: Service account name used for pods and rolebindings to form a cluster in the project.
|
- description: Service account name used for pods and rolebindings to form a cluster in the project.
|
||||||
displayName: Service Account
|
displayName: Service Account
|
||||||
name: SERVICE_ACCOUNT
|
name: SERVICE_ACCOUNT
|
||||||
value: patroni-persistent
|
value: patroni-persistent
|
||||||
- description: The size of the persistent volume to create.
|
- description: The size of the persistent volume to create.
|
||||||
displayName: Persistent Volume Size
|
displayName: Persistent Volume Size
|
||||||
name: PVC_SIZE
|
name: PVC_SIZE
|
||||||
value: 5Gi
|
value: 5Gi
|
||||||
|
|||||||
+2
-2
@@ -15,9 +15,9 @@ pipeline {
|
|||||||
script {
|
script {
|
||||||
openshift.withCluster() {
|
openshift.withCluster() {
|
||||||
openshift.withProject() {
|
openshift.withProject() {
|
||||||
def pgbench = openshift.newApp( "https://github.com/stewartshea/docker-pgbench/", "--name=pgbench", "-e PGPASSWORD=postgres", "-e PGUSER=postgres", "-e PGHOST=patroni-persistent-master", "-e PGDATABASE=postgres", "-e TEST_CLIENT_COUNT=20", "-e TEST_DURATION=120" )
|
def pgbench = openshift.newApp( "https://github.com/stewartshea/docker-pgbench/", "--name=pgbench", "-e PGPASSWORD=postgres", "-e PGUSER=postgres", "-e PGHOST=patroni-persistent-primary", "-e PGDATABASE=postgres", "-e TEST_CLIENT_COUNT=20", "-e TEST_DURATION=120" )
|
||||||
def pgbenchdc = openshift.selector( "dc", "pgbench" )
|
def pgbenchdc = openshift.selector( "dc", "pgbench" )
|
||||||
timeout(5) {
|
timeout(5) {
|
||||||
pgbenchdc.rollout().status()
|
pgbenchdc.rollout().status()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+8
-4
@@ -13,13 +13,17 @@ def hiddenimports():
|
|||||||
sys.path.pop(0)
|
sys.path.pop(0)
|
||||||
|
|
||||||
|
|
||||||
|
def resources():
|
||||||
|
import os
|
||||||
|
res_dir = 'patroni/postgresql/available_parameters/'
|
||||||
|
exts = set(f.split('.')[-1] for f in os.listdir(res_dir))
|
||||||
|
return [(res_dir + '*.' + e, res_dir) for e in exts if e.lower() in {'yml', 'yaml'}]
|
||||||
|
|
||||||
|
|
||||||
a = Analysis(['patroni/__main__.py'],
|
a = Analysis(['patroni/__main__.py'],
|
||||||
pathex=[],
|
pathex=[],
|
||||||
binaries=None,
|
binaries=None,
|
||||||
datas=[
|
datas=resources(),
|
||||||
('patroni/postgresql/available_parameters/*.yml', 'patroni/postgresql/available_parameters'),
|
|
||||||
('patroni/postgresql/available_parameters/*.yaml', 'patroni/postgresql/available_parameters'),
|
|
||||||
],
|
|
||||||
hiddenimports=hiddenimports(),
|
hiddenimports=hiddenimports(),
|
||||||
hookspath=[],
|
hookspath=[],
|
||||||
runtime_hooks=[],
|
runtime_hooks=[],
|
||||||
|
|||||||
+66
-34
@@ -12,12 +12,13 @@ import time
|
|||||||
from argparse import Namespace
|
from argparse import Namespace
|
||||||
from typing import Any, Dict, List, Optional, TYPE_CHECKING
|
from typing import Any, Dict, List, Optional, TYPE_CHECKING
|
||||||
|
|
||||||
from patroni import MIN_PSYCOPG2, MIN_PSYCOPG3, parse_version
|
from patroni import global_config, MIN_PSYCOPG2, MIN_PSYCOPG3, parse_version
|
||||||
from patroni.daemon import AbstractPatroniDaemon, abstract_main, get_base_arg_parser
|
from patroni.daemon import abstract_main, AbstractPatroniDaemon, get_base_arg_parser
|
||||||
from patroni.tags import Tags
|
from patroni.tags import Tags
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from .config import Config
|
from .config import Config
|
||||||
|
from .dcs import Cluster
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -63,10 +64,14 @@ class Patroni(AbstractPatroniDaemon, Tags):
|
|||||||
self.dcs = get_dcs(self.config)
|
self.dcs = get_dcs(self.config)
|
||||||
self.request = PatroniRequest(self.config, True)
|
self.request = PatroniRequest(self.config, True)
|
||||||
|
|
||||||
self.ensure_unique_name()
|
cluster = self.ensure_dcs_access()
|
||||||
|
self.ensure_unique_name(cluster)
|
||||||
|
|
||||||
self.watchdog = Watchdog(self.config)
|
self.watchdog = Watchdog(self.config)
|
||||||
self.load_dynamic_configuration()
|
self.apply_dynamic_configuration(cluster)
|
||||||
|
|
||||||
|
# Initialize global config
|
||||||
|
global_config.update(None, self.config.dynamic_configuration)
|
||||||
|
|
||||||
self.postgresql = Postgresql(self.config['postgresql'], self.dcs.mpp)
|
self.postgresql = Postgresql(self.config['postgresql'], self.dcs.mpp)
|
||||||
self.api = RestApiServer(self, self.config['restapi'])
|
self.api = RestApiServer(self, self.config['restapi'])
|
||||||
@@ -76,40 +81,49 @@ class Patroni(AbstractPatroniDaemon, Tags):
|
|||||||
self.next_run = time.time()
|
self.next_run = time.time()
|
||||||
self.scheduled_restart: Dict[str, Any] = {}
|
self.scheduled_restart: Dict[str, Any] = {}
|
||||||
|
|
||||||
def load_dynamic_configuration(self) -> None:
|
def ensure_dcs_access(self, sleep_time: int = 5) -> 'Cluster':
|
||||||
"""Load Patroni dynamic configuration.
|
"""Continuously attempt to retrieve cluster from DCS with delay.
|
||||||
|
|
||||||
Load dynamic configuration from the DCS, if `/config` key is available in the DCS, otherwise fall back to
|
:param sleep_time: seconds to wait between retry attempts after dcs connection raise :exc:`DCSError`.
|
||||||
|
|
||||||
|
:returns: a PostgreSQL or MPP implementation of :class:`Cluster`.
|
||||||
|
"""
|
||||||
|
from patroni.exceptions import DCSError
|
||||||
|
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
return self.dcs.get_cluster()
|
||||||
|
except DCSError:
|
||||||
|
logger.warning('Can not get cluster from dcs')
|
||||||
|
time.sleep(sleep_time)
|
||||||
|
|
||||||
|
def apply_dynamic_configuration(self, cluster: 'Cluster') -> None:
|
||||||
|
"""Apply Patroni dynamic configuration.
|
||||||
|
|
||||||
|
Apply dynamic configuration from the DCS, if `/config` key is available in the DCS, otherwise fall back to
|
||||||
``bootstrap.dcs`` section from the configuration file.
|
``bootstrap.dcs`` section from the configuration file.
|
||||||
|
|
||||||
If the DCS connection fails returning the exception :class:`~patroni.exceptions.DCSError` an attempt will be
|
|
||||||
remade every 5 seconds.
|
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
This method is called only once, at the time when Patroni is started.
|
This method is called only once, at the time when Patroni is started.
|
||||||
"""
|
|
||||||
from patroni.exceptions import DCSError
|
|
||||||
while True:
|
|
||||||
try:
|
|
||||||
cluster = self.dcs.get_cluster()
|
|
||||||
if cluster and cluster.config and cluster.config.data:
|
|
||||||
if self.config.set_dynamic_configuration(cluster.config):
|
|
||||||
self.dcs.reload_config(self.config)
|
|
||||||
self.watchdog.reload_config(self.config)
|
|
||||||
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
|
||||||
if self.config.set_dynamic_configuration(self.config['bootstrap']['dcs']):
|
|
||||||
self.dcs.reload_config(self.config)
|
|
||||||
self.watchdog.reload_config(self.config)
|
|
||||||
break
|
|
||||||
except DCSError:
|
|
||||||
logger.warning('Can not get cluster from dcs')
|
|
||||||
time.sleep(5)
|
|
||||||
|
|
||||||
def ensure_unique_name(self) -> None:
|
:param cluster: a PostgreSQL or MPP implementation of :class:`Cluster`.
|
||||||
"""A helper method to prevent splitbrain from operator naming error."""
|
"""
|
||||||
|
if cluster and cluster.config and cluster.config.data:
|
||||||
|
if self.config.set_dynamic_configuration(cluster.config):
|
||||||
|
self.dcs.reload_config(self.config)
|
||||||
|
self.watchdog.reload_config(self.config)
|
||||||
|
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
||||||
|
if self.config.set_dynamic_configuration(self.config['bootstrap']['dcs']):
|
||||||
|
self.dcs.reload_config(self.config)
|
||||||
|
self.watchdog.reload_config(self.config)
|
||||||
|
|
||||||
|
def ensure_unique_name(self, cluster: 'Cluster') -> None:
|
||||||
|
"""A helper method to prevent splitbrain from operator naming error.
|
||||||
|
|
||||||
|
:param cluster: a PostgreSQL or MPP implementation of :class:`Cluster`.
|
||||||
|
"""
|
||||||
from patroni.dcs import Member
|
from patroni.dcs import Member
|
||||||
|
|
||||||
cluster = self.dcs.get_cluster()
|
|
||||||
if not cluster:
|
if not cluster:
|
||||||
return
|
return
|
||||||
member = cluster.get_member(self.config['name'], False)
|
member = cluster.get_member(self.config['name'], False)
|
||||||
@@ -198,13 +212,15 @@ class Patroni(AbstractPatroniDaemon, Tags):
|
|||||||
the change and cache the new dynamic configuration values in ``patroni.dynamic.json`` file under Postgres data
|
the change and cache the new dynamic configuration values in ``patroni.dynamic.json`` file under Postgres data
|
||||||
directory.
|
directory.
|
||||||
"""
|
"""
|
||||||
|
from patroni.postgresql.misc import PostgresqlRole
|
||||||
|
|
||||||
logger.info(self.ha.run_cycle())
|
logger.info(self.ha.run_cycle())
|
||||||
|
|
||||||
if self.dcs.cluster and self.dcs.cluster.config and self.dcs.cluster.config.data \
|
if self.dcs.cluster and self.dcs.cluster.config and self.dcs.cluster.config.data \
|
||||||
and self.config.set_dynamic_configuration(self.dcs.cluster.config):
|
and self.config.set_dynamic_configuration(self.dcs.cluster.config):
|
||||||
self.reload_config()
|
self.reload_config()
|
||||||
|
|
||||||
if self.postgresql.role != 'uninitialized':
|
if self.postgresql.role != PostgresqlRole.UNINITIALIZED:
|
||||||
self.config.save_cache()
|
self.config.save_cache()
|
||||||
|
|
||||||
self.schedule_next_run()
|
self.schedule_next_run()
|
||||||
@@ -241,6 +257,10 @@ def process_arguments() -> Namespace:
|
|||||||
* ``--validate-config`` -- used to validate the Patroni configuration file
|
* ``--validate-config`` -- used to validate the Patroni configuration file
|
||||||
* ``--generate-config`` -- used to generate Patroni configuration from a running PostgreSQL instance
|
* ``--generate-config`` -- used to generate Patroni configuration from a running PostgreSQL instance
|
||||||
* ``--generate-sample-config`` -- used to generate a sample Patroni configuration
|
* ``--generate-sample-config`` -- used to generate a sample Patroni configuration
|
||||||
|
* ``--ignore-listen-port`` | ``-i`` -- used to ignore ``listen`` ports already in use.
|
||||||
|
Can be used only with ``--validate-config``
|
||||||
|
* ``--print`` | ``-p`` -- used to print out local configuration (incl. environment configuration overrides).
|
||||||
|
Can be used only with ``--validate-config``
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
If running with ``--generate-config``, ``--generate-sample-config`` or ``--validate-flag`` will exit
|
If running with ``--generate-config``, ``--generate-sample-config`` or ``--validate-flag`` will exit
|
||||||
@@ -259,6 +279,12 @@ def process_arguments() -> Namespace:
|
|||||||
help='Generate a Patroni yaml configuration file for a running instance')
|
help='Generate a Patroni yaml configuration file for a running instance')
|
||||||
parser.add_argument('--dsn', help='Optional DSN string of the instance to be used as a source \
|
parser.add_argument('--dsn', help='Optional DSN string of the instance to be used as a source \
|
||||||
for config generation. Superuser connection is required.')
|
for config generation. Superuser connection is required.')
|
||||||
|
parser.add_argument('--ignore-listen-port', '-i', action='store_true',
|
||||||
|
help='Ignore `listen` ports already in use.\
|
||||||
|
Can only be used with --validate-config')
|
||||||
|
parser.add_argument('--print', '-p', action='store_true',
|
||||||
|
help='Print out local configuration (incl. environment configuration overrides).\
|
||||||
|
Can only be used with --validate-config')
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
if args.generate_sample_config:
|
if args.generate_sample_config:
|
||||||
@@ -268,15 +294,21 @@ def process_arguments() -> Namespace:
|
|||||||
generate_config(args.configfile, False, args.dsn)
|
generate_config(args.configfile, False, args.dsn)
|
||||||
sys.exit(0)
|
sys.exit(0)
|
||||||
elif args.validate_config:
|
elif args.validate_config:
|
||||||
from patroni.validator import schema
|
|
||||||
from patroni.config import Config, ConfigParseError
|
from patroni.config import Config, ConfigParseError
|
||||||
|
from patroni.validator import populate_validate_params, schema
|
||||||
|
|
||||||
|
populate_validate_params(ignore_listen_port=args.ignore_listen_port)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
Config(args.configfile, validator=schema)
|
config = Config(args.configfile, validator=schema)
|
||||||
sys.exit()
|
|
||||||
except ConfigParseError as e:
|
except ConfigParseError as e:
|
||||||
sys.exit(e.value)
|
sys.exit(e.value)
|
||||||
|
|
||||||
|
if args.print:
|
||||||
|
import yaml
|
||||||
|
yaml.safe_dump(config.local_configuration, sys.stdout, default_flow_style=False, allow_unicode=True)
|
||||||
|
sys.exit()
|
||||||
|
|
||||||
return args
|
return args
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+135
-84
@@ -7,43 +7,43 @@ utilises the API to perform these functions.
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import base64
|
import base64
|
||||||
|
import datetime
|
||||||
import hmac
|
import hmac
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import time
|
|
||||||
import traceback
|
|
||||||
import dateutil.parser
|
|
||||||
import datetime
|
|
||||||
import os
|
import os
|
||||||
import socket
|
import socket
|
||||||
import sys
|
import sys
|
||||||
|
import time
|
||||||
|
import traceback
|
||||||
|
|
||||||
from http.server import BaseHTTPRequestHandler, HTTPServer
|
from http.server import BaseHTTPRequestHandler, HTTPServer
|
||||||
from ipaddress import ip_address, ip_network, IPv4Network, IPv6Network
|
from ipaddress import ip_address, ip_network, IPv4Network, IPv6Network
|
||||||
from socketserver import ThreadingMixIn
|
from socketserver import ThreadingMixIn
|
||||||
from threading import Thread
|
from threading import Thread
|
||||||
from urllib.parse import urlparse, parse_qs
|
from typing import Any, Callable, cast, Dict, Iterator, List, Optional, Tuple, TYPE_CHECKING, Union
|
||||||
|
from urllib.parse import parse_qs, urlparse
|
||||||
|
|
||||||
from typing import Any, Callable, Dict, Iterator, List, Optional, Tuple, TYPE_CHECKING, Union
|
import dateutil.parser
|
||||||
|
|
||||||
from . import global_config, psycopg
|
from . import global_config, psycopg
|
||||||
from .__main__ import Patroni
|
from .__main__ import Patroni
|
||||||
from .dcs import Cluster
|
from .dcs import Cluster
|
||||||
from .exceptions import PostgresConnectionException, PostgresException
|
from .exceptions import PostgresConnectionException, PostgresException
|
||||||
from .postgresql.misc import postgres_version_to_int
|
from .postgresql.misc import postgres_version_to_int, PostgresqlRole, PostgresqlState
|
||||||
from .utils import deep_compare, enable_keepalive, parse_bool, patch_config, Retry, \
|
from .utils import cluster_as_json, deep_compare, enable_keepalive, parse_bool, \
|
||||||
RetryFailedError, parse_int, split_host_port, tzutc, uri, cluster_as_json
|
parse_int, patch_config, Retry, RetryFailedError, split_host_port, tzutc, uri
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def check_access(func: Callable[..., None]) -> Callable[..., None]:
|
def check_access(*args: Any, **kwargs: Any) -> Callable[..., Any]:
|
||||||
"""Check the source ip, authorization header, or client certificates.
|
"""Check the source ip, authorization header, or client certificates.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
The actual logic to check access is implemented through :func:`RestApiServer.check_access`.
|
The actual logic to check access is implemented through :func:`RestApiServer.check_access`.
|
||||||
|
|
||||||
:param func: function to be decorated.
|
Optionally it is possible to skip source ip check by specifying ``allowlist_check_members=False``.
|
||||||
|
|
||||||
:returns: a decorator that executes *func* only if :func:`RestApiServer.check_access` returns ``True``.
|
:returns: a decorator that executes *func* only if :func:`RestApiServer.check_access` returns ``True``.
|
||||||
|
|
||||||
@@ -60,19 +60,31 @@ def check_access(func: Callable[..., None]) -> Callable[..., None]:
|
|||||||
... @check_access
|
... @check_access
|
||||||
... def do_PUT_foo(self):
|
... def do_PUT_foo(self):
|
||||||
... print('In do_PUT_foo')
|
... print('In do_PUT_foo')
|
||||||
|
... @check_access(allowlist_check_members=False)
|
||||||
|
... def do_POST_bar(self):
|
||||||
|
... print('In do_POST_bar')
|
||||||
|
|
||||||
>>> f = Foo()
|
>>> f = Foo()
|
||||||
>>> f.do_PUT_foo()
|
>>> f.do_PUT_foo()
|
||||||
In FooServer: Foo
|
In FooServer: Foo
|
||||||
In do_PUT_foo
|
In do_PUT_foo
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
allowlist_check_members = kwargs.get('allowlist_check_members', True)
|
||||||
|
|
||||||
def wrapper(self: 'RestApiHandler', *args: Any, **kwargs: Any) -> None:
|
def inner_decorator(func: Callable[..., Any]) -> Callable[..., Any]:
|
||||||
if self.server.check_access(self):
|
def wrapper(self: 'RestApiHandler', *args: Any, **kwargs: Any) -> Any:
|
||||||
return func(self, *args, **kwargs)
|
if self.server.check_access(self, allowlist_check_members=allowlist_check_members):
|
||||||
|
return func(self, *args, **kwargs)
|
||||||
|
|
||||||
return wrapper
|
return wrapper
|
||||||
|
|
||||||
|
# A hacky way to have decorators that work with and without parameters.
|
||||||
|
if len(args) == 1 and callable(args[0]):
|
||||||
|
# The first parameter is a function, it means decorator is used as "@check_access"
|
||||||
|
return inner_decorator(args[0])
|
||||||
|
else:
|
||||||
|
# @check_access(allowlist_check_members=False) case
|
||||||
|
return inner_decorator
|
||||||
|
|
||||||
|
|
||||||
class RestApiHandler(BaseHTTPRequestHandler):
|
class RestApiHandler(BaseHTTPRequestHandler):
|
||||||
@@ -254,6 +266,14 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
|
|
||||||
* HTTP status ``200``: if up and running and without ``noloadbalance`` tag.
|
* HTTP status ``200``: if up and running and without ``noloadbalance`` tag.
|
||||||
|
|
||||||
|
* ``/quorum``:
|
||||||
|
|
||||||
|
* HTTP status ``200``: if up and running as a quorum synchronous standby.
|
||||||
|
|
||||||
|
* ``/read-only-quorum``:
|
||||||
|
|
||||||
|
* HTTP status ``200``: if up and running as a quorum synchronous standby or primary.
|
||||||
|
|
||||||
* ``/synchronous`` or ``/sync``:
|
* ``/synchronous`` or ``/sync``:
|
||||||
|
|
||||||
* HTTP status ``200``: if up and running as a synchronous standby.
|
* HTTP status ``200``: if up and running as a synchronous standby.
|
||||||
@@ -290,12 +310,13 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
"""
|
"""
|
||||||
path = '/primary' if self.path == '/' else self.path
|
path = '/primary' if self.path == '/' else self.path
|
||||||
response = self.get_postgresql_status()
|
response = self.get_postgresql_status()
|
||||||
|
latest_end_lsn = response.pop('latest_end_lsn', 0)
|
||||||
|
|
||||||
patroni = self.server.patroni
|
patroni = self.server.patroni
|
||||||
cluster = patroni.dcs.cluster
|
cluster = patroni.dcs.cluster
|
||||||
config = global_config.from_cluster(cluster)
|
config = global_config.from_cluster(cluster)
|
||||||
|
|
||||||
leader_optime = cluster and cluster.last_lsn or 0
|
leader_optime = max(cluster and cluster.status.last_lsn or 0, latest_end_lsn)
|
||||||
replayed_location = response.get('xlog', {}).get('replayed_location', 0)
|
replayed_location = response.get('xlog', {}).get('replayed_location', 0)
|
||||||
max_replica_lag = parse_int(self.path_query.get('lag', [sys.maxsize])[0], 'B')
|
max_replica_lag = parse_int(self.path_query.get('lag', [sys.maxsize])[0], 'B')
|
||||||
if max_replica_lag is None:
|
if max_replica_lag is None:
|
||||||
@@ -303,17 +324,19 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
is_lagging = leader_optime and leader_optime > replayed_location + max_replica_lag
|
is_lagging = leader_optime and leader_optime > replayed_location + max_replica_lag
|
||||||
|
|
||||||
replica_status_code = 200 if not patroni.noloadbalance and not is_lagging and \
|
replica_status_code = 200 if not patroni.noloadbalance and not is_lagging and \
|
||||||
response.get('role') == 'replica' and response.get('state') == 'running' else 503
|
response.get('role') == PostgresqlRole.REPLICA and response.get('state') == PostgresqlState.RUNNING else 503
|
||||||
|
|
||||||
if not cluster and response.get('pause'):
|
if not cluster and response.get('pause'):
|
||||||
leader_status_code = 200 if response.get('role') in ('master', 'primary', 'standby_leader') else 503
|
leader_status_code = 200 if response.get('role') in (PostgresqlRole.PRIMARY,
|
||||||
primary_status_code = 200 if response.get('role') in ('master', 'primary') else 503
|
PostgresqlRole.STANDBY_LEADER) else 503
|
||||||
standby_leader_status_code = 200 if response.get('role') == 'standby_leader' else 503
|
primary_status_code = 200 if response.get('role') == PostgresqlRole.PRIMARY else 503
|
||||||
|
standby_leader_status_code = 200 if response.get('role') == PostgresqlRole.STANDBY_LEADER else 503
|
||||||
elif patroni.ha.is_leader():
|
elif patroni.ha.is_leader():
|
||||||
leader_status_code = 200
|
leader_status_code = 200
|
||||||
if config.is_standby_cluster:
|
if config.is_standby_cluster:
|
||||||
primary_status_code = replica_status_code = 503
|
primary_status_code = replica_status_code = 503
|
||||||
standby_leader_status_code = 200 if response.get('role') in ('replica', 'standby_leader') else 503
|
standby_leader_status_code =\
|
||||||
|
200 if response.get('role') in (PostgresqlRole.REPLICA, PostgresqlRole.STANDBY_LEADER) else 503
|
||||||
else:
|
else:
|
||||||
primary_status_code = 200
|
primary_status_code = 200
|
||||||
standby_leader_status_code = 503
|
standby_leader_status_code = 503
|
||||||
@@ -334,16 +357,24 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
ignore_tags = True
|
ignore_tags = True
|
||||||
elif 'replica' in path:
|
elif 'replica' in path:
|
||||||
status_code = replica_status_code
|
status_code = replica_status_code
|
||||||
elif 'read-only' in path and 'sync' not in path:
|
elif 'read-only' in path and 'sync' not in path and 'quorum' not in path:
|
||||||
status_code = 200 if 200 in (primary_status_code, standby_leader_status_code) else replica_status_code
|
status_code = 200 if 200 in (primary_status_code, standby_leader_status_code) else replica_status_code
|
||||||
elif 'health' in path:
|
elif 'health' in path:
|
||||||
status_code = 200 if response.get('state') == 'running' else 503
|
status_code = 200 if response.get('state') == PostgresqlState.RUNNING else 503
|
||||||
elif cluster: # dcs is available
|
elif cluster: # dcs is available
|
||||||
|
is_quorum = response.get('quorum_standby')
|
||||||
is_synchronous = response.get('sync_standby')
|
is_synchronous = response.get('sync_standby')
|
||||||
if path in ('/sync', '/synchronous') and is_synchronous:
|
if path in ('/sync', '/synchronous') and is_synchronous:
|
||||||
status_code = replica_status_code
|
status_code = replica_status_code
|
||||||
elif path in ('/async', '/asynchronous') and not is_synchronous:
|
elif path == '/quorum' and is_quorum:
|
||||||
status_code = replica_status_code
|
status_code = replica_status_code
|
||||||
|
elif path in ('/async', '/asynchronous') and not is_synchronous and not is_quorum:
|
||||||
|
status_code = replica_status_code
|
||||||
|
elif path == '/read-only-quorum':
|
||||||
|
if 200 in (primary_status_code, standby_leader_status_code):
|
||||||
|
status_code = 200
|
||||||
|
elif is_quorum:
|
||||||
|
status_code = replica_status_code
|
||||||
elif path in ('/read-only-sync', '/read-only-synchronous'):
|
elif path in ('/read-only-sync', '/read-only-synchronous'):
|
||||||
if 200 in (primary_status_code, standby_leader_status_code):
|
if 200 in (primary_status_code, standby_leader_status_code):
|
||||||
status_code = 200
|
status_code = 200
|
||||||
@@ -407,7 +438,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
patroni: Patroni = self.server.patroni
|
patroni: Patroni = self.server.patroni
|
||||||
is_primary = patroni.postgresql.role in ('master', 'primary') and patroni.postgresql.is_running()
|
is_primary = patroni.postgresql.role == PostgresqlRole.PRIMARY and patroni.postgresql.is_running()
|
||||||
# We can tolerate Patroni problems longer on the replica.
|
# We can tolerate Patroni problems longer on the replica.
|
||||||
# On the primary the liveness probe most likely will start failing only after the leader key expired.
|
# On the primary the liveness probe most likely will start failing only after the leader key expired.
|
||||||
# It should not be a big problem because replicas will see that the primary is still alive via REST API call.
|
# It should not be a big problem because replicas will see that the primary is still alive via REST API call.
|
||||||
@@ -433,7 +464,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
patroni = self.server.patroni
|
patroni = self.server.patroni
|
||||||
if patroni.ha.is_leader():
|
if patroni.ha.is_leader():
|
||||||
status_code = 200
|
status_code = 200
|
||||||
elif patroni.postgresql.state == 'running':
|
elif patroni.postgresql.state == PostgresqlState.RUNNING:
|
||||||
status_code = 200 if patroni.dcs.cluster else 503
|
status_code = 200 if patroni.dcs.cluster else 503
|
||||||
else:
|
else:
|
||||||
status_code = 503
|
status_code = 503
|
||||||
@@ -446,6 +477,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
Postgres.
|
Postgres.
|
||||||
"""
|
"""
|
||||||
response = self.get_postgresql_status(True)
|
response = self.get_postgresql_status(True)
|
||||||
|
response.pop('latest_end_lsn', None)
|
||||||
self._write_status_response(200, response)
|
self._write_status_response(200, response)
|
||||||
|
|
||||||
def do_GET_cluster(self) -> None:
|
def do_GET_cluster(self) -> None:
|
||||||
@@ -504,12 +536,12 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
* ``patroni_version``: Patroni version without periods, e.g. ``030002`` for Patroni ``3.0.2``;
|
* ``patroni_version``: Patroni version without periods, e.g. ``030002`` for Patroni ``3.0.2``;
|
||||||
* ``patroni_postgres_running``: ``1`` if PostgreSQL is running, else ``0``;
|
* ``patroni_postgres_running``: ``1`` if PostgreSQL is running, else ``0``;
|
||||||
* ``patroni_postmaster_start_time``: epoch timestamp since Postmaster was started;
|
* ``patroni_postmaster_start_time``: epoch timestamp since Postmaster was started;
|
||||||
* ``patroni_master``: ``1`` if this node holds the leader lock, else ``0``;
|
* ``patroni_primary``: ``1`` if this node holds the leader lock, else ``0``;
|
||||||
* ``patroni_primary``: same as ``patroni_master``;
|
|
||||||
* ``patroni_xlog_location``: ``pg_wal_lsn_diff(pg_current_wal_flush_lsn(), '0/0')`` if leader, else ``0``;
|
* ``patroni_xlog_location``: ``pg_wal_lsn_diff(pg_current_wal_flush_lsn(), '0/0')`` if leader, else ``0``;
|
||||||
* ``patroni_standby_leader``: ``1`` if standby leader node, else ``0``;
|
* ``patroni_standby_leader``: ``1`` if standby leader node, else ``0``;
|
||||||
* ``patroni_replica``: ``1`` if a replica, else ``0``;
|
* ``patroni_replica``: ``1`` if a replica, else ``0``;
|
||||||
* ``patroni_sync_standby``: ``1`` if a sync replica, else ``0``;
|
* ``patroni_sync_standby``: ``1`` if a sync replica, else ``0``;
|
||||||
|
* ``patroni_quorum_standby``: ``1`` if a quorum sync replica, else ``0``;
|
||||||
* ``patroni_xlog_received_location``: ``pg_wal_lsn_diff(pg_last_wal_receive_lsn(), '0/0')``;
|
* ``patroni_xlog_received_location``: ``pg_wal_lsn_diff(pg_last_wal_receive_lsn(), '0/0')``;
|
||||||
* ``patroni_xlog_replayed_location``: ``pg_wal_lsn_diff(pg_last_wal_replay_lsn(), '0/0)``;
|
* ``patroni_xlog_replayed_location``: ``pg_wal_lsn_diff(pg_last_wal_replay_lsn(), '0/0)``;
|
||||||
* ``patroni_xlog_replayed_timestamp``: ``pg_last_xact_replay_timestamp``;
|
* ``patroni_xlog_replayed_timestamp``: ``pg_last_xact_replay_timestamp``;
|
||||||
@@ -543,7 +575,8 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
|
|
||||||
metrics.append("# HELP patroni_postgres_running Value is 1 if Postgres is running, 0 otherwise.")
|
metrics.append("# HELP patroni_postgres_running Value is 1 if Postgres is running, 0 otherwise.")
|
||||||
metrics.append("# TYPE patroni_postgres_running gauge")
|
metrics.append("# TYPE patroni_postgres_running gauge")
|
||||||
metrics.append("patroni_postgres_running{0} {1}".format(labels, int(postgres['state'] == 'running')))
|
metrics.append("patroni_postgres_running{0} {1}".format(
|
||||||
|
labels, int(postgres['state'] == PostgresqlState.RUNNING)))
|
||||||
|
|
||||||
metrics.append("# HELP patroni_postmaster_start_time Epoch seconds since Postgres started.")
|
metrics.append("# HELP patroni_postmaster_start_time Epoch seconds since Postgres started.")
|
||||||
metrics.append("# TYPE patroni_postmaster_start_time gauge")
|
metrics.append("# TYPE patroni_postmaster_start_time gauge")
|
||||||
@@ -551,13 +584,9 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
postmaster_start_time = (postmaster_start_time - epoch).total_seconds() if postmaster_start_time else 0
|
postmaster_start_time = (postmaster_start_time - epoch).total_seconds() if postmaster_start_time else 0
|
||||||
metrics.append("patroni_postmaster_start_time{0} {1}".format(labels, postmaster_start_time))
|
metrics.append("patroni_postmaster_start_time{0} {1}".format(labels, postmaster_start_time))
|
||||||
|
|
||||||
metrics.append("# HELP patroni_master Value is 1 if this node is the leader, 0 otherwise.")
|
|
||||||
metrics.append("# TYPE patroni_master gauge")
|
|
||||||
metrics.append("patroni_master{0} {1}".format(labels, int(postgres['role'] in ('master', 'primary'))))
|
|
||||||
|
|
||||||
metrics.append("# HELP patroni_primary Value is 1 if this node is the leader, 0 otherwise.")
|
metrics.append("# HELP patroni_primary Value is 1 if this node is the leader, 0 otherwise.")
|
||||||
metrics.append("# TYPE patroni_primary gauge")
|
metrics.append("# TYPE patroni_primary gauge")
|
||||||
metrics.append("patroni_primary{0} {1}".format(labels, int(postgres['role'] in ('master', 'primary'))))
|
metrics.append("patroni_primary{0} {1}".format(labels, int(postgres['role'] == PostgresqlRole.PRIMARY)))
|
||||||
|
|
||||||
metrics.append("# HELP patroni_xlog_location Current location of the Postgres"
|
metrics.append("# HELP patroni_xlog_location Current location of the Postgres"
|
||||||
" transaction log, 0 if this node is not the leader.")
|
" transaction log, 0 if this node is not the leader.")
|
||||||
@@ -566,16 +595,21 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
|
|
||||||
metrics.append("# HELP patroni_standby_leader Value is 1 if this node is the standby_leader, 0 otherwise.")
|
metrics.append("# HELP patroni_standby_leader Value is 1 if this node is the standby_leader, 0 otherwise.")
|
||||||
metrics.append("# TYPE patroni_standby_leader gauge")
|
metrics.append("# TYPE patroni_standby_leader gauge")
|
||||||
metrics.append("patroni_standby_leader{0} {1}".format(labels, int(postgres['role'] == 'standby_leader')))
|
metrics.append("patroni_standby_leader{0} {1}".format(labels,
|
||||||
|
int(postgres['role'] == PostgresqlRole.STANDBY_LEADER)))
|
||||||
|
|
||||||
metrics.append("# HELP patroni_replica Value is 1 if this node is a replica, 0 otherwise.")
|
metrics.append("# HELP patroni_replica Value is 1 if this node is a replica, 0 otherwise.")
|
||||||
metrics.append("# TYPE patroni_replica gauge")
|
metrics.append("# TYPE patroni_replica gauge")
|
||||||
metrics.append("patroni_replica{0} {1}".format(labels, int(postgres['role'] == 'replica')))
|
metrics.append("patroni_replica{0} {1}".format(labels, int(postgres['role'] == PostgresqlRole.REPLICA)))
|
||||||
|
|
||||||
metrics.append("# HELP patroni_sync_standby Value is 1 if this node is a sync standby replica, 0 otherwise.")
|
metrics.append("# HELP patroni_sync_standby Value is 1 if this node is a sync standby, 0 otherwise.")
|
||||||
metrics.append("# TYPE patroni_sync_standby gauge")
|
metrics.append("# TYPE patroni_sync_standby gauge")
|
||||||
metrics.append("patroni_sync_standby{0} {1}".format(labels, int(postgres.get('sync_standby', False))))
|
metrics.append("patroni_sync_standby{0} {1}".format(labels, int(postgres.get('sync_standby', False))))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_quorum_standby Value is 1 if this node is a quorum standby, 0 otherwise.")
|
||||||
|
metrics.append("# TYPE patroni_quorum_standby gauge")
|
||||||
|
metrics.append("patroni_quorum_standby{0} {1}".format(labels, int(postgres.get('quorum_standby', False))))
|
||||||
|
|
||||||
metrics.append("# HELP patroni_xlog_received_location Current location of the received"
|
metrics.append("# HELP patroni_xlog_received_location Current location of the received"
|
||||||
" Postgres transaction log, 0 if this node is not a replica.")
|
" Postgres transaction log, 0 if this node is not a replica.")
|
||||||
metrics.append("# TYPE patroni_xlog_received_location counter")
|
metrics.append("# TYPE patroni_xlog_received_location counter")
|
||||||
@@ -627,7 +661,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
|
|
||||||
metrics.append("# HELP patroni_postgres_timeline Postgres timeline of this node (if running), 0 otherwise.")
|
metrics.append("# HELP patroni_postgres_timeline Postgres timeline of this node (if running), 0 otherwise.")
|
||||||
metrics.append("# TYPE patroni_postgres_timeline counter")
|
metrics.append("# TYPE patroni_postgres_timeline counter")
|
||||||
metrics.append("patroni_postgres_timeline{0} {1}".format(labels, postgres.get('timeline', 0)))
|
metrics.append("patroni_postgres_timeline{0} {1}".format(labels, postgres.get('timeline') or 0))
|
||||||
|
|
||||||
metrics.append("# HELP patroni_dcs_last_seen Epoch timestamp when DCS was last contacted successfully"
|
metrics.append("# HELP patroni_dcs_last_seen Epoch timestamp when DCS was last contacted successfully"
|
||||||
" by Patroni.")
|
" by Patroni.")
|
||||||
@@ -670,9 +704,9 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
content_length = int(self.headers.get('content-length') or 0)
|
content_length = int(self.headers.get('content-length') or 0)
|
||||||
if content_length == 0 and body_is_optional:
|
if content_length == 0 and body_is_optional:
|
||||||
return {}
|
return {}
|
||||||
request: Union[Dict[str, Any], Any] = json.loads(self.rfile.read(content_length).decode('utf-8'))
|
request = json.loads(self.rfile.read(content_length).decode('utf-8'))
|
||||||
if isinstance(request, dict) and (request or body_is_optional):
|
if isinstance(request, dict) and (request or body_is_optional):
|
||||||
return request
|
return cast(Dict[str, Any], request)
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('Bad request')
|
logger.exception('Bad request')
|
||||||
self.send_error(400)
|
self.send_error(400)
|
||||||
@@ -747,12 +781,12 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
else:
|
else:
|
||||||
self.send_error(502)
|
self.send_error(502)
|
||||||
|
|
||||||
@check_access
|
@check_access(allowlist_check_members=False)
|
||||||
def do_POST_failsafe(self) -> None:
|
def do_POST_failsafe(self) -> None:
|
||||||
"""Handle a ``POST`` request to ``/failsafe`` path.
|
"""Handle a ``POST`` request to ``/failsafe`` path.
|
||||||
|
|
||||||
Writes a response with HTTP status ``200`` if this node is a Standby, or with HTTP status ``500`` if this is
|
Writes a response with HTTP status ``200`` if this node is a Standby, or with HTTP status ``500`` if this is
|
||||||
the primary.
|
the primary. In addition to that it returns absolute value of received/replayed LSN in the ``lsn`` header.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
If ``failsafe_mode`` is not enabled, then write a response with HTTP status ``502``.
|
If ``failsafe_mode`` is not enabled, then write a response with HTTP status ``502``.
|
||||||
@@ -760,9 +794,11 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
if self.server.patroni.ha.is_failsafe_mode():
|
if self.server.patroni.ha.is_failsafe_mode():
|
||||||
request = self._read_json_content()
|
request = self._read_json_content()
|
||||||
if request:
|
if request:
|
||||||
message = self.server.patroni.ha.update_failsafe(request) or 'Accepted'
|
ret = self.server.patroni.ha.update_failsafe(request)
|
||||||
|
headers = {'lsn': str(ret)} if isinstance(ret, int) else {}
|
||||||
|
message = ret if isinstance(ret, str) else 'Accepted'
|
||||||
code = 200 if message == 'Accepted' else 500
|
code = 200 if message == 'Accepted' else 500
|
||||||
self.write_response(code, message)
|
self.write_response(code, message, headers=headers)
|
||||||
else:
|
else:
|
||||||
self.send_error(502)
|
self.send_error(502)
|
||||||
|
|
||||||
@@ -828,7 +864,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
* ``schedule``: timestamp at which the restart should occur;
|
* ``schedule``: timestamp at which the restart should occur;
|
||||||
* ``role``: restart only nodes which role is ``role``. Can be either:
|
* ``role``: restart only nodes which role is ``role``. Can be either:
|
||||||
|
|
||||||
* ``primary`` (or ``master``); or
|
* ``primary`; or
|
||||||
* ``replica``.
|
* ``replica``.
|
||||||
|
|
||||||
* ``postgres_version``: restart only nodes which PostgreSQL version is less than ``postgres_version``, e.g.
|
* ``postgres_version``: restart only nodes which PostgreSQL version is less than ``postgres_version``, e.g.
|
||||||
@@ -857,7 +893,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
If it's not able to parse the request body, then the request is silently discarded.
|
If it's not able to parse the request body, then the request is silently discarded.
|
||||||
"""
|
"""
|
||||||
status_code = 500
|
status_code = 500
|
||||||
data = 'restart failed'
|
data = PostgresqlState.RESTART_FAILED
|
||||||
request = self._read_json_content(body_is_optional=True)
|
request = self._read_json_content(body_is_optional=True)
|
||||||
cluster = self.server.patroni.dcs.get_cluster()
|
cluster = self.server.patroni.dcs.get_cluster()
|
||||||
if request is None:
|
if request is None:
|
||||||
@@ -877,9 +913,9 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
status_code = _
|
status_code = _
|
||||||
break
|
break
|
||||||
elif k == 'role':
|
elif k == 'role':
|
||||||
if request[k] not in ('master', 'primary', 'replica'):
|
if request[k] not in (PostgresqlRole.PRIMARY, PostgresqlRole.STANDBY_LEADER, PostgresqlRole.REPLICA):
|
||||||
status_code = 400
|
status_code = 400
|
||||||
data = "PostgreSQL role should be either primary or replica"
|
data = "PostgreSQL role should be either primary, standby_leader, or replica"
|
||||||
break
|
break
|
||||||
elif k == 'postgres_version':
|
elif k == 'postgres_version':
|
||||||
try:
|
try:
|
||||||
@@ -1035,16 +1071,17 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
|
|
||||||
:returns: a string with the error message or ``None`` if good nodes are found.
|
:returns: a string with the error message or ``None`` if good nodes are found.
|
||||||
"""
|
"""
|
||||||
is_synchronous_mode = global_config.from_cluster(cluster).is_synchronous_mode
|
config = global_config.from_cluster(cluster)
|
||||||
if leader and (not cluster.leader or cluster.leader.name != leader):
|
if leader and (not cluster.leader or cluster.leader.name != leader):
|
||||||
return 'leader name does not match'
|
return 'leader name does not match'
|
||||||
if candidate:
|
if candidate:
|
||||||
if action == 'switchover' and is_synchronous_mode and not cluster.sync.matches(candidate):
|
if action == 'switchover' and config.is_synchronous_mode\
|
||||||
|
and not config.is_quorum_commit_mode and not cluster.sync.matches(candidate):
|
||||||
return 'candidate name does not match with sync_standby'
|
return 'candidate name does not match with sync_standby'
|
||||||
members = [m for m in cluster.members if m.name == candidate]
|
members = [m for m in cluster.members if m.name == candidate]
|
||||||
if not members:
|
if not members:
|
||||||
return 'candidate does not exists'
|
return 'candidate does not exists'
|
||||||
elif is_synchronous_mode:
|
elif config.is_synchronous_mode and not config.is_quorum_commit_mode:
|
||||||
members = [m for m in cluster.members if cluster.sync.matches(m.name)]
|
members = [m for m in cluster.members if cluster.sync.matches(m.name)]
|
||||||
if not members:
|
if not members:
|
||||||
return action + ' is not possible: can not find sync_standby'
|
return action + ' is not possible: can not find sync_standby'
|
||||||
@@ -1115,7 +1152,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
data = 'Switchover is possible only to a specific candidate in a paused state'
|
data = 'Switchover is possible only to a specific candidate in a paused state'
|
||||||
|
|
||||||
if action == 'failover' and leader:
|
if action == 'failover' and leader:
|
||||||
logger.warning('received failover request with leader specifed - performing switchover instead')
|
logger.warning('received failover request with leader specified - performing switchover instead')
|
||||||
action = 'switchover'
|
action = 'switchover'
|
||||||
|
|
||||||
if not data and leader == candidate:
|
if not data and leader == candidate:
|
||||||
@@ -1230,20 +1267,19 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
|
|
||||||
:returns: a dict with the status of Postgres/Patroni. The keys are:
|
:returns: a dict with the status of Postgres/Patroni. The keys are:
|
||||||
|
|
||||||
* ``state``: Postgres state among ``stopping``, ``stopped``, ``stop failed``, ``crashed``, ``running``,
|
* ``state``: one of :class:`~patroni.postgresql.misc.PostgresqlState` or ``unknown``;
|
||||||
``starting``, ``start failed``, ``restarting``, ``restart failed``, ``initializing new cluster``,
|
|
||||||
``initdb failed``, ``running custom bootstrap script``, ``custom bootstrap failed``,
|
|
||||||
``creating replica``, or ``unknown``;
|
|
||||||
* ``postmaster_start_time``: ``pg_postmaster_start_time()``;
|
* ``postmaster_start_time``: ``pg_postmaster_start_time()``;
|
||||||
* ``role``: ``replica`` or ``master`` based on ``pg_is_in_recovery()`` output;
|
* ``role``: :class:`~patroni.postgresql.misc.PostgresqlRole.REPLICA` or
|
||||||
|
:class:`~patroni.postgresql.misc.PostgresqlRole.PRIMARY` based on ``pg_is_in_recovery()`` output;
|
||||||
* ``server_version``: Postgres version without periods, e.g. ``150002`` for Postgres ``15.2``;
|
* ``server_version``: Postgres version without periods, e.g. ``150002`` for Postgres ``15.2``;
|
||||||
|
* ``latest_end_lsn``: latest_end_lsn value from ``pg_stat_get_wal_receiver()``, only on replica nodes;
|
||||||
* ``xlog``: dictionary. Its structure depends on ``role``:
|
* ``xlog``: dictionary. Its structure depends on ``role``:
|
||||||
|
|
||||||
* If ``master``:
|
* If :class:`~patroni.postgresql.misc.PostgresqlRole.PRIMARY`:
|
||||||
|
|
||||||
* ``location``: ``pg_current_wal_flush_lsn()``
|
* ``location``: ``pg_current_wal_flush_lsn()``
|
||||||
|
|
||||||
* If ``replica``:
|
* If :class:`~patroni.postgresql.misc.PostgresqlRole.REPLICA`:
|
||||||
|
|
||||||
* ``received_location``: ``pg_wal_lsn_diff(pg_last_wal_receive_lsn(), '0/0')``;
|
* ``received_location``: ``pg_wal_lsn_diff(pg_last_wal_receive_lsn(), '0/0')``;
|
||||||
* ``replayed_location``: ``pg_wal_lsn_diff(pg_last_wal_replay_lsn(), '0/0)``;
|
* ``replayed_location``: ``pg_wal_lsn_diff(pg_last_wal_replay_lsn(), '0/0)``;
|
||||||
@@ -1251,6 +1287,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
* ``paused``: ``pg_is_wal_replay_paused()``;
|
* ``paused``: ``pg_is_wal_replay_paused()``;
|
||||||
|
|
||||||
* ``sync_standby``: ``True`` if replication mode is synchronous and this is a sync standby;
|
* ``sync_standby``: ``True`` if replication mode is synchronous and this is a sync standby;
|
||||||
|
* ``quorum_standby``: ``True`` if replication mode is quorum and this is a quorum standby;
|
||||||
* ``timeline``: PostgreSQL primary node timeline;
|
* ``timeline``: PostgreSQL primary node timeline;
|
||||||
* ``replication``: :class:`list` of :class:`dict` entries, one for each replication connection. Each entry
|
* ``replication``: :class:`list` of :class:`dict` entries, one for each replication connection. Each entry
|
||||||
contains the following keys:
|
contains the following keys:
|
||||||
@@ -1273,27 +1310,31 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
config = global_config.from_cluster(cluster)
|
config = global_config.from_cluster(cluster)
|
||||||
try:
|
try:
|
||||||
|
|
||||||
if postgresql.state not in ('running', 'restarting', 'starting'):
|
if postgresql.state not in (PostgresqlState.RUNNING, PostgresqlState.RESTARTING,
|
||||||
|
PostgresqlState.STARTING):
|
||||||
raise RetryFailedError('')
|
raise RetryFailedError('')
|
||||||
replication_state = ('(pg_catalog.pg_stat_get_wal_receiver()).status'
|
replication_state = ("pg_catalog.pg_{0}_{1}_diff(wr.latest_end_lsn, '0/0')::bigint, wr.status"
|
||||||
if postgresql.major_version >= 90600 else 'NULL') + ", " +\
|
if postgresql.major_version >= 90600 else "NULL, NULL") + ", " +\
|
||||||
("pg_catalog.current_setting('restore_command')" if postgresql.major_version >= 120000 else "NULL")
|
("pg_catalog.current_setting('restore_command')" if postgresql.major_version >= 120000 else "NULL") +\
|
||||||
|
", " + ("pg_catalog.pg_wal_lsn_diff(wr.written_lsn, '0/0')::bigint"
|
||||||
|
if postgresql.major_version >= 130000 else "NULL")
|
||||||
stmt = ("SELECT " + postgresql.POSTMASTER_START_TIME + ", " + postgresql.TL_LSN + ","
|
stmt = ("SELECT " + postgresql.POSTMASTER_START_TIME + ", " + postgresql.TL_LSN + ","
|
||||||
" pg_catalog.pg_last_xact_replay_timestamp(), " + replication_state + ","
|
" pg_catalog.pg_last_xact_replay_timestamp(), " + replication_state + ","
|
||||||
" pg_catalog.array_to_json(pg_catalog.array_agg(pg_catalog.row_to_json(ri))) "
|
" (SELECT pg_catalog.array_to_json(pg_catalog.array_agg(pg_catalog.row_to_json(ri))) "
|
||||||
"FROM (SELECT (SELECT rolname FROM pg_catalog.pg_authid WHERE oid = usesysid) AS usename,"
|
"FROM (SELECT (SELECT rolname FROM pg_catalog.pg_authid WHERE oid = usesysid) AS usename,"
|
||||||
" application_name, client_addr, w.state, sync_state, sync_priority"
|
" application_name, client_addr, w.state, sync_state, sync_priority"
|
||||||
" FROM pg_catalog.pg_stat_get_wal_senders() w, pg_catalog.pg_stat_get_activity(pid)) AS ri")
|
" FROM pg_catalog.pg_stat_get_wal_senders() w, pg_catalog.pg_stat_get_activity(pid)) AS ri)") +\
|
||||||
|
(" FROM pg_catalog.pg_stat_get_wal_receiver() AS wr" if postgresql.major_version >= 90600 else "")
|
||||||
|
|
||||||
row = self.query(stmt.format(postgresql.wal_name, postgresql.lsn_name,
|
row = self.query(stmt.format(postgresql.wal_name, postgresql.lsn_name,
|
||||||
postgresql.wal_flush), retry=retry)[0]
|
postgresql.wal_flush), retry=retry)[0]
|
||||||
result = {
|
result = {
|
||||||
'state': postgresql.state,
|
'state': postgresql.state,
|
||||||
'postmaster_start_time': row[0],
|
'postmaster_start_time': row[0],
|
||||||
'role': 'replica' if row[1] == 0 else 'master',
|
'role': PostgresqlRole.REPLICA if row[1] == 0 else PostgresqlRole.PRIMARY,
|
||||||
'server_version': postgresql.server_version,
|
'server_version': postgresql.server_version,
|
||||||
'xlog': ({
|
'xlog': ({
|
||||||
'received_location': row[4] or row[3],
|
'received_location': row[10] or row[4] or row[3],
|
||||||
'replayed_location': row[3],
|
'replayed_location': row[3],
|
||||||
'replayed_timestamp': row[6],
|
'replayed_timestamp': row[6],
|
||||||
'paused': row[5]} if row[1] == 0 else {
|
'paused': row[5]} if row[1] == 0 else {
|
||||||
@@ -1301,12 +1342,12 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
if result['role'] == 'replica' and config.is_standby_cluster:
|
if result['role'] == PostgresqlRole.REPLICA and config.is_standby_cluster:
|
||||||
result['role'] = postgresql.role
|
result['role'] = postgresql.role
|
||||||
|
|
||||||
if result['role'] == 'replica' and config.is_synchronous_mode\
|
if result['role'] == PostgresqlRole.REPLICA and config.is_synchronous_mode\
|
||||||
and cluster and cluster.sync.matches(postgresql.name):
|
and cluster and cluster.sync.matches(postgresql.name):
|
||||||
result['sync_standby'] = True
|
result['quorum_standby' if global_config.is_quorum_commit_mode else 'sync_standby'] = True
|
||||||
|
|
||||||
if row[1] > 0:
|
if row[1] > 0:
|
||||||
result['timeline'] = row[1]
|
result['timeline'] = row[1]
|
||||||
@@ -1315,16 +1356,19 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
if not cluster or cluster.is_unlocked() or not cluster.leader else cluster.leader.timeline
|
if not cluster or cluster.is_unlocked() or not cluster.leader else cluster.leader.timeline
|
||||||
result['timeline'] = postgresql.replica_cached_timeline(leader_timeline)
|
result['timeline'] = postgresql.replica_cached_timeline(leader_timeline)
|
||||||
|
|
||||||
replication_state = postgresql.replication_state_from_parameters(row[1] > 0, row[7], row[8])
|
if row[7]:
|
||||||
|
result['latest_end_lsn'] = row[7]
|
||||||
|
|
||||||
|
replication_state = postgresql.replication_state_from_parameters(row[1] > 0, row[8], row[9])
|
||||||
if replication_state:
|
if replication_state:
|
||||||
result['replication_state'] = replication_state
|
result['replication_state'] = replication_state
|
||||||
|
|
||||||
if row[9]:
|
if row[11]:
|
||||||
result['replication'] = row[9]
|
result['replication'] = row[11]
|
||||||
|
|
||||||
except (psycopg.Error, RetryFailedError, PostgresConnectionException):
|
except (psycopg.Error, RetryFailedError, PostgresConnectionException):
|
||||||
state = postgresql.state
|
state = postgresql.state
|
||||||
if state == 'running':
|
if state == PostgresqlState.RUNNING:
|
||||||
logger.exception('get_postgresql_status')
|
logger.exception('get_postgresql_status')
|
||||||
state = 'unknown'
|
state = 'unknown'
|
||||||
result: Dict[str, Any] = {'state': state, 'role': postgresql.role}
|
result: Dict[str, Any] = {'state': state, 'role': postgresql.role}
|
||||||
@@ -1499,7 +1543,7 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.debug('Failed to parse url %s: %r', member.api_url, e)
|
logger.debug('Failed to parse url %s: %r', member.api_url, e)
|
||||||
|
|
||||||
def check_access(self, rh: RestApiHandler) -> Optional[bool]:
|
def check_access(self, rh: RestApiHandler, allowlist_check_members: bool = True) -> Optional[bool]:
|
||||||
"""Ensure client has enough privileges to perform a given request.
|
"""Ensure client has enough privileges to perform a given request.
|
||||||
|
|
||||||
Write a response back to the client if any issue is observed, and the HTTP status may be:
|
Write a response back to the client if any issue is observed, and the HTTP status may be:
|
||||||
@@ -1512,12 +1556,17 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
* a client certificate is expected by the server, but is missing in the request.
|
* a client certificate is expected by the server, but is missing in the request.
|
||||||
|
|
||||||
:param rh: the request which access should be checked.
|
:param rh: the request which access should be checked.
|
||||||
|
:param allowlist_check_members: whether we should check the source ip against existing cluster members.
|
||||||
|
|
||||||
:returns: ``True`` if client access verification succeeded, otherwise ``None``.
|
:returns: ``True`` if client access verification succeeded, otherwise ``None``.
|
||||||
"""
|
"""
|
||||||
if self.__allowlist or self.__allowlist_include_members:
|
allowlist_check_members = allowlist_check_members and bool(self.__allowlist_include_members)
|
||||||
|
if self.__allowlist or allowlist_check_members:
|
||||||
incoming_ip = ip_address(rh.client_address[0])
|
incoming_ip = ip_address(rh.client_address[0])
|
||||||
if not any(incoming_ip in net for net in self.__allowlist + tuple(self.__members_ips())):
|
|
||||||
|
members_ips = tuple(self.__members_ips()) if allowlist_check_members else ()
|
||||||
|
|
||||||
|
if not any(incoming_ip in net for net in self.__allowlist + members_ips):
|
||||||
return rh.write_response(403, 'Access is denied')
|
return rh.write_response(403, 'Access is denied')
|
||||||
|
|
||||||
if not hasattr(rh.request, 'getpeercert') or not rh.request.getpeercert(): # valid client cert isn't present
|
if not hasattr(rh.request, 'getpeercert') or not rh.request.getpeercert(): # valid client cert isn't present
|
||||||
@@ -1563,13 +1612,16 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
if hostname in ('', '*'):
|
if hostname in ('', '*'):
|
||||||
hostname = None
|
hostname = None
|
||||||
|
|
||||||
info = socket.getaddrinfo(hostname, port, socket.AF_UNSPEC, socket.SOCK_STREAM, 0, socket.AI_PASSIVE)
|
# Filter out unexpected results when python is compiled with --disable-ipv6 and running on IPv6 system.
|
||||||
|
info = [(a[0], a[4][0], a[4][1])
|
||||||
|
for a in socket.getaddrinfo(hostname, port, socket.AF_UNSPEC, socket.SOCK_STREAM, 0, socket.AI_PASSIVE)
|
||||||
|
if isinstance(a[4][0], str) and isinstance(a[4][1], int)]
|
||||||
# in case dual stack is not supported we want IPv4 to be preferred over IPv6
|
# in case dual stack is not supported we want IPv4 to be preferred over IPv6
|
||||||
info.sort(key=lambda x: x[0] == socket.AF_INET, reverse=not dual_stack)
|
info.sort(key=lambda x: x[0] == socket.AF_INET, reverse=not dual_stack)
|
||||||
|
|
||||||
self.address_family = info[0][0]
|
self.address_family = info[0][0]
|
||||||
try:
|
try:
|
||||||
HTTPServer.__init__(self, info[0][-1][:2], RestApiHandler)
|
HTTPServer.__init__(self, (info[0][1], info[0][2]), RestApiHandler)
|
||||||
except socket.error:
|
except socket.error:
|
||||||
logger.error(
|
logger.error(
|
||||||
"Couldn't start a service on '%s:%s', please check your `restapi.listen` configuration", hostname, port)
|
"Couldn't start a service on '%s:%s', please check your `restapi.listen` configuration", hostname, port)
|
||||||
@@ -1686,12 +1738,11 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
|
|
||||||
:returns: serial number of the certificate configured through ``restapi.certfile`` setting.
|
:returns: serial number of the certificate configured through ``restapi.certfile`` setting.
|
||||||
"""
|
"""
|
||||||
if self.__ssl_options.get('certfile'):
|
certfile: Optional[str] = self.__ssl_options.get('certfile')
|
||||||
|
if certfile:
|
||||||
import ssl
|
import ssl
|
||||||
try:
|
try:
|
||||||
crt: Dict[str, Any] = ssl._ssl._test_decode_cert(self.__ssl_options['certfile']) # pyright: ignore
|
crt = cast(Dict[str, Any], ssl._ssl._test_decode_cert(certfile)) # pyright: ignore
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
|
||||||
assert isinstance(crt, dict)
|
|
||||||
return crt.get('serialNumber')
|
return crt.get('serialNumber')
|
||||||
except ssl.SSLError as e:
|
except ssl.SSLError as e:
|
||||||
logger.error('Failed to get serial number from certificate %s: %r', self.__ssl_options['certfile'], e)
|
logger.error('Failed to get serial number from certificate %s: %r', self.__ssl_options['certfile'], e)
|
||||||
|
|||||||
+49
-4
@@ -1,9 +1,10 @@
|
|||||||
"""Patroni custom object types somewhat like :mod:`collections` module.
|
"""Patroni custom object types somewhat like :mod:`collections` module.
|
||||||
|
|
||||||
Provides a case insensitive :class:`dict` and :class:`set` object types.
|
Provides a case insensitive :class:`dict` and :class:`set` object types, and `EMPTY_DICT` frozen dictionary object.
|
||||||
"""
|
"""
|
||||||
from collections import OrderedDict
|
from collections import OrderedDict
|
||||||
from typing import Any, Collection, Dict, Iterator, KeysView, MutableMapping, MutableSet, Optional
|
from copy import deepcopy
|
||||||
|
from typing import Any, Collection, Dict, Iterator, KeysView, Mapping, MutableMapping, MutableSet, Optional
|
||||||
|
|
||||||
|
|
||||||
class CaseInsensitiveSet(MutableSet[str]):
|
class CaseInsensitiveSet(MutableSet[str]):
|
||||||
@@ -48,7 +49,7 @@ class CaseInsensitiveSet(MutableSet[str]):
|
|||||||
"""
|
"""
|
||||||
return str(set(self._values.values()))
|
return str(set(self._values.values()))
|
||||||
|
|
||||||
def __contains__(self, value: str) -> bool:
|
def __contains__(self, value: object) -> bool:
|
||||||
"""Check if set contains *value*.
|
"""Check if set contains *value*.
|
||||||
|
|
||||||
The check is performed case-insensitively.
|
The check is performed case-insensitively.
|
||||||
@@ -57,7 +58,7 @@ class CaseInsensitiveSet(MutableSet[str]):
|
|||||||
|
|
||||||
:returns: ``True`` if *value* is already in the set, ``False`` otherwise.
|
:returns: ``True`` if *value* is already in the set, ``False`` otherwise.
|
||||||
"""
|
"""
|
||||||
return value.lower() in self._values
|
return isinstance(value, str) and value.lower() in self._values
|
||||||
|
|
||||||
def __iter__(self) -> Iterator[str]:
|
def __iter__(self) -> Iterator[str]:
|
||||||
"""Iterate over the values in this set.
|
"""Iterate over the values in this set.
|
||||||
@@ -207,3 +208,47 @@ class CaseInsensitiveDict(MutableMapping[str, Any]):
|
|||||||
"<CaseInsensitiveDict{'A': 'B', 'c': 'd'} at ..."
|
"<CaseInsensitiveDict{'A': 'B', 'c': 'd'} at ..."
|
||||||
"""
|
"""
|
||||||
return '<{0}{1} at {2:x}>'.format(type(self).__name__, dict(self.items()), id(self))
|
return '<{0}{1} at {2:x}>'.format(type(self).__name__, dict(self.items()), id(self))
|
||||||
|
|
||||||
|
|
||||||
|
class _FrozenDict(Mapping[str, Any]):
|
||||||
|
"""Frozen dictionary object."""
|
||||||
|
|
||||||
|
def __init__(self, *args: Any, **kwargs: Any) -> None:
|
||||||
|
"""Create a new instance of :class:`_FrozenDict` with given data."""
|
||||||
|
self.__values: Dict[str, Any] = dict(*args, **kwargs)
|
||||||
|
|
||||||
|
def __iter__(self) -> Iterator[str]:
|
||||||
|
"""Iterate over keys of this dict.
|
||||||
|
|
||||||
|
:yields: each key present in the dict. Yields each key with its last case that has been stored.
|
||||||
|
"""
|
||||||
|
return iter(self.__values)
|
||||||
|
|
||||||
|
def __len__(self) -> int:
|
||||||
|
"""Get the length of this dict.
|
||||||
|
|
||||||
|
:returns: number of keys in the dict.
|
||||||
|
|
||||||
|
:Example:
|
||||||
|
|
||||||
|
>>> len(_FrozenDict())
|
||||||
|
0
|
||||||
|
"""
|
||||||
|
return len(self.__values)
|
||||||
|
|
||||||
|
def __getitem__(self, key: str) -> Any:
|
||||||
|
"""Get the value corresponding to *key*.
|
||||||
|
|
||||||
|
:returns: value corresponding to *key*.
|
||||||
|
"""
|
||||||
|
return self.__values[key]
|
||||||
|
|
||||||
|
def copy(self) -> Dict[str, Any]:
|
||||||
|
"""Create a copy of this dict.
|
||||||
|
|
||||||
|
:return: a new dict object with the same keys and values of this dict.
|
||||||
|
"""
|
||||||
|
return deepcopy(self.__values)
|
||||||
|
|
||||||
|
|
||||||
|
EMPTY_DICT = _FrozenDict()
|
||||||
|
|||||||
+50
-54
@@ -1,24 +1,25 @@
|
|||||||
"""Facilities related to Patroni configuration."""
|
"""Facilities related to Patroni configuration."""
|
||||||
import re
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
import shutil
|
import shutil
|
||||||
import tempfile
|
import tempfile
|
||||||
import yaml
|
|
||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
from typing import Any, Callable, Collection, Dict, List, Optional, Union, TYPE_CHECKING
|
from typing import Any, Callable, cast, Collection, Dict, List, Optional, TYPE_CHECKING, Union
|
||||||
|
|
||||||
|
import yaml
|
||||||
|
|
||||||
from . import PATRONI_ENV_PREFIX
|
from . import PATRONI_ENV_PREFIX
|
||||||
from .collections import CaseInsensitiveDict
|
from .collections import CaseInsensitiveDict, EMPTY_DICT
|
||||||
from .dcs import ClusterConfig
|
from .dcs import ClusterConfig
|
||||||
from .exceptions import ConfigParseError
|
from .exceptions import ConfigParseError
|
||||||
from .file_perm import pg_perm
|
from .file_perm import pg_perm
|
||||||
from .postgresql.config import ConfigHandler
|
from .postgresql.config import ConfigHandler
|
||||||
from .validator import IntValidator
|
|
||||||
from .utils import deep_compare, parse_bool, parse_int, patch_config
|
from .utils import deep_compare, parse_bool, parse_int, patch_config
|
||||||
|
from .validator import IntValidator
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -33,7 +34,8 @@ _AUTH_ALLOWED_PARAMETERS = (
|
|||||||
'sslcrl',
|
'sslcrl',
|
||||||
'sslcrldir',
|
'sslcrldir',
|
||||||
'gssencmode',
|
'gssencmode',
|
||||||
'channel_binding'
|
'channel_binding',
|
||||||
|
'sslnegotiation'
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -145,7 +147,7 @@ class Config(object):
|
|||||||
self._cache_file = os.path.join(self._data_dir, self.__CACHE_FILENAME)
|
self._cache_file = os.path.join(self._data_dir, self.__CACHE_FILENAME)
|
||||||
if validator: # patronictl uses validator=None
|
if validator: # patronictl uses validator=None
|
||||||
self._load_cache() # we don't want to load anything from local cache for ctl
|
self._load_cache() # we don't want to load anything from local cache for ctl
|
||||||
self._validate_failover_tags() # irrelevant for ctl
|
self._validate_contradictory_tags() # irrelevant for ctl
|
||||||
self._cache_needs_saving = False
|
self._cache_needs_saving = False
|
||||||
|
|
||||||
@property
|
@property
|
||||||
@@ -186,6 +188,7 @@ class Config(object):
|
|||||||
|
|
||||||
:raises:
|
:raises:
|
||||||
:class:`ConfigParseError`: if *path* is invalid.
|
:class:`ConfigParseError`: if *path* is invalid.
|
||||||
|
:class:`ConfigParseError`: if *path* does not contain dict (empty file or no mapping values).
|
||||||
"""
|
"""
|
||||||
if os.path.isfile(path):
|
if os.path.isfile(path):
|
||||||
files = [path]
|
files = [path]
|
||||||
@@ -200,7 +203,10 @@ class Config(object):
|
|||||||
for fname in files:
|
for fname in files:
|
||||||
with open(fname) as f:
|
with open(fname) as f:
|
||||||
config = yaml.safe_load(f)
|
config = yaml.safe_load(f)
|
||||||
patch_config(overall_config, config)
|
if not isinstance(config, dict):
|
||||||
|
logger.error('%s does not contain a dict', fname)
|
||||||
|
raise ConfigParseError(f'invalid config file {fname}')
|
||||||
|
patch_config(overall_config, cast(Dict[Any, Any], config))
|
||||||
return overall_config
|
return overall_config
|
||||||
|
|
||||||
def _load_config_file(self) -> Dict[str, Any]:
|
def _load_config_file(self) -> Dict[str, Any]:
|
||||||
@@ -357,7 +363,7 @@ class Config(object):
|
|||||||
new_configuration = self._build_effective_configuration(self._dynamic_configuration, configuration)
|
new_configuration = self._build_effective_configuration(self._dynamic_configuration, configuration)
|
||||||
self._local_configuration = configuration
|
self._local_configuration = configuration
|
||||||
self.__effective_configuration = new_configuration
|
self.__effective_configuration = new_configuration
|
||||||
self._validate_failover_tags()
|
self._validate_contradictory_tags()
|
||||||
return True
|
return True
|
||||||
else:
|
else:
|
||||||
logger.info('No local configuration items changed.')
|
logger.info('No local configuration items changed.')
|
||||||
@@ -388,7 +394,6 @@ class Config(object):
|
|||||||
* ``cluster_name``: set through ``scope`` local configuration or through ``PATRONI_SCOPE`` environment
|
* ``cluster_name``: set through ``scope`` local configuration or through ``PATRONI_SCOPE`` environment
|
||||||
variable;
|
variable;
|
||||||
* ``hot_standby``: always enabled;
|
* ``hot_standby``: always enabled;
|
||||||
* ``wal_log_hints``: always enabled.
|
|
||||||
|
|
||||||
:param parameters: Postgres parameters to be processed. Should be the parsed YAML value of
|
:param parameters: Postgres parameters to be processed. Should be the parsed YAML value of
|
||||||
``postgresql.parameters`` configuration, either from local or from dynamic configuration.
|
``postgresql.parameters`` configuration, either from local or from dynamic configuration.
|
||||||
@@ -445,14 +450,14 @@ class Config(object):
|
|||||||
|
|
||||||
for name, value in dynamic_configuration.items():
|
for name, value in dynamic_configuration.items():
|
||||||
if name == 'postgresql':
|
if name == 'postgresql':
|
||||||
for name, value in (value or {}).items():
|
for name, value in (value or EMPTY_DICT).items():
|
||||||
if name == 'parameters':
|
if name == 'parameters':
|
||||||
config['postgresql'][name].update(self._process_postgresql_parameters(value))
|
config['postgresql'][name].update(self._process_postgresql_parameters(value))
|
||||||
elif name not in ('connect_address', 'proxy_address', 'listen',
|
elif name not in ('connect_address', 'proxy_address', 'listen',
|
||||||
'config_dir', 'data_dir', 'pgpass', 'authentication'):
|
'config_dir', 'data_dir', 'pgpass', 'authentication'):
|
||||||
config['postgresql'][name] = deepcopy(value)
|
config['postgresql'][name] = deepcopy(value)
|
||||||
elif name == 'standby_cluster':
|
elif name == 'standby_cluster':
|
||||||
for name, value in (value or {}).items():
|
for name, value in (value or EMPTY_DICT).items():
|
||||||
if name in self.__DEFAULT_CONFIG['standby_cluster']:
|
if name in self.__DEFAULT_CONFIG['standby_cluster']:
|
||||||
config['standby_cluster'][name] = deepcopy(value)
|
config['standby_cluster'][name] = deepcopy(value)
|
||||||
elif name in config: # only variables present in __DEFAULT_CONFIG allowed to be overridden from DCS
|
elif name in config: # only variables present in __DEFAULT_CONFIG allowed to be overridden from DCS
|
||||||
@@ -536,7 +541,8 @@ class Config(object):
|
|||||||
_set_section_values('postgresql', ['listen', 'connect_address', 'proxy_address',
|
_set_section_values('postgresql', ['listen', 'connect_address', 'proxy_address',
|
||||||
'config_dir', 'data_dir', 'pgpass', 'bin_dir'])
|
'config_dir', 'data_dir', 'pgpass', 'bin_dir'])
|
||||||
_set_section_values('log', ['type', 'level', 'traceback_level', 'format', 'dateformat', 'static_fields',
|
_set_section_values('log', ['type', 'level', 'traceback_level', 'format', 'dateformat', 'static_fields',
|
||||||
'max_queue_size', 'dir', 'file_size', 'file_num', 'loggers'])
|
'max_queue_size', 'dir', 'mode', 'file_size', 'file_num', 'loggers',
|
||||||
|
'deduplicate_heartbeat_logs'])
|
||||||
_set_section_values('raft', ['data_dir', 'self_addr', 'partner_addrs', 'password', 'bind_addr'])
|
_set_section_values('raft', ['data_dir', 'self_addr', 'partner_addrs', 'password', 'bind_addr'])
|
||||||
|
|
||||||
for binary in ('pg_ctl', 'initdb', 'pg_controldata', 'pg_basebackup', 'postgres', 'pg_isready', 'pg_rewind'):
|
for binary in ('pg_ctl', 'initdb', 'pg_controldata', 'pg_basebackup', 'postgres', 'pg_isready', 'pg_rewind'):
|
||||||
@@ -545,7 +551,8 @@ class Config(object):
|
|||||||
ret['postgresql'].setdefault('bin_name', {})[binary] = value
|
ret['postgresql'].setdefault('bin_name', {})[binary] = value
|
||||||
|
|
||||||
# parse all values retrieved from the environment as Python objects, according to the expected type
|
# parse all values retrieved from the environment as Python objects, according to the expected type
|
||||||
for first, second in (('restapi', 'allowlist_include_members'), ('ctl', 'insecure')):
|
for first, second in (('restapi', 'allowlist_include_members'), ('ctl', 'insecure'),
|
||||||
|
('log', 'deduplicate_heartbeat_logs')):
|
||||||
value = ret.get(first, {}).pop(second, None)
|
value = ret.get(first, {}).pop(second, None)
|
||||||
if value:
|
if value:
|
||||||
value = parse_bool(value)
|
value = parse_bool(value)
|
||||||
@@ -553,7 +560,7 @@ class Config(object):
|
|||||||
ret[first][second] = value
|
ret[first][second] = value
|
||||||
|
|
||||||
for first, params in (('restapi', ('request_queue_size',)),
|
for first, params in (('restapi', ('request_queue_size',)),
|
||||||
('log', ('max_queue_size', 'file_size', 'file_num'))):
|
('log', ('max_queue_size', 'file_size', 'file_num', 'mode'))):
|
||||||
for second in params:
|
for second in params:
|
||||||
value = ret.get(first, {}).pop(second, None)
|
value = ret.get(first, {}).pop(second, None)
|
||||||
if value:
|
if value:
|
||||||
@@ -656,7 +663,7 @@ class Config(object):
|
|||||||
'SERVICE_TAGS', 'NAMESPACE', 'CONTEXT', 'USE_ENDPOINTS', 'SCOPE_LABEL', 'ROLE_LABEL',
|
'SERVICE_TAGS', 'NAMESPACE', 'CONTEXT', 'USE_ENDPOINTS', 'SCOPE_LABEL', 'ROLE_LABEL',
|
||||||
'POD_IP', 'PORTS', 'LABELS', 'BYPASS_API_SERVICE', 'RETRIABLE_HTTP_CODES', 'KEY_PASSWORD',
|
'POD_IP', 'PORTS', 'LABELS', 'BYPASS_API_SERVICE', 'RETRIABLE_HTTP_CODES', 'KEY_PASSWORD',
|
||||||
'USE_SSL', 'SET_ACLS', 'GROUP', 'DATABASE', 'LEADER_LABEL_VALUE', 'FOLLOWER_LABEL_VALUE',
|
'USE_SSL', 'SET_ACLS', 'GROUP', 'DATABASE', 'LEADER_LABEL_VALUE', 'FOLLOWER_LABEL_VALUE',
|
||||||
'STANDBY_LEADER_LABEL_VALUE', 'TMP_ROLE_LABEL', 'AUTH_DATA') and name:
|
'STANDBY_LEADER_LABEL_VALUE', 'TMP_ROLE_LABEL', 'AUTH_DATA', 'BOOTSTRAP_LABELS') and name:
|
||||||
value = os.environ.pop(param)
|
value = os.environ.pop(param)
|
||||||
if name == 'CITUS':
|
if name == 'CITUS':
|
||||||
if suffix == 'GROUP':
|
if suffix == 'GROUP':
|
||||||
@@ -667,7 +674,7 @@ class Config(object):
|
|||||||
value = value and parse_int(value)
|
value = value and parse_int(value)
|
||||||
elif suffix in ('HOSTS', 'PORTS', 'CHECKS', 'SERVICE_TAGS', 'RETRIABLE_HTTP_CODES'):
|
elif suffix in ('HOSTS', 'PORTS', 'CHECKS', 'SERVICE_TAGS', 'RETRIABLE_HTTP_CODES'):
|
||||||
value = value and _parse_list(value)
|
value = value and _parse_list(value)
|
||||||
elif suffix in ('LABELS', 'SET_ACLS', 'AUTH_DATA'):
|
elif suffix in ('LABELS', 'SET_ACLS', 'AUTH_DATA', 'BOOTSTRAP_LABELS'):
|
||||||
value = _parse_dict(value)
|
value = _parse_dict(value)
|
||||||
elif suffix in ('USE_PROXIES', 'REGISTER_SERVICE', 'USE_ENDPOINTS', 'BYPASS_API_SERVICE', 'VERIFY'):
|
elif suffix in ('USE_PROXIES', 'REGISTER_SERVICE', 'USE_ENDPOINTS', 'BYPASS_API_SERVICE', 'VERIFY'):
|
||||||
value = parse_bool(value)
|
value = parse_bool(value)
|
||||||
@@ -677,23 +684,6 @@ class Config(object):
|
|||||||
if dcs in ret:
|
if dcs in ret:
|
||||||
ret[dcs].update(_get_auth(dcs))
|
ret[dcs].update(_get_auth(dcs))
|
||||||
|
|
||||||
users = {}
|
|
||||||
for param in list(os.environ.keys()):
|
|
||||||
if param.startswith(PATRONI_ENV_PREFIX):
|
|
||||||
name, suffix = (param[len(PATRONI_ENV_PREFIX):].rsplit('_', 1) + [''])[:2]
|
|
||||||
# PATRONI_<username>_PASSWORD=<password>, PATRONI_<username>_OPTIONS=<option1,option2,...>
|
|
||||||
# CREATE USER "<username>" WITH <OPTIONS> PASSWORD '<password>'
|
|
||||||
if name and suffix == 'PASSWORD':
|
|
||||||
password = os.environ.pop(param)
|
|
||||||
if password:
|
|
||||||
users[name] = {'password': password}
|
|
||||||
options = os.environ.pop(param[:-9] + '_OPTIONS', None) # replace "_PASSWORD" with "_OPTIONS"
|
|
||||||
options = options and _parse_list(options)
|
|
||||||
if options:
|
|
||||||
users[name]['options'] = options
|
|
||||||
if users:
|
|
||||||
ret['bootstrap']['users'] = users
|
|
||||||
|
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
def _build_effective_configuration(self, dynamic_configuration: Dict[str, Any],
|
def _build_effective_configuration(self, dynamic_configuration: Dict[str, Any],
|
||||||
@@ -711,8 +701,8 @@ class Config(object):
|
|||||||
config = self._safe_copy_dynamic_configuration(dynamic_configuration)
|
config = self._safe_copy_dynamic_configuration(dynamic_configuration)
|
||||||
for name, value in local_configuration.items():
|
for name, value in local_configuration.items():
|
||||||
if name == 'citus': # remove invalid citus configuration
|
if name == 'citus': # remove invalid citus configuration
|
||||||
if isinstance(value, dict) and isinstance(value.get('group'), int)\
|
if isinstance(value, dict) and isinstance(cast(Dict[str, Any], value).get('group'), int) \
|
||||||
and isinstance(value.get('database'), str):
|
and isinstance(cast(Dict[str, Any], value).get('database'), str):
|
||||||
config[name] = value
|
config[name] = value
|
||||||
elif name == 'postgresql':
|
elif name == 'postgresql':
|
||||||
for name, value in (value or {}).items():
|
for name, value in (value or {}).items():
|
||||||
@@ -757,7 +747,7 @@ class Config(object):
|
|||||||
if 'citus' in config:
|
if 'citus' in config:
|
||||||
bootstrap = config.setdefault('bootstrap', {})
|
bootstrap = config.setdefault('bootstrap', {})
|
||||||
dcs = bootstrap.setdefault('dcs', {})
|
dcs = bootstrap.setdefault('dcs', {})
|
||||||
dcs.setdefault('synchronous_mode', True)
|
dcs.setdefault('synchronous_mode', 'quorum')
|
||||||
|
|
||||||
updated_fields = (
|
updated_fields = (
|
||||||
'name',
|
'name',
|
||||||
@@ -814,25 +804,31 @@ class Config(object):
|
|||||||
"""
|
"""
|
||||||
return deepcopy(self.__effective_configuration)
|
return deepcopy(self.__effective_configuration)
|
||||||
|
|
||||||
def _validate_failover_tags(self) -> None:
|
def _validate_contradictory_tags(self) -> None:
|
||||||
"""Check ``nofailover``/``failover_priority`` config and warn user if it's contradictory.
|
"""Check boolean/priority tags' config and warn user if it's contradictory.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
To preserve sanity (and backwards compatibility) the ``nofailover`` tag will still exist. A contradictory
|
To preserve sanity (and backwards compatibility) the ``nofailover``/``nosync`` tag will still exist.
|
||||||
configuration is one where ``nofailover`` is ``True`` but ``failover_priority > 0``, or where
|
A contradictory configuration is one where ``nofailover``/``nosync`` is ``True`` but
|
||||||
``nofailover`` is ``False``, but ``failover_priority <= 0``. Essentially, ``nofailover`` and
|
``failover_priority > 0``/``sync_priority > 0``, or where ``nofailover``/``nosync`` is ``False``,
|
||||||
``failover_priority`` are communicating different things.
|
but ``failover_priority <= 0``/``sync_priority <= 0``. Essentially, ``nofailover``/``nosync`` and
|
||||||
|
``failover_priority``/``sync_priority`` are communicating different things.
|
||||||
This checks for this edge case (which is a misconfiguration on the part of the user) and warns them.
|
This checks for this edge case (which is a misconfiguration on the part of the user) and warns them.
|
||||||
The behaviour is as if ``failover_priority`` were not provided (i.e ``nofailover`` is the
|
The behaviour is as if ``failover_priority``/``sync_priority`` were not provided
|
||||||
bedrock source of truth)
|
(i.e ``nofailover``/``nosync`` is the bedrock source of truth).
|
||||||
"""
|
"""
|
||||||
tags = self.get('tags', {})
|
tags = self.get('tags', {})
|
||||||
if 'nofailover' not in tags:
|
|
||||||
return
|
def validate_tag(bool_name: str, priority_name: str) -> None:
|
||||||
nofailover_tag = tags.get('nofailover')
|
if bool_name not in tags:
|
||||||
failover_priority_tag = parse_int(tags.get('failover_priority'))
|
return
|
||||||
if failover_priority_tag is not None \
|
bool_tag = tags.get(bool_name)
|
||||||
and (bool(nofailover_tag) is True and failover_priority_tag > 0
|
priority_tag = parse_int(tags.get(priority_name))
|
||||||
or bool(nofailover_tag) is False and failover_priority_tag <= 0):
|
if priority_tag is not None \
|
||||||
logger.warning('Conflicting configuration between nofailover: %s and failover_priority: %s. '
|
and (bool(bool_tag) is True and priority_tag > 0
|
||||||
'Defaulting to nofailover: %s', nofailover_tag, failover_priority_tag, nofailover_tag)
|
or bool(bool_tag) is False and priority_tag <= 0):
|
||||||
|
logger.warning('Conflicting configuration between %s: %s and %s: %s. Defaulting to %s: %s',
|
||||||
|
bool_name, bool_tag, priority_name, priority_tag, bool_name, bool_tag)
|
||||||
|
|
||||||
|
validate_tag('nofailover', 'failover_priority')
|
||||||
|
validate_tag('nosync', 'sync_priority')
|
||||||
|
|||||||
+39
-27
@@ -2,19 +2,22 @@
|
|||||||
import abc
|
import abc
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import psutil
|
|
||||||
import socket
|
import socket
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
|
from contextlib import contextmanager
|
||||||
|
from getpass import getpass, getuser
|
||||||
|
from typing import Any, Dict, Iterator, List, Optional, TextIO, Tuple, TYPE_CHECKING, Union
|
||||||
|
|
||||||
|
import psutil
|
||||||
import yaml
|
import yaml
|
||||||
|
|
||||||
from getpass import getuser, getpass
|
|
||||||
from contextlib import contextmanager
|
|
||||||
from typing import Any, Dict, Iterator, List, Optional, TextIO, Tuple, TYPE_CHECKING, Union
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from psycopg import Cursor
|
from psycopg import Cursor
|
||||||
from psycopg2 import cursor
|
from psycopg2 import cursor
|
||||||
|
|
||||||
from . import psycopg
|
from . import psycopg
|
||||||
|
from .collections import EMPTY_DICT
|
||||||
from .config import Config
|
from .config import Config
|
||||||
from .exceptions import PatroniException
|
from .exceptions import PatroniException
|
||||||
from .log import PatroniLogger
|
from .log import PatroniLogger
|
||||||
@@ -22,7 +25,6 @@ from .postgresql.config import ConfigHandler, parse_dsn
|
|||||||
from .postgresql.misc import postgres_major_version_to_int
|
from .postgresql.misc import postgres_major_version_to_int
|
||||||
from .utils import get_major_version, parse_bool, patch_config, read_stripped
|
from .utils import get_major_version, parse_bool, patch_config, read_stripped
|
||||||
|
|
||||||
|
|
||||||
# Mapping between the libpq connection parameters and the environment variables.
|
# Mapping between the libpq connection parameters and the environment variables.
|
||||||
# This dict should be kept in sync with `patroni.utils._AUTH_ALLOWED_PARAMETERS`
|
# This dict should be kept in sync with `patroni.utils._AUTH_ALLOWED_PARAMETERS`
|
||||||
# (we use "username" in the Patroni config for some reason, other parameter names are the same).
|
# (we use "username" in the Patroni config for some reason, other parameter names are the same).
|
||||||
@@ -37,7 +39,8 @@ _AUTH_ALLOWED_PARAMETERS_MAPPING = {
|
|||||||
'sslcrl': 'PGSSLCRL',
|
'sslcrl': 'PGSSLCRL',
|
||||||
'sslcrldir': 'PGSSLCRLDIR',
|
'sslcrldir': 'PGSSLCRLDIR',
|
||||||
'gssencmode': 'PGGSSENCMODE',
|
'gssencmode': 'PGGSSENCMODE',
|
||||||
'channel_binding': 'PGCHANNELBINDING'
|
'channel_binding': 'PGCHANNELBINDING',
|
||||||
|
'sslnegotiation': 'PGSSLNEGOTIATION'
|
||||||
}
|
}
|
||||||
NO_VALUE_MSG = '#FIXME'
|
NO_VALUE_MSG = '#FIXME'
|
||||||
|
|
||||||
@@ -51,13 +54,15 @@ def get_address() -> Tuple[str, str]:
|
|||||||
:returns: tuple consisting of the hostname returned by :func:`~socket.gethostname`
|
:returns: tuple consisting of the hostname returned by :func:`~socket.gethostname`
|
||||||
and the first element in the sorted list of the addresses returned by :func:`~socket.getaddrinfo`.
|
and the first element in the sorted list of the addresses returned by :func:`~socket.getaddrinfo`.
|
||||||
Sorting guarantees it will prefer IPv4.
|
Sorting guarantees it will prefer IPv4.
|
||||||
If an exception occured, hostname and ip values are equal to :data:`~patroni.config_generator.NO_VALUE_MSG`.
|
If an exception occurred, hostname and ip values are equal to :data:`~patroni.config_generator.NO_VALUE_MSG`.
|
||||||
"""
|
"""
|
||||||
hostname = None
|
hostname = None
|
||||||
try:
|
try:
|
||||||
hostname = socket.gethostname()
|
hostname = socket.gethostname()
|
||||||
return hostname, sorted(socket.getaddrinfo(hostname, 0, socket.AF_UNSPEC, socket.SOCK_STREAM, 0),
|
# Filter out unexpected results when python is compiled with --disable-ipv6 and running on IPv6 system.
|
||||||
key=lambda x: x[0])[0][4][0]
|
addrs = [(a[0], a[4][0]) for a in socket.getaddrinfo(hostname, 0, socket.AF_UNSPEC, socket.SOCK_STREAM, 0)
|
||||||
|
if isinstance(a[4][0], str)]
|
||||||
|
return hostname, sorted(addrs, key=lambda x: x[0])[0][1]
|
||||||
except Exception as err:
|
except Exception as err:
|
||||||
logging.warning('Failed to obtain address: %r', err)
|
logging.warning('Failed to obtain address: %r', err)
|
||||||
return NO_VALUE_MSG, NO_VALUE_MSG
|
return NO_VALUE_MSG, NO_VALUE_MSG
|
||||||
@@ -71,8 +76,6 @@ class AbstractConfigGenerator(abc.ABC):
|
|||||||
:ivar config: dictionary used for the generated configuration storage.
|
:ivar config: dictionary used for the generated configuration storage.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
_HOSTNAME, _IP = get_address()
|
|
||||||
|
|
||||||
def __init__(self, output_file: Optional[str]) -> None:
|
def __init__(self, output_file: Optional[str]) -> None:
|
||||||
"""Set up the output file (if passed), helper vars and the minimal config structure.
|
"""Set up the output file (if passed), helper vars and the minimal config structure.
|
||||||
|
|
||||||
@@ -91,12 +94,14 @@ class AbstractConfigGenerator(abc.ABC):
|
|||||||
:returns: dictionary with the values gathered from Patroni env, hopefully defined hostname and ip address
|
:returns: dictionary with the values gathered from Patroni env, hopefully defined hostname and ip address
|
||||||
(otherwise set to :data:`~patroni.config_generator.NO_VALUE_MSG`), and some sane defaults.
|
(otherwise set to :data:`~patroni.config_generator.NO_VALUE_MSG`), and some sane defaults.
|
||||||
"""
|
"""
|
||||||
|
_HOSTNAME, _IP = get_address()
|
||||||
|
|
||||||
template_config: Dict[str, Any] = {
|
template_config: Dict[str, Any] = {
|
||||||
'scope': NO_VALUE_MSG,
|
'scope': NO_VALUE_MSG,
|
||||||
'name': cls._HOSTNAME,
|
'name': _HOSTNAME,
|
||||||
'restapi': {
|
'restapi': {
|
||||||
'connect_address': cls._IP + ':8008',
|
'connect_address': _IP + ':8008',
|
||||||
'listen': cls._IP + ':8008'
|
'listen': _IP + ':8008'
|
||||||
},
|
},
|
||||||
'log': {
|
'log': {
|
||||||
'type': PatroniLogger.DEFAULT_TYPE,
|
'type': PatroniLogger.DEFAULT_TYPE,
|
||||||
@@ -107,8 +112,8 @@ class AbstractConfigGenerator(abc.ABC):
|
|||||||
},
|
},
|
||||||
'postgresql': {
|
'postgresql': {
|
||||||
'data_dir': NO_VALUE_MSG,
|
'data_dir': NO_VALUE_MSG,
|
||||||
'connect_address': cls._IP + ':5432',
|
'connect_address': _IP + ':5432',
|
||||||
'listen': cls._IP + ':5432',
|
'listen': _IP + ':5432',
|
||||||
'bin_dir': '',
|
'bin_dir': '',
|
||||||
'authentication': {
|
'authentication': {
|
||||||
'superuser': {
|
'superuser': {
|
||||||
@@ -123,9 +128,11 @@ class AbstractConfigGenerator(abc.ABC):
|
|||||||
},
|
},
|
||||||
'tags': {
|
'tags': {
|
||||||
'failover_priority': 1,
|
'failover_priority': 1,
|
||||||
|
'sync_priority': 1,
|
||||||
'noloadbalance': False,
|
'noloadbalance': False,
|
||||||
'clonefrom': True,
|
'clonefrom': True,
|
||||||
'nosync': False,
|
'nosync': False,
|
||||||
|
'nostream': False,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -225,7 +232,7 @@ class AbstractConfigGenerator(abc.ABC):
|
|||||||
class SampleConfigGenerator(AbstractConfigGenerator):
|
class SampleConfigGenerator(AbstractConfigGenerator):
|
||||||
"""Object representing the generated sample Patroni config.
|
"""Object representing the generated sample Patroni config.
|
||||||
|
|
||||||
Sane defults are used based on the gathered PG version.
|
Sane defaults are used based on the gathered PG version.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@property
|
@property
|
||||||
@@ -243,7 +250,8 @@ class SampleConfigGenerator(AbstractConfigGenerator):
|
|||||||
See :func:`~patroni.postgresql.misc.postgres_major_version_to_int` and
|
See :func:`~patroni.postgresql.misc.postgres_major_version_to_int` and
|
||||||
:func:`~patroni.utils.get_major_version`.
|
:func:`~patroni.utils.get_major_version`.
|
||||||
"""
|
"""
|
||||||
postgres_bin = ((self.config.get('postgresql') or {}).get('bin_name') or {}).get('postgres', 'postgres')
|
postgres_bin = ((self.config.get('postgresql')
|
||||||
|
or EMPTY_DICT).get('bin_name') or EMPTY_DICT).get('postgres', 'postgres')
|
||||||
return postgres_major_version_to_int(get_major_version(self.config['postgresql'].get('bin_dir'), postgres_bin))
|
return postgres_major_version_to_int(get_major_version(self.config['postgresql'].get('bin_dir'), postgres_bin))
|
||||||
|
|
||||||
def generate(self) -> None:
|
def generate(self) -> None:
|
||||||
@@ -265,7 +273,8 @@ class SampleConfigGenerator(AbstractConfigGenerator):
|
|||||||
wal_level = 'hot_standby' if self.pg_major < 90600 else 'replica'
|
wal_level = 'hot_standby' if self.pg_major < 90600 else 'replica'
|
||||||
self.config['bootstrap']['dcs']['postgresql']['parameters']['wal_level'] = wal_level
|
self.config['bootstrap']['dcs']['postgresql']['parameters']['wal_level'] = wal_level
|
||||||
|
|
||||||
self.config['bootstrap']['dcs']['postgresql']['use_pg_rewind'] = True
|
self.config['bootstrap']['dcs']['postgresql']['use_pg_rewind'] = \
|
||||||
|
parse_bool(self.config['bootstrap']['dcs']['postgresql']['parameters']['wal_log_hints']) is True
|
||||||
if self.pg_major >= 110000:
|
if self.pg_major >= 110000:
|
||||||
self.config['postgresql']['authentication'].setdefault(
|
self.config['postgresql']['authentication'].setdefault(
|
||||||
'rewind', {'username': 'rewind_user'}).setdefault('password', NO_VALUE_MSG)
|
'rewind', {'username': 'rewind_user'}).setdefault('password', NO_VALUE_MSG)
|
||||||
@@ -308,7 +317,7 @@ class RunningClusterConfigGenerator(AbstractConfigGenerator):
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def _required_pg_params(self) -> List[str]:
|
def _required_pg_params(self) -> List[str]:
|
||||||
"""PG configuration prameters that have to be always present in the generated config.
|
"""PG configuration parameters that have to be always present in the generated config.
|
||||||
|
|
||||||
:returns: list of the parameter names.
|
:returns: list of the parameter names.
|
||||||
"""
|
"""
|
||||||
@@ -324,7 +333,7 @@ class RunningClusterConfigGenerator(AbstractConfigGenerator):
|
|||||||
:exc:`~patroni.exceptions.PatroniException`: if:
|
:exc:`~patroni.exceptions.PatroniException`: if:
|
||||||
|
|
||||||
* pid could not be obtained from the ``postmaster.pid`` file; or
|
* pid could not be obtained from the ``postmaster.pid`` file; or
|
||||||
* :exc:`OSError` occured during ``postmaster.pid`` file handling; or
|
* :exc:`OSError` occurred during ``postmaster.pid`` file handling; or
|
||||||
* the obtained postmaster pid doesn't exist.
|
* the obtained postmaster pid doesn't exist.
|
||||||
"""
|
"""
|
||||||
postmaster_pid = None
|
postmaster_pid = None
|
||||||
@@ -347,7 +356,7 @@ class RunningClusterConfigGenerator(AbstractConfigGenerator):
|
|||||||
"""Get cursor for the PG connection established based on the stored information.
|
"""Get cursor for the PG connection established based on the stored information.
|
||||||
|
|
||||||
:raises:
|
:raises:
|
||||||
:exc:`~patroni.exceptions.PatroniException`: if :exc:`psycopg.Error` occured.
|
:exc:`~patroni.exceptions.PatroniException`: if :exc:`psycopg.Error` occurred.
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
conn = psycopg.connect(dsn=self.dsn,
|
conn = psycopg.connect(dsn=self.dsn,
|
||||||
@@ -396,8 +405,9 @@ class RunningClusterConfigGenerator(AbstractConfigGenerator):
|
|||||||
else:
|
else:
|
||||||
self.config['bootstrap']['dcs']['postgresql']['parameters'][param] = value
|
self.config['bootstrap']['dcs']['postgresql']['parameters'][param] = value
|
||||||
|
|
||||||
|
connect_ip = self.config['postgresql']['connect_address'].rsplit(':')[0]
|
||||||
connect_port = self.parsed_dsn.get('port', os.getenv('PGPORT', helper_dict['port']))
|
connect_port = self.parsed_dsn.get('port', os.getenv('PGPORT', helper_dict['port']))
|
||||||
self.config['postgresql']['connect_address'] = f'{self._IP}:{connect_port}'
|
self.config['postgresql']['connect_address'] = f'{connect_ip}:{connect_port}'
|
||||||
self.config['postgresql']['listen'] = f'{helper_dict["listen_addresses"]}:{helper_dict["port"]}'
|
self.config['postgresql']['listen'] = f'{helper_dict["listen_addresses"]}:{helper_dict["port"]}'
|
||||||
|
|
||||||
def _set_su_params(self) -> None:
|
def _set_su_params(self) -> None:
|
||||||
@@ -410,8 +420,10 @@ class RunningClusterConfigGenerator(AbstractConfigGenerator):
|
|||||||
val = self.parsed_dsn.get(conn_param, os.getenv(env_var))
|
val = self.parsed_dsn.get(conn_param, os.getenv(env_var))
|
||||||
if val:
|
if val:
|
||||||
su_params[conn_param] = val
|
su_params[conn_param] = val
|
||||||
patroni_env_su_username = ((self.config.get('authentication') or {}).get('superuser') or {}).get('username')
|
patroni_env_su_username = ((self.config.get('authentication')
|
||||||
patroni_env_su_pwd = ((self.config.get('authentication') or {}).get('superuser') or {}).get('password')
|
or EMPTY_DICT).get('superuser') or EMPTY_DICT).get('username')
|
||||||
|
patroni_env_su_pwd = ((self.config.get('authentication')
|
||||||
|
or EMPTY_DICT).get('superuser') or EMPTY_DICT).get('password')
|
||||||
# because we use "username" in the config for some reason
|
# because we use "username" in the config for some reason
|
||||||
su_params['username'] = su_params.pop('user', patroni_env_su_username) or getuser()
|
su_params['username'] = su_params.pop('user', patroni_env_su_username) or getuser()
|
||||||
su_params['password'] = su_params.get('password', patroni_env_su_pwd) or \
|
su_params['password'] = su_params.get('password', patroni_env_su_pwd) or \
|
||||||
@@ -430,7 +442,7 @@ class RunningClusterConfigGenerator(AbstractConfigGenerator):
|
|||||||
are located outside of ``PGDATA`` and Patroni doesn't have write permissions for them.
|
are located outside of ``PGDATA`` and Patroni doesn't have write permissions for them.
|
||||||
|
|
||||||
:raises:
|
:raises:
|
||||||
:exc:`~patroni.exceptions.PatroniException`: if :exc:`OSError` occured during the conf files handling.
|
:exc:`~patroni.exceptions.PatroniException`: if :exc:`OSError` occurred during the conf files handling.
|
||||||
"""
|
"""
|
||||||
default_hba_path = os.path.join(self.config['postgresql']['data_dir'], 'pg_hba.conf')
|
default_hba_path = os.path.join(self.config['postgresql']['data_dir'], 'pg_hba.conf')
|
||||||
if self.config['postgresql']['parameters']['hba_file'] == default_hba_path:
|
if self.config['postgresql']['parameters']['hba_file'] == default_hba_path:
|
||||||
@@ -470,7 +482,7 @@ class RunningClusterConfigGenerator(AbstractConfigGenerator):
|
|||||||
with self._get_connection_cursor() as cur:
|
with self._get_connection_cursor() as cur:
|
||||||
self.pg_major = getattr(cur.connection, 'server_version', 0)
|
self.pg_major = getattr(cur.connection, 'server_version', 0)
|
||||||
|
|
||||||
if not parse_bool(cur.connection.info.parameter_status('is_superuser')):
|
if not parse_bool(getattr(cur.connection, 'get_parameter_status')('is_superuser')):
|
||||||
raise PatroniException('The provided user does not have superuser privilege')
|
raise PatroniException('The provided user does not have superuser privilege')
|
||||||
|
|
||||||
self._set_pg_params(cur)
|
self._set_pg_params(cur)
|
||||||
|
|||||||
+133
-100
@@ -12,12 +12,9 @@
|
|||||||
If it is also missing in the configuration file we assume that this is just a normal Patroni cluster (not Citus).
|
If it is also missing in the configuration file we assume that this is just a normal Patroni cluster (not Citus).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import click
|
|
||||||
import codecs
|
import codecs
|
||||||
import copy
|
import copy
|
||||||
import datetime
|
import datetime
|
||||||
import dateutil.parser
|
|
||||||
import dateutil.tz
|
|
||||||
import difflib
|
import difflib
|
||||||
import io
|
import io
|
||||||
import json
|
import json
|
||||||
@@ -28,32 +25,50 @@ import shutil
|
|||||||
import subprocess
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
import tempfile
|
import tempfile
|
||||||
import urllib3
|
|
||||||
import time
|
import time
|
||||||
import yaml
|
|
||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from prettytable import ALL, FRAME, PrettyTable
|
from enum import Enum
|
||||||
|
from typing import Any, Dict, Iterator, List, Optional, Tuple, TYPE_CHECKING, Union
|
||||||
from urllib.parse import urlparse
|
from urllib.parse import urlparse
|
||||||
from typing import Any, Dict, Iterator, List, Optional, Union, Tuple, TYPE_CHECKING
|
|
||||||
|
import click
|
||||||
|
import dateutil.parser
|
||||||
|
import dateutil.tz
|
||||||
|
import urllib3
|
||||||
|
import yaml
|
||||||
|
|
||||||
|
from prettytable import PrettyTable
|
||||||
|
|
||||||
|
try: # pragma: no cover
|
||||||
|
from prettytable import HRuleStyle
|
||||||
|
hrule_all = HRuleStyle.ALL
|
||||||
|
hrule_frame = HRuleStyle.FRAME
|
||||||
|
except ImportError: # pragma: no cover
|
||||||
|
from prettytable import ALL as hrule_all, FRAME as hrule_frame
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from psycopg import Cursor
|
from psycopg import Cursor
|
||||||
from psycopg2 import cursor
|
from psycopg2 import cursor
|
||||||
|
|
||||||
try:
|
try: # pragma: no cover
|
||||||
from ydiff import markup_to_pager, PatchStream # pyright: ignore [reportMissingModuleSource]
|
from ydiff import markup_to_pager # pyright: ignore [reportMissingModuleSource]
|
||||||
|
try:
|
||||||
|
from ydiff import PatchStream # pyright: ignore [reportMissingModuleSource]
|
||||||
|
except ImportError:
|
||||||
|
PatchStream = iter
|
||||||
except ImportError: # pragma: no cover
|
except ImportError: # pragma: no cover
|
||||||
from cdiff import markup_to_pager, PatchStream # pyright: ignore [reportMissingModuleSource]
|
from cdiff import markup_to_pager, PatchStream # pyright: ignore [reportMissingModuleSource]
|
||||||
|
|
||||||
from . import global_config
|
from . import global_config
|
||||||
from .config import Config
|
from .config import Config
|
||||||
from .dcs import get_dcs as _get_dcs, AbstractDCS, Cluster, Member
|
from .dcs import AbstractDCS, Cluster, get_dcs as _get_dcs, Member
|
||||||
from .exceptions import PatroniException
|
from .exceptions import PatroniException
|
||||||
from .postgresql.misc import postgres_version_to_int
|
from .postgresql.misc import postgres_version_to_int, PostgresqlRole, PostgresqlState
|
||||||
from .postgresql.mpp import get_mpp
|
from .postgresql.mpp import get_mpp
|
||||||
from .utils import cluster_as_json, patch_config, polling_loop
|
|
||||||
from .request import PatroniRequest
|
from .request import PatroniRequest
|
||||||
|
from .utils import cluster_as_json, patch_config, polling_loop
|
||||||
from .version import __version__
|
from .version import __version__
|
||||||
|
|
||||||
CONFIG_DIR_PATH = click.get_app_dir('patroni')
|
CONFIG_DIR_PATH = click.get_app_dir('patroni')
|
||||||
@@ -66,6 +81,20 @@ DCS_DEFAULTS: Dict[str, Dict[str, Any]] = {
|
|||||||
'etcd3': {'port': 2379, 'template': "etcd3:\n host: '{host}:{port}'"}}
|
'etcd3': {'port': 2379, 'template': "etcd3:\n host: '{host}:{port}'"}}
|
||||||
|
|
||||||
|
|
||||||
|
class CtlPostgresqlRole(str, Enum):
|
||||||
|
|
||||||
|
LEADER = 'leader'
|
||||||
|
PRIMARY = 'primary'
|
||||||
|
STANDBY_LEADER = 'standby-leader'
|
||||||
|
REPLICA = 'replica'
|
||||||
|
STANDBY = 'standby'
|
||||||
|
ANY = 'any'
|
||||||
|
|
||||||
|
def __repr__(self) -> str:
|
||||||
|
"""Get a string representation of a :class:`CtlPostgresqlRole` member."""
|
||||||
|
return self.value
|
||||||
|
|
||||||
|
|
||||||
class PatroniCtlException(click.ClickException):
|
class PatroniCtlException(click.ClickException):
|
||||||
"""Raised upon issues faced by ``patronictl`` utility."""
|
"""Raised upon issues faced by ``patronictl`` utility."""
|
||||||
|
|
||||||
@@ -275,7 +304,7 @@ arg_cluster_name = click.argument('cluster_name', required=False,
|
|||||||
option_default_citus_group = click.option('--group', required=False, type=int, help='Citus group',
|
option_default_citus_group = click.option('--group', required=False, type=int, help='Citus group',
|
||||||
default=lambda: _get_configuration().get('citus', {}).get('group'))
|
default=lambda: _get_configuration().get('citus', {}).get('group'))
|
||||||
option_citus_group = click.option('--group', required=False, type=int, help='Citus group')
|
option_citus_group = click.option('--group', required=False, type=int, help='Citus group')
|
||||||
role_choice = click.Choice(['leader', 'primary', 'standby-leader', 'replica', 'standby', 'any', 'master'])
|
role_choice = click.Choice([role.value for role in CtlPostgresqlRole])
|
||||||
|
|
||||||
|
|
||||||
@click.group(cls=click.Group)
|
@click.group(cls=click.Group)
|
||||||
@@ -294,7 +323,7 @@ def ctl(ctx: click.Context, config_file: str, dcs_url: Optional[str], insecure:
|
|||||||
.. note::
|
.. note::
|
||||||
Besides *dcs_url* and *insecure*, which are used to override DCS configuration section and ``ctl.insecure``
|
Besides *dcs_url* and *insecure*, which are used to override DCS configuration section and ``ctl.insecure``
|
||||||
setting, you can also override the value of ``log.level``, by default ``WARNING``, through either of these
|
setting, you can also override the value of ``log.level``, by default ``WARNING``, through either of these
|
||||||
environemnt variables:
|
environment variables:
|
||||||
* ``LOGLEVEL``
|
* ``LOGLEVEL``
|
||||||
* ``PATRONI_LOGLEVEL``
|
* ``PATRONI_LOGLEVEL``
|
||||||
* ``PATRONI_LOG_LEVEL``
|
* ``PATRONI_LOG_LEVEL``
|
||||||
@@ -303,7 +332,7 @@ def ctl(ctx: click.Context, config_file: str, dcs_url: Optional[str], insecure:
|
|||||||
:param config_file: path to the configuration file.
|
:param config_file: path to the configuration file.
|
||||||
:param dcs_url: the DCS URL in the format ``DCS://HOST:PORT``, e.g. ``etcd3://random.com:2399``. If given override
|
:param dcs_url: the DCS URL in the format ``DCS://HOST:PORT``, e.g. ``etcd3://random.com:2399``. If given override
|
||||||
whatever DCS is set in the configuration file.
|
whatever DCS is set in the configuration file.
|
||||||
:param insecure: if ``True`` allow SSL connections without client certiticates. Override what is configured through
|
:param insecure: if ``True`` allow SSL connections without client certificates. Override what is configured through
|
||||||
``ctl.insecure` in the configuration file.
|
``ctl.insecure` in the configuration file.
|
||||||
"""
|
"""
|
||||||
level = 'WARNING'
|
level = 'WARNING'
|
||||||
@@ -325,6 +354,10 @@ def is_citus_cluster() -> bool:
|
|||||||
return click.get_current_context().obj['__mpp'].is_enabled()
|
return click.get_current_context().obj['__mpp'].is_enabled()
|
||||||
|
|
||||||
|
|
||||||
|
# Cache DCS instances for given scope and group
|
||||||
|
__dcs_cache: Dict[Tuple[str, Optional[int]], AbstractDCS] = {}
|
||||||
|
|
||||||
|
|
||||||
def get_dcs(scope: str, group: Optional[int]) -> AbstractDCS:
|
def get_dcs(scope: str, group: Optional[int]) -> AbstractDCS:
|
||||||
"""Get the DCS object.
|
"""Get the DCS object.
|
||||||
|
|
||||||
@@ -338,6 +371,8 @@ def get_dcs(scope: str, group: Optional[int]) -> AbstractDCS:
|
|||||||
:raises:
|
:raises:
|
||||||
:class:`PatroniCtlException`: if not suitable DCS configuration could be found.
|
:class:`PatroniCtlException`: if not suitable DCS configuration could be found.
|
||||||
"""
|
"""
|
||||||
|
if (scope, group) in __dcs_cache:
|
||||||
|
return __dcs_cache[(scope, group)]
|
||||||
config = _get_configuration()
|
config = _get_configuration()
|
||||||
config.update({'scope': scope, 'patronictl': True})
|
config.update({'scope': scope, 'patronictl': True})
|
||||||
if group is not None:
|
if group is not None:
|
||||||
@@ -348,6 +383,7 @@ def get_dcs(scope: str, group: Optional[int]) -> AbstractDCS:
|
|||||||
if is_citus_cluster() and group is None:
|
if is_citus_cluster() and group is None:
|
||||||
dcs.is_mpp_coordinator = lambda: True
|
dcs.is_mpp_coordinator = lambda: True
|
||||||
click.get_current_context().obj['__mpp'] = dcs.mpp
|
click.get_current_context().obj['__mpp'] = dcs.mpp
|
||||||
|
__dcs_cache[(scope, group)] = dcs
|
||||||
return dcs
|
return dcs
|
||||||
except PatroniException as e:
|
except PatroniException as e:
|
||||||
raise PatroniCtlException(str(e))
|
raise PatroniCtlException(str(e))
|
||||||
@@ -423,7 +459,7 @@ def print_output(columns: Optional[List[str]], rows: List[List[Any]], alignment:
|
|||||||
else:
|
else:
|
||||||
# If any value is multi-line, then add horizontal between all table rows while printing to get a clear
|
# If any value is multi-line, then add horizontal between all table rows while printing to get a clear
|
||||||
# visual separation of rows.
|
# visual separation of rows.
|
||||||
hrules = ALL if any(any(isinstance(c, str) and '\n' in c for c in r) for r in rows) else FRAME
|
hrules = hrule_all if any(any(isinstance(c, str) and '\n' in c for c in r) for r in rows) else hrule_frame
|
||||||
table = PatronictlPrettyTable(header, columns, hrules=hrules)
|
table = PatronictlPrettyTable(header, columns, hrules=hrules)
|
||||||
table.align = 'l'
|
table.align = 'l'
|
||||||
for k, v in (alignment or {}).items():
|
for k, v in (alignment or {}).items():
|
||||||
@@ -472,46 +508,42 @@ def watching(w: bool, watch: Optional[int], max_count: Optional[int] = None, cle
|
|||||||
yield 0
|
yield 0
|
||||||
|
|
||||||
|
|
||||||
def get_all_members(cluster: Cluster, group: Optional[int], role: str = 'leader') -> Iterator[Member]:
|
def get_all_members(cluster: Cluster, group: Optional[int],
|
||||||
|
role: CtlPostgresqlRole = CtlPostgresqlRole.LEADER) -> Iterator[Member]:
|
||||||
"""Get all cluster members that have the given *role*.
|
"""Get all cluster members that have the given *role*.
|
||||||
|
|
||||||
:param cluster: the Patroni cluster.
|
:param cluster: the Patroni cluster.
|
||||||
:param group: filter which Citus group we should get members from. If ``None`` get from all groups.
|
:param group: filter which Citus group we should get members from. If ``None`` get from all groups.
|
||||||
:param role: role to filter members. Can be one among:
|
:param role: role to filter members. One of :class:`CtlPostgresqlRole` values.
|
||||||
|
|
||||||
* ``primary`` or ``master``: the primary PostgreSQL instance;
|
|
||||||
* ``replica`` or ``standby``: a standby PostgreSQL instance;
|
|
||||||
* ``leader``: the leader of a Patroni cluster. Can also be used to get the leader of a Patroni standby cluster;
|
|
||||||
* ``standby-leader``: the leader of a Patroni standby cluster;
|
|
||||||
* ``any``: matches any node independent of its role.
|
|
||||||
|
|
||||||
:yields: members that have the given *role*.
|
:yields: members that have the given *role*.
|
||||||
"""
|
"""
|
||||||
clusters = {0: cluster}
|
clusters = {0: cluster}
|
||||||
if is_citus_cluster() and group is None:
|
if is_citus_cluster() and group is None:
|
||||||
clusters.update(cluster.workers)
|
clusters.update(cluster.workers)
|
||||||
if role in ('leader', 'master', 'primary', 'standby-leader'):
|
if role in (CtlPostgresqlRole.LEADER, CtlPostgresqlRole.PRIMARY, CtlPostgresqlRole.STANDBY_LEADER):
|
||||||
# In the DCS the members' role can be one among: ``primary``, ``master``, ``replica`` or ``standby_leader``.
|
# In the DCS the members' role can be one among: ``primary``, ``master``, ``replica`` or ``standby_leader``.
|
||||||
# ``primary`` and ``master`` are the same thing, so we map both to ``master`` to have a simpler ``if``.
|
# ``primary`` and ``master`` are the same thing.
|
||||||
# In a future release we might remove ``master`` from the available roles for the DCS members.
|
|
||||||
role = {'primary': 'master', 'standby-leader': 'standby_leader'}.get(role, role)
|
|
||||||
for cluster in clusters.values():
|
for cluster in clusters.values():
|
||||||
if cluster.leader is not None and cluster.leader.name and\
|
if cluster.leader is not None and cluster.leader.name and\
|
||||||
(role == 'leader'
|
(role == CtlPostgresqlRole.LEADER
|
||||||
or cluster.leader.data.get('role') != 'master' and role == 'standby_leader'
|
or cluster.leader.data.get('role') not in (PostgresqlRole.PRIMARY, PostgresqlRole.MASTER)
|
||||||
or cluster.leader.data.get('role') != 'standby_leader' and role == 'master'):
|
and role == CtlPostgresqlRole.STANDBY_LEADER
|
||||||
|
or cluster.leader.data.get('role') != PostgresqlRole.STANDBY_LEADER
|
||||||
|
and role == CtlPostgresqlRole.PRIMARY):
|
||||||
yield cluster.leader.member
|
yield cluster.leader.member
|
||||||
return
|
return
|
||||||
|
|
||||||
for cluster in clusters.values():
|
for cluster in clusters.values():
|
||||||
leader_name = (cluster.leader.member.name if cluster.leader else None)
|
leader_name = (cluster.leader.member.name if cluster.leader else None)
|
||||||
for m in cluster.members:
|
for m in cluster.members:
|
||||||
if role == 'any' or role in ('replica', 'standby') and m.name != leader_name:
|
if role == CtlPostgresqlRole.ANY or\
|
||||||
|
role in (CtlPostgresqlRole.REPLICA, CtlPostgresqlRole.STANDBY) and m.name != leader_name:
|
||||||
yield m
|
yield m
|
||||||
|
|
||||||
|
|
||||||
def get_any_member(cluster: Cluster, group: Optional[int],
|
def get_any_member(cluster: Cluster, group: Optional[int],
|
||||||
role: Optional[str] = None, member: Optional[str] = None) -> Optional[Member]:
|
role: Optional[CtlPostgresqlRole] = None, member: Optional[str] = None) -> Optional[Member]:
|
||||||
"""Get the first found cluster member that has the given *role*.
|
"""Get the first found cluster member that has the given *role*.
|
||||||
|
|
||||||
:param cluster: the Patroni cluster.
|
:param cluster: the Patroni cluster.
|
||||||
@@ -527,9 +559,9 @@ def get_any_member(cluster: Cluster, group: Optional[int],
|
|||||||
if member is not None:
|
if member is not None:
|
||||||
if role is not None:
|
if role is not None:
|
||||||
raise PatroniCtlException('--role and --member are mutually exclusive options')
|
raise PatroniCtlException('--role and --member are mutually exclusive options')
|
||||||
role = 'any'
|
role = CtlPostgresqlRole.ANY
|
||||||
elif role is None:
|
elif role is None:
|
||||||
role = 'leader'
|
role = CtlPostgresqlRole.LEADER
|
||||||
|
|
||||||
for m in get_all_members(cluster, group, role):
|
for m in get_all_members(cluster, group, role):
|
||||||
if member is None or m.name == member:
|
if member is None or m.name == member:
|
||||||
@@ -553,7 +585,8 @@ def get_all_members_leader_first(cluster: Cluster) -> Iterator[Member]:
|
|||||||
|
|
||||||
|
|
||||||
def get_cursor(cluster: Cluster, group: Optional[int], connect_parameters: Dict[str, Any],
|
def get_cursor(cluster: Cluster, group: Optional[int], connect_parameters: Dict[str, Any],
|
||||||
role: Optional[str] = None, member_name: Optional[str] = None) -> Union['cursor', 'Cursor[Any]', None]:
|
role: Optional[CtlPostgresqlRole] = None,
|
||||||
|
member_name: Optional[str] = None) -> Union['cursor', 'Cursor[Any]', None]:
|
||||||
"""Get a cursor object to execute queries against a member that has the given *role* or *member_name*.
|
"""Get a cursor object to execute queries against a member that has the given *role* or *member_name*.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
@@ -592,7 +625,7 @@ def get_cursor(cluster: Cluster, group: Optional[int], connect_parameters: Dict[
|
|||||||
# If we want ``any`` node we are fine to return the cursor. ``None`` is similar to ``any`` at this point, as it's
|
# If we want ``any`` node we are fine to return the cursor. ``None`` is similar to ``any`` at this point, as it's
|
||||||
# been dealt with through :func:`get_any_member`.
|
# been dealt with through :func:`get_any_member`.
|
||||||
# If we want the Patroni leader node, :func:`get_any_member` already checks that for us
|
# If we want the Patroni leader node, :func:`get_any_member` already checks that for us
|
||||||
if role in (None, 'any', 'leader'):
|
if role in (None, CtlPostgresqlRole.ANY, CtlPostgresqlRole.LEADER):
|
||||||
return cursor
|
return cursor
|
||||||
|
|
||||||
# If we want something other than ``any`` or ``leader``, then we do not rely only on the DCS information about
|
# If we want something other than ``any`` or ``leader``, then we do not rely only on the DCS information about
|
||||||
@@ -601,8 +634,9 @@ def get_cursor(cluster: Cluster, group: Optional[int], connect_parameters: Dict[
|
|||||||
row = cursor.fetchone()
|
row = cursor.fetchone()
|
||||||
in_recovery = not row or row[0]
|
in_recovery = not row or row[0]
|
||||||
|
|
||||||
if in_recovery and role in ('replica', 'standby', 'standby-leader')\
|
if in_recovery and\
|
||||||
or not in_recovery and role in ('master', 'primary'):
|
role in (CtlPostgresqlRole.REPLICA, CtlPostgresqlRole.STANDBY, CtlPostgresqlRole.STANDBY_LEADER) or\
|
||||||
|
not in_recovery and role == CtlPostgresqlRole.PRIMARY:
|
||||||
return cursor
|
return cursor
|
||||||
|
|
||||||
conn.close()
|
conn.close()
|
||||||
@@ -610,7 +644,7 @@ def get_cursor(cluster: Cluster, group: Optional[int], connect_parameters: Dict[
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def get_members(cluster: Cluster, cluster_name: str, member_names: List[str], role: str,
|
def get_members(cluster: Cluster, cluster_name: str, member_names: List[str], role: CtlPostgresqlRole,
|
||||||
force: bool, action: str, ask_confirmation: bool = True, group: Optional[int] = None) -> List[Member]:
|
force: bool, action: str, ask_confirmation: bool = True, group: Optional[int] = None) -> List[Member]:
|
||||||
"""Get the list of members based on the given filters.
|
"""Get the list of members based on the given filters.
|
||||||
|
|
||||||
@@ -692,7 +726,7 @@ def get_members(cluster: Cluster, cluster_name: str, member_names: List[str], ro
|
|||||||
|
|
||||||
|
|
||||||
def confirm_members_action(members: List[Member], force: bool, action: str,
|
def confirm_members_action(members: List[Member], force: bool, action: str,
|
||||||
scheduled_at: Optional[datetime.datetime] = None) -> None:
|
scheduled_at: Optional[datetime.datetime] = None, pending: Optional[bool] = None) -> None:
|
||||||
"""Ask for confirmation if *action* should be taken by *members*.
|
"""Ask for confirmation if *action* should be taken by *members*.
|
||||||
|
|
||||||
:param members: list of member which will take the *action*.
|
:param members: list of member which will take the *action*.
|
||||||
@@ -711,8 +745,13 @@ def confirm_members_action(members: List[Member], force: bool, action: str,
|
|||||||
"""
|
"""
|
||||||
if scheduled_at:
|
if scheduled_at:
|
||||||
if not force:
|
if not force:
|
||||||
confirm = click.confirm('Are you sure you want to schedule {0} of members {1} at {2}?'
|
if pending:
|
||||||
.format(action, ', '.join([m.name for m in members]), scheduled_at))
|
confirm = click.confirm('The nodes needing a restart will be identified at the scheduled time, and '
|
||||||
|
'might be different from the ones showing pending restart right now. '
|
||||||
|
'Is this fine?')
|
||||||
|
else:
|
||||||
|
confirm = click.confirm('Are you sure you want to schedule {0} of members {1} at {2}?'
|
||||||
|
.format(action, ', '.join([m.name for m in members]), scheduled_at))
|
||||||
if not confirm:
|
if not confirm:
|
||||||
raise PatroniCtlException('Aborted scheduled {0}'.format(action))
|
raise PatroniCtlException('Aborted scheduled {0}'.format(action))
|
||||||
else:
|
else:
|
||||||
@@ -728,7 +767,7 @@ def confirm_members_action(members: List[Member], force: bool, action: str,
|
|||||||
@click.option('--member', '-m', help='Generate a dsn for this member', type=str)
|
@click.option('--member', '-m', help='Generate a dsn for this member', type=str)
|
||||||
@arg_cluster_name
|
@arg_cluster_name
|
||||||
@option_citus_group
|
@option_citus_group
|
||||||
def dsn(cluster_name: str, group: Optional[int], role: Optional[str], member: Optional[str]) -> None:
|
def dsn(cluster_name: str, group: Optional[int], role: Optional[CtlPostgresqlRole], member: Optional[str]) -> None:
|
||||||
"""Process ``dsn`` command of ``patronictl`` utility.
|
"""Process ``dsn`` command of ``patronictl`` utility.
|
||||||
|
|
||||||
Get DSN to connect to *member*.
|
Get DSN to connect to *member*.
|
||||||
@@ -774,7 +813,7 @@ def dsn(cluster_name: str, group: Optional[int], role: Optional[str], member: Op
|
|||||||
def query(
|
def query(
|
||||||
cluster_name: str,
|
cluster_name: str,
|
||||||
group: Optional[int],
|
group: Optional[int],
|
||||||
role: Optional[str],
|
role: Optional[CtlPostgresqlRole],
|
||||||
member: Optional[str],
|
member: Optional[str],
|
||||||
w: bool,
|
w: bool,
|
||||||
watch: Optional[int],
|
watch: Optional[int],
|
||||||
@@ -841,7 +880,7 @@ def query(
|
|||||||
|
|
||||||
|
|
||||||
def query_member(cluster: Cluster, group: Optional[int], cursor: Union['cursor', 'Cursor[Any]', None],
|
def query_member(cluster: Cluster, group: Optional[int], cursor: Union['cursor', 'Cursor[Any]', None],
|
||||||
member: Optional[str], role: Optional[str], command: str,
|
member: Optional[str], role: Optional[CtlPostgresqlRole], command: str,
|
||||||
connect_parameters: Dict[str, Any]) -> Tuple[List[List[Any]], Optional[List[Any]]]:
|
connect_parameters: Dict[str, Any]) -> Tuple[List[List[Any]], Optional[List[Any]]]:
|
||||||
"""Execute SQL *command* against a member.
|
"""Execute SQL *command* against a member.
|
||||||
|
|
||||||
@@ -879,7 +918,7 @@ def query_member(cluster: Cluster, group: Optional[int], cursor: Union['cursor',
|
|||||||
if member is not None:
|
if member is not None:
|
||||||
message = f'No connection to member {member} is available'
|
message = f'No connection to member {member} is available'
|
||||||
elif role is not None:
|
elif role is not None:
|
||||||
message = f'No connection to role {role} is available'
|
message = f'No connection to role {role!r} is available'
|
||||||
else:
|
else:
|
||||||
message = 'No connection is available'
|
message = 'No connection is available'
|
||||||
logging.debug(message)
|
logging.debug(message)
|
||||||
@@ -953,7 +992,7 @@ def check_response(response: urllib3.response.HTTPResponse, member_name: str,
|
|||||||
:param action_name: action associated with the *response*.
|
:param action_name: action associated with the *response*.
|
||||||
:param silent_success: if a status message should be skipped upon a successful *response*.
|
:param silent_success: if a status message should be skipped upon a successful *response*.
|
||||||
|
|
||||||
:returns: ``True`` if the response indicates a sucessful operation (HTTP status < ``400``), ``False`` otherwise.
|
:returns: ``True`` if the response indicates a successful operation (HTTP status < ``400``), ``False`` otherwise.
|
||||||
"""
|
"""
|
||||||
if response.status >= 400:
|
if response.status >= 400:
|
||||||
click.echo('Failed: {0} for member {1}, status code={2}, ({3})'.format(
|
click.echo('Failed: {0} for member {1}, status code={2}, ({3})'.format(
|
||||||
@@ -1006,9 +1045,11 @@ def parse_scheduled(scheduled: Optional[str]) -> Optional[datetime.datetime]:
|
|||||||
@click.argument('cluster_name')
|
@click.argument('cluster_name')
|
||||||
@click.argument('member_names', nargs=-1)
|
@click.argument('member_names', nargs=-1)
|
||||||
@option_citus_group
|
@option_citus_group
|
||||||
@click.option('--role', '-r', help='Reload only members with this role', type=role_choice, default='any')
|
@click.option('--role', '-r', help='Reload only members with this role',
|
||||||
|
type=role_choice, default=CtlPostgresqlRole.ANY)
|
||||||
@option_force
|
@option_force
|
||||||
def reload(cluster_name: str, member_names: List[str], group: Optional[int], force: bool, role: str) -> None:
|
def reload(cluster_name: str, member_names: List[str], group: Optional[int],
|
||||||
|
force: bool, role: CtlPostgresqlRole) -> None:
|
||||||
"""Process ``reload`` command of ``patronictl`` utility.
|
"""Process ``reload`` command of ``patronictl`` utility.
|
||||||
|
|
||||||
Reload configuration of cluster members based on given filters.
|
Reload configuration of cluster members based on given filters.
|
||||||
@@ -1043,7 +1084,8 @@ def reload(cluster_name: str, member_names: List[str], group: Optional[int], for
|
|||||||
@click.argument('cluster_name')
|
@click.argument('cluster_name')
|
||||||
@click.argument('member_names', nargs=-1)
|
@click.argument('member_names', nargs=-1)
|
||||||
@option_citus_group
|
@option_citus_group
|
||||||
@click.option('--role', '-r', help='Restart only members with this role', type=role_choice, default='any')
|
@click.option('--role', '-r', help='Restart only members with this role', type=role_choice,
|
||||||
|
default=CtlPostgresqlRole.ANY)
|
||||||
@click.option('--any', 'p_any', help='Restart a single member only', is_flag=True)
|
@click.option('--any', 'p_any', help='Restart a single member only', is_flag=True)
|
||||||
@click.option('--scheduled', help='Timestamp of a scheduled restart in unambiguous format (e.g. ISO 8601)',
|
@click.option('--scheduled', help='Timestamp of a scheduled restart in unambiguous format (e.g. ISO 8601)',
|
||||||
default=None)
|
default=None)
|
||||||
@@ -1053,7 +1095,7 @@ def reload(cluster_name: str, member_names: List[str], group: Optional[int], for
|
|||||||
@click.option('--timeout', help='Return error and fail over if necessary when restarting takes longer than this.')
|
@click.option('--timeout', help='Return error and fail over if necessary when restarting takes longer than this.')
|
||||||
@option_force
|
@option_force
|
||||||
def restart(cluster_name: str, group: Optional[int], member_names: List[str],
|
def restart(cluster_name: str, group: Optional[int], member_names: List[str],
|
||||||
force: bool, role: str, p_any: bool, scheduled: Optional[str], version: Optional[str],
|
force: bool, role: CtlPostgresqlRole, p_any: bool, scheduled: Optional[str], version: Optional[str],
|
||||||
pending: bool, timeout: Optional[str]) -> None:
|
pending: bool, timeout: Optional[str]) -> None:
|
||||||
"""Process ``restart`` command of ``patronictl`` utility.
|
"""Process ``restart`` command of ``patronictl`` utility.
|
||||||
|
|
||||||
@@ -1085,20 +1127,25 @@ def restart(cluster_name: str, group: Optional[int], member_names: List[str],
|
|||||||
type=str, default='now')
|
type=str, default='now')
|
||||||
|
|
||||||
scheduled_at = parse_scheduled(scheduled)
|
scheduled_at = parse_scheduled(scheduled)
|
||||||
confirm_members_action(members, force, 'restart', scheduled_at)
|
|
||||||
|
content: Dict[str, Any] = {}
|
||||||
|
if pending:
|
||||||
|
content['restart_pending'] = True
|
||||||
|
# For scheduled restarts we don't filter the members now. If they really need a restart might change
|
||||||
|
# until the scheduled time.
|
||||||
|
if not scheduled_at:
|
||||||
|
members = [m for m in members if m.data.get('pending_restart', False)]
|
||||||
|
|
||||||
if p_any:
|
if p_any:
|
||||||
random.shuffle(members)
|
random.shuffle(members)
|
||||||
members = members[:1]
|
members = members[:1]
|
||||||
|
|
||||||
|
confirm_members_action(members, force, 'restart', scheduled_at, pending)
|
||||||
|
|
||||||
if version is None and not force:
|
if version is None and not force:
|
||||||
version = click.prompt('Restart if the PostgreSQL version is less than provided (e.g. 9.5.2) ',
|
version = click.prompt('Restart if the PostgreSQL version is less than provided (e.g. 9.5.2) ',
|
||||||
type=str, default='')
|
type=str, default='')
|
||||||
|
|
||||||
content: Dict[str, Any] = {}
|
|
||||||
if pending:
|
|
||||||
content['restart_pending'] = True
|
|
||||||
|
|
||||||
if version:
|
if version:
|
||||||
try:
|
try:
|
||||||
postgres_version_to_int(version)
|
postgres_version_to_int(version)
|
||||||
@@ -1155,7 +1202,8 @@ def reinit(cluster_name: str, group: Optional[int], member_names: List[str], for
|
|||||||
:param wait: wait for the operation to complete.
|
:param wait: wait for the operation to complete.
|
||||||
"""
|
"""
|
||||||
cluster = get_dcs(cluster_name, group).get_cluster()
|
cluster = get_dcs(cluster_name, group).get_cluster()
|
||||||
members = get_members(cluster, cluster_name, member_names, 'replica', force, 'reinitialize', group=group)
|
members = get_members(cluster, cluster_name, member_names, CtlPostgresqlRole.REPLICA,
|
||||||
|
force, 'reinitialize', group=group)
|
||||||
|
|
||||||
wait_on_members: List[Member] = []
|
wait_on_members: List[Member] = []
|
||||||
for member in members:
|
for member in members:
|
||||||
@@ -1181,14 +1229,14 @@ def reinit(cluster_name: str, group: Optional[int], member_names: List[str], for
|
|||||||
time.sleep(2)
|
time.sleep(2)
|
||||||
for member in wait_on_members:
|
for member in wait_on_members:
|
||||||
data = json.loads(request_patroni(member, 'get', 'patroni').data.decode('utf-8'))
|
data = json.loads(request_patroni(member, 'get', 'patroni').data.decode('utf-8'))
|
||||||
if data.get('state') != 'creating replica':
|
if data.get('state') != PostgresqlState.CREATING_REPLICA:
|
||||||
click.echo('Reinitialize is completed on: {0}'.format(member.name))
|
click.echo('Reinitialize is completed on: {0}'.format(member.name))
|
||||||
wait_on_members.remove(member)
|
wait_on_members.remove(member)
|
||||||
|
|
||||||
|
|
||||||
def _do_failover_or_switchover(action: str, cluster_name: str, group: Optional[int],
|
def _do_failover_or_switchover(action: str, cluster_name: str, group: Optional[int], candidate: Optional[str],
|
||||||
switchover_leader: Optional[str], candidate: Optional[str],
|
force: bool, switchover_leader: Optional[str] = None,
|
||||||
force: bool, scheduled: Optional[str] = None) -> None:
|
switchover_scheduled: Optional[str] = None) -> None:
|
||||||
"""Perform a failover or a switchover operation in the cluster.
|
"""Perform a failover or a switchover operation in the cluster.
|
||||||
|
|
||||||
Informational messages are printed in the console during the operation, as well as the list of members before and
|
Informational messages are printed in the console during the operation, as well as the list of members before and
|
||||||
@@ -1201,10 +1249,11 @@ def _do_failover_or_switchover(action: str, cluster_name: str, group: Optional[i
|
|||||||
:param cluster_name: name of the Patroni cluster.
|
:param cluster_name: name of the Patroni cluster.
|
||||||
:param group: filter Citus group within we should perform a failover or switchover. If ``None``, user will be
|
:param group: filter Citus group within we should perform a failover or switchover. If ``None``, user will be
|
||||||
prompted for filling it -- unless *force* is ``True``, in which case an exception is raised.
|
prompted for filling it -- unless *force* is ``True``, in which case an exception is raised.
|
||||||
:param switchover_leader: name of the leader member passed as switchover option.
|
|
||||||
:param candidate: name of a standby member to be promoted. Nodes that are tagged with ``nofailover`` cannot be used.
|
:param candidate: name of a standby member to be promoted. Nodes that are tagged with ``nofailover`` cannot be used.
|
||||||
:param force: perform the failover or switchover without asking for confirmations.
|
:param force: perform the failover or switchover without asking for confirmations.
|
||||||
:param scheduled: timestamp when the switchover should be scheduled to occur. If ``now`` perform immediately.
|
:param switchover_leader: name of the leader passed to the switchover command if any.
|
||||||
|
:param switchover_scheduled: timestamp when the switchover should be scheduled to occur. If ``now``,
|
||||||
|
perform immediately.
|
||||||
|
|
||||||
:raises:
|
:raises:
|
||||||
:class:`PatroniCtlException`: if:
|
:class:`PatroniCtlException`: if:
|
||||||
@@ -1283,12 +1332,12 @@ def _do_failover_or_switchover(action: str, cluster_name: str, group: Optional[i
|
|||||||
scheduled_at = None
|
scheduled_at = None
|
||||||
|
|
||||||
if action == 'switchover':
|
if action == 'switchover':
|
||||||
if scheduled is None and not force:
|
if switchover_scheduled is None and not force:
|
||||||
next_hour = (datetime.datetime.now() + datetime.timedelta(hours=1)).strftime('%Y-%m-%dT%H:%M')
|
next_hour = (datetime.datetime.now() + datetime.timedelta(hours=1)).strftime('%Y-%m-%dT%H:%M')
|
||||||
scheduled = click.prompt('When should the switchover take place (e.g. ' + next_hour + ' ) ',
|
switchover_scheduled = click.prompt('When should the switchover take place (e.g. ' + next_hour + ' ) ',
|
||||||
type=str, default='now')
|
type=str, default='now')
|
||||||
|
|
||||||
scheduled_at = parse_scheduled(scheduled)
|
scheduled_at = parse_scheduled(switchover_scheduled)
|
||||||
if scheduled_at:
|
if scheduled_at:
|
||||||
if config.is_paused:
|
if config.is_paused:
|
||||||
raise PatroniCtlException("Can't schedule switchover in the paused state")
|
raise PatroniCtlException("Can't schedule switchover in the paused state")
|
||||||
@@ -1346,20 +1395,12 @@ def _do_failover_or_switchover(action: str, cluster_name: str, group: Optional[i
|
|||||||
@ctl.command('failover', help='Failover to a replica')
|
@ctl.command('failover', help='Failover to a replica')
|
||||||
@arg_cluster_name
|
@arg_cluster_name
|
||||||
@option_citus_group
|
@option_citus_group
|
||||||
@click.option('--leader', '--primary', '--master', 'leader', help='The name of the current leader', default=None)
|
|
||||||
@click.option('--candidate', help='The name of the candidate', default=None)
|
@click.option('--candidate', help='The name of the candidate', default=None)
|
||||||
@option_force
|
@option_force
|
||||||
def failover(cluster_name: str, group: Optional[int],
|
def failover(cluster_name: str, group: Optional[int], candidate: Optional[str], force: bool) -> None:
|
||||||
leader: Optional[str], candidate: Optional[str], force: bool) -> None:
|
|
||||||
"""Process ``failover`` command of ``patronictl`` utility.
|
"""Process ``failover`` command of ``patronictl`` utility.
|
||||||
|
|
||||||
Perform a failover operation immediately in the cluster.
|
Perform a failover operation immediately in the cluster.
|
||||||
|
|
||||||
.. note::
|
|
||||||
If *leader* is given perform a switchover instead of a failover.
|
|
||||||
This behavior is deprecated. ``--leader`` option support will be
|
|
||||||
removed in the next major release.
|
|
||||||
|
|
||||||
.. seealso::
|
.. seealso::
|
||||||
Refer to :func:`_do_failover_or_switchover` for details.
|
Refer to :func:`_do_failover_or_switchover` for details.
|
||||||
|
|
||||||
@@ -1367,23 +1408,16 @@ def failover(cluster_name: str, group: Optional[int],
|
|||||||
:param group: filter Citus group within we should perform a failover or switchover. If ``None``, user will be
|
:param group: filter Citus group within we should perform a failover or switchover. If ``None``, user will be
|
||||||
prompted for filling it -- unless *force* is ``True``, in which case an exception is raised by
|
prompted for filling it -- unless *force* is ``True``, in which case an exception is raised by
|
||||||
:func:`_do_failover_or_switchover`.
|
:func:`_do_failover_or_switchover`.
|
||||||
:param leader: name of the current leader member.
|
|
||||||
:param candidate: name of a standby member to be promoted. Nodes that are tagged with ``nofailover`` cannot be used.
|
:param candidate: name of a standby member to be promoted. Nodes that are tagged with ``nofailover`` cannot be used.
|
||||||
:param force: perform the failover or switchover without asking for confirmations.
|
:param force: perform the failover or switchover without asking for confirmations.
|
||||||
"""
|
"""
|
||||||
action = 'failover'
|
_do_failover_or_switchover('failover', cluster_name, group, candidate, force)
|
||||||
if leader:
|
|
||||||
action = 'switchover'
|
|
||||||
click.echo(click.style(
|
|
||||||
'Supplying a leader name using this command is deprecated and will be removed in a future version of'
|
|
||||||
' Patroni, change your scripts to use `switchover` instead.\nExecuting switchover!', fg='red'))
|
|
||||||
_do_failover_or_switchover(action, cluster_name, group, leader, candidate, force)
|
|
||||||
|
|
||||||
|
|
||||||
@ctl.command('switchover', help='Switchover to a replica')
|
@ctl.command('switchover', help='Switchover to a replica')
|
||||||
@arg_cluster_name
|
@arg_cluster_name
|
||||||
@option_citus_group
|
@option_citus_group
|
||||||
@click.option('--leader', '--primary', '--master', 'leader', help='The name of the current leader', default=None)
|
@click.option('--leader', '--primary', 'leader', help='The name of the current leader', default=None)
|
||||||
@click.option('--candidate', help='The name of the candidate', default=None)
|
@click.option('--candidate', help='The name of the candidate', default=None)
|
||||||
@click.option('--scheduled', help='Timestamp of a scheduled switchover in unambiguous format (e.g. ISO 8601)',
|
@click.option('--scheduled', help='Timestamp of a scheduled switchover in unambiguous format (e.g. ISO 8601)',
|
||||||
default=None)
|
default=None)
|
||||||
@@ -1406,7 +1440,7 @@ def switchover(cluster_name: str, group: Optional[int], leader: Optional[str],
|
|||||||
:param force: perform the switchover without asking for confirmations.
|
:param force: perform the switchover without asking for confirmations.
|
||||||
:param scheduled: timestamp when the switchover should be scheduled to occur. If ``now`` perform immediately.
|
:param scheduled: timestamp when the switchover should be scheduled to occur. If ``now`` perform immediately.
|
||||||
"""
|
"""
|
||||||
_do_failover_or_switchover('switchover', cluster_name, group, leader, candidate, force, scheduled)
|
_do_failover_or_switchover('switchover', cluster_name, group, candidate, force, leader, scheduled)
|
||||||
|
|
||||||
|
|
||||||
def generate_topology(level: int, member: Dict[str, Any],
|
def generate_topology(level: int, member: Dict[str, Any],
|
||||||
@@ -1431,7 +1465,7 @@ def generate_topology(level: int, member: Dict[str, Any],
|
|||||||
+ postgresql2
|
+ postgresql2
|
||||||
|
|
||||||
:param level: the current level being inspected in the *topology*.
|
:param level: the current level being inspected in the *topology*.
|
||||||
:param member: information about the current member being inspected in *level* of *topology*. Should countain at
|
:param member: information about the current member being inspected in *level* of *topology*. Should contain at
|
||||||
least this key:
|
least this key:
|
||||||
* ``name``: name of the node, according to ``name`` configuration;
|
* ``name``: name of the node, according to ``name`` configuration;
|
||||||
|
|
||||||
@@ -1458,7 +1492,7 @@ def generate_topology(level: int, member: Dict[str, Any],
|
|||||||
def topology_sort(members: List[Dict[str, Any]]) -> Iterator[Dict[str, Any]]:
|
def topology_sort(members: List[Dict[str, Any]]) -> Iterator[Dict[str, Any]]:
|
||||||
"""Sort *members* according to their level in the replication topology tree.
|
"""Sort *members* according to their level in the replication topology tree.
|
||||||
|
|
||||||
:param members: list of members in the cluster. Each item should countain at least these keys:
|
:param members: list of members in the cluster. Each item should contain at least these keys:
|
||||||
|
|
||||||
* ``name``: name of the node, according to ``name`` configuration;
|
* ``name``: name of the node, according to ``name`` configuration;
|
||||||
* ``role``: ``leader``, ``standby_leader`` or ``replica``.
|
* ``role``: ``leader``, ``standby_leader`` or ``replica``.
|
||||||
@@ -1518,10 +1552,7 @@ def output_members(cluster: Cluster, name: str, extended: bool = False,
|
|||||||
* ``Member``: name of the Patroni node, as per ``name`` configuration;
|
* ``Member``: name of the Patroni node, as per ``name`` configuration;
|
||||||
* ``Host``: hostname (or IP) and port, as per ``postgresql.listen`` configuration;
|
* ``Host``: hostname (or IP) and port, as per ``postgresql.listen`` configuration;
|
||||||
* ``Role``: ``Leader``, ``Standby Leader``, ``Sync Standby`` or ``Replica``;
|
* ``Role``: ``Leader``, ``Standby Leader``, ``Sync Standby`` or ``Replica``;
|
||||||
* ``State``: ``stopping``, ``stopped``, ``stop failed``, ``crashed``, ``running``, ``starting``,
|
* ``State``: one of :class:`~patroni.postgresql.misc.PostgresqlState`;
|
||||||
``start failed``, ``restarting``, ``restart failed``, ``initializing new cluster``, ``initdb failed``,
|
|
||||||
``running custom bootstrap script``, ``custom bootstrap failed``, ``creating replica``, ``streaming``,
|
|
||||||
``in archive recovery``, and so on;
|
|
||||||
* ``TL``: current timeline in Postgres;
|
* ``TL``: current timeline in Postgres;
|
||||||
``Lag in MB``: replication lag.
|
``Lag in MB``: replication lag.
|
||||||
|
|
||||||
@@ -1697,10 +1728,10 @@ def timestamp(precision: int = 6) -> str:
|
|||||||
@option_citus_group
|
@option_citus_group
|
||||||
@click.argument('member_names', nargs=-1)
|
@click.argument('member_names', nargs=-1)
|
||||||
@click.argument('target', type=click.Choice(['restart', 'switchover']))
|
@click.argument('target', type=click.Choice(['restart', 'switchover']))
|
||||||
@click.option('--role', '-r', help='Flush only members with this role', type=role_choice, default='any')
|
@click.option('--role', '-r', help='Flush only members with this role', type=role_choice, default=CtlPostgresqlRole.ANY)
|
||||||
@option_force
|
@option_force
|
||||||
def flush(cluster_name: str, group: Optional[int],
|
def flush(cluster_name: str, group: Optional[int],
|
||||||
member_names: List[str], force: bool, role: str, target: str) -> None:
|
member_names: List[str], force: bool, role: CtlPostgresqlRole, target: str) -> None:
|
||||||
"""Process ``flush`` command of ``patronictl`` utility.
|
"""Process ``flush`` command of ``patronictl`` utility.
|
||||||
|
|
||||||
Discard scheduled restart or switchover events.
|
Discard scheduled restart or switchover events.
|
||||||
@@ -1891,12 +1922,13 @@ def show_diff(before_editing: str, after_editing: str) -> None:
|
|||||||
unified_diff = difflib.unified_diff(listify(before_editing), listify(after_editing))
|
unified_diff = difflib.unified_diff(listify(before_editing), listify(after_editing))
|
||||||
|
|
||||||
if sys.stdout.isatty():
|
if sys.stdout.isatty():
|
||||||
buf = io.StringIO()
|
buf = io.BytesIO()
|
||||||
for line in unified_diff:
|
for line in unified_diff:
|
||||||
buf.write(str(line))
|
buf.write(line.encode('utf-8'))
|
||||||
buf.seek(0)
|
buf.seek(0)
|
||||||
|
|
||||||
class opts:
|
class opts:
|
||||||
|
theme = 'default'
|
||||||
side_by_side = False
|
side_by_side = False
|
||||||
width = 80
|
width = 80
|
||||||
tab_width = 8
|
tab_width = 8
|
||||||
@@ -2025,7 +2057,7 @@ def invoke_editor(before_editing: str, cluster_name: str) -> Tuple[str, Dict[str
|
|||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
Requires an editor program, and uses first found among:
|
Requires an editor program, and uses first found among:
|
||||||
* Program given by ``EDITOR`` environemnt variable; or
|
* Program given by ``EDITOR`` environment variable; or
|
||||||
* ``editor``; or
|
* ``editor``; or
|
||||||
* ``vi``.
|
* ``vi``.
|
||||||
|
|
||||||
@@ -2049,9 +2081,10 @@ def invoke_editor(before_editing: str, cluster_name: str) -> Tuple[str, Dict[str
|
|||||||
if not editor_cmd:
|
if not editor_cmd:
|
||||||
raise PatroniCtlException('EDITOR environment variable is not set. editor or vi are not available')
|
raise PatroniCtlException('EDITOR environment variable is not set. editor or vi are not available')
|
||||||
|
|
||||||
|
safe_cluster_name = cluster_name.replace("/", "_")
|
||||||
with temporary_file(contents=before_editing.encode('utf-8'),
|
with temporary_file(contents=before_editing.encode('utf-8'),
|
||||||
suffix='.yaml',
|
suffix='.yaml',
|
||||||
prefix='{0}-config-'.format(cluster_name)) as tmpfile:
|
prefix='{0}-config-'.format(safe_cluster_name)) as tmpfile:
|
||||||
ret = subprocess.call([editor_cmd, tmpfile])
|
ret = subprocess.call([editor_cmd, tmpfile])
|
||||||
if ret:
|
if ret:
|
||||||
raise PatroniCtlException("Editor exited with return code {0}".format(ret))
|
raise PatroniCtlException("Editor exited with return code {0}".format(ret))
|
||||||
@@ -2179,7 +2212,7 @@ def version(cluster_name: str, group: Optional[int], member_names: List[str]) ->
|
|||||||
|
|
||||||
click.echo("")
|
click.echo("")
|
||||||
cluster = get_dcs(cluster_name, group).get_cluster()
|
cluster = get_dcs(cluster_name, group).get_cluster()
|
||||||
for m in get_all_members(cluster, group, 'any'):
|
for m in get_all_members(cluster, group, CtlPostgresqlRole.ANY):
|
||||||
if m.api_url:
|
if m.api_url:
|
||||||
if not member_names or m.name in member_names:
|
if not member_names or m.name in member_names:
|
||||||
try:
|
try:
|
||||||
|
|||||||
+9
-1
@@ -7,6 +7,7 @@ from __future__ import print_function
|
|||||||
|
|
||||||
import abc
|
import abc
|
||||||
import argparse
|
import argparse
|
||||||
|
import logging
|
||||||
import os
|
import os
|
||||||
import signal
|
import signal
|
||||||
import sys
|
import sys
|
||||||
@@ -17,6 +18,8 @@ from typing import Any, Optional, Type, TYPE_CHECKING
|
|||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from .config import Config
|
from .config import Config
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def get_base_arg_parser() -> argparse.ArgumentParser:
|
def get_base_arg_parser() -> argparse.ArgumentParser:
|
||||||
"""Create a basic argument parser with the arguments used for both patroni and raft controller daemon.
|
"""Create a basic argument parser with the arguments used for both patroni and raft controller daemon.
|
||||||
@@ -132,8 +135,13 @@ class AbstractPatroniDaemon(abc.ABC):
|
|||||||
"""Run the daemon process.
|
"""Run the daemon process.
|
||||||
|
|
||||||
Start the logger thread and keep running execution cycles until a SIGTERM is eventually received. Also reload
|
Start the logger thread and keep running execution cycles until a SIGTERM is eventually received. Also reload
|
||||||
configuration uppon receiving SIGHUP.
|
configuration upon receiving SIGHUP.
|
||||||
"""
|
"""
|
||||||
|
try: # pragma: no cover
|
||||||
|
from systemd import daemon # pyright: ignore
|
||||||
|
daemon.notify("READY=1") # pyright: ignore
|
||||||
|
except ImportError: # pragma: no cover
|
||||||
|
logger.info("Systemd integration is not supported")
|
||||||
self.logger.start()
|
self.logger.start()
|
||||||
while not self.received_sigterm:
|
while not self.received_sigterm:
|
||||||
if self._received_sighup:
|
if self._received_sighup:
|
||||||
|
|||||||
+361
-114
@@ -5,29 +5,29 @@ import json
|
|||||||
import logging
|
import logging
|
||||||
import re
|
import re
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
from random import randint
|
from random import randint
|
||||||
from threading import Event, Lock
|
from threading import Event, Lock
|
||||||
from typing import Any, Callable, Collection, Dict, Iterator, List, \
|
from typing import Any, Callable, cast, Collection, Dict, Iterator, \
|
||||||
NamedTuple, Optional, Tuple, Type, TYPE_CHECKING, Union
|
List, NamedTuple, Optional, Set, Tuple, Type, TYPE_CHECKING, Union
|
||||||
from urllib.parse import urlparse, urlunparse, parse_qsl
|
from urllib.parse import parse_qsl, urlparse, urlunparse
|
||||||
|
|
||||||
import dateutil.parser
|
import dateutil.parser
|
||||||
|
|
||||||
from .. import global_config
|
from .. import global_config
|
||||||
from ..dynamic_loader import iter_classes, iter_modules
|
from ..dynamic_loader import iter_classes, iter_modules
|
||||||
from ..exceptions import PatroniFatalException
|
from ..exceptions import PatroniFatalException
|
||||||
from ..utils import deep_compare, uri
|
|
||||||
from ..tags import Tags
|
from ..tags import Tags
|
||||||
from ..utils import parse_int
|
from ..utils import deep_compare, parse_int, uri
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from ..config import Config
|
from ..config import Config
|
||||||
from ..postgresql import Postgresql
|
from ..postgresql import Postgresql
|
||||||
|
from ..postgresql.misc import PostgresqlRole
|
||||||
from ..postgresql.mpp import AbstractMPP
|
from ..postgresql.mpp import AbstractMPP
|
||||||
|
|
||||||
SLOT_ADVANCE_AVAILABLE_VERSION = 110000
|
|
||||||
slot_name_re = re.compile('^[a-z0-9_]{1,63}$')
|
slot_name_re = re.compile('^[a-z0-9_]{1,63}$')
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -85,6 +85,8 @@ def dcs_modules() -> List[str]:
|
|||||||
|
|
||||||
:returns: list of known module names with absolute python module path namespace, e.g. ``patroni.dcs.etcd``.
|
:returns: list of known module names with absolute python module path namespace, e.g. ``patroni.dcs.etcd``.
|
||||||
"""
|
"""
|
||||||
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
|
assert isinstance(__package__, str)
|
||||||
return iter_modules(__package__)
|
return iter_modules(__package__)
|
||||||
|
|
||||||
|
|
||||||
@@ -101,6 +103,8 @@ def iter_dcs_classes(
|
|||||||
|
|
||||||
:returns: an iterator of tuples, each containing the module ``name`` and the imported DCS class object.
|
:returns: an iterator of tuples, each containing the module ``name`` and the imported DCS class object.
|
||||||
"""
|
"""
|
||||||
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
|
assert isinstance(__package__, str)
|
||||||
return iter_classes(__package__, AbstractDCS, config)
|
return iter_classes(__package__, AbstractDCS, config)
|
||||||
|
|
||||||
|
|
||||||
@@ -216,7 +220,7 @@ class Member(Tags, NamedTuple('Member',
|
|||||||
|
|
||||||
return None
|
return None
|
||||||
|
|
||||||
def conn_kwargs(self, auth: Union[Any, Dict[str, Any], None] = None) -> Dict[str, Any]:
|
def conn_kwargs(self, auth: Optional[Any] = None) -> Dict[str, Any]:
|
||||||
"""Give keyword arguments used for PostgreSQL connection settings.
|
"""Give keyword arguments used for PostgreSQL connection settings.
|
||||||
|
|
||||||
:param auth: Authentication properties - can be defined as anything supported by the ``psycopg2`` or
|
:param auth: Authentication properties - can be defined as anything supported by the ``psycopg2`` or
|
||||||
@@ -252,11 +256,24 @@ class Member(Tags, NamedTuple('Member',
|
|||||||
|
|
||||||
# apply any remaining authentication parameters
|
# apply any remaining authentication parameters
|
||||||
if auth and isinstance(auth, dict):
|
if auth and isinstance(auth, dict):
|
||||||
ret.update({k: v for k, v in auth.items() if v is not None})
|
ret.update({k: v for k, v in cast(Dict[str, Any], auth).items() if v is not None})
|
||||||
if 'username' in auth:
|
if 'username' in auth:
|
||||||
ret['user'] = ret.pop('username')
|
ret['user'] = ret.pop('username')
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
|
def get_endpoint_url(self, endpoint: Optional[str] = None) -> str:
|
||||||
|
"""Get URL from member :attr:`~Member.api_url` and endpoint.
|
||||||
|
|
||||||
|
:param endpoint: URL path of REST API.
|
||||||
|
|
||||||
|
:returns: full URL for this REST API.
|
||||||
|
"""
|
||||||
|
url = self.api_url or ''
|
||||||
|
if endpoint:
|
||||||
|
scheme, netloc, _, _, _, _ = urlparse(url)
|
||||||
|
url = urlunparse((scheme, netloc, endpoint, '', '', ''))
|
||||||
|
return url
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def api_url(self) -> Optional[str]:
|
def api_url(self) -> Optional[str]:
|
||||||
"""The ``api_url`` value from :attr:`~Member.data` if defined."""
|
"""The ``api_url`` value from :attr:`~Member.data` if defined."""
|
||||||
@@ -279,8 +296,10 @@ class Member(Tags, NamedTuple('Member',
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def is_running(self) -> bool:
|
def is_running(self) -> bool:
|
||||||
"""``True`` if the member :attr:`~Member.state` is ``running``."""
|
"""``True`` if the member :attr:`~Member.state` is :class:`~patroni.postgresql.misc.PostgresqlState.RUNNING`."""
|
||||||
return self.state == 'running'
|
from ..postgresql.misc import PostgresqlState
|
||||||
|
|
||||||
|
return self.state == PostgresqlState.RUNNING
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def patroni_version(self) -> Optional[Tuple[int, ...]]:
|
def patroni_version(self) -> Optional[Tuple[int, ...]]:
|
||||||
@@ -398,10 +417,13 @@ class Leader(NamedTuple):
|
|||||||
|
|
||||||
>>> Leader(1, '', Member.from_node(1, '', '', '{"version":"z"}')).checkpoint_after_promote
|
>>> Leader(1, '', Member.from_node(1, '', '', '{"version":"z"}')).checkpoint_after_promote
|
||||||
"""
|
"""
|
||||||
|
from ..postgresql.misc import PostgresqlRole
|
||||||
|
|
||||||
version = self.member.patroni_version
|
version = self.member.patroni_version
|
||||||
# 1.5.6 is the last version which doesn't expose checkpoint_after_promote: false
|
# 1.5.6 is the last version which doesn't expose checkpoint_after_promote: false
|
||||||
if version and version > (1, 5, 6):
|
if version and version > (1, 5, 6):
|
||||||
return self.data.get('role') in ('master', 'primary') and 'checkpoint_after_promote' not in self.data
|
return self.data.get('role') in (PostgresqlRole.MASTER, PostgresqlRole.PRIMARY) \
|
||||||
|
and 'checkpoint_after_promote' not in self.data
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
@@ -545,11 +567,15 @@ class SyncState(NamedTuple):
|
|||||||
:ivar version: modification version of a synchronization key in a Configuration Store.
|
:ivar version: modification version of a synchronization key in a Configuration Store.
|
||||||
:ivar leader: reference to member that was leader.
|
:ivar leader: reference to member that was leader.
|
||||||
:ivar sync_standby: synchronous standby list (comma delimited) which are last synchronized to leader.
|
:ivar sync_standby: synchronous standby list (comma delimited) which are last synchronized to leader.
|
||||||
|
:ivar quorum: if the node from :attr:`~SyncState.sync_standby` list is doing a leader race it should
|
||||||
|
see at least :attr:`~SyncState.quorum` other nodes from the
|
||||||
|
:attr:`~SyncState.sync_standby` + :attr:`~SyncState.leader` list.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
version: Optional[_Version]
|
version: Optional[_Version]
|
||||||
leader: Optional[str]
|
leader: Optional[str]
|
||||||
sync_standby: Optional[str]
|
sync_standby: Optional[str]
|
||||||
|
quorum: int
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def from_node(version: Optional[_Version], value: Union[str, Dict[str, Any], None]) -> 'SyncState':
|
def from_node(version: Optional[_Version], value: Union[str, Dict[str, Any], None]) -> 'SyncState':
|
||||||
@@ -584,7 +610,9 @@ class SyncState(NamedTuple):
|
|||||||
if value and isinstance(value, str):
|
if value and isinstance(value, str):
|
||||||
value = json.loads(value)
|
value = json.loads(value)
|
||||||
assert isinstance(value, dict)
|
assert isinstance(value, dict)
|
||||||
return SyncState(version, value.get('leader'), value.get('sync_standby'))
|
leader = value.get('leader')
|
||||||
|
quorum = value.get('quorum')
|
||||||
|
return SyncState(version, leader, value.get('sync_standby'), int(quorum) if leader and quorum else 0)
|
||||||
except (AssertionError, TypeError, ValueError):
|
except (AssertionError, TypeError, ValueError):
|
||||||
return SyncState.empty(version)
|
return SyncState.empty(version)
|
||||||
|
|
||||||
@@ -596,7 +624,7 @@ class SyncState(NamedTuple):
|
|||||||
|
|
||||||
:returns: empty synchronisation state object.
|
:returns: empty synchronisation state object.
|
||||||
"""
|
"""
|
||||||
return SyncState(version, None, None)
|
return SyncState(version, None, None, 0)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def is_empty(self) -> bool:
|
def is_empty(self) -> bool:
|
||||||
@@ -614,10 +642,17 @@ class SyncState(NamedTuple):
|
|||||||
return list(filter(lambda a: a, [s.strip() for s in value.split(',')]))
|
return list(filter(lambda a: a, [s.strip() for s in value.split(',')]))
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def members(self) -> List[str]:
|
def voters(self) -> List[str]:
|
||||||
""":attr:`~SyncState.sync_standby` as list or an empty list if undefined or object considered ``empty``."""
|
""":attr:`~SyncState.sync_standby` as list or an empty list if undefined or object considered ``empty``."""
|
||||||
return self._str_to_list(self.sync_standby) if not self.is_empty and self.sync_standby else []
|
return self._str_to_list(self.sync_standby) if not self.is_empty and self.sync_standby else []
|
||||||
|
|
||||||
|
@property
|
||||||
|
def members(self) -> List[str]:
|
||||||
|
""":attr:`~SyncState.sync_standby` and :attr:`~SyncState.leader` as list
|
||||||
|
or an empty list if object considered ``empty``.
|
||||||
|
"""
|
||||||
|
return [] if not self.leader else [self.leader] + self.voters
|
||||||
|
|
||||||
def matches(self, name: Optional[str], check_leader: bool = False) -> bool:
|
def matches(self, name: Optional[str], check_leader: bool = False) -> bool:
|
||||||
"""Checks if node is presented in the /sync state.
|
"""Checks if node is presented in the /sync state.
|
||||||
|
|
||||||
@@ -631,7 +666,7 @@ class SyncState(NamedTuple):
|
|||||||
the sync state.
|
the sync state.
|
||||||
|
|
||||||
:Example:
|
:Example:
|
||||||
>>> s = SyncState(1, 'foo', 'bar,zoo')
|
>>> s = SyncState(1, 'foo', 'bar,zoo', 0)
|
||||||
|
|
||||||
>>> s.matches('foo')
|
>>> s.matches('foo')
|
||||||
False
|
False
|
||||||
@@ -722,9 +757,11 @@ class Status(NamedTuple):
|
|||||||
|
|
||||||
:ivar last_lsn: :class:`int` object containing position of last known leader LSN.
|
:ivar last_lsn: :class:`int` object containing position of last known leader LSN.
|
||||||
:ivar slots: state of permanent replication slots on the primary in the format: ``{"slot_name": int}``.
|
:ivar slots: state of permanent replication slots on the primary in the format: ``{"slot_name": int}``.
|
||||||
|
:ivar retain_slots: list physical replication slots for members that exist in the cluster.
|
||||||
"""
|
"""
|
||||||
last_lsn: int
|
last_lsn: int
|
||||||
slots: Optional[Dict[str, int]]
|
slots: Optional[Dict[str, int]]
|
||||||
|
retain_slots: List[str]
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def empty() -> 'Status':
|
def empty() -> 'Status':
|
||||||
@@ -732,13 +769,20 @@ class Status(NamedTuple):
|
|||||||
|
|
||||||
:returns: empty :class:`Status` object.
|
:returns: empty :class:`Status` object.
|
||||||
"""
|
"""
|
||||||
return Status(0, None)
|
return Status(0, None, [])
|
||||||
|
|
||||||
|
def is_empty(self):
|
||||||
|
"""Validate definition of all attributes of this :class:`Status` instance.
|
||||||
|
|
||||||
|
:returns: ``True`` if all attributes of the current :class:`Status` are unpopulated.
|
||||||
|
"""
|
||||||
|
return self.last_lsn == 0 and self.slots is None and not self.retain_slots
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def from_node(value: Union[str, Dict[str, Any], None]) -> 'Status':
|
def from_node(value: Union[str, Dict[str, Any], None]) -> 'Status':
|
||||||
"""Factory method to parse *value* as :class:`Status` object.
|
"""Factory method to parse *value* as :class:`Status` object.
|
||||||
|
|
||||||
:param value: JSON serialized string
|
:param value: JSON serialized string or :class:`dict` object.
|
||||||
|
|
||||||
:returns: constructed :class:`Status` object.
|
:returns: constructed :class:`Status` object.
|
||||||
"""
|
"""
|
||||||
@@ -749,7 +793,7 @@ class Status(NamedTuple):
|
|||||||
return Status.empty()
|
return Status.empty()
|
||||||
|
|
||||||
if isinstance(value, int): # legacy
|
if isinstance(value, int): # legacy
|
||||||
return Status(value, None)
|
return Status(value, None, [])
|
||||||
|
|
||||||
if not isinstance(value, dict):
|
if not isinstance(value, dict):
|
||||||
return Status.empty()
|
return Status.empty()
|
||||||
@@ -768,7 +812,16 @@ class Status(NamedTuple):
|
|||||||
if not isinstance(slots, dict):
|
if not isinstance(slots, dict):
|
||||||
slots = None
|
slots = None
|
||||||
|
|
||||||
return Status(last_lsn, slots)
|
retain_slots: Union[str, List[str], None] = value.get('retain_slots')
|
||||||
|
if isinstance(retain_slots, str):
|
||||||
|
try:
|
||||||
|
retain_slots = json.loads(retain_slots)
|
||||||
|
except Exception:
|
||||||
|
retain_slots = []
|
||||||
|
if not isinstance(retain_slots, list):
|
||||||
|
retain_slots = []
|
||||||
|
|
||||||
|
return Status(last_lsn, slots, retain_slots)
|
||||||
|
|
||||||
|
|
||||||
class Cluster(NamedTuple('Cluster',
|
class Cluster(NamedTuple('Cluster',
|
||||||
@@ -810,14 +863,13 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
return super(Cluster, cls).__new__(cls, *args, **kwargs)
|
return super(Cluster, cls).__new__(cls, *args, **kwargs)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def last_lsn(self) -> int:
|
def slots(self) -> Dict[str, int]:
|
||||||
"""Last known leader LSN."""
|
"""State of permanent replication slots on the primary in the format: ``{"slot_name": int}``.
|
||||||
return self.status.last_lsn
|
|
||||||
|
|
||||||
@property
|
.. note::
|
||||||
def slots(self) -> Optional[Dict[str, int]]:
|
We are trying to be foolproof here and for values that can't be parsed to :class:`int` will return ``0``.
|
||||||
"""State of permanent replication slots on the primary in the format: ``{"slot_name": int}``."""
|
"""
|
||||||
return self.status.slots
|
return {k: parse_int(v) or 0 for k, v in (self.status.slots or {}).items()}
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def empty() -> 'Cluster':
|
def empty() -> 'Cluster':
|
||||||
@@ -829,9 +881,9 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
|
|
||||||
:returns: ``True`` if all attributes of the current :class:`Cluster` are unpopulated.
|
:returns: ``True`` if all attributes of the current :class:`Cluster` are unpopulated.
|
||||||
"""
|
"""
|
||||||
return all((self.initialize is None, self.config is None, self.leader is None, self.last_lsn == 0,
|
return all((self.initialize is None, self.config is None, self.leader is None, self.status.is_empty(),
|
||||||
self.members == [], self.failover is None, self.sync.version is None,
|
self.members == [], self.failover is None, self.sync.version is None,
|
||||||
self.history is None, self.slots is None, self.failsafe is None, self.workers == {}))
|
self.history is None, self.failsafe is None, self.workers == {}))
|
||||||
|
|
||||||
def __len__(self) -> int:
|
def __len__(self) -> int:
|
||||||
"""Implement ``len`` function capability.
|
"""Implement ``len`` function capability.
|
||||||
@@ -845,7 +897,8 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
|
|
||||||
>>> assert bool(cluster) is False
|
>>> assert bool(cluster) is False
|
||||||
|
|
||||||
>>> cluster = Cluster(None, None, None, Status(0, None), [1, 2, 3], None, SyncState.empty(), None, None, {})
|
>>> status = Status(0, None, [])
|
||||||
|
>>> cluster = Cluster(None, None, None, status, [1, 2, 3], None, SyncState.empty(), None, None, {})
|
||||||
>>> len(cluster)
|
>>> len(cluster)
|
||||||
1
|
1
|
||||||
|
|
||||||
@@ -902,17 +955,19 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
return candidates[randint(0, len(candidates) - 1)] if candidates else self.leader
|
return candidates[randint(0, len(candidates) - 1)] if candidates else self.leader
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def is_physical_slot(value: Union[Any, Dict[str, Any]]) -> bool:
|
def is_physical_slot(value: Any) -> bool:
|
||||||
"""Check whether provided configuration is for permanent physical replication slot.
|
"""Check whether provided configuration is for permanent physical replication slot.
|
||||||
|
|
||||||
:param value: configuration of the permanent replication slot.
|
:param value: configuration of the permanent replication slot.
|
||||||
|
|
||||||
:returns: ``True`` if *value* is a physical replication slot, otherwise ``False``.
|
:returns: ``True`` if *value* is a physical replication slot, otherwise ``False``.
|
||||||
"""
|
"""
|
||||||
return not value or isinstance(value, dict) and value.get('type', 'physical') == 'physical'
|
return not value \
|
||||||
|
or (isinstance(value, dict) and not Cluster.is_logical_slot(cast(Dict[str, Any], value))
|
||||||
|
and cast(Dict[str, Any], value).get('type', 'physical') == 'physical')
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def is_logical_slot(value: Union[Any, Dict[str, Any]]) -> bool:
|
def is_logical_slot(value: Any) -> bool:
|
||||||
"""Check whether provided configuration is for permanent logical replication slot.
|
"""Check whether provided configuration is for permanent logical replication slot.
|
||||||
|
|
||||||
:param value: configuration of the permanent replication slot.
|
:param value: configuration of the permanent replication slot.
|
||||||
@@ -920,35 +975,38 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
:returns: ``True`` if *value* is a logical replication slot, otherwise ``False``.
|
:returns: ``True`` if *value* is a logical replication slot, otherwise ``False``.
|
||||||
"""
|
"""
|
||||||
return isinstance(value, dict) \
|
return isinstance(value, dict) \
|
||||||
and value.get('type', 'logical') == 'logical' \
|
and cast(Dict[str, Any], value).get('type', 'logical') == 'logical' \
|
||||||
and bool(value.get('database') and value.get('plugin'))
|
and bool(cast(Dict[str, Any], value).get('database') and cast(Dict[str, Any], value).get('plugin'))
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def __permanent_slots(self) -> Dict[str, Union[Dict[str, Any], Any]]:
|
def __permanent_slots(self) -> Dict[str, Union[Dict[str, Any], Any]]:
|
||||||
"""Dictionary of permanent replication slots with their known LSN."""
|
"""Dictionary of permanent replication slots with their known LSN."""
|
||||||
ret: Dict[str, Union[Dict[str, Any], Any]] = global_config.permanent_slots
|
ret: Dict[str, Union[Dict[str, Any], Any]] = global_config.permanent_slots
|
||||||
|
|
||||||
members: Dict[str, int] = {slot_name_from_member_name(m.name): m.lsn or 0 for m in self.members}
|
members: Dict[str, int] = {slot_name_from_member_name(m.name): m.lsn or 0
|
||||||
slots: Dict[str, int] = {k: parse_int(v) or 0 for k, v in (self.slots or {}).items()}
|
for m in self.members if m.replicatefrom}
|
||||||
|
slots: Dict[str, int] = self.slots
|
||||||
for name, value in list(ret.items()):
|
for name, value in list(ret.items()):
|
||||||
if not value:
|
if not value:
|
||||||
value = ret[name] = {}
|
value = ret[name] = {}
|
||||||
if isinstance(value, dict):
|
if isinstance(value, dict):
|
||||||
# for permanent physical slots we want to get MAX LSN from the `Cluster.slots` and from the
|
# For permanent physical slots we want to get MAX LSN from the `Cluster.slots` and from the
|
||||||
# member with the matching name. It is necessary because we may have the replication slot on
|
# member that does cascading replication with the matching name (see `replicatefrom` tag).
|
||||||
# the primary that is streaming from the other standby node using the `replicatefrom` tag.
|
# It is necessary because we may have the permanent replication slot on the primary for this node.
|
||||||
lsn = max(members.get(name, 0) if self.is_physical_slot(value) else 0, slots.get(name, 0))
|
lsn = max(members.get(name, 0) if self.is_physical_slot(value) else 0, slots.get(name, 0))
|
||||||
if lsn:
|
if lsn:
|
||||||
value['lsn'] = lsn
|
value['lsn'] = lsn
|
||||||
else:
|
else:
|
||||||
# Don't let anyone set 'lsn' in the global configuration :)
|
# Don't let anyone set 'lsn' in the global configuration :)
|
||||||
value.pop('lsn', None)
|
value.pop('lsn', None) # pyright: ignore [reportUnknownMemberType]
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def __permanent_physical_slots(self) -> Dict[str, Any]:
|
def permanent_physical_slots(self) -> Dict[str, Any]:
|
||||||
"""Dictionary of permanent ``physical`` replication slots."""
|
"""Dictionary of permanent ``physical`` replication slots."""
|
||||||
return {name: value for name, value in self.__permanent_slots.items() if self.is_physical_slot(value)}
|
return {name: value for name, value in self.__permanent_slots.items() if self.is_physical_slot(value)
|
||||||
|
and ((global_config.is_standby_cluster and value.get('cluster_type') != 'primary')
|
||||||
|
or (not global_config.is_standby_cluster and value.get('cluster_type') != 'standby'))}
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def __permanent_logical_slots(self) -> Dict[str, Any]:
|
def __permanent_logical_slots(self) -> Dict[str, Any]:
|
||||||
@@ -956,7 +1014,8 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
return {name: value for name, value in self.__permanent_slots.items() if self.is_logical_slot(value)}
|
return {name: value for name, value in self.__permanent_slots.items() if self.is_logical_slot(value)}
|
||||||
|
|
||||||
def get_replication_slots(self, postgresql: 'Postgresql', member: Tags, *,
|
def get_replication_slots(self, postgresql: 'Postgresql', member: Tags, *,
|
||||||
role: Optional[str] = None, show_error: bool = False) -> Dict[str, Dict[str, Any]]:
|
role: Optional['PostgresqlRole'] = None,
|
||||||
|
show_error: bool = False) -> Dict[str, Dict[str, Any]]:
|
||||||
"""Lookup configured slot names in the DCS, report issues found and merge with permanent slots.
|
"""Lookup configured slot names in the DCS, report issues found and merge with permanent slots.
|
||||||
|
|
||||||
Will log an error if:
|
Will log an error if:
|
||||||
@@ -965,7 +1024,8 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
|
|
||||||
:param postgresql: reference to :class:`Postgresql` object.
|
:param postgresql: reference to :class:`Postgresql` object.
|
||||||
:param member: reference to an object implementing :class:`Tags` interface.
|
:param member: reference to an object implementing :class:`Tags` interface.
|
||||||
:param role: role of the node, if not set will be taken from *postgresql*.
|
:param role: role of the node, if not set will be taken from *postgresql*
|
||||||
|
One of :class:`~patroni.postgresql.misc.PostgresqlRole` values.
|
||||||
:param show_error: if ``True`` report error if any disabled logical slots or conflicting slot names are found.
|
:param show_error: if ``True`` report error if any disabled logical slots or conflicting slot names are found.
|
||||||
|
|
||||||
:returns: final dictionary of slot names, after merging with permanent slots and performing sanity checks.
|
:returns: final dictionary of slot names, after merging with permanent slots and performing sanity checks.
|
||||||
@@ -973,11 +1033,12 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
name = member.name if isinstance(member, Member) else postgresql.name
|
name = member.name if isinstance(member, Member) else postgresql.name
|
||||||
role = role or postgresql.role
|
role = role or postgresql.role
|
||||||
|
|
||||||
slots: Dict[str, Dict[str, str]] = self._get_members_slots(name, role)
|
slots: Dict[str, Dict[str, Any]] = self._get_members_slots(name, role,
|
||||||
|
member.nofailover, postgresql.can_advance_slots)
|
||||||
permanent_slots: Dict[str, Any] = self._get_permanent_slots(postgresql, member, role)
|
permanent_slots: Dict[str, Any] = self._get_permanent_slots(postgresql, member, role)
|
||||||
|
|
||||||
disabled_permanent_logical_slots: List[str] = self._merge_permanent_slots(
|
disabled_permanent_logical_slots: List[str] = self._merge_permanent_slots(
|
||||||
slots, permanent_slots, name, postgresql.major_version)
|
slots, permanent_slots, name, role, postgresql.can_advance_slots)
|
||||||
|
|
||||||
if disabled_permanent_logical_slots and show_error:
|
if disabled_permanent_logical_slots and show_error:
|
||||||
logger.error("Permanent logical replication slots supported by Patroni only starting from PostgreSQL 11. "
|
logger.error("Permanent logical replication slots supported by Patroni only starting from PostgreSQL 11. "
|
||||||
@@ -985,8 +1046,8 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
|
|
||||||
return slots
|
return slots
|
||||||
|
|
||||||
def _merge_permanent_slots(self, slots: Dict[str, Dict[str, str]], permanent_slots: Dict[str, Any], name: str,
|
def _merge_permanent_slots(self, slots: Dict[str, Dict[str, Any]], permanent_slots: Dict[str, Any],
|
||||||
major_version: int) -> List[str]:
|
name: str, role: 'PostgresqlRole', can_advance_slots: bool) -> List[str]:
|
||||||
"""Merge replication *slots* for members with *permanent_slots*.
|
"""Merge replication *slots* for members with *permanent_slots*.
|
||||||
|
|
||||||
Perform validation of configured permanent slot name, skipping invalid names.
|
Perform validation of configured permanent slot name, skipping invalid names.
|
||||||
@@ -996,11 +1057,19 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
|
|
||||||
:param slots: Slot names with existing attributes if known.
|
:param slots: Slot names with existing attributes if known.
|
||||||
:param name: name of this node.
|
:param name: name of this node.
|
||||||
|
:param role: role of the node. One of :class:`~patroni.postgresql.misc.PostgresqlRole` values.
|
||||||
:param permanent_slots: dictionary containing slot name key and slot information values.
|
:param permanent_slots: dictionary containing slot name key and slot information values.
|
||||||
:param major_version: postgresql major version.
|
:param can_advance_slots: ``True`` if ``pg_replication_slot_advance()`` function is available,
|
||||||
|
``False`` otherwise.
|
||||||
|
|
||||||
:returns: List of disabled permanent, logical slot names, if postgresql version < 11.
|
:returns: List of disabled permanent, logical slot names, if postgresql version < 11.
|
||||||
"""
|
"""
|
||||||
|
from ..postgresql.misc import PostgresqlRole
|
||||||
|
|
||||||
|
name = slot_name_from_member_name(name)
|
||||||
|
topology = {slot_name_from_member_name(m.name): m.replicatefrom and slot_name_from_member_name(m.replicatefrom)
|
||||||
|
for m in self.members}
|
||||||
|
|
||||||
disabled_permanent_logical_slots: List[str] = []
|
disabled_permanent_logical_slots: List[str] = []
|
||||||
|
|
||||||
for slot_name, value in permanent_slots.items():
|
for slot_name, value in permanent_slots.items():
|
||||||
@@ -1009,19 +1078,27 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
logger.error("Slot name may only contain lower case letters, numbers, and the underscore chars")
|
logger.error("Slot name may only contain lower case letters, numbers, and the underscore chars")
|
||||||
continue
|
continue
|
||||||
|
|
||||||
value = deepcopy(value) if value else {'type': 'physical'}
|
tmp = deepcopy(value) if value else {'type': 'physical'}
|
||||||
if isinstance(value, dict):
|
if isinstance(tmp, dict):
|
||||||
|
value = cast(Dict[str, Any], tmp)
|
||||||
if 'type' not in value:
|
if 'type' not in value:
|
||||||
value['type'] = 'logical' if value.get('database') and value.get('plugin') else 'physical'
|
value['type'] = 'logical' if value.get('database') and value.get('plugin') else 'physical'
|
||||||
|
|
||||||
if value['type'] == 'physical':
|
if value['type'] == 'physical':
|
||||||
# Don't try to create permanent physical replication slot for yourself
|
# Don't try to create permanent physical replication slot for yourself
|
||||||
if slot_name != slot_name_from_member_name(name):
|
if slot_name not in slots and slot_name != name:
|
||||||
slots[slot_name] = value
|
# On the leader we expected to have permanent slots active, except the case when it is a slot
|
||||||
|
# for a cascading replica. Lets consider a configuration with C being a permanent slot. In this
|
||||||
|
# case we should have the following: A(B: active, C: inactive) <- B (C: active) <- C
|
||||||
|
# We don't consider the same situation on node B, because if node C doesn't exists, we will not
|
||||||
|
# be able to know its `replicatefrom` tag value.
|
||||||
|
expected_active = not topology.get(slot_name) and role in (PostgresqlRole.PRIMARY,
|
||||||
|
PostgresqlRole.STANDBY_LEADER)
|
||||||
|
slots[slot_name] = {**value, 'expected_active': expected_active}
|
||||||
continue
|
continue
|
||||||
|
|
||||||
if self.is_logical_slot(value):
|
if self.is_logical_slot(value):
|
||||||
if major_version < SLOT_ADVANCE_AVAILABLE_VERSION:
|
if not can_advance_slots:
|
||||||
disabled_permanent_logical_slots.append(slot_name)
|
disabled_permanent_logical_slots.append(slot_name)
|
||||||
elif slot_name in slots:
|
elif slot_name in slots:
|
||||||
logger.error("Permanent logical replication slot {'%s': %s} is conflicting with"
|
logger.error("Permanent logical replication slot {'%s': %s} is conflicting with"
|
||||||
@@ -1033,12 +1110,14 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
logger.error("Bad value for slot '%s' in permanent_slots: %s", slot_name, permanent_slots[slot_name])
|
logger.error("Bad value for slot '%s' in permanent_slots: %s", slot_name, permanent_slots[slot_name])
|
||||||
return disabled_permanent_logical_slots
|
return disabled_permanent_logical_slots
|
||||||
|
|
||||||
def _get_permanent_slots(self, postgresql: 'Postgresql', tags: Tags, role: str) -> Dict[str, Any]:
|
def _get_permanent_slots(self, postgresql: 'Postgresql', tags: Tags, role: 'PostgresqlRole') -> Dict[str, Any]:
|
||||||
"""Get configured permanent replication slots.
|
"""Get configured permanent replication slots.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
Permanent replication slots are only considered if ``use_slots`` configuration is enabled.
|
Permanent replication slots are only considered if ``use_slots`` configuration is enabled.
|
||||||
A node that is not supposed to become a leader (*nofailover*) will not have permanent replication slots.
|
A node that is not supposed to become a leader (*nofailover*) will not have permanent replication slots.
|
||||||
|
Also node with disabled streaming (*nostream*) and its cascading followers must not have permanent
|
||||||
|
logical slots due to lack of feedback from node to primary, which makes them unsafe to use.
|
||||||
|
|
||||||
In a standby cluster we only support physical replication slots.
|
In a standby cluster we only support physical replication slots.
|
||||||
|
|
||||||
@@ -1047,54 +1126,117 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
|
|
||||||
:param postgresql: reference to :class:`Postgresql` object.
|
:param postgresql: reference to :class:`Postgresql` object.
|
||||||
:param tags: reference to an object implementing :class:`Tags` interface.
|
:param tags: reference to an object implementing :class:`Tags` interface.
|
||||||
:param role: role of the node -- ``primary``, ``standby_leader`` or ``replica``.
|
:param role: role of the node. One of :class:`~patroni.postgresql.misc.PostgresqlRole` values.
|
||||||
|
|
||||||
:returns: dictionary of permanent slot names mapped to attributes.
|
:returns: dictionary of permanent slot names mapped to attributes.
|
||||||
"""
|
"""
|
||||||
|
from ..postgresql.misc import PostgresqlRole
|
||||||
|
|
||||||
if not global_config.use_slots or tags.nofailover:
|
if not global_config.use_slots or tags.nofailover:
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
if global_config.is_standby_cluster:
|
if global_config.is_standby_cluster or self.get_slot_name_on_primary(postgresql.name, tags) is None:
|
||||||
return self.__permanent_physical_slots \
|
return self.permanent_physical_slots\
|
||||||
if postgresql.major_version >= SLOT_ADVANCE_AVAILABLE_VERSION or role == 'standby_leader' else {}
|
if postgresql.can_advance_slots or role == PostgresqlRole.STANDBY_LEADER else {}
|
||||||
|
|
||||||
return self.__permanent_slots if postgresql.major_version >= SLOT_ADVANCE_AVAILABLE_VERSION\
|
return self.__permanent_slots if postgresql.can_advance_slots or role == PostgresqlRole.PRIMARY \
|
||||||
or role in ('master', 'primary') else self.__permanent_logical_slots
|
else self.__permanent_logical_slots
|
||||||
|
|
||||||
def _get_members_slots(self, name: str, role: str) -> Dict[str, Dict[str, str]]:
|
def _get_members_slots(self, name: str, role: 'PostgresqlRole', nofailover: bool,
|
||||||
"""Get physical replication slots configuration for members that sourcing from this node.
|
can_advance_slots: bool) -> Dict[str, Dict[str, Any]]:
|
||||||
|
"""Get physical replication slots configuration for a given member.
|
||||||
|
|
||||||
If the ``replicatefrom`` tag is set on the member - we should not create the replication slot for it on
|
There are following situations possible:
|
||||||
the current primary, because that member would replicate from elsewhere. We still create the slot if
|
|
||||||
the ``replicatefrom`` destination member is currently not a member of the cluster (fallback to the
|
* If the ``nostream`` tag is set on the member - we should not have the replication slot for it
|
||||||
primary), or if ``replicatefrom`` destination member happens to be the current primary.
|
on the current primary or any other member even if ``replicatefrom`` is set, because
|
||||||
|
``nostream`` disables WAL streaming.
|
||||||
|
|
||||||
|
* PostgreSQL is 11 and newer and configuration allows retention of member replication slots. In this case
|
||||||
|
we want to have replication slots for every member except the case when we have ``nofailover`` tag set.
|
||||||
|
|
||||||
|
* PostgreSQL is older than 11 or configuration doesn't allow member slots retention. In this case we want:
|
||||||
|
|
||||||
|
* On primary have replication slots for all members that don't have ``replicatefrom`` tag pointing
|
||||||
|
to the existing member.
|
||||||
|
|
||||||
|
* On replica node have replication slots only for members which ``replicatefrom`` tag pointing to us.
|
||||||
|
|
||||||
Will log an error if:
|
Will log an error if:
|
||||||
|
|
||||||
* Conflicting slot names between members are found
|
* Conflicting slot names between members are found
|
||||||
|
|
||||||
:param name: name of this node.
|
:param name: name of this node.
|
||||||
:param role: role of this node, if this is a ``primary`` or ``standby_leader`` return list of members
|
:param role: role of this node, ``primary``, ``standby_leader``, or ``replica``.
|
||||||
replicating from this node. If not then return a list of members replicating as cascaded
|
:param nofailover: ``True`` if this node is tagged to not be a failover candidate, ``False`` otherwise.
|
||||||
replicas from this node.
|
:param can_advance_slots: ``True`` if ``pg_replication_slot_advance()`` function is available,
|
||||||
|
``False`` otherwise.
|
||||||
|
|
||||||
:returns: dictionary of physical replication slots that should exist on a given node.
|
:returns: dictionary of physical replication slots that should exist on a given node.
|
||||||
"""
|
"""
|
||||||
|
from ..postgresql.misc import PostgresqlRole
|
||||||
|
|
||||||
if not global_config.use_slots:
|
if not global_config.use_slots:
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
# we always want to exclude the member with our name from the list
|
# we always want to exclude the member with our name from the list,
|
||||||
members = filter(lambda m: m.name != name, self.members)
|
# also exclude members with disabled WAL streaming
|
||||||
|
members = filter(lambda m: m.name != name and not m.nostream, self.members)
|
||||||
|
|
||||||
if role in ('master', 'primary', 'standby_leader'):
|
def leader_filter(member: Member) -> bool:
|
||||||
members = [m for m in members if m.replicatefrom is None
|
"""Check whether provided *member* should replicate from the current node when it is running as a leader.
|
||||||
or m.replicatefrom == name or not self.has_member(m.replicatefrom)]
|
|
||||||
|
:param member: a :class:`Member` object.
|
||||||
|
|
||||||
|
:returns: ``True`` if provided member should replicate from the current node, ``False`` otherwise.
|
||||||
|
"""
|
||||||
|
return member.replicatefrom is None or\
|
||||||
|
member.replicatefrom == name or\
|
||||||
|
not self.has_member(member.replicatefrom)
|
||||||
|
|
||||||
|
def replica_filter(member: Member) -> bool:
|
||||||
|
"""Check whether provided *member* should replicate from the current node when it is running as a replica.
|
||||||
|
|
||||||
|
..note::
|
||||||
|
We only consider members with ``replicatefrom`` tag that matches our name and always exclude the leader.
|
||||||
|
|
||||||
|
:param member: a :class:`Member` object.
|
||||||
|
|
||||||
|
:returns: ``True`` if provided member should replicate from the current node, ``False`` otherwise.
|
||||||
|
"""
|
||||||
|
return member.replicatefrom == name and member.name != self.leader_name
|
||||||
|
|
||||||
|
# In case when retention of replication slots is possible the `expected_active` function
|
||||||
|
# will be used to figure out whether the replication slot is expected to be active.
|
||||||
|
# Otherwise it will be used to find replication slots that should exist on a current node.
|
||||||
|
expected_active = leader_filter if role in (PostgresqlRole.PRIMARY, PostgresqlRole.STANDBY_LEADER) \
|
||||||
|
else replica_filter
|
||||||
|
|
||||||
|
if can_advance_slots and global_config.member_slots_ttl > 0:
|
||||||
|
# if the node does only cascading and can't become the leader, we
|
||||||
|
# want only to have slots for members that could connect to it.
|
||||||
|
members = [m for m in members if not nofailover or m.replicatefrom == name]
|
||||||
else:
|
else:
|
||||||
# only manage slots for replicas that replicate from this one, except for the leader among them
|
members = [m for m in members if expected_active(m)]
|
||||||
members = [m for m in members if m.replicatefrom == name and m.name != self.leader_name]
|
|
||||||
|
|
||||||
slots = {slot_name_from_member_name(m.name): {'type': 'physical'} for m in members}
|
leader_patroni_version = self.leader and self.leader.member.patroni_version
|
||||||
if len(slots) < len(members):
|
slots: Dict[str, int] = self.slots
|
||||||
|
ret: Dict[str, Dict[str, Any]] = {}
|
||||||
|
for member in members:
|
||||||
|
slot_name = slot_name_from_member_name(member.name)
|
||||||
|
lsn = slots.get(slot_name, 0)
|
||||||
|
if member.replicatefrom or leader_patroni_version and leader_patroni_version < (4, 0, 0):
|
||||||
|
# `/status` key is maintained by the leader, but `member` may be connected to some other node.
|
||||||
|
# In that case, the slot in the leader is inactive and doesn't advance, so we use the LSN
|
||||||
|
# reported by the member to advance replication slot LSN.
|
||||||
|
# `max` is only a fallback so we take the LSN from the slot when there is no feedback from the member.
|
||||||
|
lsn = max(member.lsn or 0, lsn)
|
||||||
|
ret[slot_name] = {'type': 'physical', 'lsn': lsn, 'expected_active': expected_active(member)}
|
||||||
|
slot_name = slot_name_from_member_name(name)
|
||||||
|
ret.update({slot: {'type': 'physical'} for slot in self.status.retain_slots
|
||||||
|
if not nofailover and slot not in ret and slot != slot_name})
|
||||||
|
|
||||||
|
if len(ret) < len(members):
|
||||||
# Find which names are conflicting for a nicer error message
|
# Find which names are conflicting for a nicer error message
|
||||||
slot_conflicts: Dict[str, List[str]] = defaultdict(list)
|
slot_conflicts: Dict[str, List[str]] = defaultdict(list)
|
||||||
for member in members:
|
for member in members:
|
||||||
@@ -1102,7 +1244,7 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
logger.error("Following cluster members share a replication slot name: %s",
|
logger.error("Following cluster members share a replication slot name: %s",
|
||||||
"; ".join(f"{', '.join(v)} map to {k}"
|
"; ".join(f"{', '.join(v)} map to {k}"
|
||||||
for k, v in slot_conflicts.items() if len(v) > 1))
|
for k, v in slot_conflicts.items() if len(v) > 1))
|
||||||
return slots
|
return ret
|
||||||
|
|
||||||
def has_permanent_slots(self, postgresql: 'Postgresql', member: Tags) -> bool:
|
def has_permanent_slots(self, postgresql: 'Postgresql', member: Tags) -> bool:
|
||||||
"""Check if our node has permanent replication slots configured.
|
"""Check if our node has permanent replication slots configured.
|
||||||
@@ -1110,28 +1252,38 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
:param postgresql: reference to :class:`Postgresql` object.
|
:param postgresql: reference to :class:`Postgresql` object.
|
||||||
:param member: reference to an object implementing :class:`Tags` interface for
|
:param member: reference to an object implementing :class:`Tags` interface for
|
||||||
the node that we are checking permanent logical replication slots for.
|
the node that we are checking permanent logical replication slots for.
|
||||||
|
|
||||||
:returns: ``True`` if there are permanent replication slots configured, otherwise ``False``.
|
:returns: ``True`` if there are permanent replication slots configured, otherwise ``False``.
|
||||||
"""
|
"""
|
||||||
role = 'replica'
|
from ..postgresql.misc import PostgresqlRole
|
||||||
members_slots: Dict[str, Dict[str, str]] = self._get_members_slots(postgresql.name, role)
|
|
||||||
|
role = PostgresqlRole.REPLICA
|
||||||
|
members_slots: Dict[str, Dict[str, str]] = self._get_members_slots(postgresql.name, role,
|
||||||
|
member.nofailover,
|
||||||
|
postgresql.can_advance_slots)
|
||||||
permanent_slots: Dict[str, Any] = self._get_permanent_slots(postgresql, member, role)
|
permanent_slots: Dict[str, Any] = self._get_permanent_slots(postgresql, member, role)
|
||||||
slots = deepcopy(members_slots)
|
slots = deepcopy(members_slots)
|
||||||
self._merge_permanent_slots(slots, permanent_slots, postgresql.name, postgresql.major_version)
|
self._merge_permanent_slots(slots, permanent_slots, postgresql.name, role, postgresql.can_advance_slots)
|
||||||
return len(slots) > len(members_slots) or any(self.is_physical_slot(v) for v in permanent_slots.values())
|
return len(slots) > len(members_slots) or any(self.is_physical_slot(v) for v in permanent_slots.values())
|
||||||
|
|
||||||
def filter_permanent_slots(self, postgresql: 'Postgresql', slots: Dict[str, int]) -> Dict[str, int]:
|
def maybe_filter_permanent_slots(self, postgresql: 'Postgresql', slots: Dict[str, int]) -> Dict[str, int]:
|
||||||
"""Filter out all non-permanent slots from provided *slots* dict.
|
"""Filter out all non-permanent slots from provided *slots* dict.
|
||||||
|
|
||||||
|
.. note::
|
||||||
|
In case if retention of replication slots for members is enabled we will not do
|
||||||
|
any filtering, because we need to publish LSN values for members replication slots,
|
||||||
|
so that other nodes can use them to advance LSN, like they do it for permanent slots.
|
||||||
|
|
||||||
:param postgresql: reference to :class:`Postgresql` object.
|
:param postgresql: reference to :class:`Postgresql` object.
|
||||||
:param slots: slot names with LSN values.
|
:param slots: slot names with LSN values.
|
||||||
|
|
||||||
:returns: a :class:`dict` object that contains only slots that are known to be permanent.
|
:returns: a :class:`dict` object that contains only slots that are known to be permanent.
|
||||||
"""
|
"""
|
||||||
if postgresql.major_version < SLOT_ADVANCE_AVAILABLE_VERSION:
|
from ..postgresql.misc import PostgresqlRole
|
||||||
return {} # for legacy PostgreSQL we don't support permanent slots on standby nodes
|
|
||||||
|
|
||||||
permanent_slots: Dict[str, Any] = self._get_permanent_slots(postgresql, RemoteMember('', {}), 'replica')
|
if global_config.member_slots_ttl > 0:
|
||||||
|
return slots
|
||||||
|
|
||||||
|
permanent_slots: Dict[str, Any] = self._get_permanent_slots(postgresql, RemoteMember('', {}),
|
||||||
|
PostgresqlRole.REPLICA)
|
||||||
members_slots = {slot_name_from_member_name(m.name) for m in self.members}
|
members_slots = {slot_name_from_member_name(m.name) for m in self.members}
|
||||||
|
|
||||||
return {name: value for name, value in slots.items() if name in permanent_slots
|
return {name: value for name, value in slots.items() if name in permanent_slots
|
||||||
@@ -1147,7 +1299,9 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
|
|
||||||
:returns: ``True`` if any detected replications slots are ``logical``, otherwise ``False``.
|
:returns: ``True`` if any detected replications slots are ``logical``, otherwise ``False``.
|
||||||
"""
|
"""
|
||||||
slots = self.get_replication_slots(postgresql, member, role='replica').values()
|
from ..postgresql.misc import PostgresqlRole
|
||||||
|
|
||||||
|
slots = self.get_replication_slots(postgresql, member, role=PostgresqlRole.REPLICA).values()
|
||||||
return any(v for v in slots if v.get("type") == "logical")
|
return any(v for v in slots if v.get("type") == "logical")
|
||||||
|
|
||||||
def should_enforce_hot_standby_feedback(self, postgresql: 'Postgresql', member: Tags) -> bool:
|
def should_enforce_hot_standby_feedback(self, postgresql: 'Postgresql', member: Tags) -> bool:
|
||||||
@@ -1168,11 +1322,15 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
|
|
||||||
if global_config.use_slots:
|
if global_config.use_slots:
|
||||||
name = member.name if isinstance(member, Member) else postgresql.name
|
name = member.name if isinstance(member, Member) else postgresql.name
|
||||||
|
|
||||||
|
if not self.get_slot_name_on_primary(name, member):
|
||||||
|
return False
|
||||||
|
|
||||||
members = [m for m in self.members if m.replicatefrom == name and m.name != self.leader_name]
|
members = [m for m in self.members if m.replicatefrom == name and m.name != self.leader_name]
|
||||||
return any(self.should_enforce_hot_standby_feedback(postgresql, m) for m in members)
|
return any(self.should_enforce_hot_standby_feedback(postgresql, m) for m in members)
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def get_slot_name_on_primary(self, name: str, tags: Tags) -> str:
|
def get_slot_name_on_primary(self, name: str, tags: Tags) -> Optional[str]:
|
||||||
"""Get the name of physical replication slot for this node on the primary.
|
"""Get the name of physical replication slot for this node on the primary.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
@@ -1186,9 +1344,18 @@ class Cluster(NamedTuple('Cluster',
|
|||||||
|
|
||||||
:returns: the slot name on the primary that is in use for physical replication on this node.
|
:returns: the slot name on the primary that is in use for physical replication on this node.
|
||||||
"""
|
"""
|
||||||
replicatefrom = self.get_member(tags.replicatefrom, False) if tags.replicatefrom else None
|
seen_nodes: Set[str] = set()
|
||||||
return self.get_slot_name_on_primary(replicatefrom.name, replicatefrom) \
|
while True:
|
||||||
if isinstance(replicatefrom, Member) else slot_name_from_member_name(name)
|
seen_nodes.add(name)
|
||||||
|
if tags.nostream:
|
||||||
|
return None
|
||||||
|
replicatefrom = self.get_member(tags.replicatefrom, False) \
|
||||||
|
if tags.replicatefrom and tags.replicatefrom != name else None
|
||||||
|
if not isinstance(replicatefrom, Member):
|
||||||
|
return slot_name_from_member_name(name)
|
||||||
|
if replicatefrom.name in seen_nodes:
|
||||||
|
return None
|
||||||
|
name, tags = replicatefrom.name, replicatefrom
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def timeline(self) -> int:
|
def timeline(self) -> int:
|
||||||
@@ -1339,8 +1506,9 @@ class AbstractDCS(abc.ABC):
|
|||||||
def __init__(self, config: Dict[str, Any], mpp: 'AbstractMPP') -> None:
|
def __init__(self, config: Dict[str, Any], mpp: 'AbstractMPP') -> None:
|
||||||
"""Prepare DCS paths, MPP object, initial values for state information and processing dependencies.
|
"""Prepare DCS paths, MPP object, initial values for state information and processing dependencies.
|
||||||
|
|
||||||
:ivar config: :class:`dict`, reference to config section of selected DCS.
|
:param config: :class:`dict`, reference to config section of selected DCS.
|
||||||
i.e.: ``zookeeper`` for zookeeper, ``etcd`` for etcd, etc...
|
i.e.: ``zookeeper`` for zookeeper, ``etcd`` for etcd, etc...
|
||||||
|
:param mpp: an object implementing :class:`AbstractMPP` interface.
|
||||||
"""
|
"""
|
||||||
self._mpp = mpp
|
self._mpp = mpp
|
||||||
self._name = config['name']
|
self._name = config['name']
|
||||||
@@ -1353,7 +1521,8 @@ class AbstractDCS(abc.ABC):
|
|||||||
self._cluster_thread_lock = Lock()
|
self._cluster_thread_lock = Lock()
|
||||||
self._last_lsn: int = 0
|
self._last_lsn: int = 0
|
||||||
self._last_seen: int = 0
|
self._last_seen: int = 0
|
||||||
self._last_status: Dict[str, Any] = {}
|
self._last_status: Dict[str, Any] = {'retain_slots': []}
|
||||||
|
self._last_retain_slots: Dict[str, float] = {}
|
||||||
self._last_failsafe: Optional[Dict[str, str]] = {}
|
self._last_failsafe: Optional[Dict[str, str]] = {}
|
||||||
self.event = Event()
|
self.event = Event()
|
||||||
|
|
||||||
@@ -1574,15 +1743,16 @@ class AbstractDCS(abc.ABC):
|
|||||||
self.reset_cluster()
|
self.reset_cluster()
|
||||||
raise
|
raise
|
||||||
|
|
||||||
self._last_seen = int(time.time())
|
|
||||||
self._last_status = {self._OPTIME: cluster.last_lsn}
|
|
||||||
if cluster.slots:
|
|
||||||
self._last_status['slots'] = cluster.slots
|
|
||||||
self._last_failsafe = cluster.failsafe
|
|
||||||
|
|
||||||
with self._cluster_thread_lock:
|
with self._cluster_thread_lock:
|
||||||
self._cluster = cluster
|
self._cluster = cluster
|
||||||
self._cluster_valid_till = time.time() + self.ttl
|
self._cluster_valid_till = time.time() + self.ttl
|
||||||
|
|
||||||
|
self._last_seen = int(time.time())
|
||||||
|
self._last_status = {self._OPTIME: cluster.status.last_lsn, 'retain_slots': cluster.status.retain_slots}
|
||||||
|
if cluster.status.slots:
|
||||||
|
self._last_status['slots'] = cluster.status.slots
|
||||||
|
self._last_failsafe = cluster.failsafe
|
||||||
|
|
||||||
return cluster
|
return cluster
|
||||||
|
|
||||||
@property
|
@property
|
||||||
@@ -1638,6 +1808,14 @@ class AbstractDCS(abc.ABC):
|
|||||||
|
|
||||||
:param value: JSON serializable dictionary with current WAL LSN and ``confirmed_flush_lsn`` of permanent slots.
|
:param value: JSON serializable dictionary with current WAL LSN and ``confirmed_flush_lsn`` of permanent slots.
|
||||||
"""
|
"""
|
||||||
|
# This method is always called with ``optime`` key, rest of the keys are optional.
|
||||||
|
# In case if we know old values (stored in self._last_status), we will copy them over.
|
||||||
|
for name in ('slots', 'retain_slots'):
|
||||||
|
if name not in value and self._last_status.get(name):
|
||||||
|
value[name] = self._last_status[name]
|
||||||
|
# if the key is present, but the value is None, we will not write such pair.
|
||||||
|
value = {k: v for k, v in value.items() if v is not None}
|
||||||
|
|
||||||
if not deep_compare(self._last_status, value) and self._write_status(json.dumps(value, separators=(',', ':'))):
|
if not deep_compare(self._last_status, value) and self._write_status(json.dumps(value, separators=(',', ':'))):
|
||||||
self._last_status = value
|
self._last_status = value
|
||||||
cluster = self.cluster
|
cluster = self.cluster
|
||||||
@@ -1669,6 +1847,49 @@ class AbstractDCS(abc.ABC):
|
|||||||
"""Stored value of :attr:`~AbstractDCS._last_failsafe`."""
|
"""Stored value of :attr:`~AbstractDCS._last_failsafe`."""
|
||||||
return self._last_failsafe
|
return self._last_failsafe
|
||||||
|
|
||||||
|
def _build_retain_slots(self, cluster: Cluster, slots: Optional[Dict[str, int]]) -> Optional[List[str]]:
|
||||||
|
"""Handle retention policy of physical replication slots for cluster members.
|
||||||
|
|
||||||
|
When the member key is missing we want to keep its replication slot for a while, so that WAL segments
|
||||||
|
will not be already absent when it comes back online. It is being solved by storing the list of
|
||||||
|
replication slots representing members in the ``retain_slots`` field of the ``/status`` key.
|
||||||
|
|
||||||
|
This method handles retention policy by keeping the list of such replication slots in memory
|
||||||
|
and removing names when they were observed longer than ``member_slots_ttl`` ago.
|
||||||
|
|
||||||
|
:param cluster: :class:`Cluster` object with information about the current cluster state.
|
||||||
|
:param slots: slot names with LSN values that exist on the leader node and consists
|
||||||
|
of slots for cluster members and permanent replication slots.
|
||||||
|
|
||||||
|
:returns: the list of replication slots to be written to ``/status`` key or ``None``.
|
||||||
|
"""
|
||||||
|
timestamp = time.time()
|
||||||
|
|
||||||
|
if slots: # if slots is not empty it implies we are running v11+
|
||||||
|
members: Set[str] = set()
|
||||||
|
found_self = False
|
||||||
|
for member in cluster.members:
|
||||||
|
found_self = member.name == self._name
|
||||||
|
if not member.nostream:
|
||||||
|
members.add(slot_name_from_member_name(member.name))
|
||||||
|
if not found_self:
|
||||||
|
# It could be that the member key for our node is not in DCS and we can't check tags.nostream.
|
||||||
|
# In this case our name will falsely appear in `retain_slots`, but only temporary.
|
||||||
|
members.add(slot_name_from_member_name(self._name))
|
||||||
|
|
||||||
|
permanent_slots = cluster.permanent_physical_slots
|
||||||
|
# we want to have in ``retain_slots`` only non-permanent member slots
|
||||||
|
self._last_retain_slots.update({name: timestamp for name in slots
|
||||||
|
if name in members and name not in permanent_slots})
|
||||||
|
# retention
|
||||||
|
for name, value in list(self._last_retain_slots.items()):
|
||||||
|
if value + global_config.member_slots_ttl <= timestamp:
|
||||||
|
logger.info("Replication slot '%s' for absent cluster member is expired after %d sec.",
|
||||||
|
name, global_config.member_slots_ttl)
|
||||||
|
del self._last_retain_slots[name]
|
||||||
|
|
||||||
|
return list(sorted(self._last_retain_slots.keys())) or None
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
def _update_leader(self, leader: Leader) -> bool:
|
def _update_leader(self, leader: Leader) -> bool:
|
||||||
"""Update ``leader`` key (or session) ttl.
|
"""Update ``leader`` key (or session) ttl.
|
||||||
@@ -1686,24 +1907,25 @@ class AbstractDCS(abc.ABC):
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
def update_leader(self,
|
def update_leader(self,
|
||||||
leader: Leader,
|
cluster: Cluster,
|
||||||
last_lsn: Optional[int],
|
last_lsn: Optional[int],
|
||||||
slots: Optional[Dict[str, int]] = None,
|
slots: Optional[Dict[str, int]] = None,
|
||||||
failsafe: Optional[Dict[str, str]] = None) -> bool:
|
failsafe: Optional[Dict[str, str]] = None) -> bool:
|
||||||
"""Update ``leader`` key (or session) ttl and optime/leader.
|
"""Update ``leader`` key (or session) ttl, ``/status``, and ``/failsafe`` keys.
|
||||||
|
|
||||||
:param leader: :class:`Leader` object with information about the leader.
|
:param cluster: :class:`Cluster` object with information about the current cluster state.
|
||||||
:param last_lsn: absolute WAL LSN in bytes.
|
:param last_lsn: absolute WAL LSN in bytes.
|
||||||
:param slots: dictionary with permanent slots ``confirmed_flush_lsn``.
|
:param slots: dictionary with permanent slots ``confirmed_flush_lsn``.
|
||||||
:param failsafe: if defined dictionary passed to :meth:`~AbstractDCS.write_failsafe`.
|
:param failsafe: if defined dictionary passed to :meth:`~AbstractDCS.write_failsafe`.
|
||||||
|
|
||||||
:returns: ``True`` if ``leader`` key (or session) has been updated successfully.
|
:returns: ``True`` if ``leader`` key (or session) has been updated successfully.
|
||||||
"""
|
"""
|
||||||
ret = self._update_leader(leader)
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
|
assert isinstance(cluster.leader, Leader)
|
||||||
|
ret = self._update_leader(cluster.leader)
|
||||||
if ret and last_lsn:
|
if ret and last_lsn:
|
||||||
status: Dict[str, Any] = {self._OPTIME: last_lsn}
|
status: Dict[str, Any] = {self._OPTIME: last_lsn, 'slots': slots or None,
|
||||||
if slots:
|
'retain_slots': self._build_retain_slots(cluster, slots)}
|
||||||
status['slots'] = slots
|
|
||||||
self.write_status(status)
|
self.write_status(status)
|
||||||
|
|
||||||
if ret and failsafe is not None:
|
if ret and failsafe is not None:
|
||||||
@@ -1727,6 +1949,23 @@ class AbstractDCS(abc.ABC):
|
|||||||
:returns: ``True`` if key has been created successfully.
|
:returns: ``True`` if key has been created successfully.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
def acquire_leader_lock(self) -> bool:
|
||||||
|
"""Attempt to acquire leader lock.
|
||||||
|
|
||||||
|
.. note::
|
||||||
|
This method wraps :meth:`~AbstractDCS.attempt_to_acquire_leader`: and is
|
||||||
|
used to reset retention time of physical replication slots that representing
|
||||||
|
members of the cluster when current node is to be promoted to the leader.
|
||||||
|
|
||||||
|
:returns: ``True`` if the leader key has been created successfully.
|
||||||
|
"""
|
||||||
|
ret = self.attempt_to_acquire_leader()
|
||||||
|
if ret:
|
||||||
|
timestamp = time.time()
|
||||||
|
# every time we promote we need to reset retention time for slots recorded in the /status key
|
||||||
|
self._last_retain_slots = {name: timestamp for name in self._last_status['retain_slots']}
|
||||||
|
return ret
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
def set_failover_value(self, value: str, version: Optional[Any] = None) -> bool:
|
def set_failover_value(self, value: str, version: Optional[Any] = None) -> bool:
|
||||||
"""Create or update ``/failover`` key.
|
"""Create or update ``/failover`` key.
|
||||||
@@ -1849,18 +2088,23 @@ class AbstractDCS(abc.ABC):
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def sync_state(leader: Optional[str], sync_standby: Optional[Collection[str]]) -> Dict[str, Any]:
|
def sync_state(leader: Optional[str], sync_standby: Optional[Collection[str]],
|
||||||
|
quorum: Optional[int]) -> Dict[str, Any]:
|
||||||
"""Build ``sync_state`` dictionary.
|
"""Build ``sync_state`` dictionary.
|
||||||
|
|
||||||
:param leader: name of the leader node that manages ``/sync`` key.
|
:param leader: name of the leader node that manages ``/sync`` key.
|
||||||
:param sync_standby: collection of currently known synchronous standby node names.
|
:param sync_standby: collection of currently known synchronous standby node names.
|
||||||
|
:param quorum: if the node from :attr:`~SyncState.sync_standby` list is doing a leader race it should
|
||||||
|
see at least :attr:`~SyncState.quorum` other nodes from the
|
||||||
|
:attr:`~SyncState.sync_standby` + :attr:`~SyncState.leader` list
|
||||||
|
|
||||||
:returns: dictionary that later could be serialized to JSON or saved directly to DCS.
|
:returns: dictionary that later could be serialized to JSON or saved directly to DCS.
|
||||||
"""
|
"""
|
||||||
return {'leader': leader, 'sync_standby': ','.join(sorted(sync_standby)) if sync_standby else None}
|
return {'leader': leader, 'quorum': quorum,
|
||||||
|
'sync_standby': ','.join(sorted(sync_standby)) if sync_standby else None}
|
||||||
|
|
||||||
def write_sync_state(self, leader: Optional[str], sync_standby: Optional[Collection[str]],
|
def write_sync_state(self, leader: Optional[str], sync_standby: Optional[Collection[str]],
|
||||||
version: Optional[Any] = None) -> Optional[SyncState]:
|
quorum: Optional[int], version: Optional[Any] = None) -> Optional[SyncState]:
|
||||||
"""Write the new synchronous state to DCS.
|
"""Write the new synchronous state to DCS.
|
||||||
|
|
||||||
Calls :meth:`~AbstractDCS.sync_state` to build a dictionary and then calls DCS specific
|
Calls :meth:`~AbstractDCS.sync_state` to build a dictionary and then calls DCS specific
|
||||||
@@ -1869,10 +2113,13 @@ class AbstractDCS(abc.ABC):
|
|||||||
:param leader: name of the leader node that manages ``/sync`` key.
|
:param leader: name of the leader node that manages ``/sync`` key.
|
||||||
:param sync_standby: collection of currently known synchronous standby node names.
|
:param sync_standby: collection of currently known synchronous standby node names.
|
||||||
:param version: for conditional update of the key/object.
|
:param version: for conditional update of the key/object.
|
||||||
|
:param quorum: if the node from :attr:`~SyncState.sync_standby` list is doing a leader race it should
|
||||||
|
see at least :attr:`~SyncState.quorum` other nodes from the
|
||||||
|
:attr:`~SyncState.sync_standby` + :attr:`~SyncState.leader` list
|
||||||
|
|
||||||
:returns: the new :class:`SyncState` object or ``None``.
|
:returns: the new :class:`SyncState` object or ``None``.
|
||||||
"""
|
"""
|
||||||
sync_value = self.sync_state(leader, sync_standby)
|
sync_value = self.sync_state(leader, sync_standby, quorum)
|
||||||
ret = self.set_sync_state_value(json.dumps(sync_value, separators=(',', ':')), version)
|
ret = self.set_sync_state_value(json.dumps(sync_value, separators=(',', ':')), version)
|
||||||
if not isinstance(ret, bool):
|
if not isinstance(ret, bool):
|
||||||
return SyncState.from_node(ret, sync_value)
|
return SyncState.from_node(ret, sync_value)
|
||||||
|
|||||||
+28
-21
@@ -1,4 +1,5 @@
|
|||||||
from __future__ import absolute_import
|
from __future__ import absolute_import
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
@@ -6,20 +7,24 @@ import re
|
|||||||
import socket
|
import socket
|
||||||
import ssl
|
import ssl
|
||||||
import time
|
import time
|
||||||
import urllib3
|
|
||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from consul import ConsulException, NotFound, base
|
|
||||||
from http.client import HTTPException
|
from http.client import HTTPException
|
||||||
|
from typing import Any, Callable, Dict, List, Mapping, NamedTuple, Optional, Tuple, TYPE_CHECKING, Union
|
||||||
|
from urllib.parse import quote, urlencode, urlparse
|
||||||
|
|
||||||
|
import urllib3
|
||||||
|
|
||||||
|
from consul import base, Check, ConsulException, NotFound
|
||||||
from urllib3.exceptions import HTTPError
|
from urllib3.exceptions import HTTPError
|
||||||
from urllib.parse import urlencode, urlparse, quote
|
|
||||||
from typing import Any, Callable, Dict, List, Mapping, NamedTuple, Optional, Union, Tuple, TYPE_CHECKING
|
|
||||||
|
|
||||||
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member, Status, SyncState, \
|
|
||||||
TimelineHistory, ReturnFalseException, catch_return_false_exception
|
|
||||||
from ..exceptions import DCSError
|
from ..exceptions import DCSError
|
||||||
|
from ..postgresql.misc import PostgresqlRole, PostgresqlState
|
||||||
from ..postgresql.mpp import AbstractMPP
|
from ..postgresql.mpp import AbstractMPP
|
||||||
from ..utils import deep_compare, parse_bool, Retry, RetryFailedError, split_host_port, uri, USER_AGENT
|
from ..utils import deep_compare, parse_bool, Retry, RetryFailedError, split_host_port, uri, USER_AGENT
|
||||||
|
from . import AbstractDCS, catch_return_false_exception, Cluster, ClusterConfig, \
|
||||||
|
Failover, Leader, Member, ReturnFalseException, Status, SyncState, TimelineHistory
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from ..config import Config
|
from ..config import Config
|
||||||
|
|
||||||
@@ -188,7 +193,7 @@ class ConsulClient(base.Consul):
|
|||||||
return HTTPClient(**kwargs)
|
return HTTPClient(**kwargs)
|
||||||
|
|
||||||
def connect(self, *args: Any, **kwargs: Any) -> HTTPClient:
|
def connect(self, *args: Any, **kwargs: Any) -> HTTPClient:
|
||||||
return self.http_connect(*args, **kwargs)
|
return self.http_connect(*args, **kwargs) # pragma: no cover
|
||||||
|
|
||||||
def reload_config(self, config: Dict[str, Any]) -> None:
|
def reload_config(self, config: Dict[str, Any]) -> None:
|
||||||
self.http.token = self.token = config.get('token')
|
self.http.token = self.token = config.get('token')
|
||||||
@@ -444,8 +449,9 @@ class Consul(AbstractDCS):
|
|||||||
|
|
||||||
:returns: all MPP groups as :class:`dict`, with group IDs as keys and :class:`Cluster` objects as values.
|
:returns: all MPP groups as :class:`dict`, with group IDs as keys and :class:`Cluster` objects as values.
|
||||||
"""
|
"""
|
||||||
|
results: Optional[List[Dict[str, Any]]]
|
||||||
_, results = self.retry(self._client.kv.get, path, recurse=True, consistency=self._consistency)
|
_, results = self.retry(self._client.kv.get, path, recurse=True, consistency=self._consistency)
|
||||||
clusters: Dict[int, Dict[str, Cluster]] = defaultdict(dict)
|
clusters: Dict[int, Dict[str, Dict[str, Any]]] = defaultdict(dict)
|
||||||
for node in results or []:
|
for node in results or []:
|
||||||
key = node['Key'][len(path):].split('/', 1)
|
key = node['Key'][len(path):].split('/', 1)
|
||||||
if len(key) == 2 and self._mpp.group_re.match(key[0]):
|
if len(key) == 2 and self._mpp.group_re.match(key[0]):
|
||||||
@@ -521,16 +527,14 @@ class Consul(AbstractDCS):
|
|||||||
api_parts = api_parts._replace(path='/{0}'.format(role))
|
api_parts = api_parts._replace(path='/{0}'.format(role))
|
||||||
conn_url: str = data['conn_url']
|
conn_url: str = data['conn_url']
|
||||||
conn_parts = urlparse(conn_url)
|
conn_parts = urlparse(conn_url)
|
||||||
check = base.Check.http(api_parts.geturl(), self._service_check_interval,
|
check = Check.http(api_parts.geturl(), self._service_check_interval,
|
||||||
deregister='{0}s'.format(self._client.http.ttl * 10))
|
deregister='{0}s'.format(self._client.http.ttl * 10))
|
||||||
if self._service_check_tls_server_name is not None:
|
if self._service_check_tls_server_name is not None:
|
||||||
check['TLSServerName'] = self._service_check_tls_server_name
|
check['TLSServerName'] = self._service_check_tls_server_name
|
||||||
tags = self._service_tags[:]
|
tags = self._service_tags[:]
|
||||||
tags.append(role)
|
tags.append(role)
|
||||||
if role == 'master':
|
if data['role'] == PostgresqlRole.PRIMARY:
|
||||||
tags.append('primary')
|
tags.append(PostgresqlRole.MASTER)
|
||||||
elif role == 'primary':
|
|
||||||
tags.append('master')
|
|
||||||
self._previous_loop_service_tags = self._service_tags
|
self._previous_loop_service_tags = self._service_tags
|
||||||
self._previous_loop_token = self._client.token
|
self._previous_loop_token = self._client.token
|
||||||
|
|
||||||
@@ -543,13 +547,13 @@ class Consul(AbstractDCS):
|
|||||||
'enable_tag_override': True,
|
'enable_tag_override': True,
|
||||||
}
|
}
|
||||||
|
|
||||||
if state == 'stopped' or (not self._register_service and self._previous_loop_register_service):
|
if state == PostgresqlState.STOPPED or (not self._register_service and self._previous_loop_register_service):
|
||||||
self._previous_loop_register_service = self._register_service
|
self._previous_loop_register_service = self._register_service
|
||||||
return self.deregister_service(params['service_id'])
|
return self.deregister_service(params['service_id'])
|
||||||
|
|
||||||
self._previous_loop_register_service = self._register_service
|
self._previous_loop_register_service = self._register_service
|
||||||
if role in ['master', 'primary', 'replica', 'standby-leader']:
|
if data['role'] in [PostgresqlRole.PRIMARY, PostgresqlRole.REPLICA, PostgresqlRole.STANDBY_LEADER]:
|
||||||
if state != 'running':
|
if state != PostgresqlState.RUNNING:
|
||||||
return
|
return
|
||||||
return self.register_service(service_name, **params)
|
return self.register_service(service_name, **params)
|
||||||
|
|
||||||
@@ -577,14 +581,17 @@ class Consul(AbstractDCS):
|
|||||||
try:
|
try:
|
||||||
return retry(self._client.kv.put, self.leader_path, self._name, acquire=self._session)
|
return retry(self._client.kv.put, self.leader_path, self._name, acquire=self._session)
|
||||||
except InvalidSession:
|
except InvalidSession:
|
||||||
logger.error('Our session disappeared from Consul. Will try to get a new one and retry attempt')
|
|
||||||
self._session = None
|
self._session = None
|
||||||
retry.ensure_deadline(0)
|
|
||||||
|
if not retry.ensure_deadline(0):
|
||||||
|
logger.error('Our session disappeared from Consul. Deadline exceeded, giving up')
|
||||||
|
return False
|
||||||
|
|
||||||
|
logger.error('Our session disappeared from Consul. Will try to get a new one and retry attempt')
|
||||||
|
|
||||||
retry(self._do_refresh_session)
|
retry(self._do_refresh_session)
|
||||||
|
|
||||||
retry.ensure_deadline(1, ConsulError('_do_attempt_to_acquire_leader timeout'))
|
retry.ensure_deadline(1, ConsulError('_do_attempt_to_acquire_leader timeout'))
|
||||||
|
|
||||||
return retry(self._client.kv.put, self.leader_path, self._name, acquire=self._session)
|
return retry(self._client.kv.put, self.leader_path, self._name, acquire=self._session)
|
||||||
|
|
||||||
@catch_return_false_exception
|
@catch_return_false_exception
|
||||||
@@ -676,7 +683,7 @@ class Consul(AbstractDCS):
|
|||||||
def set_sync_state_value(self, value: str, version: Optional[int] = None) -> Union[int, bool]:
|
def set_sync_state_value(self, value: str, version: Optional[int] = None) -> Union[int, bool]:
|
||||||
retry = self._retry.copy()
|
retry = self._retry.copy()
|
||||||
ret = retry(self._client.kv.put, self.sync_path, value, cas=version)
|
ret = retry(self._client.kv.put, self.sync_path, value, cas=version)
|
||||||
if ret: # We have no other choise, only read after write :(
|
if ret: # We have no other choice, only read after write :(
|
||||||
if not retry.ensure_deadline(0.5):
|
if not retry.ensure_deadline(0.5):
|
||||||
return False
|
return False
|
||||||
_, ret = self.retry(self._client.kv.get, self.sync_path, consistency='consistent')
|
_, ret = self.retry(self._client.kv.get, self.sync_path, consistency='consistent')
|
||||||
|
|||||||
+58
-16
@@ -1,32 +1,36 @@
|
|||||||
from __future__ import absolute_import
|
from __future__ import absolute_import
|
||||||
|
|
||||||
import abc
|
import abc
|
||||||
import etcd
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import urllib3.util.connection
|
|
||||||
import random
|
import random
|
||||||
import socket
|
import socket
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
from dns.exception import DNSException
|
|
||||||
from dns import resolver
|
|
||||||
from http.client import HTTPException
|
from http.client import HTTPException
|
||||||
from queue import Queue
|
from queue import Queue
|
||||||
from threading import Thread
|
from threading import Thread
|
||||||
from typing import Any, Callable, Collection, Dict, List, Optional, Union, Tuple, Type, TYPE_CHECKING
|
from typing import Any, Callable, Collection, Dict, List, Optional, Tuple, Type, TYPE_CHECKING, Union
|
||||||
from urllib.parse import urlparse
|
from urllib.parse import urlparse
|
||||||
from urllib3 import Timeout
|
|
||||||
from urllib3.exceptions import HTTPError, ReadTimeoutError, ProtocolError
|
|
||||||
|
|
||||||
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member, Status, SyncState, \
|
import etcd
|
||||||
TimelineHistory, ReturnFalseException, catch_return_false_exception
|
import urllib3.util.connection
|
||||||
|
|
||||||
|
from dns import resolver
|
||||||
|
from dns.exception import DNSException
|
||||||
|
from urllib3 import Timeout
|
||||||
|
from urllib3.exceptions import HTTPError, ProtocolError, ReadTimeoutError
|
||||||
|
|
||||||
from ..exceptions import DCSError
|
from ..exceptions import DCSError
|
||||||
from ..postgresql.mpp import AbstractMPP
|
from ..postgresql.mpp import AbstractMPP
|
||||||
from ..request import get as requests_get
|
from ..request import get as requests_get
|
||||||
from ..utils import Retry, RetryFailedError, split_host_port, uri, USER_AGENT
|
from ..utils import Retry, RetryFailedError, split_host_port, uri, USER_AGENT
|
||||||
|
from . import AbstractDCS, catch_return_false_exception, Cluster, ClusterConfig, \
|
||||||
|
Failover, Leader, Member, ReturnFalseException, Status, SyncState, TimelineHistory
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from ..config import Config
|
from ..config import Config
|
||||||
|
|
||||||
@@ -37,11 +41,16 @@ class EtcdRaftInternal(etcd.EtcdException):
|
|||||||
"""Raft Internal Error"""
|
"""Raft Internal Error"""
|
||||||
|
|
||||||
|
|
||||||
|
class StaleEtcdNode(Exception):
|
||||||
|
"""Node is stale (raft term is older than previous known)."""
|
||||||
|
|
||||||
|
|
||||||
class EtcdError(DCSError):
|
class EtcdError(DCSError):
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
||||||
_AddrInfo = Tuple[socket.AddressFamily, socket.SocketKind, int, str, Union[Tuple[str, int], Tuple[str, int, int, int]]]
|
_AddrInfo = Tuple[socket.AddressFamily, socket.SocketKind, int, str,
|
||||||
|
Union[Tuple[str, int], Tuple[str, int, int, int], Tuple[int, bytes]]]
|
||||||
|
|
||||||
|
|
||||||
class DnsCachingResolver(Thread):
|
class DnsCachingResolver(Thread):
|
||||||
@@ -97,6 +106,8 @@ class AbstractEtcdClientWithFailover(abc.ABC, etcd.Client):
|
|||||||
ERROR_CLS: Type[Exception]
|
ERROR_CLS: Type[Exception]
|
||||||
|
|
||||||
def __init__(self, config: Dict[str, Any], dns_resolver: DnsCachingResolver, cache_ttl: int = 300) -> None:
|
def __init__(self, config: Dict[str, Any], dns_resolver: DnsCachingResolver, cache_ttl: int = 300) -> None:
|
||||||
|
self._cluster_id = None
|
||||||
|
self._raft_term = 0
|
||||||
self._dns_resolver = dns_resolver
|
self._dns_resolver = dns_resolver
|
||||||
self.set_machines_cache_ttl(cache_ttl)
|
self.set_machines_cache_ttl(cache_ttl)
|
||||||
self._machines_cache_updated = 0
|
self._machines_cache_updated = 0
|
||||||
@@ -114,6 +125,32 @@ class AbstractEtcdClientWithFailover(abc.ABC, etcd.Client):
|
|||||||
self._read_options.add('retry')
|
self._read_options.add('retry')
|
||||||
self._del_conditions.add('retry')
|
self._del_conditions.add('retry')
|
||||||
|
|
||||||
|
def _check_cluster_raft_term(self, cluster_id: Optional[str], value: Union[None, str, int]) -> None:
|
||||||
|
"""Check that observed Raft Term in Etcd cluster is increasing.
|
||||||
|
|
||||||
|
If we observe that the new value is smaller than the previously known one, it could be an
|
||||||
|
indicator that we connected to a stale node and should switch to some other node.
|
||||||
|
However, we need to reset the memorized value when we notice that Cluster ID changed.
|
||||||
|
"""
|
||||||
|
if not (cluster_id and value):
|
||||||
|
return
|
||||||
|
|
||||||
|
if self._cluster_id and self._cluster_id != cluster_id:
|
||||||
|
logger.warning('Etcd Cluster ID changed from %s to %s', self._cluster_id, cluster_id)
|
||||||
|
self._raft_term = 0
|
||||||
|
self._cluster_id = cluster_id
|
||||||
|
|
||||||
|
try:
|
||||||
|
raft_term = int(value)
|
||||||
|
except Exception:
|
||||||
|
return
|
||||||
|
|
||||||
|
if raft_term < self._raft_term:
|
||||||
|
logger.warning('Connected to Etcd node with term %d. Old known term %d. Switching to another node.',
|
||||||
|
raft_term, self._raft_term)
|
||||||
|
raise StaleEtcdNode
|
||||||
|
self._raft_term = raft_term
|
||||||
|
|
||||||
def _calculate_timeouts(self, etcd_nodes: int, timeout: Optional[float] = None) -> Tuple[int, float, int]:
|
def _calculate_timeouts(self, etcd_nodes: int, timeout: Optional[float] = None) -> Tuple[int, float, int]:
|
||||||
"""Calculate a request timeout and number of retries per single etcd node.
|
"""Calculate a request timeout and number of retries per single etcd node.
|
||||||
In case if the timeout per node is too small (less than one second) we will reduce the number of nodes.
|
In case if the timeout per node is too small (less than one second) we will reduce the number of nodes.
|
||||||
@@ -222,7 +259,7 @@ class AbstractEtcdClientWithFailover(abc.ABC, etcd.Client):
|
|||||||
def _do_http_request(self, retry: Optional[Retry], machines_cache: List[str],
|
def _do_http_request(self, retry: Optional[Retry], machines_cache: List[str],
|
||||||
request_executor: Callable[..., urllib3.response.HTTPResponse],
|
request_executor: Callable[..., urllib3.response.HTTPResponse],
|
||||||
method: str, path: str, fields: Optional[Dict[str, Any]] = None,
|
method: str, path: str, fields: Optional[Dict[str, Any]] = None,
|
||||||
**kwargs: Any) -> urllib3.response.HTTPResponse:
|
**kwargs: Any) -> Any:
|
||||||
is_watch_request = isinstance(fields, dict) and fields.get('wait') == 'true'
|
is_watch_request = isinstance(fields, dict) and fields.get('wait') == 'true'
|
||||||
if fields is not None:
|
if fields is not None:
|
||||||
kwargs['fields'] = fields
|
kwargs['fields'] = fields
|
||||||
@@ -236,8 +273,8 @@ class AbstractEtcdClientWithFailover(abc.ABC, etcd.Client):
|
|||||||
if some_request_failed:
|
if some_request_failed:
|
||||||
self.set_base_uri(base_uri)
|
self.set_base_uri(base_uri)
|
||||||
self._refresh_machines_cache()
|
self._refresh_machines_cache()
|
||||||
return response
|
return self._handle_server_response(response)
|
||||||
except (HTTPError, HTTPException, socket.error, socket.timeout) as e:
|
except (HTTPError, HTTPException, socket.error, socket.timeout, StaleEtcdNode) as e:
|
||||||
self.http.clear()
|
self.http.clear()
|
||||||
if not retry:
|
if not retry:
|
||||||
if len(machines_cache) == 1:
|
if len(machines_cache) == 1:
|
||||||
@@ -280,8 +317,7 @@ class AbstractEtcdClientWithFailover(abc.ABC, etcd.Client):
|
|||||||
|
|
||||||
while True:
|
while True:
|
||||||
try:
|
try:
|
||||||
response = self._do_http_request(retry, machines_cache, request_executor, method, path, **kwargs)
|
return self._do_http_request(retry, machines_cache, request_executor, method, path, **kwargs)
|
||||||
return self._handle_server_response(response)
|
|
||||||
except etcd.EtcdWatchTimedOut:
|
except etcd.EtcdWatchTimedOut:
|
||||||
raise
|
raise
|
||||||
except etcd.EtcdConnectionFailed as ex:
|
except etcd.EtcdConnectionFailed as ex:
|
||||||
@@ -346,7 +382,9 @@ class AbstractEtcdClientWithFailover(abc.ABC, etcd.Client):
|
|||||||
def _get_machines_cache_from_dns(self, host: str, port: int) -> List[str]:
|
def _get_machines_cache_from_dns(self, host: str, port: int) -> List[str]:
|
||||||
"""One host might be resolved into multiple ip addresses. We will make list out of it"""
|
"""One host might be resolved into multiple ip addresses. We will make list out of it"""
|
||||||
if self.protocol == 'http':
|
if self.protocol == 'http':
|
||||||
ret = [uri(self.protocol, res[-1][:2]) for res in self._dns_resolver.resolve(host, port)]
|
# Filter out unexpected results when python is compiled with --disable-ipv6 and running on IPv6 system.
|
||||||
|
ret = [uri(self.protocol, (res[4][0], res[4][1])) for res in self._dns_resolver.resolve(host, port)
|
||||||
|
if isinstance(res[4][0], str) and isinstance(res[4][1], int)]
|
||||||
if ret:
|
if ret:
|
||||||
return list(set(ret))
|
return list(set(ret))
|
||||||
return [uri(self.protocol, (host, port))]
|
return [uri(self.protocol, (host, port))]
|
||||||
@@ -456,6 +494,10 @@ class EtcdClient(AbstractEtcdClientWithFailover):
|
|||||||
def _prepare_get_members(self, etcd_nodes: int) -> Dict[str, Any]:
|
def _prepare_get_members(self, etcd_nodes: int) -> Dict[str, Any]:
|
||||||
return self._prepare_common_parameters(etcd_nodes)
|
return self._prepare_common_parameters(etcd_nodes)
|
||||||
|
|
||||||
|
def _handle_server_response(self, response: urllib3.response.HTTPResponse) -> Any:
|
||||||
|
self._check_cluster_raft_term(response.headers.get('x-etcd-cluster-id'), response.headers.get('x-raft-term'))
|
||||||
|
return super(EtcdClient, self)._handle_server_response(response)
|
||||||
|
|
||||||
def _get_members(self, base_uri: str, **kwargs: Any) -> List[str]:
|
def _get_members(self, base_uri: str, **kwargs: Any) -> List[str]:
|
||||||
response = self.http.request(self._MGET, base_uri + self.version_prefix + '/machines', **kwargs)
|
response = self.http.request(self._MGET, base_uri + self.version_prefix + '/machines', **kwargs)
|
||||||
data = self._handle_server_response(response).data.decode('utf-8')
|
data = self._handle_server_response(response).data.decode('utf-8')
|
||||||
|
|||||||
+32
-36
@@ -1,26 +1,30 @@
|
|||||||
from __future__ import absolute_import
|
from __future__ import absolute_import
|
||||||
|
|
||||||
import base64
|
import base64
|
||||||
import etcd
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import socket
|
import socket
|
||||||
import sys
|
import sys
|
||||||
import time
|
import time
|
||||||
import urllib3
|
|
||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from enum import IntEnum
|
from enum import IntEnum
|
||||||
from urllib3.exceptions import ReadTimeoutError, ProtocolError
|
|
||||||
from threading import Condition, Lock, Thread
|
from threading import Condition, Lock, Thread
|
||||||
from typing import Any, Callable, Collection, Dict, Iterator, List, Optional, Tuple, Type, TYPE_CHECKING, Union
|
from typing import Any, Callable, Collection, Dict, Iterator, List, Optional, Tuple, Type, TYPE_CHECKING, Union
|
||||||
|
|
||||||
from . import ClusterConfig, Cluster, Failover, Leader, Member, Status, SyncState, \
|
import etcd
|
||||||
TimelineHistory, catch_return_false_exception
|
import urllib3
|
||||||
from .etcd import AbstractEtcdClientWithFailover, AbstractEtcd, catch_etcd_errors, DnsCachingResolver, Retry
|
|
||||||
|
from urllib3.exceptions import ProtocolError, ReadTimeoutError
|
||||||
|
|
||||||
|
from ..collections import EMPTY_DICT
|
||||||
from ..exceptions import DCSError, PatroniException
|
from ..exceptions import DCSError, PatroniException
|
||||||
from ..postgresql.mpp import AbstractMPP
|
from ..postgresql.mpp import AbstractMPP
|
||||||
from ..utils import deep_compare, enable_keepalive, iter_response_objects, RetryFailedError, USER_AGENT
|
from ..utils import deep_compare, enable_keepalive, iter_response_objects, RetryFailedError, USER_AGENT
|
||||||
|
from . import catch_return_false_exception, Cluster, ClusterConfig, \
|
||||||
|
Failover, Leader, Member, Status, SyncState, TimelineHistory
|
||||||
|
from .etcd import AbstractEtcd, AbstractEtcdClientWithFailover, catch_etcd_errors, DnsCachingResolver, Retry
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -147,8 +151,7 @@ errStringToClientError = {getattr(s, 'error'): s for s in Etcd3ClientError.get_s
|
|||||||
errCodeToClientError = {getattr(s, 'code'): s for s in Etcd3ClientError.__subclasses__()}
|
errCodeToClientError = {getattr(s, 'code'): s for s in Etcd3ClientError.__subclasses__()}
|
||||||
|
|
||||||
|
|
||||||
def _raise_for_data(data: Union[bytes, str, Dict[str, Union[Any, Dict[str, Any]]]],
|
def _raise_for_data(data: Union[bytes, str, Dict[str, Any]], status_code: Optional[int] = None) -> Etcd3ClientError:
|
||||||
status_code: Optional[int] = None) -> Etcd3ClientError:
|
|
||||||
try:
|
try:
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
assert isinstance(data, dict)
|
assert isinstance(data, dict)
|
||||||
@@ -198,12 +201,6 @@ def build_range_request(key: str, range_end: Union[bytes, str, None] = None) ->
|
|||||||
return fields
|
return fields
|
||||||
|
|
||||||
|
|
||||||
class ReAuthenticateMode(IntEnum):
|
|
||||||
NOT_REQUIRED = 0
|
|
||||||
REQUIRED = 1
|
|
||||||
WITHOUT_WATCHER_RESTART = 2
|
|
||||||
|
|
||||||
|
|
||||||
def _handle_auth_errors(func: Callable[..., Any]) -> Any:
|
def _handle_auth_errors(func: Callable[..., Any]) -> Any:
|
||||||
def wrapper(self: 'Etcd3Client', *args: Any, **kwargs: Any) -> Any:
|
def wrapper(self: 'Etcd3Client', *args: Any, **kwargs: Any) -> Any:
|
||||||
return self.handle_auth_errors(func, *args, **kwargs)
|
return self.handle_auth_errors(func, *args, **kwargs)
|
||||||
@@ -215,7 +212,7 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
|
|||||||
ERROR_CLS = Etcd3Error
|
ERROR_CLS = Etcd3Error
|
||||||
|
|
||||||
def __init__(self, config: Dict[str, Any], dns_resolver: DnsCachingResolver, cache_ttl: int = 300) -> None:
|
def __init__(self, config: Dict[str, Any], dns_resolver: DnsCachingResolver, cache_ttl: int = 300) -> None:
|
||||||
self._reauthenticate_reason = ReAuthenticateMode.NOT_REQUIRED
|
self._reauthenticate = False
|
||||||
self._token = None
|
self._token = None
|
||||||
self._cluster_version: Tuple[int, ...] = tuple()
|
self._cluster_version: Tuple[int, ...] = tuple()
|
||||||
super(Etcd3Client, self).__init__({**config, 'version_prefix': '/v3beta'}, dns_resolver, cache_ttl)
|
super(Etcd3Client, self).__init__({**config, 'version_prefix': '/v3beta'}, dns_resolver, cache_ttl)
|
||||||
@@ -244,6 +241,10 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
|
|||||||
try:
|
try:
|
||||||
data = data.decode('utf-8')
|
data = data.decode('utf-8')
|
||||||
ret: Dict[str, Any] = json.loads(data)
|
ret: Dict[str, Any] = json.loads(data)
|
||||||
|
|
||||||
|
header = ret.get('header', EMPTY_DICT)
|
||||||
|
self._check_cluster_raft_term(header.get('cluster_id'), header.get('raft_term'))
|
||||||
|
|
||||||
if response.status < 400:
|
if response.status < 400:
|
||||||
return ret
|
return ret
|
||||||
except (TypeError, ValueError, UnicodeError) as e:
|
except (TypeError, ValueError, UnicodeError) as e:
|
||||||
@@ -294,7 +295,7 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
|
|||||||
fields['retry'] = retry
|
fields['retry'] = retry
|
||||||
return self.api_execute(self.version_prefix + method, self._MPOST, fields)
|
return self.api_execute(self.version_prefix + method, self._MPOST, fields)
|
||||||
|
|
||||||
def authenticate(self, *, restart_watcher: bool = True, retry: Optional[Retry] = None) -> bool:
|
def authenticate(self, *, retry: Optional[Retry] = None) -> bool:
|
||||||
if self._use_proxies and not self._cluster_version:
|
if self._use_proxies and not self._cluster_version:
|
||||||
kwargs = self._prepare_common_parameters(1)
|
kwargs = self._prepare_common_parameters(1)
|
||||||
self._ensure_version_prefix(self._base_uri, **kwargs)
|
self._ensure_version_prefix(self._base_uri, **kwargs)
|
||||||
@@ -316,20 +317,18 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
|
|||||||
|
|
||||||
def handle_auth_errors(self: 'Etcd3Client', func: Callable[..., Any], *args: Any,
|
def handle_auth_errors(self: 'Etcd3Client', func: Callable[..., Any], *args: Any,
|
||||||
retry: Optional[Retry] = None, **kwargs: Any) -> Any:
|
retry: Optional[Retry] = None, **kwargs: Any) -> Any:
|
||||||
|
reauthenticated = False
|
||||||
exc = None
|
exc = None
|
||||||
while True:
|
while True:
|
||||||
if self._reauthenticate_reason:
|
if self._reauthenticate:
|
||||||
if self.username and self.password:
|
if self.username and self.password:
|
||||||
self.authenticate(
|
self.authenticate(retry=retry)
|
||||||
restart_watcher=self._reauthenticate_reason != ReAuthenticateMode.WITHOUT_WATCHER_RESTART,
|
self._reauthenticate = False
|
||||||
retry=retry)
|
|
||||||
self._reauthenticate_reason = ReAuthenticateMode.NOT_REQUIRED
|
|
||||||
if retry:
|
|
||||||
retry.ensure_deadline(0)
|
|
||||||
else:
|
else:
|
||||||
msg = 'Username or password not set, authentication is not possible'
|
msg = 'Username or password not set, authentication is not possible'
|
||||||
logger.fatal(msg)
|
logger.fatal(msg)
|
||||||
raise exc or Etcd3Exception(msg)
|
raise exc or Etcd3Exception(msg)
|
||||||
|
reauthenticated = True
|
||||||
|
|
||||||
try:
|
try:
|
||||||
return func(self, *args, retry=retry, **kwargs)
|
return func(self, *args, retry=retry, **kwargs)
|
||||||
@@ -347,11 +346,12 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
|
|||||||
except AuthOldRevision as e:
|
except AuthOldRevision as e:
|
||||||
logger.error('Auth token is for old revision of auth store')
|
logger.error('Auth token is for old revision of auth store')
|
||||||
exc = e
|
exc = e
|
||||||
self._reauthenticate_reason = ReAuthenticateMode.WITHOUT_WATCHER_RESTART \
|
self._reauthenticate = True
|
||||||
if isinstance(exc, AuthOldRevision) else ReAuthenticateMode.REQUIRED
|
if retry:
|
||||||
if not retry:
|
logger.error('retry = %s', retry)
|
||||||
|
retry.ensure_deadline(0.5, exc)
|
||||||
|
elif reauthenticated:
|
||||||
raise exc
|
raise exc
|
||||||
retry.ensure_deadline(0.5, exc)
|
|
||||||
|
|
||||||
@_handle_auth_errors
|
@_handle_auth_errors
|
||||||
def range(self, key: str, range_end: Union[bytes, str, None] = None, serializable: bool = True,
|
def range(self, key: str, range_end: Union[bytes, str, None] = None, serializable: bool = True,
|
||||||
@@ -603,12 +603,6 @@ class PatroniEtcd3Client(Etcd3Client):
|
|||||||
super(PatroniEtcd3Client, self).set_base_uri(value)
|
super(PatroniEtcd3Client, self).set_base_uri(value)
|
||||||
self._restart_watcher()
|
self._restart_watcher()
|
||||||
|
|
||||||
def authenticate(self, *, restart_watcher: bool = True, retry: Optional[Retry] = None) -> bool:
|
|
||||||
ret = super(PatroniEtcd3Client, self).authenticate(restart_watcher=restart_watcher, retry=retry)
|
|
||||||
if ret and restart_watcher:
|
|
||||||
self._restart_watcher()
|
|
||||||
return ret
|
|
||||||
|
|
||||||
def _wait_cache(self, timeout: float) -> None:
|
def _wait_cache(self, timeout: float) -> None:
|
||||||
stop_time = time.time() + timeout
|
stop_time = time.time() + timeout
|
||||||
while self._kv_cache and not self._kv_cache.is_ready():
|
while self._kv_cache and not self._kv_cache.is_ready():
|
||||||
@@ -866,14 +860,16 @@ class Etcd3(AbstractEtcd):
|
|||||||
try:
|
try:
|
||||||
return _retry(self._client.put, self.leader_path, self._name, self._lease, create_revision='0')
|
return _retry(self._client.put, self.leader_path, self._name, self._lease, create_revision='0')
|
||||||
except LeaseNotFound:
|
except LeaseNotFound:
|
||||||
logger.error('Our lease disappeared from Etcd. Will try to get a new one and retry attempt')
|
|
||||||
self._lease = None
|
self._lease = None
|
||||||
retry.ensure_deadline(0)
|
if not retry.ensure_deadline(0):
|
||||||
|
logger.error('Our lease disappeared from Etcd. Deadline exceeded, giving up')
|
||||||
|
return False
|
||||||
|
|
||||||
|
logger.error('Our lease disappeared from Etcd. Will try to get a new one and retry attempt')
|
||||||
|
|
||||||
_retry(self._do_refresh_lease)
|
_retry(self._do_refresh_lease)
|
||||||
|
|
||||||
retry.ensure_deadline(1, Etcd3Error('_do_attempt_to_acquire_leader timeout'))
|
retry.ensure_deadline(1, Etcd3Error('_do_attempt_to_acquire_leader timeout'))
|
||||||
|
|
||||||
return _retry(self._client.put, self.leader_path, self._name, self._lease, create_revision='0')
|
return _retry(self._client.put, self.leader_path, self._name, self._lease, create_revision='0')
|
||||||
|
|
||||||
@catch_return_false_exception
|
@catch_return_false_exception
|
||||||
|
|||||||
@@ -3,13 +3,13 @@ import logging
|
|||||||
import random
|
import random
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from typing import Any, Callable, Dict, List, Union
|
from typing import Any, Callable, cast, Dict, List, Union
|
||||||
|
|
||||||
from . import Cluster
|
|
||||||
from .zookeeper import ZooKeeper
|
|
||||||
from ..postgresql.mpp import AbstractMPP
|
from ..postgresql.mpp import AbstractMPP
|
||||||
from ..request import get as requests_get
|
from ..request import get as requests_get
|
||||||
from ..utils import uri
|
from ..utils import uri
|
||||||
|
from . import Cluster
|
||||||
|
from .zookeeper import ZooKeeper
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -41,8 +41,9 @@ class ExhibitorEnsembleProvider(object):
|
|||||||
|
|
||||||
if isinstance(json, dict) and 'servers' in json and 'port' in json:
|
if isinstance(json, dict) and 'servers' in json and 'port' in json:
|
||||||
self._next_poll = time.time() + self._poll_interval
|
self._next_poll = time.time() + self._poll_interval
|
||||||
servers: List[str] = json['servers']
|
servers: List[str] = cast(Dict[str, Any], json)['servers']
|
||||||
zookeeper_hosts = ','.join([h + ':' + str(json['port']) for h in sorted(servers)])
|
port = str(cast(Dict[str, Any], json)['port'])
|
||||||
|
zookeeper_hosts = ','.join([h + ':' + port for h in sorted(servers)])
|
||||||
if self._zookeeper_hosts != zookeeper_hosts:
|
if self._zookeeper_hosts != zookeeper_hosts:
|
||||||
logger.info('ZooKeeper connection string has changed: %s => %s', self._zookeeper_hosts, zookeeper_hosts)
|
logger.info('ZooKeeper connection string has changed: %s => %s', self._zookeeper_hosts, zookeeper_hosts)
|
||||||
self._zookeeper_hosts = zookeeper_hosts
|
self._zookeeper_hosts = zookeeper_hosts
|
||||||
@@ -50,7 +51,7 @@ class ExhibitorEnsembleProvider(object):
|
|||||||
return True
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def _query_exhibitors(self, exhibitors: List[str]) -> Union[Dict[str, Any], Any]:
|
def _query_exhibitors(self, exhibitors: List[str]) -> Any:
|
||||||
random.shuffle(exhibitors)
|
random.shuffle(exhibitors)
|
||||||
for host in exhibitors:
|
for host in exhibitors:
|
||||||
try:
|
try:
|
||||||
|
|||||||
+64
-37
@@ -9,21 +9,26 @@ import random
|
|||||||
import socket
|
import socket
|
||||||
import tempfile
|
import tempfile
|
||||||
import time
|
import time
|
||||||
import urllib3
|
|
||||||
import yaml
|
|
||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
from http.client import HTTPException
|
from http.client import HTTPException
|
||||||
from urllib3.exceptions import HTTPError
|
|
||||||
from threading import Condition, Lock, Thread
|
from threading import Condition, Lock, Thread
|
||||||
from typing import Any, Callable, Collection, Dict, List, Optional, Tuple, Type, Union, TYPE_CHECKING
|
from typing import Any, Callable, Collection, Dict, List, Optional, Tuple, Type, TYPE_CHECKING, Union
|
||||||
|
|
||||||
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member, Status, SyncState, TimelineHistory
|
import urllib3
|
||||||
|
import yaml
|
||||||
|
|
||||||
|
from urllib3.exceptions import HTTPError
|
||||||
|
|
||||||
|
from ..collections import EMPTY_DICT
|
||||||
from ..exceptions import DCSError
|
from ..exceptions import DCSError
|
||||||
|
from ..postgresql.misc import PostgresqlRole, PostgresqlState
|
||||||
from ..postgresql.mpp import AbstractMPP
|
from ..postgresql.mpp import AbstractMPP
|
||||||
from ..utils import deep_compare, iter_response_objects, keepalive_socket_options, \
|
from ..utils import deep_compare, iter_response_objects, \
|
||||||
Retry, RetryFailedError, tzutc, uri, USER_AGENT
|
keepalive_socket_options, Retry, RetryFailedError, tzutc, uri, USER_AGENT
|
||||||
|
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member, Status, SyncState, TimelineHistory
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from ..config import Config
|
from ..config import Config
|
||||||
|
|
||||||
@@ -470,7 +475,7 @@ class K8sClient(object):
|
|||||||
if len(args) == 3: # name, namespace, body
|
if len(args) == 3: # name, namespace, body
|
||||||
body = args[2]
|
body = args[2]
|
||||||
elif action == 'create': # namespace, body
|
elif action == 'create': # namespace, body
|
||||||
body = args[1]
|
body = args[1] # pyright: ignore [reportGeneralTypeIssues]
|
||||||
elif action == 'delete': # name, namespace
|
elif action == 'delete': # name, namespace
|
||||||
body = kwargs.pop('body', None)
|
body = kwargs.pop('body', None)
|
||||||
else:
|
else:
|
||||||
@@ -509,7 +514,7 @@ class KubernetesRetriableException(k8s_client.rest.ApiException):
|
|||||||
@property
|
@property
|
||||||
def sleeptime(self) -> Optional[int]:
|
def sleeptime(self) -> Optional[int]:
|
||||||
try:
|
try:
|
||||||
return int((self.headers or {}).get('retry-after', ''))
|
return int((self.headers or EMPTY_DICT).get('retry-after', ''))
|
||||||
except Exception:
|
except Exception:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
@@ -644,7 +649,7 @@ class ObjectCache(Thread):
|
|||||||
with self._object_cache_lock:
|
with self._object_cache_lock:
|
||||||
return self._object_cache.get(name)
|
return self._object_cache.get(name)
|
||||||
|
|
||||||
def _process_event(self, event: Dict[str, Union[Any, Dict[str, Union[Any, Dict[str, Any]]]]]) -> None:
|
def _process_event(self, event: Dict[str, Any]) -> None:
|
||||||
ev_type = event['type']
|
ev_type = event['type']
|
||||||
obj = event['object']
|
obj = event['object']
|
||||||
name = obj['metadata']['name']
|
name = obj['metadata']['name']
|
||||||
@@ -654,7 +659,7 @@ class ObjectCache(Thread):
|
|||||||
obj = K8sObject(obj)
|
obj = K8sObject(obj)
|
||||||
success, old_value = self.set(name, obj)
|
success, old_value = self.set(name, obj)
|
||||||
if success:
|
if success:
|
||||||
new_value = (obj.metadata.annotations or {}).get(self._annotations_map.get(name))
|
new_value = (obj.metadata.annotations or EMPTY_DICT).get(self._annotations_map.get(name, ''))
|
||||||
elif ev_type == 'DELETED':
|
elif ev_type == 'DELETED':
|
||||||
success, old_value = self.delete(name, obj['metadata']['resourceVersion'])
|
success, old_value = self.delete(name, obj['metadata']['resourceVersion'])
|
||||||
else:
|
else:
|
||||||
@@ -662,7 +667,7 @@ class ObjectCache(Thread):
|
|||||||
|
|
||||||
if success and obj.get('kind') != 'Pod':
|
if success and obj.get('kind') != 'Pod':
|
||||||
if old_value:
|
if old_value:
|
||||||
old_value = (old_value.metadata.annotations or {}).get(self._annotations_map.get(name))
|
old_value = (old_value.metadata.annotations or EMPTY_DICT).get(self._annotations_map.get(name, ''))
|
||||||
|
|
||||||
value_changed = old_value != new_value and \
|
value_changed = old_value != new_value and \
|
||||||
(name != self._dcs.config_path or old_value is not None and new_value is not None)
|
(name != self._dcs.config_path or old_value is not None and new_value is not None)
|
||||||
@@ -752,10 +757,12 @@ class Kubernetes(AbstractDCS):
|
|||||||
self._label_selector = ','.join('{0}={1}'.format(k, v) for k, v in self._labels.items())
|
self._label_selector = ','.join('{0}={1}'.format(k, v) for k, v in self._labels.items())
|
||||||
self._namespace = config.get('namespace') or 'default'
|
self._namespace = config.get('namespace') or 'default'
|
||||||
self._role_label = config.get('role_label', 'role')
|
self._role_label = config.get('role_label', 'role')
|
||||||
self._leader_label_value = config.get('leader_label_value', 'master')
|
self._leader_label_value = config.get('leader_label_value', 'primary')
|
||||||
self._follower_label_value = config.get('follower_label_value', 'replica')
|
self._follower_label_value = config.get('follower_label_value', 'replica')
|
||||||
self._standby_leader_label_value = config.get('standby_leader_label_value', 'master')
|
self._standby_leader_label_value = config.get('standby_leader_label_value', 'primary')
|
||||||
self._tmp_role_label = config.get('tmp_role_label')
|
self._tmp_role_label = config.get('tmp_role_label')
|
||||||
|
self._bootstrap_labels: Dict[str, str] = {str(k): str(v)
|
||||||
|
for k, v in (config.get('bootstrap_labels') or EMPTY_DICT).items()}
|
||||||
self._ca_certs = os.environ.get('PATRONI_KUBERNETES_CACERT', config.get('cacert')) or SERVICE_CERT_FILENAME
|
self._ca_certs = os.environ.get('PATRONI_KUBERNETES_CACERT', config.get('cacert')) or SERVICE_CERT_FILENAME
|
||||||
super(Kubernetes, self).__init__({**config, 'namespace': ''}, mpp)
|
super(Kubernetes, self).__init__({**config, 'namespace': ''}, mpp)
|
||||||
if self._mpp.is_enabled():
|
if self._mpp.is_enabled():
|
||||||
@@ -844,7 +851,7 @@ class Kubernetes(AbstractDCS):
|
|||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def member(pod: K8sObject) -> Member:
|
def member(pod: K8sObject) -> Member:
|
||||||
annotations = pod.metadata.annotations or {}
|
annotations = pod.metadata.annotations or EMPTY_DICT
|
||||||
member = Member.from_node(pod.metadata.resource_version, pod.metadata.name, None, annotations.get('status', ''))
|
member = Member.from_node(pod.metadata.resource_version, pod.metadata.name, None, annotations.get('status', ''))
|
||||||
member.data['pod_labels'] = pod.metadata.labels
|
member.data['pod_labels'] = pod.metadata.labels
|
||||||
return member
|
return member
|
||||||
@@ -925,7 +932,7 @@ class Kubernetes(AbstractDCS):
|
|||||||
failover = nodes.get(path + self._FAILOVER)
|
failover = nodes.get(path + self._FAILOVER)
|
||||||
metadata = failover and failover.metadata
|
metadata = failover and failover.metadata
|
||||||
failover = metadata and Failover.from_node(metadata.resource_version,
|
failover = metadata and Failover.from_node(metadata.resource_version,
|
||||||
(metadata.annotations or {}).copy())
|
(metadata.annotations or EMPTY_DICT).copy())
|
||||||
|
|
||||||
# get synchronization state
|
# get synchronization state
|
||||||
sync = nodes.get(path + self._SYNC)
|
sync = nodes.get(path + self._SYNC)
|
||||||
@@ -1046,9 +1053,10 @@ class Kubernetes(AbstractDCS):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
def __target_ref(self, leader_ip: str, latest_subsets: List[K8sObject], pod: K8sObject) -> K8sObject:
|
def __target_ref(self, leader_ip: str, latest_subsets: List[K8sObject], pod: K8sObject) -> K8sObject:
|
||||||
# we want to re-use existing target_ref if possible
|
# we want to reuse existing target_ref if possible
|
||||||
|
empty_addresses: List[K8sObject] = []
|
||||||
for subset in latest_subsets:
|
for subset in latest_subsets:
|
||||||
for address in subset.addresses or []:
|
for address in subset.addresses or empty_addresses:
|
||||||
if address.ip == leader_ip and address.target_ref and address.target_ref.name == self._name:
|
if address.ip == leader_ip and address.target_ref and address.target_ref.name == self._name:
|
||||||
return address.target_ref
|
return address.target_ref
|
||||||
return k8s_client.V1ObjectReference(kind='Pod', uid=pod.metadata.uid, namespace=self._namespace,
|
return k8s_client.V1ObjectReference(kind='Pod', uid=pod.metadata.uid, namespace=self._namespace,
|
||||||
@@ -1056,7 +1064,8 @@ class Kubernetes(AbstractDCS):
|
|||||||
|
|
||||||
def _map_subsets(self, endpoints: Dict[str, Any], ips: List[str]) -> None:
|
def _map_subsets(self, endpoints: Dict[str, Any], ips: List[str]) -> None:
|
||||||
leader = self._kinds.get(self.leader_path)
|
leader = self._kinds.get(self.leader_path)
|
||||||
latest_subsets = leader and leader.subsets or []
|
empty_addresses: List[K8sObject] = []
|
||||||
|
latest_subsets = leader and leader.subsets or empty_addresses
|
||||||
if not ips:
|
if not ips:
|
||||||
# We want to have subsets empty
|
# We want to have subsets empty
|
||||||
if latest_subsets:
|
if latest_subsets:
|
||||||
@@ -1209,23 +1218,28 @@ class Kubernetes(AbstractDCS):
|
|||||||
|
|
||||||
self._kinds.set(self.leader_path, kind)
|
self._kinds.set(self.leader_path, kind)
|
||||||
|
|
||||||
if not retry.ensure_deadline(0.5):
|
kind_annotations = kind and kind.metadata.annotations or EMPTY_DICT
|
||||||
return False
|
|
||||||
|
|
||||||
kind_annotations = kind and kind.metadata.annotations or {}
|
|
||||||
kind_resource_version = kind and kind.metadata.resource_version
|
kind_resource_version = kind and kind.metadata.resource_version
|
||||||
|
|
||||||
# There is different leader or resource_version in cache didn't change
|
# There is different leader or resource_version in cache didn't change
|
||||||
if kind and (kind_annotations.get(self._LEADER) != self._name or kind_resource_version == resource_version):
|
if kind and (kind_annotations.get(self._LEADER) != self._name or kind_resource_version == resource_version):
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
# We can get 409 because we do at least one retry, and the first update might have succeeded,
|
||||||
|
# therefore we will check if annotations on the read object match expectations.
|
||||||
|
if all(kind_annotations.get(k) == v for k, v in annotations.items()):
|
||||||
|
return True
|
||||||
|
|
||||||
|
if not retry.ensure_deadline(0.5):
|
||||||
|
return False
|
||||||
|
|
||||||
return bool(_run_and_handle_exceptions(self._patch_or_create, self.leader_path, annotations,
|
return bool(_run_and_handle_exceptions(self._patch_or_create, self.leader_path, annotations,
|
||||||
kind_resource_version, ips=ips, retry=_retry))
|
kind_resource_version, ips=ips, retry=_retry))
|
||||||
|
|
||||||
def update_leader(self, leader: Leader, last_lsn: Optional[int],
|
def update_leader(self, cluster: Cluster, last_lsn: Optional[int],
|
||||||
slots: Optional[Dict[str, int]] = None, failsafe: Optional[Dict[str, str]] = None) -> bool:
|
slots: Optional[Dict[str, int]] = None, failsafe: Optional[Dict[str, str]] = None) -> bool:
|
||||||
kind = self._kinds.get(self.leader_path)
|
kind = self._kinds.get(self.leader_path)
|
||||||
kind_annotations = kind and kind.metadata.annotations or {}
|
kind_annotations = kind and kind.metadata.annotations or EMPTY_DICT
|
||||||
|
|
||||||
if kind and kind_annotations.get(self._LEADER) != self._name:
|
if kind and kind_annotations.get(self._LEADER) != self._name:
|
||||||
return False
|
return False
|
||||||
@@ -1238,6 +1252,8 @@ class Kubernetes(AbstractDCS):
|
|||||||
if last_lsn:
|
if last_lsn:
|
||||||
annotations[self._OPTIME] = str(last_lsn)
|
annotations[self._OPTIME] = str(last_lsn)
|
||||||
annotations['slots'] = json.dumps(slots, separators=(',', ':')) if slots else None
|
annotations['slots'] = json.dumps(slots, separators=(',', ':')) if slots else None
|
||||||
|
retain_slots = self._build_retain_slots(cluster, slots)
|
||||||
|
annotations['retain_slots'] = json.dumps(retain_slots) if retain_slots else None
|
||||||
|
|
||||||
if failsafe is not None:
|
if failsafe is not None:
|
||||||
annotations[self._FAILSAFE] = json.dumps(failsafe, separators=(',', ':')) if failsafe else None
|
annotations[self._FAILSAFE] = json.dumps(failsafe, separators=(',', ':')) if failsafe else None
|
||||||
@@ -1303,28 +1319,36 @@ class Kubernetes(AbstractDCS):
|
|||||||
def touch_member(self, data: Dict[str, Any]) -> bool:
|
def touch_member(self, data: Dict[str, Any]) -> bool:
|
||||||
cluster = self.cluster
|
cluster = self.cluster
|
||||||
if cluster and cluster.leader and cluster.leader.name == self._name:
|
if cluster and cluster.leader and cluster.leader.name == self._name:
|
||||||
role = self._standby_leader_label_value if data['role'] == 'standby_leader' else self._leader_label_value
|
role = self._standby_leader_label_value \
|
||||||
tmp_role = 'master'
|
if data['role'] == PostgresqlRole.STANDBY_LEADER else self._leader_label_value
|
||||||
elif data['state'] == 'running' and data['role'] not in ('master', 'primary'):
|
tmp_role = 'primary'
|
||||||
|
elif data['state'] == PostgresqlState.RUNNING and data['role'] != PostgresqlRole.PRIMARY:
|
||||||
role = {'replica': self._follower_label_value}.get(data['role'], data['role'])
|
role = {'replica': self._follower_label_value}.get(data['role'], data['role'])
|
||||||
tmp_role = data['role']
|
tmp_role = data['role']
|
||||||
else:
|
else:
|
||||||
role = None
|
role = None
|
||||||
tmp_role = None
|
tmp_role = None
|
||||||
|
|
||||||
role_labels = {self._role_label: role}
|
updated_labels = {self._role_label: role}
|
||||||
if self._tmp_role_label:
|
if self._tmp_role_label:
|
||||||
role_labels[self._tmp_role_label] = tmp_role
|
updated_labels[self._tmp_role_label] = tmp_role
|
||||||
|
|
||||||
|
if self._bootstrap_labels:
|
||||||
|
if data['state'] in (PostgresqlState.INITDB, PostgresqlState.CUSTOM_BOOTSTRAP,
|
||||||
|
PostgresqlState.BOOTSTRAP_STARTING, PostgresqlState.CREATING_REPLICA):
|
||||||
|
updated_labels.update(self._bootstrap_labels)
|
||||||
|
else:
|
||||||
|
updated_labels.update({k: None for k, _ in self._bootstrap_labels.items()})
|
||||||
|
|
||||||
member = cluster and cluster.get_member(self._name, fallback_to_leader=False)
|
member = cluster and cluster.get_member(self._name, fallback_to_leader=False)
|
||||||
pod_labels = member and member.data.pop('pod_labels', None)
|
pod_labels = member and member.data.pop('pod_labels', None)
|
||||||
ret = member and pod_labels is not None\
|
ret = member and pod_labels is not None\
|
||||||
and all(pod_labels.get(k) == v for k, v in role_labels.items())\
|
and all(pod_labels.get(k) == v for k, v in updated_labels.items())\
|
||||||
and deep_compare(data, member.data)
|
and deep_compare(data, member.data)
|
||||||
|
|
||||||
if not ret:
|
if not ret:
|
||||||
metadata = {'namespace': self._namespace, 'name': self._name, 'labels': role_labels,
|
metadata: Dict[str, Any] = {'namespace': self._namespace, 'name': self._name, 'labels': updated_labels,
|
||||||
'annotations': {'status': json.dumps(data, separators=(',', ':'))}}
|
'annotations': {'status': json.dumps(data, separators=(',', ':'))}}
|
||||||
body = k8s_client.V1Pod(metadata=k8s_client.V1ObjectMeta(**metadata))
|
body = k8s_client.V1Pod(metadata=k8s_client.V1ObjectMeta(**metadata))
|
||||||
ret = self._api.patch_namespaced_pod(self._name, self._namespace, body)
|
ret = self._api.patch_namespaced_pod(self._name, self._namespace, body)
|
||||||
if ret:
|
if ret:
|
||||||
@@ -1346,7 +1370,7 @@ class Kubernetes(AbstractDCS):
|
|||||||
def delete_leader(self, leader: Optional[Leader], last_lsn: Optional[int] = None) -> bool:
|
def delete_leader(self, leader: Optional[Leader], last_lsn: Optional[int] = None) -> bool:
|
||||||
ret = False
|
ret = False
|
||||||
kind = self._kinds.get(self.leader_path)
|
kind = self._kinds.get(self.leader_path)
|
||||||
if kind and (kind.metadata.annotations or {}).get(self._LEADER) == self._name:
|
if kind and (kind.metadata.annotations or EMPTY_DICT).get(self._LEADER) == self._name:
|
||||||
annotations: Dict[str, Optional[str]] = {self._LEADER: None}
|
annotations: Dict[str, Optional[str]] = {self._LEADER: None}
|
||||||
if last_lsn:
|
if last_lsn:
|
||||||
annotations[self._OPTIME] = str(last_lsn)
|
annotations[self._OPTIME] = str(last_lsn)
|
||||||
@@ -1370,15 +1394,18 @@ class Kubernetes(AbstractDCS):
|
|||||||
raise NotImplementedError # pragma: no cover
|
raise NotImplementedError # pragma: no cover
|
||||||
|
|
||||||
def write_sync_state(self, leader: Optional[str], sync_standby: Optional[Collection[str]],
|
def write_sync_state(self, leader: Optional[str], sync_standby: Optional[Collection[str]],
|
||||||
version: Optional[str] = None) -> Optional[SyncState]:
|
quorum: Optional[int], version: Optional[str] = None) -> Optional[SyncState]:
|
||||||
"""Prepare and write annotations to $SCOPE-sync Endpoint or ConfigMap.
|
"""Prepare and write annotations to $SCOPE-sync Endpoint or ConfigMap.
|
||||||
|
|
||||||
:param leader: name of the leader node that manages /sync key
|
:param leader: name of the leader node that manages /sync key
|
||||||
:param sync_standby: collection of currently known synchronous standby node names
|
:param sync_standby: collection of currently known synchronous standby node names
|
||||||
|
:param quorum: if the node from sync_standby list is doing a leader race it should
|
||||||
|
see at least quorum other nodes from the sync_standby + leader list
|
||||||
:param version: last known `resource_version` for conditional update of the object
|
:param version: last known `resource_version` for conditional update of the object
|
||||||
:returns: the new :class:`SyncState` object or None
|
:returns: the new :class:`SyncState` object or None
|
||||||
"""
|
"""
|
||||||
sync_state = self.sync_state(leader, sync_standby)
|
sync_state = self.sync_state(leader, sync_standby, quorum)
|
||||||
|
sync_state['quorum'] = str(sync_state['quorum']) if sync_state['quorum'] is not None else None
|
||||||
ret = self.patch_or_create(self.sync_path, sync_state, version, False)
|
ret = self.patch_or_create(self.sync_path, sync_state, version, False)
|
||||||
if not isinstance(ret, bool):
|
if not isinstance(ret, bool):
|
||||||
return SyncState.from_node(ret.metadata.resource_version, sync_state)
|
return SyncState.from_node(ret.metadata.resource_version, sync_state)
|
||||||
@@ -1390,7 +1417,7 @@ class Kubernetes(AbstractDCS):
|
|||||||
:param version: last known `resource_version` for conditional update of the object
|
:param version: last known `resource_version` for conditional update of the object
|
||||||
:returns: `True` if "delete" was successful
|
:returns: `True` if "delete" was successful
|
||||||
"""
|
"""
|
||||||
return self.write_sync_state(None, None, version=version) is not None
|
return self.write_sync_state(None, None, None, version=version) is not None
|
||||||
|
|
||||||
def watch(self, leader_version: Optional[str], timeout: float) -> bool:
|
def watch(self, leader_version: Optional[str], timeout: float) -> bool:
|
||||||
if self.__do_not_watch:
|
if self.__do_not_watch:
|
||||||
|
|||||||
+7
-5
@@ -5,17 +5,19 @@ import threading
|
|||||||
import time
|
import time
|
||||||
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from pysyncobj import SyncObj, SyncObjConf, replicated, FAIL_REASON
|
from typing import Any, Callable, Collection, Dict, List, Optional, Set, TYPE_CHECKING, Union
|
||||||
|
|
||||||
|
from pysyncobj import FAIL_REASON, replicated, SyncObj, SyncObjConf
|
||||||
from pysyncobj.dns_resolver import globalDnsResolver
|
from pysyncobj.dns_resolver import globalDnsResolver
|
||||||
from pysyncobj.node import TCPNode
|
from pysyncobj.node import TCPNode
|
||||||
from pysyncobj.transport import TCPTransport, CONNECTION_STATE
|
from pysyncobj.transport import CONNECTION_STATE, TCPTransport
|
||||||
from pysyncobj.utility import TcpUtility
|
from pysyncobj.utility import TcpUtility
|
||||||
from typing import Any, Callable, Collection, Dict, List, Optional, Set, Union, TYPE_CHECKING
|
|
||||||
|
|
||||||
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, Status, SyncState, TimelineHistory
|
|
||||||
from ..exceptions import DCSError
|
from ..exceptions import DCSError
|
||||||
from ..postgresql.mpp import AbstractMPP
|
from ..postgresql.mpp import AbstractMPP
|
||||||
from ..utils import validate_directory
|
from ..utils import validate_directory
|
||||||
|
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member, Status, SyncState, TimelineHistory
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from ..config import Config
|
from ..config import Config
|
||||||
|
|
||||||
@@ -253,7 +255,7 @@ class KVStoreTTL(DynMemberSyncObj):
|
|||||||
self.__limb.pop(key)
|
self.__limb.pop(key)
|
||||||
self._expire(key, value, callback=callback)
|
self._expire(key, value, callback=callback)
|
||||||
|
|
||||||
def get(self, key: str, recursive: bool = False) -> Union[None, Dict[str, Any], Dict[str, Dict[str, Any]]]:
|
def get(self, key: str, recursive: bool = False) -> Optional[Dict[str, Any]]:
|
||||||
if not recursive:
|
if not recursive:
|
||||||
return self.__data.get(key)
|
return self.__data.get(key)
|
||||||
return {k: v for k, v in self.__data.items() if k.startswith(key)}
|
return {k: v for k, v in self.__data.items() if k.startswith(key)}
|
||||||
|
|||||||
@@ -4,18 +4,20 @@ import select
|
|||||||
import socket
|
import socket
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from kazoo.client import KazooClient, KazooState, KazooRetry
|
from typing import Any, Callable, cast, Dict, List, Optional, Tuple, TYPE_CHECKING, Union
|
||||||
from kazoo.exceptions import ConnectionClosedError, NoNodeError, NodeExistsError, SessionExpiredError
|
|
||||||
|
from kazoo.client import KazooClient, KazooRetry, KazooState
|
||||||
|
from kazoo.exceptions import ConnectionClosedError, NodeExistsError, NoNodeError, SessionExpiredError
|
||||||
from kazoo.handlers.threading import AsyncResult, SequentialThreadingHandler
|
from kazoo.handlers.threading import AsyncResult, SequentialThreadingHandler
|
||||||
from kazoo.protocol.states import KeeperState, WatchedEvent, ZnodeStat
|
from kazoo.protocol.states import KeeperState, WatchedEvent, ZnodeStat
|
||||||
from kazoo.retry import RetryFailedError
|
from kazoo.retry import RetryFailedError
|
||||||
from kazoo.security import ACL, make_acl
|
from kazoo.security import ACL, make_acl
|
||||||
from typing import Any, Callable, Dict, List, Optional, Union, Tuple, TYPE_CHECKING
|
|
||||||
|
|
||||||
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, Status, SyncState, TimelineHistory
|
|
||||||
from ..exceptions import DCSError
|
from ..exceptions import DCSError
|
||||||
from ..postgresql.mpp import AbstractMPP
|
from ..postgresql.mpp import AbstractMPP
|
||||||
from ..utils import deep_compare
|
from ..utils import deep_compare
|
||||||
|
from . import AbstractDCS, Cluster, ClusterConfig, Failover, Leader, Member, Status, SyncState, TimelineHistory
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from ..config import Config
|
from ..config import Config
|
||||||
|
|
||||||
@@ -178,7 +180,8 @@ class ZooKeeper(AbstractDCS):
|
|||||||
return int(self._client._session_timeout / 1000.0)
|
return int(self._client._session_timeout / 1000.0)
|
||||||
|
|
||||||
def set_retry_timeout(self, retry_timeout: int) -> None:
|
def set_retry_timeout(self, retry_timeout: int) -> None:
|
||||||
retry = self._client.retry if isinstance(self._client.retry, KazooRetry) else self._client._retry
|
old_kazoo = isinstance(self._client.retry, KazooRetry) # pyright: ignore [reportUnnecessaryIsInstance]
|
||||||
|
retry = cast(KazooRetry, self._client.retry) if old_kazoo else self._client._retry
|
||||||
retry.deadline = retry_timeout
|
retry.deadline = retry_timeout
|
||||||
|
|
||||||
def get_node(
|
def get_node(
|
||||||
|
|||||||
@@ -5,9 +5,9 @@ import logging
|
|||||||
import os
|
import os
|
||||||
import pkgutil
|
import pkgutil
|
||||||
import sys
|
import sys
|
||||||
from types import ModuleType
|
|
||||||
|
|
||||||
from typing import Any, Dict, Iterator, List, Optional, Set, Tuple, TYPE_CHECKING, Type, TypeVar, Union
|
from types import ModuleType
|
||||||
|
from typing import Any, Dict, Iterator, List, Optional, Set, Tuple, Type, TYPE_CHECKING, TypeVar, Union
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from .config import Config
|
from .config import Config
|
||||||
@@ -38,8 +38,11 @@ def iter_modules(package: str) -> List[str]:
|
|||||||
for importer in pkgutil.iter_importers():
|
for importer in pkgutil.iter_importers():
|
||||||
if hasattr(importer, 'toc'):
|
if hasattr(importer, 'toc'):
|
||||||
toc |= getattr(importer, 'toc')
|
toc |= getattr(importer, 'toc')
|
||||||
dots = module_prefix.count('.') # search for modules only on the same level
|
# If it found the pyinstaller toc then use it, otherwise fall through to default
|
||||||
return [module for module in toc if module.startswith(module_prefix) and module.count('.') == dots]
|
# behavior which works in pyinstaller >= 4.4
|
||||||
|
if len(toc) > 0:
|
||||||
|
dots = module_prefix.count('.') # search for modules only on the same level
|
||||||
|
return [module for module in toc if module.startswith(module_prefix) and module.count('.') == dots]
|
||||||
|
|
||||||
# here we are making an assumption that the package which is calling this function is already imported
|
# here we are making an assumption that the package which is calling this function is already imported
|
||||||
pkg_file = sys.modules[package].__file__
|
pkg_file = sys.modules[package].__file__
|
||||||
|
|||||||
+11
-3
@@ -39,19 +39,27 @@ class __FilePermissions:
|
|||||||
def __init__(self) -> None:
|
def __init__(self) -> None:
|
||||||
"""Create a :class:`__FilePermissions` object and set default permissions."""
|
"""Create a :class:`__FilePermissions` object and set default permissions."""
|
||||||
self.__set_owner_permissions()
|
self.__set_owner_permissions()
|
||||||
self.__set_umask()
|
self.__orig_umask = self.__set_umask()
|
||||||
|
|
||||||
def __set_umask(self) -> None:
|
def __set_umask(self) -> int:
|
||||||
"""Set umask value based on calculations.
|
"""Set umask value based on calculations.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
Should only be called once either :meth:`__set_owner_permissions`
|
Should only be called once either :meth:`__set_owner_permissions`
|
||||||
or :meth:`__set_group_permissions` has been executed.
|
or :meth:`__set_group_permissions` has been executed.
|
||||||
|
|
||||||
|
:returns: the previous value of the umask or ``0022`` if umask call failed.
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
os.umask(self.__pg_mode_mask)
|
return os.umask(self.__pg_mode_mask)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error('Can not set umask to %03o: %r', self.__pg_mode_mask, e)
|
logger.error('Can not set umask to %03o: %r', self.__pg_mode_mask, e)
|
||||||
|
return 0o22
|
||||||
|
|
||||||
|
@property
|
||||||
|
def orig_umask(self) -> int:
|
||||||
|
"""Original umask value."""
|
||||||
|
return self.__orig_umask
|
||||||
|
|
||||||
def __set_owner_permissions(self) -> None:
|
def __set_owner_permissions(self) -> None:
|
||||||
"""Make directories/files accessible only by the owner."""
|
"""Make directories/files accessible only by the owner."""
|
||||||
|
|||||||
+29
-11
@@ -8,8 +8,9 @@ import sys
|
|||||||
import types
|
import types
|
||||||
|
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
from typing import Any, Dict, List, Optional, Union, TYPE_CHECKING
|
from typing import Any, cast, Dict, List, Optional, TYPE_CHECKING
|
||||||
|
|
||||||
|
from .collections import EMPTY_DICT
|
||||||
from .utils import parse_bool, parse_int
|
from .utils import parse_bool, parse_int
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
@@ -44,19 +45,20 @@ class GlobalConfig(types.ModuleType):
|
|||||||
"""
|
"""
|
||||||
return bool(cluster and cluster.config and cluster.config.modify_version)
|
return bool(cluster and cluster.config and cluster.config.modify_version)
|
||||||
|
|
||||||
def update(self, cluster: Optional['Cluster']) -> None:
|
def update(self, cluster: Optional['Cluster'], default: Optional[Dict[str, Any]] = None) -> None:
|
||||||
"""Update with the new global configuration from the :class:`Cluster` object view.
|
"""Update with the new global configuration from the :class:`Cluster` object view.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
Global configuration is updated only when configuration in the *cluster* view is valid.
|
|
||||||
|
|
||||||
Update happens in-place and is executed only from the main heartbeat thread.
|
Update happens in-place and is executed only from the main heartbeat thread.
|
||||||
|
|
||||||
:param cluster: the currently known cluster state from DCS.
|
:param cluster: the currently known cluster state from DCS.
|
||||||
|
:param default: default configuration, which will be used if there is no valid *cluster.config*.
|
||||||
"""
|
"""
|
||||||
# Try to protect from the case when DCS was wiped out
|
# Try to protect from the case when DCS was wiped out
|
||||||
if self._cluster_has_valid_config(cluster):
|
if self._cluster_has_valid_config(cluster):
|
||||||
self.__config = cluster.config.data # pyright: ignore [reportOptionalMemberAccess]
|
self.__config = cluster.config.data # pyright: ignore [reportOptionalMemberAccess]
|
||||||
|
elif default:
|
||||||
|
self.__config = default
|
||||||
|
|
||||||
def from_cluster(self, cluster: Optional['Cluster']) -> 'GlobalConfig':
|
def from_cluster(self, cluster: Optional['Cluster']) -> 'GlobalConfig':
|
||||||
"""Return :class:`GlobalConfig` instance from the provided :class:`Cluster` object view.
|
"""Return :class:`GlobalConfig` instance from the provided :class:`Cluster` object view.
|
||||||
@@ -103,17 +105,23 @@ class GlobalConfig(types.ModuleType):
|
|||||||
"""``True`` if cluster is in maintenance mode."""
|
"""``True`` if cluster is in maintenance mode."""
|
||||||
return self.check_mode('pause')
|
return self.check_mode('pause')
|
||||||
|
|
||||||
|
@property
|
||||||
|
def is_quorum_commit_mode(self) -> bool:
|
||||||
|
""":returns: ``True`` if quorum commit replication is requested"""
|
||||||
|
return str(self.get('synchronous_mode')).lower() == 'quorum'
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def is_synchronous_mode(self) -> bool:
|
def is_synchronous_mode(self) -> bool:
|
||||||
"""``True`` if synchronous replication is requested and it is not a standby cluster config."""
|
"""``True`` if synchronous replication is requested and it is not a standby cluster config."""
|
||||||
return self.check_mode('synchronous_mode') and not self.is_standby_cluster
|
return (self.check_mode('synchronous_mode') is True or self.is_quorum_commit_mode) \
|
||||||
|
and not self.is_standby_cluster
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def is_synchronous_mode_strict(self) -> bool:
|
def is_synchronous_mode_strict(self) -> bool:
|
||||||
"""``True`` if at least one synchronous node is required."""
|
"""``True`` if at least one synchronous node is required."""
|
||||||
return self.check_mode('synchronous_mode_strict')
|
return self.check_mode('synchronous_mode_strict')
|
||||||
|
|
||||||
def get_standby_cluster_config(self) -> Union[Dict[str, Any], Any]:
|
def get_standby_cluster_config(self) -> Any:
|
||||||
"""Get ``standby_cluster`` configuration.
|
"""Get ``standby_cluster`` configuration.
|
||||||
|
|
||||||
:returns: a copy of ``standby_cluster`` configuration.
|
:returns: a copy of ``standby_cluster`` configuration.
|
||||||
@@ -125,18 +133,20 @@ class GlobalConfig(types.ModuleType):
|
|||||||
"""``True`` if global configuration has a valid ``standby_cluster`` section."""
|
"""``True`` if global configuration has a valid ``standby_cluster`` section."""
|
||||||
config = self.get_standby_cluster_config()
|
config = self.get_standby_cluster_config()
|
||||||
return isinstance(config, dict) and\
|
return isinstance(config, dict) and\
|
||||||
bool(config.get('host') or config.get('port') or config.get('restore_command'))
|
any(cast(Dict[str, Any], config).get(p) for p in ('host', 'port', 'restore_command'))
|
||||||
|
|
||||||
def get_int(self, name: str, default: int = 0) -> int:
|
def get_int(self, name: str, default: int = 0, base_unit: Optional[str] = None) -> int:
|
||||||
"""Gets current value of *name* from the global configuration and try to return it as :class:`int`.
|
"""Gets current value of *name* from the global configuration and try to return it as :class:`int`.
|
||||||
|
|
||||||
:param name: name of the parameter.
|
:param name: name of the parameter.
|
||||||
:param default: default value if *name* is not in the configuration or invalid.
|
:param default: default value if *name* is not in the configuration or invalid.
|
||||||
|
:param base_unit: an optional base unit to convert value of *name* parameter to.
|
||||||
|
Not used if the value does not contain a unit.
|
||||||
|
|
||||||
:returns: currently configured value of *name* from the global configuration or *default* if it is not set or
|
:returns: currently configured value of *name* from the global configuration or *default* if it is not set or
|
||||||
invalid.
|
invalid.
|
||||||
"""
|
"""
|
||||||
ret = parse_int(self.get(name))
|
ret = parse_int(self.get(name), base_unit)
|
||||||
return default if ret is None else ret
|
return default if ret is None else ret
|
||||||
|
|
||||||
@property
|
@property
|
||||||
@@ -213,7 +223,7 @@ class GlobalConfig(types.ModuleType):
|
|||||||
@property
|
@property
|
||||||
def use_slots(self) -> bool:
|
def use_slots(self) -> bool:
|
||||||
"""``True`` if cluster is configured to use replication slots."""
|
"""``True`` if cluster is configured to use replication slots."""
|
||||||
return bool(parse_bool((self.get('postgresql') or {}).get('use_slots', True)))
|
return bool(parse_bool((self.get('postgresql') or EMPTY_DICT).get('use_slots', True)))
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def permanent_slots(self) -> Dict[str, Any]:
|
def permanent_slots(self) -> Dict[str, Any]:
|
||||||
@@ -221,7 +231,15 @@ class GlobalConfig(types.ModuleType):
|
|||||||
return deepcopy(self.get('permanent_replication_slots')
|
return deepcopy(self.get('permanent_replication_slots')
|
||||||
or self.get('permanent_slots')
|
or self.get('permanent_slots')
|
||||||
or self.get('slots')
|
or self.get('slots')
|
||||||
or {})
|
or EMPTY_DICT.copy())
|
||||||
|
|
||||||
|
@property
|
||||||
|
def member_slots_ttl(self) -> int:
|
||||||
|
"""Currently configured value of ``member_slots_ttl`` from the global configuration converted to seconds.
|
||||||
|
|
||||||
|
Assume ``1800`` if it is not set or invalid.
|
||||||
|
"""
|
||||||
|
return self.get_int('member_slots_ttl', 1800, base_unit='s')
|
||||||
|
|
||||||
|
|
||||||
sys.modules[__name__] = GlobalConfig()
|
sys.modules[__name__] = GlobalConfig()
|
||||||
|
|||||||
+525
-180
File diff suppressed because it is too large
Load Diff
+91
-19
@@ -8,19 +8,52 @@ import os
|
|||||||
import sys
|
import sys
|
||||||
|
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
|
from io import TextIOWrapper
|
||||||
from logging.handlers import RotatingFileHandler
|
from logging.handlers import RotatingFileHandler
|
||||||
from queue import Queue, Full
|
from queue import Full, Queue
|
||||||
from threading import Lock, Thread
|
from threading import Lock, Thread
|
||||||
|
from typing import Any, cast, Dict, List, Optional, TYPE_CHECKING, Union
|
||||||
|
|
||||||
from typing import Any, Dict, List, Optional, Union, TYPE_CHECKING
|
from .file_perm import pg_perm
|
||||||
|
from .utils import deep_compare, parse_int
|
||||||
from .utils import deep_compare
|
|
||||||
|
|
||||||
type_logformat = Union[List[Union[str, Dict[str, Any], Any]], str, Any]
|
type_logformat = Union[List[Union[str, Dict[str, Any], Any]], str, Any]
|
||||||
|
|
||||||
_LOGGER = logging.getLogger(__name__)
|
_LOGGER = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class PatroniFileHandler(RotatingFileHandler):
|
||||||
|
"""Wrapper of :class:`RotatingFileHandler` to handle permissions of log files. """
|
||||||
|
|
||||||
|
def __init__(self, filename: str, mode: Optional[int]) -> None:
|
||||||
|
"""Create a new :class:`PatroniFileHandler` instance.
|
||||||
|
|
||||||
|
:param filename: basename for log files.
|
||||||
|
:param mode: permissions for log files.
|
||||||
|
"""
|
||||||
|
self.set_log_file_mode(mode)
|
||||||
|
super(PatroniFileHandler, self).__init__(filename)
|
||||||
|
|
||||||
|
def set_log_file_mode(self, mode: Optional[int]) -> None:
|
||||||
|
"""Set mode for Patroni log files.
|
||||||
|
|
||||||
|
:param mode: permissions for log files.
|
||||||
|
|
||||||
|
.. note::
|
||||||
|
If *mode* is not specified, we calculate it from the `umask` value.
|
||||||
|
"""
|
||||||
|
self._log_file_mode = 0o666 & ~pg_perm.orig_umask if mode is None else mode
|
||||||
|
|
||||||
|
def _open(self) -> TextIOWrapper:
|
||||||
|
"""Open a new log file and assign permissions.
|
||||||
|
|
||||||
|
:returns: the resulting stream.
|
||||||
|
"""
|
||||||
|
ret = super(PatroniFileHandler, self)._open()
|
||||||
|
os.chmod(self.baseFilename, self._log_file_mode)
|
||||||
|
return ret
|
||||||
|
|
||||||
|
|
||||||
def debug_exception(self: logging.Logger, msg: object, *args: Any, **kwargs: Any) -> None:
|
def debug_exception(self: logging.Logger, msg: object, *args: Any, **kwargs: Any) -> None:
|
||||||
"""Add full stack trace info to debug log messages and partial to others.
|
"""Add full stack trace info to debug log messages and partial to others.
|
||||||
|
|
||||||
@@ -62,6 +95,15 @@ def error_exception(self: logging.Logger, msg: object, *args: Any, **kwargs: Any
|
|||||||
self.error(msg, *args, exc_info=exc_info, **kwargs)
|
self.error(msg, *args, exc_info=exc_info, **kwargs)
|
||||||
|
|
||||||
|
|
||||||
|
def _type(value: Any) -> str:
|
||||||
|
"""Get type of the *value*.
|
||||||
|
|
||||||
|
:param value: any arbitrary value.
|
||||||
|
:returns: a string with a type name.
|
||||||
|
"""
|
||||||
|
return value.__class__.__name__
|
||||||
|
|
||||||
|
|
||||||
class QueueHandler(logging.Handler):
|
class QueueHandler(logging.Handler):
|
||||||
"""Queue-based logging handler.
|
"""Queue-based logging handler.
|
||||||
|
|
||||||
@@ -292,7 +334,7 @@ class PatroniLogger(Thread):
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
if not isinstance(logformat, str):
|
if not isinstance(logformat, str):
|
||||||
_LOGGER.warning('Expected log format to be a string when log type is plain, but got "%s"', type(logformat))
|
_LOGGER.warning('Expected log format to be a string when log type is plain, but got "%s"', _type(logformat))
|
||||||
logformat = PatroniLogger.DEFAULT_FORMAT
|
logformat = PatroniLogger.DEFAULT_FORMAT
|
||||||
|
|
||||||
return logging.Formatter(logformat, dateformat)
|
return logging.Formatter(logformat, dateformat)
|
||||||
@@ -316,6 +358,7 @@ class PatroniLogger(Thread):
|
|||||||
jsonformat = logformat
|
jsonformat = logformat
|
||||||
rename_fields = {}
|
rename_fields = {}
|
||||||
elif isinstance(logformat, list):
|
elif isinstance(logformat, list):
|
||||||
|
logformat = cast(List[Any], logformat)
|
||||||
log_fields: List[str] = []
|
log_fields: List[str] = []
|
||||||
rename_fields: Dict[str, str] = {}
|
rename_fields: Dict[str, str] = {}
|
||||||
|
|
||||||
@@ -323,6 +366,7 @@ class PatroniLogger(Thread):
|
|||||||
if isinstance(field, str):
|
if isinstance(field, str):
|
||||||
log_fields.append(field)
|
log_fields.append(field)
|
||||||
elif isinstance(field, dict):
|
elif isinstance(field, dict):
|
||||||
|
field = cast(Dict[str, Any], field)
|
||||||
for original_field, renamed_field in field.items():
|
for original_field, renamed_field in field.items():
|
||||||
if isinstance(renamed_field, str):
|
if isinstance(renamed_field, str):
|
||||||
log_fields.append(original_field)
|
log_fields.append(original_field)
|
||||||
@@ -330,13 +374,13 @@ class PatroniLogger(Thread):
|
|||||||
else:
|
else:
|
||||||
_LOGGER.warning(
|
_LOGGER.warning(
|
||||||
'Expected renamed log field to be a string, but got "%s"',
|
'Expected renamed log field to be a string, but got "%s"',
|
||||||
type(renamed_field)
|
_type(renamed_field)
|
||||||
)
|
)
|
||||||
|
|
||||||
else:
|
else:
|
||||||
_LOGGER.warning(
|
_LOGGER.warning(
|
||||||
'Expected each item of log format to be a string or dictionary, but got "%s"',
|
'Expected each item of log format to be a string or dictionary, but got "%s"',
|
||||||
type(field)
|
_type(field)
|
||||||
)
|
)
|
||||||
|
|
||||||
if len(log_fields) > 0:
|
if len(log_fields) > 0:
|
||||||
@@ -346,12 +390,19 @@ class PatroniLogger(Thread):
|
|||||||
else:
|
else:
|
||||||
jsonformat = PatroniLogger.DEFAULT_FORMAT
|
jsonformat = PatroniLogger.DEFAULT_FORMAT
|
||||||
rename_fields = {}
|
rename_fields = {}
|
||||||
_LOGGER.warning('Expected log format to be a string or a list, but got "%s"', type(logformat))
|
_LOGGER.warning('Expected log format to be a string or a list, but got "%s"', _type(logformat))
|
||||||
|
|
||||||
try:
|
try:
|
||||||
from pythonjsonlogger import jsonlogger
|
try:
|
||||||
|
from pythonjsonlogger import json as jsonlogger # pyright: ignore
|
||||||
|
except ImportError: # pragma: no cover
|
||||||
|
from pythonjsonlogger import jsonlogger
|
||||||
|
if hasattr(jsonlogger, 'RESERVED_ATTRS') \
|
||||||
|
and 'taskName' not in jsonlogger.RESERVED_ATTRS: # pyright: ignore [reportPrivateImportUsage]
|
||||||
|
# compatibility with python 3.12, that added a new attribute to LogRecord
|
||||||
|
jsonlogger.RESERVED_ATTRS += ('taskName',) # pyright: ignore
|
||||||
|
|
||||||
return jsonlogger.JsonFormatter(
|
return jsonlogger.JsonFormatter( # pyright: ignore [reportPrivateImportUsage]
|
||||||
jsonformat,
|
jsonformat,
|
||||||
dateformat,
|
dateformat,
|
||||||
rename_fields=rename_fields,
|
rename_fields=rename_fields,
|
||||||
@@ -377,7 +428,7 @@ class PatroniLogger(Thread):
|
|||||||
static_fields = config.get('static_fields', {})
|
static_fields = config.get('static_fields', {})
|
||||||
|
|
||||||
if dateformat is not None and not isinstance(dateformat, str):
|
if dateformat is not None and not isinstance(dateformat, str):
|
||||||
_LOGGER.warning('Expected log dateformat to be a string, but got "%s"', type(dateformat))
|
_LOGGER.warning('Expected log dateformat to be a string, but got "%s"', _type(dateformat))
|
||||||
dateformat = None
|
dateformat = None
|
||||||
|
|
||||||
if logtype == 'json':
|
if logtype == 'json':
|
||||||
@@ -410,14 +461,16 @@ class PatroniLogger(Thread):
|
|||||||
handler = self.log_handler
|
handler = self.log_handler
|
||||||
|
|
||||||
if 'dir' in config:
|
if 'dir' in config:
|
||||||
if not isinstance(handler, RotatingFileHandler):
|
mode = parse_int(config.get('mode'))
|
||||||
handler = RotatingFileHandler(os.path.join(config['dir'], __name__))
|
if not isinstance(handler, PatroniFileHandler):
|
||||||
|
handler = PatroniFileHandler(os.path.join(config['dir'], __name__), mode)
|
||||||
handler.maxBytes = int(config.get('file_size', 25000000)) # pyright: ignore [reportGeneralTypeIssues]
|
handler.set_log_file_mode(mode)
|
||||||
|
max_file_size = int(config.get('file_size', 25000000))
|
||||||
|
handler.maxBytes = max_file_size # pyright: ignore [reportAttributeAccessIssue]
|
||||||
handler.backupCount = int(config.get('file_num', 4))
|
handler.backupCount = int(config.get('file_num', 4))
|
||||||
# we can't use `if not isinstance(handler, logging.StreamHandler)` below,
|
# we can't use `if not isinstance(handler, logging.StreamHandler)` below,
|
||||||
# because RotatingFileHandler is a child of StreamHandler!!!
|
# because RotatingFileHandler and PatroniFileHandler are children of StreamHandler!!!
|
||||||
elif handler is None or isinstance(handler, RotatingFileHandler):
|
elif handler is None or isinstance(handler, PatroniFileHandler):
|
||||||
handler = logging.StreamHandler()
|
handler = logging.StreamHandler()
|
||||||
|
|
||||||
is_new_handler = handler != self.log_handler
|
is_new_handler = handler != self.log_handler
|
||||||
@@ -440,7 +493,7 @@ class PatroniLogger(Thread):
|
|||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
It is used to remove different handlers that were configured previous to a reload in the configuration,
|
It is used to remove different handlers that were configured previous to a reload in the configuration,
|
||||||
e.g. if we are switching from :class:`~logging.handlers.RotatingFileHandler` to
|
e.g. if we are switching from :class:`PatroniFileHandler` to
|
||||||
class:`~logging.StreamHandler` and vice-versa.
|
class:`~logging.StreamHandler` and vice-versa.
|
||||||
"""
|
"""
|
||||||
while True:
|
while True:
|
||||||
@@ -464,6 +517,7 @@ class PatroniLogger(Thread):
|
|||||||
self._root_logger.removeHandler(self._proxy_handler)
|
self._root_logger.removeHandler(self._proxy_handler)
|
||||||
|
|
||||||
prev_record = None
|
prev_record = None
|
||||||
|
prev_hb_msg = ''
|
||||||
|
|
||||||
while True:
|
while True:
|
||||||
self._close_old_handlers()
|
self._close_old_handlers()
|
||||||
@@ -482,8 +536,16 @@ class PatroniLogger(Thread):
|
|||||||
prev_record, record = record, None
|
prev_record, record = record, None
|
||||||
else:
|
else:
|
||||||
if prev_record and prev_record.thread == record.thread:
|
if prev_record and prev_record.thread == record.thread:
|
||||||
if not (record.msg.startswith('no action. ') or record.msg.startswith('PAUSE: no action')):
|
if self._is_heartbeat_msg(record):
|
||||||
|
config = self._config or {}
|
||||||
|
deduplicate_heartbeat_logs = config.get('deduplicate_heartbeat_logs', False)
|
||||||
|
if record.msg == prev_hb_msg and deduplicate_heartbeat_logs:
|
||||||
|
record = None
|
||||||
|
else:
|
||||||
|
prev_hb_msg = record.msg
|
||||||
|
else:
|
||||||
self.log_handler.handle(prev_record)
|
self.log_handler.handle(prev_record)
|
||||||
|
prev_hb_msg = None
|
||||||
prev_record = None
|
prev_record = None
|
||||||
|
|
||||||
if record:
|
if record:
|
||||||
@@ -491,6 +553,16 @@ class PatroniLogger(Thread):
|
|||||||
|
|
||||||
self._queue_handler.queue.task_done()
|
self._queue_handler.queue.task_done()
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _is_heartbeat_msg(record: logging.LogRecord) -> bool:
|
||||||
|
"""Checks if the given record contains a heartbeat message.
|
||||||
|
|
||||||
|
:param record: the record to check.
|
||||||
|
|
||||||
|
:returns: ``True`` if the record contains a heartbeat message, ``False`` otherwise.
|
||||||
|
"""
|
||||||
|
return record.msg.startswith('no action. ') or record.msg.startswith('PAUSE: no action')
|
||||||
|
|
||||||
def shutdown(self) -> None:
|
def shutdown(self) -> None:
|
||||||
"""Shut down the logger thread."""
|
"""Shut down the logger thread."""
|
||||||
try:
|
try:
|
||||||
|
|||||||
+140
-99
@@ -9,28 +9,30 @@ import time
|
|||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
|
from enum import IntEnum
|
||||||
|
from threading import current_thread, Lock
|
||||||
|
from typing import Any, Callable, Dict, Iterator, List, Optional, Tuple, TYPE_CHECKING, Union
|
||||||
|
|
||||||
from dateutil import tz
|
from dateutil import tz
|
||||||
from psutil import TimeoutExpired
|
from psutil import TimeoutExpired
|
||||||
from threading import current_thread, Lock
|
|
||||||
from typing import Any, Callable, Dict, Iterator, List, Optional, Union, Tuple, TYPE_CHECKING
|
|
||||||
|
|
||||||
|
from .. import global_config, psycopg
|
||||||
|
from ..async_executor import CriticalTask
|
||||||
|
from ..collections import CaseInsensitiveDict, CaseInsensitiveSet, EMPTY_DICT
|
||||||
|
from ..dcs import Cluster, Leader, Member, slot_name_from_member_name
|
||||||
|
from ..exceptions import PostgresConnectionException
|
||||||
|
from ..tags import Tags
|
||||||
|
from ..utils import data_directory_is_empty, parse_int, polling_loop, Retry, RetryFailedError
|
||||||
from .bootstrap import Bootstrap
|
from .bootstrap import Bootstrap
|
||||||
from .callback_executor import CallbackAction, CallbackExecutor
|
from .callback_executor import CallbackAction, CallbackExecutor
|
||||||
from .cancellable import CancellableSubprocess
|
from .cancellable import CancellableSubprocess
|
||||||
from .config import ConfigHandler, mtime
|
from .config import ConfigHandler, mtime
|
||||||
from .connection import ConnectionPool, get_connection_cursor
|
from .connection import ConnectionPool, get_connection_cursor
|
||||||
from .misc import parse_history, parse_lsn, postgres_major_version_to_int
|
from .misc import parse_history, parse_lsn, postgres_major_version_to_int, PostgresqlRole, PostgresqlState
|
||||||
from .mpp import AbstractMPP
|
from .mpp import AbstractMPP
|
||||||
from .postmaster import PostmasterProcess
|
from .postmaster import PostmasterProcess
|
||||||
from .slots import SlotsHandler
|
from .slots import SlotsHandler
|
||||||
from .sync import SyncHandler
|
from .sync import SyncHandler
|
||||||
from .. import global_config, psycopg
|
|
||||||
from ..async_executor import CriticalTask
|
|
||||||
from ..collections import CaseInsensitiveSet, CaseInsensitiveDict
|
|
||||||
from ..dcs import Cluster, Leader, Member, SLOT_ADVANCE_AVAILABLE_VERSION
|
|
||||||
from ..exceptions import PostgresConnectionException
|
|
||||||
from ..utils import Retry, RetryFailedError, polling_loop, data_directory_is_empty, parse_int
|
|
||||||
from ..tags import Tags
|
|
||||||
|
|
||||||
if TYPE_CHECKING: # pragma: no cover
|
if TYPE_CHECKING: # pragma: no cover
|
||||||
from psycopg import Connection as Connection3, Cursor
|
from psycopg import Connection as Connection3, Cursor
|
||||||
@@ -38,14 +40,24 @@ if TYPE_CHECKING: # pragma: no cover
|
|||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
STATE_RUNNING = 'running'
|
|
||||||
STATE_REJECT = 'rejecting connections'
|
|
||||||
STATE_NO_RESPONSE = 'not responding'
|
|
||||||
STATE_UNKNOWN = 'unknown'
|
|
||||||
|
|
||||||
STOP_POLLING_INTERVAL = 1
|
STOP_POLLING_INTERVAL = 1
|
||||||
|
|
||||||
|
|
||||||
|
class PgIsReadyStatus(IntEnum):
|
||||||
|
"""Possible PostgreSQL connection status ``pg_isready`` utility can report.
|
||||||
|
|
||||||
|
:cvar RUNNING: return code 0. PostgreSQL is accepting connections normally.
|
||||||
|
:cvar REJECT: return code 1. PostgreSQL is rejecting connections.
|
||||||
|
:cvar NO_RESPONSE: return code 2. There was no response to the connection attempt.
|
||||||
|
:cvar UNKNOWN: Return code 3. No connection attempt was made, something went wrong.
|
||||||
|
"""
|
||||||
|
|
||||||
|
RUNNING = 0
|
||||||
|
REJECT = 1
|
||||||
|
NO_RESPONSE = 2
|
||||||
|
UNKNOWN = 3
|
||||||
|
|
||||||
|
|
||||||
@contextmanager
|
@contextmanager
|
||||||
def null_context():
|
def null_context():
|
||||||
yield
|
yield
|
||||||
@@ -75,16 +87,18 @@ class Postgresql(object):
|
|||||||
self._major_version = self.get_major_version()
|
self._major_version = self.get_major_version()
|
||||||
|
|
||||||
self._state_lock = Lock()
|
self._state_lock = Lock()
|
||||||
self.set_state('stopped')
|
self.set_state(PostgresqlState.STOPPED)
|
||||||
|
|
||||||
self._pending_restart_reason = CaseInsensitiveDict()
|
self._pending_restart_reason = CaseInsensitiveDict()
|
||||||
self.connection_pool = ConnectionPool()
|
self.connection_pool = ConnectionPool()
|
||||||
self._connection = self.connection_pool.get('heartbeat')
|
self._connection = self.connection_pool.get('heartbeat')
|
||||||
self.mpp_handler = mpp.get_handler_impl(self)
|
self.mpp_handler = mpp.get_handler_impl(self)
|
||||||
|
self._bin_dir = config.get('bin_dir') or ''
|
||||||
|
self._role_lock = Lock()
|
||||||
|
self.set_role(PostgresqlRole.UNINITIALIZED)
|
||||||
self.config = ConfigHandler(self, config)
|
self.config = ConfigHandler(self, config)
|
||||||
self.config.check_directories()
|
self.config.check_directories()
|
||||||
|
|
||||||
self._bin_dir = config.get('bin_dir') or ''
|
|
||||||
self.bootstrap = Bootstrap(self)
|
self.bootstrap = Bootstrap(self)
|
||||||
self.bootstrapping = False
|
self.bootstrapping = False
|
||||||
self.__thread_ident = current_thread().ident
|
self.__thread_ident = current_thread().ident
|
||||||
@@ -106,12 +120,11 @@ class Postgresql(object):
|
|||||||
self._is_leader_retry = Retry(max_tries=1, deadline=config['retry_timeout'] / 2.0, max_delay=1,
|
self._is_leader_retry = Retry(max_tries=1, deadline=config['retry_timeout'] / 2.0, max_delay=1,
|
||||||
retry_exceptions=PostgresConnectionException)
|
retry_exceptions=PostgresConnectionException)
|
||||||
|
|
||||||
self._role_lock = Lock()
|
|
||||||
self.set_role(self.get_postgres_role_from_data_directory())
|
self.set_role(self.get_postgres_role_from_data_directory())
|
||||||
self._state_entry_timestamp = 0
|
self._state_entry_timestamp = 0
|
||||||
|
|
||||||
self._cluster_info_state = {}
|
self._cluster_info_state = {}
|
||||||
self._has_permanent_slots = True
|
self._should_query_slots = True
|
||||||
self._enforce_hot_standby_feedback = False
|
self._enforce_hot_standby_feedback = False
|
||||||
self._cached_replica_timeline = None
|
self._cached_replica_timeline = None
|
||||||
|
|
||||||
@@ -122,27 +135,27 @@ class Postgresql(object):
|
|||||||
|
|
||||||
if self.is_running():
|
if self.is_running():
|
||||||
# If we found postmaster process we need to figure out whether postgres is accepting connections
|
# If we found postmaster process we need to figure out whether postgres is accepting connections
|
||||||
self.set_state('starting')
|
self.set_state(PostgresqlState.STARTING)
|
||||||
self.check_startup_state_changed()
|
self.check_startup_state_changed()
|
||||||
|
|
||||||
if self.state == 'running': # we are "joining" already running postgres
|
if self.state == PostgresqlState.RUNNING: # we are "joining" already running postgres
|
||||||
# we know that PostgreSQL is accepting connections and can read some GUC's from pg_settings
|
# we know that PostgreSQL is accepting connections and can read some GUC's from pg_settings
|
||||||
self.config.load_current_server_parameters()
|
self.config.load_current_server_parameters()
|
||||||
|
|
||||||
self.set_role('master' if self.is_primary() else 'replica')
|
self.set_role(PostgresqlRole.PRIMARY if self.is_primary() else PostgresqlRole.REPLICA)
|
||||||
|
|
||||||
hba_saved = self.config.replace_pg_hba()
|
hba_saved = self.config.replace_pg_hba()
|
||||||
ident_saved = self.config.replace_pg_ident()
|
ident_saved = self.config.replace_pg_ident()
|
||||||
|
|
||||||
if self.major_version < 120000 or self.role in ('master', 'primary'):
|
if self.major_version < 120000 or self.role == PostgresqlRole.PRIMARY:
|
||||||
# If PostgreSQL is running as a primary or we run PostgreSQL that is older than 12 we can
|
# If PostgreSQL is running as a primary or we run PostgreSQL that is older than 12 we can
|
||||||
# call reload_config() once again (the first call happened in the ConfigHandler constructor),
|
# call reload_config() once again (the first call happened in the ConfigHandler constructor),
|
||||||
# so that it can figure out if config files should be updated and pg_ctl reload executed.
|
# so that it can figure out if config files should be updated and pg_ctl reload executed.
|
||||||
self.config.reload_config(config, sighup=bool(hba_saved or ident_saved))
|
self.config.reload_config(config, sighup=bool(hba_saved or ident_saved))
|
||||||
elif hba_saved or ident_saved:
|
elif hba_saved or ident_saved:
|
||||||
self.reload()
|
self.reload()
|
||||||
elif not self.is_running() and self.role in ('master', 'primary'):
|
elif not self.is_running() and self.role == PostgresqlRole.PRIMARY:
|
||||||
self.set_role('demoted')
|
self.set_role(PostgresqlRole.DEMOTED)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def create_replica_methods(self) -> List[str]:
|
def create_replica_methods(self) -> List[str]:
|
||||||
@@ -181,6 +194,11 @@ class Postgresql(object):
|
|||||||
def lsn_name(self) -> str:
|
def lsn_name(self) -> str:
|
||||||
return 'lsn' if self._major_version >= 100000 else 'location'
|
return 'lsn' if self._major_version >= 100000 else 'location'
|
||||||
|
|
||||||
|
@property
|
||||||
|
def supports_quorum_commit(self) -> bool:
|
||||||
|
"""``True`` if quorum commit is supported by Postgres."""
|
||||||
|
return self._major_version >= 100000
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def supports_multiple_sync(self) -> bool:
|
def supports_multiple_sync(self) -> bool:
|
||||||
""":returns: `True` if Postgres version supports more than one synchronous node."""
|
""":returns: `True` if Postgres version supports more than one synchronous node."""
|
||||||
@@ -189,7 +207,7 @@ class Postgresql(object):
|
|||||||
@property
|
@property
|
||||||
def can_advance_slots(self) -> bool:
|
def can_advance_slots(self) -> bool:
|
||||||
"""``True`` if :attr:``major_version`` is greater than 110000."""
|
"""``True`` if :attr:``major_version`` is greater than 110000."""
|
||||||
return self.major_version >= SLOT_ADVANCE_AVAILABLE_VERSION
|
return self.major_version >= 110000
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def cluster_info_query(self) -> str:
|
def cluster_info_query(self) -> str:
|
||||||
@@ -219,23 +237,27 @@ class Postgresql(object):
|
|||||||
" pg_catalog.pg_stat_get_activity(w.pid)"
|
" pg_catalog.pg_stat_get_activity(w.pid)"
|
||||||
" WHERE w.state = 'streaming') r)").format(self.wal_name, self.lsn_name)
|
" WHERE w.state = 'streaming') r)").format(self.wal_name, self.lsn_name)
|
||||||
if global_config.is_synchronous_mode
|
if global_config.is_synchronous_mode
|
||||||
and self.role in ('master', 'primary', 'promoted') else "'on', '', NULL")
|
and self.role in (PostgresqlRole.PRIMARY, PostgresqlRole.PROMOTED) else "'on', '', NULL")
|
||||||
|
|
||||||
if self._major_version >= 90600:
|
if self._major_version >= 90600:
|
||||||
|
filter_failover = ' WHERE NOT failover' if self._major_version >= 170000 else ''
|
||||||
extra = ("pg_catalog.current_setting('restore_command')" if self._major_version >= 120000 else "NULL") +\
|
extra = ("pg_catalog.current_setting('restore_command')" if self._major_version >= 120000 else "NULL") +\
|
||||||
", " + ("(SELECT pg_catalog.json_agg(s.*) FROM (SELECT slot_name, slot_type as type, datoid::bigint, "
|
", " + ("(SELECT pg_catalog.json_agg(s.*) FROM (SELECT slot_name, slot_type as type, datoid::bigint, "
|
||||||
"plugin, catalog_xmin, pg_catalog.pg_wal_lsn_diff(confirmed_flush_lsn, '0/0')::bigint"
|
"plugin, catalog_xmin, pg_catalog.pg_wal_lsn_diff(confirmed_flush_lsn, '0/0')::bigint"
|
||||||
" AS confirmed_flush_lsn, pg_catalog.pg_wal_lsn_diff(restart_lsn, '0/0')::bigint"
|
" AS confirmed_flush_lsn, pg_catalog.pg_wal_lsn_diff(restart_lsn, '0/0')::bigint"
|
||||||
" AS restart_lsn FROM pg_catalog.pg_get_replication_slots()) AS s)"
|
f" AS restart_lsn, xmin FROM pg_catalog.pg_get_replication_slots(){filter_failover}) AS s)"
|
||||||
if self._has_permanent_slots and self.can_advance_slots else "NULL") + extra
|
if self._should_query_slots and self.can_advance_slots else "NULL") + extra
|
||||||
extra = (", CASE WHEN latest_end_lsn IS NULL THEN NULL ELSE received_tli END,"
|
|
||||||
" slot_name, conninfo, status, {0} FROM pg_catalog.pg_stat_get_wal_receiver()").format(extra)
|
written_lsn = ("pg_catalog.pg_wal_lsn_diff(written_lsn, '0/0')::bigint"
|
||||||
if self.role == 'standby_leader':
|
if self._major_version >= 130000 else "NULL")
|
||||||
|
extra = (", CASE WHEN latest_end_lsn IS NULL THEN NULL ELSE received_tli END, {0}, slot_name, "
|
||||||
|
"conninfo, status, {1} FROM pg_catalog.pg_stat_get_wal_receiver()").format(written_lsn, extra)
|
||||||
|
if self.role == PostgresqlRole.STANDBY_LEADER:
|
||||||
extra = "timeline_id" + extra + ", pg_catalog.pg_control_checkpoint()"
|
extra = "timeline_id" + extra + ", pg_catalog.pg_control_checkpoint()"
|
||||||
else:
|
else:
|
||||||
extra = "0" + extra
|
extra = "0" + extra
|
||||||
else:
|
else:
|
||||||
extra = "0, NULL, NULL, NULL, NULL, NULL, NULL" + extra
|
extra = "0, NULL, NULL, NULL, NULL, NULL, NULL, NULL" + extra
|
||||||
|
|
||||||
return ("SELECT " + self.TL_LSN + ", {3}").format(self.wal_name, self.lsn_name, self.wal_flush, extra)
|
return ("SELECT " + self.TL_LSN + ", {3}").format(self.wal_name, self.lsn_name, self.wal_flush, extra)
|
||||||
|
|
||||||
@@ -272,7 +294,7 @@ class Postgresql(object):
|
|||||||
|
|
||||||
:returns: path to Postgres binary named *cmd*.
|
:returns: path to Postgres binary named *cmd*.
|
||||||
"""
|
"""
|
||||||
return os.path.join(self._bin_dir, (self.config.get('bin_name', {}) or {}).get(cmd, cmd))
|
return os.path.join(self._bin_dir, (self.config.get('bin_name', {}) or EMPTY_DICT).get(cmd, cmd))
|
||||||
|
|
||||||
def pg_ctl(self, cmd: str, *args: str, **kwargs: Any) -> bool:
|
def pg_ctl(self, cmd: str, *args: str, **kwargs: Any) -> bool:
|
||||||
"""Builds and executes pg_ctl command
|
"""Builds and executes pg_ctl command
|
||||||
@@ -293,10 +315,10 @@ class Postgresql(object):
|
|||||||
initdb = [self.pgcommand('initdb')] + list(args) + [self.data_dir]
|
initdb = [self.pgcommand('initdb')] + list(args) + [self.data_dir]
|
||||||
return subprocess.call(initdb, **kwargs) == 0
|
return subprocess.call(initdb, **kwargs) == 0
|
||||||
|
|
||||||
def pg_isready(self) -> str:
|
def pg_isready(self) -> PgIsReadyStatus:
|
||||||
"""Runs pg_isready to see if PostgreSQL is accepting connections.
|
"""Runs pg_isready to see if PostgreSQL is accepting connections.
|
||||||
|
|
||||||
:returns: 'ok' if PostgreSQL is up, 'reject' if starting up, 'no_resopnse' if not up."""
|
:returns: one of :class:`PgIsReadyStatus` values."""
|
||||||
|
|
||||||
r = self.connection_pool.conn_kwargs
|
r = self.connection_pool.conn_kwargs
|
||||||
cmd = [self.pgcommand('pg_isready'), '-p', r['port'], '-d', self._database]
|
cmd = [self.pgcommand('pg_isready'), '-p', r['port'], '-d', self._database]
|
||||||
@@ -310,11 +332,10 @@ class Postgresql(object):
|
|||||||
cmd.extend(['-U', r['user']])
|
cmd.extend(['-U', r['user']])
|
||||||
|
|
||||||
ret = subprocess.call(cmd)
|
ret = subprocess.call(cmd)
|
||||||
return_codes = {0: STATE_RUNNING,
|
try:
|
||||||
1: STATE_REJECT,
|
return PgIsReadyStatus(ret)
|
||||||
2: STATE_NO_RESPONSE,
|
except ValueError:
|
||||||
3: STATE_UNKNOWN}
|
return PgIsReadyStatus.UNKNOWN
|
||||||
return return_codes.get(ret, STATE_UNKNOWN)
|
|
||||||
|
|
||||||
def reload_config(self, config: Dict[str, Any], sighup: bool = False) -> None:
|
def reload_config(self, config: Dict[str, Any], sighup: bool = False) -> None:
|
||||||
self.config.reload_config(config, sighup)
|
self.config.reload_config(config, sighup)
|
||||||
@@ -345,13 +366,13 @@ class Postgresql(object):
|
|||||||
self._sysid = data.get('Database system identifier', '')
|
self._sysid = data.get('Database system identifier', '')
|
||||||
return self._sysid
|
return self._sysid
|
||||||
|
|
||||||
def get_postgres_role_from_data_directory(self) -> str:
|
def get_postgres_role_from_data_directory(self) -> PostgresqlRole:
|
||||||
if self.data_directory_empty() or not self.controldata():
|
if self.data_directory_empty() or not self.controldata():
|
||||||
return 'uninitialized'
|
return PostgresqlRole.UNINITIALIZED
|
||||||
elif self.config.recovery_conf_exists():
|
elif self.config.recovery_conf_exists():
|
||||||
return 'replica'
|
return PostgresqlRole.REPLICA
|
||||||
else:
|
else:
|
||||||
return 'master'
|
return PostgresqlRole.PRIMARY
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def server_version(self) -> int:
|
def server_version(self) -> int:
|
||||||
@@ -378,7 +399,7 @@ class Postgresql(object):
|
|||||||
try:
|
try:
|
||||||
return self._connection.query(sql, *params)
|
return self._connection.query(sql, *params)
|
||||||
except PostgresConnectionException as exc:
|
except PostgresConnectionException as exc:
|
||||||
if self.state == 'restarting':
|
if self.state == PostgresqlState.RESTARTING:
|
||||||
raise RetryFailedError('cluster is being restarted') from exc
|
raise RetryFailedError('cluster is being restarted') from exc
|
||||||
raise
|
raise
|
||||||
|
|
||||||
@@ -414,11 +435,10 @@ class Postgresql(object):
|
|||||||
return data_directory_is_empty(self._data_dir)
|
return data_directory_is_empty(self._data_dir)
|
||||||
|
|
||||||
def replica_method_options(self, method: str) -> Dict[str, Any]:
|
def replica_method_options(self, method: str) -> Dict[str, Any]:
|
||||||
return deepcopy(self.config.get(method, {}) or {})
|
return deepcopy(self.config.get(method, {}) or EMPTY_DICT.copy())
|
||||||
|
|
||||||
def replica_method_can_work_without_replication_connection(self, method: str) -> bool:
|
def replica_method_can_work_without_replication_connection(self, method: str) -> bool:
|
||||||
return method != 'basebackup' and bool(self.replica_method_options(method).get('no_master')
|
return method != 'basebackup' and bool(self.replica_method_options(method).get('no_leader'))
|
||||||
or self.replica_method_options(method).get('no_leader'))
|
|
||||||
|
|
||||||
def can_create_replica_without_replication_connection(self, replica_methods: Optional[List[str]]) -> bool:
|
def can_create_replica_without_replication_connection(self, replica_methods: Optional[List[str]]) -> bool:
|
||||||
""" go through the replication methods to see if there are ones
|
""" go through the replication methods to see if there are ones
|
||||||
@@ -463,7 +483,7 @@ class Postgresql(object):
|
|||||||
# to have a logical slot or in case if it is the cascading replica.
|
# to have a logical slot or in case if it is the cascading replica.
|
||||||
self.set_enforce_hot_standby_feedback(not global_config.is_standby_cluster and self.can_advance_slots
|
self.set_enforce_hot_standby_feedback(not global_config.is_standby_cluster and self.can_advance_slots
|
||||||
and cluster.should_enforce_hot_standby_feedback(self, tags))
|
and cluster.should_enforce_hot_standby_feedback(self, tags))
|
||||||
self._has_permanent_slots = cluster.has_permanent_slots(self, tags)
|
self._should_query_slots = global_config.member_slots_ttl > 0 or cluster.has_permanent_slots(self, tags)
|
||||||
|
|
||||||
def _cluster_info_state_get(self, name: str) -> Optional[Any]:
|
def _cluster_info_state_get(self, name: str) -> Optional[Any]:
|
||||||
if not self._cluster_info_state:
|
if not self._cluster_info_state:
|
||||||
@@ -471,17 +491,17 @@ class Postgresql(object):
|
|||||||
result = self._is_leader_retry(self._query, self.cluster_info_query)[0]
|
result = self._is_leader_retry(self._query, self.cluster_info_query)[0]
|
||||||
cluster_info_state = dict(zip(['timeline', 'wal_position', 'replayed_location',
|
cluster_info_state = dict(zip(['timeline', 'wal_position', 'replayed_location',
|
||||||
'received_location', 'replay_paused', 'pg_control_timeline',
|
'received_location', 'replay_paused', 'pg_control_timeline',
|
||||||
'received_tli', 'slot_name', 'conninfo', 'receiver_state',
|
'received_tli', 'write_location', 'slot_name', 'conninfo',
|
||||||
'restore_command', 'slots', 'synchronous_commit',
|
'receiver_state', 'restore_command', 'slots', 'synchronous_commit',
|
||||||
'synchronous_standby_names', 'pg_stat_replication'], result))
|
'synchronous_standby_names', 'pg_stat_replication'], result))
|
||||||
if self._has_permanent_slots and self.can_advance_slots:
|
if self._should_query_slots and self.can_advance_slots:
|
||||||
cluster_info_state['slots'] =\
|
cluster_info_state['slots'] =\
|
||||||
self.slots_handler.process_permanent_slots(cluster_info_state['slots'])
|
self.slots_handler.process_permanent_slots(cluster_info_state['slots'])
|
||||||
self._cluster_info_state = cluster_info_state
|
self._cluster_info_state = cluster_info_state
|
||||||
except RetryFailedError as e: # SELECT failed two times
|
except RetryFailedError as e: # SELECT failed two times
|
||||||
self._cluster_info_state = {'error': str(e)}
|
self._cluster_info_state = {'error': str(e)}
|
||||||
if not self.is_starting() and self.pg_isready() == STATE_REJECT:
|
if not self.is_starting() and self.pg_isready() == PgIsReadyStatus.REJECT:
|
||||||
self.set_state('starting')
|
self.set_state(PostgresqlState.STARTING)
|
||||||
|
|
||||||
if 'error' in self._cluster_info_state:
|
if 'error' in self._cluster_info_state:
|
||||||
raise PostgresConnectionException(self._cluster_info_state['error'])
|
raise PostgresConnectionException(self._cluster_info_state['error'])
|
||||||
@@ -492,10 +512,24 @@ class Postgresql(object):
|
|||||||
return self._cluster_info_state_get('replayed_location')
|
return self._cluster_info_state_get('replayed_location')
|
||||||
|
|
||||||
def received_location(self) -> Optional[int]:
|
def received_location(self) -> Optional[int]:
|
||||||
return self._cluster_info_state_get('received_location')
|
write = self._cluster_info_state_get('write_location')
|
||||||
|
received = self._cluster_info_state_get('received_location')
|
||||||
|
return max(received, write) if received and write else write or received
|
||||||
|
|
||||||
def slots(self) -> Dict[str, int]:
|
def slots(self) -> Dict[str, int]:
|
||||||
return self._cluster_info_state_get('slots') or {}
|
"""Get replication slots state.
|
||||||
|
|
||||||
|
..note::
|
||||||
|
Since this methods is supposed to be used only by the leader and only to publish state of
|
||||||
|
replication slots to DCS so that other nodes can advance LSN on respective replication slots,
|
||||||
|
we are also adding our own name to the list. All slots that shouldn't be published to DCS
|
||||||
|
later will be filtered out by :meth:`~Cluster.maybe_filter_permanent_slots` method.
|
||||||
|
|
||||||
|
:returns: A :class:`dict` object with replication slot names and LSNs as absolute values.
|
||||||
|
"""
|
||||||
|
return {**(self._cluster_info_state_get('slots') or {}),
|
||||||
|
slot_name_from_member_name(self.name): self.last_operation()} \
|
||||||
|
if self.can_advance_slots else {}
|
||||||
|
|
||||||
def primary_slot_name(self) -> Optional[str]:
|
def primary_slot_name(self) -> Optional[str]:
|
||||||
return self._cluster_info_state_get('slot_name')
|
return self._cluster_info_state_get('slot_name')
|
||||||
@@ -523,7 +557,7 @@ class Postgresql(object):
|
|||||||
"""Figure out the replication state from input parameters.
|
"""Figure out the replication state from input parameters.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
This method could be only called when Postgres is up, running and queries are successfuly executed.
|
This method could be only called when Postgres is up, running and queries are successfully executed.
|
||||||
|
|
||||||
:is_primary: `True` is postgres is not running in recovery
|
:is_primary: `True` is postgres is not running in recovery
|
||||||
:receiver_state: value from `pg_stat_get_wal_receiver.state` or None if Postgres is older than 9.6
|
:receiver_state: value from `pg_stat_get_wal_receiver.state` or None if Postgres is older than 9.6
|
||||||
@@ -558,7 +592,7 @@ class Postgresql(object):
|
|||||||
return bool(self._cluster_info_state_get('timeline'))
|
return bool(self._cluster_info_state_get('timeline'))
|
||||||
except PostgresConnectionException:
|
except PostgresConnectionException:
|
||||||
logger.warning('Failed to determine PostgreSQL state from the connection, falling back to cached role')
|
logger.warning('Failed to determine PostgreSQL state from the connection, falling back to cached role')
|
||||||
return bool(self.is_running() and self.role in ('master', 'primary'))
|
return bool(self.is_running() and self.role == PostgresqlRole.PRIMARY)
|
||||||
|
|
||||||
def replay_paused(self) -> bool:
|
def replay_paused(self) -> bool:
|
||||||
return self._cluster_info_state_get('replay_paused') or False
|
return self._cluster_info_state_get('replay_paused') or False
|
||||||
@@ -638,6 +672,7 @@ class Postgresql(object):
|
|||||||
if self._postmaster_proc.is_running():
|
if self._postmaster_proc.is_running():
|
||||||
return self._postmaster_proc
|
return self._postmaster_proc
|
||||||
self._postmaster_proc = None
|
self._postmaster_proc = None
|
||||||
|
self._available_gucs = None
|
||||||
|
|
||||||
# we noticed that postgres was restarted, force syncing of replication slots and check of logical slots
|
# we noticed that postgres was restarted, force syncing of replication slots and check of logical slots
|
||||||
self.slots_handler.schedule()
|
self.slots_handler.schedule()
|
||||||
@@ -659,7 +694,7 @@ class Postgresql(object):
|
|||||||
|
|
||||||
if self.callback and cb_type in self.callback:
|
if self.callback and cb_type in self.callback:
|
||||||
cmd = self.callback[cb_type]
|
cmd = self.callback[cb_type]
|
||||||
role = 'master' if self.role == 'promoted' else self.role
|
role = PostgresqlRole.PRIMARY if self.role == PostgresqlRole.PROMOTED else self.role
|
||||||
try:
|
try:
|
||||||
cmd = shlex.split(self.callback[cb_type]) + [cb_type, role, self.scope]
|
cmd = shlex.split(self.callback[cb_type]) + [cb_type, role, self.scope]
|
||||||
self._callback_executor.call(cmd)
|
self._callback_executor.call(cmd)
|
||||||
@@ -667,20 +702,20 @@ class Postgresql(object):
|
|||||||
logger.exception('callback %s %r %s %s failed', cmd, cb_type, role, self.scope)
|
logger.exception('callback %s %r %s %s failed', cmd, cb_type, role, self.scope)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def role(self) -> str:
|
def role(self) -> PostgresqlRole:
|
||||||
with self._role_lock:
|
with self._role_lock:
|
||||||
return self._role
|
return self._role
|
||||||
|
|
||||||
def set_role(self, value: str) -> None:
|
def set_role(self, value: PostgresqlRole) -> None:
|
||||||
with self._role_lock:
|
with self._role_lock:
|
||||||
self._role = value
|
self._role = value
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def state(self) -> str:
|
def state(self) -> PostgresqlState:
|
||||||
with self._state_lock:
|
with self._state_lock:
|
||||||
return self._state
|
return self._state
|
||||||
|
|
||||||
def set_state(self, value: str) -> None:
|
def set_state(self, value: PostgresqlState) -> None:
|
||||||
with self._state_lock:
|
with self._state_lock:
|
||||||
self._state = value
|
self._state = value
|
||||||
self._state_entry_timestamp = time.time()
|
self._state_entry_timestamp = time.time()
|
||||||
@@ -689,7 +724,7 @@ class Postgresql(object):
|
|||||||
return time.time() - self._state_entry_timestamp
|
return time.time() - self._state_entry_timestamp
|
||||||
|
|
||||||
def is_starting(self) -> bool:
|
def is_starting(self) -> bool:
|
||||||
return self.state == 'starting'
|
return self.state in (PostgresqlState.STARTING, PostgresqlState.BOOTSTRAP_STARTING)
|
||||||
|
|
||||||
def wait_for_port_open(self, postmaster: PostmasterProcess, timeout: float) -> bool:
|
def wait_for_port_open(self, postmaster: PostmasterProcess, timeout: float) -> bool:
|
||||||
"""Waits until PostgreSQL opens ports."""
|
"""Waits until PostgreSQL opens ports."""
|
||||||
@@ -699,12 +734,12 @@ class Postgresql(object):
|
|||||||
|
|
||||||
if not postmaster.is_running():
|
if not postmaster.is_running():
|
||||||
logger.error('postmaster is not running')
|
logger.error('postmaster is not running')
|
||||||
self.set_state('start failed')
|
self.set_state(PostgresqlState.START_FAILED)
|
||||||
return False
|
return False
|
||||||
|
|
||||||
isready = self.pg_isready()
|
isready = self.pg_isready()
|
||||||
if isready != STATE_NO_RESPONSE:
|
if isready != PgIsReadyStatus.NO_RESPONSE:
|
||||||
if isready not in [STATE_REJECT, STATE_RUNNING]:
|
if isready not in [PgIsReadyStatus.REJECT, PgIsReadyStatus.RUNNING]:
|
||||||
logger.warning("Can't determine PostgreSQL startup status, assuming running")
|
logger.warning("Can't determine PostgreSQL startup status, assuming running")
|
||||||
return True
|
return True
|
||||||
|
|
||||||
@@ -712,7 +747,7 @@ class Postgresql(object):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
def start(self, timeout: Optional[float] = None, task: Optional[CriticalTask] = None,
|
def start(self, timeout: Optional[float] = None, task: Optional[CriticalTask] = None,
|
||||||
block_callbacks: bool = False, role: Optional[str] = None,
|
block_callbacks: bool = False, role: Optional[PostgresqlRole] = None,
|
||||||
after_start: Optional[Callable[..., Any]] = None) -> Optional[bool]:
|
after_start: Optional[Callable[..., Any]] = None) -> Optional[bool]:
|
||||||
"""Start PostgreSQL
|
"""Start PostgreSQL
|
||||||
|
|
||||||
@@ -727,9 +762,11 @@ class Postgresql(object):
|
|||||||
# patroni.
|
# patroni.
|
||||||
self.connection_pool.close()
|
self.connection_pool.close()
|
||||||
|
|
||||||
|
state = (PostgresqlState.BOOTSTRAP_STARTING if self.bootstrap.running_custom_bootstrap
|
||||||
|
else PostgresqlState.STARTING)
|
||||||
if self.is_running():
|
if self.is_running():
|
||||||
logger.error('Cannot start PostgreSQL because one is already running.')
|
logger.error('Cannot start PostgreSQL because one is already running.')
|
||||||
self.set_state('starting')
|
self.set_state(state)
|
||||||
return True
|
return True
|
||||||
|
|
||||||
if not block_callbacks:
|
if not block_callbacks:
|
||||||
@@ -737,7 +774,7 @@ class Postgresql(object):
|
|||||||
|
|
||||||
self.set_role(role or self.get_postgres_role_from_data_directory())
|
self.set_role(role or self.get_postgres_role_from_data_directory())
|
||||||
|
|
||||||
self.set_state('starting')
|
self.set_state(state)
|
||||||
self.set_pending_restart_reason(CaseInsensitiveDict())
|
self.set_pending_restart_reason(CaseInsensitiveDict())
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@@ -812,7 +849,7 @@ class Postgresql(object):
|
|||||||
cur.execute('CHECKPOINT')
|
cur.execute('CHECKPOINT')
|
||||||
except psycopg.Error:
|
except psycopg.Error:
|
||||||
logger.exception('Exception during CHECKPOINT')
|
logger.exception('Exception during CHECKPOINT')
|
||||||
return 'not accessible or not healty'
|
return 'not accessible or not healthy'
|
||||||
|
|
||||||
def stop(self, mode: str = 'fast', block_callbacks: bool = False, checkpoint: Optional[bool] = None,
|
def stop(self, mode: str = 'fast', block_callbacks: bool = False, checkpoint: Optional[bool] = None,
|
||||||
on_safepoint: Optional[Callable[..., Any]] = None, on_shutdown: Optional[Callable[[int, int], Any]] = None,
|
on_safepoint: Optional[Callable[..., Any]] = None, on_shutdown: Optional[Callable[[int, int], Any]] = None,
|
||||||
@@ -836,12 +873,12 @@ class Postgresql(object):
|
|||||||
# block_callbacks is used during restart to avoid
|
# block_callbacks is used during restart to avoid
|
||||||
# running start/stop callbacks in addition to restart ones
|
# running start/stop callbacks in addition to restart ones
|
||||||
if not block_callbacks:
|
if not block_callbacks:
|
||||||
self.set_state('stopped')
|
self.set_state(PostgresqlState.STOPPED)
|
||||||
if pg_signaled:
|
if pg_signaled:
|
||||||
self.call_nowait(CallbackAction.ON_STOP)
|
self.call_nowait(CallbackAction.ON_STOP)
|
||||||
else:
|
else:
|
||||||
logger.warning('pg_ctl stop failed')
|
logger.warning('pg_ctl stop failed')
|
||||||
self.set_state('stop failed')
|
self.set_state(PostgresqlState.STOP_FAILED)
|
||||||
return success
|
return success
|
||||||
|
|
||||||
def _do_stop(self, mode: str, block_callbacks: bool, checkpoint: bool,
|
def _do_stop(self, mode: str, block_callbacks: bool, checkpoint: bool,
|
||||||
@@ -857,7 +894,7 @@ class Postgresql(object):
|
|||||||
self.checkpoint(timeout=stop_timeout)
|
self.checkpoint(timeout=stop_timeout)
|
||||||
|
|
||||||
if not block_callbacks:
|
if not block_callbacks:
|
||||||
self.set_state('stopping')
|
self.set_state(PostgresqlState.STOPPING)
|
||||||
|
|
||||||
# invoke user-directed before stop script
|
# invoke user-directed before stop script
|
||||||
self._before_stop()
|
self._before_stop()
|
||||||
@@ -947,28 +984,28 @@ class Postgresql(object):
|
|||||||
def check_startup_state_changed(self) -> bool:
|
def check_startup_state_changed(self) -> bool:
|
||||||
"""Checks if PostgreSQL has completed starting up or failed or still starting.
|
"""Checks if PostgreSQL has completed starting up or failed or still starting.
|
||||||
|
|
||||||
Should only be called when state == 'starting'
|
Should only be called when state == 'starting [after custom bootstrap]'
|
||||||
|
|
||||||
:returns: True if state was changed from 'starting'
|
:returns: True if state was changed from 'starting'
|
||||||
"""
|
"""
|
||||||
ready = self.pg_isready()
|
ready = self.pg_isready()
|
||||||
|
|
||||||
if ready == STATE_REJECT:
|
if ready == PgIsReadyStatus.REJECT:
|
||||||
return False
|
return False
|
||||||
elif ready == STATE_NO_RESPONSE:
|
elif ready == PgIsReadyStatus.NO_RESPONSE:
|
||||||
ret = not self.is_running()
|
ret = not self.is_running()
|
||||||
if ret:
|
if ret:
|
||||||
self.set_state('start failed')
|
self.set_state(PostgresqlState.START_FAILED)
|
||||||
self.slots_handler.schedule(False) # TODO: can remove this?
|
self.slots_handler.schedule(False) # TODO: can remove this?
|
||||||
self.config.save_configuration_files(True) # TODO: maybe remove this?
|
self.config.save_configuration_files(True) # TODO: maybe remove this?
|
||||||
return ret
|
return ret
|
||||||
else:
|
else:
|
||||||
if ready != STATE_RUNNING:
|
if ready != PgIsReadyStatus.RUNNING:
|
||||||
# Bad configuration or unexpected OS error. No idea of PostgreSQL status.
|
# Bad configuration or unexpected OS error. No idea of PostgreSQL status.
|
||||||
# Let the main loop of run cycle clean up the mess.
|
# Let the main loop of run cycle clean up the mess.
|
||||||
logger.warning("%s status returned from pg_isready",
|
logger.warning("%s status returned from pg_isready",
|
||||||
"Unknown" if ready == STATE_UNKNOWN else "Invalid")
|
"Unknown" if ready == PgIsReadyStatus.UNKNOWN else "Invalid")
|
||||||
self.set_state('running')
|
self.set_state(PostgresqlState.RUNNING)
|
||||||
self.slots_handler.schedule()
|
self.slots_handler.schedule()
|
||||||
self.config.save_configuration_files(True)
|
self.config.save_configuration_files(True)
|
||||||
# TODO: __cb_pending can be None here after PostgreSQL restarts on its own. Do we want to call the callback?
|
# TODO: __cb_pending can be None here after PostgreSQL restarts on its own. Do we want to call the callback?
|
||||||
@@ -992,10 +1029,10 @@ class Postgresql(object):
|
|||||||
return None
|
return None
|
||||||
time.sleep(1)
|
time.sleep(1)
|
||||||
|
|
||||||
return self.state == 'running'
|
return self.state == PostgresqlState.RUNNING
|
||||||
|
|
||||||
def restart(self, timeout: Optional[float] = None, task: Optional[CriticalTask] = None,
|
def restart(self, timeout: Optional[float] = None, task: Optional[CriticalTask] = None,
|
||||||
block_callbacks: bool = False, role: Optional[str] = None,
|
block_callbacks: bool = False, role: Optional[PostgresqlRole] = None,
|
||||||
before_shutdown: Optional[Callable[..., Any]] = None,
|
before_shutdown: Optional[Callable[..., Any]] = None,
|
||||||
after_start: Optional[Callable[..., Any]] = None) -> Optional[bool]:
|
after_start: Optional[Callable[..., Any]] = None) -> Optional[bool]:
|
||||||
"""Restarts PostgreSQL.
|
"""Restarts PostgreSQL.
|
||||||
@@ -1005,13 +1042,14 @@ class Postgresql(object):
|
|||||||
|
|
||||||
:returns: True when restart was successful and timeout did not expire when waiting.
|
:returns: True when restart was successful and timeout did not expire when waiting.
|
||||||
"""
|
"""
|
||||||
self.set_state('restarting')
|
self.set_state(PostgresqlState.RESTARTING)
|
||||||
if not block_callbacks:
|
if not block_callbacks:
|
||||||
self.__cb_pending = CallbackAction.ON_RESTART
|
self.__cb_pending = CallbackAction.ON_RESTART
|
||||||
ret = self.stop(block_callbacks=True, before_shutdown=before_shutdown)\
|
ret = self.stop(block_callbacks=True, before_shutdown=before_shutdown)\
|
||||||
and self.start(timeout, task, True, role, after_start)
|
and self.start(timeout, task, True, role, after_start)
|
||||||
if not ret and not self.is_starting():
|
if not ret and not self.is_starting():
|
||||||
self.set_state('restart failed ({0})'.format(self.state))
|
logger.warning('restart failed (%r)', self.state)
|
||||||
|
self.set_state(PostgresqlState.RESTART_FAILED)
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
def is_healthy(self) -> bool:
|
def is_healthy(self) -> bool:
|
||||||
@@ -1033,7 +1071,7 @@ class Postgresql(object):
|
|||||||
def controldata(self) -> Dict[str, str]:
|
def controldata(self) -> Dict[str, str]:
|
||||||
""" return the contents of pg_controldata, or non-True value if pg_controldata call failed """
|
""" return the contents of pg_controldata, or non-True value if pg_controldata call failed """
|
||||||
# Don't try to call pg_controldata during backup restore
|
# Don't try to call pg_controldata during backup restore
|
||||||
if self._version_file_exists() and self.state != 'creating replica':
|
if self._version_file_exists() and self.state != PostgresqlState.CREATING_REPLICA:
|
||||||
try:
|
try:
|
||||||
env = {**os.environ, 'LANG': 'C', 'LC_ALL': 'C'}
|
env = {**os.environ, 'LANG': 'C', 'LC_ALL': 'C'}
|
||||||
data = subprocess.check_output([self.pgcommand('pg_controldata'), self._data_dir], env=env)
|
data = subprocess.check_output([self.pgcommand('pg_controldata'), self._data_dir], env=env)
|
||||||
@@ -1041,8 +1079,8 @@ class Postgresql(object):
|
|||||||
data = filter(lambda e: ':' in e, data.decode('utf-8').splitlines())
|
data = filter(lambda e: ':' in e, data.decode('utf-8').splitlines())
|
||||||
# pg_controldata output depends on major version. Some of parameters are prefixed by 'Current '
|
# pg_controldata output depends on major version. Some of parameters are prefixed by 'Current '
|
||||||
return {k.replace('Current ', '', 1): v.strip() for k, v in map(lambda e: e.split(':', 1), data)}
|
return {k.replace('Current ', '', 1): v.strip() for k, v in map(lambda e: e.split(':', 1), data)}
|
||||||
except subprocess.CalledProcessError:
|
except Exception as e:
|
||||||
logger.exception("Error when calling pg_controldata")
|
logger.error("Error when calling pg_controldata: %r", e)
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
def waldump(self, timeline: Union[int, str], lsn: str, limit: int) -> Tuple[Optional[bytes], Optional[bytes]]:
|
def waldump(self, timeline: Union[int, str], lsn: str, limit: int) -> Tuple[Optional[bytes], Optional[bytes]]:
|
||||||
@@ -1102,14 +1140,15 @@ class Postgresql(object):
|
|||||||
logger.exception('Failed to read and parse %s', (history_path,))
|
logger.exception('Failed to read and parse %s', (history_path,))
|
||||||
return history
|
return history
|
||||||
|
|
||||||
def follow(self, member: Union[Leader, Member, None], role: str = 'replica',
|
def follow(self, member: Union[Leader, Member, None], role: PostgresqlRole = PostgresqlRole.REPLICA,
|
||||||
timeout: Optional[float] = None, do_reload: bool = False) -> Optional[bool]:
|
timeout: Optional[float] = None, do_reload: bool = False) -> Optional[bool]:
|
||||||
"""Reconfigure postgres to follow a new member or use different recovery parameters.
|
"""Reconfigure postgres to follow a new member or use different recovery parameters.
|
||||||
|
|
||||||
Method may call `on_role_change` callback if role is changing.
|
Method may call `on_role_change` callback if role is changing.
|
||||||
|
|
||||||
:param member: The member to follow
|
:param member: The member to follow
|
||||||
:param role: The desired role, normally 'replica', but could also be a 'standby_leader'
|
:param role: The desired role, one of :class:`~misc.PostgresqlRole` values, normally
|
||||||
|
:class:`~misc.PostgresqlRole.REPLICA`, but could also be a :class:`~misc.PostgresqlRole.STANDBY_LEADER`
|
||||||
:param timeout: start timeout, how long should the `start()` method wait for postgres accepting connections
|
:param timeout: start timeout, how long should the `start()` method wait for postgres accepting connections
|
||||||
:param do_reload: indicates that after updating postgresql.conf we just need to do a reload instead of restart
|
:param do_reload: indicates that after updating postgresql.conf we just need to do a reload instead of restart
|
||||||
|
|
||||||
@@ -1127,8 +1166,9 @@ class Postgresql(object):
|
|||||||
# and we know for sure that postgres was already running before, we will only execute on_role_change
|
# and we know for sure that postgres was already running before, we will only execute on_role_change
|
||||||
# callback and prevent execution of on_restart/on_start callback.
|
# callback and prevent execution of on_restart/on_start callback.
|
||||||
# If the role remains the same (replica or standby_leader), we will execute on_start or on_restart
|
# If the role remains the same (replica or standby_leader), we will execute on_start or on_restart
|
||||||
change_role = self.cb_called and (self.role in ('master', 'primary', 'demoted')
|
change_role = self.cb_called and \
|
||||||
or not {'standby_leader', 'replica'} - {self.role, role})
|
(self.role in (PostgresqlRole.PRIMARY, PostgresqlRole.DEMOTED)
|
||||||
|
or not {PostgresqlRole.STANDBY_LEADER, PostgresqlRole.REPLICA} - {self.role, role})
|
||||||
if change_role:
|
if change_role:
|
||||||
self.__cb_pending = CallbackAction.NOOP
|
self.__cb_pending = CallbackAction.NOOP
|
||||||
|
|
||||||
@@ -1153,7 +1193,7 @@ class Postgresql(object):
|
|||||||
for _ in polling_loop(wait_seconds):
|
for _ in polling_loop(wait_seconds):
|
||||||
data = self.controldata()
|
data = self.controldata()
|
||||||
if data.get('Database cluster state') == 'in production':
|
if data.get('Database cluster state') == 'in production':
|
||||||
self.set_role('master')
|
self.set_role(PostgresqlRole.PRIMARY)
|
||||||
return True
|
return True
|
||||||
|
|
||||||
def _pre_promote(self) -> bool:
|
def _pre_promote(self) -> bool:
|
||||||
@@ -1188,7 +1228,7 @@ class Postgresql(object):
|
|||||||
|
|
||||||
def promote(self, wait_seconds: int, task: CriticalTask,
|
def promote(self, wait_seconds: int, task: CriticalTask,
|
||||||
before_promote: Optional[Callable[..., Any]] = None) -> Optional[bool]:
|
before_promote: Optional[Callable[..., Any]] = None) -> Optional[bool]:
|
||||||
if self.role in ('promoted', 'master', 'primary'):
|
if self.role in (PostgresqlRole.PROMOTED, PostgresqlRole.PRIMARY):
|
||||||
return True
|
return True
|
||||||
|
|
||||||
ret = self._pre_promote()
|
ret = self._pre_promote()
|
||||||
@@ -1212,7 +1252,7 @@ class Postgresql(object):
|
|||||||
|
|
||||||
ret = self.pg_ctl('promote', '-W')
|
ret = self.pg_ctl('promote', '-W')
|
||||||
if ret:
|
if ret:
|
||||||
self.set_role('promoted')
|
self.set_role(PostgresqlRole.PROMOTED)
|
||||||
self.call_nowait(CallbackAction.ON_ROLE_CHANGE)
|
self.call_nowait(CallbackAction.ON_ROLE_CHANGE)
|
||||||
ret = self._wait_promote(wait_seconds)
|
ret = self._wait_promote(wait_seconds)
|
||||||
return ret
|
return ret
|
||||||
@@ -1232,8 +1272,9 @@ class Postgresql(object):
|
|||||||
received_location = self.received_location()
|
received_location = self.received_location()
|
||||||
pg_control_timeline = self._cluster_info_state_get('pg_control_timeline')
|
pg_control_timeline = self._cluster_info_state_get('pg_control_timeline')
|
||||||
else:
|
else:
|
||||||
timeline, wal_position, replayed_location, received_location, _, pg_control_timeline = \
|
timeline, wal_position, replayed_location, received_location, _, pg_control_timeline, _, write_location = \
|
||||||
self._query(self.cluster_info_query)[0][:6]
|
self._query(self.cluster_info_query)[0][:8]
|
||||||
|
received_location = max(received_location or 0, write_location or 0)
|
||||||
|
|
||||||
wal_position = self._wal_position(bool(timeline), wal_position, received_location, replayed_location)
|
wal_position = self._wal_position(bool(timeline), wal_position, received_location, replayed_location)
|
||||||
return timeline, wal_position, pg_control_timeline
|
return timeline, wal_position, pg_control_timeline
|
||||||
@@ -1319,7 +1360,7 @@ class Postgresql(object):
|
|||||||
logger.exception("Could not rename data directory %s", self._data_dir)
|
logger.exception("Could not rename data directory %s", self._data_dir)
|
||||||
|
|
||||||
def remove_data_directory(self) -> None:
|
def remove_data_directory(self) -> None:
|
||||||
self.set_role('uninitialized')
|
self.set_role(PostgresqlRole.UNINITIALIZED)
|
||||||
logger.info('Removing data directory: %s', self._data_dir)
|
logger.info('Removing data directory: %s', self._data_dir)
|
||||||
try:
|
try:
|
||||||
if os.path.islink(self._data_dir):
|
if os.path.islink(self._data_dir):
|
||||||
|
|||||||
@@ -1,7 +1,25 @@
|
|||||||
parameters:
|
parameters:
|
||||||
|
allow_alter_system:
|
||||||
|
- type: Bool
|
||||||
|
version_from: 170000
|
||||||
allow_in_place_tablespaces:
|
allow_in_place_tablespaces:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 150000
|
version_from: 150000
|
||||||
|
- type: Bool
|
||||||
|
version_from: 140005
|
||||||
|
version_till: 140099
|
||||||
|
- type: Bool
|
||||||
|
version_from: 130008
|
||||||
|
version_till: 130099
|
||||||
|
- type: Bool
|
||||||
|
version_from: 120012
|
||||||
|
version_till: 120099
|
||||||
|
- type: Bool
|
||||||
|
version_from: 110017
|
||||||
|
version_till: 110099
|
||||||
|
- type: Bool
|
||||||
|
version_from: 100022
|
||||||
|
version_till: 100099
|
||||||
allow_system_table_mods:
|
allow_system_table_mods:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
@@ -245,6 +263,12 @@ parameters:
|
|||||||
version_from: 90300
|
version_from: 90300
|
||||||
min_val: 0
|
min_val: 0
|
||||||
max_val: 1000
|
max_val: 1000
|
||||||
|
commit_timestamp_buffers:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 0
|
||||||
|
max_val: 131072
|
||||||
|
unit: 8kB
|
||||||
compute_query_id:
|
compute_query_id:
|
||||||
- type: EnumBool
|
- type: EnumBool
|
||||||
version_from: 140000
|
version_from: 140000
|
||||||
@@ -299,6 +323,7 @@ parameters:
|
|||||||
db_user_namespace:
|
db_user_namespace:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
|
version_till: 170000
|
||||||
deadlock_timeout:
|
deadlock_timeout:
|
||||||
- type: Integer
|
- type: Integer
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
@@ -313,6 +338,12 @@ parameters:
|
|||||||
debug_io_direct:
|
debug_io_direct:
|
||||||
- type: String
|
- type: String
|
||||||
version_from: 160000
|
version_from: 160000
|
||||||
|
debug_logical_replication_streaming:
|
||||||
|
- type: Enum
|
||||||
|
version_from: 170000
|
||||||
|
possible_values:
|
||||||
|
- buffered
|
||||||
|
- immediate
|
||||||
debug_parallel_query:
|
debug_parallel_query:
|
||||||
- type: EnumBool
|
- type: EnumBool
|
||||||
version_from: 160000
|
version_from: 160000
|
||||||
@@ -406,6 +437,9 @@ parameters:
|
|||||||
enable_gathermerge:
|
enable_gathermerge:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 100000
|
version_from: 100000
|
||||||
|
enable_group_by_reordering:
|
||||||
|
- type: Bool
|
||||||
|
version_from: 170000
|
||||||
enable_hashagg:
|
enable_hashagg:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
@@ -466,6 +500,9 @@ parameters:
|
|||||||
event_source:
|
event_source:
|
||||||
- type: String
|
- type: String
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
|
event_triggers:
|
||||||
|
- type: Bool
|
||||||
|
version_from: 170000
|
||||||
exit_on_error:
|
exit_on_error:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
@@ -607,6 +644,12 @@ parameters:
|
|||||||
ignore_system_indexes:
|
ignore_system_indexes:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
|
io_combine_limit:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 1
|
||||||
|
max_val: 32
|
||||||
|
unit: 8kB
|
||||||
IntervalStyle:
|
IntervalStyle:
|
||||||
- type: Enum
|
- type: Enum
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
@@ -875,6 +918,7 @@ parameters:
|
|||||||
logical_replication_mode:
|
logical_replication_mode:
|
||||||
- type: Enum
|
- type: Enum
|
||||||
version_from: 160000
|
version_from: 160000
|
||||||
|
version_till: 170000
|
||||||
possible_values:
|
possible_values:
|
||||||
- buffered
|
- buffered
|
||||||
- immediate
|
- immediate
|
||||||
@@ -919,6 +963,11 @@ parameters:
|
|||||||
version_from: 100000
|
version_from: 100000
|
||||||
min_val: 0
|
min_val: 0
|
||||||
max_val: 262143
|
max_val: 262143
|
||||||
|
max_notify_queue_pages:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 64
|
||||||
|
max_val: 2147483647
|
||||||
max_parallel_apply_workers_per_subscription:
|
max_parallel_apply_workers_per_subscription:
|
||||||
- type: Integer
|
- type: Integer
|
||||||
version_from: 160000
|
version_from: 160000
|
||||||
@@ -1072,9 +1121,28 @@ parameters:
|
|||||||
min_val: 2
|
min_val: 2
|
||||||
max_val: 2147483647
|
max_val: 2147483647
|
||||||
unit: MB
|
unit: MB
|
||||||
|
multixact_member_buffers:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 16
|
||||||
|
max_val: 131072
|
||||||
|
unit: 8kB
|
||||||
|
multixact_offset_buffers:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 16
|
||||||
|
max_val: 131072
|
||||||
|
unit: 8kB
|
||||||
|
notify_buffers:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 16
|
||||||
|
max_val: 131072
|
||||||
|
unit: 8kB
|
||||||
old_snapshot_threshold:
|
old_snapshot_threshold:
|
||||||
- type: Integer
|
- type: Integer
|
||||||
version_from: 90600
|
version_from: 90600
|
||||||
|
version_till: 170000
|
||||||
min_val: -1
|
min_val: -1
|
||||||
max_val: 86400
|
max_val: 86400
|
||||||
unit: min
|
unit: min
|
||||||
@@ -1169,6 +1237,24 @@ parameters:
|
|||||||
restart_after_crash:
|
restart_after_crash:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
|
restrict_nonsystem_relation_kind:
|
||||||
|
- type: String
|
||||||
|
version_from: 170000
|
||||||
|
- type: String
|
||||||
|
version_from: 160004
|
||||||
|
version_till: 160099
|
||||||
|
- type: String
|
||||||
|
version_from: 150008
|
||||||
|
version_till: 150099
|
||||||
|
- type: String
|
||||||
|
version_from: 140013
|
||||||
|
version_till: 140099
|
||||||
|
- type: String
|
||||||
|
version_from: 130016
|
||||||
|
version_till: 130099
|
||||||
|
- type: String
|
||||||
|
version_from: 120020
|
||||||
|
version_till: 120099
|
||||||
row_security:
|
row_security:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90500
|
version_from: 90500
|
||||||
@@ -1191,6 +1277,12 @@ parameters:
|
|||||||
version_from: 90300
|
version_from: 90300
|
||||||
min_val: 0
|
min_val: 0
|
||||||
max_val: 1.79769e+308
|
max_val: 1.79769e+308
|
||||||
|
serializable_buffers:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 16
|
||||||
|
max_val: 131072
|
||||||
|
unit: 8kB
|
||||||
session_preload_libraries:
|
session_preload_libraries:
|
||||||
- type: String
|
- type: String
|
||||||
version_from: 90400
|
version_from: 90400
|
||||||
@@ -1300,6 +1392,15 @@ parameters:
|
|||||||
- type: String
|
- type: String
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
version_till: 150000
|
version_till: 150000
|
||||||
|
subtransaction_buffers:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 0
|
||||||
|
max_val: 131072
|
||||||
|
unit: 8kB
|
||||||
|
summarize_wal:
|
||||||
|
- type: Bool
|
||||||
|
version_from: 170000
|
||||||
superuser_reserved_connections:
|
superuser_reserved_connections:
|
||||||
- type: Integer
|
- type: Integer
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
@@ -1310,9 +1411,15 @@ parameters:
|
|||||||
version_from: 90600
|
version_from: 90600
|
||||||
min_val: 0
|
min_val: 0
|
||||||
max_val: 262143
|
max_val: 262143
|
||||||
|
sync_replication_slots:
|
||||||
|
- type: Bool
|
||||||
|
version_from: 170000
|
||||||
synchronize_seqscans:
|
synchronize_seqscans:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
|
synchronized_standby_slots:
|
||||||
|
- type: String
|
||||||
|
version_from: 170000
|
||||||
synchronous_commit:
|
synchronous_commit:
|
||||||
- type: EnumBool
|
- type: EnumBool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
@@ -1394,12 +1501,16 @@ parameters:
|
|||||||
timezone_abbreviations:
|
timezone_abbreviations:
|
||||||
- type: String
|
- type: String
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
|
trace_connection_negotiation:
|
||||||
|
- type: Bool
|
||||||
|
version_from: 170000
|
||||||
trace_notify:
|
trace_notify:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
trace_recovery_messages:
|
trace_recovery_messages:
|
||||||
- type: Enum
|
- type: Enum
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
|
version_till: 170000
|
||||||
possible_values:
|
possible_values:
|
||||||
- debug5
|
- debug5
|
||||||
- debug4
|
- debug4
|
||||||
@@ -1452,6 +1563,12 @@ parameters:
|
|||||||
track_wal_io_timing:
|
track_wal_io_timing:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 140000
|
version_from: 140000
|
||||||
|
transaction_buffers:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 0
|
||||||
|
max_val: 131072
|
||||||
|
unit: 8kB
|
||||||
transaction_deferrable:
|
transaction_deferrable:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
@@ -1466,6 +1583,12 @@ parameters:
|
|||||||
transaction_read_only:
|
transaction_read_only:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
|
transaction_timeout:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 0
|
||||||
|
max_val: 2147483647
|
||||||
|
unit: ms
|
||||||
transform_null_equals:
|
transform_null_equals:
|
||||||
- type: Bool
|
- type: Bool
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
@@ -1664,6 +1787,12 @@ parameters:
|
|||||||
min_val: 0
|
min_val: 0
|
||||||
max_val: 2147483647
|
max_val: 2147483647
|
||||||
unit: kB
|
unit: kB
|
||||||
|
wal_summary_keep_time:
|
||||||
|
- type: Integer
|
||||||
|
version_from: 170000
|
||||||
|
min_val: 0
|
||||||
|
max_val: 35791394
|
||||||
|
unit: min
|
||||||
wal_sync_method:
|
wal_sync_method:
|
||||||
- type: Enum
|
- type: Enum
|
||||||
version_from: 90300
|
version_from: 90300
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user