mirror of
https://github.com/outbackdingo/patroni.git
synced 2026-08-26 15:40:21 +00:00
Compare commits
@@ -18,11 +18,14 @@ def install_requirements(what):
|
|||||||
finally:
|
finally:
|
||||||
sys.path = old_path
|
sys.path = old_path
|
||||||
requirements = ['mock>=2.0.0', 'flake8', 'pytest', 'pytest-cov'] if what == 'all' else ['behave']
|
requirements = ['mock>=2.0.0', 'flake8', 'pytest', 'pytest-cov'] if what == 'all' else ['behave']
|
||||||
requirements += ['psycopg2-binary', 'coverage']
|
requirements += ['coverage']
|
||||||
|
# try to split tests between psycopg2 and psycopg3
|
||||||
|
requirements += ['psycopg[binary]'] if sys.version_info >= (3, 6, 0) and\
|
||||||
|
(sys.platform != 'darwin' or what == 'etcd3') else ['psycopg2-binary']
|
||||||
for r in read('requirements.txt').split('\n'):
|
for r in read('requirements.txt').split('\n'):
|
||||||
r = r.strip()
|
r = r.strip()
|
||||||
if r != '':
|
if r != '':
|
||||||
extras = {e for e, v in EXTRAS_REQUIRE.items() if v and r.startswith(v[0])}
|
extras = {e for e, v in EXTRAS_REQUIRE.items() if v and any(r.startswith(x) for x in v)}
|
||||||
if not extras or what == 'all' or what in extras:
|
if not extras or what == 'all' or what in extras:
|
||||||
requirements.append(r)
|
requirements.append(r)
|
||||||
|
|
||||||
@@ -33,13 +36,17 @@ def install_requirements(what):
|
|||||||
|
|
||||||
|
|
||||||
def install_packages(what):
|
def install_packages(what):
|
||||||
|
from mapping import versions
|
||||||
|
|
||||||
packages = {
|
packages = {
|
||||||
'zookeeper': ['zookeeper', 'zookeeper-bin', 'zookeeperd'],
|
'zookeeper': ['zookeeper', 'zookeeper-bin', 'zookeeperd'],
|
||||||
'consul': ['consul'],
|
'consul': ['consul'],
|
||||||
}
|
}
|
||||||
packages['exhibitor'] = packages['zookeeper']
|
packages['exhibitor'] = packages['zookeeper']
|
||||||
packages = packages.get(what, [])
|
packages = packages.get(what, [])
|
||||||
ver = str({'etcd': '9.6', 'etcd3': '9.6', 'consul': 10, 'exhibitor': 11, 'kubernetes': 12, 'raft': 13}.get(what))
|
ver = versions.get(what)
|
||||||
|
subprocess.call(['sudo', 'sed', '-i', 's/pgdg main.*$/pgdg main {0}/'.format(ver),
|
||||||
|
'/etc/apt/sources.list.d/pgdg.list'])
|
||||||
subprocess.call(['sudo', 'apt-get', 'update', '-y'])
|
subprocess.call(['sudo', 'apt-get', 'update', '-y'])
|
||||||
return subprocess.call(['sudo', 'apt-get', 'install', '-y', 'postgresql-' + ver, 'expect-dev', 'wget'] + packages)
|
return subprocess.call(['sudo', 'apt-get', 'install', '-y', 'postgresql-' + ver, 'expect-dev', 'wget'] + packages)
|
||||||
|
|
||||||
@@ -103,7 +110,7 @@ def install_etcd():
|
|||||||
|
|
||||||
|
|
||||||
def install_postgres():
|
def install_postgres():
|
||||||
version = os.environ.get('PGVERSION', '12.1-1')
|
version = os.environ.get('PGVERSION', '14.1-1')
|
||||||
platform = {'darwin': 'osx', 'win32': 'windows-x64', 'cygwin': 'windows-x64'}[sys.platform]
|
platform = {'darwin': 'osx', 'win32': 'windows-x64', 'cygwin': 'windows-x64'}[sys.platform]
|
||||||
name = 'postgresql-{0}-{1}-binaries.zip'.format(version, platform)
|
name = 'postgresql-{0}-{1}-binaries.zip'.format(version, platform)
|
||||||
get_file('http://get.enterprisedb.com/postgresql/' + name, name)
|
get_file('http://get.enterprisedb.com/postgresql/' + name, name)
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
versions = {'etcd': '9.6', 'etcd3': '14', 'consul': '13', 'exhibitor': '12', 'raft': '11', 'kubernetes': '14'}
|
||||||
@@ -23,18 +23,21 @@ def main():
|
|||||||
|
|
||||||
env = os.environ.copy()
|
env = os.environ.copy()
|
||||||
if sys.platform.startswith('linux'):
|
if sys.platform.startswith('linux'):
|
||||||
version = {'etcd': '9.6', 'etcd3': '9.6', 'consul': 10, 'exhibitor': 11, 'kubernetes': 12, 'raft': 13}.get(what)
|
from mapping import versions
|
||||||
|
|
||||||
|
version = versions.get(what)
|
||||||
path = '/usr/lib/postgresql/{0}/bin:.'.format(version)
|
path = '/usr/lib/postgresql/{0}/bin:.'.format(version)
|
||||||
unbuffer = ['timeout', '600', 'unbuffer']
|
unbuffer = ['timeout', '900', 'unbuffer']
|
||||||
|
args = ['--tags=-skip'] if what == 'etcd' else []
|
||||||
else:
|
else:
|
||||||
path = os.path.abspath(os.path.join('pgsql', 'bin'))
|
path = os.path.abspath(os.path.join('pgsql', 'bin'))
|
||||||
if sys.platform == 'darwin':
|
if sys.platform == 'darwin':
|
||||||
path += ':.'
|
path += ':.'
|
||||||
unbuffer = []
|
args = unbuffer = []
|
||||||
env['PATH'] = path + os.pathsep + env['PATH']
|
env['PATH'] = path + os.pathsep + env['PATH']
|
||||||
env['DCS'] = what
|
env['DCS'] = what
|
||||||
|
|
||||||
ret = subprocess.call(unbuffer + [sys.executable, '-m', 'behave'], env=env)
|
ret = subprocess.call(unbuffer + [sys.executable, '-m', 'behave'] + args, env=env)
|
||||||
|
|
||||||
if ret != 0:
|
if ret != 0:
|
||||||
if subprocess.call('grep . features/output/*_failed/*postgres?.*', shell=True) != 0:
|
if subprocess.call('grep . features/output/*_failed/*postgres?.*', shell=True) != 0:
|
||||||
|
|||||||
@@ -30,15 +30,6 @@ jobs:
|
|||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
if: matrix.os != 'windows'
|
if: matrix.os != 'windows'
|
||||||
|
|
||||||
- name: Set up Python 3.5
|
|
||||||
uses: actions/setup-python@v2
|
|
||||||
with:
|
|
||||||
python-version: 3.5
|
|
||||||
- name: Install dependencies
|
|
||||||
run: python .github/workflows/install_deps.py
|
|
||||||
- name: Run tests and flake8
|
|
||||||
run: python .github/workflows/run_tests.py
|
|
||||||
|
|
||||||
- name: Set up Python 3.6
|
- name: Set up Python 3.6
|
||||||
uses: actions/setup-python@v2
|
uses: actions/setup-python@v2
|
||||||
with:
|
with:
|
||||||
@@ -75,6 +66,15 @@ jobs:
|
|||||||
- name: Run tests and flake8
|
- name: Run tests and flake8
|
||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
|
- name: Set up Python 3.10
|
||||||
|
uses: actions/setup-python@v2
|
||||||
|
with:
|
||||||
|
python-version: '3.10'
|
||||||
|
- name: Install dependencies
|
||||||
|
run: python .github/workflows/install_deps.py
|
||||||
|
- name: Run tests and flake8
|
||||||
|
run: python .github/workflows/run_tests.py
|
||||||
|
|
||||||
- name: Combine coverage
|
- name: Combine coverage
|
||||||
run: python .github/workflows/run_tests.py combine
|
run: python .github/workflows/run_tests.py combine
|
||||||
|
|
||||||
@@ -88,50 +88,7 @@ jobs:
|
|||||||
GITHUB_TOKEN: ${{ secrets.github_token }}
|
GITHUB_TOKEN: ${{ secrets.github_token }}
|
||||||
run: python -m coveralls --service=github
|
run: python -m coveralls --service=github
|
||||||
|
|
||||||
- name: Run codacy-coverage-reporter
|
|
||||||
uses: codacy/codacy-coverage-reporter-action@master
|
|
||||||
env:
|
|
||||||
SECRETS_AVAILABLE: ${{ secrets.CODACY_PROJECT_TOKEN != '' }}
|
|
||||||
with:
|
|
||||||
project-token: ${{ secrets.CODACY_PROJECT_TOKEN }}
|
|
||||||
coverage-reports: coverage.xml
|
|
||||||
if: ${{ matrix.os == 'ubuntu' && env.SECRETS_AVAILABLE == 'true' }}
|
|
||||||
|
|
||||||
behave:
|
behave:
|
||||||
runs-on: ${{ matrix.os }}-latest
|
|
||||||
env:
|
|
||||||
DCS: ${{ matrix.dcs }}
|
|
||||||
ETCDVERSION: 3.3.13
|
|
||||||
strategy:
|
|
||||||
fail-fast: false
|
|
||||||
matrix:
|
|
||||||
os: [ubuntu]
|
|
||||||
python-version: [2.7, 3.5, 3.8]
|
|
||||||
dcs: [etcd, etcd3, consul, exhibitor, kubernetes, raft]
|
|
||||||
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v1
|
|
||||||
- name: Set up Python
|
|
||||||
uses: actions/setup-python@v2
|
|
||||||
with:
|
|
||||||
python-version: ${{ matrix.python-version }}
|
|
||||||
- name: Install dependencies
|
|
||||||
run: python .github/workflows/install_deps.py
|
|
||||||
- name: Run behave tests
|
|
||||||
run: python .github/workflows/run_tests.py
|
|
||||||
- uses: actions/setup-python@v2
|
|
||||||
with:
|
|
||||||
python-version: 3.9
|
|
||||||
- name: Install coveralls
|
|
||||||
run: python -m pip install coveralls
|
|
||||||
- name: Upload Coverage
|
|
||||||
env:
|
|
||||||
COVERALLS_FLAG_NAME: behave-${{ matrix.os }}-${{ matrix.dcs }}-${{ matrix.python-version }}
|
|
||||||
COVERALLS_PARALLEL: 'true'
|
|
||||||
GITHUB_TOKEN: ${{ secrets.github_token }}
|
|
||||||
run: python -m coveralls --service=github
|
|
||||||
|
|
||||||
behavem:
|
|
||||||
runs-on: ${{ matrix.os }}-latest
|
runs-on: ${{ matrix.os }}-latest
|
||||||
env:
|
env:
|
||||||
DCS: ${{ matrix.dcs }}
|
DCS: ${{ matrix.dcs }}
|
||||||
@@ -140,9 +97,22 @@ jobs:
|
|||||||
strategy:
|
strategy:
|
||||||
fail-fast: false
|
fail-fast: false
|
||||||
matrix:
|
matrix:
|
||||||
os: [macos] #, windows]
|
os: [ubuntu]
|
||||||
python-version: [3.7]
|
python-version: [2.7, 3.6, 3.9]
|
||||||
dcs: [etcd, etcd3, raft]
|
dcs: [etcd, etcd3, consul, exhibitor, kubernetes, raft]
|
||||||
|
exclude:
|
||||||
|
- dcs: kubernetes
|
||||||
|
python-version: 2.7
|
||||||
|
include:
|
||||||
|
- os: macos
|
||||||
|
python-version: 3.7
|
||||||
|
dcs: raft
|
||||||
|
- os: macos
|
||||||
|
python-version: 3.8
|
||||||
|
dcs: etcd
|
||||||
|
- os: macos
|
||||||
|
python-version: '3.10'
|
||||||
|
dcs: etcd3
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v1
|
- uses: actions/checkout@v1
|
||||||
@@ -150,10 +120,16 @@ jobs:
|
|||||||
uses: actions/setup-python@v2
|
uses: actions/setup-python@v2
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python-version }}
|
python-version: ${{ matrix.python-version }}
|
||||||
|
- name: Add postgresql apt repo
|
||||||
|
run: sudo sh -c 'echo "deb http://apt.postgresql.org/pub/repos/apt $(lsb_release -cs)-pgdg main" > /etc/apt/sources.list.d/pgdg.list'
|
||||||
|
if: matrix.os == 'ubuntu'
|
||||||
- name: Install dependencies
|
- name: Install dependencies
|
||||||
run: python .github/workflows/install_deps.py
|
run: python .github/workflows/install_deps.py
|
||||||
- name: Run behave tests
|
- name: Run behave tests
|
||||||
run: python .github/workflows/run_tests.py
|
run: python .github/workflows/run_tests.py
|
||||||
|
- uses: actions/setup-python@v2
|
||||||
|
with:
|
||||||
|
python-version: '3.10'
|
||||||
- name: Install coveralls
|
- name: Install coveralls
|
||||||
run: python -m pip install coveralls
|
run: python -m pip install coveralls
|
||||||
- name: Upload Coverage
|
- name: Upload Coverage
|
||||||
@@ -165,7 +141,7 @@ jobs:
|
|||||||
|
|
||||||
coveralls-finish:
|
coveralls-finish:
|
||||||
name: Finalize coveralls.io
|
name: Finalize coveralls.io
|
||||||
needs: [unit, behave, behavem]
|
needs: [unit, behave]
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/setup-python@v2
|
- uses: actions/setup-python@v2
|
||||||
|
|||||||
+13
-7
@@ -1,4 +1,4 @@
|
|||||||
|Build Status| |Coverage Status|
|
|Tests Status| |Coverage Status|
|
||||||
|
|
||||||
Patroni: A Template for PostgreSQL HA with ZooKeeper, etcd or Consul
|
Patroni: A Template for PostgreSQL HA with ZooKeeper, etcd or Consul
|
||||||
--------------------------------------------------------------------
|
--------------------------------------------------------------------
|
||||||
@@ -12,7 +12,7 @@ Patroni is a template for you to create your own customized, high-availability s
|
|||||||
|
|
||||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely.
|
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely.
|
||||||
|
|
||||||
Currently supported PostgreSQL versions: 9.3 to 13.
|
Currently supported PostgreSQL versions: 9.3 to 14.
|
||||||
|
|
||||||
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the `Kubernetes <https://github.com/zalando/patroni/blob/master/docs/kubernetes.rst>`__ chapter of the Patroni documentation.
|
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the `Kubernetes <https://github.com/zalando/patroni/blob/master/docs/kubernetes.rst>`__ chapter of the Patroni documentation.
|
||||||
|
|
||||||
@@ -61,7 +61,7 @@ To install requirements on a Mac, run the following:
|
|||||||
|
|
||||||
brew install postgresql etcd haproxy libyaml python
|
brew install postgresql etcd haproxy libyaml python
|
||||||
|
|
||||||
**Psycopg2**
|
**Psycopg**
|
||||||
|
|
||||||
Starting from `psycopg2-2.8 <http://initd.org/psycopg/articles/2019/04/04/psycopg-28-released/>`__ the binary version of psycopg2 will no longer be installed by default. Installing it from the source code requires C compiler and postgres+python dev packages.
|
Starting from `psycopg2-2.8 <http://initd.org/psycopg/articles/2019/04/04/psycopg-28-released/>`__ the binary version of psycopg2 will no longer be installed by default. Installing it from the source code requires C compiler and postgres+python dev packages.
|
||||||
Since in the python world it is not possible to specify dependency as ``psycopg2 OR psycopg2-binary`` you will have to decide how to install it.
|
Since in the python world it is not possible to specify dependency as ``psycopg2 OR psycopg2-binary`` you will have to decide how to install it.
|
||||||
@@ -88,6 +88,12 @@ There are a few options available:
|
|||||||
|
|
||||||
pip install psycopg2>=2.5.4
|
pip install psycopg2>=2.5.4
|
||||||
|
|
||||||
|
4. Use psycopg 3.0 instead of psycopg2
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
|
pip install psycopg[binary]
|
||||||
|
|
||||||
**General installation for pip**
|
**General installation for pip**
|
||||||
|
|
||||||
Patroni can be installed with pip:
|
Patroni can be installed with pip:
|
||||||
@@ -119,7 +125,7 @@ For example, the command in order to install Patroni together with dependencies
|
|||||||
|
|
||||||
pip install patroni[etcd,aws]
|
pip install patroni[etcd,aws]
|
||||||
|
|
||||||
Note that external tools to call in the replica creation or custom bootstap scripts (i.e. WAL-E) should be installed independently of Patroni.
|
Note that external tools to call in the replica creation or custom bootstrap scripts (i.e. WAL-E) should be installed independently of Patroni.
|
||||||
|
|
||||||
=======================
|
=======================
|
||||||
Running and Configuring
|
Running and Configuring
|
||||||
@@ -171,7 +177,7 @@ Applications Should Not Use Superusers
|
|||||||
|
|
||||||
When connecting from an application, always use a non-superuser. Patroni requires access to the database to function properly. By using a superuser from an application, you can potentially use the entire connection pool, including the connections reserved for superusers, with the ``superuser_reserved_connections`` setting. If Patroni cannot access the Primary because the connection pool is full, behavior will be undesirable.
|
When connecting from an application, always use a non-superuser. Patroni requires access to the database to function properly. By using a superuser from an application, you can potentially use the entire connection pool, including the connections reserved for superusers, with the ``superuser_reserved_connections`` setting. If Patroni cannot access the Primary because the connection pool is full, behavior will be undesirable.
|
||||||
|
|
||||||
.. |Build Status| image:: https://travis-ci.org/zalando/patroni.svg?branch=master
|
.. |Tests Status| image:: https://github.com/zalando/patroni/actions/workflows/tests.yaml/badge.svg
|
||||||
:target: https://travis-ci.org/zalando/patroni
|
:target: https://github.com/zalando/patroni/actions/workflows/tests.yaml?query=branch%3Amaster
|
||||||
.. |Coverage Status| image:: https://coveralls.io/repos/zalando/patroni/badge.svg?branch=master
|
.. |Coverage Status| image:: https://coveralls.io/repos/zalando/patroni/badge.svg?branch=master
|
||||||
:target: https://coveralls.io/r/zalando/patroni?branch=master
|
:target: https://coveralls.io/github/zalando/patroni?branch=master
|
||||||
|
|||||||
@@ -59,7 +59,8 @@ Etcd
|
|||||||
- **PATRONI\_ETCD\_USE\_PROXIES**: If this parameter is set to true, Patroni will consider **hosts** as a list of proxies and will not perform a topology discovery of etcd cluster but stick to a fixed list of **hosts**.
|
- **PATRONI\_ETCD\_USE\_PROXIES**: If this parameter is set to true, Patroni will consider **hosts** as a list of proxies and will not perform a topology discovery of etcd cluster but stick to a fixed list of **hosts**.
|
||||||
- **PATRONI\_ETCD\_PROTOCOL**: http or https, if not specified http is used. If the **url** or **proxy** is specified - will take protocol from them.
|
- **PATRONI\_ETCD\_PROTOCOL**: http or https, if not specified http is used. If the **url** or **proxy** is specified - will take protocol from them.
|
||||||
- **PATRONI\_ETCD\_HOST**: the host:port for the etcd endpoint.
|
- **PATRONI\_ETCD\_HOST**: the host:port for the etcd endpoint.
|
||||||
- **PATRONI\_ETCD\_SRV**: Domain to search the SRV record(s) for cluster autodiscovery.
|
- **PATRONI\_ETCD\_SRV**: Domain to search the SRV record(s) for cluster autodiscovery. Patroni will try to query these SRV service names for specified domain (in that order until first success): ``_etcd-client-ssl``, ``_etcd-client``, ``_etcd-ssl``, ``_etcd``, ``_etcd-server-ssl``, ``_etcd-server``. If SRV records for ``_etcd-server-ssl`` or ``_etcd-server`` are retrieved then ETCD peer protocol is used do query ETCD for available members. Otherwise hosts from SRV records will be used.
|
||||||
|
- **PATRONI\_ETCD\_SRV\_SUFFIX**: Configures a suffix to the SRV name that is queried during discovery. Use this flag to differentiate between multiple etcd clusters under the same domain. Works only with conjunction with **PATRONI\_ETCD\_SRV**. For example, if ``PATRONI_ETCD_SRV_SUFFIX=foo`` and ``PATRONI_ETCD_SRV=example.org`` are set, the following DNS SRV query is made:``_etcd-client-ssl-foo._tcp.example.com`` (and so on for every possible ETCD SRV service name).
|
||||||
- **PATRONI\_ETCD\_USERNAME**: username for etcd authentication.
|
- **PATRONI\_ETCD\_USERNAME**: username for etcd authentication.
|
||||||
- **PATRONI\_ETCD\_PASSWORD**: password for etcd authentication.
|
- **PATRONI\_ETCD\_PASSWORD**: password for etcd authentication.
|
||||||
- **PATRONI\_ETCD\_CACERT**: The ca certificate. If present it will enable validation.
|
- **PATRONI\_ETCD\_CACERT**: The ca certificate. If present it will enable validation.
|
||||||
@@ -83,6 +84,7 @@ ZooKeeper
|
|||||||
- **PATRONI\_ZOOKEEPER\_KEY**: (optional) File with the client key.
|
- **PATRONI\_ZOOKEEPER\_KEY**: (optional) File with the client key.
|
||||||
- **PATRONI\_ZOOKEEPER\_KEY\_PASSWORD**: (optional) The client key password.
|
- **PATRONI\_ZOOKEEPER\_KEY\_PASSWORD**: (optional) The client key password.
|
||||||
- **PATRONI\_ZOOKEEPER\_VERIFY**: (optional) Whether to verify certificate or not. Defaults to ``true``.
|
- **PATRONI\_ZOOKEEPER\_VERIFY**: (optional) Whether to verify certificate or not. Defaults to ``true``.
|
||||||
|
- **PATRONI\_ZOOKEEPER\_SET\_ACLS**: (optional) If set, configure Kazoo to apply a default ACL to each ZNode that it creates. ACLs will assume 'x509' schema and should be specified as a dictionary with the principal as the key and one or more permissions as a list in the value. Permissions may be one of ``CREATE``, ``READ``, ``WRITE``, ``DELETE`` or ``ADMIN``. For example, ``set_acls: {CN=principal1: [CREATE, READ], CN=principal2: [ALL]}``.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
It is required to install ``kazoo>=2.6.0`` to support SSL.
|
It is required to install ``kazoo>=2.6.0`` to support SSL.
|
||||||
@@ -105,6 +107,7 @@ Kubernetes
|
|||||||
- **PATRONI\_KUBERNETES\_USE\_ENDPOINTS**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
- **PATRONI\_KUBERNETES\_USE\_ENDPOINTS**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||||
- **PATRONI\_KUBERNETES\_POD\_IP**: (optional) IP address of the pod Patroni is running in. This value is required when `PATRONI_KUBERNETES_USE_ENDPOINTS` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
- **PATRONI\_KUBERNETES\_POD\_IP**: (optional) IP address of the pod Patroni is running in. This value is required when `PATRONI_KUBERNETES_USE_ENDPOINTS` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||||
- **PATRONI\_KUBERNETES\_PORTS**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``PATRONI_KUBERNETES_PORTS='[{"name": "postgresql", "port": 5432}]'`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `PATRONI_KUBERNETES_USE_ENDPOINTS` is set.
|
- **PATRONI\_KUBERNETES\_PORTS**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``PATRONI_KUBERNETES_PORTS='[{"name": "postgresql", "port": 5432}]'`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `PATRONI_KUBERNETES_USE_ENDPOINTS` is set.
|
||||||
|
- **PATRONI\_KUBERNETES\_CACERT**: (optional) Specifies the file with the CA_BUNDLE file with certificates of trusted CAs to use while verifying Kubernetes API SSL certs. If not provided, patroni will use the value provided by the ServiceAccount secret.
|
||||||
|
|
||||||
Raft
|
Raft
|
||||||
----
|
----
|
||||||
@@ -131,6 +134,7 @@ PostgreSQL
|
|||||||
- **PATRONI\_REPLICATION\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **PATRONI\_REPLICATION\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **PATRONI\_REPLICATION\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **PATRONI\_REPLICATION\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **PATRONI\_REPLICATION\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **PATRONI\_REPLICATION\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **PATRONI\_REPLICATION\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **PATRONI\_REPLICATION\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **PATRONI\_REPLICATION\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **PATRONI\_REPLICATION\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **PATRONI\_REPLICATION\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
- **PATRONI\_SUPERUSER\_USERNAME**: name for the superuser, set during initialization (initdb) and later used by Patroni to connect to the postgres. Also this user is used by pg_rewind.
|
- **PATRONI\_SUPERUSER\_USERNAME**: name for the superuser, set during initialization (initdb) and later used by Patroni to connect to the postgres. Also this user is used by pg_rewind.
|
||||||
@@ -141,6 +145,7 @@ PostgreSQL
|
|||||||
- **PATRONI\_SUPERUSER\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **PATRONI\_SUPERUSER\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **PATRONI\_SUPERUSER\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **PATRONI\_SUPERUSER\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **PATRONI\_SUPERUSER\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **PATRONI\_SUPERUSER\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **PATRONI\_SUPERUSER\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **PATRONI\_SUPERUSER\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **PATRONI\_SUPERUSER\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **PATRONI\_SUPERUSER\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **PATRONI\_SUPERUSER\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
- **PATRONI\_REWIND\_USERNAME**: name for the user for ``pg_rewind``; the user will be created during initialization of postgres 11+ and all necessary `permissions <https://www.postgresql.org/docs/11/app-pgrewind.html#id-1.9.5.8.8>`__ will be granted.
|
- **PATRONI\_REWIND\_USERNAME**: name for the user for ``pg_rewind``; the user will be created during initialization of postgres 11+ and all necessary `permissions <https://www.postgresql.org/docs/11/app-pgrewind.html#id-1.9.5.8.8>`__ will be granted.
|
||||||
@@ -151,6 +156,7 @@ PostgreSQL
|
|||||||
- **PATRONI\_REWIND\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **PATRONI\_REWIND\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **PATRONI\_REWIND\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **PATRONI\_REWIND\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **PATRONI\_REWIND\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **PATRONI\_REWIND\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **PATRONI\_REWIND\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **PATRONI\_REWIND\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **PATRONI\_REWIND\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **PATRONI\_REWIND\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **PATRONI\_REWIND\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
|
|
||||||
@@ -166,6 +172,8 @@ REST API
|
|||||||
- **PATRONI\_RESTAPI\_CAFILE**: Specifies the file with the CA_BUNDLE with certificates of trusted CAs to use while verifying client certs.
|
- **PATRONI\_RESTAPI\_CAFILE**: Specifies the file with the CA_BUNDLE with certificates of trusted CAs to use while verifying client certs.
|
||||||
- **PATRONI\_RESTAPI\_CIPHERS**: (optional) Specifies the permitted cipher suites (e.g. "ECDHE-RSA-AES256-GCM-SHA384:DHE-RSA-AES256-GCM-SHA384:ECDHE-RSA-AES128-GCM-SHA256:DHE-RSA-AES128-GCM-SHA256:!SSLv1:!SSLv2:!SSLv3:!TLSv1:!TLSv1.1")
|
- **PATRONI\_RESTAPI\_CIPHERS**: (optional) Specifies the permitted cipher suites (e.g. "ECDHE-RSA-AES256-GCM-SHA384:DHE-RSA-AES256-GCM-SHA384:ECDHE-RSA-AES128-GCM-SHA256:DHE-RSA-AES128-GCM-SHA256:!SSLv1:!SSLv2:!SSLv3:!TLSv1:!TLSv1.1")
|
||||||
- **PATRONI\_RESTAPI\_VERIFY\_CLIENT**: ``none`` (default), ``optional`` or ``required``. When ``none`` REST API will not check client certificates. When ``required`` client certificates are required for all REST API calls. When ``optional`` client certificates are required for all unsafe REST API endpoints. When ``required`` is used, then client authentication succeeds, if the certificate signature verification succeeds. For ``optional`` the client cert will only be checked for ``PUT``, ``POST``, ``PATCH``, and ``DELETE`` requests.
|
- **PATRONI\_RESTAPI\_VERIFY\_CLIENT**: ``none`` (default), ``optional`` or ``required``. When ``none`` REST API will not check client certificates. When ``required`` client certificates are required for all REST API calls. When ``optional`` client certificates are required for all unsafe REST API endpoints. When ``required`` is used, then client authentication succeeds, if the certificate signature verification succeeds. For ``optional`` the client cert will only be checked for ``PUT``, ``POST``, ``PATCH``, and ``DELETE`` requests.
|
||||||
|
- **PATRONI\_RESTAPI\_ALLOWLIST**: (optional): Specifies the set of hosts that are allowed to call unsafe REST API endpoints. The single element could be a host name, an IP address or a network address using CIDR notation. By default ``allow all`` is used. In case if ``allowlist`` or ``allowlist_include_members`` are set, anything that is not included is rejected.
|
||||||
|
- **PATRONI\_RESTAPI\_ALLOWLIST\_INCLUDE\_MEMBERS**: (optional): If set to ``true`` it allows accessing unsafe REST API endpoints from other cluster members registered in DCS (IP address or hostname is taken from the members ``api_url``). Be careful, it might happen that OS will use a different IP for outgoing connections.
|
||||||
- **PATRONI\_RESTAPI\_HTTP\_EXTRA\_HEADERS**: (optional) HTTP headers let the REST API server pass additional information with an HTTP response.
|
- **PATRONI\_RESTAPI\_HTTP\_EXTRA\_HEADERS**: (optional) HTTP headers let the REST API server pass additional information with an HTTP response.
|
||||||
- **PATRONI\_RESTAPI\_HTTPS\_EXTRA\_HEADERS**: (optional) HTTPS headers let the REST API server pass additional information with an HTTP response when TLS is enabled. This will also pass additional information set in ``http_extra_headers``.
|
- **PATRONI\_RESTAPI\_HTTPS\_EXTRA\_HEADERS**: (optional) HTTPS headers let the REST API server pass additional information with an HTTP response when TLS is enabled. This will also pass additional information set in ``http_extra_headers``.
|
||||||
|
|
||||||
|
|||||||
+8
-2
@@ -35,7 +35,7 @@ To install requirements on a Mac, run the following:
|
|||||||
|
|
||||||
.. _psycopg2_install_options:
|
.. _psycopg2_install_options:
|
||||||
|
|
||||||
**Psycopg2**
|
**Psycopg**
|
||||||
|
|
||||||
Starting from `psycopg2-2.8 <http://initd.org/psycopg/articles/2019/04/04/psycopg-28-released/>`__ the binary version of psycopg2 will no longer be installed by default. Installing it from the source code requires C compiler and postgres+python dev packages.
|
Starting from `psycopg2-2.8 <http://initd.org/psycopg/articles/2019/04/04/psycopg-28-released/>`__ the binary version of psycopg2 will no longer be installed by default. Installing it from the source code requires C compiler and postgres+python dev packages.
|
||||||
Since in the python world it is not possible to specify dependency as ``psycopg2 OR psycopg2-binary`` you will have to decide how to install it.
|
Since in the python world it is not possible to specify dependency as ``psycopg2 OR psycopg2-binary`` you will have to decide how to install it.
|
||||||
@@ -62,6 +62,12 @@ There are a few options available:
|
|||||||
|
|
||||||
pip install psycopg2>=2.5.4
|
pip install psycopg2>=2.5.4
|
||||||
|
|
||||||
|
4. Use psycopg 3.0 instead of psycopg2
|
||||||
|
|
||||||
|
::
|
||||||
|
|
||||||
|
pip install psycopg[binary]>=3.0.0
|
||||||
|
|
||||||
**General installation for pip**
|
**General installation for pip**
|
||||||
|
|
||||||
Patroni can be installed with pip:
|
Patroni can be installed with pip:
|
||||||
@@ -166,7 +172,7 @@ When connecting from an application, always use a non-superuser. Patroni require
|
|||||||
|
|
||||||
Testing Your HA Solution
|
Testing Your HA Solution
|
||||||
--------------------------------------
|
--------------------------------------
|
||||||
Testing an HA solution is a time consuming process, with many variables. This is particularly true considering a cross-platform application. You need a trained system administrator or a consultant to do this work. It is not something we can cover in depth in the documentaiton.
|
Testing an HA solution is a time consuming process, with many variables. This is particularly true considering a cross-platform application. You need a trained system administrator or a consultant to do this work. It is not something we can cover in depth in the documentation.
|
||||||
|
|
||||||
That said, here are some pieces of your infrastructure you should be sure to test:
|
That said, here are some pieces of your infrastructure you should be sure to test:
|
||||||
|
|
||||||
|
|||||||
+25
-4
@@ -34,7 +34,7 @@ Dynamic configuration is stored in the DCS (Distributed Configuration Store) and
|
|||||||
- **restore\_command**: command to restore WAL records from the remote master to standby leader, can be different from the list defined in :ref:`postgresql_settings`
|
- **restore\_command**: command to restore WAL records from the remote master to standby leader, can be different from the list defined in :ref:`postgresql_settings`
|
||||||
- **archive\_cleanup\_command**: cleanup command for standby leader
|
- **archive\_cleanup\_command**: cleanup command for standby leader
|
||||||
- **recovery\_min\_apply\_delay**: how long to wait before actually apply WAL records on a standby leader
|
- **recovery\_min\_apply\_delay**: how long to wait before actually apply WAL records on a standby leader
|
||||||
- **slots**: define permanent replication slots. These slots will be preserved during switchover/failover. Patroni will try to create slots before opening connections to the cluster.
|
- **slots**: define permanent replication slots. These slots will be preserved during switchover/failover. The logical slots are copied from the primary to a standby with restart, and after that their position advanced every **loop_wait** seconds (if necessary). Copying logical slot files performed via ``libpq`` connection and using either rewind or superuser credentials (see **postgresql.authentication** section). There is always a chance that the logical slot position on the replica is a bit behind the former primary, therefore application should be prepared that some messages could be received the second time after the failover. The easiest way of doing so - tracking ``confirmed_flush_lsn``. Enabling permanent logical replication slots requires **postgresql.use_slots** to be set and will also automatically enable the ``hot_standby_feedback``. Since the failover of logical replication slots is unsafe on PostgreSQL 9.6 and older and PostgreSQL version 10 is missing some important functions, the feature only works with PostgreSQL 11+.
|
||||||
- **my_slot_name**: the name of replication slot. If the permanent slot name matches with the name of the current primary it will not be created. Everything else is the responsibility of the operator to make sure that there are no clashes in names between replication slots automatically created by Patroni for members and permanent replication slots.
|
- **my_slot_name**: the name of replication slot. If the permanent slot name matches with the name of the current primary it will not be created. Everything else is the responsibility of the operator to make sure that there are no clashes in names between replication slots automatically created by Patroni for members and permanent replication slots.
|
||||||
- **type**: slot type. Could be ``physical`` or ``logical``. If the slot is logical, you have to additionally define ``database`` and ``plugin``.
|
- **type**: slot type. Could be ``physical`` or ``logical``. If the slot is logical, you have to additionally define ``database`` and ``plugin``.
|
||||||
- **database**: the database name where logical slots should be created.
|
- **database**: the database name where logical slots should be created.
|
||||||
@@ -156,7 +156,8 @@ Most of the parameters are optional, but you have to specify one of the **host**
|
|||||||
- **use\_proxies**: If this parameter is set to true, Patroni will consider **hosts** as a list of proxies and will not perform a topology discovery of etcd cluster.
|
- **use\_proxies**: If this parameter is set to true, Patroni will consider **hosts** as a list of proxies and will not perform a topology discovery of etcd cluster.
|
||||||
- **url**: url for the etcd.
|
- **url**: url for the etcd.
|
||||||
- **proxy**: proxy url for the etcd. If you are connecting to the etcd using proxy, use this parameter instead of **url**.
|
- **proxy**: proxy url for the etcd. If you are connecting to the etcd using proxy, use this parameter instead of **url**.
|
||||||
- **srv**: Domain to search the SRV record(s) for cluster autodiscovery.
|
- **srv**: Domain to search the SRV record(s) for cluster autodiscovery. Patroni will try to query these SRV service names for specified domain (in that order until first success): ``_etcd-client-ssl``, ``_etcd-client``, ``_etcd-ssl``, ``_etcd``, ``_etcd-server-ssl``, ``_etcd-server``. If SRV records for ``_etcd-server-ssl`` or ``_etcd-server`` are retrieved then ETCD peer protocol is used do query ETCD for available members. Otherwise hosts from SRV records will be used.
|
||||||
|
- **srv\_suffix**: Configures a suffix to the SRV name that is queried during discovery. Use this flag to differentiate between multiple etcd clusters under the same domain. Works only with conjunction with **srv**. For example, if ``srv_suffix: foo`` and ``srv: example.org`` are set, the following DNS SRV query is made:``_etcd-client-ssl-foo._tcp.example.com`` (and so on for every possible ETCD SRV service name).
|
||||||
- **protocol**: (optional) http or https, if not specified http is used. If the **url** or **proxy** is specified - will take protocol from them.
|
- **protocol**: (optional) http or https, if not specified http is used. If the **url** or **proxy** is specified - will take protocol from them.
|
||||||
- **username**: (optional) username for etcd authentication.
|
- **username**: (optional) username for etcd authentication.
|
||||||
- **password**: (optional) password for etcd authentication.
|
- **password**: (optional) password for etcd authentication.
|
||||||
@@ -181,6 +182,7 @@ ZooKeeper
|
|||||||
- **key**: (optional) File with the client key.
|
- **key**: (optional) File with the client key.
|
||||||
- **key_password**: (optional) The client key password.
|
- **key_password**: (optional) The client key password.
|
||||||
- **verify**: (optional) Whether to verify certificate or not. Defaults to ``true``.
|
- **verify**: (optional) Whether to verify certificate or not. Defaults to ``true``.
|
||||||
|
- **set_acls**: (optional) If set, configure Kazoo to apply a default ACL to each ZNode that it creates. ACLs will assume 'x509' schema and should be specified as a dictionary with the principal as the key and one or more permissions as a list in the value. Permissions may be one of ``CREATE``, ``READ``, ``WRITE``, ``DELETE`` or ``ADMIN``. For example, ``set_acls: {CN=principal1: [CREATE, READ], CN=principal2: [ALL]}``.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
It is required to install ``kazoo>=2.6.0`` to support SSL.
|
It is required to install ``kazoo>=2.6.0`` to support SSL.
|
||||||
@@ -204,6 +206,7 @@ Kubernetes
|
|||||||
- **use\_endpoints**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
- **use\_endpoints**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||||
- **pod\_ip**: (optional) IP address of the pod Patroni is running in. This value is required when `use_endpoints` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
- **pod\_ip**: (optional) IP address of the pod Patroni is running in. This value is required when `use_endpoints` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||||
- **ports**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``kubernetes.ports: [{"name": "postgresql", "port": 5432}]`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `kubernetes.use_endpoints` is set.
|
- **ports**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``kubernetes.ports: [{"name": "postgresql", "port": 5432}]`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `kubernetes.use_endpoints` is set.
|
||||||
|
- **cacert**: (optional) Specifies the file with the CA_BUNDLE file with certificates of trusted CAs to use while verifying Kubernetes API SSL certs. If not provided, patroni will use the value provided by the ServiceAccount secret.
|
||||||
|
|
||||||
|
|
||||||
.. _raft_settings:
|
.. _raft_settings:
|
||||||
@@ -228,7 +231,7 @@ Raft
|
|||||||
|
|
||||||
- Q: Where to get the ``syncobj_admin`` utility?
|
- Q: Where to get the ``syncobj_admin`` utility?
|
||||||
|
|
||||||
A: It is installed together with ``pysyncobj`` module (python RAFT implementation), which is Patroni dependancy.
|
A: It is installed together with ``pysyncobj`` module (python RAFT implementation), which is Patroni dependency.
|
||||||
|
|
||||||
- Q: it is possible to run Patroni node without adding in to the consensus?
|
- Q: it is possible to run Patroni node without adding in to the consensus?
|
||||||
|
|
||||||
@@ -254,6 +257,7 @@ PostgreSQL
|
|||||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
- **replication**:
|
- **replication**:
|
||||||
@@ -265,6 +269,7 @@ PostgreSQL
|
|||||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
- **rewind**:
|
- **rewind**:
|
||||||
@@ -276,6 +281,7 @@ PostgreSQL
|
|||||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
|
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||||
- **callbacks**: callback scripts to run on certain actions. Patroni will pass the action, role and cluster name. (See scripts/aws.py as an example of how to write them.)
|
- **callbacks**: callback scripts to run on certain actions. Patroni will pass the action, role and cluster name. (See scripts/aws.py as an example of how to write them.)
|
||||||
@@ -293,6 +299,7 @@ PostgreSQL
|
|||||||
- **bin\_dir**: Path to PostgreSQL binaries (pg_ctl, pg_rewind, pg_basebackup, postgres). The default value is an empty string meaning that PATH environment variable will be used to find the executables.
|
- **bin\_dir**: Path to PostgreSQL binaries (pg_ctl, pg_rewind, pg_basebackup, postgres). The default value is an empty string meaning that PATH environment variable will be used to find the executables.
|
||||||
- **listen**: IP address + port that Postgres listens to; must be accessible from other nodes in the cluster, if you're using streaming replication. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node.
|
- **listen**: IP address + port that Postgres listens to; must be accessible from other nodes in the cluster, if you're using streaming replication. Multiple comma-separated addresses are permitted, as long as the port component is appended after to the last one with a colon, i.e. ``listen: 127.0.0.1,127.0.0.2:5432``. Patroni will use the first address from this list to establish local connections to the PostgreSQL node.
|
||||||
- **use\_unix\_socket**: specifies that Patroni should prefer to use unix sockets to connect to the cluster. Default value is ``false``. If ``unix_socket_directories`` is defined, Patroni will use the first suitable value from it to connect to the cluster and fallback to tcp if nothing is suitable. If ``unix_socket_directories`` is not specified in ``postgresql.parameters``, Patroni will assume that the default value should be used and omit ``host`` from the connection parameters.
|
- **use\_unix\_socket**: specifies that Patroni should prefer to use unix sockets to connect to the cluster. Default value is ``false``. If ``unix_socket_directories`` is defined, Patroni will use the first suitable value from it to connect to the cluster and fallback to tcp if nothing is suitable. If ``unix_socket_directories`` is not specified in ``postgresql.parameters``, Patroni will assume that the default value should be used and omit ``host`` from the connection parameters.
|
||||||
|
- **use\_unix\_socket\_repl**: specifies that Patroni should prefer to use unix sockets for replication user cluster connection. Default value is ``false``. If ``unix_socket_directories`` is defined, Patroni will use the first suitable value from it to connect to the cluster and fallback to tcp if nothing is suitable. If ``unix_socket_directories`` is not specified in ``postgresql.parameters``, Patroni will assume that the default value should be used and omit ``host`` from the connection parameters.
|
||||||
- **pgpass**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup, the post_init script and under some other circumstances. The location must be writable by Patroni.
|
- **pgpass**: path to the `.pgpass <https://www.postgresql.org/docs/current/static/libpq-pgpass.html>`__ password file. Patroni creates this file before executing pg\_basebackup, the post_init script and under some other circumstances. The location must be writable by Patroni.
|
||||||
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower.
|
- **recovery\_conf**: additional configuration settings written to recovery.conf when configuring follower.
|
||||||
- **custom\_conf** : path to an optional custom ``postgresql.conf`` file, that will be used in place of ``postgresql.base.conf``. The file must exist on all cluster nodes, be readable by PostgreSQL and will be included from its location on the real ``postgresql.conf``. Note that Patroni will not monitor this file for changes, nor backup it. However, its settings can still be overridden by Patroni's own configuration facilities - see :ref:`dynamic configuration <dynamic_configuration>` for details.
|
- **custom\_conf** : path to an optional custom ``postgresql.conf`` file, that will be used in place of ``postgresql.base.conf``. The file must exist on all cluster nodes, be readable by PostgreSQL and will be included from its location on the real ``postgresql.conf``. Note that Patroni will not monitor this file for changes, nor backup it. However, its settings can still be overridden by Patroni's own configuration facilities - see :ref:`dynamic configuration <dynamic_configuration>` for details.
|
||||||
@@ -322,10 +329,12 @@ REST API
|
|||||||
- **password**: Basic-auth password to protect unsafe REST API endpoints.
|
- **password**: Basic-auth password to protect unsafe REST API endpoints.
|
||||||
- **certfile**: (optional): Specifies the file with the certificate in the PEM format. If the certfile is not specified or is left empty, the API server will work without SSL.
|
- **certfile**: (optional): Specifies the file with the certificate in the PEM format. If the certfile is not specified or is left empty, the API server will work without SSL.
|
||||||
- **keyfile**: (optional): Specifies the file with the secret key in the PEM format.
|
- **keyfile**: (optional): Specifies the file with the secret key in the PEM format.
|
||||||
- **keyfile_password**: (optional): Specifies a password for decrypting the keyfile.
|
- **keyfile\_password**: (optional): Specifies a password for decrypting the keyfile.
|
||||||
- **cafile**: (optional): Specifies the file with the CA_BUNDLE with certificates of trusted CAs to use while verifying client certs.
|
- **cafile**: (optional): Specifies the file with the CA_BUNDLE with certificates of trusted CAs to use while verifying client certs.
|
||||||
- **ciphers**: (optional): Specifies the permitted cipher suites (e.g. "ECDHE-RSA-AES256-GCM-SHA384:DHE-RSA-AES256-GCM-SHA384:ECDHE-RSA-AES128-GCM-SHA256:DHE-RSA-AES128-GCM-SHA256:!SSLv1:!SSLv2:!SSLv3:!TLSv1:!TLSv1.1")
|
- **ciphers**: (optional): Specifies the permitted cipher suites (e.g. "ECDHE-RSA-AES256-GCM-SHA384:DHE-RSA-AES256-GCM-SHA384:ECDHE-RSA-AES128-GCM-SHA256:DHE-RSA-AES128-GCM-SHA256:!SSLv1:!SSLv2:!SSLv3:!TLSv1:!TLSv1.1")
|
||||||
- **verify\_client**: (optional): ``none`` (default), ``optional`` or ``required``. When ``none`` REST API will not check client certificates. When ``required`` client certificates are required for all REST API calls. When ``optional`` client certificates are required for all unsafe REST API endpoints. When ``required`` is used, then client authentication succeeds, if the certificate signature verification succeeds. For ``optional`` the client cert will only be checked for ``PUT``, ``POST``, ``PATCH``, and ``DELETE`` requests.
|
- **verify\_client**: (optional): ``none`` (default), ``optional`` or ``required``. When ``none`` REST API will not check client certificates. When ``required`` client certificates are required for all REST API calls. When ``optional`` client certificates are required for all unsafe REST API endpoints. When ``required`` is used, then client authentication succeeds, if the certificate signature verification succeeds. For ``optional`` the client cert will only be checked for ``PUT``, ``POST``, ``PATCH``, and ``DELETE`` requests.
|
||||||
|
- **allowlist**: (optional): Specifies the set of hosts that are allowed to call unsafe REST API endpoints. The single element could be a host name, an IP address or a network address using CIDR notation. By default ``allow all`` is used. In case if ``allowlist`` or ``allowlist_include_members`` are set, anything that is not included is rejected.
|
||||||
|
- **allowlist\_include\_members**: (optional): If set to ``true`` it allows accessing unsafe REST API endpoints from other cluster members registered in DCS (IP address or hostname is taken from the members ``api_url``). Be careful, it might happen that OS will use a different IP for outgoing connections.
|
||||||
- **http\_extra\_headers**: (optional): HTTP headers let the REST API server pass additional information with an HTTP response.
|
- **http\_extra\_headers**: (optional): HTTP headers let the REST API server pass additional information with an HTTP response.
|
||||||
- **https\_extra\_headers**: (optional): HTTPS headers let the REST API server pass additional information with an HTTP response when TLS is enabled. This will also pass additional information set in ``http_extra_headers``.
|
- **https\_extra\_headers**: (optional): HTTPS headers let the REST API server pass additional information with an HTTP response when TLS is enabled. This will also pass additional information set in ``http_extra_headers``.
|
||||||
|
|
||||||
@@ -358,6 +367,7 @@ CTL
|
|||||||
- **cacert**: Specifies the file with the CA_BUNDLE file or directory with certificates of trusted CAs to use while verifying REST API SSL certs. If not provided patronictl will use the value provided for REST API "cafile" parameter.
|
- **cacert**: Specifies the file with the CA_BUNDLE file or directory with certificates of trusted CAs to use while verifying REST API SSL certs. If not provided patronictl will use the value provided for REST API "cafile" parameter.
|
||||||
- **certfile**: Specifies the file with the client certificate in the PEM format. If not provided patronictl will use the value provided for REST API "certfile" parameter.
|
- **certfile**: Specifies the file with the client certificate in the PEM format. If not provided patronictl will use the value provided for REST API "certfile" parameter.
|
||||||
- **keyfile**: Specifies the file with the client secret key in the PEM format. If not provided patronictl will use the value provided for REST API "keyfile" parameter.
|
- **keyfile**: Specifies the file with the client secret key in the PEM format. If not provided patronictl will use the value provided for REST API "keyfile" parameter.
|
||||||
|
- **keyfile\_password**: Specifies a password for decrypting the keyfile. If not provided patronictl will use the value provided for REST API "keyfile\_password" parameter.
|
||||||
|
|
||||||
Watchdog
|
Watchdog
|
||||||
--------
|
--------
|
||||||
@@ -365,6 +375,8 @@ Watchdog
|
|||||||
- **device**: Path to watchdog device. Defaults to ``/dev/watchdog``.
|
- **device**: Path to watchdog device. Defaults to ``/dev/watchdog``.
|
||||||
- **safety_margin**: Number of seconds of safety margin between watchdog triggering and leader key expiration.
|
- **safety_margin**: Number of seconds of safety margin between watchdog triggering and leader key expiration.
|
||||||
|
|
||||||
|
.. _tags_settings:
|
||||||
|
|
||||||
Tags
|
Tags
|
||||||
----
|
----
|
||||||
- **nofailover**: ``true`` or ``false``, controls whether this node is allowed to participate in the leader race and become a leader. Defaults to ``false``
|
- **nofailover**: ``true`` or ``false``, controls whether this node is allowed to participate in the leader race and become a leader. Defaults to ``false``
|
||||||
@@ -372,3 +384,12 @@ Tags
|
|||||||
- **noloadbalance**: ``true`` or ``false``. If set to ``true`` the node will return HTTP Status Code 503 for the ``GET /replica`` REST API health-check and therefore will be excluded from the load-balancing. Defaults to ``false``.
|
- **noloadbalance**: ``true`` or ``false``. If set to ``true`` the node will return HTTP Status Code 503 for the ``GET /replica`` REST API health-check and therefore will be excluded from the load-balancing. Defaults to ``false``.
|
||||||
- **replicatefrom**: The IP address/hostname of another replica. Used to support cascading replication.
|
- **replicatefrom**: The IP address/hostname of another replica. Used to support cascading replication.
|
||||||
- **nosync**: ``true`` or ``false``. If set to ``true`` the node will never be selected as a synchronous replica.
|
- **nosync**: ``true`` or ``false``. If set to ``true`` the node will never be selected as a synchronous replica.
|
||||||
|
|
||||||
|
In addition to these predefined tags, you can also add your own ones:
|
||||||
|
|
||||||
|
- **key1**: ``true``
|
||||||
|
- **key2**: ``false``
|
||||||
|
- **key3**: ``1.4``
|
||||||
|
- **key4**: ``"RandomString"``
|
||||||
|
|
||||||
|
Tags are visible in the :ref:`REST API <rest_api>` and ``patronictl list`` You can also check for an instance health using these tags. If the tag isn't defined for an instance, or if the respective value doesn't match the querying value, it will return HTTP Status Code 503.
|
||||||
|
|||||||
+4
-1
@@ -194,4 +194,7 @@ intersphinx_mapping = {'https://docs.python.org/': None}
|
|||||||
# A possibility to have an own stylesheet, to add new rules or override existing ones
|
# A possibility to have an own stylesheet, to add new rules or override existing ones
|
||||||
# For the latter case, the CSS specificity of the rules should be higher than the default ones
|
# For the latter case, the CSS specificity of the rules should be higher than the default ones
|
||||||
def setup(app):
|
def setup(app):
|
||||||
app.add_stylesheet("custom.css")
|
if hasattr(app, 'add_css_file'):
|
||||||
|
app.add_css_file('custom.css')
|
||||||
|
else:
|
||||||
|
app.add_stylesheet('custom.css')
|
||||||
|
|||||||
+1
-1
@@ -10,7 +10,7 @@ Patroni is a template for you to create your own customized, high-availability s
|
|||||||
|
|
||||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
|
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
|
||||||
|
|
||||||
Currently supported PostgreSQL versions: 9.3 to 13.
|
Currently supported PostgreSQL versions: 9.3 to 14.
|
||||||
|
|
||||||
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the :ref:`Kubernetes <kubernetes>` chapter of the Patroni documentation.
|
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the :ref:`Kubernetes <kubernetes>` chapter of the Patroni documentation.
|
||||||
|
|
||||||
|
|||||||
+309
-11
@@ -3,6 +3,304 @@
|
|||||||
Release notes
|
Release notes
|
||||||
=============
|
=============
|
||||||
|
|
||||||
|
Version 2.1.3
|
||||||
|
-------------
|
||||||
|
|
||||||
|
**New features**
|
||||||
|
|
||||||
|
- Added support for encrypted TLS keys for ``patronictl`` (Alexander Kukushkin)
|
||||||
|
|
||||||
|
It could be configured via ``ctl.keyfile_password`` or the ``PATRONI_CTL_KEYFILE_PASSWORD`` environment variable.
|
||||||
|
|
||||||
|
- Added more metrics to the /metrics endpoint (Alexandre Pereira)
|
||||||
|
|
||||||
|
Specifically, ``patroni_pending_restart`` and ``patroni_is_paused``.
|
||||||
|
|
||||||
|
- Make it possible to specify multiple hosts in the standby cluster configuration (Michael Banck)
|
||||||
|
|
||||||
|
If the standby cluster is replicating from the Patroni cluster it might be nice to rely on client-side failover which is available in ``libpq`` since PostgreSQL v10. That is, the ``primary_conninfo`` on the standby leader and ``pg_rewind`` setting ``target_session_attrs=read-write`` in the connection string. The ``pgpass`` file will be generated with multiple lines (one line per host), and instead of calling ``CHECKPOINT`` on the primary cluster nodes the standby cluster will wait for ``pg_control`` to be updated.
|
||||||
|
|
||||||
|
**Stability improvements**
|
||||||
|
|
||||||
|
- Compatibility with legacy ``psycopg2`` (Alexander)
|
||||||
|
|
||||||
|
For example, the ``psycopg2`` installed from Ubuntu 18.04 packages doesn't have the ``UndefinedFile`` exception yet.
|
||||||
|
|
||||||
|
- Restart ``etcd3`` watcher if all Etcd nodes don't respond (Alexander)
|
||||||
|
|
||||||
|
If the watcher is alive the ``get_cluster()`` method continues returning stale information even if all Etcd nodes are failing.
|
||||||
|
|
||||||
|
- Don't remove the leader lock in the standby cluster while paused (Alexander)
|
||||||
|
|
||||||
|
Previously the lock was maintained only by the node that was running as a primary and not a standby leader.
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Fixed bug in the standby-leader bootstrap (Alexander)
|
||||||
|
|
||||||
|
Patroni was considering bootstrap as failed if Postgres didn't start accepting connections after 60 seconds. The bug was introduced in the 2.1.2 release.
|
||||||
|
|
||||||
|
- Fixed bug with failover to a cascading standby (Alexander)
|
||||||
|
|
||||||
|
When figuring out which slots should be created on cascading standby we forgot to take into account that the leader might be absent.
|
||||||
|
|
||||||
|
- Fixed small issues in Postgres config validator (Alexander)
|
||||||
|
|
||||||
|
Integer parameters introduced in PostgreSQL v14 were failing to validate because min and max values were quoted in the validator.py
|
||||||
|
|
||||||
|
- Use replication credentials when checking leader status (Alexander)
|
||||||
|
|
||||||
|
It could be that the ``remove_data_directory_on_diverged_timelines`` is set, but there is no ``rewind_credentials`` defined and superuser access between nodes is not allowed.
|
||||||
|
|
||||||
|
- Fixed "port in use" error on REST API certificate replacement (Ants Aasma)
|
||||||
|
|
||||||
|
When switching certificates there was a race condition with a concurrent API request. If there is one active during the replacement period then the replacement will error out with a port in use error and Patroni gets stuck in a state without an active API server.
|
||||||
|
|
||||||
|
- Fixed a bug in cluster bootstrap if passwords contain ``%`` characters (Bastien Wirtz)
|
||||||
|
|
||||||
|
The bootstrap method executes the ``DO`` block, with all parameters properly quoted, but the ``cursor.execute()`` method didn't like an empty list with parameters passed.
|
||||||
|
|
||||||
|
- Fixed the "AttributeError: no attribute 'leader'" exception (Hrvoje Milković)
|
||||||
|
|
||||||
|
It could happen if the synchronous mode is enabled and the DCS content was wiped out.
|
||||||
|
|
||||||
|
- Fix bug in divergence timeline check (Alexander)
|
||||||
|
|
||||||
|
Patroni was falsely assuming that timelines have diverged. For pg_rewind it didn't create any problem, but if pg_rewind is not allowed and the ``remove_data_directory_on_diverged_timelines`` is set, it resulted in reinitializing the former leader.
|
||||||
|
|
||||||
|
|
||||||
|
Version 2.1.2
|
||||||
|
-------------
|
||||||
|
|
||||||
|
**New features**
|
||||||
|
|
||||||
|
- Compatibility with ``psycopg>=3.0`` (Alexander Kukushkin)
|
||||||
|
|
||||||
|
By default ``psycopg2`` is preferred. `psycopg>=3.0` will be used only if ``psycopg2`` is not available or its version is too old.
|
||||||
|
|
||||||
|
- Add ``dcs_last_seen`` field to the REST API (Michael Banck)
|
||||||
|
|
||||||
|
This field notes the last time (as unix epoch) a cluster member has successfully communicated with the DCS. This is useful to identify and/or analyze network partitions.
|
||||||
|
|
||||||
|
- Release the leader lock when ``pg_controldata`` reports "shut down" (Alexander)
|
||||||
|
|
||||||
|
To solve the problem of slow switchover/shutdown in case ``archive_command`` is slow/failing, Patroni will remove the leader key immediately after ``pg_controldata`` started reporting PGDATA as ``shut down`` cleanly and it verified that there is at least one replica that received all changes. If there are no replicas that fulfill this condition the leader key is not removed and the old behavior is retained, i.e. Patroni will keep updating the lock.
|
||||||
|
|
||||||
|
- Add ``sslcrldir`` connection parameter support (Kostiantyn Nemchenko)
|
||||||
|
|
||||||
|
The new connection parameter was introduced in the PostgreSQL v14.
|
||||||
|
|
||||||
|
- Allow setting ACLs for ZNodes in Zookeeper (Alwyn Davis)
|
||||||
|
|
||||||
|
Introduce a new configuration option ``zookeeper.set_acls`` so that Kazoo will apply a default ACL for each ZNode that it creates.
|
||||||
|
|
||||||
|
|
||||||
|
**Stability improvements**
|
||||||
|
|
||||||
|
- Delay the next attempt of recovery till next HA loop (Alexander)
|
||||||
|
|
||||||
|
If Postgres crashed due to out of disk space (for example) and fails to start because of that Patroni is too eagerly trying to recover it flooding logs.
|
||||||
|
|
||||||
|
- Add log before demoting, which can take some time (Michael)
|
||||||
|
|
||||||
|
It can take some time for the demote to finish and it might not be obvious from looking at the logs what exactly is going on.
|
||||||
|
|
||||||
|
- Improve "I am" status messages (Michael)
|
||||||
|
|
||||||
|
``no action. I am a secondary ({0})`` vs ``no action. I am ({0}), a secondary``
|
||||||
|
|
||||||
|
- Cast to int ``wal_keep_segments`` when converting to ``wal_keep_size`` (Jorge Solórzano)
|
||||||
|
|
||||||
|
It is possible to specify ``wal_keep_segments`` as a string in the global :ref:`dynamic configuration <dynamic_configuration>` and due to Python being a dynamically typed language the string was simply multiplied. Example: ``wal_keep_segments: "100"`` was converted to ``100100100100100100100100100100100100100100100100MB``.
|
||||||
|
|
||||||
|
- Allow switchover only to sync nodes when synchronous replication is enabled (Alexander)
|
||||||
|
|
||||||
|
In addition to that do the leader race only against known synchronous nodes.
|
||||||
|
|
||||||
|
- Use cached role as a fallback when Postgres is slow (Alexander)
|
||||||
|
|
||||||
|
In some extreme cases Postgres could be so slow that the normal monitoring query does not finish in a few seconds. The ``statement_timeout`` exception not being properly handled could lead to the situation where Postgres was not demoted on time when the leader key expired or the update failed. In case of such exception Patroni will use the cached ``role`` to determine whether Postgres is running as a primary.
|
||||||
|
|
||||||
|
- Avoid unnecessary updates of the member ZNode (Alexander)
|
||||||
|
|
||||||
|
If no values have changed in the members data, the update should not happen.
|
||||||
|
|
||||||
|
- Optimize checkpoint after promote (Alexander)
|
||||||
|
|
||||||
|
Avoid doing ``CHECKPOINT`` if the latest timeline is already stored in ``pg_control``. It helps to avoid unnecessary ``CHECKPOINT`` right after initializing the new cluster with ``initdb``.
|
||||||
|
|
||||||
|
- Prefer members without ``nofailover`` when picking sync nodes (Alexander)
|
||||||
|
|
||||||
|
Previously sync nodes were selected only based on the replication lag, hence the node with ``nofailover`` tag had the same chances to become synchronous as any other node. That behavior was confusing and dangerous at the same time because in case of a failed primary the failover could not happen automatically.
|
||||||
|
|
||||||
|
- Remove duplicate hosts from the etcd machine cache (Michael)
|
||||||
|
|
||||||
|
Advertised client URLs in the etcd cluster could be misconfigured. Removing duplicates in Patroni in this case is a low-hanging fruit.
|
||||||
|
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Skip temporary replication slots while doing slot management (Alexander)
|
||||||
|
|
||||||
|
Starting from v10 ``pg_basebackup`` creates a temporary replication slot for WAL streaming and Patroni was trying to drop it because the slot name looks unknown. In order to fix it, we skip all temporary slots when querying ``pg_stat_replication_slots`` view.
|
||||||
|
|
||||||
|
- Ensure ``pg_replication_slot_advance()`` doesn't timeout (Alexander)
|
||||||
|
|
||||||
|
Patroni was using the default ``statement_timeout`` in this case and once the call failed there are very high chances that it will never recover, resulting in increased size of ``pg_wal`` and ``pg_catalog`` bloat.
|
||||||
|
|
||||||
|
- The ``/status`` wasn't updated on demote (Alexander)
|
||||||
|
|
||||||
|
After demoting PostgreSQL the old leader updates the last LSN in DCS. Starting from ``2.1.0`` the new ``/status`` key was introduced, but the optime was still written to the ``/optime/leader``.
|
||||||
|
|
||||||
|
- Handle DCS exceptions when demoting (Alexander)
|
||||||
|
|
||||||
|
While demoting the master due to failure to update the leader lock it could happen that DCS goes completely down and the ``get_cluster()`` call raises an exception. Not being handled properly it results in Postgres remaining stopped until DCS recovers.
|
||||||
|
|
||||||
|
- The ``use_unix_socket_repl`` didn't work is some cases (Alexander)
|
||||||
|
|
||||||
|
Specifically, if ``postgresql.unix_socket_directories`` is not set. In this case Patroni is supposed to use the default value from ``libpq``.
|
||||||
|
|
||||||
|
- Fix a few issues with Patroni REST API (Alexander)
|
||||||
|
|
||||||
|
The ``clusters_unlocked`` sometimes could be not defined, what resulted in exceptions in the ``GET /metrics`` endpoint. In addition to that the error handling method was assuming that the ``connect_address`` tuple always has two elements, while in fact there could be more in case of IPv6.
|
||||||
|
|
||||||
|
- Wait for newly promoted node to finish recovery before deciding to rewind (Alexander)
|
||||||
|
|
||||||
|
It could take some time before the actual promote happens and the new timeline is created. Without waiting replicas could come to the conclusion that rewind isn't required.
|
||||||
|
|
||||||
|
- Handle missing timelines in a history file when deciding to rewind (Alexander)
|
||||||
|
|
||||||
|
If the current replica timeline is missing in the history file on the primary the replica was falsely assuming that rewind isn't required.
|
||||||
|
|
||||||
|
|
||||||
|
Version 2.1.1
|
||||||
|
-------------
|
||||||
|
|
||||||
|
**New features**
|
||||||
|
|
||||||
|
- Support for ETCD SRV name suffix (David Pavlicek)
|
||||||
|
|
||||||
|
Etcd allows to differentiate between multiple Etcd clusters under the same domain and from now on Patroni also supports it.
|
||||||
|
|
||||||
|
- Enrich history with the new leader (huiyalin525)
|
||||||
|
|
||||||
|
It adds the new column to the ``patronictl history`` output.
|
||||||
|
|
||||||
|
- Make the CA bundle configurable for in-cluster Kubernetes config (Aron Parsons)
|
||||||
|
|
||||||
|
By default Patroni is using ``/var/run/secrets/kubernetes.io/serviceaccount/ca.crt`` and this new feature allows specifying the custom ``kubernetes.cacert``.
|
||||||
|
|
||||||
|
- Support dynamically registering/deregistering as a Consul service and changing tags (Tommy Li)
|
||||||
|
|
||||||
|
Previously it required Patroni restart.
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Avoid unnecessary reload of REST API (Alexander Kukushkin)
|
||||||
|
|
||||||
|
The previous release added a feature of reloading REST API certificates if changed on disk. Unfortunately, the reload was happening unconditionally right after the start.
|
||||||
|
|
||||||
|
- Don't resolve cluster members when ``etcd.use_proxies`` is set (Alexander)
|
||||||
|
|
||||||
|
When starting up Patroni checks the healthiness of Etcd cluster by querying the list of members. In addition to that, it also tried to resolve their hostnames, which is not necessary when working with Etcd via proxy and was causing unnecessary warnings.
|
||||||
|
|
||||||
|
- Skip rows with NULL values in the ``pg_stat_replication`` (Alexander)
|
||||||
|
|
||||||
|
It seems that the ``pg_stat_replication`` view could contain NULL values in the ``replay_lsn``, ``flush_lsn``, or ``write_lsn`` fields even when ``state = 'streaming'``.
|
||||||
|
|
||||||
|
|
||||||
|
Version 2.1.0
|
||||||
|
-------------
|
||||||
|
|
||||||
|
This version adds compatibility with PostgreSQL v14, makes logical replication slots to survive failover/switchover, implements support of allowlist for REST API, and also reducing the number of logs to one line per heart-beat.
|
||||||
|
|
||||||
|
**New features**
|
||||||
|
|
||||||
|
- Compatibility with PostgreSQL v14 (Alexander Kukushkin)
|
||||||
|
|
||||||
|
Unpause WAL replay if Patroni is not in a "pause" mode itself. It could be "paused" due to the change of certain parameters like for example ``max_connections`` on the primary.
|
||||||
|
|
||||||
|
- Failover logical slots (Alexander)
|
||||||
|
|
||||||
|
Make logical replication slots survive failover/switchover on PostgreSQL v11+. The replication slot if copied from the primary to the replica with restart and later the `pg_replication_slot_advance() <https://www.postgresql.org/docs/11/functions-admin.html#id-1.5.8.31.8.5.2.2.8.1.1>`__ function is used to move it forward. As a result, the slot will already exist before the failover and no events should be lost, but, there is a chance that some events could be delivered more than once.
|
||||||
|
|
||||||
|
- Implemented allowlist for Patroni REST API (Alexander)
|
||||||
|
|
||||||
|
If configured, only IP's that matching rules would be allowed to call unsafe endpoints. In addition to that, it is possible to automatically include IP's of members of the cluster to the list.
|
||||||
|
|
||||||
|
- Added support of replication connections via unix socket (Mohamad El-Rifai)
|
||||||
|
|
||||||
|
Previously Patroni was always using TCP for replication connection what could cause some issues with SSL verification. Using unix sockets allows exempt replication user from SSL verification.
|
||||||
|
|
||||||
|
- Health check on user-defined tags (Arman Jafari Tehrani)
|
||||||
|
|
||||||
|
Along with :ref:`predefined tags: <tags_settings>` it is possible to specify any number of custom tags that become visible in the ``patronictl list`` output and in the REST API. From now on it is possible to use custom tags in health checks.
|
||||||
|
|
||||||
|
- Added Prometheus ``/metrics`` endpoint (Mark Mercado, Michael Banck)
|
||||||
|
|
||||||
|
The endpoint exposing the same metrics as ``/patroni``.
|
||||||
|
|
||||||
|
- Reduced chattiness of Patroni logs (Alexander)
|
||||||
|
|
||||||
|
When everything goes normal, only one line will be written for every run of HA loop.
|
||||||
|
|
||||||
|
|
||||||
|
**Breaking changes**
|
||||||
|
|
||||||
|
- The old ``permanent logical replication slots`` feature will no longer work with PostgreSQL v10 and older (Alexander)
|
||||||
|
|
||||||
|
The strategy of creating the logical slots after performing a promotion can't guaranty that no logical events are lost and therefore disabled.
|
||||||
|
|
||||||
|
- The ``/leader`` endpoint always returns 200 if the node holds the lock (Alexander)
|
||||||
|
|
||||||
|
Promoting the standby cluster requires updating load-balancer health checks, which is not very convenient and easy to forget. To solve it, we change the behavior of the ``/leader`` health check endpoint. It will return 200 without taking into account whether the cluster is normal or the ``standby_cluster``.
|
||||||
|
|
||||||
|
|
||||||
|
**Improvements in Raft support**
|
||||||
|
|
||||||
|
- Reliable support of Raft traffic encryption (Alexander)
|
||||||
|
|
||||||
|
Due to the different issues in the ``PySyncObj`` the encryption support was very unstable
|
||||||
|
|
||||||
|
- Handle DNS issues in Raft implementation (Alexander)
|
||||||
|
|
||||||
|
If ``self_addr`` and/or ``partner_addrs`` are configured using the DNS name instead of IP's the ``PySyncObj`` was effectively doing resolve only once when the object is created. It was causing problems when the same node was coming back online with a different IP.
|
||||||
|
|
||||||
|
|
||||||
|
**Stability improvements**
|
||||||
|
|
||||||
|
- Compatibility with ``psycopg2-2.9+`` (Alexander)
|
||||||
|
|
||||||
|
In ``psycopg2`` the ``autocommit = True`` is ignored in the ``with connection`` block, which breaks replication protocol connections.
|
||||||
|
|
||||||
|
- Fix excessive HA loop runs with Zookeeper (Alexander)
|
||||||
|
|
||||||
|
Update of member ZNodes was causing a chain reaction and resulted in running the HA loops multiple times in a row.
|
||||||
|
|
||||||
|
- Reload if REST API certificate is changed on disk (Michael Todorovic)
|
||||||
|
|
||||||
|
If the REST API certificate file was updated in place Patroni didn't perform a reload.
|
||||||
|
|
||||||
|
- Don't create pgpass dir if kerberos auth is used (Kostiantyn Nemchenko)
|
||||||
|
|
||||||
|
Kerberos and password authentication are mutually exclusive.
|
||||||
|
|
||||||
|
- Fixed little issues with custom bootstrap (Alexander)
|
||||||
|
|
||||||
|
Start Postgres with ``hot_standby=off`` only when we do a PITR and restart it after PITR is done.
|
||||||
|
|
||||||
|
|
||||||
|
**Bugfixes**
|
||||||
|
|
||||||
|
- Compatibility with ``kazoo-2.7+`` (Alexander)
|
||||||
|
|
||||||
|
Since Patroni is handling retries on its own, it is relying on the old behavior of ``kazoo`` that requests to a Zookeeper cluster are immediately discarded when there are no connections available.
|
||||||
|
|
||||||
|
- Explicitly request the version of Etcd v3 cluster when it is known that we are connecting via proxy (Alexander)
|
||||||
|
|
||||||
|
Patroni is working with Etcd v3 cluster via gPRC-gateway and it depending on the cluster version different endpoints (``/v3``, ``/v3beta``, or ``/v3alpha``) must be used. The version was resolved only together with the cluster topology, but since the latter was never done when connecting via proxy.
|
||||||
|
|
||||||
|
|
||||||
Version 2.0.2
|
Version 2.0.2
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
@@ -682,7 +980,7 @@ Version 1.6.1
|
|||||||
|
|
||||||
- Some improvements in logging infrastructure (Alexander Kukushkin)
|
- Some improvements in logging infrastructure (Alexander Kukushkin)
|
||||||
|
|
||||||
Previously threre was a possibility to loose the last few log lines on shutdown because the logging thread was a ``daemon`` thread.
|
Previously there was a possibility to loose the last few log lines on shutdown because the logging thread was a ``daemon`` thread.
|
||||||
|
|
||||||
- Use ``spawn`` multiprocessing start method on python 3.4+ (Maciej Kowalczyk)
|
- Use ``spawn`` multiprocessing start method on python 3.4+ (Maciej Kowalczyk)
|
||||||
|
|
||||||
@@ -732,7 +1030,7 @@ Version 1.6.1
|
|||||||
|
|
||||||
If the method is executed from the REST API thread, it requires a separate cursor object to be created.
|
If the method is executed from the REST API thread, it requires a separate cursor object to be created.
|
||||||
|
|
||||||
- Fix the problem of not promoting the sync standby that had a name contaning upper case letters (Alexander Kukushkin)
|
- Fix the problem of not promoting the sync standby that had a name containing upper case letters (Alexander Kukushkin)
|
||||||
|
|
||||||
We converted the name to the lower case because Postgres was doing the same while comparing the ``application_name`` with the value in ``synchronous_standby_names``.
|
We converted the name to the lower case because Postgres was doing the same while comparing the ``application_name`` with the value in ``synchronous_standby_names``.
|
||||||
|
|
||||||
@@ -1008,7 +1306,7 @@ Compatibility and bugfix release.
|
|||||||
|
|
||||||
- Fix broken compatibility with postgres 9.3 (Alexander)
|
- Fix broken compatibility with postgres 9.3 (Alexander)
|
||||||
|
|
||||||
When opening a replication connection we should specify replication=1, beacuse 9.3 does not understand replication='database'
|
When opening a replication connection we should specify replication=1, because 9.3 does not understand replication='database'
|
||||||
|
|
||||||
- Make sure we refresh Consul session at least once per HA loop and improve handling of consul sessions exceptions (Alexander)
|
- Make sure we refresh Consul session at least once per HA loop and improve handling of consul sessions exceptions (Alexander)
|
||||||
|
|
||||||
@@ -1093,7 +1391,7 @@ This version enables Patroni HA cluster to operate in a standby mode, introduces
|
|||||||
|
|
||||||
- Immediately reserve the WAL position upon creation of the replication slot (Alexander Kukushkin)
|
- Immediately reserve the WAL position upon creation of the replication slot (Alexander Kukushkin)
|
||||||
|
|
||||||
Starting from 9.6, `pg_create_physical_replication_slot` function provides an additional boolean parameter `immediately_reserve`. When it is set to `false`, which is also the default, the slot doesn't reserve the WAL position until it receives the first client connection, potentially losing some segments required by the client in a time window between the slot creation and the intiial client connection.
|
Starting from 9.6, `pg_create_physical_replication_slot` function provides an additional boolean parameter `immediately_reserve`. When it is set to `false`, which is also the default, the slot doesn't reserve the WAL position until it receives the first client connection, potentially losing some segments required by the client in a time window between the slot creation and the initial client connection.
|
||||||
|
|
||||||
- Fix bug in strict synchronous replication (Alexander Kukushkin)
|
- Fix bug in strict synchronous replication (Alexander Kukushkin)
|
||||||
|
|
||||||
@@ -1333,7 +1631,7 @@ This version adds support for using Kubernetes as a DCS, allowing to run Patroni
|
|||||||
|
|
||||||
**Upgrade notice**
|
**Upgrade notice**
|
||||||
|
|
||||||
Installing Patroni via pip will no longer bring in dependencies for (such as libraries for Etcd, Zookeper, Consul or Kubernetes, or support for AWS). In order to enable them one need to list them in pip install command explicitely, for instance `pip install patroni[etcd,kubernetes]`.
|
Installing Patroni via pip will no longer bring in dependencies for (such as libraries for Etcd, Zookeper, Consul or Kubernetes, or support for AWS). In order to enable them one need to list them in pip install command explicitly, for instance `pip install patroni[etcd,kubernetes]`.
|
||||||
|
|
||||||
**Kubernetes support**
|
**Kubernetes support**
|
||||||
|
|
||||||
@@ -1352,7 +1650,7 @@ In addition to using Endpoints, Patroni supports ConfigMaps. You can find more i
|
|||||||
|
|
||||||
- Remove leader key on shutdown only when we have the lock (Ants)
|
- Remove leader key on shutdown only when we have the lock (Ants)
|
||||||
|
|
||||||
Unconditional removal was generating unnecessary and missleading exceptions.
|
Unconditional removal was generating unnecessary and misleading exceptions.
|
||||||
|
|
||||||
**Improvements in patronictl**
|
**Improvements in patronictl**
|
||||||
|
|
||||||
@@ -1383,7 +1681,7 @@ In addition to using Endpoints, Patroni supports ConfigMaps. You can find more i
|
|||||||
|
|
||||||
- Alter the behavior of ``patronictl failover`` (Alexander)
|
- Alter the behavior of ``patronictl failover`` (Alexander)
|
||||||
|
|
||||||
It will work even if there is no leader, but in that case you will have to explicitely specify a node which should become the new leader.
|
It will work even if there is no leader, but in that case you will have to explicitly specify a node which should become the new leader.
|
||||||
|
|
||||||
**Expose information about timeline and history**
|
**Expose information about timeline and history**
|
||||||
|
|
||||||
@@ -1399,7 +1697,7 @@ In addition to using Endpoints, Patroni supports ConfigMaps. You can find more i
|
|||||||
|
|
||||||
- Add new /sync and /async endpoints (Alexander, Oleksii Kliukin)
|
- Add new /sync and /async endpoints (Alexander, Oleksii Kliukin)
|
||||||
|
|
||||||
Those endpoints (also accessible as /synchronous and /asynchronous) return 200 only for synchronous and asynchornous replicas correspondingly (exclusing those marked as `noloadbalance`).
|
Those endpoints (also accessible as /synchronous and /asynchronous) return 200 only for synchronous and asynchronous replicas correspondingly (exclusing those marked as `noloadbalance`).
|
||||||
|
|
||||||
**Allow multiple hosts for Etcd**
|
**Allow multiple hosts for Etcd**
|
||||||
|
|
||||||
@@ -1487,7 +1785,7 @@ Version 1.3.4
|
|||||||
|
|
||||||
- Pass the consul token as a header (Andrew Colin Kissa)
|
- Pass the consul token as a header (Andrew Colin Kissa)
|
||||||
|
|
||||||
Headers are now the prefered way to pass the token to the consul `API <https://www.consul.io/api/index.html#authentication>`__.
|
Headers are now the preferred way to pass the token to the consul `API <https://www.consul.io/api/index.html#authentication>`__.
|
||||||
|
|
||||||
|
|
||||||
- Advanced configuration for Consul (Alexander Kukushkin)
|
- Advanced configuration for Consul (Alexander Kukushkin)
|
||||||
@@ -1805,7 +2103,7 @@ In addition, patronictl supports new ``pause`` and ``resume`` commands to toggle
|
|||||||
Originally, ping_timeout and connect_timeout values were calculated from the negotiated session timeout. Patroni loop_wait was not taken into account. As
|
Originally, ping_timeout and connect_timeout values were calculated from the negotiated session timeout. Patroni loop_wait was not taken into account. As
|
||||||
a result, a single retry could take more time than the session timeout, forcing Patroni to release the lock and demote.
|
a result, a single retry could take more time than the session timeout, forcing Patroni to release the lock and demote.
|
||||||
|
|
||||||
This change set ping and connect timeout to half of the value of loop_wait, speeding up detection of connection issues and leaving enough time to retry the connection attempt before loosing the lock.
|
This change set ping and connect timeout to half of the value of loop_wait, speeding up detection of connection issues and leaving enough time to retry the connection attempt before losing the lock.
|
||||||
|
|
||||||
- Update Etcd topology only after original request succeed (Alexander)
|
- Update Etcd topology only after original request succeed (Alexander)
|
||||||
|
|
||||||
@@ -1883,7 +2181,7 @@ When upgrading from v0.90 or below, always upgrade all replicas before the maste
|
|||||||
|
|
||||||
See the :ref:`dynamic configuration <dynamic_configuration>` for the details on which parameters can be changed and the order of processing difference configuration sources.
|
See the :ref:`dynamic configuration <dynamic_configuration>` for the details on which parameters can be changed and the order of processing difference configuration sources.
|
||||||
|
|
||||||
The configuration file format *has changed* since the v0.90. Patroni is still compatible with the old configuration files, but in order to take advantage of the bootstrap parameters one needs to change it. Users are encourage to update them by referring to the :ref:`dynamic configuraton documentation page <dynamic_configuration>`.
|
The configuration file format *has changed* since the v0.90. Patroni is still compatible with the old configuration files, but in order to take advantage of the bootstrap parameters one needs to change it. Users are encourage to update them by referring to the :ref:`dynamic configuration documentation page <dynamic_configuration>`.
|
||||||
|
|
||||||
**More flexible configuration***
|
**More flexible configuration***
|
||||||
|
|
||||||
|
|||||||
+17
-5
@@ -9,14 +9,17 @@ Health check endpoints
|
|||||||
----------------------
|
----------------------
|
||||||
For all health check ``GET`` requests Patroni returns a JSON document with the status of the node, along with the HTTP status code. If you don't want or don't need the JSON document, you might consider using the ``OPTIONS`` method instead of ``GET``.
|
For all health check ``GET`` requests Patroni returns a JSON document with the status of the node, along with the HTTP status code. If you don't want or don't need the JSON document, you might consider using the ``OPTIONS`` method instead of ``GET``.
|
||||||
|
|
||||||
- The following requests to Patroni REST API will return HTTP status code **200** only when the Patroni node is running as the leader:
|
- The following requests to Patroni REST API will return HTTP status code **200** only when the Patroni node is running as the primary with leader lock:
|
||||||
|
|
||||||
- ``GET /``
|
- ``GET /``
|
||||||
- ``GET /master``
|
- ``GET /master``
|
||||||
- ``GET /leader``
|
|
||||||
- ``GET /primary``
|
- ``GET /primary``
|
||||||
- ``GET /read-write``
|
- ``GET /read-write``
|
||||||
|
|
||||||
|
- ``GET /standby-leader``: returns HTTP status code **200** only when the Patroni node is running as the leader in a :ref:`standby cluster <standby_cluster>`.
|
||||||
|
|
||||||
|
- ``GET /leader``: returns HTTP status code **200** when the Patroni node has the leader lock. The major difference from the two previous endpoints is that it doesn't take into account whether PostgreSQL is running as the ``primary`` or the ``standby_leader``.
|
||||||
|
|
||||||
- ``GET /replica``: replica health check endpoint. It returns HTTP status code **200** only when the Patroni node is in the state ``running``, the role is ``replica`` and ``noloadbalance`` tag is not set.
|
- ``GET /replica``: replica health check endpoint. It returns HTTP status code **200** only when the Patroni node is in the state ``running``, the role is ``replica`` and ``noloadbalance`` tag is not set.
|
||||||
|
|
||||||
- ``GET /replica?lag=<max-lag>``: replica check endpoint. In addition to checks from ``replica``, it also checks replication latency and returns status code **200** only when it is below specified value. The key cluster.last_leader_operation from DCS is used for Leader wal position and compute latency on replica for performance reasons. max-lag can be specified in bytes (integer) or in human readable values, for e.g. 16kB, 64MB, 1GB.
|
- ``GET /replica?lag=<max-lag>``: replica check endpoint. In addition to checks from ``replica``, it also checks replication latency and returns status code **200** only when it is below specified value. The key cluster.last_leader_operation from DCS is used for Leader wal position and compute latency on replica for performance reasons. max-lag can be specified in bytes (integer) or in human readable values, for e.g. 16kB, 64MB, 1GB.
|
||||||
@@ -26,9 +29,18 @@ For all health check ``GET`` requests Patroni returns a JSON document with the s
|
|||||||
- ``GET /replica?lag=10MB``
|
- ``GET /replica?lag=10MB``
|
||||||
- ``GET /replica?lag=1GB``
|
- ``GET /replica?lag=1GB``
|
||||||
|
|
||||||
- ``GET /read-only``: like the above endpoint, but also includes the primary.
|
- ``GET /replica?tag_key1=value1&tag_key2=value2``: replica check endpoint. In addition, It will also check for user defined tags ``key1`` and ``key2`` and their respective values in the **tags** section of the yaml configuration management. If the tag isn't defined for an instance, or if the value in the yaml configuration doesn't match the querying value, it will return HTTP Status Code 503.
|
||||||
|
|
||||||
- ``GET /standby-leader``: returns HTTP status code **200** only when the Patroni node is running as the leader in a :ref:`standby cluster <standby_cluster>`.
|
In the following requests, since we are checking for the leader or standby-leader status, Patroni doesn't apply any of the user defined tags and they will be ignored.
|
||||||
|
- ``GET /?tag_key1=value1&tag_key2=value2``
|
||||||
|
- ``GET /master?tag_key1=value1&tag_key2=value2``
|
||||||
|
- ``GET /leader?tag_key1=value1&tag_key2=value2``
|
||||||
|
- ``GET /primary?tag_key1=value1&tag_key2=value2``
|
||||||
|
- ``GET /read-write?tag_key1=value1&tag_key2=value2``
|
||||||
|
- ``GET /standby_leader?tag_key1=value1&tag_key2=value2``
|
||||||
|
- ``GET /standby-leader?tag_key1=value1&tag_key2=value2``
|
||||||
|
|
||||||
|
- ``GET /read-only``: like the above endpoint, but also includes the primary.
|
||||||
|
|
||||||
- ``GET /synchronous`` or ``GET /sync``: returns HTTP status code **200** only when the Patroni node is running as a synchronous standby.
|
- ``GET /synchronous`` or ``GET /sync``: returns HTTP status code **200** only when the Patroni node is running as a synchronous standby.
|
||||||
|
|
||||||
@@ -45,7 +57,7 @@ For all health check ``GET`` requests Patroni returns a JSON document with the s
|
|||||||
|
|
||||||
- ``GET /liveness``: always returns HTTP status code **200** what only indicates that Patroni is running. Could be used for ``livenessProbe``.
|
- ``GET /liveness``: always returns HTTP status code **200** what only indicates that Patroni is running. Could be used for ``livenessProbe``.
|
||||||
|
|
||||||
- ``GET /readiness``: returns HTTP status code **200** when the Patroni node is running as the leader or when PostgreSQL is up and running. The endpoint could be used for ``readinessProbe`` when it is not possible to use Kubenetes endpoints for leader elections (OpenShift).
|
- ``GET /readiness``: returns HTTP status code **200** when the Patroni node is running as the leader or when PostgreSQL is up and running. The endpoint could be used for ``readinessProbe`` when it is not possible to use Kubernetes endpoints for leader elections (OpenShift).
|
||||||
|
|
||||||
Both, ``readiness`` and ``liveness`` endpoints are very light-weight and not executing any SQL. Probes should be configured in such a way that they start failing about time when the leader key is expiring. With the default value of ``ttl``, which is ``30s`` example probes would look like:
|
Both, ``readiness`` and ``liveness`` endpoints are very light-weight and not executing any SQL. Probes should be configured in such a way that they start failing about time when the leader key is expiring. With the default value of ``ttl``, which is ``30s`` example probes would look like:
|
||||||
|
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ Feature: basic replication
|
|||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
|
|
||||||
Scenario: check stuck sync replica
|
Scenario: check stuck sync replica
|
||||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"maximum_lag_on_syncnode": 15000000, "postgresql": {"parameters": {"synchronous_commit": "remote_apply"}}}
|
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"pause": true, "maximum_lag_on_syncnode": 15000000, "postgresql": {"parameters": {"synchronous_commit": "remote_apply"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I create table on postgres0
|
And I create table on postgres0
|
||||||
And table mytest is present on postgres1 after 2 seconds
|
And table mytest is present on postgres1 after 2 seconds
|
||||||
@@ -43,7 +43,7 @@ Feature: basic replication
|
|||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
When I issue a GET request to http://127.0.0.1:8010/async
|
When I issue a GET request to http://127.0.0.1:8010/async
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
When I issue a PATCH request to http://127.0.0.1:8008/config with {"maximum_lag_on_syncnode": -1, "postgresql": {"parameters": {"synchronous_commit": "on"}}}
|
When I issue a PATCH request to http://127.0.0.1:8008/config with {"pause": null, "maximum_lag_on_syncnode": -1, "postgresql": {"parameters": {"synchronous_commit": "on"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I drop table on postgres0
|
And I drop table on postgres0
|
||||||
|
|
||||||
|
|||||||
@@ -1,17 +0,0 @@
|
|||||||
#!/usr/bin/env python
|
|
||||||
import os
|
|
||||||
import psycopg2
|
|
||||||
import sys
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
if not (len(sys.argv) >= 3 and sys.argv[3] == "master"):
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
os.environ['PGPASSWORD'] = 'zalando'
|
|
||||||
connection = psycopg2.connect(host='127.0.0.1', port=sys.argv[1], user='postgres')
|
|
||||||
cursor = connection.cursor()
|
|
||||||
cursor.execute("SELECT slot_name FROM pg_replication_slots WHERE slot_type = 'logical'")
|
|
||||||
|
|
||||||
with open("data/postgres0/label", "w") as label:
|
|
||||||
label.write(next(iter(cursor.fetchone()), ""))
|
|
||||||
+15
-13
@@ -1,7 +1,6 @@
|
|||||||
import abc
|
import abc
|
||||||
import datetime
|
import datetime
|
||||||
import os
|
import os
|
||||||
import psycopg2
|
|
||||||
import json
|
import json
|
||||||
import shutil
|
import shutil
|
||||||
import signal
|
import signal
|
||||||
@@ -13,6 +12,8 @@ import threading
|
|||||||
import time
|
import time
|
||||||
import yaml
|
import yaml
|
||||||
|
|
||||||
|
import patroni.psycopg as psycopg
|
||||||
|
|
||||||
from six.moves.BaseHTTPServer import BaseHTTPRequestHandler, HTTPServer
|
from six.moves.BaseHTTPServer import BaseHTTPRequestHandler, HTTPServer
|
||||||
|
|
||||||
|
|
||||||
@@ -180,6 +181,7 @@ class PatroniController(AbstractController):
|
|||||||
config['postgresql']['data_dir'] = self._data_dir
|
config['postgresql']['data_dir'] = self._data_dir
|
||||||
config['postgresql']['basebackup'] = [{'checkpoint': 'fast'}]
|
config['postgresql']['basebackup'] = [{'checkpoint': 'fast'}]
|
||||||
config['postgresql']['use_unix_socket'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets
|
config['postgresql']['use_unix_socket'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets
|
||||||
|
config['postgresql']['use_unix_socket_repl'] = os.name != 'nt' # windows doesn't yet support unix-domain sockets
|
||||||
config['postgresql']['pgpass'] = os.path.join(tempfile.gettempdir(), 'pgpass_' + name)
|
config['postgresql']['pgpass'] = os.path.join(tempfile.gettempdir(), 'pgpass_' + name)
|
||||||
config['postgresql']['parameters'].update({
|
config['postgresql']['parameters'].update({
|
||||||
'logging_collector': 'on', 'log_destination': 'csvlog', 'log_directory': self._output_dir,
|
'logging_collector': 'on', 'log_destination': 'csvlog', 'log_directory': self._output_dir,
|
||||||
@@ -204,16 +206,16 @@ class PatroniController(AbstractController):
|
|||||||
|
|
||||||
user = config['postgresql'].get('authentication', config['postgresql']).get('superuser', {})
|
user = config['postgresql'].get('authentication', config['postgresql']).get('superuser', {})
|
||||||
self._connkwargs = {k: user[n] for n, k in [('username', 'user'), ('password', 'password')] if n in user}
|
self._connkwargs = {k: user[n] for n, k in [('username', 'user'), ('password', 'password')] if n in user}
|
||||||
self._connkwargs.update({'host': host, 'port': self.__PORT, 'database': 'postgres'})
|
self._connkwargs.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres'})
|
||||||
|
|
||||||
self._replication = config['postgresql'].get('authentication', config['postgresql']).get('replication', {})
|
self._replication = config['postgresql'].get('authentication', config['postgresql']).get('replication', {})
|
||||||
self._replication.update({'host': host, 'port': self.__PORT, 'database': 'postgres'})
|
self._replication.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres'})
|
||||||
|
|
||||||
return patroni_config_path
|
return patroni_config_path
|
||||||
|
|
||||||
def _connection(self):
|
def _connection(self):
|
||||||
if not self._conn or self._conn.closed != 0:
|
if not self._conn or self._conn.closed != 0:
|
||||||
self._conn = psycopg2.connect(**self._connkwargs)
|
self._conn = psycopg.connect(**self._connkwargs)
|
||||||
self._conn.autocommit = True
|
self._conn.autocommit = True
|
||||||
return self._conn
|
return self._conn
|
||||||
|
|
||||||
@@ -227,7 +229,7 @@ class PatroniController(AbstractController):
|
|||||||
cursor = self._cursor()
|
cursor = self._cursor()
|
||||||
cursor.execute(query)
|
cursor.execute(query)
|
||||||
return cursor
|
return cursor
|
||||||
except psycopg2.Error:
|
except psycopg.Error:
|
||||||
if not fail_ok:
|
if not fail_ok:
|
||||||
raise
|
raise
|
||||||
|
|
||||||
@@ -267,7 +269,7 @@ class PatroniController(AbstractController):
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def backup_source(self):
|
def backup_source(self):
|
||||||
return 'postgres://{username}:{password}@{host}:{port}/{database}'.format(**self._replication)
|
return 'postgres://{username}:{password}@{host}:{port}/{dbname}'.format(**self._replication)
|
||||||
|
|
||||||
def backup(self, dest=os.path.join('data', 'basebackup')):
|
def backup(self, dest=os.path.join('data', 'basebackup')):
|
||||||
subprocess.call(PatroniPoolController.BACKUP_SCRIPT + ['--walmethod=none',
|
subprocess.call(PatroniPoolController.BACKUP_SCRIPT + ['--walmethod=none',
|
||||||
@@ -590,10 +592,12 @@ class ExhibitorController(ZooKeeperController):
|
|||||||
class RaftController(AbstractDcsController):
|
class RaftController(AbstractDcsController):
|
||||||
|
|
||||||
CONTROLLER_ADDR = 'localhost:1234'
|
CONTROLLER_ADDR = 'localhost:1234'
|
||||||
|
PASSWORD = '12345'
|
||||||
|
|
||||||
def __init__(self, context):
|
def __init__(self, context):
|
||||||
super(RaftController, self).__init__(context)
|
super(RaftController, self).__init__(context)
|
||||||
os.environ.update(PATRONI_RAFT_PARTNER_ADDRS="'" + self.CONTROLLER_ADDR + "'", RAFT_PORT='1234')
|
os.environ.update(PATRONI_RAFT_PARTNER_ADDRS="'" + self.CONTROLLER_ADDR + "'",
|
||||||
|
PATRONI_RAFT_PASSWORD=self.PASSWORD, RAFT_PORT='1234')
|
||||||
self._raft = None
|
self._raft = None
|
||||||
|
|
||||||
def _start(self):
|
def _start(self):
|
||||||
@@ -614,18 +618,16 @@ class RaftController(AbstractDcsController):
|
|||||||
|
|
||||||
def cleanup_service_tree(self):
|
def cleanup_service_tree(self):
|
||||||
from patroni.dcs.raft import KVStoreTTL
|
from patroni.dcs.raft import KVStoreTTL
|
||||||
from pysyncobj import SyncObjConf
|
|
||||||
|
|
||||||
if self._raft:
|
if self._raft:
|
||||||
self._raft.destroy()
|
self._raft.destroy()
|
||||||
self._raft._SyncObj__thread.join()
|
|
||||||
self.stop()
|
self.stop()
|
||||||
os.makedirs(self._work_directory)
|
os.makedirs(self._work_directory)
|
||||||
self.start()
|
self.start()
|
||||||
|
|
||||||
ready_event = threading.Event()
|
ready_event = threading.Event()
|
||||||
conf = SyncObjConf(appendEntriesUseBatch=False, dynamicMembershipChange=True, onReady=ready_event.set)
|
self._raft = KVStoreTTL(ready_event.set, None, None, partner_addrs=[self.CONTROLLER_ADDR], password=self.PASSWORD)
|
||||||
self._raft = KVStoreTTL(None, [self.CONTROLLER_ADDR], conf)
|
self._raft.startAutoTick()
|
||||||
ready_event.wait()
|
ready_event.wait()
|
||||||
|
|
||||||
|
|
||||||
@@ -658,7 +660,7 @@ class PatroniPoolController(object):
|
|||||||
def output_dir(self):
|
def output_dir(self):
|
||||||
return self._output_dir
|
return self._output_dir
|
||||||
|
|
||||||
def start(self, name, max_wait_limit=20, custom_config=None):
|
def start(self, name, max_wait_limit=40, custom_config=None):
|
||||||
if name not in self._processes:
|
if name not in self._processes:
|
||||||
self._processes[name] = PatroniController(self._context, name, self.patroni_path,
|
self._processes[name] = PatroniController(self._context, name, self.patroni_path,
|
||||||
self._output_dir, custom_config)
|
self._output_dir, custom_config)
|
||||||
@@ -868,7 +870,7 @@ class WatchdogMonitor(object):
|
|||||||
return triggered
|
return triggered
|
||||||
|
|
||||||
|
|
||||||
# actions to execute on start/stop of the tests and before running invidual features
|
# actions to execute on start/stop of the tests and before running individual features
|
||||||
def before_all(context):
|
def before_all(context):
|
||||||
os.environ.update({'PATRONI_RESTAPI_USERNAME': 'username', 'PATRONI_RESTAPI_PASSWORD': 'password'})
|
os.environ.update({'PATRONI_RESTAPI_USERNAME': 'username', 'PATRONI_RESTAPI_PASSWORD': 'password'})
|
||||||
context.ci = any(a in os.environ for a in ('TRAVIS_BUILD_NUMBER', 'BUILD_NUMBER', 'GITHUB_ACTIONS'))
|
context.ci = any(a in os.environ for a in ('TRAVIS_BUILD_NUMBER', 'BUILD_NUMBER', 'GITHUB_ACTIONS'))
|
||||||
|
|||||||
@@ -51,11 +51,11 @@ Scenario: check the scheduled restart
|
|||||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"postgresql": {"parameters": {"superuser_reserved_connections": "6"}}}
|
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"postgresql": {"parameters": {"superuser_reserved_connections": "6"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 5 seconds
|
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 5 seconds
|
||||||
Given I issue a scheduled restart at http://127.0.0.1:8008 in 3 seconds with {"role": "replica"}
|
Given I issue a scheduled restart at http://127.0.0.1:8008 in 5 seconds with {"role": "replica"}
|
||||||
Then I receive a response code 202
|
Then I receive a response code 202
|
||||||
And I sleep for 4 seconds
|
And I sleep for 8 seconds
|
||||||
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 10 seconds
|
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 10 seconds
|
||||||
Given I issue a scheduled restart at http://127.0.0.1:8008 in 3 seconds with {"restart_pending": "True"}
|
Given I issue a scheduled restart at http://127.0.0.1:8008 in 5 seconds with {"restart_pending": "True"}
|
||||||
Then I receive a response code 202
|
Then I receive a response code 202
|
||||||
And Response on GET http://127.0.0.1:8008/patroni does not contain pending_restart after 10 seconds
|
And Response on GET http://127.0.0.1:8008/patroni does not contain pending_restart after 10 seconds
|
||||||
And postgres0 role is the primary after 10 seconds
|
And postgres0 role is the primary after 10 seconds
|
||||||
@@ -71,6 +71,7 @@ Scenario: check API requests for the primary-replica pair in the pause mode
|
|||||||
When I run patronictl.py restart batman postgres1 --force
|
When I run patronictl.py restart batman postgres1 --force
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
Then replication works from postgres0 to postgres1 after 20 seconds
|
Then replication works from postgres0 to postgres1 after 20 seconds
|
||||||
|
And I sleep for 2 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8009/replica
|
When I issue a GET request to http://127.0.0.1:8009/replica
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I receive a response state running
|
And I receive a response state running
|
||||||
@@ -103,12 +104,12 @@ Scenario: check the switchover via the API in the pause mode
|
|||||||
Then I receive a response code 503
|
Then I receive a response code 503
|
||||||
|
|
||||||
Scenario: check the scheduled switchover
|
Scenario: check the scheduled switchover
|
||||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 3 seconds
|
Given I issue a scheduled switchover from postgres1 to postgres0 in 10 seconds
|
||||||
Then I receive a response returncode 1
|
Then I receive a response returncode 1
|
||||||
And I receive a response output "Can't schedule switchover in the paused state"
|
And I receive a response output "Can't schedule switchover in the paused state"
|
||||||
When I run patronictl.py resume batman
|
When I run patronictl.py resume batman
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 3 seconds
|
Given I issue a scheduled switchover from postgres1 to postgres0 in 5 seconds
|
||||||
Then I receive a response returncode 0
|
Then I receive a response returncode 0
|
||||||
And postgres0 is a leader after 20 seconds
|
And postgres0 is a leader after 20 seconds
|
||||||
And postgres0 role is the primary after 10 seconds
|
And postgres0 role is the primary after 10 seconds
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
Feature: standby cluster
|
Feature: standby cluster
|
||||||
Scenario: check permanent logical slots are preserved on failover/switchover
|
Scenario: prepare the cluster with logical slots
|
||||||
Given I start postgres1
|
Given I start postgres1
|
||||||
Then postgres1 is a leader after 10 seconds
|
Then postgres1 is a leader after 10 seconds
|
||||||
And there is a non empty initialize key in DCS after 15 seconds
|
And there is a non empty initialize key in DCS after 15 seconds
|
||||||
@@ -10,15 +10,24 @@ Feature: standby cluster
|
|||||||
When I issue a PATCH request to http://127.0.0.1:8009/config with {"slots": {"test_logical": {"type": "logical", "database": "postgres", "plugin": "test_decoding"}}}
|
When I issue a PATCH request to http://127.0.0.1:8009/config with {"slots": {"test_logical": {"type": "logical", "database": "postgres", "plugin": "test_decoding"}}}
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And I do a backup of postgres1
|
And I do a backup of postgres1
|
||||||
When I start postgres0 with callback configured
|
When I start postgres0
|
||||||
Then "members/postgres0" key in DCS has state=running after 10 seconds
|
Then "members/postgres0" key in DCS has state=running after 10 seconds
|
||||||
And replication works from postgres1 to postgres0 after 15 seconds
|
And replication works from postgres1 to postgres0 after 15 seconds
|
||||||
|
|
||||||
|
@skip
|
||||||
|
Scenario: check permanent logical slots are synced to the replica
|
||||||
|
Given I run patronictl.py restart batman postgres1 --force
|
||||||
|
Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds
|
||||||
|
When I add the table replicate_me to postgres1
|
||||||
|
And I get all changes from logical slot test_logical on postgres1
|
||||||
|
Then Logical slot test_logical is in sync between postgres0 and postgres1 after 10 seconds
|
||||||
|
|
||||||
|
Scenario: Detach exiting node from the cluster
|
||||||
When I shut down postgres1
|
When I shut down postgres1
|
||||||
Then postgres0 is a leader after 10 seconds
|
Then postgres0 is a leader after 10 seconds
|
||||||
And "members/postgres0" key in DCS has role=master after 3 seconds
|
And "members/postgres0" key in DCS has role=master after 3 seconds
|
||||||
When I issue a GET request to http://127.0.0.1:8008/
|
When I issue a GET request to http://127.0.0.1:8008/
|
||||||
Then I receive a response code 200
|
Then I receive a response code 200
|
||||||
And there is a label with "test_logical" in postgres0 data directory
|
|
||||||
|
|
||||||
Scenario: check replication of a single table in a standby cluster
|
Scenario: check replication of a single table in a standby cluster
|
||||||
Given I start postgres1 in a standby cluster batman1 as a clone of postgres0
|
Given I start postgres1 in a standby cluster batman1 as a clone of postgres0
|
||||||
@@ -35,6 +44,7 @@ Feature: standby cluster
|
|||||||
When I start postgres2 in a cluster batman1
|
When I start postgres2 in a cluster batman1
|
||||||
Then postgres2 role is the replica after 24 seconds
|
Then postgres2 role is the replica after 24 seconds
|
||||||
And table foo is present on postgres2 after 20 seconds
|
And table foo is present on postgres2 after 20 seconds
|
||||||
|
And postgres1 does not have a logical replication slot named test_logical
|
||||||
|
|
||||||
Scenario: check failover
|
Scenario: check failover
|
||||||
When I kill postgres1
|
When I kill postgres1
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
import psycopg2 as pg
|
import patroni.psycopg as pg
|
||||||
|
|
||||||
from behave import step, then
|
from behave import step, then
|
||||||
from time import sleep, time
|
from time import sleep, time
|
||||||
|
|||||||
@@ -76,6 +76,8 @@ def do_request(context, request_method, url, data):
|
|||||||
data = data and json.loads(data)
|
data = data and json.loads(data)
|
||||||
try:
|
try:
|
||||||
r = request_executor.request(request_method, url, data)
|
r = request_executor.request(request_method, url, data)
|
||||||
|
if request_method == 'PATCH' and r.status == 409:
|
||||||
|
r = request_executor.request(request_method, url, data)
|
||||||
except Exception:
|
except Exception:
|
||||||
context.status_code = context.response = None
|
context.status_code = context.response = None
|
||||||
else:
|
else:
|
||||||
|
|||||||
+25
-1
@@ -1,5 +1,7 @@
|
|||||||
|
import time
|
||||||
|
|
||||||
from behave import step, then
|
from behave import step, then
|
||||||
import psycopg2 as pg
|
import patroni.psycopg as pg
|
||||||
|
|
||||||
|
|
||||||
@step('I create a logical replication slot {slot_name} on {pg_name:w} with the {plugin:w} plugin')
|
@step('I create a logical replication slot {slot_name} on {pg_name:w} with the {plugin:w} plugin')
|
||||||
@@ -34,3 +36,25 @@ def does_not_have_logical_replication_slot(context, pg_name, slot_name):
|
|||||||
assert not row, "Found unexpected replication slot named {0}".format(slot_name)
|
assert not row, "Found unexpected replication slot named {0}".format(slot_name)
|
||||||
except pg.Error:
|
except pg.Error:
|
||||||
assert False, "Error looking for slot {0} on {1}".format(slot_name, pg_name)
|
assert False, "Error looking for slot {0} on {1}".format(slot_name, pg_name)
|
||||||
|
|
||||||
|
|
||||||
|
@step('Logical slot {slot_name:w} is in sync between {pg_name1:w} and {pg_name2:w} after {time_limit:d} seconds')
|
||||||
|
def logical_slots_in_sync(context, slot_name, pg_name1, pg_name2, time_limit):
|
||||||
|
time_limit *= context.timeout_multiplier
|
||||||
|
max_time = time.time() + int(time_limit)
|
||||||
|
while time.time() < max_time:
|
||||||
|
try:
|
||||||
|
query = "SELECT confirmed_flush_lsn FROM pg_replication_slots WHERE slot_name = '{0}'".format(slot_name)
|
||||||
|
slot1 = context.pctl.query(pg_name1, query).fetchone()
|
||||||
|
slot2 = context.pctl.query(pg_name2, query).fetchone()
|
||||||
|
if slot1[0] == slot2[0]:
|
||||||
|
return
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
time.sleep(1)
|
||||||
|
assert False, "Logical slot {0} is not in sync between {1} and {2}".format(slot_name, pg_name1, pg_name2)
|
||||||
|
|
||||||
|
|
||||||
|
@step('I get all changes from logical slot {slot_name:w} on {pg_name:w}')
|
||||||
|
def logical_slot_get_changes(context, slot_name, pg_name):
|
||||||
|
context.pctl.query(pg_name, "SELECT * FROM pg_logical_slot_get_changes('{0}', NULL, NULL)".format(slot_name))
|
||||||
|
|||||||
@@ -14,17 +14,6 @@ executable = sys.executable if os.name != 'nt' else sys.executable.replace('\\',
|
|||||||
callback = executable + " features/callback2.py "
|
callback = executable + " features/callback2.py "
|
||||||
|
|
||||||
|
|
||||||
@step('I start {name:w} with callback configured')
|
|
||||||
def start_patroni_with_callbacks(context, name):
|
|
||||||
return context.pctl.start(name, custom_config={
|
|
||||||
"postgresql": {
|
|
||||||
"callbacks": {
|
|
||||||
"on_role_change": executable + " features/callback.py"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
|
|
||||||
@step('I start {name:w} in a cluster {cluster_name:w}')
|
@step('I start {name:w} in a cluster {cluster_name:w}')
|
||||||
def start_patroni(context, name, cluster_name):
|
def start_patroni(context, name, cluster_name):
|
||||||
return context.pctl.start(name, custom_config={
|
return context.pctl.start(name, custom_config={
|
||||||
@@ -55,7 +44,8 @@ def start_patroni_standby_cluster(context, name, cluster_name, name2):
|
|||||||
"port": port,
|
"port": port,
|
||||||
"primary_slot_name": "pm_1",
|
"primary_slot_name": "pm_1",
|
||||||
"create_replica_methods": ["backup_restore", "basebackup"]
|
"create_replica_methods": ["backup_restore", "basebackup"]
|
||||||
}
|
},
|
||||||
|
"postgresql": {"parameters": {"wal_level": "logical"}}
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
"postgresql": {
|
"postgresql": {
|
||||||
@@ -67,7 +57,7 @@ def start_patroni_standby_cluster(context, name, cluster_name, name2):
|
|||||||
|
|
||||||
@step('{pg_name1:w} is replicating from {pg_name2:w} after {timeout:d} seconds')
|
@step('{pg_name1:w} is replicating from {pg_name2:w} after {timeout:d} seconds')
|
||||||
def check_replication_status(context, pg_name1, pg_name2, timeout):
|
def check_replication_status(context, pg_name1, pg_name2, timeout):
|
||||||
bound_time = time.time() + timeout
|
bound_time = time.time() + timeout * context.timeout_multiplier
|
||||||
|
|
||||||
while time.time() < bound_time:
|
while time.time() < bound_time:
|
||||||
cur = context.pctl.query(
|
cur = context.pctl.query(
|
||||||
|
|||||||
@@ -20,6 +20,10 @@ metadata:
|
|||||||
spec:
|
spec:
|
||||||
replicas: 3
|
replicas: 3
|
||||||
serviceName: *cluster_name
|
serviceName: *cluster_name
|
||||||
|
selector:
|
||||||
|
matchLabels:
|
||||||
|
application: patroni
|
||||||
|
cluster-name: *cluster_name
|
||||||
template:
|
template:
|
||||||
metadata:
|
metadata:
|
||||||
labels:
|
labels:
|
||||||
|
|||||||
+1
-1
@@ -1,5 +1,5 @@
|
|||||||
#!/usr/bin/env python
|
#!/usr/bin/env python
|
||||||
from patroni import main
|
from patroni.__main__ import main
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
|
|||||||
+21
-185
@@ -1,141 +1,8 @@
|
|||||||
import logging
|
|
||||||
import os
|
|
||||||
import signal
|
|
||||||
import sys
|
import sys
|
||||||
import time
|
|
||||||
|
|
||||||
from .daemon import AbstractPatroniDaemon, abstract_main
|
|
||||||
from .version import __version__
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
PATRONI_ENV_PREFIX = 'PATRONI_'
|
PATRONI_ENV_PREFIX = 'PATRONI_'
|
||||||
KUBERNETES_ENV_PREFIX = 'KUBERNETES_'
|
KUBERNETES_ENV_PREFIX = 'KUBERNETES_'
|
||||||
|
MIN_PSYCOPG2 = (2, 5, 4)
|
||||||
|
|
||||||
class Patroni(AbstractPatroniDaemon):
|
|
||||||
|
|
||||||
def __init__(self, config):
|
|
||||||
from patroni.api import RestApiServer
|
|
||||||
from patroni.dcs import get_dcs
|
|
||||||
from patroni.ha import Ha
|
|
||||||
from patroni.postgresql import Postgresql
|
|
||||||
from patroni.request import PatroniRequest
|
|
||||||
from patroni.watchdog import Watchdog
|
|
||||||
|
|
||||||
super(Patroni, self).__init__(config)
|
|
||||||
|
|
||||||
self.version = __version__
|
|
||||||
self.dcs = get_dcs(self.config)
|
|
||||||
self.watchdog = Watchdog(self.config)
|
|
||||||
self.load_dynamic_configuration()
|
|
||||||
|
|
||||||
self.postgresql = Postgresql(self.config['postgresql'])
|
|
||||||
self.api = RestApiServer(self, self.config['restapi'])
|
|
||||||
self.request = PatroniRequest(self.config, True)
|
|
||||||
self.ha = Ha(self)
|
|
||||||
|
|
||||||
self.tags = self.get_tags()
|
|
||||||
self.next_run = time.time()
|
|
||||||
self.scheduled_restart = {}
|
|
||||||
|
|
||||||
def load_dynamic_configuration(self):
|
|
||||||
from patroni.exceptions import DCSError
|
|
||||||
while True:
|
|
||||||
try:
|
|
||||||
cluster = self.dcs.get_cluster()
|
|
||||||
if cluster and cluster.config and cluster.config.data:
|
|
||||||
if self.config.set_dynamic_configuration(cluster.config):
|
|
||||||
self.dcs.reload_config(self.config)
|
|
||||||
self.watchdog.reload_config(self.config)
|
|
||||||
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
|
||||||
if self.config.set_dynamic_configuration(self.config['bootstrap']['dcs']):
|
|
||||||
self.dcs.reload_config(self.config)
|
|
||||||
break
|
|
||||||
except DCSError:
|
|
||||||
logger.warning('Can not get cluster from dcs')
|
|
||||||
time.sleep(5)
|
|
||||||
|
|
||||||
def get_tags(self):
|
|
||||||
return {tag: value for tag, value in self.config.get('tags', {}).items()
|
|
||||||
if tag not in ('clonefrom', 'nofailover', 'noloadbalance', 'nosync') or value}
|
|
||||||
|
|
||||||
@property
|
|
||||||
def nofailover(self):
|
|
||||||
return bool(self.tags.get('nofailover', False))
|
|
||||||
|
|
||||||
@property
|
|
||||||
def nosync(self):
|
|
||||||
return bool(self.tags.get('nosync', False))
|
|
||||||
|
|
||||||
def reload_config(self, sighup=False, local=False):
|
|
||||||
try:
|
|
||||||
super(Patroni, self).reload_config(sighup, local)
|
|
||||||
if local:
|
|
||||||
self.tags = self.get_tags()
|
|
||||||
self.request.reload_config(self.config)
|
|
||||||
self.api.reload_config(self.config['restapi'])
|
|
||||||
self.watchdog.reload_config(self.config)
|
|
||||||
self.postgresql.reload_config(self.config['postgresql'], sighup)
|
|
||||||
self.dcs.reload_config(self.config)
|
|
||||||
except Exception:
|
|
||||||
logger.exception('Failed to reload config_file=%s', self.config.config_file)
|
|
||||||
|
|
||||||
@property
|
|
||||||
def replicatefrom(self):
|
|
||||||
return self.tags.get('replicatefrom')
|
|
||||||
|
|
||||||
@property
|
|
||||||
def noloadbalance(self):
|
|
||||||
return bool(self.tags.get('noloadbalance', False))
|
|
||||||
|
|
||||||
def schedule_next_run(self):
|
|
||||||
self.next_run += self.dcs.loop_wait
|
|
||||||
current_time = time.time()
|
|
||||||
nap_time = self.next_run - current_time
|
|
||||||
if nap_time <= 0:
|
|
||||||
self.next_run = current_time
|
|
||||||
# Release the GIL so we don't starve anyone waiting on async_executor lock
|
|
||||||
time.sleep(0.001)
|
|
||||||
# Warn user that Patroni is not keeping up
|
|
||||||
logger.warning("Loop time exceeded, rescheduling immediately.")
|
|
||||||
elif self.ha.watch(nap_time):
|
|
||||||
self.next_run = time.time()
|
|
||||||
|
|
||||||
def run(self):
|
|
||||||
self.api.start()
|
|
||||||
self.next_run = time.time()
|
|
||||||
super(Patroni, self).run()
|
|
||||||
|
|
||||||
def _run_cycle(self):
|
|
||||||
logger.info(self.ha.run_cycle())
|
|
||||||
|
|
||||||
if self.dcs.cluster and self.dcs.cluster.config and self.dcs.cluster.config.data \
|
|
||||||
and self.config.set_dynamic_configuration(self.dcs.cluster.config):
|
|
||||||
self.reload_config()
|
|
||||||
|
|
||||||
if self.postgresql.role != 'uninitialized':
|
|
||||||
self.config.save_cache()
|
|
||||||
|
|
||||||
self.schedule_next_run()
|
|
||||||
|
|
||||||
def _shutdown(self):
|
|
||||||
try:
|
|
||||||
self.api.shutdown()
|
|
||||||
except Exception:
|
|
||||||
logger.exception('Exception during RestApi.shutdown')
|
|
||||||
try:
|
|
||||||
self.ha.shutdown()
|
|
||||||
except Exception:
|
|
||||||
logger.exception('Exception during Ha.shutdown')
|
|
||||||
|
|
||||||
|
|
||||||
def patroni_main():
|
|
||||||
from multiprocessing import freeze_support
|
|
||||||
from patroni.validator import schema
|
|
||||||
|
|
||||||
freeze_support()
|
|
||||||
abstract_main(Patroni, schema)
|
|
||||||
|
|
||||||
|
|
||||||
def fatal(string, *args):
|
def fatal(string, *args):
|
||||||
@@ -143,63 +10,32 @@ def fatal(string, *args):
|
|||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|
||||||
|
|
||||||
def check_psycopg2():
|
def parse_version(version):
|
||||||
min_psycopg2 = (2, 5, 4)
|
def _parse_version(version):
|
||||||
min_psycopg2_str = '.'.join(map(str, min_psycopg2))
|
|
||||||
|
|
||||||
def parse_version(version):
|
|
||||||
for e in version.split('.'):
|
for e in version.split('.'):
|
||||||
try:
|
try:
|
||||||
yield int(e)
|
yield int(e)
|
||||||
except ValueError:
|
except ValueError:
|
||||||
break
|
break
|
||||||
|
return tuple(_parse_version(version.split(' ')[0]))
|
||||||
|
|
||||||
|
|
||||||
|
# We pass MIN_PSYCOPG2 and parse_version as arguments to simplify usage of check_psycopg from the setup.py
|
||||||
|
def check_psycopg(_min_psycopg2=MIN_PSYCOPG2, _parse_version=parse_version):
|
||||||
|
min_psycopg2_str = '.'.join(map(str, _min_psycopg2))
|
||||||
|
|
||||||
try:
|
try:
|
||||||
import psycopg2
|
from psycopg2 import __version__
|
||||||
version_str = psycopg2.__version__.split(' ')[0]
|
if _parse_version(__version__) >= _min_psycopg2:
|
||||||
version = tuple(parse_version(version_str))
|
return
|
||||||
if version < min_psycopg2:
|
version_str = __version__.split(' ')[0]
|
||||||
fatal('Patroni requires psycopg2>={0}, but only {1} is available', min_psycopg2_str, version_str)
|
|
||||||
except ImportError:
|
except ImportError:
|
||||||
fatal('Patroni requires psycopg2>={0} or psycopg2-binary', min_psycopg2_str)
|
version_str = None
|
||||||
|
|
||||||
|
try:
|
||||||
def main():
|
from psycopg import __version__
|
||||||
if os.getpid() != 1:
|
except ImportError:
|
||||||
check_psycopg2()
|
error = 'Patroni requires psycopg2>={0}, psycopg2-binary, or psycopg>=3.0'.format(min_psycopg2_str)
|
||||||
return patroni_main()
|
if version_str:
|
||||||
|
error += ', but only psycopg2=={0} is available'.format(version_str)
|
||||||
# Patroni started with PID=1, it looks like we are in the container
|
fatal(error)
|
||||||
pid = 0
|
|
||||||
|
|
||||||
# Looks like we are in a docker, so we will act like init
|
|
||||||
def sigchld_handler(signo, stack_frame):
|
|
||||||
try:
|
|
||||||
while True:
|
|
||||||
ret = os.waitpid(-1, os.WNOHANG)
|
|
||||||
if ret == (0, 0):
|
|
||||||
break
|
|
||||||
elif ret[0] != pid:
|
|
||||||
logger.info('Reaped pid=%s, exit status=%s', *ret)
|
|
||||||
except OSError:
|
|
||||||
pass
|
|
||||||
|
|
||||||
def passtochild(signo, stack_frame):
|
|
||||||
if pid:
|
|
||||||
os.kill(pid, signo)
|
|
||||||
|
|
||||||
if os.name != 'nt':
|
|
||||||
signal.signal(signal.SIGCHLD, sigchld_handler)
|
|
||||||
signal.signal(signal.SIGHUP, passtochild)
|
|
||||||
signal.signal(signal.SIGQUIT, passtochild)
|
|
||||||
signal.signal(signal.SIGUSR1, passtochild)
|
|
||||||
signal.signal(signal.SIGUSR2, passtochild)
|
|
||||||
signal.signal(signal.SIGINT, passtochild)
|
|
||||||
signal.signal(signal.SIGABRT, passtochild)
|
|
||||||
signal.signal(signal.SIGTERM, passtochild)
|
|
||||||
|
|
||||||
import multiprocessing
|
|
||||||
patroni = multiprocessing.Process(target=patroni_main)
|
|
||||||
patroni.start()
|
|
||||||
pid = patroni.pid
|
|
||||||
patroni.join()
|
|
||||||
|
|||||||
+178
-1
@@ -1,4 +1,181 @@
|
|||||||
from patroni import main
|
import logging
|
||||||
|
import os
|
||||||
|
import signal
|
||||||
|
import time
|
||||||
|
|
||||||
|
from .daemon import AbstractPatroniDaemon, abstract_main
|
||||||
|
|
||||||
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class Patroni(AbstractPatroniDaemon):
|
||||||
|
|
||||||
|
def __init__(self, config):
|
||||||
|
from .api import RestApiServer
|
||||||
|
from .dcs import get_dcs
|
||||||
|
from .ha import Ha
|
||||||
|
from .postgresql import Postgresql
|
||||||
|
from .request import PatroniRequest
|
||||||
|
from .version import __version__
|
||||||
|
from .watchdog import Watchdog
|
||||||
|
|
||||||
|
super(Patroni, self).__init__(config)
|
||||||
|
|
||||||
|
self.version = __version__
|
||||||
|
self.dcs = get_dcs(self.config)
|
||||||
|
self.watchdog = Watchdog(self.config)
|
||||||
|
self.load_dynamic_configuration()
|
||||||
|
|
||||||
|
self.postgresql = Postgresql(self.config['postgresql'])
|
||||||
|
self.api = RestApiServer(self, self.config['restapi'])
|
||||||
|
self.request = PatroniRequest(self.config, True)
|
||||||
|
self.ha = Ha(self)
|
||||||
|
|
||||||
|
self.tags = self.get_tags()
|
||||||
|
self.next_run = time.time()
|
||||||
|
self.scheduled_restart = {}
|
||||||
|
|
||||||
|
def load_dynamic_configuration(self):
|
||||||
|
from patroni.exceptions import DCSError
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
cluster = self.dcs.get_cluster()
|
||||||
|
if cluster and cluster.config and cluster.config.data:
|
||||||
|
if self.config.set_dynamic_configuration(cluster.config):
|
||||||
|
self.dcs.reload_config(self.config)
|
||||||
|
self.watchdog.reload_config(self.config)
|
||||||
|
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
||||||
|
if self.config.set_dynamic_configuration(self.config['bootstrap']['dcs']):
|
||||||
|
self.dcs.reload_config(self.config)
|
||||||
|
break
|
||||||
|
except DCSError:
|
||||||
|
logger.warning('Can not get cluster from dcs')
|
||||||
|
time.sleep(5)
|
||||||
|
|
||||||
|
def get_tags(self):
|
||||||
|
return {tag: value for tag, value in self.config.get('tags', {}).items()
|
||||||
|
if tag not in ('clonefrom', 'nofailover', 'noloadbalance', 'nosync') or value}
|
||||||
|
|
||||||
|
@property
|
||||||
|
def nofailover(self):
|
||||||
|
return bool(self.tags.get('nofailover', False))
|
||||||
|
|
||||||
|
@property
|
||||||
|
def nosync(self):
|
||||||
|
return bool(self.tags.get('nosync', False))
|
||||||
|
|
||||||
|
def reload_config(self, sighup=False, local=False):
|
||||||
|
try:
|
||||||
|
super(Patroni, self).reload_config(sighup, local)
|
||||||
|
if local:
|
||||||
|
self.tags = self.get_tags()
|
||||||
|
self.request.reload_config(self.config)
|
||||||
|
if local or sighup and self.api.reload_local_certificate():
|
||||||
|
self.api.reload_config(self.config['restapi'])
|
||||||
|
self.watchdog.reload_config(self.config)
|
||||||
|
self.postgresql.reload_config(self.config['postgresql'], sighup)
|
||||||
|
self.dcs.reload_config(self.config)
|
||||||
|
except Exception:
|
||||||
|
logger.exception('Failed to reload config_file=%s', self.config.config_file)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def replicatefrom(self):
|
||||||
|
return self.tags.get('replicatefrom')
|
||||||
|
|
||||||
|
@property
|
||||||
|
def noloadbalance(self):
|
||||||
|
return bool(self.tags.get('noloadbalance', False))
|
||||||
|
|
||||||
|
def schedule_next_run(self):
|
||||||
|
self.next_run += self.dcs.loop_wait
|
||||||
|
current_time = time.time()
|
||||||
|
nap_time = self.next_run - current_time
|
||||||
|
if nap_time <= 0:
|
||||||
|
self.next_run = current_time
|
||||||
|
# Release the GIL so we don't starve anyone waiting on async_executor lock
|
||||||
|
time.sleep(0.001)
|
||||||
|
# Warn user that Patroni is not keeping up
|
||||||
|
logger.warning("Loop time exceeded, rescheduling immediately.")
|
||||||
|
elif self.ha.watch(nap_time):
|
||||||
|
self.next_run = time.time()
|
||||||
|
|
||||||
|
def run(self):
|
||||||
|
self.api.start()
|
||||||
|
self.next_run = time.time()
|
||||||
|
super(Patroni, self).run()
|
||||||
|
|
||||||
|
def _run_cycle(self):
|
||||||
|
logger.info(self.ha.run_cycle())
|
||||||
|
|
||||||
|
if self.dcs.cluster and self.dcs.cluster.config and self.dcs.cluster.config.data \
|
||||||
|
and self.config.set_dynamic_configuration(self.dcs.cluster.config):
|
||||||
|
self.reload_config()
|
||||||
|
|
||||||
|
if self.postgresql.role != 'uninitialized':
|
||||||
|
self.config.save_cache()
|
||||||
|
|
||||||
|
self.schedule_next_run()
|
||||||
|
|
||||||
|
def _shutdown(self):
|
||||||
|
try:
|
||||||
|
self.api.shutdown()
|
||||||
|
except Exception:
|
||||||
|
logger.exception('Exception during RestApi.shutdown')
|
||||||
|
try:
|
||||||
|
self.ha.shutdown()
|
||||||
|
except Exception:
|
||||||
|
logger.exception('Exception during Ha.shutdown')
|
||||||
|
|
||||||
|
|
||||||
|
def patroni_main():
|
||||||
|
from multiprocessing import freeze_support
|
||||||
|
from patroni.validator import schema
|
||||||
|
|
||||||
|
freeze_support()
|
||||||
|
abstract_main(Patroni, schema)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
if os.getpid() != 1:
|
||||||
|
from . import check_psycopg
|
||||||
|
|
||||||
|
check_psycopg()
|
||||||
|
return patroni_main()
|
||||||
|
|
||||||
|
# Patroni started with PID=1, it looks like we are in the container
|
||||||
|
pid = 0
|
||||||
|
|
||||||
|
# Looks like we are in a docker, so we will act like init
|
||||||
|
def sigchld_handler(signo, stack_frame):
|
||||||
|
try:
|
||||||
|
while True:
|
||||||
|
ret = os.waitpid(-1, os.WNOHANG)
|
||||||
|
if ret == (0, 0):
|
||||||
|
break
|
||||||
|
elif ret[0] != pid:
|
||||||
|
logger.info('Reaped pid=%s, exit status=%s', *ret)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def passtochild(signo, stack_frame):
|
||||||
|
if pid:
|
||||||
|
os.kill(pid, signo)
|
||||||
|
|
||||||
|
if os.name != 'nt':
|
||||||
|
signal.signal(signal.SIGCHLD, sigchld_handler)
|
||||||
|
signal.signal(signal.SIGHUP, passtochild)
|
||||||
|
signal.signal(signal.SIGQUIT, passtochild)
|
||||||
|
signal.signal(signal.SIGUSR1, passtochild)
|
||||||
|
signal.signal(signal.SIGUSR2, passtochild)
|
||||||
|
signal.signal(signal.SIGINT, passtochild)
|
||||||
|
signal.signal(signal.SIGABRT, passtochild)
|
||||||
|
signal.signal(signal.SIGTERM, passtochild)
|
||||||
|
|
||||||
|
import multiprocessing
|
||||||
|
patroni = multiprocessing.Process(target=patroni_main)
|
||||||
|
patroni.start()
|
||||||
|
pid = patroni.pid
|
||||||
|
patroni.join()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
|
|||||||
+222
-29
@@ -2,7 +2,6 @@ import base64
|
|||||||
import hmac
|
import hmac
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import psycopg2
|
|
||||||
import time
|
import time
|
||||||
import traceback
|
import traceback
|
||||||
import dateutil.parser
|
import dateutil.parser
|
||||||
@@ -12,11 +11,13 @@ import six
|
|||||||
import socket
|
import socket
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
|
from ipaddress import ip_address, ip_network as _ip_network
|
||||||
from six.moves.BaseHTTPServer import BaseHTTPRequestHandler, HTTPServer
|
from six.moves.BaseHTTPServer import BaseHTTPRequestHandler, HTTPServer
|
||||||
from six.moves.socketserver import ThreadingMixIn
|
from six.moves.socketserver import ThreadingMixIn
|
||||||
from six.moves.urllib_parse import urlparse, parse_qs
|
from six.moves.urllib_parse import urlparse, parse_qs
|
||||||
from threading import Thread
|
from threading import Thread
|
||||||
|
|
||||||
|
from . import psycopg
|
||||||
from .exceptions import PostgresConnectionException, PostgresException
|
from .exceptions import PostgresConnectionException, PostgresException
|
||||||
from .postgresql.misc import postgres_version_to_int
|
from .postgresql.misc import postgres_version_to_int
|
||||||
from .utils import deep_compare, enable_keepalive, parse_bool, patch_config, Retry, \
|
from .utils import deep_compare, enable_keepalive, parse_bool, patch_config, Retry, \
|
||||||
@@ -25,6 +26,10 @@ from .utils import deep_compare, enable_keepalive, parse_bool, patch_config, Ret
|
|||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def ip_network(value):
|
||||||
|
return _ip_network(value.decode('utf-8') if six.PY2 else value, False)
|
||||||
|
|
||||||
|
|
||||||
class RestApiHandler(BaseHTTPRequestHandler):
|
class RestApiHandler(BaseHTTPRequestHandler):
|
||||||
|
|
||||||
def _write_status_code_only(self, status_code):
|
def _write_status_code_only(self, status_code):
|
||||||
@@ -45,19 +50,19 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
self.wfile.write(body.encode('utf-8'))
|
self.wfile.write(body.encode('utf-8'))
|
||||||
|
|
||||||
def _write_json_response(self, status_code, response):
|
def _write_json_response(self, status_code, response):
|
||||||
self._write_response(status_code, json.dumps(response), content_type='application/json')
|
self._write_response(status_code, json.dumps(response, default=str), content_type='application/json')
|
||||||
|
|
||||||
def check_auth(func):
|
def check_access(func):
|
||||||
"""Decorator function to check authorization header or client certificates
|
"""Decorator function to check the source ip, authorization header. or client certificates
|
||||||
|
|
||||||
Usage example:
|
Usage example:
|
||||||
@check_auth
|
@check_access
|
||||||
def do_PUT_foo():
|
def do_PUT_foo():
|
||||||
pass
|
pass
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def wrapper(self, *args, **kwargs):
|
def wrapper(self, *args, **kwargs):
|
||||||
if self.server.check_auth(self):
|
if self.server.check_access(self):
|
||||||
return func(self, *args, **kwargs)
|
return func(self, *args, **kwargs)
|
||||||
|
|
||||||
return wrapper
|
return wrapper
|
||||||
@@ -97,7 +102,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
patroni = self.server.patroni
|
patroni = self.server.patroni
|
||||||
cluster = patroni.dcs.cluster
|
cluster = patroni.dcs.cluster
|
||||||
|
|
||||||
leader_optime = cluster and cluster.last_leader_operation or 0
|
leader_optime = cluster and cluster.last_lsn or 0
|
||||||
replayed_location = response.get('xlog', {}).get('replayed_location', 0)
|
replayed_location = response.get('xlog', {}).get('replayed_location', 0)
|
||||||
max_replica_lag = parse_int(self.path_query.get('lag', [sys.maxsize])[0], 'B')
|
max_replica_lag = parse_int(self.path_query.get('lag', [sys.maxsize])[0], 'B')
|
||||||
if max_replica_lag is None:
|
if max_replica_lag is None:
|
||||||
@@ -108,9 +113,11 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
response.get('role') == 'replica' and response.get('state') == 'running' else 503
|
response.get('role') == 'replica' and response.get('state') == 'running' else 503
|
||||||
|
|
||||||
if not cluster and patroni.ha.is_paused():
|
if not cluster and patroni.ha.is_paused():
|
||||||
|
leader_status_code = 200 if response.get('role') in ('master', 'standby_leader') else 503
|
||||||
primary_status_code = 200 if response.get('role') == 'master' else 503
|
primary_status_code = 200 if response.get('role') == 'master' else 503
|
||||||
standby_leader_status_code = 200 if response.get('role') == 'standby_leader' else 503
|
standby_leader_status_code = 200 if response.get('role') == 'standby_leader' else 503
|
||||||
elif patroni.ha.is_leader():
|
elif patroni.ha.is_leader():
|
||||||
|
leader_status_code = 200
|
||||||
if patroni.ha.is_standby_cluster():
|
if patroni.ha.is_standby_cluster():
|
||||||
primary_status_code = replica_status_code = 503
|
primary_status_code = replica_status_code = 503
|
||||||
standby_leader_status_code = 200 if response.get('role') in ('replica', 'standby_leader') else 503
|
standby_leader_status_code = 200 if response.get('role') in ('replica', 'standby_leader') else 503
|
||||||
@@ -118,14 +125,20 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
primary_status_code = 200
|
primary_status_code = 200
|
||||||
standby_leader_status_code = 503
|
standby_leader_status_code = 503
|
||||||
else:
|
else:
|
||||||
primary_status_code = standby_leader_status_code = 503
|
leader_status_code = primary_status_code = standby_leader_status_code = 503
|
||||||
|
|
||||||
status_code = 503
|
status_code = 503
|
||||||
|
|
||||||
|
ignore_tags = False
|
||||||
if 'standby_leader' in path or 'standby-leader' in path:
|
if 'standby_leader' in path or 'standby-leader' in path:
|
||||||
status_code = standby_leader_status_code
|
status_code = standby_leader_status_code
|
||||||
elif 'master' in path or 'leader' in path or 'primary' in path or 'read-write' in path:
|
ignore_tags = True
|
||||||
|
elif 'leader' in path:
|
||||||
|
status_code = leader_status_code
|
||||||
|
ignore_tags = True
|
||||||
|
elif 'master' in path or 'primary' in path or 'read-write' in path:
|
||||||
status_code = primary_status_code
|
status_code = primary_status_code
|
||||||
|
ignore_tags = True
|
||||||
elif 'replica' in path:
|
elif 'replica' in path:
|
||||||
status_code = replica_status_code
|
status_code = replica_status_code
|
||||||
elif 'read-only' in path:
|
elif 'read-only' in path:
|
||||||
@@ -140,6 +153,25 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
elif path in ('/async', '/asynchronous') and not is_synchronous:
|
elif path in ('/async', '/asynchronous') and not is_synchronous:
|
||||||
status_code = replica_status_code
|
status_code = replica_status_code
|
||||||
|
|
||||||
|
# check for user defined tags in query params
|
||||||
|
if not ignore_tags and status_code == 200:
|
||||||
|
qs_tag_prefix = "tag_"
|
||||||
|
for qs_key, qs_value in self.path_query.items():
|
||||||
|
if not qs_key.startswith(qs_tag_prefix):
|
||||||
|
continue
|
||||||
|
qs_key = qs_key[len(qs_tag_prefix):]
|
||||||
|
qs_value = qs_value[0]
|
||||||
|
instance_tag_value = patroni.tags.get(qs_key)
|
||||||
|
# tag not registered for instance
|
||||||
|
if instance_tag_value is None:
|
||||||
|
status_code = 503
|
||||||
|
break
|
||||||
|
if not isinstance(instance_tag_value, six.string_types):
|
||||||
|
instance_tag_value = str(instance_tag_value).lower()
|
||||||
|
if instance_tag_value != qs_value:
|
||||||
|
status_code = 503
|
||||||
|
break
|
||||||
|
|
||||||
if write_status_code_only: # when haproxy sends OPTIONS request it reads only status code and nothing more
|
if write_status_code_only: # when haproxy sends OPTIONS request it reads only status code and nothing more
|
||||||
self._write_status_code_only(status_code)
|
self._write_status_code_only(status_code)
|
||||||
else:
|
else:
|
||||||
@@ -180,6 +212,99 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
else:
|
else:
|
||||||
self.send_error(502)
|
self.send_error(502)
|
||||||
|
|
||||||
|
def do_GET_metrics(self):
|
||||||
|
postgres = self.get_postgresql_status(True)
|
||||||
|
patroni = self.server.patroni
|
||||||
|
epoch = datetime.datetime(1970, 1, 1, tzinfo=tzutc)
|
||||||
|
|
||||||
|
metrics = []
|
||||||
|
|
||||||
|
scope_label = '{{scope="{0}"}}'.format(patroni.postgresql.scope)
|
||||||
|
metrics.append("# HELP patroni_version Patroni semver without periods.")
|
||||||
|
metrics.append("# TYPE patroni_version gauge")
|
||||||
|
padded_semver = ''.join([x.zfill(2) for x in patroni.version.split('.')]) # 2.0.2 => 020002
|
||||||
|
metrics.append("patroni_version{0} {1}".format(scope_label, padded_semver))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_postgres_running Value is 1 if Postgres is running, 0 otherwise.")
|
||||||
|
metrics.append("# TYPE patroni_postgres_running gauge")
|
||||||
|
metrics.append("patroni_postgres_running{0} {1}".format(scope_label, int(postgres['state'] == 'running')))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_postmaster_start_time Epoch seconds since Postgres started.")
|
||||||
|
metrics.append("# TYPE patroni_postmaster_start_time gauge")
|
||||||
|
postmaster_start_time = postgres.get('postmaster_start_time')
|
||||||
|
postmaster_start_time = (postmaster_start_time - epoch).total_seconds() if postmaster_start_time else 0
|
||||||
|
metrics.append("patroni_postmaster_start_time{0} {1}".format(scope_label, postmaster_start_time))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_master Value is 1 if this node is the leader, 0 otherwise.")
|
||||||
|
metrics.append("# TYPE patroni_master gauge")
|
||||||
|
metrics.append("patroni_master{0} {1}".format(scope_label, int(postgres['role'] == 'master')))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_xlog_location Current location of the Postgres"
|
||||||
|
" transaction log, 0 if this node is not the leader.")
|
||||||
|
metrics.append("# TYPE patroni_xlog_location counter")
|
||||||
|
metrics.append("patroni_xlog_location{0} {1}".format(scope_label, postgres.get('xlog', {}).get('location', 0)))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_standby_leader Value is 1 if this node is the standby_leader, 0 otherwise.")
|
||||||
|
metrics.append("# TYPE patroni_standby_leader gauge")
|
||||||
|
metrics.append("patroni_standby_leader{0} {1}".format(scope_label, int(postgres['role'] == 'standby_leader')))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_replica Value is 1 if this node is a replica, 0 otherwise.")
|
||||||
|
metrics.append("# TYPE patroni_replica gauge")
|
||||||
|
metrics.append("patroni_replica{0} {1}".format(scope_label, int(postgres['role'] == 'replica')))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_xlog_received_location Current location of the received"
|
||||||
|
" Postgres transaction log, 0 if this node is not a replica.")
|
||||||
|
metrics.append("# TYPE patroni_xlog_received_location counter")
|
||||||
|
metrics.append("patroni_xlog_received_location{0} {1}"
|
||||||
|
.format(scope_label, postgres.get('xlog', {}).get('received_location', 0)))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_xlog_replayed_location Current location of the replayed"
|
||||||
|
" Postgres transaction log, 0 if this node is not a replica.")
|
||||||
|
metrics.append("# TYPE patroni_xlog_replayed_location counter")
|
||||||
|
metrics.append("patroni_xlog_replayed_location{0} {1}"
|
||||||
|
.format(scope_label, postgres.get('xlog', {}).get('replayed_location', 0)))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_xlog_replayed_timestamp Current timestamp of the replayed"
|
||||||
|
" Postgres transaction log, 0 if null.")
|
||||||
|
metrics.append("# TYPE patroni_xlog_replayed_timestamp gauge")
|
||||||
|
replayed_timestamp = postgres.get('xlog', {}).get('replayed_timestamp')
|
||||||
|
replayed_timestamp = (replayed_timestamp - epoch).total_seconds() if replayed_timestamp else 0
|
||||||
|
metrics.append("patroni_xlog_replayed_timestamp{0} {1}".format(scope_label, replayed_timestamp))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_xlog_paused Value is 1 if the Postgres xlog is paused, 0 otherwise.")
|
||||||
|
metrics.append("# TYPE patroni_xlog_paused gauge")
|
||||||
|
metrics.append("patroni_xlog_paused{0} {1}"
|
||||||
|
.format(scope_label, int(postgres.get('xlog', {}).get('paused', False) is True)))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_postgres_server_version Version of Postgres (if running), 0 otherwise.")
|
||||||
|
metrics.append("# TYPE patroni_postgres_server_version gauge")
|
||||||
|
metrics.append("patroni_postgres_server_version {0} {1}".format(scope_label, postgres.get('server_version', 0)))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_cluster_unlocked Value is 1 if the cluster is unlocked, 0 if locked.")
|
||||||
|
metrics.append("# TYPE patroni_cluster_unlocked gauge")
|
||||||
|
metrics.append("patroni_cluster_unlocked{0} {1}".format(scope_label, int(postgres.get('cluster_unlocked', 0))))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_postgres_timeline Postgres timeline of this node (if running), 0 otherwise.")
|
||||||
|
metrics.append("# TYPE patroni_postgres_timeline counter")
|
||||||
|
metrics.append("patroni_postgres_timeline{0} {1}".format(scope_label, postgres.get('timeline', 0)))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_dcs_last_seen Epoch timestamp when DCS was last contacted successfully"
|
||||||
|
" by Patroni.")
|
||||||
|
metrics.append("# TYPE patroni_dcs_last_seen gauge")
|
||||||
|
metrics.append("patroni_dcs_last_seen{0} {1}".format(scope_label, postgres.get('dcs_last_seen', 0)))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_pending_restart Value is 1 if the node needs a restart, 0 otherwise.")
|
||||||
|
metrics.append("# TYPE patroni_pending_restart gauge")
|
||||||
|
metrics.append("patroni_pending_restart{0} {1}"
|
||||||
|
.format(scope_label, int(patroni.postgresql.pending_restart)))
|
||||||
|
|
||||||
|
metrics.append("# HELP patroni_is_paused Value is 1 if auto failover is disabled, 0 otherwise.")
|
||||||
|
metrics.append("# TYPE patroni_is_paused gauge")
|
||||||
|
metrics.append("patroni_is_paused{0} {1}"
|
||||||
|
.format(scope_label, int(patroni.ha.is_paused())))
|
||||||
|
|
||||||
|
self._write_response(200, '\n'.join(metrics)+'\n', content_type='text/plain')
|
||||||
|
|
||||||
def _read_json_content(self, body_is_optional=False):
|
def _read_json_content(self, body_is_optional=False):
|
||||||
if 'content-length' not in self.headers:
|
if 'content-length' not in self.headers:
|
||||||
return self.send_error(411) if not body_is_optional else {}
|
return self.send_error(411) if not body_is_optional else {}
|
||||||
@@ -194,7 +319,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
logger.exception('Bad request')
|
logger.exception('Bad request')
|
||||||
self.send_error(400)
|
self.send_error(400)
|
||||||
|
|
||||||
@check_auth
|
@check_access
|
||||||
def do_PATCH_config(self):
|
def do_PATCH_config(self):
|
||||||
request = self._read_json_content()
|
request = self._read_json_content()
|
||||||
if request:
|
if request:
|
||||||
@@ -209,7 +334,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
self.server.patroni.ha.wakeup()
|
self.server.patroni.ha.wakeup()
|
||||||
self._write_json_response(200, data)
|
self._write_json_response(200, data)
|
||||||
|
|
||||||
@check_auth
|
@check_access
|
||||||
def do_PUT_config(self):
|
def do_PUT_config(self):
|
||||||
request = self._read_json_content()
|
request = self._read_json_content()
|
||||||
if request:
|
if request:
|
||||||
@@ -220,7 +345,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
return self.send_error(502)
|
return self.send_error(502)
|
||||||
self._write_json_response(200, request)
|
self._write_json_response(200, request)
|
||||||
|
|
||||||
@check_auth
|
@check_access
|
||||||
def do_POST_reload(self):
|
def do_POST_reload(self):
|
||||||
self.server.patroni.sighup_handler()
|
self.server.patroni.sighup_handler()
|
||||||
self._write_response(202, 'reload scheduled')
|
self._write_response(202, 'reload scheduled')
|
||||||
@@ -246,7 +371,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
status_code = 422
|
status_code = 422
|
||||||
return (status_code, error, scheduled_at)
|
return (status_code, error, scheduled_at)
|
||||||
|
|
||||||
@check_auth
|
@check_access
|
||||||
def do_POST_restart(self):
|
def do_POST_restart(self):
|
||||||
status_code = 500
|
status_code = 500
|
||||||
data = 'restart failed'
|
data = 'restart failed'
|
||||||
@@ -307,7 +432,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
status_code = 409
|
status_code = 409
|
||||||
self._write_response(status_code, data)
|
self._write_response(status_code, data)
|
||||||
|
|
||||||
@check_auth
|
@check_access
|
||||||
def do_DELETE_restart(self):
|
def do_DELETE_restart(self):
|
||||||
if self.server.patroni.ha.delete_future_restart():
|
if self.server.patroni.ha.delete_future_restart():
|
||||||
data = "scheduled restart deleted"
|
data = "scheduled restart deleted"
|
||||||
@@ -317,7 +442,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
code = 404
|
code = 404
|
||||||
self._write_response(code, data)
|
self._write_response(code, data)
|
||||||
|
|
||||||
@check_auth
|
@check_access
|
||||||
def do_DELETE_switchover(self):
|
def do_DELETE_switchover(self):
|
||||||
failover = self.server.patroni.dcs.get_cluster().failover
|
failover = self.server.patroni.dcs.get_cluster().failover
|
||||||
if failover and failover.scheduled_at:
|
if failover and failover.scheduled_at:
|
||||||
@@ -331,7 +456,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
code = 404
|
code = 404
|
||||||
self._write_response(code, data)
|
self._write_response(code, data)
|
||||||
|
|
||||||
@check_auth
|
@check_access
|
||||||
def do_POST_reinitialize(self):
|
def do_POST_reinitialize(self):
|
||||||
request = self._read_json_content(body_is_optional=True)
|
request = self._read_json_content(body_is_optional=True)
|
||||||
|
|
||||||
@@ -363,7 +488,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
if not cluster.failover:
|
if not cluster.failover:
|
||||||
return 503, action.title() + ' failed'
|
return 503, action.title() + ' failed'
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.debug('Exception occured during polling %s result: %s', action, e)
|
logger.debug('Exception occurred during polling %s result: %s', action, e)
|
||||||
return 503, action.title() + ' status unknown'
|
return 503, action.title() + ' status unknown'
|
||||||
|
|
||||||
def is_failover_possible(self, cluster, leader, candidate, action):
|
def is_failover_possible(self, cluster, leader, candidate, action):
|
||||||
@@ -388,7 +513,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
return None
|
return None
|
||||||
return action + ' is not possible: no good candidates have been found'
|
return action + ' is not possible: no good candidates have been found'
|
||||||
|
|
||||||
@check_auth
|
@check_access
|
||||||
def do_POST_failover(self, action='failover'):
|
def do_POST_failover(self, action='failover'):
|
||||||
request = self._read_json_content()
|
request = self._read_json_content()
|
||||||
(status_code, data) = (400, '')
|
(status_code, data) = (400, '')
|
||||||
@@ -476,7 +601,7 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
if postgresql.state not in ('running', 'restarting', 'starting'):
|
if postgresql.state not in ('running', 'restarting', 'starting'):
|
||||||
raise RetryFailedError('')
|
raise RetryFailedError('')
|
||||||
stmt = ("SELECT " + postgresql.POSTMASTER_START_TIME + ", " + postgresql.TL_LSN + ","
|
stmt = ("SELECT " + postgresql.POSTMASTER_START_TIME + ", " + postgresql.TL_LSN + ","
|
||||||
" pg_catalog.to_char(pg_catalog.pg_last_xact_replay_timestamp(), 'YYYY-MM-DD HH24:MI:SS.MS TZ'),"
|
" pg_catalog.pg_last_xact_replay_timestamp(),"
|
||||||
" pg_catalog.array_to_json(pg_catalog.array_agg(pg_catalog.row_to_json(ri))) "
|
" pg_catalog.array_to_json(pg_catalog.array_agg(pg_catalog.row_to_json(ri))) "
|
||||||
"FROM (SELECT (SELECT rolname FROM pg_authid WHERE oid = usesysid) AS usename,"
|
"FROM (SELECT (SELECT rolname FROM pg_authid WHERE oid = usesysid) AS usename,"
|
||||||
" application_name, client_addr, w.state, sync_state, sync_priority"
|
" application_name, client_addr, w.state, sync_state, sync_priority"
|
||||||
@@ -489,7 +614,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
'postmaster_start_time': row[0],
|
'postmaster_start_time': row[0],
|
||||||
'role': 'replica' if row[1] == 0 else 'master',
|
'role': 'replica' if row[1] == 0 else 'master',
|
||||||
'server_version': postgresql.server_version,
|
'server_version': postgresql.server_version,
|
||||||
'cluster_unlocked': bool(not cluster or cluster.is_unlocked()),
|
|
||||||
'xlog': ({
|
'xlog': ({
|
||||||
'received_location': row[4] or row[3],
|
'received_location': row[4] or row[3],
|
||||||
'replayed_location': row[3],
|
'replayed_location': row[3],
|
||||||
@@ -511,13 +635,17 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
|||||||
if row[7]:
|
if row[7]:
|
||||||
result['replication'] = row[7]
|
result['replication'] = row[7]
|
||||||
|
|
||||||
return result
|
except (psycopg.Error, RetryFailedError, PostgresConnectionException):
|
||||||
except (psycopg2.Error, RetryFailedError, PostgresConnectionException):
|
|
||||||
state = postgresql.state
|
state = postgresql.state
|
||||||
if state == 'running':
|
if state == 'running':
|
||||||
logger.exception('get_postgresql_status')
|
logger.exception('get_postgresql_status')
|
||||||
state = 'unknown'
|
state = 'unknown'
|
||||||
return {'state': state, 'role': postgresql.role}
|
result = {'state': state, 'role': postgresql.role}
|
||||||
|
|
||||||
|
if not cluster or cluster.is_unlocked():
|
||||||
|
result['cluster_unlocked'] = True
|
||||||
|
result['dcs_last_seen'] = self.server.patroni.dcs.last_seen
|
||||||
|
return result
|
||||||
|
|
||||||
def handle_one_request(self):
|
def handle_one_request(self):
|
||||||
self.__start_time = time.time()
|
self.__start_time = time.time()
|
||||||
@@ -536,7 +664,8 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
self.patroni = patroni
|
self.patroni = patroni
|
||||||
self.__listen = None
|
self.__listen = None
|
||||||
self.__ssl_options = None
|
self.__ssl_options = None
|
||||||
self.http_extra_headers = {}
|
self.__ssl_serial_number = None
|
||||||
|
self._received_new_cert = False
|
||||||
self.reload_config(config)
|
self.reload_config(config)
|
||||||
self.daemon = True
|
self.daemon = True
|
||||||
|
|
||||||
@@ -546,7 +675,7 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
with self.patroni.postgresql.connection().cursor() as cursor:
|
with self.patroni.postgresql.connection().cursor() as cursor:
|
||||||
cursor.execute(sql, params)
|
cursor.execute(sql, params)
|
||||||
return [r for r in cursor]
|
return [r for r in cursor]
|
||||||
except psycopg2.Error as e:
|
except psycopg.Error as e:
|
||||||
if cursor and cursor.connection.closed == 0:
|
if cursor and cursor.connection.closed == 0:
|
||||||
raise e
|
raise e
|
||||||
raise PostgresConnectionException('connection problems')
|
raise PostgresConnectionException('connection problems')
|
||||||
@@ -568,7 +697,35 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
if not auth_header.startswith('Basic ') or not self.check_basic_auth_key(auth_header[6:]):
|
if not auth_header.startswith('Basic ') or not self.check_basic_auth_key(auth_header[6:]):
|
||||||
return 'not authenticated'
|
return 'not authenticated'
|
||||||
|
|
||||||
def check_auth(self, rh):
|
@staticmethod
|
||||||
|
def __resolve_ips(host, port):
|
||||||
|
try:
|
||||||
|
for _, _, _, _, sa in socket.getaddrinfo(host, port, 0, socket.SOCK_STREAM, socket.IPPROTO_TCP):
|
||||||
|
yield ip_network(sa[0])
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Failed to resolve %s: %r', host, e)
|
||||||
|
|
||||||
|
def __members_ips(self):
|
||||||
|
cluster = self.patroni.dcs.cluster
|
||||||
|
if self.__allowlist_include_members and cluster:
|
||||||
|
for member in cluster.members:
|
||||||
|
if member.api_url:
|
||||||
|
try:
|
||||||
|
r = urlparse(member.api_url)
|
||||||
|
host = r.hostname
|
||||||
|
port = r.port or (443 if r.scheme == 'https' else 80)
|
||||||
|
for ip in self.__resolve_ips(host, port):
|
||||||
|
yield ip
|
||||||
|
except Exception as e:
|
||||||
|
logger.debug('Failed to parse url %s: %r', member.api_url, e)
|
||||||
|
|
||||||
|
def check_access(self, rh):
|
||||||
|
if self.__allowlist or self.__allowlist_include_members:
|
||||||
|
incoming_ip = rh.client_address[0]
|
||||||
|
incoming_ip = ip_address(incoming_ip.decode('utf-8') if six.PY2 else incoming_ip)
|
||||||
|
if not any(incoming_ip in net for net in self.__allowlist + tuple(self.__members_ips())):
|
||||||
|
return rh._write_response(403, 'Access is denied')
|
||||||
|
|
||||||
if not hasattr(rh.request, 'getpeercert') or not rh.request.getpeercert(): # valid client cert isn't present
|
if not hasattr(rh.request, 'getpeercert') or not rh.request.getpeercert(): # valid client cert isn't present
|
||||||
if self.__protocol == 'https' and self.__ssl_options.get('verify_client') in ('required', 'optional'):
|
if self.__protocol == 'https' and self.__ssl_options.get('verify_client') in ('required', 'optional'):
|
||||||
return rh._write_response(403, 'client certificate required')
|
return rh._write_response(403, 'client certificate required')
|
||||||
@@ -621,9 +778,12 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
reloading_config = self.__listen is not None # changing config in runtime
|
reloading_config = self.__listen is not None # changing config in runtime
|
||||||
if reloading_config:
|
if reloading_config:
|
||||||
self.shutdown()
|
self.shutdown()
|
||||||
|
# Rely on ThreadingMixIn.server_close() to have all requests terminate before we continue
|
||||||
|
self.server_close()
|
||||||
|
|
||||||
self.__listen = listen
|
self.__listen = listen
|
||||||
self.__ssl_options = ssl_options
|
self.__ssl_options = ssl_options
|
||||||
|
self._received_new_cert = False # reset to False after reload_config()
|
||||||
|
|
||||||
self.__httpserver_init(host, port)
|
self.__httpserver_init(host, port)
|
||||||
Thread.__init__(self, target=self.serve_forever)
|
Thread.__init__(self, target=self.serve_forever)
|
||||||
@@ -646,6 +806,7 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
ctx.verify_mode = modes[verify_client]
|
ctx.verify_mode = modes[verify_client]
|
||||||
else:
|
else:
|
||||||
logger.error('Bad value in the "restapi.verify_client": %s', verify_client)
|
logger.error('Bad value in the "restapi.verify_client": %s', verify_client)
|
||||||
|
self.__ssl_serial_number = self.get_certificate_serial_number()
|
||||||
self.socket = ctx.wrap_socket(self.socket, server_side=True)
|
self.socket = ctx.wrap_socket(self.socket, server_side=True)
|
||||||
if reloading_config:
|
if reloading_config:
|
||||||
self.start()
|
self.start()
|
||||||
@@ -673,10 +834,42 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
_, request = request # SSLSocket
|
_, request = request # SSLSocket
|
||||||
return super(RestApiServer, self).shutdown_request(request)
|
return super(RestApiServer, self).shutdown_request(request)
|
||||||
|
|
||||||
|
def get_certificate_serial_number(self):
|
||||||
|
if self.__ssl_options.get('certfile'):
|
||||||
|
import ssl
|
||||||
|
try:
|
||||||
|
crt = ssl._ssl._test_decode_cert(self.__ssl_options['certfile'])
|
||||||
|
return crt.get('serialNumber')
|
||||||
|
except ssl.SSLError as e:
|
||||||
|
logger.error('Failed to get serial number from certificate %s: %r', self.__ssl_options['certfile'], e)
|
||||||
|
|
||||||
|
def reload_local_certificate(self):
|
||||||
|
if self.__protocol == 'https':
|
||||||
|
on_disk_cert_serial_number = self.get_certificate_serial_number()
|
||||||
|
if on_disk_cert_serial_number != self.__ssl_serial_number:
|
||||||
|
self._received_new_cert = True
|
||||||
|
self.__ssl_serial_number = on_disk_cert_serial_number
|
||||||
|
return True
|
||||||
|
|
||||||
|
def _build_allowlist(self, value):
|
||||||
|
if isinstance(value, list):
|
||||||
|
for v in value:
|
||||||
|
if '/' in v: # netmask
|
||||||
|
try:
|
||||||
|
yield ip_network(v)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Invalid value "%s" in the allowlist: %r', v, e)
|
||||||
|
else: # ip or hostname, try to resolve it
|
||||||
|
for ip in self.__resolve_ips(v, 8080):
|
||||||
|
yield ip
|
||||||
|
|
||||||
def reload_config(self, config):
|
def reload_config(self, config):
|
||||||
if 'listen' not in config: # changing config in runtime
|
if 'listen' not in config: # changing config in runtime
|
||||||
raise ValueError('Can not find "restapi.listen" config')
|
raise ValueError('Can not find "restapi.listen" config')
|
||||||
|
|
||||||
|
self.__allowlist = tuple(self._build_allowlist(config.get('allowlist')))
|
||||||
|
self.__allowlist_include_members = config.get('allowlist_include_members')
|
||||||
|
|
||||||
ssl_options = {n: config[n] for n in ('certfile', 'keyfile', 'keyfile_password',
|
ssl_options = {n: config[n] for n in ('certfile', 'keyfile', 'keyfile_password',
|
||||||
'cafile', 'ciphers') if n in config}
|
'cafile', 'ciphers') if n in config}
|
||||||
|
|
||||||
@@ -686,7 +879,7 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
if isinstance(config.get('verify_client'), six.string_types):
|
if isinstance(config.get('verify_client'), six.string_types):
|
||||||
ssl_options['verify_client'] = config['verify_client'].lower()
|
ssl_options['verify_client'] = config['verify_client'].lower()
|
||||||
|
|
||||||
if self.__listen != config['listen'] or self.__ssl_options != ssl_options:
|
if self.__listen != config['listen'] or self.__ssl_options != ssl_options or self._received_new_cert:
|
||||||
self.__initialize(config['listen'], ssl_options)
|
self.__initialize(config['listen'], ssl_options)
|
||||||
|
|
||||||
self.__auth_key = base64.b64encode(config['auth'].encode('utf-8')) if 'auth' in config else None
|
self.__auth_key = base64.b64encode(config['auth'].encode('utf-8')) if 'auth' in config else None
|
||||||
@@ -694,6 +887,6 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
|||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def handle_error(request, client_address):
|
def handle_error(request, client_address):
|
||||||
address, port = client_address
|
logger.warning('Exception happened during processing of request from %s:%s',
|
||||||
logger.warning('Exception happened during processing of request from {}:{}'.format(address, port))
|
client_address[0], client_address[1])
|
||||||
logger.warning(traceback.format_exc())
|
logger.warning(traceback.format_exc())
|
||||||
|
|||||||
+46
-25
@@ -24,6 +24,7 @@ _AUTH_ALLOWED_PARAMETERS = (
|
|||||||
'sslpassword',
|
'sslpassword',
|
||||||
'sslrootcert',
|
'sslrootcert',
|
||||||
'sslcrl',
|
'sslcrl',
|
||||||
|
'sslcrldir',
|
||||||
'gssencmode',
|
'gssencmode',
|
||||||
'channel_binding'
|
'channel_binding'
|
||||||
)
|
)
|
||||||
@@ -232,7 +233,7 @@ class Config(object):
|
|||||||
for name, value in (value or {}).items():
|
for name, value in (value or {}).items():
|
||||||
if name in self.__DEFAULT_CONFIG['standby_cluster']:
|
if name in self.__DEFAULT_CONFIG['standby_cluster']:
|
||||||
config['standby_cluster'][name] = deepcopy(value)
|
config['standby_cluster'][name] = deepcopy(value)
|
||||||
elif name in config: # only variables present in __DEFAULT_CONFIG allowed to be overriden from DCS
|
elif name in config: # only variables present in __DEFAULT_CONFIG allowed to be overridden from DCS
|
||||||
if name in ('synchronous_mode', 'synchronous_mode_strict'):
|
if name in ('synchronous_mode', 'synchronous_mode_strict'):
|
||||||
config[name] = value
|
config[name] = value
|
||||||
else:
|
else:
|
||||||
@@ -268,11 +269,42 @@ class Config(object):
|
|||||||
|
|
||||||
_set_section_values('restapi', ['listen', 'connect_address', 'certfile', 'keyfile', 'keyfile_password',
|
_set_section_values('restapi', ['listen', 'connect_address', 'certfile', 'keyfile', 'keyfile_password',
|
||||||
'cafile', 'ciphers', 'verify_client', 'http_extra_headers',
|
'cafile', 'ciphers', 'verify_client', 'http_extra_headers',
|
||||||
'https_extra_headers'])
|
'https_extra_headers', 'allowlist', 'allowlist_include_members'])
|
||||||
_set_section_values('ctl', ['insecure', 'cacert', 'certfile', 'keyfile'])
|
_set_section_values('ctl', ['insecure', 'cacert', 'certfile', 'keyfile', 'keyfile_password'])
|
||||||
_set_section_values('postgresql', ['listen', 'connect_address', 'config_dir', 'data_dir', 'pgpass', 'bin_dir'])
|
_set_section_values('postgresql', ['listen', 'connect_address', 'config_dir', 'data_dir', 'pgpass', 'bin_dir'])
|
||||||
_set_section_values('log', ['level', 'traceback_level', 'format', 'dateformat', 'max_queue_size',
|
_set_section_values('log', ['level', 'traceback_level', 'format', 'dateformat', 'max_queue_size',
|
||||||
'dir', 'file_size', 'file_num', 'loggers'])
|
'dir', 'file_size', 'file_num', 'loggers'])
|
||||||
|
_set_section_values('raft', ['data_dir', 'self_addr', 'partner_addrs', 'password', 'bind_addr'])
|
||||||
|
|
||||||
|
for first, second in (('restapi', 'allowlist_include_members'), ('ctl', 'insecure')):
|
||||||
|
value = ret.get(first, {}).pop(second, None)
|
||||||
|
if value:
|
||||||
|
value = parse_bool(value)
|
||||||
|
if value is not None:
|
||||||
|
ret[first][second] = value
|
||||||
|
|
||||||
|
for second in ('max_queue_size', 'file_size', 'file_num'):
|
||||||
|
value = ret.get('log', {}).pop(second, None)
|
||||||
|
if value:
|
||||||
|
value = parse_int(value)
|
||||||
|
if value is not None:
|
||||||
|
ret['log'][second] = value
|
||||||
|
|
||||||
|
def _parse_list(value):
|
||||||
|
if not (value.strip().startswith('-') or '[' in value):
|
||||||
|
value = '[{0}]'.format(value)
|
||||||
|
try:
|
||||||
|
return yaml.safe_load(value)
|
||||||
|
except Exception:
|
||||||
|
logger.exception('Exception when parsing list %s', value)
|
||||||
|
return None
|
||||||
|
|
||||||
|
for first, second in (('raft', 'partner_addrs'), ('restapi', 'allowlist')):
|
||||||
|
value = ret.get(first, {}).pop(second, None)
|
||||||
|
if value:
|
||||||
|
value = _parse_list(value)
|
||||||
|
if value:
|
||||||
|
ret[first][second] = value
|
||||||
|
|
||||||
def _parse_dict(value):
|
def _parse_dict(value):
|
||||||
if not value.strip().startswith('{'):
|
if not value.strip().startswith('{'):
|
||||||
@@ -283,11 +315,13 @@ class Config(object):
|
|||||||
logger.exception('Exception when parsing dict %s', value)
|
logger.exception('Exception when parsing dict %s', value)
|
||||||
return None
|
return None
|
||||||
|
|
||||||
value = ret.get('log', {}).pop('loggers', None)
|
for first, params in (('restapi', ('http_extra_headers', 'https_extra_headers')), ('log', ('loggers',))):
|
||||||
if value:
|
for second in params:
|
||||||
value = _parse_dict(value)
|
value = ret.get(first, {}).pop(second, None)
|
||||||
if value:
|
if value:
|
||||||
ret['log']['loggers'] = value
|
value = _parse_dict(value)
|
||||||
|
if value:
|
||||||
|
ret[first][second] = value
|
||||||
|
|
||||||
def _get_auth(name, params=None):
|
def _get_auth(name, params=None):
|
||||||
ret = {}
|
ret = {}
|
||||||
@@ -310,36 +344,23 @@ class Config(object):
|
|||||||
if authentication:
|
if authentication:
|
||||||
ret['postgresql']['authentication'] = authentication
|
ret['postgresql']['authentication'] = authentication
|
||||||
|
|
||||||
def _parse_list(value):
|
|
||||||
if not (value.strip().startswith('-') or '[' in value):
|
|
||||||
value = '[{0}]'.format(value)
|
|
||||||
try:
|
|
||||||
return yaml.safe_load(value)
|
|
||||||
except Exception:
|
|
||||||
logger.exception('Exception when parsing list %s', value)
|
|
||||||
return None
|
|
||||||
|
|
||||||
_set_section_values('raft', ['data_dir', 'self_addr', 'partner_addrs', 'password', 'bind_addr'])
|
|
||||||
if 'raft' in ret and 'partner_addrs' in ret['raft']:
|
|
||||||
ret['raft']['partner_addrs'] = _parse_list(ret['raft']['partner_addrs'])
|
|
||||||
|
|
||||||
for param in list(os.environ.keys()):
|
for param in list(os.environ.keys()):
|
||||||
if param.startswith(PATRONI_ENV_PREFIX):
|
if param.startswith(PATRONI_ENV_PREFIX):
|
||||||
# PATRONI_(ETCD|CONSUL|ZOOKEEPER|EXHIBITOR|...)_(HOSTS?|PORT|..)
|
# PATRONI_(ETCD|CONSUL|ZOOKEEPER|EXHIBITOR|...)_(HOSTS?|PORT|..)
|
||||||
name, suffix = (param[8:].split('_', 1) + [''])[:2]
|
name, suffix = (param[8:].split('_', 1) + [''])[:2]
|
||||||
if suffix in ('HOST', 'HOSTS', 'PORT', 'USE_PROXIES', 'PROTOCOL', 'SRV', 'URL', 'PROXY',
|
if suffix in ('HOST', 'HOSTS', 'PORT', 'USE_PROXIES', 'PROTOCOL', 'SRV', 'SRV_SUFFIX', 'URL', 'PROXY',
|
||||||
'CACERT', 'CERT', 'KEY', 'VERIFY', 'TOKEN', 'CHECKS', 'DC', 'CONSISTENCY',
|
'CACERT', 'CERT', 'KEY', 'VERIFY', 'TOKEN', 'CHECKS', 'DC', 'CONSISTENCY',
|
||||||
'REGISTER_SERVICE', 'SERVICE_CHECK_INTERVAL', 'NAMESPACE', 'CONTEXT',
|
'REGISTER_SERVICE', 'SERVICE_CHECK_INTERVAL', 'NAMESPACE', 'CONTEXT',
|
||||||
'USE_ENDPOINTS', 'SCOPE_LABEL', 'ROLE_LABEL', 'POD_IP', 'PORTS', 'LABELS',
|
'USE_ENDPOINTS', 'SCOPE_LABEL', 'ROLE_LABEL', 'POD_IP', 'PORTS', 'LABELS',
|
||||||
'BYPASS_API_SERVICE', 'KEY_PASSWORD', 'USE_SSL') and name:
|
'BYPASS_API_SERVICE', 'KEY_PASSWORD', 'USE_SSL', 'SET_ACLS') and name:
|
||||||
value = os.environ.pop(param)
|
value = os.environ.pop(param)
|
||||||
if suffix == 'PORT':
|
if suffix == 'PORT':
|
||||||
value = value and parse_int(value)
|
value = value and parse_int(value)
|
||||||
elif suffix in ('HOSTS', 'PORTS', 'CHECKS'):
|
elif suffix in ('HOSTS', 'PORTS', 'CHECKS'):
|
||||||
value = value and _parse_list(value)
|
value = value and _parse_list(value)
|
||||||
elif suffix == 'LABELS':
|
elif suffix in ('LABELS', 'SET_ACLS'):
|
||||||
value = _parse_dict(value)
|
value = _parse_dict(value)
|
||||||
elif suffix in ('USE_PROXIES', 'REGISTER_SERVICE', 'USE_ENDPOINTS', 'BYPASS_API_SERVICE'):
|
elif suffix in ('USE_PROXIES', 'REGISTER_SERVICE', 'USE_ENDPOINTS', 'BYPASS_API_SERVICE', 'VERIFY'):
|
||||||
value = parse_bool(value)
|
value = parse_bool(value)
|
||||||
if value:
|
if value:
|
||||||
ret[name.lower()][suffix.lower()] = value
|
ret[name.lower()][suffix.lower()] = value
|
||||||
|
|||||||
+16
-13
@@ -264,13 +264,13 @@ def get_cursor(cluster, connect_parameters, role='master', member=None):
|
|||||||
|
|
||||||
params = member.conn_kwargs(connect_parameters)
|
params = member.conn_kwargs(connect_parameters)
|
||||||
params.update({'fallback_application_name': 'Patroni ctl', 'connect_timeout': '5'})
|
params.update({'fallback_application_name': 'Patroni ctl', 'connect_timeout': '5'})
|
||||||
if 'database' in connect_parameters:
|
if 'dbname' in connect_parameters:
|
||||||
params['database'] = connect_parameters['database']
|
params['dbname'] = connect_parameters['dbname']
|
||||||
else:
|
else:
|
||||||
params.pop('database')
|
params.pop('dbname')
|
||||||
|
|
||||||
import psycopg2
|
from . import psycopg
|
||||||
conn = psycopg2.connect(**params)
|
conn = psycopg.connect(**params)
|
||||||
conn.autocommit = True
|
conn.autocommit = True
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
if role == 'any':
|
if role == 'any':
|
||||||
@@ -401,7 +401,7 @@ def query(
|
|||||||
if password:
|
if password:
|
||||||
connect_parameters['password'] = click.prompt('Password', hide_input=True, type=str)
|
connect_parameters['password'] = click.prompt('Password', hide_input=True, type=str)
|
||||||
if dbname:
|
if dbname:
|
||||||
connect_parameters['database'] = dbname
|
connect_parameters['dbname'] = dbname
|
||||||
|
|
||||||
if p_file is not None:
|
if p_file is not None:
|
||||||
command = p_file.read()
|
command = p_file.read()
|
||||||
@@ -418,7 +418,7 @@ def query(
|
|||||||
|
|
||||||
|
|
||||||
def query_member(cluster, cursor, member, role, command, connect_parameters):
|
def query_member(cluster, cursor, member, role, command, connect_parameters):
|
||||||
import psycopg2
|
from . import psycopg
|
||||||
try:
|
try:
|
||||||
if cursor is None:
|
if cursor is None:
|
||||||
cursor = get_cursor(cluster, connect_parameters, role=role, member=member)
|
cursor = get_cursor(cluster, connect_parameters, role=role, member=member)
|
||||||
@@ -433,11 +433,11 @@ def query_member(cluster, cursor, member, role, command, connect_parameters):
|
|||||||
|
|
||||||
cursor.execute(command)
|
cursor.execute(command)
|
||||||
return cursor.fetchall(), [d.name for d in cursor.description]
|
return cursor.fetchall(), [d.name for d in cursor.description]
|
||||||
except (psycopg2.OperationalError, psycopg2.DatabaseError) as oe:
|
except psycopg.DatabaseError as de:
|
||||||
logging.debug(oe)
|
logging.debug(de)
|
||||||
if cursor is not None and not cursor.connection.closed:
|
if cursor is not None and not cursor.connection.closed:
|
||||||
cursor.connection.close()
|
cursor.connection.close()
|
||||||
message = oe.pgcode or oe.pgerror or str(oe)
|
message = de.diag.sqlstate or str(de)
|
||||||
message = message.replace('\n', ' ')
|
message = message.replace('\n', ' ')
|
||||||
return [[timestamp(0), 'ERROR, SQLSTATE: {0}'.format(message)]], None
|
return [[timestamp(0), 'ERROR, SQLSTATE: {0}'.format(message)]], None
|
||||||
|
|
||||||
@@ -1299,10 +1299,13 @@ def version(obj, cluster_name, member_names):
|
|||||||
def history(obj, cluster_name, fmt):
|
def history(obj, cluster_name, fmt):
|
||||||
cluster = get_dcs(obj, cluster_name).get_cluster()
|
cluster = get_dcs(obj, cluster_name).get_cluster()
|
||||||
history = cluster.history and cluster.history.lines or []
|
history = cluster.history and cluster.history.lines or []
|
||||||
|
table_header_row = ['TL', 'LSN', 'Reason', 'Timestamp', 'New Leader']
|
||||||
for line in history:
|
for line in history:
|
||||||
if len(line) < 4:
|
if len(line) < len(table_header_row):
|
||||||
line.append('')
|
add_column_num = len(table_header_row) - len(line)
|
||||||
print_output(['TL', 'LSN', 'Reason', 'Timestamp'], history, {'TL': 'r', 'LSN': 'r'}, fmt)
|
for _ in range(add_column_num):
|
||||||
|
line.append('')
|
||||||
|
print_output(table_header_row, history, {'TL': 'r', 'LSN': 'r'}, fmt)
|
||||||
|
|
||||||
|
|
||||||
def format_pg_version(version):
|
def format_pg_version(version):
|
||||||
|
|||||||
+154
-45
@@ -13,12 +13,13 @@ import time
|
|||||||
|
|
||||||
from collections import defaultdict, namedtuple
|
from collections import defaultdict, namedtuple
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
from patroni.exceptions import PatroniFatalException
|
|
||||||
from patroni.utils import parse_bool, uri
|
|
||||||
from random import randint
|
from random import randint
|
||||||
from six.moves.urllib_parse import urlparse, urlunparse, parse_qsl
|
from six.moves.urllib_parse import urlparse, urlunparse, parse_qsl
|
||||||
from threading import Event, Lock
|
from threading import Event, Lock
|
||||||
|
|
||||||
|
from ..exceptions import PatroniFatalException
|
||||||
|
from ..utils import deep_compare, parse_bool, uri
|
||||||
|
|
||||||
slot_name_re = re.compile('^[a-z0-9_]{1,63}$')
|
slot_name_re = re.compile('^[a-z0-9_]{1,63}$')
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -67,7 +68,11 @@ def dcs_modules():
|
|||||||
|
|
||||||
if getattr(sys, 'frozen', False):
|
if getattr(sys, 'frozen', False):
|
||||||
toc = set()
|
toc = set()
|
||||||
for importer in pkgutil.iter_importers(dcs_dirname):
|
# dcs_dirname may contain a dot, which causes pkgutil.iter_importers()
|
||||||
|
# to misinterpret the path as a package name. This can be avoided
|
||||||
|
# altogether by not passing a path at all, because PyInstaller's
|
||||||
|
# FrozenImporter is a singleton and registered as top-level finder.
|
||||||
|
for importer in pkgutil.iter_importers():
|
||||||
if hasattr(importer, 'toc'):
|
if hasattr(importer, 'toc'):
|
||||||
toc |= importer.toc
|
toc |= importer.toc
|
||||||
return [module for module in toc if module.startswith(module_prefix) and module.count('.') == 2]
|
return [module for module in toc if module.startswith(module_prefix) and module.count('.') == 2]
|
||||||
@@ -133,6 +138,8 @@ class Member(namedtuple('Member', 'index,name,session,data')):
|
|||||||
else:
|
else:
|
||||||
try:
|
try:
|
||||||
data = json.loads(data)
|
data = json.loads(data)
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
data = {}
|
||||||
except (TypeError, ValueError):
|
except (TypeError, ValueError):
|
||||||
data = {}
|
data = {}
|
||||||
return Member(index, name, session, data)
|
return Member(index, name, session, data)
|
||||||
@@ -153,7 +160,7 @@ class Member(namedtuple('Member', 'index,name,session,data')):
|
|||||||
defaults = {
|
defaults = {
|
||||||
"host": None,
|
"host": None,
|
||||||
"port": None,
|
"port": None,
|
||||||
"database": None
|
"dbname": None
|
||||||
}
|
}
|
||||||
ret = self.data.get('conn_kwargs')
|
ret = self.data.get('conn_kwargs')
|
||||||
if ret:
|
if ret:
|
||||||
@@ -167,7 +174,7 @@ class Member(namedtuple('Member', 'index,name,session,data')):
|
|||||||
ret = {
|
ret = {
|
||||||
'host': r.hostname,
|
'host': r.hostname,
|
||||||
'port': r.port or 5432,
|
'port': r.port or 5432,
|
||||||
'database': r.path[1:]
|
'dbname': r.path[1:]
|
||||||
}
|
}
|
||||||
self.data['conn_kwargs'] = ret.copy()
|
self.data['conn_kwargs'] = ret.copy()
|
||||||
|
|
||||||
@@ -206,6 +213,15 @@ class Member(namedtuple('Member', 'index,name,session,data')):
|
|||||||
def is_running(self):
|
def is_running(self):
|
||||||
return self.state == 'running'
|
return self.state == 'running'
|
||||||
|
|
||||||
|
@property
|
||||||
|
def version(self):
|
||||||
|
version = self.data.get('version')
|
||||||
|
if version:
|
||||||
|
try:
|
||||||
|
return tuple(map(int, version.split('.')))
|
||||||
|
except Exception:
|
||||||
|
logger.debug('Failed to parse Patroni version %s', version)
|
||||||
|
|
||||||
|
|
||||||
class RemoteMember(Member):
|
class RemoteMember(Member):
|
||||||
""" Represents a remote master for a standby cluster
|
""" Represents a remote master for a standby cluster
|
||||||
@@ -259,14 +275,10 @@ class Leader(namedtuple('Leader', 'index,session,member')):
|
|||||||
"""
|
"""
|
||||||
>>> Leader(1, '', Member.from_node(1, '', '', '{"version":"z"}')).checkpoint_after_promote
|
>>> Leader(1, '', Member.from_node(1, '', '', '{"version":"z"}')).checkpoint_after_promote
|
||||||
"""
|
"""
|
||||||
version = self.data.get('version')
|
version = self.member.version
|
||||||
if version:
|
# 1.5.6 is the last version which doesn't expose checkpoint_after_promote: false
|
||||||
try:
|
if version and version > (1, 5, 6):
|
||||||
# 1.5.6 is the last version which doesn't expose checkpoint_after_promote: false
|
return self.data.get('role') == 'master' and 'checkpoint_after_promote' not in self.data
|
||||||
if tuple(map(int, version.split('.'))) > (1, 5, 6):
|
|
||||||
return self.data['role'] == 'master' and 'checkpoint_after_promote' not in self.data
|
|
||||||
except Exception:
|
|
||||||
logger.debug('Failed to parse Patroni version %s', version)
|
|
||||||
|
|
||||||
|
|
||||||
class Failover(namedtuple('Failover', 'index,leader,candidate,scheduled_at')):
|
class Failover(namedtuple('Failover', 'index,leader,candidate,scheduled_at')):
|
||||||
@@ -432,23 +444,28 @@ class TimelineHistory(namedtuple('TimelineHistory', 'index,value,lines')):
|
|||||||
return TimelineHistory(index, value, lines)
|
return TimelineHistory(index, value, lines)
|
||||||
|
|
||||||
|
|
||||||
class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_leader_operation,members,failover,sync,history')):
|
class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_lsn,members,failover,sync,history,slots')):
|
||||||
|
|
||||||
"""Immutable object (namedtuple) which represents PostgreSQL cluster.
|
"""Immutable object (namedtuple) which represents PostgreSQL cluster.
|
||||||
Consists of the following fields:
|
Consists of the following fields:
|
||||||
:param initialize: shows whether this cluster has initialization key stored in DC or not.
|
:param initialize: shows whether this cluster has initialization key stored in DC or not.
|
||||||
:param config: global dynamic configuration, reference to `ClusterConfig` object
|
:param config: global dynamic configuration, reference to `ClusterConfig` object
|
||||||
:param leader: `Leader` object which represents current leader of the cluster
|
:param leader: `Leader` object which represents current leader of the cluster
|
||||||
:param last_leader_operation: int or long object containing position of last known leader operation.
|
:param last_lsn: int or long object containing position of last known leader LSN.
|
||||||
This value is stored in `/optime/leader` key
|
This value is stored in the `/status` key or `/optime/leader` (legacy) key
|
||||||
:param members: list of Member object, all PostgreSQL cluster members including leader
|
:param members: list of Member object, all PostgreSQL cluster members including leader
|
||||||
:param failover: reference to `Failover` object
|
:param failover: reference to `Failover` object
|
||||||
:param sync: reference to `SyncState` object, last observed synchronous replication state.
|
:param sync: reference to `SyncState` object, last observed synchronous replication state.
|
||||||
:param history: reference to `TimelineHistory` object
|
:param history: reference to `TimelineHistory` object
|
||||||
|
:param slots: state of permanent logical replication slots on the primary in the format: {"slot_name": int}
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
@property
|
||||||
|
def leader_name(self):
|
||||||
|
return self.leader and self.leader.name
|
||||||
|
|
||||||
def is_unlocked(self):
|
def is_unlocked(self):
|
||||||
return not (self.leader and self.leader.name)
|
return not self.leader_name
|
||||||
|
|
||||||
def has_member(self, member_name):
|
def has_member(self, member_name):
|
||||||
return any(m for m in self.members if m.name == member_name)
|
return any(m for m in self.members if m.name == member_name)
|
||||||
@@ -470,22 +487,41 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_leader_operat
|
|||||||
def is_synchronous_mode(self):
|
def is_synchronous_mode(self):
|
||||||
return self.check_mode('synchronous_mode')
|
return self.check_mode('synchronous_mode')
|
||||||
|
|
||||||
def get_replication_slots(self, my_name, role):
|
@property
|
||||||
|
def __permanent_slots(self):
|
||||||
|
return self.config and self.config.permanent_slots or {}
|
||||||
|
|
||||||
|
@property
|
||||||
|
def __permanent_physical_slots(self):
|
||||||
|
return {name: value for name, value in self.__permanent_slots.items()
|
||||||
|
if not value or isinstance(value, dict) and value.get('type', 'physical') == 'physical'}
|
||||||
|
|
||||||
|
@property
|
||||||
|
def __permanent_logical_slots(self):
|
||||||
|
return {name: value for name, value in self.__permanent_slots.items() if isinstance(value, dict)
|
||||||
|
and value.get('type', 'logical') == 'logical' and value.get('database') and value.get('plugin')}
|
||||||
|
|
||||||
|
@property
|
||||||
|
def use_slots(self):
|
||||||
|
return self.config and (self.config.data.get('postgresql') or {}).get('use_slots', True)
|
||||||
|
|
||||||
|
def get_replication_slots(self, my_name, role, nofailover, major_version, show_error=False):
|
||||||
# if the replicatefrom tag is set on the member - we should not create the replication slot for it on
|
# if the replicatefrom tag is set on the member - we should not create the replication slot for it on
|
||||||
# the current master, because that member would replicate from elsewhere. We still create the slot if
|
# the current master, because that member would replicate from elsewhere. We still create the slot if
|
||||||
# the replicatefrom destination member is currently not a member of the cluster (fallback to the
|
# the replicatefrom destination member is currently not a member of the cluster (fallback to the
|
||||||
# master), or if replicatefrom destination member happens to be the current master
|
# master), or if replicatefrom destination member happens to be the current master
|
||||||
use_slots = self.config and self.config.data.get('postgresql', {}).get('use_slots', True)
|
use_slots = self.use_slots
|
||||||
if role in ('master', 'standby_leader'):
|
if role in ('master', 'standby_leader'):
|
||||||
slot_members = [m.name for m in self.members if use_slots and m.name != my_name and
|
slot_members = [m.name for m in self.members if use_slots and m.name != my_name and
|
||||||
(m.replicatefrom is None or m.replicatefrom == my_name or
|
(m.replicatefrom is None or m.replicatefrom == my_name or
|
||||||
not self.has_member(m.replicatefrom))]
|
not self.has_member(m.replicatefrom))]
|
||||||
permanent_slots = (self.config and self.config.permanent_slots or {}).copy()
|
permanent_slots = self.__permanent_slots if use_slots and \
|
||||||
|
role == 'master' else self.__permanent_physical_slots
|
||||||
else:
|
else:
|
||||||
# only manage slots for replicas that replicate from this one, except for the leader among them
|
# only manage slots for replicas that replicate from this one, except for the leader among them
|
||||||
slot_members = [m.name for m in self.members if use_slots and
|
slot_members = [m.name for m in self.members if use_slots and
|
||||||
m.replicatefrom == my_name and m.name != self.leader.name]
|
m.replicatefrom == my_name and m.name != self.leader_name]
|
||||||
permanent_slots = {}
|
permanent_slots = self.__permanent_logical_slots if use_slots and not nofailover else {}
|
||||||
|
|
||||||
slots = {slot_name_from_member_name(name): {'type': 'physical'} for name in slot_members}
|
slots = {slot_name_from_member_name(name): {'type': 'physical'} for name in slot_members}
|
||||||
|
|
||||||
@@ -499,6 +535,7 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_leader_operat
|
|||||||
for k, v in slot_conflicts.items() if len(v) > 1))
|
for k, v in slot_conflicts.items() if len(v) > 1))
|
||||||
|
|
||||||
# "merge" replication slots for members with permanent_replication_slots
|
# "merge" replication slots for members with permanent_replication_slots
|
||||||
|
disabled_permanent_logical_slots = []
|
||||||
for name, value in permanent_slots.items():
|
for name, value in permanent_slots.items():
|
||||||
if not slot_name_re.match(name):
|
if not slot_name_re.match(name):
|
||||||
logger.error("Invalid permanent replication slot name '%s'", name)
|
logger.error("Invalid permanent replication slot name '%s'", name)
|
||||||
@@ -516,7 +553,9 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_leader_operat
|
|||||||
slots[name] = value
|
slots[name] = value
|
||||||
continue
|
continue
|
||||||
elif value['type'] == 'logical' and value.get('database') and value.get('plugin'):
|
elif value['type'] == 'logical' and value.get('database') and value.get('plugin'):
|
||||||
if name in slots:
|
if major_version < 110000:
|
||||||
|
disabled_permanent_logical_slots.append(name)
|
||||||
|
elif name in slots:
|
||||||
logger.error("Permanent logical replication slot {'%s': %s} is conflicting with" +
|
logger.error("Permanent logical replication slot {'%s': %s} is conflicting with" +
|
||||||
" physical replication slot for cluster member", name, value)
|
" physical replication slot for cluster member", name, value)
|
||||||
else:
|
else:
|
||||||
@@ -525,20 +564,53 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_leader_operat
|
|||||||
|
|
||||||
logger.error("Bad value for slot '%s' in permanent_slots: %s", name, permanent_slots[name])
|
logger.error("Bad value for slot '%s' in permanent_slots: %s", name, permanent_slots[name])
|
||||||
|
|
||||||
|
if disabled_permanent_logical_slots and show_error:
|
||||||
|
logger.error("Permanent logical replication slots supported by Patroni only starting from PostgreSQL 11. "
|
||||||
|
"Following slots will not be created: %s.", disabled_permanent_logical_slots)
|
||||||
|
|
||||||
return slots
|
return slots
|
||||||
|
|
||||||
def has_permanent_logical_slots(self, name):
|
def has_permanent_logical_slots(self, my_name, nofailover, major_version=110000):
|
||||||
slots = self.get_replication_slots(name, 'master').values()
|
if major_version < 110000:
|
||||||
|
return False
|
||||||
|
slots = self.get_replication_slots(my_name, 'replica', nofailover, major_version).values()
|
||||||
return any(v for v in slots if v.get("type") == "logical")
|
return any(v for v in slots if v.get("type") == "logical")
|
||||||
|
|
||||||
|
def should_enforce_hot_standby_feedback(self, my_name, nofailover, major_version):
|
||||||
|
"""
|
||||||
|
The hot_standby_feedback must be enabled if the current replica has logical slots
|
||||||
|
or it is working as a cascading replica for the other node that has logical slots.
|
||||||
|
"""
|
||||||
|
|
||||||
|
if major_version < 110000:
|
||||||
|
return False
|
||||||
|
|
||||||
|
if self.has_permanent_logical_slots(my_name, nofailover, major_version):
|
||||||
|
return True
|
||||||
|
|
||||||
|
if self.use_slots:
|
||||||
|
members = [m for m in self.members if m.replicatefrom == my_name and m.name != self.leader_name]
|
||||||
|
return any(self.should_enforce_hot_standby_feedback(m.name, m.nofailover, major_version) for m in members)
|
||||||
|
return False
|
||||||
|
|
||||||
|
def get_my_slot_name_on_primary(self, my_name, replicatefrom):
|
||||||
|
"""
|
||||||
|
P <-- I <-- L
|
||||||
|
In case of cascading replication we have to check not our physical slot,
|
||||||
|
but slot of the replica that connects us to the primary.
|
||||||
|
"""
|
||||||
|
|
||||||
|
m = self.get_member(replicatefrom, False) if replicatefrom else None
|
||||||
|
return self.get_my_slot_name_on_primary(m.name, m.replicatefrom) if m else slot_name_from_member_name(my_name)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def timeline(self):
|
def timeline(self):
|
||||||
"""
|
"""
|
||||||
>>> Cluster(0, 0, 0, 0, 0, 0, 0, 0).timeline
|
>>> Cluster(0, 0, 0, 0, 0, 0, 0, 0, 0).timeline
|
||||||
0
|
0
|
||||||
>>> Cluster(0, 0, 0, 0, 0, 0, 0, TimelineHistory.from_node(1, '[]')).timeline
|
>>> Cluster(0, 0, 0, 0, 0, 0, 0, TimelineHistory.from_node(1, '[]'), 0).timeline
|
||||||
1
|
1
|
||||||
>>> Cluster(0, 0, 0, 0, 0, 0, 0, TimelineHistory.from_node(1, '[["a"]]')).timeline
|
>>> Cluster(0, 0, 0, 0, 0, 0, 0, TimelineHistory.from_node(1, '[["a"]]'), 0).timeline
|
||||||
0
|
0
|
||||||
"""
|
"""
|
||||||
if self.history:
|
if self.history:
|
||||||
@@ -551,6 +623,10 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_leader_operat
|
|||||||
return 1
|
return 1
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
@property
|
||||||
|
def min_version(self):
|
||||||
|
return next(iter(sorted(filter(lambda v: v, [m.version for m in self.members])) + [None]))
|
||||||
|
|
||||||
|
|
||||||
@six.add_metaclass(abc.ABCMeta)
|
@six.add_metaclass(abc.ABCMeta)
|
||||||
class AbstractDCS(object):
|
class AbstractDCS(object):
|
||||||
@@ -562,7 +638,8 @@ class AbstractDCS(object):
|
|||||||
_HISTORY = 'history'
|
_HISTORY = 'history'
|
||||||
_MEMBERS = 'members/'
|
_MEMBERS = 'members/'
|
||||||
_OPTIME = 'optime'
|
_OPTIME = 'optime'
|
||||||
_LEADER_OPTIME = _OPTIME + '/' + _LEADER
|
_STATUS = 'status' # JSON, contains "leader_lsn" and confirmed_flush_lsn of logical "slots" on the leader
|
||||||
|
_LEADER_OPTIME = _OPTIME + '/' + _LEADER # legacy
|
||||||
_SYNC = 'sync'
|
_SYNC = 'sync'
|
||||||
|
|
||||||
def __init__(self, config):
|
def __init__(self, config):
|
||||||
@@ -578,7 +655,9 @@ class AbstractDCS(object):
|
|||||||
self._cluster = None
|
self._cluster = None
|
||||||
self._cluster_valid_till = 0
|
self._cluster_valid_till = 0
|
||||||
self._cluster_thread_lock = Lock()
|
self._cluster_thread_lock = Lock()
|
||||||
self._last_leader_operation = ''
|
self._last_lsn = ''
|
||||||
|
self._last_seen = 0
|
||||||
|
self._last_status = {}
|
||||||
self.event = Event()
|
self.event = Event()
|
||||||
|
|
||||||
def client_path(self, path):
|
def client_path(self, path):
|
||||||
@@ -612,6 +691,10 @@ class AbstractDCS(object):
|
|||||||
def history_path(self):
|
def history_path(self):
|
||||||
return self.client_path(self._HISTORY)
|
return self.client_path(self._HISTORY)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def status_path(self):
|
||||||
|
return self.client_path(self._STATUS)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def leader_optime_path(self):
|
def leader_optime_path(self):
|
||||||
return self.client_path(self._LEADER_OPTIME)
|
return self.client_path(self._LEADER_OPTIME)
|
||||||
@@ -644,6 +727,10 @@ class AbstractDCS(object):
|
|||||||
def loop_wait(self):
|
def loop_wait(self):
|
||||||
return self._loop_wait
|
return self._loop_wait
|
||||||
|
|
||||||
|
@property
|
||||||
|
def last_seen(self):
|
||||||
|
return self._last_seen
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
def _load_cluster(self):
|
def _load_cluster(self):
|
||||||
"""Internally this method should build `Cluster` object which
|
"""Internally this method should build `Cluster` object which
|
||||||
@@ -666,6 +753,8 @@ class AbstractDCS(object):
|
|||||||
self.reset_cluster()
|
self.reset_cluster()
|
||||||
raise
|
raise
|
||||||
|
|
||||||
|
self._last_seen = int(time.time())
|
||||||
|
|
||||||
with self._cluster_thread_lock:
|
with self._cluster_thread_lock:
|
||||||
self._cluster = cluster
|
self._cluster = cluster
|
||||||
self._cluster_valid_till = time.time() + self.ttl
|
self._cluster_valid_till = time.time() + self.ttl
|
||||||
@@ -682,14 +771,30 @@ class AbstractDCS(object):
|
|||||||
self._cluster_valid_till = 0
|
self._cluster_valid_till = 0
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
def _write_leader_optime(self, last_operation):
|
def _write_leader_optime(self, last_lsn):
|
||||||
"""write current xlog location into `/optime/leader` key in DCS
|
"""write current WAL LSN into `/optime/leader` key in DCS
|
||||||
:param last_operation: absolute xlog location in bytes
|
|
||||||
|
:param last_lsn: absolute WAL LSN in bytes
|
||||||
:returns: `!True` on success."""
|
:returns: `!True` on success."""
|
||||||
|
|
||||||
def write_leader_optime(self, last_operation):
|
def write_leader_optime(self, last_lsn):
|
||||||
if self._last_leader_operation != last_operation and self._write_leader_optime(last_operation):
|
self.write_status({self._OPTIME: last_lsn})
|
||||||
self._last_leader_operation = last_operation
|
|
||||||
|
@abc.abstractmethod
|
||||||
|
def _write_status(self, value):
|
||||||
|
"""write current WAL LSN and confirmed_flush_lsn of permanent slots into the `/status` key in DCS
|
||||||
|
|
||||||
|
:param value: status serialized in JSON forman
|
||||||
|
:returns: `!True` on success."""
|
||||||
|
|
||||||
|
def write_status(self, value):
|
||||||
|
if not deep_compare(self._last_status, value) and self._write_status(json.dumps(value, separators=(',', ':'))):
|
||||||
|
self._last_status = value
|
||||||
|
cluster = self.cluster
|
||||||
|
min_version = cluster and cluster.min_version
|
||||||
|
if min_version and min_version < (2, 1, 0) and self._last_lsn != value[self._OPTIME]:
|
||||||
|
self._last_lsn = value[self._OPTIME]
|
||||||
|
self._write_leader_optime(str(value[self._OPTIME]))
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
@@ -701,16 +806,20 @@ class AbstractDCS(object):
|
|||||||
You have to use CAS (Compare And Swap) operation in order to update leader key,
|
You have to use CAS (Compare And Swap) operation in order to update leader key,
|
||||||
for example for etcd `prevValue` parameter must be used."""
|
for example for etcd `prevValue` parameter must be used."""
|
||||||
|
|
||||||
def update_leader(self, last_operation, access_is_restricted=False):
|
def update_leader(self, last_lsn, slots=None):
|
||||||
"""Update leader key (or session) ttl and optime/leader
|
"""Update leader key (or session) ttl and optime/leader
|
||||||
|
|
||||||
:param last_operation: absolute xlog location in bytes
|
:param last_lsn: absolute WAL LSN in bytes
|
||||||
|
:param slots: dict with permanent slots confirmed_flush_lsn
|
||||||
:returns: `!True` if leader key (or session) has been updated successfully.
|
:returns: `!True` if leader key (or session) has been updated successfully.
|
||||||
If not, `!False` must be returned and current instance would be demoted."""
|
If not, `!False` must be returned and current instance would be demoted."""
|
||||||
|
|
||||||
ret = self._update_leader()
|
ret = self._update_leader()
|
||||||
if ret and last_operation:
|
if ret and last_lsn:
|
||||||
self.write_leader_optime(last_operation)
|
status = {self._OPTIME: last_lsn}
|
||||||
|
if slots:
|
||||||
|
status['slots'] = slots
|
||||||
|
self.write_status(status)
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
@@ -779,13 +888,13 @@ class AbstractDCS(object):
|
|||||||
"""Remove leader key from DCS.
|
"""Remove leader key from DCS.
|
||||||
This method should remove leader key if current instance is the leader"""
|
This method should remove leader key if current instance is the leader"""
|
||||||
|
|
||||||
def delete_leader(self, last_operation=None):
|
def delete_leader(self, last_lsn=None):
|
||||||
"""Update optime/leader and voluntarily remove leader key from DCS.
|
"""Update optime/leader and voluntarily remove leader key from DCS.
|
||||||
This method should remove leader key if current instance is the leader.
|
This method should remove leader key if current instance is the leader.
|
||||||
:param last_operation: latest checkpoint location in bytes"""
|
:param last_lsn: latest checkpoint location in bytes"""
|
||||||
|
|
||||||
if last_operation:
|
if last_lsn:
|
||||||
self.write_leader_optime(last_operation)
|
self.write_status({self._OPTIME: last_lsn})
|
||||||
return self._delete_leader()
|
return self._delete_leader()
|
||||||
|
|
||||||
@abc.abstractmethod
|
@abc.abstractmethod
|
||||||
@@ -828,4 +937,4 @@ class AbstractDCS(object):
|
|||||||
:returns: `!True` if you would like to reschedule the next run of ha cycle"""
|
:returns: `!True` if you would like to reschedule the next run of ha cycle"""
|
||||||
|
|
||||||
self.event.wait(timeout)
|
self.event.wait(timeout)
|
||||||
return self.event.isSet()
|
return self.event.is_set()
|
||||||
|
|||||||
+65
-19
@@ -111,7 +111,7 @@ class HTTPClient(object):
|
|||||||
# According to the documentation a small random amount of additional wait time is added to the
|
# According to the documentation a small random amount of additional wait time is added to the
|
||||||
# supplied maximum wait time to spread out the wake up time of any concurrent requests. This adds
|
# supplied maximum wait time to spread out the wake up time of any concurrent requests. This adds
|
||||||
# up to wait / 16 additional time to the maximum duration. Since our goal is actually getting a
|
# up to wait / 16 additional time to the maximum duration. Since our goal is actually getting a
|
||||||
# response rather read timeout we will add to the timeout a sligtly bigger value.
|
# response rather read timeout we will add to the timeout a slightly bigger value.
|
||||||
kwargs['timeout'] = timeout + max(timeout/15.0, 1)
|
kwargs['timeout'] = timeout + max(timeout/15.0, 1)
|
||||||
else:
|
else:
|
||||||
kwargs['timeout'] = self._read_timeout
|
kwargs['timeout'] = self._read_timeout
|
||||||
@@ -227,12 +227,11 @@ class Consul(AbstractDCS):
|
|||||||
self._last_session_refresh = 0
|
self._last_session_refresh = 0
|
||||||
self.__session_checks = config.get('checks', [])
|
self.__session_checks = config.get('checks', [])
|
||||||
self._register_service = config.get('register_service', False)
|
self._register_service = config.get('register_service', False)
|
||||||
|
self._previous_loop_register_service = self._register_service
|
||||||
|
self._service_tags = sorted(config.get('service_tags', []))
|
||||||
|
self._previous_loop_service_tags = self._service_tags
|
||||||
if self._register_service:
|
if self._register_service:
|
||||||
self._service_tags = config.get('service_tags', [])
|
self._set_service_name()
|
||||||
self._service_name = service_name_from_scope_name(self._scope)
|
|
||||||
if self._scope != self._service_name:
|
|
||||||
logger.warning('Using %s as consul service name instead of scope name %s', self._service_name,
|
|
||||||
self._scope)
|
|
||||||
self._service_check_interval = config.get('service_check_interval', '5s')
|
self._service_check_interval = config.get('service_check_interval', '5s')
|
||||||
if not self._ctl:
|
if not self._ctl:
|
||||||
self.create_session()
|
self.create_session()
|
||||||
@@ -250,7 +249,18 @@ class Consul(AbstractDCS):
|
|||||||
|
|
||||||
def reload_config(self, config):
|
def reload_config(self, config):
|
||||||
super(Consul, self).reload_config(config)
|
super(Consul, self).reload_config(config)
|
||||||
self._client.reload_config(config.get('consul', {}))
|
|
||||||
|
consul_config = config.get('consul', {})
|
||||||
|
self._client.reload_config(consul_config)
|
||||||
|
self._previous_loop_service_tags = self._service_tags
|
||||||
|
self._service_tags = sorted(consul_config.get('service_tags', []))
|
||||||
|
|
||||||
|
should_register_service = consul_config.get('register_service', False)
|
||||||
|
if should_register_service and not self._register_service:
|
||||||
|
self._set_service_name()
|
||||||
|
|
||||||
|
self._previous_loop_register_service = self._register_service
|
||||||
|
self._register_service = should_register_service
|
||||||
|
|
||||||
def set_ttl(self, ttl):
|
def set_ttl(self, ttl):
|
||||||
if self._client.http.set_ttl(ttl/2.0): # Consul multiplies the TTL by 2x
|
if self._client.http.set_ttl(ttl/2.0): # Consul multiplies the TTL by 2x
|
||||||
@@ -337,9 +347,24 @@ class Consul(AbstractDCS):
|
|||||||
history = nodes.get(self._HISTORY)
|
history = nodes.get(self._HISTORY)
|
||||||
history = history and TimelineHistory.from_node(history['ModifyIndex'], history['Value'])
|
history = history and TimelineHistory.from_node(history['ModifyIndex'], history['Value'])
|
||||||
|
|
||||||
# get last leader operation
|
# get last known leader lsn and slots
|
||||||
last_leader_operation = nodes.get(self._LEADER_OPTIME)
|
status = nodes.get(self._STATUS)
|
||||||
last_leader_operation = 0 if last_leader_operation is None else int(last_leader_operation['Value'])
|
if status:
|
||||||
|
try:
|
||||||
|
status = json.loads(status['Value'])
|
||||||
|
last_lsn = status.get(self._OPTIME)
|
||||||
|
slots = status.get('slots')
|
||||||
|
except Exception:
|
||||||
|
slots = last_lsn = None
|
||||||
|
else:
|
||||||
|
last_lsn = nodes.get(self._LEADER_OPTIME)
|
||||||
|
last_lsn = last_lsn and last_lsn['Value']
|
||||||
|
slots = None
|
||||||
|
|
||||||
|
try:
|
||||||
|
last_lsn = int(last_lsn)
|
||||||
|
except Exception:
|
||||||
|
last_lsn = 0
|
||||||
|
|
||||||
# get list of members
|
# get list of members
|
||||||
members = [self.member(n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
members = [self.member(n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
||||||
@@ -366,9 +391,9 @@ class Consul(AbstractDCS):
|
|||||||
sync = nodes.get(self._SYNC)
|
sync = nodes.get(self._SYNC)
|
||||||
sync = SyncState.from_node(sync and sync['ModifyIndex'], sync and sync['Value'])
|
sync = SyncState.from_node(sync and sync['ModifyIndex'], sync and sync['Value'])
|
||||||
|
|
||||||
return Cluster(initialize, config, leader, last_leader_operation, members, failover, sync, history)
|
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||||
except NotFound:
|
except NotFound:
|
||||||
return Cluster(None, None, None, None, [], None, None, None)
|
return Cluster(None, None, None, None, [], None, None, None, None)
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('get_cluster')
|
logger.exception('get_cluster')
|
||||||
raise ConsulError('Consul is not responding properly')
|
raise ConsulError('Consul is not responding properly')
|
||||||
@@ -387,14 +412,18 @@ class Consul(AbstractDCS):
|
|||||||
self._client.kv.delete(self.member_path)
|
self._client.kv.delete(self.member_path)
|
||||||
create_member = True
|
create_member = True
|
||||||
|
|
||||||
|
if self._register_service or self._previous_loop_register_service:
|
||||||
|
try:
|
||||||
|
self.update_service(not create_member and member and member.data or {}, data)
|
||||||
|
except Exception:
|
||||||
|
logger.exception('update_service')
|
||||||
|
|
||||||
if not create_member and member and deep_compare(data, member.data):
|
if not create_member and member and deep_compare(data, member.data):
|
||||||
return True
|
return True
|
||||||
|
|
||||||
try:
|
try:
|
||||||
args = {} if permanent else {'acquire': self._session}
|
args = {} if permanent else {'acquire': self._session}
|
||||||
self._client.kv.put(self.member_path, json.dumps(data, separators=(',', ':')), **args)
|
self._client.kv.put(self.member_path, json.dumps(data, separators=(',', ':')), **args)
|
||||||
if self._register_service:
|
|
||||||
self.update_service(not create_member and member and member.data or {}, data)
|
|
||||||
return True
|
return True
|
||||||
except InvalidSession:
|
except InvalidSession:
|
||||||
self._session = None
|
self._session = None
|
||||||
@@ -403,6 +432,11 @@ class Consul(AbstractDCS):
|
|||||||
logger.exception('touch_member')
|
logger.exception('touch_member')
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
def _set_service_name(self):
|
||||||
|
self._service_name = service_name_from_scope_name(self._scope)
|
||||||
|
if self._scope != self._service_name:
|
||||||
|
logger.warning('Using %s as consul service name instead of scope name %s', self._service_name, self._scope)
|
||||||
|
|
||||||
@catch_consul_errors
|
@catch_consul_errors
|
||||||
def register_service(self, service_name, **kwargs):
|
def register_service(self, service_name, **kwargs):
|
||||||
logger.info('Register service %s, params %s', service_name, kwargs)
|
logger.info('Register service %s, params %s', service_name, kwargs)
|
||||||
@@ -426,17 +460,22 @@ class Consul(AbstractDCS):
|
|||||||
deregister='{0}s'.format(self._client.http.ttl * 10))
|
deregister='{0}s'.format(self._client.http.ttl * 10))
|
||||||
tags = self._service_tags[:]
|
tags = self._service_tags[:]
|
||||||
tags.append(role)
|
tags.append(role)
|
||||||
|
self._previous_loop_service_tags = self._service_tags
|
||||||
|
|
||||||
params = {
|
params = {
|
||||||
'service_id': '{0}/{1}'.format(self._scope, self._name),
|
'service_id': '{0}/{1}'.format(self._scope, self._name),
|
||||||
'address': conn_parts.hostname,
|
'address': conn_parts.hostname,
|
||||||
'port': conn_parts.port,
|
'port': conn_parts.port,
|
||||||
'check': check,
|
'check': check,
|
||||||
'tags': tags
|
'tags': tags,
|
||||||
|
'enable_tag_override': True,
|
||||||
}
|
}
|
||||||
|
|
||||||
if state == 'stopped':
|
if state == 'stopped' or (not self._register_service and self._previous_loop_register_service):
|
||||||
|
self._previous_loop_register_service = self._register_service
|
||||||
return self.deregister_service(params['service_id'])
|
return self.deregister_service(params['service_id'])
|
||||||
|
|
||||||
|
self._previous_loop_register_service = self._register_service
|
||||||
if role in ['master', 'replica', 'standby-leader']:
|
if role in ['master', 'replica', 'standby-leader']:
|
||||||
if state != 'running':
|
if state != 'running':
|
||||||
return
|
return
|
||||||
@@ -455,7 +494,10 @@ class Consul(AbstractDCS):
|
|||||||
if old_data.get(key) != new_data[key]:
|
if old_data.get(key) != new_data[key]:
|
||||||
update = True
|
update = True
|
||||||
|
|
||||||
if force or update:
|
if (
|
||||||
|
force or update or self._register_service != self._previous_loop_register_service
|
||||||
|
or self._service_tags != self._previous_loop_service_tags
|
||||||
|
):
|
||||||
return self._update_service(new_data)
|
return self._update_service(new_data)
|
||||||
|
|
||||||
@catch_consul_errors
|
@catch_consul_errors
|
||||||
@@ -491,8 +533,12 @@ class Consul(AbstractDCS):
|
|||||||
return self._client.kv.put(self.config_path, value, cas=index)
|
return self._client.kv.put(self.config_path, value, cas=index)
|
||||||
|
|
||||||
@catch_consul_errors
|
@catch_consul_errors
|
||||||
def _write_leader_optime(self, last_operation):
|
def _write_leader_optime(self, last_lsn):
|
||||||
return self._client.kv.put(self.leader_optime_path, last_operation)
|
return self._client.kv.put(self.leader_optime_path, last_lsn)
|
||||||
|
|
||||||
|
@catch_consul_errors
|
||||||
|
def _write_status(self, value):
|
||||||
|
return self._client.kv.put(self.status_path, value)
|
||||||
|
|
||||||
@catch_consul_errors
|
@catch_consul_errors
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
|
|||||||
+36
-13
@@ -184,11 +184,12 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
|||||||
|
|
||||||
for base_uri in machines_cache:
|
for base_uri in machines_cache:
|
||||||
try:
|
try:
|
||||||
machines = list(self._get_members(base_uri, **kwargs))
|
machines = list(set(self._get_members(base_uri, **kwargs)))
|
||||||
logger.debug("Retrieved list of machines: %s", machines)
|
logger.debug("Retrieved list of machines: %s", machines)
|
||||||
if machines:
|
if machines:
|
||||||
random.shuffle(machines)
|
random.shuffle(machines)
|
||||||
self._update_dns_cache(self._dns_resolver.resolve_async, machines)
|
if not self._use_proxies:
|
||||||
|
self._update_dns_cache(self._dns_resolver.resolve_async, machines)
|
||||||
return machines
|
return machines
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
self.http.clear()
|
self.http.clear()
|
||||||
@@ -268,6 +269,7 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
|||||||
nodes, timeout, retries = self._calculate_timeouts(etcd_nodes, remaining_time)
|
nodes, timeout, retries = self._calculate_timeouts(etcd_nodes, remaining_time)
|
||||||
if nodes == 0:
|
if nodes == 0:
|
||||||
self._update_machines_cache = True
|
self._update_machines_cache = True
|
||||||
|
self.set_base_uri(self._base_uri) # trigger Etcd3 watcher restart
|
||||||
raise ex
|
raise ex
|
||||||
retry.sleep_func(sleeptime)
|
retry.sleep_func(sleeptime)
|
||||||
retry.update_delay()
|
retry.update_delay()
|
||||||
@@ -282,13 +284,14 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
|||||||
except DNSException:
|
except DNSException:
|
||||||
return []
|
return []
|
||||||
|
|
||||||
def _get_machines_cache_from_srv(self, srv):
|
def _get_machines_cache_from_srv(self, srv, srv_suffix=None):
|
||||||
"""Fetch list of etcd-cluster member by resolving _etcd-server._tcp. SRV record.
|
"""Fetch list of etcd-cluster member by resolving _etcd-server._tcp. SRV record.
|
||||||
This record should contain list of host and peer ports which could be used to run
|
This record should contain list of host and peer ports which could be used to run
|
||||||
'GET http://{host}:{port}/members' request (peer protocol)"""
|
'GET http://{host}:{port}/members' request (peer protocol)"""
|
||||||
|
|
||||||
ret = []
|
ret = []
|
||||||
for r in ['-client-ssl', '-client', '-ssl', '', '-server-ssl', '-server']:
|
for r in ['-client-ssl', '-client', '-ssl', '', '-server-ssl', '-server']:
|
||||||
|
r = '{0}-{1}'.format(r, srv_suffix) if srv_suffix else r
|
||||||
protocol = 'https' if '-ssl' in r else 'http'
|
protocol = 'https' if '-ssl' in r else 'http'
|
||||||
endpoint = '/members' if '-server' in r else ''
|
endpoint = '/members' if '-server' in r else ''
|
||||||
for host, port in self.get_srv_record('_etcd{0}._tcp.{1}'.format(r, srv)):
|
for host, port in self.get_srv_record('_etcd{0}._tcp.{1}'.format(r, srv)):
|
||||||
@@ -325,7 +328,7 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
|||||||
|
|
||||||
machines_cache = []
|
machines_cache = []
|
||||||
if 'srv' in self._config:
|
if 'srv' in self._config:
|
||||||
machines_cache = self._get_machines_cache_from_srv(self._config['srv'])
|
machines_cache = self._get_machines_cache_from_srv(self._config['srv'], self._config.get('srv_suffix'))
|
||||||
|
|
||||||
if not machines_cache and 'hosts' in self._config:
|
if not machines_cache and 'hosts' in self._config:
|
||||||
machines_cache = list(self._config['hosts'])
|
machines_cache = list(self._config['hosts'])
|
||||||
@@ -392,8 +395,9 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
|||||||
self._machines_cache_updated = time.time()
|
self._machines_cache_updated = time.time()
|
||||||
|
|
||||||
def set_base_uri(self, value):
|
def set_base_uri(self, value):
|
||||||
logger.info('Selected new etcd server %s', value)
|
if self._base_uri != value:
|
||||||
self._base_uri = value
|
logger.info('Selected new etcd server %s', value)
|
||||||
|
self._base_uri = value
|
||||||
|
|
||||||
|
|
||||||
class EtcdClient(AbstractEtcdClientWithFailover):
|
class EtcdClient(AbstractEtcdClientWithFailover):
|
||||||
@@ -602,9 +606,24 @@ class Etcd(AbstractEtcd):
|
|||||||
history = nodes.get(self._HISTORY)
|
history = nodes.get(self._HISTORY)
|
||||||
history = history and TimelineHistory.from_node(history.modifiedIndex, history.value)
|
history = history and TimelineHistory.from_node(history.modifiedIndex, history.value)
|
||||||
|
|
||||||
# get last leader operation
|
# get last know leader lsn and slots
|
||||||
last_leader_operation = nodes.get(self._LEADER_OPTIME)
|
status = nodes.get(self._STATUS)
|
||||||
last_leader_operation = 0 if last_leader_operation is None else int(last_leader_operation.value)
|
if status:
|
||||||
|
try:
|
||||||
|
status = json.loads(status.value)
|
||||||
|
last_lsn = status.get(self._OPTIME)
|
||||||
|
slots = status.get('slots')
|
||||||
|
except Exception:
|
||||||
|
slots = last_lsn = None
|
||||||
|
else:
|
||||||
|
last_lsn = nodes.get(self._LEADER_OPTIME)
|
||||||
|
last_lsn = last_lsn and last_lsn.value
|
||||||
|
slots = None
|
||||||
|
|
||||||
|
try:
|
||||||
|
last_lsn = int(last_lsn)
|
||||||
|
except Exception:
|
||||||
|
last_lsn = 0
|
||||||
|
|
||||||
# get list of members
|
# get list of members
|
||||||
members = [self.member(n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
members = [self.member(n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
||||||
@@ -626,9 +645,9 @@ class Etcd(AbstractEtcd):
|
|||||||
sync = nodes.get(self._SYNC)
|
sync = nodes.get(self._SYNC)
|
||||||
sync = SyncState.from_node(sync and sync.modifiedIndex, sync and sync.value)
|
sync = SyncState.from_node(sync and sync.modifiedIndex, sync and sync.value)
|
||||||
|
|
||||||
cluster = Cluster(initialize, config, leader, last_leader_operation, members, failover, sync, history)
|
cluster = Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||||
except etcd.EtcdKeyNotFound:
|
except etcd.EtcdKeyNotFound:
|
||||||
cluster = Cluster(None, None, None, None, [], None, None, None)
|
cluster = Cluster(None, None, None, None, [], None, None, None, None)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
self._handle_exception(e, 'get_cluster', raise_ex=EtcdError('Etcd is not responding properly'))
|
self._handle_exception(e, 'get_cluster', raise_ex=EtcdError('Etcd is not responding properly'))
|
||||||
self._has_failed = False
|
self._has_failed = False
|
||||||
@@ -665,8 +684,12 @@ class Etcd(AbstractEtcd):
|
|||||||
return self._client.write(self.config_path, value, prevIndex=index or 0)
|
return self._client.write(self.config_path, value, prevIndex=index or 0)
|
||||||
|
|
||||||
@catch_etcd_errors
|
@catch_etcd_errors
|
||||||
def _write_leader_optime(self, last_operation):
|
def _write_leader_optime(self, last_lsn):
|
||||||
return self._client.set(self.leader_optime_path, last_operation)
|
return self._client.set(self.leader_optime_path, last_lsn)
|
||||||
|
|
||||||
|
@catch_etcd_errors
|
||||||
|
def _write_status(self, value):
|
||||||
|
return self._client.set(self.status_path, value)
|
||||||
|
|
||||||
@catch_etcd_errors
|
@catch_etcd_errors
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
|
|||||||
+34
-10
@@ -262,6 +262,9 @@ class Etcd3Client(AbstractEtcdClientWithFailover):
|
|||||||
return self.api_execute(self.version_prefix + method, self._MPOST, fields)
|
return self.api_execute(self.version_prefix + method, self._MPOST, fields)
|
||||||
|
|
||||||
def authenticate(self):
|
def authenticate(self):
|
||||||
|
if self._use_proxies and self._cluster_version is None:
|
||||||
|
kwargs = self._prepare_common_parameters(1)
|
||||||
|
self._ensure_version_prefix(self._base_uri, **kwargs)
|
||||||
if self._cluster_version >= (3, 3) and self.username and self.password:
|
if self._cluster_version >= (3, 3) and self.username and self.password:
|
||||||
logger.info('Trying to authenticate on Etcd...')
|
logger.info('Trying to authenticate on Etcd...')
|
||||||
old_token, self._token = self._token, None
|
old_token, self._token = self._token, None
|
||||||
@@ -372,6 +375,7 @@ class KVCache(Thread):
|
|||||||
self._config_key = base64_encode(dcs.config_path)
|
self._config_key = base64_encode(dcs.config_path)
|
||||||
self._leader_key = base64_encode(dcs.leader_path)
|
self._leader_key = base64_encode(dcs.leader_path)
|
||||||
self._optime_key = base64_encode(dcs.leader_optime_path)
|
self._optime_key = base64_encode(dcs.leader_optime_path)
|
||||||
|
self._status_key = base64_encode(dcs.status_path)
|
||||||
self._name = base64_encode(dcs._name)
|
self._name = base64_encode(dcs._name)
|
||||||
self._is_ready = False
|
self._is_ready = False
|
||||||
self._response = None
|
self._response = None
|
||||||
@@ -418,14 +422,15 @@ class KVCache(Thread):
|
|||||||
new_value = kv.get('value')
|
new_value = kv.get('value')
|
||||||
|
|
||||||
value_changed = old_value != new_value and \
|
value_changed = old_value != new_value and \
|
||||||
(key == self._leader_key or key == self._optime_key and new_value is not None or
|
(key == self._leader_key or key in (self._optime_key, self._status_key) and new_value is not None or
|
||||||
key == self._config_key and old_value is not None and new_value is not None)
|
key == self._config_key and old_value is not None and new_value is not None)
|
||||||
|
|
||||||
if value_changed:
|
if value_changed:
|
||||||
logger.debug('%s changed from %s to %s', key, old_value, new_value)
|
logger.debug('%s changed from %s to %s', key, old_value, new_value)
|
||||||
|
|
||||||
# We also want to wake up HA loop on replicas if leader optime was updated
|
# We also want to wake up HA loop on replicas if leader optime (or status key) was updated
|
||||||
if value_changed and (key != self._optime_key or self.get(self._leader_key) != self._name):
|
if value_changed and (key not in (self._optime_key, self._status_key) or
|
||||||
|
(self.get(self._leader_key) or {}).get('value') != self._name):
|
||||||
self._dcs.event.set()
|
self._dcs.event.set()
|
||||||
|
|
||||||
def _process_message(self, message):
|
def _process_message(self, message):
|
||||||
@@ -612,7 +617,7 @@ class Etcd3(AbstractEtcd):
|
|||||||
return self.retry(self._do_refresh_lease)
|
return self.retry(self._do_refresh_lease)
|
||||||
except (Etcd3ClientError, RetryFailedError):
|
except (Etcd3ClientError, RetryFailedError):
|
||||||
logger.exception('refresh_lease')
|
logger.exception('refresh_lease')
|
||||||
raise Etcd3Error('Failed ro keepalive/grant lease')
|
raise Etcd3Error('Failed to keepalive/grant lease')
|
||||||
|
|
||||||
def create_lease(self):
|
def create_lease(self):
|
||||||
while not self._lease:
|
while not self._lease:
|
||||||
@@ -654,9 +659,24 @@ class Etcd3(AbstractEtcd):
|
|||||||
history = nodes.get(self._HISTORY)
|
history = nodes.get(self._HISTORY)
|
||||||
history = history and TimelineHistory.from_node(history['mod_revision'], history['value'])
|
history = history and TimelineHistory.from_node(history['mod_revision'], history['value'])
|
||||||
|
|
||||||
# get last leader operation
|
# get last know leader lsn and slots
|
||||||
last_leader_operation = nodes.get(self._LEADER_OPTIME)
|
status = nodes.get(self._STATUS)
|
||||||
last_leader_operation = 0 if last_leader_operation is None else int(last_leader_operation['value'])
|
if status:
|
||||||
|
try:
|
||||||
|
status = json.loads(status['value'])
|
||||||
|
last_lsn = status.get(self._OPTIME)
|
||||||
|
slots = status.get('slots')
|
||||||
|
except Exception:
|
||||||
|
slots = last_lsn = None
|
||||||
|
else:
|
||||||
|
last_lsn = nodes.get(self._LEADER_OPTIME)
|
||||||
|
last_lsn = last_lsn and last_lsn['value']
|
||||||
|
slots = None
|
||||||
|
|
||||||
|
try:
|
||||||
|
last_lsn = int(last_lsn)
|
||||||
|
except Exception:
|
||||||
|
last_lsn = 0
|
||||||
|
|
||||||
# get list of members
|
# get list of members
|
||||||
members = [self.member(n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
members = [self.member(n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
||||||
@@ -680,7 +700,7 @@ class Etcd3(AbstractEtcd):
|
|||||||
sync = nodes.get(self._SYNC)
|
sync = nodes.get(self._SYNC)
|
||||||
sync = SyncState.from_node(sync and sync['mod_revision'], sync and sync['value'])
|
sync = SyncState.from_node(sync and sync['mod_revision'], sync and sync['value'])
|
||||||
|
|
||||||
cluster = Cluster(initialize, config, leader, last_leader_operation, members, failover, sync, history)
|
cluster = Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||||
except UnsupportedEtcdVersion:
|
except UnsupportedEtcdVersion:
|
||||||
raise
|
raise
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
@@ -741,8 +761,12 @@ class Etcd3(AbstractEtcd):
|
|||||||
return self._client.put(self.config_path, value, mod_revision=index)
|
return self._client.put(self.config_path, value, mod_revision=index)
|
||||||
|
|
||||||
@catch_etcd_errors
|
@catch_etcd_errors
|
||||||
def _write_leader_optime(self, last_operation):
|
def _write_leader_optime(self, last_lsn):
|
||||||
return self._client.put(self.leader_optime_path, last_operation)
|
return self._client.put(self.leader_optime_path, last_lsn)
|
||||||
|
|
||||||
|
@catch_etcd_errors
|
||||||
|
def _write_status(self, value):
|
||||||
|
return self._client.put(self.status_path, value)
|
||||||
|
|
||||||
@catch_etcd_errors
|
@catch_etcd_errors
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
|
|||||||
+39
-23
@@ -55,16 +55,19 @@ class K8sConfig(object):
|
|||||||
if token:
|
if token:
|
||||||
self._headers['authorization'] = 'Bearer ' + token
|
self._headers['authorization'] = 'Bearer ' + token
|
||||||
|
|
||||||
def load_incluster_config(self):
|
def load_incluster_config(self, ca_certs=SERVICE_CERT_FILENAME):
|
||||||
if SERVICE_HOST_ENV_NAME not in os.environ or SERVICE_PORT_ENV_NAME not in os.environ:
|
if SERVICE_HOST_ENV_NAME not in os.environ or SERVICE_PORT_ENV_NAME not in os.environ:
|
||||||
raise self.ConfigException('Service host/port is not set.')
|
raise self.ConfigException('Service host/port is not set.')
|
||||||
if not os.environ[SERVICE_HOST_ENV_NAME] or not os.environ[SERVICE_PORT_ENV_NAME]:
|
if not os.environ[SERVICE_HOST_ENV_NAME] or not os.environ[SERVICE_PORT_ENV_NAME]:
|
||||||
raise self.ConfigException('Service host/port is set but empty.')
|
raise self.ConfigException('Service host/port is set but empty.')
|
||||||
if not os.path.isfile(SERVICE_CERT_FILENAME):
|
|
||||||
|
if not os.path.isfile(ca_certs):
|
||||||
raise self.ConfigException('Service certificate file does not exists.')
|
raise self.ConfigException('Service certificate file does not exists.')
|
||||||
with open(SERVICE_CERT_FILENAME) as f:
|
with open(ca_certs) as f:
|
||||||
if not f.read():
|
if not f.read():
|
||||||
raise self.ConfigException('Cert file exists but empty.')
|
raise self.ConfigException('Cert file exists but empty.')
|
||||||
|
self.pool_config['ca_certs'] = ca_certs
|
||||||
|
|
||||||
if not os.path.isfile(SERVICE_TOKEN_FILENAME):
|
if not os.path.isfile(SERVICE_TOKEN_FILENAME):
|
||||||
raise self.ConfigException('Service token file does not exists.')
|
raise self.ConfigException('Service token file does not exists.')
|
||||||
with open(SERVICE_TOKEN_FILENAME) as f:
|
with open(SERVICE_TOKEN_FILENAME) as f:
|
||||||
@@ -72,7 +75,6 @@ class K8sConfig(object):
|
|||||||
if not token:
|
if not token:
|
||||||
raise self.ConfigException('Token file exists but empty.')
|
raise self.ConfigException('Token file exists but empty.')
|
||||||
self._make_headers(token=token)
|
self._make_headers(token=token)
|
||||||
self.pool_config['ca_certs'] = SERVICE_CERT_FILENAME
|
|
||||||
self._server = uri('https', (os.environ[SERVICE_HOST_ENV_NAME], os.environ[SERVICE_PORT_ENV_NAME]))
|
self._server = uri('https', (os.environ[SERVICE_HOST_ENV_NAME], os.environ[SERVICE_PORT_ENV_NAME]))
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
@@ -343,12 +345,12 @@ class K8sClient(object):
|
|||||||
try:
|
try:
|
||||||
self._load_api_servers_cache()
|
self._load_api_servers_cache()
|
||||||
api_servers_cache = self.api_servers_cache
|
api_servers_cache = self.api_servers_cache
|
||||||
api_servers = len(api_servers)
|
api_servers = len(api_servers_cache)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.debug('Failed to update list of K8s master nodes: %r', e)
|
logger.debug('Failed to update list of K8s master nodes: %r', e)
|
||||||
|
|
||||||
sleeptime = retry.sleeptime
|
sleeptime = retry.sleeptime
|
||||||
remaining_time = retry.stoptime - sleeptime - time.time()
|
remaining_time = (retry.stoptime or time.time()) - sleeptime - time.time()
|
||||||
nodes, timeout, retries = self._calculate_timeouts(api_servers, remaining_time)
|
nodes, timeout, retries = self._calculate_timeouts(api_servers, remaining_time)
|
||||||
if nodes == 0:
|
if nodes == 0:
|
||||||
self._update_api_servers_cache = True
|
self._update_api_servers_cache = True
|
||||||
@@ -613,13 +615,14 @@ class Kubernetes(AbstractDCS):
|
|||||||
self._label_selector = ','.join('{0}={1}'.format(k, v) for k, v in self._labels.items())
|
self._label_selector = ','.join('{0}={1}'.format(k, v) for k, v in self._labels.items())
|
||||||
self._namespace = config.get('namespace') or 'default'
|
self._namespace = config.get('namespace') or 'default'
|
||||||
self._role_label = config.get('role_label', 'role')
|
self._role_label = config.get('role_label', 'role')
|
||||||
|
self._ca_certs = os.environ.get('PATRONI_KUBERNETES_CACERT', config.get('cacert')) or SERVICE_CERT_FILENAME
|
||||||
config['namespace'] = ''
|
config['namespace'] = ''
|
||||||
super(Kubernetes, self).__init__(config)
|
super(Kubernetes, self).__init__(config)
|
||||||
self._retry = Retry(deadline=config['retry_timeout'], max_delay=1, max_tries=-1,
|
self._retry = Retry(deadline=config['retry_timeout'], max_delay=1, max_tries=-1,
|
||||||
retry_exceptions=KubernetesRetriableException)
|
retry_exceptions=KubernetesRetriableException)
|
||||||
self._ttl = None
|
self._ttl = None
|
||||||
try:
|
try:
|
||||||
k8s_config.load_incluster_config()
|
k8s_config.load_incluster_config(ca_certs=self._ca_certs)
|
||||||
except k8s_config.ConfigException:
|
except k8s_config.ConfigException:
|
||||||
k8s_config.load_kube_config(context=config.get('context', 'local'))
|
k8s_config.load_kube_config(context=config.get('context', 'local'))
|
||||||
|
|
||||||
@@ -724,9 +727,19 @@ class Kubernetes(AbstractDCS):
|
|||||||
self._leader_resource_version = metadata.resource_version if metadata else None
|
self._leader_resource_version = metadata.resource_version if metadata else None
|
||||||
annotations = metadata and metadata.annotations or {}
|
annotations = metadata and metadata.annotations or {}
|
||||||
|
|
||||||
# get last leader operation
|
# get last known leader lsn
|
||||||
last_leader_operation = annotations.get(self._OPTIME)
|
last_lsn = annotations.get(self._OPTIME)
|
||||||
last_leader_operation = 0 if last_leader_operation is None else int(last_leader_operation)
|
try:
|
||||||
|
last_lsn = 0 if last_lsn is None else int(last_lsn)
|
||||||
|
except Exception:
|
||||||
|
last_lsn = 0
|
||||||
|
|
||||||
|
# get permanent slots state (confirmed_flush_lsn)
|
||||||
|
slots = annotations.get('slots')
|
||||||
|
try:
|
||||||
|
slots = slots and json.loads(slots)
|
||||||
|
except Exception:
|
||||||
|
slots = None
|
||||||
|
|
||||||
# get leader
|
# get leader
|
||||||
leader_record = {n: annotations.get(n) for n in (self._LEADER, 'acquireTime',
|
leader_record = {n: annotations.get(n) for n in (self._LEADER, 'acquireTime',
|
||||||
@@ -760,7 +773,7 @@ class Kubernetes(AbstractDCS):
|
|||||||
metadata = sync and sync.metadata
|
metadata = sync and sync.metadata
|
||||||
sync = SyncState.from_node(metadata and metadata.resource_version, metadata and metadata.annotations)
|
sync = SyncState.from_node(metadata and metadata.resource_version, metadata and metadata.annotations)
|
||||||
|
|
||||||
return Cluster(initialize, config, leader, last_leader_operation, members, failover, sync, history)
|
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('get_cluster')
|
logger.exception('get_cluster')
|
||||||
raise KubernetesError('Kubernetes API is not responding properly')
|
raise KubernetesError('Kubernetes API is not responding properly')
|
||||||
@@ -881,7 +894,10 @@ class Kubernetes(AbstractDCS):
|
|||||||
return logger.exception('create_config_service failed')
|
return logger.exception('create_config_service failed')
|
||||||
self._should_create_config_service = False
|
self._should_create_config_service = False
|
||||||
|
|
||||||
def _write_leader_optime(self, last_operation):
|
def _write_leader_optime(self, last_lsn):
|
||||||
|
"""Unused"""
|
||||||
|
|
||||||
|
def _write_status(self, value):
|
||||||
"""Unused"""
|
"""Unused"""
|
||||||
|
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
@@ -911,7 +927,7 @@ class Kubernetes(AbstractDCS):
|
|||||||
|
|
||||||
# Try to get the latest version directly from K8s API instead of relying on async cache
|
# Try to get the latest version directly from K8s API instead of relying on async cache
|
||||||
try:
|
try:
|
||||||
kind = retry(self._api.read_namespaced_kind, self.leader_path, self._namespace)
|
kind = _retry(self._api.read_namespaced_kind, self.leader_path, self._namespace)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error('Failed to get the leader object "%s": %r', self.leader_path, e)
|
logger.error('Failed to get the leader object "%s": %r', self.leader_path, e)
|
||||||
return False
|
return False
|
||||||
@@ -931,7 +947,7 @@ class Kubernetes(AbstractDCS):
|
|||||||
|
|
||||||
return self.patch_or_create(self.leader_path, annotations, kind_resource_version, ips=ips, retry=_retry)
|
return self.patch_or_create(self.leader_path, annotations, kind_resource_version, ips=ips, retry=_retry)
|
||||||
|
|
||||||
def update_leader(self, last_operation, access_is_restricted=False):
|
def update_leader(self, last_lsn, slots=None):
|
||||||
kind = self._kinds.get(self.leader_path)
|
kind = self._kinds.get(self.leader_path)
|
||||||
kind_annotations = kind and kind.metadata.annotations or {}
|
kind_annotations = kind and kind.metadata.annotations or {}
|
||||||
|
|
||||||
@@ -943,12 +959,12 @@ class Kubernetes(AbstractDCS):
|
|||||||
annotations = {self._LEADER: self._name, 'ttl': str(self._ttl), 'renewTime': now,
|
annotations = {self._LEADER: self._name, 'ttl': str(self._ttl), 'renewTime': now,
|
||||||
'acquireTime': leader_observed_record.get('acquireTime') or now,
|
'acquireTime': leader_observed_record.get('acquireTime') or now,
|
||||||
'transitions': leader_observed_record.get('transitions') or '0'}
|
'transitions': leader_observed_record.get('transitions') or '0'}
|
||||||
if last_operation:
|
if last_lsn:
|
||||||
annotations[self._OPTIME] = last_operation
|
annotations[self._OPTIME] = str(last_lsn)
|
||||||
|
annotations['slots'] = json.dumps(slots) if slots else None
|
||||||
|
|
||||||
resource_version = kind and kind.metadata.resource_version
|
resource_version = kind and kind.metadata.resource_version
|
||||||
ips = [] if access_is_restricted else self.__ips
|
return self._update_leader_with_retry(annotations, resource_version, self.__ips)
|
||||||
return self._update_leader_with_retry(annotations, resource_version, ips)
|
|
||||||
|
|
||||||
def attempt_to_acquire_leader(self, permanent=False):
|
def attempt_to_acquire_leader(self, permanent=False):
|
||||||
now = datetime.datetime.now(tzutc).isoformat()
|
now = datetime.datetime.now(tzutc).isoformat()
|
||||||
@@ -995,7 +1011,7 @@ class Kubernetes(AbstractDCS):
|
|||||||
def touch_member(self, data, permanent=False):
|
def touch_member(self, data, permanent=False):
|
||||||
cluster = self.cluster
|
cluster = self.cluster
|
||||||
if cluster and cluster.leader and cluster.leader.name == self._name:
|
if cluster and cluster.leader and cluster.leader.name == self._name:
|
||||||
role = 'promoted' if data['role'] in ('replica', 'promoted') else 'master'
|
role = 'master'
|
||||||
elif data['state'] == 'running' and data['role'] != 'master':
|
elif data['state'] == 'running' and data['role'] != 'master':
|
||||||
role = data['role']
|
role = data['role']
|
||||||
else:
|
else:
|
||||||
@@ -1024,17 +1040,17 @@ class Kubernetes(AbstractDCS):
|
|||||||
def _delete_leader(self):
|
def _delete_leader(self):
|
||||||
"""Unused"""
|
"""Unused"""
|
||||||
|
|
||||||
def delete_leader(self, last_operation=None):
|
def delete_leader(self, last_lsn=None):
|
||||||
kind = self._kinds.get(self.leader_path)
|
kind = self._kinds.get(self.leader_path)
|
||||||
if kind and (kind.metadata.annotations or {}).get(self._LEADER) == self._name:
|
if kind and (kind.metadata.annotations or {}).get(self._LEADER) == self._name:
|
||||||
annotations = {self._LEADER: None}
|
annotations = {self._LEADER: None}
|
||||||
if last_operation:
|
if last_lsn:
|
||||||
annotations[self._OPTIME] = last_operation
|
annotations[self._OPTIME] = last_lsn
|
||||||
self.patch_or_create(self.leader_path, annotations, kind.metadata.resource_version, True, False, [])
|
self.patch_or_create(self.leader_path, annotations, kind.metadata.resource_version, True, False, [])
|
||||||
self.reset_cluster()
|
self.reset_cluster()
|
||||||
|
|
||||||
def cancel_initialization(self):
|
def cancel_initialization(self):
|
||||||
self.patch_or_create_config({self._INITIALIZE: None}, self._config_resource_version, True)
|
return self.patch_or_create_config({self._INITIALIZE: None}, None, True)
|
||||||
|
|
||||||
@catch_kubernetes_errors
|
@catch_kubernetes_errors
|
||||||
def delete_cluster(self):
|
def delete_cluster(self):
|
||||||
|
|||||||
+136
-135
@@ -4,130 +4,124 @@ import os
|
|||||||
import threading
|
import threading
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from patroni.dcs import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
|
||||||
from ..utils import validate_directory
|
|
||||||
from pysyncobj import SyncObj, SyncObjConf, replicated, FAIL_REASON
|
from pysyncobj import SyncObj, SyncObjConf, replicated, FAIL_REASON
|
||||||
from pysyncobj.transport import Node, TCPTransport, CONNECTION_STATE
|
from pysyncobj.dns_resolver import globalDnsResolver
|
||||||
|
from pysyncobj.node import TCPNode
|
||||||
|
from pysyncobj.transport import TCPTransport, CONNECTION_STATE
|
||||||
|
from pysyncobj.utility import TcpUtility
|
||||||
|
|
||||||
|
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
||||||
|
from ..utils import validate_directory
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
class MessageNode(Node):
|
class _TCPTransport(TCPTransport):
|
||||||
|
|
||||||
def __init__(self, address):
|
|
||||||
self.address = address
|
|
||||||
|
|
||||||
|
|
||||||
class UtilityTransport(TCPTransport):
|
|
||||||
|
|
||||||
def __init__(self, syncObj, selfNode, otherNodes):
|
def __init__(self, syncObj, selfNode, otherNodes):
|
||||||
super(UtilityTransport, self).__init__(syncObj, selfNode, otherNodes)
|
super(_TCPTransport, self).__init__(syncObj, selfNode, otherNodes)
|
||||||
self._selfIsReadonlyNode = False
|
self.setOnUtilityMessageCallback('members', syncObj.getMembers)
|
||||||
|
|
||||||
def _connectIfNecessarySingle(self, node):
|
def _connectIfNecessarySingle(self, node):
|
||||||
pass
|
try:
|
||||||
|
return super(_TCPTransport, self)._connectIfNecessarySingle(node)
|
||||||
def connectionState(self, node):
|
except Exception as e:
|
||||||
return self._connections[node].state
|
logger.debug('Connection to %s failed: %r', node, e)
|
||||||
|
return False
|
||||||
def isDisconnected(self, node):
|
|
||||||
return self.connectionState(node) == CONNECTION_STATE.DISCONNECTED
|
|
||||||
|
|
||||||
def connectIfRequiredSingle(self, node):
|
|
||||||
if self.isDisconnected(node):
|
|
||||||
return self._connections[node].connect(node.ip, node.port)
|
|
||||||
|
|
||||||
def disconnectSingle(self, node):
|
|
||||||
self._connections[node].disconnect()
|
|
||||||
|
|
||||||
|
|
||||||
class SyncObjUtility(SyncObj):
|
def resolve_host(self):
|
||||||
|
return globalDnsResolver().resolve(self.host)
|
||||||
|
|
||||||
|
|
||||||
|
setattr(TCPNode, 'ip', property(resolve_host))
|
||||||
|
|
||||||
|
|
||||||
|
class SyncObjUtility(object):
|
||||||
|
|
||||||
def __init__(self, otherNodes, conf):
|
def __init__(self, otherNodes, conf):
|
||||||
autoTick = conf.autoTick
|
self._nodes = otherNodes
|
||||||
conf.autoTick = False
|
self._utility = TcpUtility(conf.password)
|
||||||
super(SyncObjUtility, self).__init__(None, otherNodes, conf, transportClass=UtilityTransport)
|
|
||||||
conf.autoTick = autoTick
|
|
||||||
self._SyncObj__transport.setOnMessageReceivedCallback(self._onMessageReceived)
|
|
||||||
self.__result = None
|
|
||||||
|
|
||||||
def setPartnerNode(self, partner):
|
def executeCommand(self, command):
|
||||||
self.__node = partner
|
try:
|
||||||
|
return self._utility.executeCommand(self.__node, command)
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
|
||||||
def sendMessage(self, message):
|
def getMembers(self):
|
||||||
# Abuse the fact that node address is send as a first message
|
for self.__node in self._nodes:
|
||||||
self._SyncObj__transport._selfNode = MessageNode(message)
|
response = self.executeCommand(['members'])
|
||||||
self._SyncObj__transport.connectIfRequiredSingle(self.__node)
|
if response:
|
||||||
while not self._SyncObj__transport.isDisconnected(self.__node):
|
return [member['addr'] for member in response]
|
||||||
self._poller.poll(0.5)
|
|
||||||
return self.__result
|
|
||||||
|
|
||||||
def _onMessageReceived(self, _, message):
|
|
||||||
self.__result = message
|
|
||||||
self._SyncObj__transport.disconnectSingle(self.__node)
|
|
||||||
|
|
||||||
|
|
||||||
class MyTCPTransport(TCPTransport):
|
|
||||||
|
|
||||||
def _onIncomingMessageReceived(self, conn, message):
|
|
||||||
if self._syncObj.encryptor and not conn.sendRandKey:
|
|
||||||
conn.sendRandKey = message
|
|
||||||
conn.recvRandKey = os.urandom(32)
|
|
||||||
conn.send(conn.recvRandKey)
|
|
||||||
return
|
|
||||||
|
|
||||||
# Utility messages
|
|
||||||
if isinstance(message, list) and message[0] == 'members':
|
|
||||||
conn.send(self._syncObj._get_members())
|
|
||||||
return True
|
|
||||||
|
|
||||||
return super(MyTCPTransport, self)._onIncomingMessageReceived(conn, message)
|
|
||||||
|
|
||||||
|
|
||||||
class DynMemberSyncObj(SyncObj):
|
class DynMemberSyncObj(SyncObj):
|
||||||
|
|
||||||
def __init__(self, selfAddress, partnerAddrs, conf):
|
def __init__(self, selfAddress, partnerAddrs, conf):
|
||||||
add_self = False
|
self.__early_apply_local_log = selfAddress is not None
|
||||||
|
self.applied_local_log = False
|
||||||
|
|
||||||
utility = SyncObjUtility(partnerAddrs, conf)
|
utility = SyncObjUtility(partnerAddrs, conf)
|
||||||
for node in utility._SyncObj__otherNodes:
|
members = utility.getMembers()
|
||||||
utility.setPartnerNode(node)
|
add_self = members and selfAddress not in members
|
||||||
response = utility.sendMessage(['members'])
|
|
||||||
if response:
|
partnerAddrs = [member for member in (members or partnerAddrs) if member != selfAddress]
|
||||||
partnerAddrs = [member['addr'] for member in response if member['addr'] != selfAddress]
|
|
||||||
add_self = selfAddress and len(partnerAddrs) == len(response)
|
super(DynMemberSyncObj, self).__init__(selfAddress, partnerAddrs, conf, transportClass=_TCPTransport)
|
||||||
break
|
|
||||||
|
|
||||||
super(DynMemberSyncObj, self).__init__(selfAddress, partnerAddrs, conf, transportClass=MyTCPTransport)
|
|
||||||
if add_self:
|
if add_self:
|
||||||
threading.Thread(target=utility.sendMessage, args=(['add', selfAddress],)).start()
|
thread = threading.Thread(target=utility.executeCommand, args=(['add', selfAddress],))
|
||||||
|
thread.daemon = True
|
||||||
|
thread.start()
|
||||||
|
|
||||||
def _get_members(self):
|
def getMembers(self, args, callback):
|
||||||
ret = [{'addr': node.id, 'leader': node == self._getLeader(),
|
callback([{'addr': node.id, 'leader': node == self._getLeader(), 'status': CONNECTION_STATE.CONNECTED
|
||||||
'status': CONNECTION_STATE.CONNECTED if node in self._SyncObj__connectedNodes
|
if self.isNodeConnected(node) else CONNECTION_STATE.DISCONNECTED} for node in self.otherNodes] +
|
||||||
else CONNECTION_STATE.DISCONNECTED} for node in self._SyncObj__otherNodes]
|
[{'addr': self.selfNode.id, 'leader': self._isLeader(), 'status': CONNECTION_STATE.CONNECTED}], None)
|
||||||
ret.append({'addr': self._SyncObj__selfNode.id, 'leader': self._isLeader(),
|
|
||||||
'status': CONNECTION_STATE.CONNECTED})
|
|
||||||
return ret
|
|
||||||
|
|
||||||
def _SyncObj__doChangeCluster(self, request, reverse=False):
|
def _onTick(self, timeToWait=0.0):
|
||||||
ret = False
|
super(DynMemberSyncObj, self)._onTick(timeToWait)
|
||||||
if not self._SyncObj__selfNode or request[0] != 'add' or reverse or request[1] != self._SyncObj__selfNode.id:
|
|
||||||
ret = super(DynMemberSyncObj, self)._SyncObj__doChangeCluster(request, reverse)
|
# The SyncObj calls onReady callback only when cluster got the leader and is ready for writes.
|
||||||
if ret:
|
# In some cases for us it is safe to "signal" the Raft object when the local log is fully applied.
|
||||||
self.forceLogCompaction()
|
# We are using the `applied_local_log` property for that, but not calling the callback function.
|
||||||
return ret
|
if self.__early_apply_local_log and not self.applied_local_log and self.raftLastApplied == self.raftCommitIndex:
|
||||||
|
self.applied_local_log = True
|
||||||
|
|
||||||
|
|
||||||
class KVStoreTTL(DynMemberSyncObj):
|
class KVStoreTTL(DynMemberSyncObj):
|
||||||
|
|
||||||
def __init__(self, selfAddress, partnerAddrs, conf, on_set=None, on_delete=None):
|
def __init__(self, on_ready, on_set, on_delete, **config):
|
||||||
|
self.__thread = None
|
||||||
self.__on_set = on_set
|
self.__on_set = on_set
|
||||||
self.__on_delete = on_delete
|
self.__on_delete = on_delete
|
||||||
self.__limb = {}
|
self.__limb = {}
|
||||||
self.__retry_timeout = None
|
self.__retry_timeout = None
|
||||||
self.__early_apply_local_log = selfAddress is not None
|
|
||||||
self.applied_local_log = False
|
self_addr = config.get('self_addr')
|
||||||
super(KVStoreTTL, self).__init__(selfAddress, partnerAddrs, conf)
|
partner_addrs = set(config.get('partner_addrs', []))
|
||||||
|
if config.get('patronictl'):
|
||||||
|
if self_addr:
|
||||||
|
partner_addrs.add(self_addr)
|
||||||
|
self_addr = None
|
||||||
|
|
||||||
|
# Create raft data_dir if necessary
|
||||||
|
raft_data_dir = config.get('data_dir', '')
|
||||||
|
if raft_data_dir != '':
|
||||||
|
validate_directory(raft_data_dir)
|
||||||
|
|
||||||
|
file_template = (self_addr or '')
|
||||||
|
file_template = file_template.replace(':', '_') if os.name == 'nt' else file_template
|
||||||
|
file_template = os.path.join(raft_data_dir, file_template)
|
||||||
|
conf = SyncObjConf(password=config.get('password'), autoTick=False, appendEntriesUseBatch=False,
|
||||||
|
bindAddress=config.get('bind_addr'), dnsFailCacheTime=(config.get('loop_wait') or 10),
|
||||||
|
dnsCacheTime=(config.get('ttl') or 30), commandsWaitLeader=config.get('commandsWaitLeader'),
|
||||||
|
fullDumpFile=(file_template + '.dump' if self_addr else None),
|
||||||
|
journalFile=(file_template + '.journal' if self_addr else None),
|
||||||
|
onReady=on_ready, dynamicMembershipChange=True)
|
||||||
|
|
||||||
|
super(KVStoreTTL, self).__init__(self_addr, partner_addrs, conf)
|
||||||
self.__data = {}
|
self.__data = {}
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
@@ -174,7 +168,7 @@ class KVStoreTTL(DynMemberSyncObj):
|
|||||||
|
|
||||||
if old_value and old_value['created'] != value['created']:
|
if old_value and old_value['created'] != value['created']:
|
||||||
value['created'] = value['updated']
|
value['created'] = value['updated']
|
||||||
value['index'] = self._SyncObj__raftLastApplied + 1
|
value['index'] = self.raftLastApplied + 1
|
||||||
|
|
||||||
self.__data[key] = value
|
self.__data[key] = value
|
||||||
if self.__on_set:
|
if self.__on_set:
|
||||||
@@ -241,27 +235,29 @@ class KVStoreTTL(DynMemberSyncObj):
|
|||||||
return {k: v for k, v in self.__data.items() if k.startswith(key)}
|
return {k: v for k, v in self.__data.items() if k.startswith(key)}
|
||||||
|
|
||||||
def _onTick(self, timeToWait=0.0):
|
def _onTick(self, timeToWait=0.0):
|
||||||
# The SyncObj starts applying the local log only when there is at least one node connected.
|
|
||||||
# We want to change this behavior and apply the local log even when there is nobody except us.
|
|
||||||
# It gives us at least some picture about the last known cluster state.
|
|
||||||
if self.__early_apply_local_log and not self.applied_local_log and self._SyncObj__needLoadDumpFile:
|
|
||||||
self._SyncObj__raftCommitIndex = self._SyncObj__getCurrentLogIndex()
|
|
||||||
self._SyncObj__raftCurrentTerm = self._SyncObj__getCurrentLogTerm()
|
|
||||||
|
|
||||||
super(KVStoreTTL, self)._onTick(timeToWait)
|
super(KVStoreTTL, self)._onTick(timeToWait)
|
||||||
|
|
||||||
# The SyncObj calls onReady callback only when cluster got the leader and is ready for writes.
|
|
||||||
# In some cases for us it is safe to "signal" the Raft object when the local log is fully applied.
|
|
||||||
# We are using the `applied_local_log` property for that, but not calling the callback function.
|
|
||||||
if self.__early_apply_local_log and not self.applied_local_log and self._SyncObj__raftCommitIndex != 1 and \
|
|
||||||
self._SyncObj__raftLastApplied == self._SyncObj__raftCommitIndex:
|
|
||||||
self.applied_local_log = True
|
|
||||||
|
|
||||||
if self._isLeader():
|
if self._isLeader():
|
||||||
self.__expire_keys()
|
self.__expire_keys()
|
||||||
else:
|
else:
|
||||||
self.__limb.clear()
|
self.__limb.clear()
|
||||||
|
|
||||||
|
def _autoTickThread(self):
|
||||||
|
self.__destroying = False
|
||||||
|
while not self.__destroying:
|
||||||
|
self.doTick(self.conf.autoTickPeriod)
|
||||||
|
|
||||||
|
def startAutoTick(self):
|
||||||
|
self.__thread = threading.Thread(target=self._autoTickThread)
|
||||||
|
self.__thread.daemon = True
|
||||||
|
self.__thread.start()
|
||||||
|
|
||||||
|
def destroy(self):
|
||||||
|
if self.__thread:
|
||||||
|
self.__destroying = True
|
||||||
|
self.__thread.join()
|
||||||
|
super(KVStoreTTL, self).destroy()
|
||||||
|
|
||||||
|
|
||||||
class Raft(AbstractDCS):
|
class Raft(AbstractDCS):
|
||||||
|
|
||||||
@@ -269,41 +265,24 @@ class Raft(AbstractDCS):
|
|||||||
super(Raft, self).__init__(config)
|
super(Raft, self).__init__(config)
|
||||||
self._ttl = int(config.get('ttl') or 30)
|
self._ttl = int(config.get('ttl') or 30)
|
||||||
|
|
||||||
self_addr = config.get('self_addr')
|
|
||||||
partner_addrs = config.get('partner_addrs', [])
|
|
||||||
if self._ctl:
|
|
||||||
if self_addr:
|
|
||||||
partner_addrs.append(self_addr)
|
|
||||||
self_addr = None
|
|
||||||
|
|
||||||
# Create raft data_dir if necessary
|
|
||||||
raft_data_dir = config.get('data_dir', '')
|
|
||||||
if raft_data_dir != '':
|
|
||||||
validate_directory(raft_data_dir)
|
|
||||||
|
|
||||||
ready_event = threading.Event()
|
ready_event = threading.Event()
|
||||||
file_template = os.path.join(config.get('data_dir', ''), (self_addr or ''))
|
self._sync_obj = KVStoreTTL(ready_event.set, self._on_set, self._on_delete, commandsWaitLeader=False, **config)
|
||||||
conf = SyncObjConf(password=config.get('password'), appendEntriesUseBatch=False,
|
self._sync_obj.startAutoTick()
|
||||||
bindAddress=config.get('bind_addr'), commandsWaitLeader=False,
|
|
||||||
fullDumpFile=(file_template + '.dump' if self_addr else None),
|
|
||||||
journalFile=(file_template + '.journal' if self_addr else None),
|
|
||||||
onReady=ready_event.set, dynamicMembershipChange=True)
|
|
||||||
|
|
||||||
self._sync_obj = KVStoreTTL(self_addr, partner_addrs, conf, self._on_set, self._on_delete)
|
|
||||||
while True:
|
while True:
|
||||||
ready_event.wait(5)
|
ready_event.wait(5)
|
||||||
if ready_event.isSet() or self._sync_obj.applied_local_log:
|
if ready_event.is_set() or self._sync_obj.applied_local_log:
|
||||||
break
|
break
|
||||||
else:
|
else:
|
||||||
logger.info('waiting on raft')
|
logger.info('waiting on raft')
|
||||||
self._sync_obj.forceLogCompaction()
|
|
||||||
self.set_retry_timeout(int(config.get('retry_timeout') or 10))
|
self.set_retry_timeout(int(config.get('retry_timeout') or 10))
|
||||||
|
|
||||||
def _on_set(self, key, value):
|
def _on_set(self, key, value):
|
||||||
leader = (self._sync_obj.get(self.leader_path) or {}).get('value')
|
leader = (self._sync_obj.get(self.leader_path) or {}).get('value')
|
||||||
if key == value['created'] == value['updated'] and \
|
if key == value['created'] == value['updated'] and \
|
||||||
(key.startswith(self.members_path) or key == self.leader_path and leader != self._name) or \
|
(key.startswith(self.members_path) or key == self.leader_path and leader != self._name) or \
|
||||||
key == self.leader_optime_path and leader != self._name or key in (self.config_path, self.sync_path):
|
key in (self.leader_optime_path, self.status_path) and leader != self._name or \
|
||||||
|
key in (self.config_path, self.sync_path):
|
||||||
self.event.set()
|
self.event.set()
|
||||||
|
|
||||||
def _on_delete(self, key):
|
def _on_delete(self, key):
|
||||||
@@ -320,6 +299,10 @@ class Raft(AbstractDCS):
|
|||||||
def set_retry_timeout(self, retry_timeout):
|
def set_retry_timeout(self, retry_timeout):
|
||||||
self._sync_obj.set_retry_timeout(retry_timeout)
|
self._sync_obj.set_retry_timeout(retry_timeout)
|
||||||
|
|
||||||
|
def reload_config(self, config):
|
||||||
|
super(Raft, self).reload_config(config)
|
||||||
|
globalDnsResolver().setTimeouts(self.ttl, self.loop_wait)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def member(key, value):
|
def member(key, value):
|
||||||
return Member.from_node(value['index'], os.path.basename(key), None, value['value'])
|
return Member.from_node(value['index'], os.path.basename(key), None, value['value'])
|
||||||
@@ -328,7 +311,7 @@ class Raft(AbstractDCS):
|
|||||||
prefix = self.client_path('')
|
prefix = self.client_path('')
|
||||||
response = self._sync_obj.get(prefix, recursive=True)
|
response = self._sync_obj.get(prefix, recursive=True)
|
||||||
if not response:
|
if not response:
|
||||||
return Cluster(None, None, None, None, [], None, None, None)
|
return Cluster(None, None, None, None, [], None, None, None, None)
|
||||||
nodes = {os.path.relpath(key, prefix).replace('\\', '/'): value for key, value in response.items()}
|
nodes = {os.path.relpath(key, prefix).replace('\\', '/'): value for key, value in response.items()}
|
||||||
|
|
||||||
# get initialize flag
|
# get initialize flag
|
||||||
@@ -343,9 +326,24 @@ class Raft(AbstractDCS):
|
|||||||
history = nodes.get(self._HISTORY)
|
history = nodes.get(self._HISTORY)
|
||||||
history = history and TimelineHistory.from_node(history['index'], history['value'])
|
history = history and TimelineHistory.from_node(history['index'], history['value'])
|
||||||
|
|
||||||
# get last leader operation
|
# get last know leader lsn and slots
|
||||||
last_leader_operation = nodes.get(self._LEADER_OPTIME)
|
status = nodes.get(self._STATUS)
|
||||||
last_leader_operation = 0 if last_leader_operation is None else int(last_leader_operation['value'])
|
if status:
|
||||||
|
try:
|
||||||
|
status = json.loads(status['value'])
|
||||||
|
last_lsn = status.get(self._OPTIME)
|
||||||
|
slots = status.get('slots')
|
||||||
|
except Exception:
|
||||||
|
slots = last_lsn = None
|
||||||
|
else:
|
||||||
|
last_lsn = nodes.get(self._LEADER_OPTIME)
|
||||||
|
last_lsn = last_lsn and last_lsn['value']
|
||||||
|
slots = None
|
||||||
|
|
||||||
|
try:
|
||||||
|
last_lsn = int(last_lsn)
|
||||||
|
except Exception:
|
||||||
|
last_lsn = 0
|
||||||
|
|
||||||
# get list of members
|
# get list of members
|
||||||
members = [self.member(k, n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
members = [self.member(k, n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1]
|
||||||
@@ -366,10 +364,13 @@ class Raft(AbstractDCS):
|
|||||||
sync = nodes.get(self._SYNC)
|
sync = nodes.get(self._SYNC)
|
||||||
sync = SyncState.from_node(sync and sync['index'], sync and sync['value'])
|
sync = SyncState.from_node(sync and sync['index'], sync and sync['value'])
|
||||||
|
|
||||||
return Cluster(initialize, config, leader, last_leader_operation, members, failover, sync, history)
|
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||||
|
|
||||||
def _write_leader_optime(self, last_operation):
|
def _write_leader_optime(self, last_lsn):
|
||||||
return self._sync_obj.set(self.leader_optime_path, last_operation, timeout=1)
|
return self._sync_obj.set(self.leader_optime_path, last_lsn, timeout=1)
|
||||||
|
|
||||||
|
def _write_status(self, value):
|
||||||
|
return self._sync_obj.set(self.status_path, value, timeout=1)
|
||||||
|
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
ret = self._sync_obj.set(self.leader_path, self._name, ttl=self._ttl, prevValue=self._name)
|
ret = self._sync_obj.set(self.leader_path, self._name, ttl=self._ttl, prevValue=self._name)
|
||||||
|
|||||||
+101
-43
@@ -4,11 +4,14 @@ import select
|
|||||||
import time
|
import time
|
||||||
|
|
||||||
from kazoo.client import KazooClient, KazooState, KazooRetry
|
from kazoo.client import KazooClient, KazooState, KazooRetry
|
||||||
from kazoo.exceptions import NoNodeError, NodeExistsError
|
from kazoo.exceptions import NoNodeError, NodeExistsError, SessionExpiredError
|
||||||
from kazoo.handlers.threading import SequentialThreadingHandler
|
from kazoo.handlers.threading import SequentialThreadingHandler
|
||||||
from patroni.dcs import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
from kazoo.protocol.states import KeeperState
|
||||||
from patroni.exceptions import DCSError
|
from kazoo.security import make_acl
|
||||||
from patroni.utils import deep_compare
|
|
||||||
|
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
||||||
|
from ..exceptions import DCSError
|
||||||
|
from ..utils import deep_compare
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -54,6 +57,20 @@ class PatroniSequentialThreadingHandler(SequentialThreadingHandler):
|
|||||||
raise select.error(9, str(e))
|
raise select.error(9, str(e))
|
||||||
|
|
||||||
|
|
||||||
|
class PatroniKazooClient(KazooClient):
|
||||||
|
|
||||||
|
def _call(self, request, async_object):
|
||||||
|
# Before kazoo==2.7.0 it wasn't possible to send requests to zookeeper if
|
||||||
|
# the connection is in the SUSPENDED state and Patroni was strongly relying on it.
|
||||||
|
# The https://github.com/python-zk/kazoo/pull/588 changed it, and now such requests are queued.
|
||||||
|
# We override the `_call()` method in order to keep the old behavior.
|
||||||
|
|
||||||
|
if self._state == KeeperState.CONNECTING:
|
||||||
|
async_object.set_exception(SessionExpiredError())
|
||||||
|
return False
|
||||||
|
return super(PatroniKazooClient, self)._call(request, async_object)
|
||||||
|
|
||||||
|
|
||||||
class ZooKeeper(AbstractDCS):
|
class ZooKeeper(AbstractDCS):
|
||||||
|
|
||||||
def __init__(self, config):
|
def __init__(self, config):
|
||||||
@@ -67,14 +84,28 @@ class ZooKeeper(AbstractDCS):
|
|||||||
'cert': 'certfile', 'key': 'keyfile', 'key_password': 'keyfile_password'}
|
'cert': 'certfile', 'key': 'keyfile', 'key_password': 'keyfile_password'}
|
||||||
kwargs = {v: config[k] for k, v in mapping.items() if k in config}
|
kwargs = {v: config[k] for k, v in mapping.items() if k in config}
|
||||||
|
|
||||||
self._client = KazooClient(hosts, handler=PatroniSequentialThreadingHandler(config['retry_timeout']),
|
if 'set_acls' in config:
|
||||||
timeout=config['ttl'], connection_retry=KazooRetry(max_delay=1, max_tries=-1,
|
kwargs['default_acl'] = []
|
||||||
sleep_func=time.sleep), command_retry=KazooRetry(deadline=config['retry_timeout'],
|
for principal, permissions in config['set_acls'].items():
|
||||||
max_delay=1, max_tries=-1, sleep_func=time.sleep), **kwargs)
|
normalizedPermissions = [p.upper() for p in permissions]
|
||||||
|
kwargs['default_acl'].append(make_acl(scheme='x509',
|
||||||
|
credential=principal,
|
||||||
|
read='READ' in normalizedPermissions,
|
||||||
|
write='WRITE' in normalizedPermissions,
|
||||||
|
create='CREATE' in normalizedPermissions,
|
||||||
|
delete='DELETE' in normalizedPermissions,
|
||||||
|
admin='ADMIN' in normalizedPermissions,
|
||||||
|
all='ALL' in normalizedPermissions))
|
||||||
|
|
||||||
|
self._client = PatroniKazooClient(hosts, handler=PatroniSequentialThreadingHandler(config['retry_timeout']),
|
||||||
|
timeout=config['ttl'], connection_retry=KazooRetry(max_delay=1, max_tries=-1,
|
||||||
|
sleep_func=time.sleep), command_retry=KazooRetry(max_delay=1, max_tries=-1,
|
||||||
|
deadline=config['retry_timeout'], sleep_func=time.sleep), **kwargs)
|
||||||
self._client.add_listener(self.session_listener)
|
self._client.add_listener(self.session_listener)
|
||||||
|
|
||||||
self._fetch_cluster = True
|
self._fetch_cluster = True
|
||||||
self._fetch_optime = True
|
self._fetch_status = True
|
||||||
|
self.__last_member_data = None
|
||||||
|
|
||||||
self._orig_kazoo_connect = self._client._connection._connect
|
self._orig_kazoo_connect = self._client._connection._connect
|
||||||
self._client._connection._connect = self._kazoo_connect
|
self._client._connection._connect = self._kazoo_connect
|
||||||
@@ -100,13 +131,13 @@ class ZooKeeper(AbstractDCS):
|
|||||||
if state in [KazooState.SUSPENDED, KazooState.LOST]:
|
if state in [KazooState.SUSPENDED, KazooState.LOST]:
|
||||||
self.cluster_watcher(None)
|
self.cluster_watcher(None)
|
||||||
|
|
||||||
def optime_watcher(self, event):
|
def status_watcher(self, event):
|
||||||
self._fetch_optime = True
|
self._fetch_status = True
|
||||||
self.event.set()
|
self.event.set()
|
||||||
|
|
||||||
def cluster_watcher(self, event):
|
def cluster_watcher(self, event):
|
||||||
self._fetch_cluster = True
|
self._fetch_cluster = True
|
||||||
self.optime_watcher(event)
|
self.status_watcher(event)
|
||||||
|
|
||||||
def reload_config(self, config):
|
def reload_config(self, config):
|
||||||
self.set_retry_timeout(config['retry_timeout'])
|
self.set_retry_timeout(config['retry_timeout'])
|
||||||
@@ -151,11 +182,29 @@ class ZooKeeper(AbstractDCS):
|
|||||||
except NoNodeError:
|
except NoNodeError:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
def get_leader_optime(self, leader):
|
def get_status(self, leader):
|
||||||
watch = self.optime_watcher if not leader or leader.name != self._name else None
|
watch = self.status_watcher if not leader or leader.name != self._name else None
|
||||||
optime = self.get_node(self.leader_optime_path, watch)
|
|
||||||
self._fetch_optime = False
|
status = self.get_node(self.status_path, watch)
|
||||||
return optime and int(optime[0]) or 0
|
if status:
|
||||||
|
try:
|
||||||
|
status = json.loads(status[0])
|
||||||
|
last_lsn = status.get(self._OPTIME)
|
||||||
|
slots = status.get('slots')
|
||||||
|
except Exception:
|
||||||
|
slots = last_lsn = None
|
||||||
|
else:
|
||||||
|
last_lsn = self.get_node(self.leader_optime_path, watch)
|
||||||
|
last_lsn = last_lsn and last_lsn[0]
|
||||||
|
slots = None
|
||||||
|
|
||||||
|
try:
|
||||||
|
last_lsn = int(last_lsn)
|
||||||
|
except Exception:
|
||||||
|
last_lsn = 0
|
||||||
|
|
||||||
|
self._fetch_status = False
|
||||||
|
return last_lsn, slots
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def member(name, value, znode):
|
def member(name, value, znode):
|
||||||
@@ -167,11 +216,10 @@ class ZooKeeper(AbstractDCS):
|
|||||||
except NoNodeError:
|
except NoNodeError:
|
||||||
return []
|
return []
|
||||||
|
|
||||||
def load_members(self, sync_standby):
|
def load_members(self):
|
||||||
members = []
|
members = []
|
||||||
for member in self.get_children(self.members_path, self.cluster_watcher):
|
for member in self.get_children(self.members_path, self.cluster_watcher):
|
||||||
watch = member in sync_standby and self.cluster_watcher or None
|
data = self.get_node(self.members_path + member)
|
||||||
data = self.get_node(self.members_path + member, watch)
|
|
||||||
if data is not None:
|
if data is not None:
|
||||||
members.append(self.member(member, *data))
|
members.append(self.member(member, *data))
|
||||||
return members
|
return members
|
||||||
@@ -199,8 +247,7 @@ class ZooKeeper(AbstractDCS):
|
|||||||
sync = SyncState.from_node(sync and sync[1].version, sync and sync[0])
|
sync = SyncState.from_node(sync and sync[1].version, sync and sync[0])
|
||||||
|
|
||||||
# get list of members
|
# get list of members
|
||||||
sync_standby = sync.leader == self._name and sync.members or []
|
members = self.load_members() if self._MEMBERS[:-1] in nodes else []
|
||||||
members = self.load_members(sync_standby) if self._MEMBERS[:-1] in nodes else []
|
|
||||||
|
|
||||||
# get leader
|
# get leader
|
||||||
leader = self.get_node(self.leader_path) if self._LEADER in nodes else None
|
leader = self.get_node(self.leader_path) if self._LEADER in nodes else None
|
||||||
@@ -218,14 +265,14 @@ class ZooKeeper(AbstractDCS):
|
|||||||
leader = Leader(leader[1].version, leader[1].ephemeralOwner, member)
|
leader = Leader(leader[1].version, leader[1].ephemeralOwner, member)
|
||||||
self._fetch_cluster = member.index == -1
|
self._fetch_cluster = member.index == -1
|
||||||
|
|
||||||
# get last leader operation
|
# get last known leader lsn and slots
|
||||||
last_leader_operation = self._OPTIME in nodes and self.get_leader_optime(leader)
|
last_lsn, slots = self.get_status(leader)
|
||||||
|
|
||||||
# failover key
|
# failover key
|
||||||
failover = self.get_node(self.failover_path, watch=self.cluster_watcher) if self._FAILOVER in nodes else None
|
failover = self.get_node(self.failover_path, watch=self.cluster_watcher) if self._FAILOVER in nodes else None
|
||||||
failover = failover and Failover.from_node(failover[1].version, failover[0])
|
failover = failover and Failover.from_node(failover[1].version, failover[0])
|
||||||
|
|
||||||
return Cluster(initialize, config, leader, last_leader_operation, members, failover, sync, history)
|
return Cluster(initialize, config, leader, last_lsn, members, failover, sync, history, slots)
|
||||||
|
|
||||||
def _load_cluster(self):
|
def _load_cluster(self):
|
||||||
cluster = self.cluster
|
cluster = self.cluster
|
||||||
@@ -236,15 +283,20 @@ class ZooKeeper(AbstractDCS):
|
|||||||
logger.exception('get_cluster')
|
logger.exception('get_cluster')
|
||||||
self.cluster_watcher(None)
|
self.cluster_watcher(None)
|
||||||
raise ZooKeeperError('ZooKeeper in not responding properly')
|
raise ZooKeeperError('ZooKeeper in not responding properly')
|
||||||
# Optime ZNode was updated or doesn't exist and we are not leader
|
# The /status ZNode was updated or doesn't exist
|
||||||
elif (self._fetch_optime and not self._fetch_cluster or not cluster.last_leader_operation) and\
|
elif self._fetch_status and not self._fetch_cluster or not cluster.last_lsn \
|
||||||
not (cluster.leader and cluster.leader.name == self._name):
|
or cluster.has_permanent_logical_slots(self._name, False) and not cluster.slots:
|
||||||
try:
|
# If current node is the leader just clear the event without fetching anything (we are updating the /status)
|
||||||
optime = self.get_leader_optime(cluster.leader)
|
if cluster.leader and cluster.leader.name == self._name:
|
||||||
cluster = Cluster(cluster.initialize, cluster.config, cluster.leader, optime,
|
self.event.clear()
|
||||||
cluster.members, cluster.failover, cluster.sync, cluster.history)
|
else:
|
||||||
except Exception:
|
try:
|
||||||
pass
|
last_lsn, slots = self.get_status(cluster.leader)
|
||||||
|
self.event.clear()
|
||||||
|
cluster = Cluster(cluster.initialize, cluster.config, cluster.leader, last_lsn,
|
||||||
|
cluster.members, cluster.failover, cluster.sync, cluster.history, slots)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
return cluster
|
return cluster
|
||||||
|
|
||||||
def _bypass_caches(self):
|
def _bypass_caches(self):
|
||||||
@@ -300,11 +352,11 @@ class ZooKeeper(AbstractDCS):
|
|||||||
def touch_member(self, data, permanent=False):
|
def touch_member(self, data, permanent=False):
|
||||||
cluster = self.cluster
|
cluster = self.cluster
|
||||||
member = cluster and cluster.get_member(self._name, fallback_to_leader=False)
|
member = cluster and cluster.get_member(self._name, fallback_to_leader=False)
|
||||||
encoded_data = json.dumps(data, separators=(',', ':')).encode('utf-8')
|
member_data = self.__last_member_data or member and member.data
|
||||||
if member and (self._client.client_id is not None and member.session != self._client.client_id[0] or
|
if member and (self._client.client_id is not None and member.session != self._client.client_id[0] or
|
||||||
not (deep_compare(member.data.get('tags', {}), data.get('tags', {})) and
|
not (deep_compare(member_data.get('tags', {}), data.get('tags', {})) and
|
||||||
member.data.get('version') == data.get('version') and
|
member_data.get('version') == data.get('version') and
|
||||||
member.data.get('checkpoint_after_promote') == data.get('checkpoint_after_promote'))):
|
member_data.get('checkpoint_after_promote') == data.get('checkpoint_after_promote'))):
|
||||||
try:
|
try:
|
||||||
self._client.delete_async(self.member_path).get(timeout=1)
|
self._client.delete_async(self.member_path).get(timeout=1)
|
||||||
except NoNodeError:
|
except NoNodeError:
|
||||||
@@ -313,13 +365,15 @@ class ZooKeeper(AbstractDCS):
|
|||||||
return False
|
return False
|
||||||
member = None
|
member = None
|
||||||
|
|
||||||
|
encoded_data = json.dumps(data, separators=(',', ':')).encode('utf-8')
|
||||||
if member:
|
if member:
|
||||||
if deep_compare(data, member.data):
|
if deep_compare(data, member_data):
|
||||||
return True
|
return True
|
||||||
else:
|
else:
|
||||||
try:
|
try:
|
||||||
self._client.create_async(self.member_path, encoded_data, makepath=True,
|
self._client.create_async(self.member_path, encoded_data, makepath=True,
|
||||||
ephemeral=not permanent).get(timeout=1)
|
ephemeral=not permanent).get(timeout=1)
|
||||||
|
self.__last_member_data = data
|
||||||
return True
|
return True
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
if not isinstance(e, NodeExistsError):
|
if not isinstance(e, NodeExistsError):
|
||||||
@@ -327,6 +381,7 @@ class ZooKeeper(AbstractDCS):
|
|||||||
return False
|
return False
|
||||||
try:
|
try:
|
||||||
self._client.set_async(self.member_path, encoded_data).get(timeout=1)
|
self._client.set_async(self.member_path, encoded_data).get(timeout=1)
|
||||||
|
self.__last_member_data = data
|
||||||
return True
|
return True
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('touch_member')
|
logger.exception('touch_member')
|
||||||
@@ -336,8 +391,11 @@ class ZooKeeper(AbstractDCS):
|
|||||||
def take_leader(self):
|
def take_leader(self):
|
||||||
return self.attempt_to_acquire_leader()
|
return self.attempt_to_acquire_leader()
|
||||||
|
|
||||||
def _write_leader_optime(self, last_operation):
|
def _write_leader_optime(self, last_lsn):
|
||||||
return self._set_or_create(self.leader_optime_path, last_operation)
|
return self._set_or_create(self.leader_optime_path, last_lsn)
|
||||||
|
|
||||||
|
def _write_status(self, value):
|
||||||
|
return self._set_or_create(self.status_path, value)
|
||||||
|
|
||||||
def _update_leader(self):
|
def _update_leader(self):
|
||||||
return True
|
return True
|
||||||
@@ -373,7 +431,7 @@ class ZooKeeper(AbstractDCS):
|
|||||||
return self.set_sync_state_value("{}", index)
|
return self.set_sync_state_value("{}", index)
|
||||||
|
|
||||||
def watch(self, leader_index, timeout):
|
def watch(self, leader_index, timeout):
|
||||||
ret = super(ZooKeeper, self).watch(leader_index, timeout)
|
ret = super(ZooKeeper, self).watch(leader_index, timeout + 0.5)
|
||||||
if ret and not self._fetch_optime:
|
if ret and not self._fetch_status:
|
||||||
self._fetch_cluster = True
|
self._fetch_cluster = True
|
||||||
return ret or self._fetch_cluster
|
return ret or self._fetch_cluster
|
||||||
|
|||||||
+149
-86
@@ -2,32 +2,36 @@ import datetime
|
|||||||
import functools
|
import functools
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import psycopg2
|
import six
|
||||||
import sys
|
import sys
|
||||||
import time
|
import time
|
||||||
import uuid
|
import uuid
|
||||||
|
|
||||||
from collections import namedtuple
|
from collections import namedtuple
|
||||||
from multiprocessing.pool import ThreadPool
|
from multiprocessing.pool import ThreadPool
|
||||||
from patroni.async_executor import AsyncExecutor, CriticalTask
|
|
||||||
from patroni.exceptions import DCSError, PostgresConnectionException, PatroniFatalException
|
|
||||||
from patroni.postgresql import ACTION_ON_START, ACTION_ON_ROLE_CHANGE
|
|
||||||
from patroni.postgresql.misc import postgres_version_to_int
|
|
||||||
from patroni.postgresql.rewind import Rewind
|
|
||||||
from patroni.utils import polling_loop, tzutc, is_standby_cluster as _is_standby_cluster, parse_int
|
|
||||||
from patroni.dcs import RemoteMember
|
|
||||||
from threading import RLock
|
from threading import RLock
|
||||||
|
|
||||||
|
from . import psycopg
|
||||||
|
from .async_executor import AsyncExecutor, CriticalTask
|
||||||
|
from .exceptions import DCSError, PostgresConnectionException, PatroniFatalException
|
||||||
|
from .postgresql import ACTION_ON_START, ACTION_ON_ROLE_CHANGE
|
||||||
|
from .postgresql.misc import postgres_version_to_int
|
||||||
|
from .postgresql.rewind import Rewind
|
||||||
|
from .utils import polling_loop, tzutc, is_standby_cluster as _is_standby_cluster, parse_int
|
||||||
|
from .dcs import RemoteMember
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
class _MemberStatus(namedtuple('_MemberStatus', ['member', 'reachable', 'in_recovery', 'timeline',
|
class _MemberStatus(namedtuple('_MemberStatus', ['member', 'reachable', 'in_recovery',
|
||||||
'wal_position', 'tags', 'watchdog_failed'])):
|
'dcs_last_seen', 'timeline', 'wal_position',
|
||||||
|
'tags', 'watchdog_failed'])):
|
||||||
"""Node status distilled from API response:
|
"""Node status distilled from API response:
|
||||||
|
|
||||||
member - dcs.Member object of the node
|
member - dcs.Member object of the node
|
||||||
reachable - `!False` if the node is not reachable or is not responding with correct JSON
|
reachable - `!False` if the node is not reachable or is not responding with correct JSON
|
||||||
in_recovery - `!True` if pg_is_in_recovery() == true
|
in_recovery - `!True` if pg_is_in_recovery() == true
|
||||||
|
dcs_last_seen - timestamp from JSON of last succesful communication with DCS
|
||||||
timeline - timeline value from JSON
|
timeline - timeline value from JSON
|
||||||
wal_position - maximum value of `replayed_location` or `received_location` from JSON
|
wal_position - maximum value of `replayed_location` or `received_location` from JSON
|
||||||
tags - dictionary with values of different tags (i.e. nofailover)
|
tags - dictionary with values of different tags (i.e. nofailover)
|
||||||
@@ -37,12 +41,14 @@ class _MemberStatus(namedtuple('_MemberStatus', ['member', 'reachable', 'in_reco
|
|||||||
def from_api_response(cls, member, json):
|
def from_api_response(cls, member, json):
|
||||||
is_master = json['role'] == 'master'
|
is_master = json['role'] == 'master'
|
||||||
timeline = json.get('timeline', 0)
|
timeline = json.get('timeline', 0)
|
||||||
|
dcs_last_seen = json.get('dcs_last_seen', 0)
|
||||||
wal = not is_master and max(json['xlog'].get('received_location', 0), json['xlog'].get('replayed_location', 0))
|
wal = not is_master and max(json['xlog'].get('received_location', 0), json['xlog'].get('replayed_location', 0))
|
||||||
return cls(member, True, not is_master, timeline, wal, json.get('tags', {}), json.get('watchdog_failed', False))
|
return cls(member, True, not is_master, dcs_last_seen, timeline, wal,
|
||||||
|
json.get('tags', {}), json.get('watchdog_failed', False))
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def unknown(cls, member):
|
def unknown(cls, member):
|
||||||
return cls(member, False, None, 0, 0, {}, False)
|
return cls(member, False, None, 0, 0, 0, {}, False)
|
||||||
|
|
||||||
def failover_limitation(self):
|
def failover_limitation(self):
|
||||||
"""Returns reason why this node can't promote or None if everything is ok."""
|
"""Returns reason why this node can't promote or None if everything is ok."""
|
||||||
@@ -66,7 +72,6 @@ class Ha(object):
|
|||||||
self.old_cluster = None
|
self.old_cluster = None
|
||||||
self._is_leader = False
|
self._is_leader = False
|
||||||
self._is_leader_lock = RLock()
|
self._is_leader_lock = RLock()
|
||||||
self._leader_access_is_restricted = False
|
|
||||||
self._was_paused = False
|
self._was_paused = False
|
||||||
self._leader_timeline = None
|
self._leader_timeline = None
|
||||||
self.recovering = False
|
self.recovering = False
|
||||||
@@ -85,7 +90,7 @@ class Ha(object):
|
|||||||
self._disable_sync = 0
|
self._disable_sync = 0
|
||||||
|
|
||||||
# We need following property to avoid shutdown of postgres when join of Patroni to the postgres
|
# We need following property to avoid shutdown of postgres when join of Patroni to the postgres
|
||||||
# already running as replica was aborted due to cluster not beeing initialized in DCS.
|
# already running as replica was aborted due to cluster not being initialized in DCS.
|
||||||
self._join_aborted = False
|
self._join_aborted = False
|
||||||
|
|
||||||
# used only in backoff after failing a pre_promote script
|
# used only in backoff after failing a pre_promote script
|
||||||
@@ -121,16 +126,12 @@ class Ha(object):
|
|||||||
|
|
||||||
def is_leader(self):
|
def is_leader(self):
|
||||||
with self._is_leader_lock:
|
with self._is_leader_lock:
|
||||||
return self._is_leader > time.time() and not self._leader_access_is_restricted
|
return self._is_leader > time.time()
|
||||||
|
|
||||||
def set_is_leader(self, value):
|
def set_is_leader(self, value):
|
||||||
with self._is_leader_lock:
|
with self._is_leader_lock:
|
||||||
self._is_leader = time.time() + self.dcs.ttl if value else 0
|
self._is_leader = time.time() + self.dcs.ttl if value else 0
|
||||||
|
|
||||||
def set_leader_access_is_restricted(self, value):
|
|
||||||
with self._is_leader_lock:
|
|
||||||
self._leader_access_is_restricted = value
|
|
||||||
|
|
||||||
def load_cluster_from_dcs(self):
|
def load_cluster_from_dcs(self):
|
||||||
cluster = self.dcs.get_cluster()
|
cluster = self.dcs.get_cluster()
|
||||||
|
|
||||||
@@ -145,20 +146,20 @@ class Ha(object):
|
|||||||
self._leader_timeline = None if cluster.is_unlocked() else cluster.leader.timeline
|
self._leader_timeline = None if cluster.is_unlocked() else cluster.leader.timeline
|
||||||
|
|
||||||
def acquire_lock(self):
|
def acquire_lock(self):
|
||||||
self.set_leader_access_is_restricted(self.cluster.has_permanent_logical_slots(self.state_handler.name))
|
|
||||||
ret = self.dcs.attempt_to_acquire_leader()
|
ret = self.dcs.attempt_to_acquire_leader()
|
||||||
self.set_is_leader(ret)
|
self.set_is_leader(ret)
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
def update_lock(self, write_leader_optime=False):
|
def update_lock(self, write_leader_optime=False):
|
||||||
last_operation = None
|
last_lsn = slots = None
|
||||||
if write_leader_optime:
|
if write_leader_optime:
|
||||||
try:
|
try:
|
||||||
last_operation = self.state_handler.last_operation()
|
last_lsn = self.state_handler.last_operation()
|
||||||
|
slots = self.state_handler.slots()
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('Exception when called state_handler.last_operation()')
|
logger.exception('Exception when called state_handler.last_operation()')
|
||||||
try:
|
try:
|
||||||
ret = self.dcs.update_leader(last_operation, self._leader_access_is_restricted)
|
ret = self.dcs.update_leader(last_lsn, slots)
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('Unexpected exception raised from update_leader, please report it as a BUG')
|
logger.exception('Unexpected exception raised from update_leader, please report it as a BUG')
|
||||||
ret = False
|
ret = False
|
||||||
@@ -191,9 +192,6 @@ class Ha(object):
|
|||||||
'version': self.patroni.version
|
'version': self.patroni.version
|
||||||
}
|
}
|
||||||
|
|
||||||
# following two lines are mainly necessary for consul, to avoid creation of master service
|
|
||||||
if data['role'] == 'master' and not self.is_leader():
|
|
||||||
data['role'] = 'promoted'
|
|
||||||
if self.is_leader() and not self._rewind.checkpoint_after_promote():
|
if self.is_leader() and not self._rewind.checkpoint_after_promote():
|
||||||
data['checkpoint_after_promote'] = False
|
data['checkpoint_after_promote'] = False
|
||||||
tags = self.get_effective_tags()
|
tags = self.get_effective_tags()
|
||||||
@@ -243,7 +241,7 @@ class Ha(object):
|
|||||||
logger.info('bootstrapped %s', msg)
|
logger.info('bootstrapped %s', msg)
|
||||||
cluster = self.dcs.get_cluster()
|
cluster = self.dcs.get_cluster()
|
||||||
node_to_follow = self._get_node_to_follow(cluster)
|
node_to_follow = self._get_node_to_follow(cluster)
|
||||||
return self.state_handler.follow(node_to_follow)
|
return self.state_handler.follow(node_to_follow) is not False
|
||||||
else:
|
else:
|
||||||
logger.error('failed to bootstrap %s', msg)
|
logger.error('failed to bootstrap %s', msg)
|
||||||
self.state_handler.remove_data_directory()
|
self.state_handler.remove_data_directory()
|
||||||
@@ -318,10 +316,7 @@ class Ha(object):
|
|||||||
if timeout == 0:
|
if timeout == 0:
|
||||||
# We are requested to prefer failing over to restarting master. But see first if there
|
# We are requested to prefer failing over to restarting master. But see first if there
|
||||||
# is anyone to fail over to.
|
# is anyone to fail over to.
|
||||||
members = self.cluster.members
|
if self.is_failover_possible(self.cluster.members):
|
||||||
if self.is_synchronous_mode():
|
|
||||||
members = [m for m in members if self.cluster.sync.matches(m.name)]
|
|
||||||
if self.is_failover_possible(members):
|
|
||||||
logger.info("Master crashed. Failing over.")
|
logger.info("Master crashed. Failing over.")
|
||||||
self.demote('immediate')
|
self.demote('immediate')
|
||||||
return 'stopped PostgreSQL to fail over after a crash'
|
return 'stopped PostgreSQL to fail over after a crash'
|
||||||
@@ -409,7 +404,7 @@ class Ha(object):
|
|||||||
self.state_handler.set_role('replica')
|
self.state_handler.set_role('replica')
|
||||||
|
|
||||||
if not node_to_follow:
|
if not node_to_follow:
|
||||||
return 'no action'
|
return 'no action. I am ({0})'.format(self.state_handler.name)
|
||||||
elif is_leader:
|
elif is_leader:
|
||||||
self.demote('immediate-nolock')
|
self.demote('immediate-nolock')
|
||||||
return demote_reason
|
return demote_reason
|
||||||
@@ -422,6 +417,9 @@ class Ha(object):
|
|||||||
if msg:
|
if msg:
|
||||||
return msg
|
return msg
|
||||||
|
|
||||||
|
if not self.is_paused():
|
||||||
|
self.state_handler.handle_parameter_change()
|
||||||
|
|
||||||
role = 'standby_leader' if isinstance(node_to_follow, RemoteMember) and self.has_lock(False) else 'replica'
|
role = 'standby_leader' if isinstance(node_to_follow, RemoteMember) and self.has_lock(False) else 'replica'
|
||||||
# It might happen that leader key in the standby cluster references non-exiting member.
|
# It might happen that leader key in the standby cluster references non-exiting member.
|
||||||
# In this case it is safe to continue running without changing recovery.conf
|
# In this case it is safe to continue running without changing recovery.conf
|
||||||
@@ -553,7 +551,7 @@ class Ha(object):
|
|||||||
if master_timeline == 1:
|
if master_timeline == 1:
|
||||||
if cluster_history:
|
if cluster_history:
|
||||||
self.dcs.set_history_value('[]')
|
self.dcs.set_history_value('[]')
|
||||||
elif not cluster_history or cluster_history[-1][0] != master_timeline - 1 or len(cluster_history[-1]) != 4:
|
elif not cluster_history or cluster_history[-1][0] != master_timeline - 1 or len(cluster_history[-1]) != 5:
|
||||||
cluster_history = {line[0]: line for line in cluster_history or []}
|
cluster_history = {line[0]: line for line in cluster_history or []}
|
||||||
history = self.state_handler.get_history(master_timeline)
|
history = self.state_handler.get_history(master_timeline)
|
||||||
if history and self.cluster.config:
|
if history and self.cluster.config:
|
||||||
@@ -561,9 +559,11 @@ class Ha(object):
|
|||||||
for line in history:
|
for line in history:
|
||||||
# enrich current history with promotion timestamps stored in DCS
|
# enrich current history with promotion timestamps stored in DCS
|
||||||
if len(line) == 3 and line[0] in cluster_history \
|
if len(line) == 3 and line[0] in cluster_history \
|
||||||
and len(cluster_history[line[0]]) == 4 \
|
and len(cluster_history[line[0]]) >= 4 \
|
||||||
and cluster_history[line[0]][1] == line[1]:
|
and cluster_history[line[0]][1] == line[1]:
|
||||||
line.append(cluster_history[line[0]][3])
|
line.append(cluster_history[line[0]][3])
|
||||||
|
if len(cluster_history[line[0]]) == 5:
|
||||||
|
line.append(cluster_history[line[0]][4])
|
||||||
self.dcs.set_history_value(json.dumps(history, separators=(',', ':')))
|
self.dcs.set_history_value(json.dumps(history, separators=(',', ':')))
|
||||||
|
|
||||||
def enforce_follow_remote_master(self, message):
|
def enforce_follow_remote_master(self, message):
|
||||||
@@ -613,8 +613,6 @@ class Ha(object):
|
|||||||
return 'Postponing promotion because synchronous replication state was updated by somebody else'
|
return 'Postponing promotion because synchronous replication state was updated by somebody else'
|
||||||
self.state_handler.config.set_synchronous_standby(['*'] if self.is_synchronous_mode_strict() else [])
|
self.state_handler.config.set_synchronous_standby(['*'] if self.is_synchronous_mode_strict() else [])
|
||||||
if self.state_handler.role != 'master':
|
if self.state_handler.role != 'master':
|
||||||
self.set_leader_access_is_restricted(self.cluster.has_permanent_logical_slots(self.state_handler.name))
|
|
||||||
|
|
||||||
def on_success():
|
def on_success():
|
||||||
self._rewind.reset_state()
|
self._rewind.reset_state()
|
||||||
logger.info("cleared rewind state after becoming the leader")
|
logger.info("cleared rewind state after becoming the leader")
|
||||||
@@ -622,8 +620,7 @@ class Ha(object):
|
|||||||
with self._async_response:
|
with self._async_response:
|
||||||
self._async_response.reset()
|
self._async_response.reset()
|
||||||
self._async_executor.try_run_async('promote', self.state_handler.promote,
|
self._async_executor.try_run_async('promote', self.state_handler.promote,
|
||||||
args=(self.dcs.loop_wait, self._async_response, on_success,
|
args=(self.dcs.loop_wait, self._async_response, on_success))
|
||||||
self._leader_access_is_restricted))
|
|
||||||
return promote_message
|
return promote_message
|
||||||
|
|
||||||
def fetch_node_status(self, member):
|
def fetch_node_status(self, member):
|
||||||
@@ -653,14 +650,13 @@ class Ha(object):
|
|||||||
:param wal_position: Current wal position.
|
:param wal_position: Current wal position.
|
||||||
:returns True when node is lagging
|
:returns True when node is lagging
|
||||||
"""
|
"""
|
||||||
lag = (self.cluster.last_leader_operation or 0) - wal_position
|
lag = (self.cluster.last_lsn or 0) - wal_position
|
||||||
return lag > self.patroni.config.get('maximum_lag_on_failover', 0)
|
return lag > self.patroni.config.get('maximum_lag_on_failover', 0)
|
||||||
|
|
||||||
def _is_healthiest_node(self, members, check_replication_lag=True):
|
def _is_healthiest_node(self, members, check_replication_lag=True):
|
||||||
"""This method tries to determine whether I am healthy enough to became a new leader candidate or not."""
|
"""This method tries to determine whether I am healthy enough to became a new leader candidate or not."""
|
||||||
|
|
||||||
# We don't call `last_operation()` here because it returns a string
|
my_wal_position = self.state_handler.last_operation()
|
||||||
_, my_wal_position, _ = self.state_handler.timeline_wal_position()
|
|
||||||
if check_replication_lag and self.is_lagging(my_wal_position):
|
if check_replication_lag and self.is_lagging(my_wal_position):
|
||||||
logger.info('My wal position exceeds maximum replication lag')
|
logger.info('My wal position exceeds maximum replication lag')
|
||||||
return False # Too far behind last reported wal position on master
|
return False # Too far behind last reported wal position on master
|
||||||
@@ -690,16 +686,21 @@ class Ha(object):
|
|||||||
logger.info('Ignoring the former leader being ahead of us')
|
logger.info('Ignoring the former leader being ahead of us')
|
||||||
return True
|
return True
|
||||||
|
|
||||||
def is_failover_possible(self, members):
|
def is_failover_possible(self, members, check_synchronous=True, cluster_lsn=None):
|
||||||
ret = False
|
ret = False
|
||||||
cluster_timeline = self.cluster.timeline
|
cluster_timeline = self.cluster.timeline
|
||||||
members = [m for m in members if m.name != self.state_handler.name and not m.nofailover and m.api_url]
|
members = [m for m in members if m.name != self.state_handler.name and not m.nofailover and m.api_url]
|
||||||
|
if check_synchronous and self.is_synchronous_mode():
|
||||||
|
members = [m for m in members if self.cluster.sync.matches(m.name)]
|
||||||
if members:
|
if members:
|
||||||
for st in self.fetch_nodes_statuses(members):
|
for st in self.fetch_nodes_statuses(members):
|
||||||
not_allowed_reason = st.failover_limitation()
|
not_allowed_reason = st.failover_limitation()
|
||||||
if not_allowed_reason:
|
if not_allowed_reason:
|
||||||
logger.info('Member %s is %s', st.member.name, not_allowed_reason)
|
logger.info('Member %s is %s', st.member.name, not_allowed_reason)
|
||||||
elif self.is_lagging(st.wal_position):
|
elif not isinstance(st.wal_position, six.integer_types):
|
||||||
|
logger.info('Member %s does not report wal_position', st.member.name)
|
||||||
|
elif cluster_lsn and st.wal_position < cluster_lsn or\
|
||||||
|
not cluster_lsn and self.is_lagging(st.wal_position):
|
||||||
logger.info('Member %s exceeds maximum replication lag', st.member.name)
|
logger.info('Member %s exceeds maximum replication lag', st.member.name)
|
||||||
elif self.check_timeline() and (not st.timeline or st.timeline < cluster_timeline):
|
elif self.check_timeline() and (not st.timeline or st.timeline < cluster_timeline):
|
||||||
logger.info('Timeline %s of member %s is behind the cluster timeline %s',
|
logger.info('Timeline %s of member %s is behind the cluster timeline %s',
|
||||||
@@ -783,6 +784,10 @@ class Ha(object):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
if self.cluster.failover:
|
if self.cluster.failover:
|
||||||
|
# When doing a switchover in synchronous mode only synchronous nodes and former leader are allowed to race
|
||||||
|
if self.is_synchronous_mode() and self.cluster.failover.leader and \
|
||||||
|
self.cluster.failover.candidate and not self.cluster.sync.matches(self.state_handler.name):
|
||||||
|
return False
|
||||||
return self.manual_failover_process_no_leader()
|
return self.manual_failover_process_no_leader()
|
||||||
|
|
||||||
if not self.watchdog.is_healthy:
|
if not self.watchdog.is_healthy:
|
||||||
@@ -791,7 +796,7 @@ class Ha(object):
|
|||||||
|
|
||||||
# When in sync mode, only last known master and sync standby are allowed to promote automatically.
|
# When in sync mode, only last known master and sync standby are allowed to promote automatically.
|
||||||
all_known_members = self.cluster.members + self.old_cluster.members
|
all_known_members = self.cluster.members + self.old_cluster.members
|
||||||
if self.is_synchronous_mode() and self.cluster.sync.leader:
|
if self.is_synchronous_mode() and self.cluster.sync and self.cluster.sync.leader:
|
||||||
if not self.cluster.sync.matches(self.state_handler.name):
|
if not self.cluster.sync.matches(self.state_handler.name):
|
||||||
return False
|
return False
|
||||||
# pick between synchronous candidates so we minimize unnecessary failovers/demotions
|
# pick between synchronous candidates so we minimize unnecessary failovers/demotions
|
||||||
@@ -802,13 +807,13 @@ class Ha(object):
|
|||||||
|
|
||||||
return self._is_healthiest_node(members.values())
|
return self._is_healthiest_node(members.values())
|
||||||
|
|
||||||
def _delete_leader(self, last_operation=None):
|
def _delete_leader(self, last_lsn=None):
|
||||||
self.set_is_leader(False)
|
self.set_is_leader(False)
|
||||||
self.dcs.delete_leader(last_operation)
|
self.dcs.delete_leader(last_lsn)
|
||||||
self.dcs.reset_cluster()
|
self.dcs.reset_cluster()
|
||||||
|
|
||||||
def release_leader_key_voluntarily(self, last_operation=None):
|
def release_leader_key_voluntarily(self, last_lsn=None):
|
||||||
self._delete_leader(last_operation)
|
self._delete_leader(last_lsn)
|
||||||
self.touch_member()
|
self.touch_member()
|
||||||
logger.info("Leader key released")
|
logger.info("Leader key released")
|
||||||
|
|
||||||
@@ -830,23 +835,44 @@ class Ha(object):
|
|||||||
'immediate-nolock': dict(stop='immediate', checkpoint=False, release=False, offline=False, async_req=True),
|
'immediate-nolock': dict(stop='immediate', checkpoint=False, release=False, offline=False, async_req=True),
|
||||||
}[mode]
|
}[mode]
|
||||||
|
|
||||||
|
logger.info('Demoting self (%s)', mode)
|
||||||
|
|
||||||
self._rewind.trigger_check_diverged_lsn()
|
self._rewind.trigger_check_diverged_lsn()
|
||||||
|
|
||||||
|
status = {'released': False}
|
||||||
|
|
||||||
|
def on_shutdown(checkpoint_location):
|
||||||
|
# Postmaster is still running, but pg_control already reports clean "shut down".
|
||||||
|
# It could happen if Postgres is still archiving the backlog of WAL files.
|
||||||
|
# If we know that there are replicas that received the shutdown checkpoint
|
||||||
|
# location, we can remove the leader key and allow them to start leader race.
|
||||||
|
if self.is_failover_possible(self.cluster.members, cluster_lsn=checkpoint_location):
|
||||||
|
self.state_handler.set_role('demoted')
|
||||||
|
with self._async_executor:
|
||||||
|
self.release_leader_key_voluntarily(checkpoint_location)
|
||||||
|
status['released'] = True
|
||||||
|
|
||||||
self.state_handler.stop(mode_control['stop'], checkpoint=mode_control['checkpoint'],
|
self.state_handler.stop(mode_control['stop'], checkpoint=mode_control['checkpoint'],
|
||||||
on_safepoint=self.watchdog.disable if self.watchdog.is_running else None,
|
on_safepoint=self.watchdog.disable if self.watchdog.is_running else None,
|
||||||
|
on_shutdown=on_shutdown if mode_control['release'] else None,
|
||||||
stop_timeout=self.master_stop_timeout())
|
stop_timeout=self.master_stop_timeout())
|
||||||
self.state_handler.set_role('demoted')
|
self.state_handler.set_role('demoted')
|
||||||
self.set_is_leader(False)
|
self.set_is_leader(False)
|
||||||
|
|
||||||
if mode_control['release']:
|
if mode_control['release']:
|
||||||
checkpoint_location = self.state_handler.latest_checkpoint_location() if mode == 'graceful' else None
|
if not status['released']:
|
||||||
with self._async_executor:
|
checkpoint_location = self.state_handler.latest_checkpoint_location() if mode == 'graceful' else None
|
||||||
self.release_leader_key_voluntarily(checkpoint_location)
|
with self._async_executor:
|
||||||
|
self.release_leader_key_voluntarily(checkpoint_location)
|
||||||
time.sleep(2) # Give a time to somebody to take the leader lock
|
time.sleep(2) # Give a time to somebody to take the leader lock
|
||||||
if mode_control['offline']:
|
if mode_control['offline']:
|
||||||
node_to_follow, leader = None, None
|
node_to_follow, leader = None, None
|
||||||
else:
|
else:
|
||||||
cluster = self.dcs.get_cluster()
|
try:
|
||||||
node_to_follow, leader = self._get_node_to_follow(cluster), cluster.leader
|
cluster = self.dcs.get_cluster()
|
||||||
|
node_to_follow, leader = self._get_node_to_follow(cluster), cluster.leader
|
||||||
|
except Exception:
|
||||||
|
node_to_follow, leader = None, None
|
||||||
|
|
||||||
# FIXME: with mode offline called from DCS exception handler and handle_long_action_in_progress
|
# FIXME: with mode offline called from DCS exception handler and handle_long_action_in_progress
|
||||||
# there could be an async action already running, calling follow from here will lead
|
# there could be an async action already running, calling follow from here will lead
|
||||||
@@ -924,7 +950,7 @@ class Ha(object):
|
|||||||
else:
|
else:
|
||||||
members = [m for m in self.cluster.members
|
members = [m for m in self.cluster.members
|
||||||
if not failover.candidate or m.name == failover.candidate]
|
if not failover.candidate or m.name == failover.candidate]
|
||||||
if self.is_failover_possible(members): # check that there are healthy members
|
if self.is_failover_possible(members, False): # check that there are healthy members
|
||||||
ret = self._async_executor.try_run_async('manual failover: demote', self.demote, ('graceful',))
|
ret = self._async_executor.try_run_async('manual failover: demote', self.demote, ('graceful',))
|
||||||
return ret or 'manual failover: demoting myself'
|
return ret or 'manual failover: demoting myself'
|
||||||
else:
|
else:
|
||||||
@@ -986,13 +1012,9 @@ class Ha(object):
|
|||||||
if self.cluster.failover and self.cluster.failover.candidate == self.state_handler.name:
|
if self.cluster.failover and self.cluster.failover.candidate == self.state_handler.name:
|
||||||
return 'waiting to become master after promote...'
|
return 'waiting to become master after promote...'
|
||||||
|
|
||||||
self._delete_leader()
|
if not self.is_standby_cluster():
|
||||||
return 'removed leader lock because postgres is not running as master'
|
self._delete_leader()
|
||||||
|
return 'removed leader lock because postgres is not running as master'
|
||||||
if self.state_handler.is_leader() and self._leader_access_is_restricted:
|
|
||||||
self.state_handler.slots_handler.sync_replication_slots(self.cluster)
|
|
||||||
self.state_handler.call_nowait(ACTION_ON_ROLE_CHANGE)
|
|
||||||
self.set_leader_access_is_restricted(False)
|
|
||||||
|
|
||||||
if self.update_lock(True):
|
if self.update_lock(True):
|
||||||
msg = self.process_manual_failover_from_leader()
|
msg = self.process_manual_failover_from_leader()
|
||||||
@@ -1006,14 +1028,14 @@ class Ha(object):
|
|||||||
# in case of standby cluster we don't really need to
|
# in case of standby cluster we don't really need to
|
||||||
# enforce anything, since the leader is not a master.
|
# enforce anything, since the leader is not a master.
|
||||||
# So just remind the role.
|
# So just remind the role.
|
||||||
msg = 'no action. i am the standby leader with the lock' \
|
msg = 'no action. I am ({0}), the standby leader with the lock'.format(self.state_handler.name) \
|
||||||
if self.state_handler.role == 'standby_leader' else \
|
if self.state_handler.role == 'standby_leader' else \
|
||||||
'promoted self to a standby leader because i had the session lock'
|
'promoted self to a standby leader because i had the session lock'
|
||||||
return self.enforce_follow_remote_master(msg)
|
return self.enforce_follow_remote_master(msg)
|
||||||
else:
|
else:
|
||||||
return self.enforce_master_role(
|
return self.enforce_master_role(
|
||||||
'no action. i am the leader with the lock',
|
'no action. I am ({0}), the leader with the lock'.format(self.state_handler.name),
|
||||||
'promoted self to leader because i had the session lock'
|
'promoted self to leader because I had the session lock'
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
# Either there is no connection to DCS or someone else acquired the lock
|
# Either there is no connection to DCS or someone else acquired the lock
|
||||||
@@ -1026,12 +1048,15 @@ class Ha(object):
|
|||||||
else:
|
else:
|
||||||
return 'not promoting because failed to update leader lock in DCS'
|
return 'not promoting because failed to update leader lock in DCS'
|
||||||
else:
|
else:
|
||||||
logger.info('does not have lock')
|
logger.debug('does not have lock')
|
||||||
|
lock_owner = self.cluster.leader and self.cluster.leader.name
|
||||||
if self.is_standby_cluster():
|
if self.is_standby_cluster():
|
||||||
return self.follow('cannot be a real master in standby cluster',
|
return self.follow('cannot be a real primary in a standby cluster',
|
||||||
'no action. i am a secondary and i am following a standby leader', refresh=False)
|
'no action. I am ({0}), a secondary, and following a standby leader ({1})'.format(
|
||||||
return self.follow('demoting self because i do not have the lock and i was a leader',
|
self.state_handler.name, lock_owner), refresh=False)
|
||||||
'no action. i am a secondary and i am following a leader', refresh=False)
|
return self.follow('demoting self because I do not have the lock and I was a leader',
|
||||||
|
'no action. I am ({0}), a secondary, and following a leader ({1})'.format(
|
||||||
|
self.state_handler.name, lock_owner), refresh=False)
|
||||||
|
|
||||||
def evaluate_scheduled_restart(self):
|
def evaluate_scheduled_restart(self):
|
||||||
if self._async_executor.busy: # Restart already in progress
|
if self._async_executor.busy: # Restart already in progress
|
||||||
@@ -1227,9 +1252,6 @@ class Ha(object):
|
|||||||
self._delete_leader()
|
self._delete_leader()
|
||||||
return 'removed leader key after trying and failing to start postgres'
|
return 'removed leader key after trying and failing to start postgres'
|
||||||
return 'failed to start postgres'
|
return 'failed to start postgres'
|
||||||
self._crash_recovery_executed = False
|
|
||||||
if self._rewind.executed and not self._rewind.failed:
|
|
||||||
self._rewind.reset_state()
|
|
||||||
return None
|
return None
|
||||||
|
|
||||||
def cancel_initialization(self):
|
def cancel_initialization(self):
|
||||||
@@ -1259,9 +1281,9 @@ class Ha(object):
|
|||||||
if not self.watchdog.activate():
|
if not self.watchdog.activate():
|
||||||
logger.error('Cancelling bootstrap because watchdog activation failed')
|
logger.error('Cancelling bootstrap because watchdog activation failed')
|
||||||
self.cancel_initialization()
|
self.cancel_initialization()
|
||||||
|
self._rewind.ensure_checkpoint_after_promote(self.wakeup)
|
||||||
self.dcs.initialize(create_new=(self.cluster.initialize is None), sysid=self.state_handler.sysid)
|
self.dcs.initialize(create_new=(self.cluster.initialize is None), sysid=self.state_handler.sysid)
|
||||||
self.dcs.set_config_value(json.dumps(self.patroni.config.dynamic_configuration, separators=(',', ':')))
|
self.dcs.set_config_value(json.dumps(self.patroni.config.dynamic_configuration, separators=(',', ':')))
|
||||||
self.state_handler.slots_handler.sync_replication_slots(self.cluster)
|
|
||||||
self.dcs.take_leader()
|
self.dcs.take_leader()
|
||||||
self.set_is_leader(True)
|
self.set_is_leader(True)
|
||||||
self.state_handler.call_nowait(ACTION_ON_START)
|
self.state_handler.call_nowait(ACTION_ON_START)
|
||||||
@@ -1317,8 +1339,12 @@ class Ha(object):
|
|||||||
def _run_cycle(self):
|
def _run_cycle(self):
|
||||||
dcs_failed = False
|
dcs_failed = False
|
||||||
try:
|
try:
|
||||||
self.state_handler.reset_cluster_info_state()
|
try:
|
||||||
self.load_cluster_from_dcs()
|
self.load_cluster_from_dcs()
|
||||||
|
self.state_handler.reset_cluster_info_state(self.cluster, self.patroni.nofailover)
|
||||||
|
except Exception:
|
||||||
|
self.state_handler.reset_cluster_info_state(None, self.patroni.nofailover)
|
||||||
|
raise
|
||||||
|
|
||||||
if self.is_paused():
|
if self.is_paused():
|
||||||
self.watchdog.disable()
|
self.watchdog.disable()
|
||||||
@@ -1350,12 +1376,24 @@ class Ha(object):
|
|||||||
if self.state_handler.bootstrapping:
|
if self.state_handler.bootstrapping:
|
||||||
return self.post_bootstrap()
|
return self.post_bootstrap()
|
||||||
|
|
||||||
if self.recovering and not self._rewind.is_needed:
|
if self.recovering:
|
||||||
self.recovering = False
|
self.recovering = False
|
||||||
# Check if we tried to recover and failed
|
|
||||||
msg = self.post_recover()
|
if not self._rewind.is_needed:
|
||||||
if msg is not None:
|
# Check if we tried to recover from postgres crash and failed
|
||||||
return msg
|
msg = self.post_recover()
|
||||||
|
if msg is not None:
|
||||||
|
return msg
|
||||||
|
|
||||||
|
# Reset some states after postgres successfully started up
|
||||||
|
self._crash_recovery_executed = False
|
||||||
|
if self._rewind.executed and not self._rewind.failed:
|
||||||
|
self._rewind.reset_state()
|
||||||
|
|
||||||
|
# The Raft cluster without a quorum takes a bit of time to stabilize.
|
||||||
|
# Therefore we want to postpone the leader race if we just started up.
|
||||||
|
if self.cluster.is_unlocked() and self.dcs.__class__.__name__ == 'Raft':
|
||||||
|
return 'started as a secondary'
|
||||||
|
|
||||||
# is data directory empty?
|
# is data directory empty?
|
||||||
if self.state_handler.data_directory_empty():
|
if self.state_handler.data_directory_empty():
|
||||||
@@ -1424,20 +1462,28 @@ class Ha(object):
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
if self.cluster.is_unlocked():
|
if self.cluster.is_unlocked():
|
||||||
return self.process_unhealthy_cluster()
|
ret = self.process_unhealthy_cluster()
|
||||||
else:
|
else:
|
||||||
msg = self.process_healthy_cluster()
|
msg = self.process_healthy_cluster()
|
||||||
return self.evaluate_scheduled_restart() or msg
|
ret = self.evaluate_scheduled_restart() or msg
|
||||||
finally:
|
finally:
|
||||||
# we might not have a valid PostgreSQL connection here if another thread
|
# we might not have a valid PostgreSQL connection here if another thread
|
||||||
# stops PostgreSQL, therefore, we only reload replication slots if no
|
# stops PostgreSQL, therefore, we only reload replication slots if no
|
||||||
# asynchronous processes are running (should be always the case for the master)
|
# asynchronous processes are running (should be always the case for the master)
|
||||||
if not self._async_executor.busy and not self.state_handler.is_starting():
|
if not self._async_executor.busy and not self.state_handler.is_starting():
|
||||||
self.state_handler.slots_handler.sync_replication_slots(self.cluster)
|
create_slots = self.state_handler.slots_handler.sync_replication_slots(self.cluster,
|
||||||
|
self.patroni.nofailover)
|
||||||
if not self.state_handler.cb_called:
|
if not self.state_handler.cb_called:
|
||||||
if not self.state_handler.is_leader():
|
if not self.state_handler.is_leader():
|
||||||
self._rewind.trigger_check_diverged_lsn()
|
self._rewind.trigger_check_diverged_lsn()
|
||||||
self.state_handler.call_nowait(ACTION_ON_START)
|
self.state_handler.call_nowait(ACTION_ON_START)
|
||||||
|
if create_slots and self.cluster.leader:
|
||||||
|
err = self._async_executor.try_run_async('copy_logical_slots',
|
||||||
|
self.state_handler.slots_handler.copy_logical_slots,
|
||||||
|
args=(self.cluster.leader, create_slots))
|
||||||
|
if not err:
|
||||||
|
ret = 'Copying logical slots {0} from the primary'.format(create_slots)
|
||||||
|
return ret
|
||||||
except DCSError:
|
except DCSError:
|
||||||
dcs_failed = True
|
dcs_failed = True
|
||||||
logger.error('Error communicating with DCS')
|
logger.error('Error communicating with DCS')
|
||||||
@@ -1445,7 +1491,7 @@ class Ha(object):
|
|||||||
self.demote('offline')
|
self.demote('offline')
|
||||||
return 'demoted self because DCS is not accessible and i was a leader'
|
return 'demoted self because DCS is not accessible and i was a leader'
|
||||||
return 'DCS is not accessible'
|
return 'DCS is not accessible'
|
||||||
except (psycopg2.Error, PostgresConnectionException):
|
except (psycopg.Error, PostgresConnectionException):
|
||||||
return 'Error communicating with PostgreSQL. Will try again later'
|
return 'Error communicating with PostgreSQL. Will try again later'
|
||||||
finally:
|
finally:
|
||||||
if not dcs_failed:
|
if not dcs_failed:
|
||||||
@@ -1468,14 +1514,31 @@ class Ha(object):
|
|||||||
self.watchdog.disable()
|
self.watchdog.disable()
|
||||||
elif not self._join_aborted:
|
elif not self._join_aborted:
|
||||||
# FIXME: If stop doesn't reach safepoint quickly enough keepalive is triggered. If shutdown checkpoint
|
# FIXME: If stop doesn't reach safepoint quickly enough keepalive is triggered. If shutdown checkpoint
|
||||||
# takes longer than ttl, then leader key is lost and replication might not have sent out all xlog.
|
# takes longer than ttl, then leader key is lost and replication might not have sent out all WAL.
|
||||||
# This might not be the desired behavior of users, as a graceful shutdown of the host can mean lost data.
|
# This might not be the desired behavior of users, as a graceful shutdown of the host can mean lost data.
|
||||||
# We probably need to something smarter here.
|
# We probably need to something smarter here.
|
||||||
disable_wd = self.watchdog.disable if self.watchdog.is_running else None
|
disable_wd = self.watchdog.disable if self.watchdog.is_running else None
|
||||||
|
|
||||||
|
status = {'deleted': False}
|
||||||
|
|
||||||
|
def _on_shutdown(checkpoint_location):
|
||||||
|
if self.is_leader():
|
||||||
|
# Postmaster is still running, but pg_control already reports clean "shut down".
|
||||||
|
# It could happen if Postgres is still archiving the backlog of WAL files.
|
||||||
|
# If we know that there are replicas that received the shutdown checkpoint
|
||||||
|
# location, we can remove the leader key and allow them to start leader race.
|
||||||
|
if self.is_failover_possible(self.cluster.members, cluster_lsn=checkpoint_location):
|
||||||
|
self.dcs.delete_leader(checkpoint_location)
|
||||||
|
status['deleted'] = True
|
||||||
|
else:
|
||||||
|
self.dcs.write_leader_optime(checkpoint_location)
|
||||||
|
|
||||||
|
on_shutdown = _on_shutdown if self.is_leader() else None
|
||||||
self.while_not_sync_standby(lambda: self.state_handler.stop(checkpoint=False, on_safepoint=disable_wd,
|
self.while_not_sync_standby(lambda: self.state_handler.stop(checkpoint=False, on_safepoint=disable_wd,
|
||||||
|
on_shutdown=on_shutdown,
|
||||||
stop_timeout=self.master_stop_timeout()))
|
stop_timeout=self.master_stop_timeout()))
|
||||||
if not self.state_handler.is_running():
|
if not self.state_handler.is_running():
|
||||||
if self.is_leader():
|
if self.is_leader() and not status['deleted']:
|
||||||
checkpoint_location = self.state_handler.latest_checkpoint_location()
|
checkpoint_location = self.state_handler.latest_checkpoint_location()
|
||||||
self.dcs.delete_leader(checkpoint_location)
|
self.dcs.delete_leader(checkpoint_location)
|
||||||
self.touch_member()
|
self.touch_member()
|
||||||
|
|||||||
+14
-1
@@ -166,6 +166,8 @@ class PatroniLogger(Thread):
|
|||||||
self._root_logger.addHandler(self._queue_handler)
|
self._root_logger.addHandler(self._queue_handler)
|
||||||
self._root_logger.removeHandler(self._proxy_handler)
|
self._root_logger.removeHandler(self._proxy_handler)
|
||||||
|
|
||||||
|
prev_record = None
|
||||||
|
|
||||||
while True:
|
while True:
|
||||||
self._close_old_handlers()
|
self._close_old_handlers()
|
||||||
|
|
||||||
@@ -173,7 +175,18 @@ class PatroniLogger(Thread):
|
|||||||
if record is None:
|
if record is None:
|
||||||
break
|
break
|
||||||
|
|
||||||
self.log_handler.handle(record)
|
if self._root_logger.level == logging.INFO:
|
||||||
|
if record.msg.startswith('Lock owner: '):
|
||||||
|
prev_record, record = record, None
|
||||||
|
else:
|
||||||
|
if prev_record and prev_record.thread == record.thread:
|
||||||
|
if not (record.msg.startswith('no action. ') or record.msg.startswith('PAUSE: no action')):
|
||||||
|
self.log_handler.handle(prev_record)
|
||||||
|
prev_record = None
|
||||||
|
|
||||||
|
if record:
|
||||||
|
self.log_handler.handle(record)
|
||||||
|
|
||||||
self._queue_handler.queue.task_done()
|
self._queue_handler.queue.task_done()
|
||||||
|
|
||||||
def shutdown(self):
|
def shutdown(self):
|
||||||
|
|||||||
+173
-56
@@ -1,28 +1,31 @@
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import psycopg2
|
import re
|
||||||
import shlex
|
import shlex
|
||||||
import shutil
|
import shutil
|
||||||
|
import six
|
||||||
import subprocess
|
import subprocess
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from copy import deepcopy
|
from copy import deepcopy
|
||||||
from dateutil import tz
|
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from patroni.postgresql.callback_executor import CallbackExecutor
|
from dateutil import tz
|
||||||
from patroni.postgresql.bootstrap import Bootstrap
|
|
||||||
from patroni.postgresql.cancellable import CancellableSubprocess
|
|
||||||
from patroni.postgresql.config import ConfigHandler, mtime
|
|
||||||
from patroni.postgresql.connection import Connection, get_connection_cursor
|
|
||||||
from patroni.postgresql.misc import parse_history, parse_lsn, postgres_major_version_to_int
|
|
||||||
from patroni.postgresql.postmaster import PostmasterProcess
|
|
||||||
from patroni.postgresql.slots import SlotsHandler
|
|
||||||
from patroni.exceptions import PostgresConnectionException
|
|
||||||
from patroni.utils import Retry, RetryFailedError, polling_loop, data_directory_is_empty, parse_int
|
|
||||||
from psutil import TimeoutExpired
|
from psutil import TimeoutExpired
|
||||||
from threading import current_thread, Lock
|
from threading import current_thread, Lock
|
||||||
|
|
||||||
|
from .bootstrap import Bootstrap
|
||||||
|
from .callback_executor import CallbackExecutor
|
||||||
|
from .cancellable import CancellableSubprocess
|
||||||
|
from .config import ConfigHandler, mtime
|
||||||
|
from .connection import Connection, get_connection_cursor
|
||||||
|
from .misc import parse_history, parse_lsn, postgres_major_version_to_int
|
||||||
|
from .postmaster import PostmasterProcess
|
||||||
|
from .slots import SlotsHandler
|
||||||
|
from .. import psycopg
|
||||||
|
from ..exceptions import PostgresConnectionException
|
||||||
|
from ..utils import Retry, RetryFailedError, polling_loop, data_directory_is_empty, parse_int
|
||||||
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -48,13 +51,13 @@ def null_context():
|
|||||||
|
|
||||||
class Postgresql(object):
|
class Postgresql(object):
|
||||||
|
|
||||||
POSTMASTER_START_TIME = "pg_catalog.to_char(pg_catalog.pg_postmaster_start_time(), 'YYYY-MM-DD HH24:MI:SS.MS TZ')"
|
POSTMASTER_START_TIME = "pg_catalog.pg_postmaster_start_time()"
|
||||||
TL_LSN = ("CASE WHEN pg_catalog.pg_is_in_recovery() THEN 0 "
|
TL_LSN = ("CASE WHEN pg_catalog.pg_is_in_recovery() THEN 0 "
|
||||||
"ELSE ('x' || pg_catalog.substr(pg_catalog.pg_{0}file_name("
|
"ELSE ('x' || pg_catalog.substr(pg_catalog.pg_{0}file_name("
|
||||||
"pg_catalog.pg_current_{0}_{1}()), 1, 8))::bit(32)::int END, " # master timeline
|
"pg_catalog.pg_current_{0}_{1}()), 1, 8))::bit(32)::int END, " # master timeline
|
||||||
"CASE WHEN pg_catalog.pg_is_in_recovery() THEN 0 "
|
"CASE WHEN pg_catalog.pg_is_in_recovery() THEN 0 "
|
||||||
"ELSE pg_catalog.pg_{0}_{1}_diff(pg_catalog.pg_current_{0}_{1}(), '0/0')::bigint END, " # write_lsn
|
"ELSE pg_catalog.pg_{0}_{1}_diff(pg_catalog.pg_current_{0}_{1}(), '0/0')::bigint END, " # write_lsn
|
||||||
"pg_catalog.pg_{0}_{1}_diff(pg_catalog.pg_last_{0}_replay_{1}(), '0/0')::bigint, "
|
"pg_catalog.pg_{0}_{1}_diff(pg_catalog.pg_last_{0}_replay_{1}(), '0/0')::bigint, "
|
||||||
"pg_catalog.pg_{0}_{1}_diff(COALESCE(pg_catalog.pg_last_{0}_receive_{1}(), '0/0'), '0/0')::bigint, "
|
"pg_catalog.pg_{0}_{1}_diff(COALESCE(pg_catalog.pg_last_{0}_receive_{1}(), '0/0'), '0/0')::bigint, "
|
||||||
"pg_catalog.pg_is_in_recovery() AND pg_catalog.pg_is_{0}_replay_paused()")
|
"pg_catalog.pg_is_in_recovery() AND pg_catalog.pg_is_{0}_replay_paused()")
|
||||||
|
|
||||||
@@ -101,15 +104,19 @@ class Postgresql(object):
|
|||||||
self._state_entry_timestamp = None
|
self._state_entry_timestamp = None
|
||||||
|
|
||||||
self._cluster_info_state = {}
|
self._cluster_info_state = {}
|
||||||
|
self._has_permanent_logical_slots = True
|
||||||
|
self._enforce_hot_standby_feedback = False
|
||||||
self._cached_replica_timeline = None
|
self._cached_replica_timeline = None
|
||||||
|
|
||||||
# Last known running process
|
# Last known running process
|
||||||
self._postmaster_proc = None
|
self._postmaster_proc = None
|
||||||
|
|
||||||
if self.is_running():
|
if self.is_running(): # we are "joining" already running postgres
|
||||||
self.set_state('running')
|
self.set_state('running')
|
||||||
self.set_role('master' if self.is_leader() else 'replica')
|
self.set_role('master' if self.is_leader() else 'replica')
|
||||||
self.config.write_postgresql_conf() # we are "joining" already running postgres
|
# postpone writing postgresql.conf for 12+ because recovery parameters are not yet known
|
||||||
|
if self.major_version < 120000 or self.is_leader():
|
||||||
|
self.config.write_postgresql_conf()
|
||||||
hba_saved = self.config.replace_pg_hba()
|
hba_saved = self.config.replace_pg_hba()
|
||||||
ident_saved = self.config.replace_pg_ident()
|
ident_saved = self.config.replace_pg_ident()
|
||||||
if hba_saved or ident_saved:
|
if hba_saved or ident_saved:
|
||||||
@@ -152,14 +159,18 @@ class Postgresql(object):
|
|||||||
@property
|
@property
|
||||||
def cluster_info_query(self):
|
def cluster_info_query(self):
|
||||||
if self._major_version >= 90600:
|
if self._major_version >= 90600:
|
||||||
|
extra = "(SELECT pg_catalog.json_agg(s.*) FROM (SELECT slot_name, slot_type as type, datoid::bigint, " +\
|
||||||
|
"plugin, catalog_xmin, pg_catalog.pg_wal_lsn_diff(confirmed_flush_lsn, '0/0')::bigint" + \
|
||||||
|
" AS confirmed_flush_lsn FROM pg_catalog.pg_get_replication_slots()) AS s)"\
|
||||||
|
if self._has_permanent_logical_slots and self._major_version >= 110000 else "NULL"
|
||||||
extra = (", CASE WHEN latest_end_lsn IS NULL THEN NULL ELSE received_tli END,"
|
extra = (", CASE WHEN latest_end_lsn IS NULL THEN NULL ELSE received_tli END,"
|
||||||
" slot_name, conninfo FROM pg_catalog.pg_stat_get_wal_receiver()")
|
" slot_name, conninfo, {0} FROM pg_catalog.pg_stat_get_wal_receiver()").format(extra)
|
||||||
if self.role == 'standby_leader':
|
if self.role == 'standby_leader':
|
||||||
extra = "timeline_id" + extra + ", pg_catalog.pg_control_checkpoint()"
|
extra = "timeline_id" + extra + ", pg_catalog.pg_control_checkpoint()"
|
||||||
else:
|
else:
|
||||||
extra = "0" + extra
|
extra = "0" + extra
|
||||||
else:
|
else:
|
||||||
extra = "0, NULL, NULL, NULL"
|
extra = "0, NULL, NULL, NULL, NULL"
|
||||||
|
|
||||||
return ("SELECT " + self.TL_LSN + ", {2}").format(self.wal_name, self.lsn_name, extra)
|
return ("SELECT " + self.TL_LSN + ", {2}").format(self.wal_name, self.lsn_name, extra)
|
||||||
|
|
||||||
@@ -253,15 +264,15 @@ class Postgresql(object):
|
|||||||
cursor = None
|
cursor = None
|
||||||
try:
|
try:
|
||||||
cursor = self._connection.cursor()
|
cursor = self._connection.cursor()
|
||||||
cursor.execute(sql, params)
|
cursor.execute(sql, params or None)
|
||||||
return cursor
|
return cursor
|
||||||
except psycopg2.Error as e:
|
except psycopg.Error as e:
|
||||||
if cursor and cursor.connection.closed == 0:
|
if cursor and cursor.connection.closed == 0:
|
||||||
# When connected via unix socket, psycopg2 can't recoginze 'connection lost'
|
# When connected via unix socket, psycopg2 can't recoginze 'connection lost'
|
||||||
# and leaves `_cursor_holder.connection.closed == 0`, but psycopg2.OperationalError
|
# and leaves `_cursor_holder.connection.closed == 0`, but psycopg2.OperationalError
|
||||||
# is still raised (what is correct). It doesn't make sense to continiue with existing
|
# is still raised (what is correct). It doesn't make sense to continiue with existing
|
||||||
# connection and we will close it, to avoid its reuse by the `cursor` method.
|
# connection and we will close it, to avoid its reuse by the `cursor` method.
|
||||||
if isinstance(e, psycopg2.OperationalError):
|
if isinstance(e, psycopg.OperationalError):
|
||||||
self._connection.close()
|
self._connection.close()
|
||||||
else:
|
else:
|
||||||
raise e
|
raise e
|
||||||
@@ -299,16 +310,41 @@ class Postgresql(object):
|
|||||||
replica_methods = self.create_replica_methods
|
replica_methods = self.create_replica_methods
|
||||||
return any(self.replica_method_can_work_without_replication_connection(m) for m in replica_methods)
|
return any(self.replica_method_can_work_without_replication_connection(m) for m in replica_methods)
|
||||||
|
|
||||||
def reset_cluster_info_state(self):
|
@property
|
||||||
|
def enforce_hot_standby_feedback(self):
|
||||||
|
return self._enforce_hot_standby_feedback
|
||||||
|
|
||||||
|
def set_enforce_hot_standby_feedback(self, value):
|
||||||
|
# If we enable or disable the hot_standby_feedback we need to update postgresql.conf and reload
|
||||||
|
if self._enforce_hot_standby_feedback != value:
|
||||||
|
self._enforce_hot_standby_feedback = value
|
||||||
|
if self.is_running():
|
||||||
|
self.config.write_postgresql_conf()
|
||||||
|
self.reload()
|
||||||
|
|
||||||
|
def reset_cluster_info_state(self, cluster, nofailover=None):
|
||||||
self._cluster_info_state = {}
|
self._cluster_info_state = {}
|
||||||
|
if cluster and cluster.config and cluster.config.modify_index:
|
||||||
|
self._has_permanent_logical_slots =\
|
||||||
|
cluster.has_permanent_logical_slots(self.name, nofailover, self.major_version)
|
||||||
|
|
||||||
|
# We want to enable hot_standby_feedback if the replica is supposed
|
||||||
|
# to have a logical slot or in case if it is the cascading replica.
|
||||||
|
self.set_enforce_hot_standby_feedback(
|
||||||
|
self._has_permanent_logical_slots or
|
||||||
|
cluster.should_enforce_hot_standby_feedback(self.name, nofailover, self.major_version))
|
||||||
|
|
||||||
def _cluster_info_state_get(self, name):
|
def _cluster_info_state_get(self, name):
|
||||||
if not self._cluster_info_state:
|
if not self._cluster_info_state:
|
||||||
try:
|
try:
|
||||||
result = self._is_leader_retry(self._query, self.cluster_info_query).fetchone()
|
result = self._is_leader_retry(self._query, self.cluster_info_query).fetchone()
|
||||||
self._cluster_info_state = dict(zip(['timeline', 'wal_position', 'replayed_location',
|
cluster_info_state = dict(zip(['timeline', 'wal_position', 'replayed_location',
|
||||||
'received_location', 'replay_paused', 'pg_control_timeline',
|
'received_location', 'replay_paused', 'pg_control_timeline',
|
||||||
'received_tli', 'slot_name', 'conninfo'], result))
|
'received_tli', 'slot_name', 'conninfo', 'slots'], result))
|
||||||
|
if self._has_permanent_logical_slots:
|
||||||
|
cluster_info_state['slots'] =\
|
||||||
|
self.slots_handler.process_permanent_slots(cluster_info_state['slots'])
|
||||||
|
self._cluster_info_state = cluster_info_state
|
||||||
except RetryFailedError as e: # SELECT failed two times
|
except RetryFailedError as e: # SELECT failed two times
|
||||||
self._cluster_info_state = {'error': str(e)}
|
self._cluster_info_state = {'error': str(e)}
|
||||||
if not self.is_starting() and self.pg_isready() == STATE_REJECT:
|
if not self.is_starting() and self.pg_isready() == STATE_REJECT:
|
||||||
@@ -325,6 +361,9 @@ class Postgresql(object):
|
|||||||
def received_location(self):
|
def received_location(self):
|
||||||
return self._cluster_info_state_get('received_location')
|
return self._cluster_info_state_get('received_location')
|
||||||
|
|
||||||
|
def slots(self):
|
||||||
|
return self._cluster_info_state_get('slots')
|
||||||
|
|
||||||
def primary_slot_name(self):
|
def primary_slot_name(self):
|
||||||
return self._cluster_info_state_get('slot_name')
|
return self._cluster_info_state_get('slot_name')
|
||||||
|
|
||||||
@@ -335,24 +374,65 @@ class Postgresql(object):
|
|||||||
return self._cluster_info_state_get('received_tli')
|
return self._cluster_info_state_get('received_tli')
|
||||||
|
|
||||||
def is_leader(self):
|
def is_leader(self):
|
||||||
return bool(self._cluster_info_state_get('timeline'))
|
try:
|
||||||
|
return bool(self._cluster_info_state_get('timeline'))
|
||||||
|
except PostgresConnectionException:
|
||||||
|
logger.warning('Failed to determine PostgreSQL state from the connection, falling back to cached role')
|
||||||
|
return bool(self.is_running() and self.role == 'master')
|
||||||
|
|
||||||
|
def replay_paused(self):
|
||||||
|
return self._cluster_info_state_get('replay_paused')
|
||||||
|
|
||||||
|
def resume_wal_replay(self):
|
||||||
|
self._query('SELECT pg_catalog.pg_{0}_replay_resume()'.format(self.wal_name))
|
||||||
|
|
||||||
|
def handle_parameter_change(self):
|
||||||
|
if self.major_version >= 140000 and self.replay_paused():
|
||||||
|
logger.info('Resuming paused WAL replay for PostgreSQL 14+')
|
||||||
|
self.resume_wal_replay()
|
||||||
|
|
||||||
def pg_control_timeline(self):
|
def pg_control_timeline(self):
|
||||||
try:
|
try:
|
||||||
|
|
||||||
return int(self.controldata().get("Latest checkpoint's TimeLineID"))
|
return int(self.controldata().get("Latest checkpoint's TimeLineID"))
|
||||||
except (TypeError, ValueError):
|
except (TypeError, ValueError):
|
||||||
logger.exception('Failed to parse timeline from pg_controldata output')
|
logger.exception('Failed to parse timeline from pg_controldata output')
|
||||||
|
|
||||||
|
def parse_wal_record(self, timeline, lsn):
|
||||||
|
out, err = self.waldump(timeline, lsn, 1)
|
||||||
|
if out and not err:
|
||||||
|
match = re.match(r'^rmgr:\s+(.+?)\s+len \(rec/tot\):\s+\d+/\s+\d+, tx:\s+\d+, '
|
||||||
|
r'lsn: ([0-9A-Fa-f]+/[0-9A-Fa-f]+), prev ([0-9A-Fa-f]+/[0-9A-Fa-f]+), '
|
||||||
|
r'.*?desc: (.+)', out.decode('utf-8'))
|
||||||
|
if match:
|
||||||
|
return match.groups()
|
||||||
|
return None, None, None, None
|
||||||
|
|
||||||
def latest_checkpoint_location(self):
|
def latest_checkpoint_location(self):
|
||||||
"""Returns checkpoint location for the cleanly shut down primary"""
|
"""Returns checkpoint location for the cleanly shut down primary.
|
||||||
|
But, if we know that the checkpoint was written to the new WAL
|
||||||
|
due to the archive_mode=on, we will return the LSN of prev wal record (SWITCH)."""
|
||||||
|
|
||||||
data = self.controldata()
|
data = self.controldata()
|
||||||
lsn = data.get('Latest checkpoint location')
|
timeline = data.get("Latest checkpoint's TimeLineID")
|
||||||
if data.get('Database cluster state') == 'shut down' and lsn:
|
lsn = checkpoint_lsn = data.get('Latest checkpoint location')
|
||||||
|
if data.get('Database cluster state') == 'shut down' and lsn and timeline:
|
||||||
try:
|
try:
|
||||||
return str(parse_lsn(lsn))
|
checkpoint_lsn = parse_lsn(checkpoint_lsn)
|
||||||
except (IndexError, ValueError) as e:
|
rm_name, lsn, prev, desc = self.parse_wal_record(timeline, lsn)
|
||||||
logger.error('Exception when parsing lsn %s: %r', lsn, e)
|
desc = desc.strip().lower()
|
||||||
|
if rm_name == 'XLOG' and parse_lsn(lsn) == checkpoint_lsn and prev and\
|
||||||
|
desc.startswith('checkpoint') and desc.endswith('shutdown'):
|
||||||
|
_, lsn, _, desc = self.parse_wal_record(timeline, prev)
|
||||||
|
prev = parse_lsn(prev)
|
||||||
|
# If the cluster is shutdown with archive_mode=on, WAL is switched before writing the checkpoint.
|
||||||
|
# In this case we want to take the LSN of previous record (switch) as the last known WAL location.
|
||||||
|
if parse_lsn(lsn) == prev and desc.strip() in ('xlog switch', 'SWITCH'):
|
||||||
|
return str(prev)
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Exception when parsing WAL pg_%sdump output: %r', self.wal_name, e)
|
||||||
|
if isinstance(checkpoint_lsn, six.integer_types):
|
||||||
|
return str(checkpoint_lsn)
|
||||||
|
|
||||||
def is_running(self):
|
def is_running(self):
|
||||||
"""Returns PostmasterProcess if one is running on the data directory or None. If most recently seen process
|
"""Returns PostmasterProcess if one is running on the data directory or None. If most recently seen process
|
||||||
@@ -362,7 +442,7 @@ class Postgresql(object):
|
|||||||
return self._postmaster_proc
|
return self._postmaster_proc
|
||||||
self._postmaster_proc = None
|
self._postmaster_proc = None
|
||||||
|
|
||||||
# we noticed that postgres was restarted, force syncing of replication
|
# we noticed that postgres was restarted, force syncing of replication slots and check of logical slots
|
||||||
self.slots_handler.schedule()
|
self.slots_handler.schedule()
|
||||||
|
|
||||||
self._postmaster_proc = PostmasterProcess.from_pidfile(self._data_dir)
|
self._postmaster_proc = PostmasterProcess.from_pidfile(self._data_dir)
|
||||||
@@ -523,12 +603,13 @@ class Postgresql(object):
|
|||||||
cur.execute('SELECT pg_catalog.pg_is_in_recovery()')
|
cur.execute('SELECT pg_catalog.pg_is_in_recovery()')
|
||||||
if cur.fetchone()[0]:
|
if cur.fetchone()[0]:
|
||||||
return 'is_in_recovery=true'
|
return 'is_in_recovery=true'
|
||||||
return cur.execute('CHECKPOINT')
|
cur.execute('CHECKPOINT')
|
||||||
except psycopg2.Error:
|
except psycopg.Error:
|
||||||
logger.exception('Exception during CHECKPOINT')
|
logger.exception('Exception during CHECKPOINT')
|
||||||
return 'not accessible or not healty'
|
return 'not accessible or not healty'
|
||||||
|
|
||||||
def stop(self, mode='fast', block_callbacks=False, checkpoint=None, on_safepoint=None, stop_timeout=None):
|
def stop(self, mode='fast', block_callbacks=False, checkpoint=None,
|
||||||
|
on_safepoint=None, on_shutdown=None, stop_timeout=None):
|
||||||
"""Stop PostgreSQL
|
"""Stop PostgreSQL
|
||||||
|
|
||||||
Supports a callback when a safepoint is reached. A safepoint is when no user backend can return a successful
|
Supports a callback when a safepoint is reached. A safepoint is when no user backend can return a successful
|
||||||
@@ -536,11 +617,12 @@ class Postgresql(object):
|
|||||||
could be added.
|
could be added.
|
||||||
|
|
||||||
:param on_safepoint: This callback is called when no user backends are running.
|
:param on_safepoint: This callback is called when no user backends are running.
|
||||||
|
:param on_shutdown: is called when pg_controldata starts reporting `Database cluster state: shut down`
|
||||||
"""
|
"""
|
||||||
if checkpoint is None:
|
if checkpoint is None:
|
||||||
checkpoint = False if mode == 'immediate' else True
|
checkpoint = False if mode == 'immediate' else True
|
||||||
|
|
||||||
success, pg_signaled = self._do_stop(mode, block_callbacks, checkpoint, on_safepoint, stop_timeout)
|
success, pg_signaled = self._do_stop(mode, block_callbacks, checkpoint, on_safepoint, on_shutdown, stop_timeout)
|
||||||
if success:
|
if success:
|
||||||
# block_callbacks is used during restart to avoid
|
# block_callbacks is used during restart to avoid
|
||||||
# running start/stop callbacks in addition to restart ones
|
# running start/stop callbacks in addition to restart ones
|
||||||
@@ -553,7 +635,7 @@ class Postgresql(object):
|
|||||||
self.set_state('stop failed')
|
self.set_state('stop failed')
|
||||||
return success
|
return success
|
||||||
|
|
||||||
def _do_stop(self, mode, block_callbacks, checkpoint, on_safepoint, stop_timeout):
|
def _do_stop(self, mode, block_callbacks, checkpoint, on_safepoint, on_shutdown, stop_timeout):
|
||||||
postmaster = self.is_running()
|
postmaster = self.is_running()
|
||||||
if not postmaster:
|
if not postmaster:
|
||||||
if on_safepoint:
|
if on_safepoint:
|
||||||
@@ -580,6 +662,22 @@ class Postgresql(object):
|
|||||||
postmaster.wait_for_user_backends_to_close()
|
postmaster.wait_for_user_backends_to_close()
|
||||||
on_safepoint()
|
on_safepoint()
|
||||||
|
|
||||||
|
if on_shutdown and mode in ('fast', 'smart'):
|
||||||
|
i = 0
|
||||||
|
# Wait for pg_controldata `Database cluster state:` to change to "shut down"
|
||||||
|
while postmaster.is_running():
|
||||||
|
data = self.controldata()
|
||||||
|
if data.get('Database cluster state', '') == 'shut down':
|
||||||
|
on_shutdown(int(self.latest_checkpoint_location()))
|
||||||
|
break
|
||||||
|
elif data.get('Database cluster state', '').startswith('shut down'): # shut down in recovery
|
||||||
|
break
|
||||||
|
elif stop_timeout and i >= stop_timeout:
|
||||||
|
stop_timeout = 0
|
||||||
|
break
|
||||||
|
time.sleep(STOP_POLLING_INTERVAL)
|
||||||
|
i += STOP_POLLING_INTERVAL
|
||||||
|
|
||||||
try:
|
try:
|
||||||
postmaster.wait(timeout=stop_timeout)
|
postmaster.wait(timeout=stop_timeout)
|
||||||
except TimeoutExpired:
|
except TimeoutExpired:
|
||||||
@@ -614,7 +712,7 @@ class Postgresql(object):
|
|||||||
while postmaster.is_running(): # Need a timeout here?
|
while postmaster.is_running(): # Need a timeout here?
|
||||||
cur.execute("SELECT 1")
|
cur.execute("SELECT 1")
|
||||||
time.sleep(STOP_POLLING_INTERVAL)
|
time.sleep(STOP_POLLING_INTERVAL)
|
||||||
except psycopg2.Error:
|
except psycopg.Error:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
def reload(self, block_callbacks=False):
|
def reload(self, block_callbacks=False):
|
||||||
@@ -724,8 +822,22 @@ class Postgresql(object):
|
|||||||
logger.exception("Error when calling pg_controldata")
|
logger.exception("Error when calling pg_controldata")
|
||||||
return {}
|
return {}
|
||||||
|
|
||||||
|
def waldump(self, timeline, lsn, limit):
|
||||||
|
cmd = self.pgcommand('pg_{0}dump'.format(self.wal_name))
|
||||||
|
env = os.environ.copy()
|
||||||
|
env.update(LANG='C', LC_ALL='C', PGDATA=self._data_dir)
|
||||||
|
try:
|
||||||
|
waldump = subprocess.Popen([cmd, '-t', str(timeline), '-s', str(lsn), '-n', str(limit)],
|
||||||
|
stdout=subprocess.PIPE, stderr=subprocess.PIPE, env=env)
|
||||||
|
out, err = waldump.communicate()
|
||||||
|
waldump.wait()
|
||||||
|
return out, err
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Failed to execute `%s -t %s -s %s -n %s`: %r', cmd, timeline, lsn, limit, e)
|
||||||
|
return None, None
|
||||||
|
|
||||||
@contextmanager
|
@contextmanager
|
||||||
def get_replication_connection_cursor(self, host='localhost', port=5432, **kwargs):
|
def get_replication_connection_cursor(self, host=None, port=5432, **kwargs):
|
||||||
conn_kwargs = self.config.replication.copy()
|
conn_kwargs = self.config.replication.copy()
|
||||||
conn_kwargs.update(host=host, port=int(port) if port else None, user=conn_kwargs.pop('username'),
|
conn_kwargs.update(host=host, port=int(port) if port else None, user=conn_kwargs.pop('username'),
|
||||||
connect_timeout=3, replication=1, options='-c statement_timeout=2000')
|
connect_timeout=3, replication=1, options='-c statement_timeout=2000')
|
||||||
@@ -759,6 +871,7 @@ class Postgresql(object):
|
|||||||
if history[-1][0] == timeline - 1:
|
if history[-1][0] == timeline - 1:
|
||||||
history_mtime = datetime.fromtimestamp(history_mtime).replace(tzinfo=tz.tzlocal())
|
history_mtime = datetime.fromtimestamp(history_mtime).replace(tzinfo=tz.tzlocal())
|
||||||
history[-1].append(history_mtime.isoformat())
|
history[-1].append(history_mtime.isoformat())
|
||||||
|
history[-1].append(self.name)
|
||||||
return history
|
return history
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('Failed to read and parse %s', (history_path,))
|
logger.exception('Failed to read and parse %s', (history_path,))
|
||||||
@@ -776,20 +889,22 @@ class Postgresql(object):
|
|||||||
if change_role:
|
if change_role:
|
||||||
self.__cb_pending = ACTION_NOOP
|
self.__cb_pending = ACTION_NOOP
|
||||||
|
|
||||||
|
ret = True
|
||||||
if self.is_running():
|
if self.is_running():
|
||||||
if do_reload:
|
if do_reload:
|
||||||
self.config.write_postgresql_conf()
|
self.config.write_postgresql_conf()
|
||||||
if self.reload(block_callbacks=change_role) and change_role:
|
ret = self.reload(block_callbacks=change_role)
|
||||||
|
if ret and change_role:
|
||||||
self.set_role(role)
|
self.set_role(role)
|
||||||
else:
|
else:
|
||||||
self.restart(block_callbacks=change_role, role=role)
|
ret = self.restart(block_callbacks=change_role, role=role)
|
||||||
else:
|
else:
|
||||||
self.start(timeout=timeout, block_callbacks=change_role, role=role)
|
ret = self.start(timeout=timeout, block_callbacks=change_role, role=role) or None
|
||||||
|
|
||||||
if change_role:
|
if change_role:
|
||||||
# TODO: postpone this until start completes, or maybe do even earlier
|
# TODO: postpone this until start completes, or maybe do even earlier
|
||||||
self.call_nowait(ACTION_ON_ROLE_CHANGE)
|
self.call_nowait(ACTION_ON_ROLE_CHANGE)
|
||||||
return True
|
return ret
|
||||||
|
|
||||||
def _wait_promote(self, wait_seconds):
|
def _wait_promote(self, wait_seconds):
|
||||||
for _ in polling_loop(wait_seconds):
|
for _ in polling_loop(wait_seconds):
|
||||||
@@ -812,7 +927,7 @@ class Postgresql(object):
|
|||||||
logger.info('pre_promote script `%s` exited with %s', cmd, ret)
|
logger.info('pre_promote script `%s` exited with %s', cmd, ret)
|
||||||
return ret == 0
|
return ret == 0
|
||||||
|
|
||||||
def promote(self, wait_seconds, task, on_success=None, access_is_restricted=False):
|
def promote(self, wait_seconds, task, on_success=None):
|
||||||
if self.role == 'master':
|
if self.role == 'master':
|
||||||
return True
|
return True
|
||||||
|
|
||||||
@@ -829,13 +944,14 @@ class Postgresql(object):
|
|||||||
logger.info("PostgreSQL promote cancelled.")
|
logger.info("PostgreSQL promote cancelled.")
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
self.slots_handler.on_promote()
|
||||||
|
|
||||||
ret = self.pg_ctl('promote', '-W')
|
ret = self.pg_ctl('promote', '-W')
|
||||||
if ret:
|
if ret:
|
||||||
self.set_role('master')
|
self.set_role('master')
|
||||||
if on_success is not None:
|
if on_success is not None:
|
||||||
on_success()
|
on_success()
|
||||||
if not access_is_restricted:
|
self.call_nowait(ACTION_ON_ROLE_CHANGE)
|
||||||
self.call_nowait(ACTION_ON_ROLE_CHANGE)
|
|
||||||
ret = self._wait_promote(wait_seconds)
|
ret = self._wait_promote(wait_seconds)
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
@@ -865,16 +981,16 @@ class Postgresql(object):
|
|||||||
try:
|
try:
|
||||||
query = "SELECT " + self.POSTMASTER_START_TIME
|
query = "SELECT " + self.POSTMASTER_START_TIME
|
||||||
if current_thread().ident == self.__thread_ident:
|
if current_thread().ident == self.__thread_ident:
|
||||||
return self.query(query).fetchone()[0]
|
return self.query(query).fetchone()[0].isoformat(sep=' ')
|
||||||
with self.connection().cursor() as cursor:
|
with self.connection().cursor() as cursor:
|
||||||
cursor.execute(query)
|
cursor.execute(query)
|
||||||
return cursor.fetchone()[0]
|
return cursor.fetchone()[0].isoformat(sep=' ')
|
||||||
except psycopg2.Error:
|
except psycopg.Error:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
def last_operation(self):
|
def last_operation(self):
|
||||||
return str(self._wal_position(self.is_leader(), self._cluster_info_state_get('wal_position'),
|
return self._wal_position(self.is_leader(), self._cluster_info_state_get('wal_position'),
|
||||||
self.received_location(), self.replayed_location()))
|
self.received_location(), self.replayed_location())
|
||||||
|
|
||||||
def configure_server_parameters(self):
|
def configure_server_parameters(self):
|
||||||
self._major_version = self.get_major_version()
|
self._major_version = self.get_major_version()
|
||||||
@@ -967,7 +1083,7 @@ class Postgresql(object):
|
|||||||
|
|
||||||
Current synchronous standby is always preferred, unless it has disconnected or does not want to be a
|
Current synchronous standby is always preferred, unless it has disconnected or does not want to be a
|
||||||
synchronous standby any longer.
|
synchronous standby any longer.
|
||||||
Parameter sync_node_maxlag(maximum_lag_on_syncnode) would help swapping unhealthy sync replica incase
|
Parameter sync_node_maxlag(maximum_lag_on_syncnode) would help swapping unhealthy sync replica in case
|
||||||
if it stops responding (or hung). Please set the value high enough so it won't unncessarily swap sync
|
if it stops responding (or hung). Please set the value high enough so it won't unncessarily swap sync
|
||||||
standbys during high loads. Any less or equal of 0 value keep the behavior backward compatible and
|
standbys during high loads. Any less or equal of 0 value keep the behavior backward compatible and
|
||||||
will not swap. Please note that it will not also swap sync standbys in case where all replicas are hung.
|
will not swap. Please note that it will not also swap sync standbys in case where all replicas are hung.
|
||||||
@@ -992,15 +1108,16 @@ class Postgresql(object):
|
|||||||
for app_name, sync_state, replica_lsn in self.query(
|
for app_name, sync_state, replica_lsn in self.query(
|
||||||
"SELECT pg_catalog.lower(application_name), sync_state, pg_{2}_{1}_diff({0}_{1}, '0/0')::bigint"
|
"SELECT pg_catalog.lower(application_name), sync_state, pg_{2}_{1}_diff({0}_{1}, '0/0')::bigint"
|
||||||
" FROM pg_catalog.pg_stat_replication"
|
" FROM pg_catalog.pg_stat_replication"
|
||||||
" WHERE state = 'streaming'"
|
" WHERE state = 'streaming' AND {0}_{1} IS NOT NULL"
|
||||||
" ORDER BY sync_state DESC, {0}_{1} DESC".format(sort_col, self.lsn_name, self.wal_name)):
|
" ORDER BY sync_state DESC, {0}_{1} DESC".format(sort_col, self.lsn_name, self.wal_name)):
|
||||||
member = members.get(app_name)
|
member = members.get(app_name)
|
||||||
if member and not member.tags.get('nosync', False):
|
if member and not member.tags.get('nosync', False):
|
||||||
replica_list.append((member.name, sync_state, replica_lsn))
|
replica_list.append((member.name, sync_state, replica_lsn, bool(member.nofailover)))
|
||||||
|
|
||||||
max_lsn = max(replica_list, key=lambda x: x[2])[2] if len(replica_list) > 1 else int(str(self.last_operation()))
|
max_lsn = max(replica_list, key=lambda x: x[2])[2] if len(replica_list) > 1 else int(str(self.last_operation()))
|
||||||
|
|
||||||
for app_name, sync_state, replica_lsn in replica_list:
|
# Prefer members without nofailover tag. We are relying on the fact that sorts are guaranteed to be stable.
|
||||||
|
for app_name, sync_state, replica_lsn, _ in sorted(replica_list, key=lambda x: x[3]):
|
||||||
if sync_node_maxlag <= 0 or max_lsn - replica_lsn <= sync_node_maxlag:
|
if sync_node_maxlag <= 0 or max_lsn - replica_lsn <= sync_node_maxlag:
|
||||||
candidates.append(app_name)
|
candidates.append(app_name)
|
||||||
if sync_state == 'sync':
|
if sync_state == 'sync':
|
||||||
|
|||||||
@@ -4,10 +4,12 @@ import shlex
|
|||||||
import tempfile
|
import tempfile
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from patroni.dcs import RemoteMember
|
|
||||||
from patroni.utils import deep_compare
|
|
||||||
from six import string_types
|
from six import string_types
|
||||||
|
|
||||||
|
from ..dcs import RemoteMember
|
||||||
|
from ..psycopg import quote_ident, quote_literal
|
||||||
|
from ..utils import deep_compare
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
@@ -53,7 +55,7 @@ class Bootstrap(object):
|
|||||||
error_handler('Error when parsing {0} option {1}: value should be string value'
|
error_handler('Error when parsing {0} option {1}: value should be string value'
|
||||||
' or a single key-value pair'.format(tool, opt))
|
' or a single key-value pair'.format(tool, opt))
|
||||||
else:
|
else:
|
||||||
error_handler('{0} options must be list ot dict'.format(tool))
|
error_handler('{0} options must be list or dict'.format(tool))
|
||||||
return user_options
|
return user_options
|
||||||
|
|
||||||
def _initdb(self, config):
|
def _initdb(self, config):
|
||||||
@@ -90,8 +92,8 @@ class Bootstrap(object):
|
|||||||
self._postgresql.configure_server_parameters()
|
self._postgresql.configure_server_parameters()
|
||||||
|
|
||||||
# make sure there is no trigger file or postgres will be automatically promoted
|
# make sure there is no trigger file or postgres will be automatically promoted
|
||||||
trigger_file = 'promote_trigger_file' if self._postgresql.major_version >= 120000 else 'trigger_file'
|
trigger_file = self._postgresql.config.triggerfile_good_name
|
||||||
trigger_file = self._postgresql.config.get('recovery_conf', {}).get(trigger_file) or 'promote'
|
trigger_file = (self._postgresql.config.get('recovery_conf') or {}).get(trigger_file) or 'promote'
|
||||||
trigger_file = os.path.abspath(os.path.join(self._postgresql.data_dir, trigger_file))
|
trigger_file = os.path.abspath(os.path.join(self._postgresql.data_dir, trigger_file))
|
||||||
if os.path.exists(trigger_file):
|
if os.path.exists(trigger_file):
|
||||||
os.unlink(trigger_file)
|
os.unlink(trigger_file)
|
||||||
@@ -297,26 +299,24 @@ class Bootstrap(object):
|
|||||||
if 'NOLOGIN' not in options and 'LOGIN' not in options:
|
if 'NOLOGIN' not in options and 'LOGIN' not in options:
|
||||||
options.append('LOGIN')
|
options.append('LOGIN')
|
||||||
|
|
||||||
params = [name]
|
|
||||||
if password:
|
if password:
|
||||||
options.extend(['PASSWORD', '%s'])
|
options.extend(['PASSWORD', quote_literal(password)])
|
||||||
params.extend([password, password])
|
|
||||||
|
|
||||||
sql = """DO $$
|
sql = """DO $$
|
||||||
BEGIN
|
BEGIN
|
||||||
SET local synchronous_commit = 'local';
|
SET local synchronous_commit = 'local';
|
||||||
PERFORM * FROM pg_authid WHERE rolname = %s;
|
PERFORM * FROM pg_catalog.pg_authid WHERE rolname = {0};
|
||||||
IF FOUND THEN
|
IF FOUND THEN
|
||||||
ALTER ROLE "{0}" WITH {1};
|
ALTER ROLE {1} WITH {2};
|
||||||
ELSE
|
ELSE
|
||||||
CREATE ROLE "{0}" WITH {1};
|
CREATE ROLE {1} WITH {2};
|
||||||
END IF;
|
END IF;
|
||||||
END;$$""".format(name, ' '.join(options))
|
END;$$""".format(quote_literal(name), quote_ident(name, self._postgresql.connection()), ' '.join(options))
|
||||||
self._postgresql.query('SET log_statement TO none')
|
self._postgresql.query('SET log_statement TO none')
|
||||||
self._postgresql.query('SET log_min_duration_statement TO -1')
|
self._postgresql.query('SET log_min_duration_statement TO -1')
|
||||||
self._postgresql.query("SET log_min_error_statement TO 'log'")
|
self._postgresql.query("SET log_min_error_statement TO 'log'")
|
||||||
try:
|
try:
|
||||||
self._postgresql.query(sql, *params)
|
self._postgresql.query(sql)
|
||||||
finally:
|
finally:
|
||||||
self._postgresql.query('RESET log_min_error_statement')
|
self._postgresql.query('RESET log_min_error_statement')
|
||||||
self._postgresql.query('RESET log_min_duration_statement')
|
self._postgresql.query('RESET log_min_duration_statement')
|
||||||
@@ -342,8 +342,8 @@ END;$$""".format(name, ' '.join(options))
|
|||||||
sql = """DO $$
|
sql = """DO $$
|
||||||
BEGIN
|
BEGIN
|
||||||
SET local synchronous_commit = 'local';
|
SET local synchronous_commit = 'local';
|
||||||
GRANT EXECUTE ON function pg_catalog.{0} TO "{1}";
|
GRANT EXECUTE ON function pg_catalog.{0} TO {1};
|
||||||
END;$$""".format(f, rewind['username'])
|
END;$$""".format(f, quote_ident(rewind['username'], self._postgresql.connection()))
|
||||||
postgresql.query(sql)
|
postgresql.query(sql)
|
||||||
|
|
||||||
for name, value in (config.get('users') or {}).items():
|
for name, value in (config.get('users') or {}).items():
|
||||||
|
|||||||
@@ -36,7 +36,7 @@ class CancellableExecutor(object):
|
|||||||
with self._lock:
|
with self._lock:
|
||||||
if self._process is not None and self._process.is_running() and not self._process_children:
|
if self._process is not None and self._process.is_running() and not self._process_children:
|
||||||
try:
|
try:
|
||||||
self._process.suspend() # Suspend the process before getting list of childrens
|
self._process.suspend() # Suspend the process before getting list of children
|
||||||
except psutil.Error as e:
|
except psutil.Error as e:
|
||||||
logger.info('Failed to suspend the process: %s', e.msg)
|
logger.info('Failed to suspend the process: %s', e.msg)
|
||||||
|
|
||||||
|
|||||||
@@ -12,6 +12,7 @@ from .validator import CaseInsensitiveDict, recovery_parameters,\
|
|||||||
transform_postgresql_parameter_value, transform_recovery_parameter_value
|
transform_postgresql_parameter_value, transform_recovery_parameter_value
|
||||||
from ..dcs import slot_name_from_member_name, RemoteMember
|
from ..dcs import slot_name_from_member_name, RemoteMember
|
||||||
from ..exceptions import PatroniFatalException
|
from ..exceptions import PatroniFatalException
|
||||||
|
from ..psycopg import quote_ident as _quote_ident
|
||||||
from ..utils import compare_values, parse_bool, parse_int, split_host_port, uri, \
|
from ..utils import compare_values, parse_bool, parse_int, split_host_port, uri, \
|
||||||
validate_directory, is_subpath
|
validate_directory, is_subpath
|
||||||
|
|
||||||
@@ -23,7 +24,7 @@ PARAMETER_RE = re.compile(r'([a-z_]+)\s*=\s*')
|
|||||||
|
|
||||||
def quote_ident(value):
|
def quote_ident(value):
|
||||||
"""Very simplified version of quote_ident"""
|
"""Very simplified version of quote_ident"""
|
||||||
return value if SYNC_STANDBY_NAME_RE.match(value) else '"' + value + '"'
|
return value if SYNC_STANDBY_NAME_RE.match(value) else _quote_ident(value)
|
||||||
|
|
||||||
|
|
||||||
def conninfo_uri_parse(dsn):
|
def conninfo_uri_parse(dsn):
|
||||||
@@ -336,8 +337,9 @@ class ConfigHandler(object):
|
|||||||
if "stats_temp_directory" in self._server_parameters:
|
if "stats_temp_directory" in self._server_parameters:
|
||||||
self.try_to_create_dir(self._server_parameters["stats_temp_directory"],
|
self.try_to_create_dir(self._server_parameters["stats_temp_directory"],
|
||||||
"'{}' is defined in stats_temp_directory, {}")
|
"'{}' is defined in stats_temp_directory, {}")
|
||||||
self.try_to_create_dir(os.path.dirname(self._pgpass),
|
if not self._krbsrvname:
|
||||||
"'{}' is defined in `postgresql.pgpass`, {}")
|
self.try_to_create_dir(os.path.dirname(self._pgpass),
|
||||||
|
"'{}' is defined in `postgresql.pgpass`, {}")
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def _configuration_to_save(self):
|
def _configuration_to_save(self):
|
||||||
@@ -352,7 +354,7 @@ class ConfigHandler(object):
|
|||||||
|
|
||||||
def save_configuration_files(self, check_custom_bootstrap=False):
|
def save_configuration_files(self, check_custom_bootstrap=False):
|
||||||
"""
|
"""
|
||||||
copy postgresql.conf to postgresql.conf.backup to be able to retrive configuration files
|
copy postgresql.conf to postgresql.conf.backup to be able to retrieve configuration files
|
||||||
- originally stored as symlinks, those are normally skipped by pg_basebackup
|
- originally stored as symlinks, those are normally skipped by pg_basebackup
|
||||||
- in case of WAL-E basebackup (see http://comments.gmane.org/gmane.comp.db.postgresql.wal-e/239)
|
- in case of WAL-E basebackup (see http://comments.gmane.org/gmane.comp.db.postgresql.wal-e/239)
|
||||||
"""
|
"""
|
||||||
@@ -387,22 +389,22 @@ class ConfigHandler(object):
|
|||||||
if 'custom_conf' not in self._config and not os.path.exists(self._postgresql_base_conf):
|
if 'custom_conf' not in self._config and not os.path.exists(self._postgresql_base_conf):
|
||||||
os.rename(self._postgresql_conf, self._postgresql_base_conf)
|
os.rename(self._postgresql_conf, self._postgresql_base_conf)
|
||||||
|
|
||||||
# In case we are using custom bootstrap from spilo image with PITR it fails if it contains increasing
|
configuration = configuration or self._server_parameters.copy()
|
||||||
# values like Max_connections. We disable hot_standby so it will accept increasing values.
|
# Due to the permanent logical replication slots configured we have to enable hot_standby_feedback
|
||||||
if self._postgresql.bootstrap.running_custom_bootstrap:
|
if self._postgresql.enforce_hot_standby_feedback:
|
||||||
configuration['hot_standby'] = 'off'
|
configuration['hot_standby_feedback'] = 'on'
|
||||||
|
|
||||||
with ConfigWriter(self._postgresql_conf) as f:
|
with ConfigWriter(self._postgresql_conf) as f:
|
||||||
include = self._config.get('custom_conf') or self._postgresql_base_conf_name
|
include = self._config.get('custom_conf') or self._postgresql_base_conf_name
|
||||||
f.writeline("include '{0}'\n".format(ConfigWriter.escape(include)))
|
f.writeline("include '{0}'\n".format(ConfigWriter.escape(include)))
|
||||||
for name, value in sorted((configuration or self._server_parameters).items()):
|
for name, value in sorted((configuration).items()):
|
||||||
value = transform_postgresql_parameter_value(self._postgresql.major_version, name, value)
|
value = transform_postgresql_parameter_value(self._postgresql.major_version, name, value)
|
||||||
if (not self._postgresql.bootstrap.running_custom_bootstrap or name != 'hba_file') \
|
if value is not None and\
|
||||||
and name not in self._RECOVERY_PARAMETERS and value is not None:
|
(name != 'hba_file' or not self._postgresql.bootstrap.running_custom_bootstrap):
|
||||||
f.write_param(name, value)
|
f.write_param(name, value)
|
||||||
# when we are doing custom bootstrap we assume that we don't know superuser password
|
# when we are doing custom bootstrap we assume that we don't know superuser password
|
||||||
# and in order to be able to change it, we are opening trust access from a certain address
|
# and in order to be able to change it, we are opening trust access from a certain address
|
||||||
# therefore we need to make sure that hba_file is not overriden
|
# therefore we need to make sure that hba_file is not overridden
|
||||||
# after changing superuser password we will "revert" all these "changes"
|
# after changing superuser password we will "revert" all these "changes"
|
||||||
if self._postgresql.bootstrap.running_custom_bootstrap or 'hba_file' not in self._server_parameters:
|
if self._postgresql.bootstrap.running_custom_bootstrap or 'hba_file' not in self._server_parameters:
|
||||||
f.write_param('hba_file', self._pg_hba_conf)
|
f.write_param('hba_file', self._pg_hba_conf)
|
||||||
@@ -476,18 +478,20 @@ class ConfigHandler(object):
|
|||||||
ret.setdefault('channel_binding', 'prefer')
|
ret.setdefault('channel_binding', 'prefer')
|
||||||
if self._krbsrvname:
|
if self._krbsrvname:
|
||||||
ret['krbsrvname'] = self._krbsrvname
|
ret['krbsrvname'] = self._krbsrvname
|
||||||
if 'database' in ret:
|
if 'dbname' in ret:
|
||||||
del ret['database']
|
del ret['dbname']
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
def format_dsn(self, params, include_dbname=False):
|
def format_dsn(self, params, include_dbname=False):
|
||||||
# A list of keywords that can be found in a conninfo string. Follows what is acceptable by libpq
|
# A list of keywords that can be found in a conninfo string. Follows what is acceptable by libpq
|
||||||
keywords = ('dbname', 'user', 'passfile' if params.get('passfile') else 'password', 'host', 'port',
|
keywords = ('dbname', 'user', 'passfile' if params.get('passfile') else 'password', 'host', 'port',
|
||||||
'sslmode', 'sslcompression', 'sslcert', 'sslkey', 'sslpassword', 'sslrootcert', 'sslcrl',
|
'sslmode', 'sslcompression', 'sslcert', 'sslkey', 'sslpassword', 'sslrootcert', 'sslcrl',
|
||||||
'application_name', 'krbsrvname', 'gssencmode', 'channel_binding')
|
'sslcrldir', 'application_name', 'krbsrvname', 'gssencmode', 'channel_binding',
|
||||||
|
'target_session_attrs')
|
||||||
if include_dbname:
|
if include_dbname:
|
||||||
params = params.copy()
|
params = params.copy()
|
||||||
params['dbname'] = params.get('database') or self._postgresql.database
|
if 'dbname' not in params:
|
||||||
|
params['dbname'] = self._postgresql.database
|
||||||
# we are abusing information about the necessity of dbname
|
# we are abusing information about the necessity of dbname
|
||||||
# dsn should contain passfile or password only if there is no dbname in it (it is used in recovery.conf)
|
# dsn should contain passfile or password only if there is no dbname in it (it is used in recovery.conf)
|
||||||
skip = {'passfile', 'password'}
|
skip = {'passfile', 'password'}
|
||||||
@@ -539,6 +543,12 @@ class ConfigHandler(object):
|
|||||||
if use_slots and not (is_remote_master and member.no_replication_slot):
|
if use_slots and not (is_remote_master and member.no_replication_slot):
|
||||||
primary_slot_name = member.primary_slot_name if is_remote_master else self._postgresql.name
|
primary_slot_name = member.primary_slot_name if is_remote_master else self._postgresql.name
|
||||||
recovery_params['primary_slot_name'] = slot_name_from_member_name(primary_slot_name)
|
recovery_params['primary_slot_name'] = slot_name_from_member_name(primary_slot_name)
|
||||||
|
# We are a standby leader and are using a replication slot. Make sure we connect to
|
||||||
|
# the leader of the main cluster (in case more than one host is specified in the
|
||||||
|
# connstr) by adding 'target_session_attrs=read-write' to primary_conninfo.
|
||||||
|
if is_remote_master and 'target_sesions_attrs' not in primary_conninfo and\
|
||||||
|
self._postgresql.major_version >= 100000:
|
||||||
|
primary_conninfo['target_session_attrs'] = 'read-write'
|
||||||
recovery_params['primary_conninfo'] = primary_conninfo
|
recovery_params['primary_conninfo'] = primary_conninfo
|
||||||
|
|
||||||
# standby_cluster config might have different parameters, we want to override them
|
# standby_cluster config might have different parameters, we want to override them
|
||||||
@@ -553,7 +563,7 @@ class ConfigHandler(object):
|
|||||||
return os.path.exists(self._recovery_conf)
|
return os.path.exists(self._recovery_conf)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def _triggerfile_good_name(self):
|
def triggerfile_good_name(self):
|
||||||
return 'trigger_file' if self._postgresql.major_version < 120000 else 'promote_trigger_file'
|
return 'trigger_file' if self._postgresql.major_version < 120000 else 'promote_trigger_file'
|
||||||
|
|
||||||
@property
|
@property
|
||||||
@@ -615,19 +625,19 @@ class ConfigHandler(object):
|
|||||||
|
|
||||||
def _check_passfile(self, passfile, wanted_primary_conninfo):
|
def _check_passfile(self, passfile, wanted_primary_conninfo):
|
||||||
# If there is a passfile in the primary_conninfo try to figure out that
|
# If there is a passfile in the primary_conninfo try to figure out that
|
||||||
# the passfile contains the line allowing connection to the given node.
|
# the passfile contains the line(s) allowing connection to the given node.
|
||||||
# We assume that the passfile was created by Patroni and therefore doing
|
# We assume that the passfile was created by Patroni and therefore doing
|
||||||
# the full match and not covering cases when host, port or user are set to '*'
|
# the full match and not covering cases when host, port or user are set to '*'
|
||||||
passfile_mtime = mtime(passfile)
|
passfile_mtime = mtime(passfile)
|
||||||
if passfile_mtime:
|
if passfile_mtime:
|
||||||
try:
|
try:
|
||||||
with open(passfile) as f:
|
with open(passfile) as f:
|
||||||
wanted_line = self._pgpass_line(wanted_primary_conninfo).strip()
|
wanted_lines = self._pgpass_line(wanted_primary_conninfo).splitlines()
|
||||||
for raw_line in f:
|
file_lines = f.read().splitlines()
|
||||||
if raw_line.strip() == wanted_line:
|
if set(wanted_lines) == set(file_lines):
|
||||||
self._passfile = passfile
|
self._passfile = passfile
|
||||||
self._passfile_mtime = passfile_mtime
|
self._passfile_mtime = passfile_mtime
|
||||||
return True
|
return True
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.info('Failed to read %s', passfile)
|
logger.info('Failed to read %s', passfile)
|
||||||
return False
|
return False
|
||||||
@@ -701,7 +711,9 @@ class ConfigHandler(object):
|
|||||||
if wal_receiver_primary_slot_name is not None:
|
if wal_receiver_primary_slot_name is not None:
|
||||||
self._current_recovery_params['primary_slot_name'][0] = wal_receiver_primary_slot_name
|
self._current_recovery_params['primary_slot_name'][0] = wal_receiver_primary_slot_name
|
||||||
|
|
||||||
required = {'restart': 0, 'reload': 0}
|
# Increment the 'reload' to enforce write of postgresql.conf when joining the running postgres
|
||||||
|
required = {'restart': 0,
|
||||||
|
'reload': int(not self._postgresql.cb_called and self._postgresql.major_version >= 120000)}
|
||||||
|
|
||||||
def record_missmatch(mtype):
|
def record_missmatch(mtype):
|
||||||
required['restart' if mtype else 'reload'] += 1
|
required['restart' if mtype else 'reload'] += 1
|
||||||
@@ -740,7 +752,12 @@ class ConfigHandler(object):
|
|||||||
return re.sub(r'([:\\])', r'\\\1', str(value))
|
return re.sub(r'([:\\])', r'\\\1', str(value))
|
||||||
|
|
||||||
record = {n: escape(record.get(n) or '*') for n in ('host', 'port', 'user', 'password')}
|
record = {n: escape(record.get(n) or '*') for n in ('host', 'port', 'user', 'password')}
|
||||||
return '{host}:{port}:*:{user}:{password}'.format(**record)
|
# 'host' could be several comma-separated hostnames, in this case
|
||||||
|
# we need to write on pgpass line per host
|
||||||
|
line = ''
|
||||||
|
for hostname in record.get('host').split(','):
|
||||||
|
line += hostname + ':{port}:*:{user}:{password}'.format(**record) + '\n'
|
||||||
|
return line.rstrip()
|
||||||
|
|
||||||
def write_pgpass(self, record):
|
def write_pgpass(self, record):
|
||||||
line = self._pgpass_line(record)
|
line = self._pgpass_line(record)
|
||||||
@@ -756,13 +773,13 @@ class ConfigHandler(object):
|
|||||||
return env
|
return env
|
||||||
|
|
||||||
def write_recovery_conf(self, recovery_params):
|
def write_recovery_conf(self, recovery_params):
|
||||||
|
self._recovery_params = recovery_params
|
||||||
if self._postgresql.major_version >= 120000:
|
if self._postgresql.major_version >= 120000:
|
||||||
if parse_bool(recovery_params.pop('standby_mode', None)):
|
if parse_bool(recovery_params.pop('standby_mode', None)):
|
||||||
open(self._standby_signal, 'w').close()
|
open(self._standby_signal, 'w').close()
|
||||||
else:
|
else:
|
||||||
self._remove_file_if_exists(self._standby_signal)
|
self._remove_file_if_exists(self._standby_signal)
|
||||||
open(self._recovery_signal, 'w').close()
|
open(self._recovery_signal, 'w').close()
|
||||||
self._recovery_params = recovery_params
|
|
||||||
else:
|
else:
|
||||||
with ConfigWriter(self._recovery_conf) as f:
|
with ConfigWriter(self._recovery_conf) as f:
|
||||||
os.chmod(self._recovery_conf, stat.S_IWRITE | stat.S_IREAD)
|
os.chmod(self._recovery_conf, stat.S_IWRITE | stat.S_IREAD)
|
||||||
@@ -806,8 +823,8 @@ class ConfigHandler(object):
|
|||||||
|
|
||||||
if self.get('recovery_conf'):
|
if self.get('recovery_conf'):
|
||||||
value = self._config['recovery_conf'].pop(self._triggerfile_wrong_name, None)
|
value = self._config['recovery_conf'].pop(self._triggerfile_wrong_name, None)
|
||||||
if self._triggerfile_good_name not in self._config['recovery_conf'] and value:
|
if self.triggerfile_good_name not in self._config['recovery_conf'] and value:
|
||||||
self._config['recovery_conf'][self._triggerfile_good_name] = value
|
self._config['recovery_conf'][self.triggerfile_good_name] = value
|
||||||
|
|
||||||
def get_server_parameters(self, config):
|
def get_server_parameters(self, config):
|
||||||
parameters = config['parameters'].copy()
|
parameters = config['parameters'].copy()
|
||||||
@@ -831,7 +848,7 @@ class ConfigHandler(object):
|
|||||||
# this exercise is improving cross version compatibility and user must set the correct parameter in the config.
|
# this exercise is improving cross version compatibility and user must set the correct parameter in the config.
|
||||||
if self._postgresql.major_version >= 130000:
|
if self._postgresql.major_version >= 130000:
|
||||||
wal_keep_segments = parameters.pop('wal_keep_segments', self.CMDLINE_OPTIONS['wal_keep_segments'][0])
|
wal_keep_segments = parameters.pop('wal_keep_segments', self.CMDLINE_OPTIONS['wal_keep_segments'][0])
|
||||||
parameters.setdefault('wal_keep_size', str(wal_keep_segments * 16) + 'MB')
|
parameters.setdefault('wal_keep_size', str(int(wal_keep_segments) * 16) + 'MB')
|
||||||
elif self._postgresql.major_version:
|
elif self._postgresql.major_version:
|
||||||
wal_keep_size = parse_int(parameters.pop('wal_keep_size', self.CMDLINE_OPTIONS['wal_keep_size'][0]), 'MB')
|
wal_keep_size = parse_int(parameters.pop('wal_keep_size', self.CMDLINE_OPTIONS['wal_keep_size'][0]), 'MB')
|
||||||
parameters.setdefault('wal_keep_segments', int((wal_keep_size + 8) / 16))
|
parameters.setdefault('wal_keep_segments', int((wal_keep_size + 8) / 16))
|
||||||
@@ -867,7 +884,7 @@ class ConfigHandler(object):
|
|||||||
ret['user'] = self._superuser['username']
|
ret['user'] = self._superuser['username']
|
||||||
del ret['username']
|
del ret['username']
|
||||||
# ensure certain Patroni configurations are available
|
# ensure certain Patroni configurations are available
|
||||||
ret.update({'database': self._postgresql.database,
|
ret.update({'dbname': self._postgresql.database,
|
||||||
'fallback_application_name': 'Patroni',
|
'fallback_application_name': 'Patroni',
|
||||||
'connect_timeout': 3,
|
'connect_timeout': 3,
|
||||||
'options': '-c statement_timeout=2000'})
|
'options': '-c statement_timeout=2000'})
|
||||||
@@ -876,25 +893,21 @@ class ConfigHandler(object):
|
|||||||
def resolve_connection_addresses(self):
|
def resolve_connection_addresses(self):
|
||||||
port = self._server_parameters['port']
|
port = self._server_parameters['port']
|
||||||
tcp_local_address = self._get_tcp_local_address()
|
tcp_local_address = self._get_tcp_local_address()
|
||||||
|
|
||||||
local_address = {'port': port}
|
|
||||||
if self._config.get('use_unix_socket'):
|
|
||||||
unix_socket_directories = self._server_parameters.get('unix_socket_directories')
|
|
||||||
if unix_socket_directories is not None:
|
|
||||||
# fallback to tcp if unix_socket_directories is set, but there are no sutable values
|
|
||||||
local_address['host'] = self._get_unix_local_address(unix_socket_directories) or tcp_local_address
|
|
||||||
|
|
||||||
# if unix_socket_directories is not specified, but use_unix_socket is set to true - do our best
|
|
||||||
# to use default value, i.e. don't specify a host neither in connection url nor arguments
|
|
||||||
else:
|
|
||||||
local_address['host'] = tcp_local_address
|
|
||||||
|
|
||||||
self._local_address = local_address
|
|
||||||
self.local_replication_address = {'host': tcp_local_address, 'port': port}
|
|
||||||
|
|
||||||
netloc = self._config.get('connect_address') or tcp_local_address + ':' + port
|
netloc = self._config.get('connect_address') or tcp_local_address + ':' + port
|
||||||
self._postgresql.connection_string = uri('postgres', netloc, self._postgresql.database)
|
|
||||||
|
|
||||||
|
unix_local_address = {'port': port}
|
||||||
|
unix_socket_directories = self._server_parameters.get('unix_socket_directories')
|
||||||
|
if unix_socket_directories is not None:
|
||||||
|
# fallback to tcp if unix_socket_directories is set, but there are no suitable values
|
||||||
|
unix_local_address['host'] = self._get_unix_local_address(unix_socket_directories) or tcp_local_address
|
||||||
|
|
||||||
|
tcp_local_address = {'host': tcp_local_address, 'port': port}
|
||||||
|
|
||||||
|
self._local_address = unix_local_address if self._config.get('use_unix_socket') else tcp_local_address
|
||||||
|
self.local_replication_address = unix_local_address\
|
||||||
|
if self._config.get('use_unix_socket_repl') else tcp_local_address
|
||||||
|
|
||||||
|
self._postgresql.connection_string = uri('postgres', netloc, self._postgresql.database)
|
||||||
self._postgresql.set_connection_kwargs(self.local_connect_kwargs)
|
self._postgresql.set_connection_kwargs(self.local_connect_kwargs)
|
||||||
|
|
||||||
def _get_pg_settings(self, names):
|
def _get_pg_settings(self, names):
|
||||||
@@ -1002,8 +1015,9 @@ class ConfigHandler(object):
|
|||||||
if self._postgresql.major_version >= 90500:
|
if self._postgresql.major_version >= 90500:
|
||||||
time.sleep(1)
|
time.sleep(1)
|
||||||
try:
|
try:
|
||||||
pending_restart = self._postgresql.query('SELECT COUNT(*) FROM pg_catalog.pg_settings'
|
pending_restart = self._postgresql.query(
|
||||||
' WHERE pending_restart').fetchone()[0] > 0
|
'SELECT COUNT(*) FROM pg_catalog.pg_settings WHERE pg_catalog.lower(name) != ALL(%s)'
|
||||||
|
' AND pending_restart', [n.lower() for n in self._RECOVERY_PARAMETERS]).fetchone()[0] > 0
|
||||||
self._postgresql.set_pending_restart(pending_restart)
|
self._postgresql.set_pending_restart(pending_restart)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.warning('Exception %r when running query', e)
|
logger.warning('Exception %r when running query', e)
|
||||||
@@ -1067,6 +1081,14 @@ class ConfigHandler(object):
|
|||||||
if cvalue > value:
|
if cvalue > value:
|
||||||
effective_configuration[name] = cvalue
|
effective_configuration[name] = cvalue
|
||||||
self._postgresql.set_pending_restart(True)
|
self._postgresql.set_pending_restart(True)
|
||||||
|
|
||||||
|
# If we are using custom bootstrap with PITR it could fail when values
|
||||||
|
# like max_connections are increased, therefore we disable hot_standby.
|
||||||
|
if self._postgresql.bootstrap.running_custom_bootstrap and \
|
||||||
|
(self._postgresql.bootstrap.keep_existing_recovery_conf or self._recovery_conf):
|
||||||
|
effective_configuration['hot_standby'] = 'off'
|
||||||
|
self._postgresql.set_pending_restart(True)
|
||||||
|
|
||||||
return effective_configuration
|
return effective_configuration
|
||||||
|
|
||||||
@property
|
@property
|
||||||
|
|||||||
@@ -1,9 +1,10 @@
|
|||||||
import logging
|
import logging
|
||||||
import psycopg2
|
|
||||||
|
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from threading import Lock
|
from threading import Lock
|
||||||
|
|
||||||
|
from .. import psycopg
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
@@ -20,7 +21,7 @@ class Connection(object):
|
|||||||
def get(self):
|
def get(self):
|
||||||
with self._lock:
|
with self._lock:
|
||||||
if not self._connection or self._connection.closed != 0:
|
if not self._connection or self._connection.closed != 0:
|
||||||
self._connection = psycopg2.connect(**self._conn_kwargs)
|
self._connection = psycopg.connect(**self._conn_kwargs)
|
||||||
self._connection.autocommit = True
|
self._connection.autocommit = True
|
||||||
self.server_version = self._connection.server_version
|
self.server_version = self._connection.server_version
|
||||||
return self._connection
|
return self._connection
|
||||||
@@ -40,7 +41,8 @@ class Connection(object):
|
|||||||
|
|
||||||
@contextmanager
|
@contextmanager
|
||||||
def get_connection_cursor(**kwargs):
|
def get_connection_cursor(**kwargs):
|
||||||
with psycopg2.connect(**kwargs) as conn:
|
conn = psycopg.connect(**kwargs)
|
||||||
conn.autocommit = True
|
conn.autocommit = True
|
||||||
with conn.cursor() as cur:
|
with conn.cursor() as cur:
|
||||||
yield cur
|
yield cur
|
||||||
|
conn.close()
|
||||||
|
|||||||
@@ -37,7 +37,7 @@ def postgres_version_to_int(pg_version):
|
|||||||
raise PostgresException('Invalid PostgreSQL version format: X.Y or X.Y.Z is accepted: {0}'.format(pg_version))
|
raise PostgresException('Invalid PostgreSQL version format: X.Y or X.Y.Z is accepted: {0}'.format(pg_version))
|
||||||
|
|
||||||
if len(components) == 2:
|
if len(components) == 2:
|
||||||
# new style verion numbers, i.e. 10.1 becomes 100001
|
# new style version numbers, i.e. 10.1 becomes 100001
|
||||||
components.insert(1, 0)
|
components.insert(1, 0)
|
||||||
|
|
||||||
return int(''.join('{0:02d}'.format(c) for c in components))
|
return int(''.join('{0:02d}'.format(c) for c in components))
|
||||||
@@ -68,3 +68,8 @@ def parse_history(data):
|
|||||||
yield values
|
yield values
|
||||||
except (IndexError, ValueError):
|
except (IndexError, ValueError):
|
||||||
logger.exception('Exception when parsing timeline history line "%s"', values)
|
logger.exception('Exception when parsing timeline history line "%s"', values)
|
||||||
|
|
||||||
|
|
||||||
|
def format_lsn(lsn, full=False):
|
||||||
|
template = '{0:X}/{1:08X}' if full else '{0:X}/{1:X}'
|
||||||
|
return template.format(lsn >> 32, lsn & 0xFFFFFFFF)
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ import subprocess
|
|||||||
from threading import Lock, Thread
|
from threading import Lock, Thread
|
||||||
|
|
||||||
from .connection import get_connection_cursor
|
from .connection import get_connection_cursor
|
||||||
from .misc import parse_history, parse_lsn
|
from .misc import format_lsn, parse_history, parse_lsn
|
||||||
from ..async_executor import CriticalTask
|
from ..async_executor import CriticalTask
|
||||||
from ..dcs import Leader
|
from ..dcs import Leader
|
||||||
|
|
||||||
@@ -17,11 +17,6 @@ REWIND_STATUS = type('Enum', (), {'INITIAL': 0, 'CHECKPOINT': 1, 'CHECK': 2, 'NE
|
|||||||
'NOT_NEED': 4, 'SUCCESS': 5, 'FAILED': 6})
|
'NOT_NEED': 4, 'SUCCESS': 5, 'FAILED': 6})
|
||||||
|
|
||||||
|
|
||||||
def format_lsn(lsn, full=False):
|
|
||||||
template = '{0:X}/{1:08X}' if full else '{0:X}/{1:X}'
|
|
||||||
return template.format(lsn >> 32, lsn & 0xFFFFFFFF)
|
|
||||||
|
|
||||||
|
|
||||||
class Rewind(object):
|
class Rewind(object):
|
||||||
|
|
||||||
def __init__(self, postgresql):
|
def __init__(self, postgresql):
|
||||||
@@ -70,26 +65,31 @@ class Rewind(object):
|
|||||||
except Exception:
|
except Exception:
|
||||||
return logger.exception('Exception when working with leader')
|
return logger.exception('Exception when working with leader')
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def check_leader_has_run_checkpoint(conn_kwargs):
|
||||||
|
try:
|
||||||
|
with get_connection_cursor(connect_timeout=3, options='-c statement_timeout=2000', **conn_kwargs) as cur:
|
||||||
|
cur.execute("SELECT NOT pg_catalog.pg_is_in_recovery()" +
|
||||||
|
" AND ('x' || pg_catalog.substr(pg_catalog.pg_walfile_name(" +
|
||||||
|
" pg_catalog.pg_current_wal_lsn()), 1, 8))::bit(32)::int = timeline_id" +
|
||||||
|
" FROM pg_catalog.pg_control_checkpoint()")
|
||||||
|
if not cur.fetchone()[0]:
|
||||||
|
return 'leader has not run a checkpoint yet'
|
||||||
|
except Exception:
|
||||||
|
logger.exception('Exception when working with leader')
|
||||||
|
return 'not accessible or not healty'
|
||||||
|
|
||||||
def _get_checkpoint_end(self, timeline, lsn):
|
def _get_checkpoint_end(self, timeline, lsn):
|
||||||
"""The checkpoint record size in WAL depends on postgres major version and platform (memory alignment).
|
"""The checkpoint record size in WAL depends on postgres major version and platform (memory alignment).
|
||||||
Hence, the only reliable way to figure out where it ends, read the record from file with the help of pg_waldump
|
Hence, the only reliable way to figure out where it ends, read the record from file with the help of pg_waldump
|
||||||
and parse the output. We are trying to read two records, and expect that it wil fail to read the second one:
|
and parse the output. We are trying to read two records, and expect that it will fail to read the second one:
|
||||||
`pg_waldump: fatal: error in WAL record at 0/182E220: invalid record length at 0/182E298: wanted 24, got 0`
|
`pg_waldump: fatal: error in WAL record at 0/182E220: invalid record length at 0/182E298: wanted 24, got 0`
|
||||||
The error message contains information about LSN of the next record, which is exactly where checkpoint ends."""
|
The error message contains information about LSN of the next record, which is exactly where checkpoint ends."""
|
||||||
|
|
||||||
cmd = self._postgresql.pgcommand('pg_{0}dump'.format(self._postgresql.wal_name))
|
|
||||||
lsn8 = format_lsn(lsn, True)
|
lsn8 = format_lsn(lsn, True)
|
||||||
lsn = format_lsn(lsn)
|
lsn = format_lsn(lsn)
|
||||||
env = os.environ.copy()
|
out, err = self._postgresql.waldump(timeline, lsn, 2)
|
||||||
env.update(LANG='C', LC_ALL='C', PGDATA=self._postgresql.data_dir)
|
if out is not None and err is not None:
|
||||||
try:
|
|
||||||
waldump = subprocess.Popen([cmd, '-t', str(timeline), '-s', lsn, '-n', '2'],
|
|
||||||
stdout=subprocess.PIPE, stderr=subprocess.PIPE, env=env)
|
|
||||||
out, err = waldump.communicate()
|
|
||||||
waldump.wait()
|
|
||||||
except Exception as e:
|
|
||||||
logger.error('Failed to execute `%s -t %s -s %s -n 2`: %r', cmd, timeline, lsn, e)
|
|
||||||
else:
|
|
||||||
out = out.decode('utf-8').rstrip().split('\n')
|
out = out.decode('utf-8').rstrip().split('\n')
|
||||||
err = err.decode('utf-8').rstrip().split('\n')
|
err = err.decode('utf-8').rstrip().split('\n')
|
||||||
pattern = 'error in WAL record at {0}: invalid record length at '.format(lsn)
|
pattern = 'error in WAL record at {0}: invalid record length at '.format(lsn)
|
||||||
@@ -102,7 +102,7 @@ class Rewind(object):
|
|||||||
return parse_lsn(err[0][i:j])
|
return parse_lsn(err[0][i:j])
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.error('Failed to parse lsn %s: %r', err[0][i:j], e)
|
logger.error('Failed to parse lsn %s: %r', err[0][i:j], e)
|
||||||
logger.error('Failed to parse `%s -t %s -s %s -n 2` output', cmd, timeline, lsn)
|
logger.error('Failed to parse pg_%sdump output', self._postgresql.wal_name)
|
||||||
logger.error(' stdout=%s', '\n'.join(out))
|
logger.error(' stdout=%s', '\n'.join(out))
|
||||||
logger.error(' stderr=%s', '\n'.join(err))
|
logger.error(' stderr=%s', '\n'.join(err))
|
||||||
|
|
||||||
@@ -166,8 +166,12 @@ class Rewind(object):
|
|||||||
|
|
||||||
def _conn_kwargs(self, member, auth):
|
def _conn_kwargs(self, member, auth):
|
||||||
ret = member.conn_kwargs(auth)
|
ret = member.conn_kwargs(auth)
|
||||||
if not ret.get('database'):
|
if not ret.get('dbname'):
|
||||||
ret['database'] = self._postgresql.database
|
ret['dbname'] = self._postgresql.database
|
||||||
|
# Add target_session_attrs in case more than one hostname is specified
|
||||||
|
# (libpq client-side failover) making sure we hit the primary
|
||||||
|
if 'target_session_attrs' not in ret and self._postgresql.major_version >= 100000:
|
||||||
|
ret['target_session_attrs'] = 'read-write'
|
||||||
return ret
|
return ret
|
||||||
|
|
||||||
def _check_timeline_and_lsn(self, leader):
|
def _check_timeline_and_lsn(self, leader):
|
||||||
@@ -175,11 +179,11 @@ class Rewind(object):
|
|||||||
if local_timeline is None or local_lsn is None:
|
if local_timeline is None or local_lsn is None:
|
||||||
return
|
return
|
||||||
|
|
||||||
if isinstance(leader, Leader):
|
if isinstance(leader, Leader) and leader.member.data.get('role') != 'master':
|
||||||
if leader.member.data.get('role') != 'master':
|
return
|
||||||
return
|
|
||||||
# standby cluster
|
if not self.check_leader_is_not_in_recovery(
|
||||||
elif not self.check_leader_is_not_in_recovery(self._conn_kwargs(leader, self._postgresql.config.replication)):
|
self._conn_kwargs(leader, self._postgresql.config.replication)):
|
||||||
return
|
return
|
||||||
|
|
||||||
history = need_rewind = None
|
history = need_rewind = None
|
||||||
@@ -193,8 +197,10 @@ class Rewind(object):
|
|||||||
elif local_timeline == master_timeline:
|
elif local_timeline == master_timeline:
|
||||||
need_rewind = False
|
need_rewind = False
|
||||||
elif master_timeline > 1:
|
elif master_timeline > 1:
|
||||||
cur.execute('TIMELINE_HISTORY %s', (master_timeline,))
|
cur.execute('TIMELINE_HISTORY {0}'.format(master_timeline))
|
||||||
history = bytes(cur.fetchone()[1]).decode('utf-8')
|
history = cur.fetchone()[1]
|
||||||
|
if not isinstance(history, six.string_types):
|
||||||
|
history = bytes(history).decode('utf-8')
|
||||||
logger.debug('master: history=%s', history)
|
logger.debug('master: history=%s', history)
|
||||||
except Exception:
|
except Exception:
|
||||||
return logger.exception('Exception when working with master via replication connection')
|
return logger.exception('Exception when working with master via replication connection')
|
||||||
@@ -214,7 +220,10 @@ class Rewind(object):
|
|||||||
need_rewind = switchpoint != self._get_checkpoint_end(local_timeline, local_lsn)
|
need_rewind = switchpoint != self._get_checkpoint_end(local_timeline, local_lsn)
|
||||||
break
|
break
|
||||||
elif parent_timeline > local_timeline:
|
elif parent_timeline > local_timeline:
|
||||||
|
need_rewind = True
|
||||||
break
|
break
|
||||||
|
else:
|
||||||
|
need_rewind = True
|
||||||
self._log_master_history(history, i)
|
self._log_master_history(history, i)
|
||||||
|
|
||||||
self._state = need_rewind and REWIND_STATUS.NEED or REWIND_STATUS.NOT_NEED
|
self._state = need_rewind and REWIND_STATUS.NEED or REWIND_STATUS.NOT_NEED
|
||||||
@@ -242,16 +251,14 @@ class Rewind(object):
|
|||||||
with self._checkpoint_task_lock:
|
with self._checkpoint_task_lock:
|
||||||
if self._checkpoint_task:
|
if self._checkpoint_task:
|
||||||
with self._checkpoint_task:
|
with self._checkpoint_task:
|
||||||
if self._checkpoint_task.result:
|
if self._checkpoint_task.result is not None:
|
||||||
self._state = REWIND_STATUS.CHECKPOINT
|
self._state = REWIND_STATUS.CHECKPOINT
|
||||||
if self._checkpoint_task.result is not False:
|
self._checkpoint_task = None
|
||||||
return
|
elif self._postgresql.get_master_timeline() == self._postgresql.pg_control_timeline():
|
||||||
|
self._state = REWIND_STATUS.CHECKPOINT
|
||||||
else:
|
else:
|
||||||
self._checkpoint_task = CriticalTask()
|
self._checkpoint_task = CriticalTask()
|
||||||
return Thread(target=self.__checkpoint, args=(self._checkpoint_task, wakeup)).start()
|
Thread(target=self.__checkpoint, args=(self._checkpoint_task, wakeup)).start()
|
||||||
|
|
||||||
if self._postgresql.get_master_timeline() == self._postgresql.pg_control_timeline():
|
|
||||||
self._state = REWIND_STATUS.CHECKPOINT
|
|
||||||
|
|
||||||
def checkpoint_after_promote(self):
|
def checkpoint_after_promote(self):
|
||||||
return self._state == REWIND_STATUS.CHECKPOINT
|
return self._state == REWIND_STATUS.CHECKPOINT
|
||||||
@@ -345,9 +352,14 @@ class Rewind(object):
|
|||||||
# running a checkpoint or
|
# running a checkpoint or
|
||||||
# waiting until Patroni on the master will expose checkpoint_after_promote=True
|
# waiting until Patroni on the master will expose checkpoint_after_promote=True
|
||||||
checkpoint_status = leader.checkpoint_after_promote if isinstance(leader, Leader) else None
|
checkpoint_status = leader.checkpoint_after_promote if isinstance(leader, Leader) else None
|
||||||
if checkpoint_status is None: # master still runs the old Patroni
|
if checkpoint_status is None: # we are the standby-cluster leader or master still runs the old Patroni
|
||||||
leader_status = self._postgresql.checkpoint(self._conn_kwargs(leader, self._postgresql.config.superuser))
|
# superuser credentials match rewind_credentials if the latter are not provided or we run 10 or older
|
||||||
if leader_status:
|
if self._postgresql.config.superuser == self._postgresql.config.rewind_credentials:
|
||||||
|
leader_status = self._postgresql.checkpoint(
|
||||||
|
self._conn_kwargs(leader, self._postgresql.config.superuser))
|
||||||
|
else: # we run 11+ and have a dedicated pg_rewind user
|
||||||
|
leader_status = self.check_leader_has_run_checkpoint(r)
|
||||||
|
if leader_status: # we tried to run/check for a checkpoint on the remote leader, but it failed
|
||||||
return logger.warning('Can not use %s for rewind: %s', leader.name, leader_status)
|
return logger.warning('Can not use %s for rewind: %s', leader.name, leader_status)
|
||||||
elif not checkpoint_status:
|
elif not checkpoint_status:
|
||||||
return logger.info('Waiting for checkpoint on %s before rewind', leader.name)
|
return logger.info('Waiting for checkpoint on %s before rewind', leader.name)
|
||||||
|
|||||||
+243
-51
@@ -1,14 +1,34 @@
|
|||||||
|
import errno
|
||||||
import logging
|
import logging
|
||||||
|
import os
|
||||||
|
import shutil
|
||||||
|
|
||||||
from patroni.postgresql.connection import get_connection_cursor
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
|
from contextlib import contextmanager
|
||||||
|
|
||||||
|
from .connection import get_connection_cursor
|
||||||
|
from .misc import format_lsn
|
||||||
|
from ..psycopg import OperationalError
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def compare_slots(s1, s2):
|
def compare_slots(s1, s2, dbid='database'):
|
||||||
return s1['type'] == s2['type'] and (s1['type'] == 'physical' or
|
return s1['type'] == s2['type'] and (s1['type'] == 'physical' or
|
||||||
s1['database'] == s2['database'] and s1['plugin'] == s2['plugin'])
|
s1.get(dbid) == s2.get(dbid) and s1['plugin'] == s2['plugin'])
|
||||||
|
|
||||||
|
|
||||||
|
def fsync_dir(path):
|
||||||
|
if os.name != 'nt':
|
||||||
|
fd = os.open(path, os.O_DIRECTORY)
|
||||||
|
try:
|
||||||
|
os.fsync(fd)
|
||||||
|
except OSError as e:
|
||||||
|
# Some filesystems don't like fsyncing directories and raise EINVAL. Ignoring it is usually safe.
|
||||||
|
if e.errno != errno.EINVAL:
|
||||||
|
raise
|
||||||
|
finally:
|
||||||
|
os.close(fd)
|
||||||
|
|
||||||
|
|
||||||
class SlotsHandler(object):
|
class SlotsHandler(object):
|
||||||
@@ -16,22 +36,67 @@ class SlotsHandler(object):
|
|||||||
def __init__(self, postgresql):
|
def __init__(self, postgresql):
|
||||||
self._postgresql = postgresql
|
self._postgresql = postgresql
|
||||||
self._replication_slots = {} # already existing replication slots
|
self._replication_slots = {} # already existing replication slots
|
||||||
|
self._unready_logical_slots = set()
|
||||||
self.schedule()
|
self.schedule()
|
||||||
|
|
||||||
def _query(self, sql, *params):
|
def _query(self, sql, *params):
|
||||||
return self._postgresql.query(sql, *params, retry=False)
|
return self._postgresql.query(sql, *params, retry=False)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _copy_items(src, dst, keys=None):
|
||||||
|
dst.update({key: src[key] for key in keys or ('datoid', 'catalog_xmin', 'confirmed_flush_lsn')})
|
||||||
|
|
||||||
|
def process_permanent_slots(self, slots):
|
||||||
|
"""This methods solves three problems at once (I know, it is weird).
|
||||||
|
|
||||||
|
The cluster_info_query from `Postgresql` is executed every HA loop and returns
|
||||||
|
information about all replication slots that exists on the current host.
|
||||||
|
Based on this information we perform the following actions:
|
||||||
|
1. For the primary we want to expose to DCS permanent logical slots, therefore the method
|
||||||
|
builds (and returns) a dict, that maps permanent logical slot names and confirmed_flush_lsns.
|
||||||
|
2. This method also detects if one of the previously known permanent slots got missing and schedules resync.
|
||||||
|
3. Updates the local cache with the fresh catalog_xmin and confirmed_flush_lsn for every known slot.
|
||||||
|
This info is used when performing the check of logical slot readiness on standbys.
|
||||||
|
"""
|
||||||
|
ret = {}
|
||||||
|
|
||||||
|
slots = {slot['slot_name']: slot for slot in slots or []}
|
||||||
|
if slots:
|
||||||
|
for name, value in slots.items():
|
||||||
|
if name in self._replication_slots:
|
||||||
|
if compare_slots(value, self._replication_slots[name], 'datoid'):
|
||||||
|
if value['type'] == 'logical':
|
||||||
|
ret[name] = value['confirmed_flush_lsn']
|
||||||
|
self._copy_items(value, self._replication_slots[name])
|
||||||
|
else:
|
||||||
|
self._schedule_load_slots = True
|
||||||
|
|
||||||
|
# It could happen that the slots was deleted in the background, we want to detect this case
|
||||||
|
if any(name not in slots for name in self._replication_slots.keys()):
|
||||||
|
self._schedule_load_slots = True
|
||||||
|
|
||||||
|
return ret
|
||||||
|
|
||||||
def load_replication_slots(self):
|
def load_replication_slots(self):
|
||||||
if self._postgresql.major_version >= 90400 and self._schedule_load_slots:
|
if self._postgresql.major_version >= 90400 and self._schedule_load_slots:
|
||||||
replication_slots = {}
|
replication_slots = {}
|
||||||
cursor = self._query('SELECT slot_name, slot_type, plugin, database FROM pg_catalog.pg_replication_slots')
|
extra = ", catalog_xmin, pg_catalog.pg_wal_lsn_diff(confirmed_flush_lsn, '0/0')::bigint"\
|
||||||
|
if self._postgresql.major_version >= 100000 else ""
|
||||||
|
skip_temp_slots = ' WHERE NOT temporary' if self._postgresql.major_version >= 100000 else ''
|
||||||
|
cursor = self._query('SELECT slot_name, slot_type, plugin, database, datoid'
|
||||||
|
'{0} FROM pg_catalog.pg_replication_slots{1}'.format(extra, skip_temp_slots))
|
||||||
for r in cursor:
|
for r in cursor:
|
||||||
value = {'type': r[1]}
|
value = {'type': r[1]}
|
||||||
if r[1] == 'logical':
|
if r[1] == 'logical':
|
||||||
value.update({'plugin': r[2], 'database': r[3]})
|
value.update(plugin=r[2], database=r[3], datoid=r[4])
|
||||||
|
if self._postgresql.major_version >= 100000:
|
||||||
|
value.update(catalog_xmin=r[5], confirmed_flush_lsn=r[6])
|
||||||
replication_slots[r[0]] = value
|
replication_slots[r[0]] = value
|
||||||
self._replication_slots = replication_slots
|
self._replication_slots = replication_slots
|
||||||
self._schedule_load_slots = False
|
self._schedule_load_slots = False
|
||||||
|
if self._force_readiness_check:
|
||||||
|
self._unready_logical_slots = set(n for n, v in replication_slots.items() if v['type'] == 'logical')
|
||||||
|
self._force_readiness_check = False
|
||||||
|
|
||||||
def ignore_replication_slot(self, cluster, name):
|
def ignore_replication_slot(self, cluster, name):
|
||||||
slot = self._replication_slots[name]
|
slot = self._replication_slots[name]
|
||||||
@@ -47,65 +112,192 @@ class SlotsHandler(object):
|
|||||||
# In normal situation rowcount should be 1, otherwise either slot doesn't exists or it is still active
|
# In normal situation rowcount should be 1, otherwise either slot doesn't exists or it is still active
|
||||||
return cursor.rowcount == 1
|
return cursor.rowcount == 1
|
||||||
|
|
||||||
def sync_replication_slots(self, cluster):
|
def _drop_incorrect_slots(self, cluster, slots):
|
||||||
|
# drop old replication slots which are not presented in desired slots
|
||||||
|
for name in set(self._replication_slots) - set(slots):
|
||||||
|
if not self.ignore_replication_slot(cluster, name) and not self.drop_replication_slot(name):
|
||||||
|
logger.error("Failed to drop replication slot '%s'", name)
|
||||||
|
self._schedule_load_slots = True
|
||||||
|
|
||||||
|
for name, value in slots.items():
|
||||||
|
if name in self._replication_slots and not compare_slots(value, self._replication_slots[name]):
|
||||||
|
logger.info("Trying to drop replication slot '%s' because value is changing from %s to %s",
|
||||||
|
name, self._replication_slots[name], value)
|
||||||
|
if self.drop_replication_slot(name):
|
||||||
|
self._replication_slots.pop(name)
|
||||||
|
else:
|
||||||
|
logger.error("Failed to drop replication slot '%s'", name)
|
||||||
|
self._schedule_load_slots = True
|
||||||
|
|
||||||
|
def _ensure_physical_slots(self, slots):
|
||||||
|
immediately_reserve = ', true' if self._postgresql.major_version >= 90600 else ''
|
||||||
|
for name, value in slots.items():
|
||||||
|
if name not in self._replication_slots and value['type'] == 'physical':
|
||||||
|
try:
|
||||||
|
self._query(("SELECT pg_catalog.pg_create_physical_replication_slot(%s{0})" +
|
||||||
|
" WHERE NOT EXISTS (SELECT 1 FROM pg_catalog.pg_replication_slots" +
|
||||||
|
" WHERE slot_type = 'physical' AND slot_name = %s)").format(
|
||||||
|
immediately_reserve), name, name)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Failed to create physical replication slot '%s'", name)
|
||||||
|
self._schedule_load_slots = True
|
||||||
|
|
||||||
|
@contextmanager
|
||||||
|
def _get_local_connection_cursor(self, **kwargs):
|
||||||
|
conn_kwargs = self._postgresql.config.local_connect_kwargs
|
||||||
|
conn_kwargs.update(kwargs)
|
||||||
|
with get_connection_cursor(**conn_kwargs) as cur:
|
||||||
|
yield cur
|
||||||
|
|
||||||
|
def _ensure_logical_slots_primary(self, slots):
|
||||||
|
# Group logical slots to be created by database name
|
||||||
|
logical_slots = defaultdict(dict)
|
||||||
|
for name, value in slots.items():
|
||||||
|
if value['type'] == 'logical':
|
||||||
|
# If the logical already exists, copy some information about it into the original structure
|
||||||
|
if self._replication_slots.get(name, {}).get('datoid'):
|
||||||
|
self._copy_items(self._replication_slots[name], value)
|
||||||
|
else:
|
||||||
|
logical_slots[value['database']][name] = value
|
||||||
|
|
||||||
|
# Create new logical slots
|
||||||
|
for database, values in logical_slots.items():
|
||||||
|
with self._get_local_connection_cursor(dbname=database) as cur:
|
||||||
|
for name, value in values.items():
|
||||||
|
try:
|
||||||
|
cur.execute("SELECT pg_catalog.pg_create_logical_replication_slot(%s, %s)" +
|
||||||
|
" WHERE NOT EXISTS (SELECT 1 FROM pg_catalog.pg_replication_slots" +
|
||||||
|
" WHERE slot_type = 'logical' AND slot_name = %s)",
|
||||||
|
(name, value['plugin'], name))
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("Failed to create logical replication slot '%s' plugin='%s': %r",
|
||||||
|
name, value['plugin'], e)
|
||||||
|
slots.pop(name)
|
||||||
|
self._schedule_load_slots = True
|
||||||
|
|
||||||
|
def _ensure_logical_slots_replica(self, cluster, slots):
|
||||||
|
advance_slots = defaultdict(dict) # Group logical slots to be advanced by database name
|
||||||
|
create_slots = [] # And collect logical slots to be created on the replica
|
||||||
|
for name, value in slots.items():
|
||||||
|
if value['type'] == 'logical':
|
||||||
|
# If the logical already exists, copy some information about it into the original structure
|
||||||
|
if self._replication_slots.get(name, {}).get('datoid'):
|
||||||
|
self._copy_items(self._replication_slots[name], value)
|
||||||
|
if name in cluster.slots:
|
||||||
|
try: # Skip slots that doesn't need to be advanced
|
||||||
|
if value['confirmed_flush_lsn'] < int(cluster.slots[name]):
|
||||||
|
advance_slots[value['database']][name] = value
|
||||||
|
except Exception as e:
|
||||||
|
logger.error('Failed to parse "%s": %r', cluster.slots[name], e)
|
||||||
|
elif name in cluster.slots: # We want to copy only slots with feedback in a DCS
|
||||||
|
create_slots.append(name)
|
||||||
|
|
||||||
|
# Advance logical slots
|
||||||
|
for database, values in advance_slots.items():
|
||||||
|
with self._get_local_connection_cursor(dbname=database, options='-c statement_timeout=0') as cur:
|
||||||
|
for name, value in values.items():
|
||||||
|
try:
|
||||||
|
cur.execute("SELECT pg_catalog.pg_replication_slot_advance(%s, %s)",
|
||||||
|
(name, format_lsn(int(cluster.slots[name]))))
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("Failed to advance logical replication slot '%s': %r", name, e)
|
||||||
|
if isinstance(e, OperationalError) and e.diag.sqlstate == '58P01': # WAL file is gone
|
||||||
|
create_slots.append(name)
|
||||||
|
self._schedule_load_slots = True
|
||||||
|
return create_slots
|
||||||
|
|
||||||
|
def sync_replication_slots(self, cluster, nofailover, replicatefrom=None):
|
||||||
|
ret = None
|
||||||
if self._postgresql.major_version >= 90400 and cluster.config:
|
if self._postgresql.major_version >= 90400 and cluster.config:
|
||||||
try:
|
try:
|
||||||
self.load_replication_slots()
|
self.load_replication_slots()
|
||||||
|
|
||||||
slots = cluster.get_replication_slots(self._postgresql.name, self._postgresql.role)
|
slots = cluster.get_replication_slots(self._postgresql.name, self._postgresql.role,
|
||||||
|
nofailover, self._postgresql.major_version, True)
|
||||||
|
|
||||||
# drop old replication slots which are not presented in desired slots
|
self._drop_incorrect_slots(cluster, slots)
|
||||||
for name in set(self._replication_slots) - set(slots):
|
|
||||||
if not self.ignore_replication_slot(cluster, name) and not self.drop_replication_slot(name):
|
|
||||||
logger.error("Failed to drop replication slot '%s'", name)
|
|
||||||
self._schedule_load_slots = True
|
|
||||||
|
|
||||||
immediately_reserve = ', true' if self._postgresql.major_version >= 90600 else ''
|
self._ensure_physical_slots(slots)
|
||||||
|
|
||||||
logical_slots = defaultdict(dict)
|
if self._postgresql.is_leader():
|
||||||
for name, value in slots.items():
|
self._unready_logical_slots.clear()
|
||||||
if name in self._replication_slots and not compare_slots(value, self._replication_slots[name]):
|
self._ensure_logical_slots_primary(slots)
|
||||||
logger.info("Trying to drop replication slot '%s' because value is changing from %s to %s",
|
elif cluster.slots and slots:
|
||||||
name, self._replication_slots[name], value)
|
self.check_logical_slots_readiness(cluster, nofailover, replicatefrom)
|
||||||
if not self.drop_replication_slot(name):
|
|
||||||
logger.error("Failed to drop replication slot '%s'", name)
|
ret = self._ensure_logical_slots_replica(cluster, slots)
|
||||||
self._schedule_load_slots = True
|
|
||||||
continue
|
|
||||||
self._replication_slots.pop(name)
|
|
||||||
if name not in self._replication_slots:
|
|
||||||
if value['type'] == 'physical':
|
|
||||||
try:
|
|
||||||
self._query(("SELECT pg_catalog.pg_create_physical_replication_slot(%s{0})" +
|
|
||||||
" WHERE NOT EXISTS (SELECT 1 FROM pg_catalog.pg_replication_slots" +
|
|
||||||
" WHERE slot_type = 'physical' AND slot_name = %s)").format(
|
|
||||||
immediately_reserve), name, name)
|
|
||||||
except Exception:
|
|
||||||
logger.exception("Failed to create physical replication slot '%s'", name)
|
|
||||||
self._schedule_load_slots = True
|
|
||||||
elif value['type'] == 'logical' and name not in self._replication_slots:
|
|
||||||
logical_slots[value['database']][name] = value
|
|
||||||
|
|
||||||
# create new logical slots
|
|
||||||
for database, values in logical_slots.items():
|
|
||||||
conn_kwargs = self._postgresql.config.local_connect_kwargs
|
|
||||||
conn_kwargs['database'] = database
|
|
||||||
with get_connection_cursor(**conn_kwargs) as cur:
|
|
||||||
for name, value in values.items():
|
|
||||||
try:
|
|
||||||
cur.execute("SELECT pg_catalog.pg_create_logical_replication_slot(%s, %s)" +
|
|
||||||
" WHERE NOT EXISTS (SELECT 1 FROM pg_catalog.pg_replication_slots" +
|
|
||||||
" WHERE slot_type = 'logical' AND slot_name = %s)",
|
|
||||||
(name, value['plugin'], name))
|
|
||||||
except Exception:
|
|
||||||
logger.exception("Failed to create logical replication slot '%s' plugin='%s'",
|
|
||||||
name, value['plugin'])
|
|
||||||
self._schedule_load_slots = True
|
|
||||||
self._replication_slots = slots
|
self._replication_slots = slots
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('Exception when changing replication slots')
|
logger.exception('Exception when changing replication slots')
|
||||||
self._schedule_load_slots = True
|
self._schedule_load_slots = True
|
||||||
|
return ret
|
||||||
|
|
||||||
|
@contextmanager
|
||||||
|
def _get_leader_connection_cursor(self, leader):
|
||||||
|
conn_kwargs = leader.conn_kwargs(self._postgresql.config.rewind_credentials)
|
||||||
|
conn_kwargs['dbname'] = self._postgresql.database
|
||||||
|
with get_connection_cursor(connect_timeout=3, options="-c statement_timeout=2000", **conn_kwargs) as cur:
|
||||||
|
yield cur
|
||||||
|
|
||||||
|
def check_logical_slots_readiness(self, cluster, nofailover, replicatefrom):
|
||||||
|
if self._unready_logical_slots:
|
||||||
|
slot_name = cluster.get_my_slot_name_on_primary(self._postgresql.name, replicatefrom)
|
||||||
|
try:
|
||||||
|
with self._get_leader_connection_cursor(cluster.leader) as cur:
|
||||||
|
cur.execute("SELECT catalog_xmin FROM pg_catalog.pg_get_replication_slots()"
|
||||||
|
" WHERE NOT pg_catalog.pg_is_in_recovery() AND slot_name = %s", (slot_name,))
|
||||||
|
if cur.rowcount < 1:
|
||||||
|
return logger.warning('Physical slot %s does not exist on the primary', slot_name)
|
||||||
|
catalog_xmin = cur.fetchone()[0]
|
||||||
|
except Exception as e:
|
||||||
|
return logger.error("Failed to check %s physical slot on the primary: %r", slot_name, e)
|
||||||
|
for name in list(self._unready_logical_slots):
|
||||||
|
value = self._replication_slots.get(name)
|
||||||
|
if not value or catalog_xmin <= value['catalog_xmin']:
|
||||||
|
self._unready_logical_slots.remove(name)
|
||||||
|
if value:
|
||||||
|
logger.info('Logical slot %s is safe to be used after a failover', name)
|
||||||
|
|
||||||
|
def copy_logical_slots(self, leader, slots):
|
||||||
|
with self._get_leader_connection_cursor(leader) as cur:
|
||||||
|
try:
|
||||||
|
cur.execute("SELECT slot_name, catalog_xmin, "
|
||||||
|
"pg_catalog.pg_wal_lsn_diff(confirmed_flush_lsn, '0/0')::bigint, "
|
||||||
|
"pg_catalog.pg_read_binary_file('pg_replslot/' || slot_name || '/state')"
|
||||||
|
" FROM pg_catalog.pg_get_replication_slots() WHERE NOT pg_catalog.pg_is_in_recovery()"
|
||||||
|
" AND slot_name = ANY(%s)", (slots,))
|
||||||
|
slots = {r[0]: {'catalog_xmin': r[1], 'confirmed_flush_lsn': r[2], 'data': r[3]} for r in cur}
|
||||||
|
except Exception as e:
|
||||||
|
logger.error("Failed to copy logical slots from the %s via postgresql connection: %r", leader.name, e)
|
||||||
|
|
||||||
|
if isinstance(slots, dict) and self._postgresql.stop():
|
||||||
|
pg_replslot_dir = os.path.join(self._postgresql.data_dir, 'pg_replslot')
|
||||||
|
for name, value in slots.items():
|
||||||
|
slot_dir = os.path.join(pg_replslot_dir, name)
|
||||||
|
slot_tmp_dir = slot_dir + '.tmp'
|
||||||
|
if os.path.exists(slot_tmp_dir):
|
||||||
|
shutil.rmtree(slot_tmp_dir)
|
||||||
|
os.makedirs(slot_tmp_dir)
|
||||||
|
fsync_dir(slot_tmp_dir)
|
||||||
|
with open(os.path.join(slot_tmp_dir, 'state'), 'wb') as f:
|
||||||
|
f.write(value['data'])
|
||||||
|
f.flush()
|
||||||
|
os.fsync(f.fileno())
|
||||||
|
if os.path.exists(slot_dir):
|
||||||
|
shutil.rmtree(slot_dir)
|
||||||
|
os.rename(slot_tmp_dir, slot_dir)
|
||||||
|
fsync_dir(slot_dir)
|
||||||
|
self._unready_logical_slots.add(name)
|
||||||
|
fsync_dir(pg_replslot_dir)
|
||||||
|
self._postgresql.start()
|
||||||
|
|
||||||
def schedule(self, value=None):
|
def schedule(self, value=None):
|
||||||
if value is None:
|
if value is None:
|
||||||
value = self._postgresql.major_version >= 90400
|
value = self._postgresql.major_version >= 90400
|
||||||
self._schedule_load_slots = value
|
self._schedule_load_slots = self._force_readiness_check = value
|
||||||
|
|
||||||
|
def on_promote(self):
|
||||||
|
if self._unready_logical_slots:
|
||||||
|
logger.warning('Logical replication slots that might be unsafe to use after promote: %s',
|
||||||
|
self._unready_logical_slots)
|
||||||
|
|||||||
@@ -108,7 +108,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
),
|
),
|
||||||
'archive_timeout': Integer(90300, None, 0, 1073741823, 's'),
|
'archive_timeout': Integer(90300, None, 0, 1073741823, 's'),
|
||||||
'array_nulls': Bool(90300, None),
|
'array_nulls': Bool(90300, None),
|
||||||
'authentication_timeout': Integer(90300, None, 1, 600, 's'),
|
'authentication_timeout': Integer(90300, None, 1, 600, 's'),
|
||||||
'autovacuum': Bool(90300, None),
|
'autovacuum': Bool(90300, None),
|
||||||
'autovacuum_analyze_scale_factor': Real(90300, None, 0, 100, None),
|
'autovacuum_analyze_scale_factor': Real(90300, None, 0, 100, None),
|
||||||
'autovacuum_analyze_threshold': Integer(90300, None, 0, 2147483647, None),
|
'autovacuum_analyze_threshold': Integer(90300, None, 0, 2147483647, None),
|
||||||
@@ -151,12 +151,14 @@ parameters = CaseInsensitiveDict({
|
|||||||
Integer(90600, None, 30, 86400, 's')
|
Integer(90600, None, 30, 86400, 's')
|
||||||
),
|
),
|
||||||
'checkpoint_warning': Integer(90300, None, 0, 2147483647, 's'),
|
'checkpoint_warning': Integer(90300, None, 0, 2147483647, 's'),
|
||||||
|
'client_connection_check_interval': Integer(140000, None, 0, 2147483647, 'ms'),
|
||||||
'client_encoding': String(90300, None),
|
'client_encoding': String(90300, None),
|
||||||
'client_min_messages': Enum(90300, None, ('debug5', 'debug4', 'debug3', 'debug2',
|
'client_min_messages': Enum(90300, None, ('debug5', 'debug4', 'debug3', 'debug2',
|
||||||
'debug1', 'log', 'notice', 'warning', 'error')),
|
'debug1', 'log', 'notice', 'warning', 'error')),
|
||||||
'cluster_name': String(90500, None),
|
'cluster_name': String(90500, None),
|
||||||
'commit_delay': Integer(90300, None, 0, 100000, None),
|
'commit_delay': Integer(90300, None, 0, 100000, None),
|
||||||
'commit_siblings': Integer(90300, None, 0, 1000, None),
|
'commit_siblings': Integer(90300, None, 0, 1000, None),
|
||||||
|
'compute_query_id': EnumBool(140000, None, ('auto',)),
|
||||||
'config_file': String(90300, None),
|
'config_file': String(90300, None),
|
||||||
'constraint_exclusion': EnumBool(90300, None, ('partition',)),
|
'constraint_exclusion': EnumBool(90300, None, ('partition',)),
|
||||||
'cpu_index_tuple_cost': Real(90300, None, 0, 1.79769e+308, None),
|
'cpu_index_tuple_cost': Real(90300, None, 0, 1.79769e+308, None),
|
||||||
@@ -176,6 +178,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
'default_table_access_method': String(120000, None),
|
'default_table_access_method': String(120000, None),
|
||||||
'default_tablespace': String(90300, None),
|
'default_tablespace': String(90300, None),
|
||||||
'default_text_search_config': String(90300, None),
|
'default_text_search_config': String(90300, None),
|
||||||
|
'default_toast_compression': Enum(140000, None, ('pglz', 'lz4')),
|
||||||
'default_transaction_deferrable': Bool(90300, None),
|
'default_transaction_deferrable': Bool(90300, None),
|
||||||
'default_transaction_isolation': Enum(90300, None, ('serializable', 'repeatable read',
|
'default_transaction_isolation': Enum(90300, None, ('serializable', 'repeatable read',
|
||||||
'read committed', 'read uncommitted')),
|
'read committed', 'read uncommitted')),
|
||||||
@@ -188,6 +191,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
),
|
),
|
||||||
'effective_cache_size': Integer(90300, None, 1, 2147483647, '8kB'),
|
'effective_cache_size': Integer(90300, None, 1, 2147483647, '8kB'),
|
||||||
'effective_io_concurrency': Integer(90300, None, 0, 1000, None),
|
'effective_io_concurrency': Integer(90300, None, 0, 1000, None),
|
||||||
|
'enable_async_append': Bool(140000, None),
|
||||||
'enable_bitmapscan': Bool(90300, None),
|
'enable_bitmapscan': Bool(90300, None),
|
||||||
'enable_gathermerge': Bool(100000, None),
|
'enable_gathermerge': Bool(100000, None),
|
||||||
'enable_hashagg': Bool(90300, None),
|
'enable_hashagg': Bool(90300, None),
|
||||||
@@ -209,6 +213,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
'escape_string_warning': Bool(90300, None),
|
'escape_string_warning': Bool(90300, None),
|
||||||
'event_source': String(90300, None),
|
'event_source': String(90300, None),
|
||||||
'exit_on_error': Bool(90300, None),
|
'exit_on_error': Bool(90300, None),
|
||||||
|
'extension_destdir': String(140000, None),
|
||||||
'external_pid_file': String(90300, None),
|
'external_pid_file': String(90300, None),
|
||||||
'extra_float_digits': Integer(90300, None, -15, 3, None),
|
'extra_float_digits': Integer(90300, None, -15, 3, None),
|
||||||
'force_parallel_mode': EnumBool(90600, None, ('regress',)),
|
'force_parallel_mode': EnumBool(90600, None, ('regress',)),
|
||||||
@@ -229,8 +234,10 @@ parameters = CaseInsensitiveDict({
|
|||||||
'hot_standby': Bool(90300, None),
|
'hot_standby': Bool(90300, None),
|
||||||
'hot_standby_feedback': Bool(90300, None),
|
'hot_standby_feedback': Bool(90300, None),
|
||||||
'huge_pages': EnumBool(90400, None, ('try',)),
|
'huge_pages': EnumBool(90400, None, ('try',)),
|
||||||
|
'huge_page_size': Integer(140000, None, 0, 2147483647, 'kB'),
|
||||||
'ident_file': String(90300, None),
|
'ident_file': String(90300, None),
|
||||||
'idle_in_transaction_session_timeout': Integer(90600, None, 0, 2147483647, 'ms'),
|
'idle_in_transaction_session_timeout': Integer(90600, None, 0, 2147483647, 'ms'),
|
||||||
|
'idle_session_timeout': Integer(140000, None, 0, 2147483647, 'ms'),
|
||||||
'ignore_checksum_failure': Bool(90300, None),
|
'ignore_checksum_failure': Bool(90300, None),
|
||||||
'ignore_invalid_pages': Bool(130000, None),
|
'ignore_invalid_pages': Bool(130000, None),
|
||||||
'ignore_system_indexes': Bool(90300, None),
|
'ignore_system_indexes': Bool(90300, None),
|
||||||
@@ -283,6 +290,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
'log_parameter_max_length_on_error': Integer(130000, None, -1, 1073741823, 'B'),
|
'log_parameter_max_length_on_error': Integer(130000, None, -1, 1073741823, 'B'),
|
||||||
'log_parser_stats': Bool(90300, None),
|
'log_parser_stats': Bool(90300, None),
|
||||||
'log_planner_stats': Bool(90300, None),
|
'log_planner_stats': Bool(90300, None),
|
||||||
|
'log_recovery_conflict_waits': Bool(140000, None),
|
||||||
'log_replication_commands': Bool(90500, None),
|
'log_replication_commands': Bool(90500, None),
|
||||||
'log_rotation_age': Integer(90300, None, 0, 35791394, 'min'),
|
'log_rotation_age': Integer(90300, None, 0, 35791394, 'min'),
|
||||||
'log_rotation_size': Integer(90300, None, 0, 2097151, 'kB'),
|
'log_rotation_size': Integer(90300, None, 0, 2097151, 'kB'),
|
||||||
@@ -336,6 +344,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
Integer(90400, 90600, 1, 8388607, None),
|
Integer(90400, 90600, 1, 8388607, None),
|
||||||
Integer(90600, None, 0, 262143, None)
|
Integer(90600, None, 0, 262143, None)
|
||||||
),
|
),
|
||||||
|
'min_dynamic_shared_memory': Integer(140000, None, 0, 2147483647, 'MB'),
|
||||||
'min_parallel_index_scan_size': Integer(100000, None, 0, 715827882, '8kB'),
|
'min_parallel_index_scan_size': Integer(100000, None, 0, 715827882, '8kB'),
|
||||||
'min_parallel_relation_size': Integer(90600, 100000, 0, 715827882, '8kB'),
|
'min_parallel_relation_size': Integer(90600, 100000, 0, 715827882, '8kB'),
|
||||||
'min_parallel_table_scan_size': Integer(100000, None, 0, 715827882, '8kB'),
|
'min_parallel_table_scan_size': Integer(100000, None, 0, 715827882, '8kB'),
|
||||||
@@ -344,7 +353,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
Integer(100000, None, 2, 2147483647, 'MB')
|
Integer(100000, None, 2, 2147483647, 'MB')
|
||||||
),
|
),
|
||||||
'old_snapshot_threshold': Integer(90600, None, -1, 86400, 'min'),
|
'old_snapshot_threshold': Integer(90600, None, -1, 86400, 'min'),
|
||||||
'operator_precedence_warning': Bool(90500, None),
|
'operator_precedence_warning': Bool(90500, 140000),
|
||||||
'parallel_leader_participation': Bool(110000, None),
|
'parallel_leader_participation': Bool(110000, None),
|
||||||
'parallel_setup_cost': Real(90600, None, 0, 1.79769e+308, None),
|
'parallel_setup_cost': Real(90600, None, 0, 1.79769e+308, None),
|
||||||
'parallel_tuple_cost': Real(90600, None, 0, 1.79769e+308, None),
|
'parallel_tuple_cost': Real(90600, None, 0, 1.79769e+308, None),
|
||||||
@@ -358,6 +367,8 @@ parameters = CaseInsensitiveDict({
|
|||||||
'pre_auth_delay': Integer(90300, None, 0, 60, 's'),
|
'pre_auth_delay': Integer(90300, None, 0, 60, 's'),
|
||||||
'quote_all_identifiers': Bool(90300, None),
|
'quote_all_identifiers': Bool(90300, None),
|
||||||
'random_page_cost': Real(90300, None, 0, 1.79769e+308, None),
|
'random_page_cost': Real(90300, None, 0, 1.79769e+308, None),
|
||||||
|
'recovery_init_sync_method': Enum(140000, None, ('fsync', 'syncfs')),
|
||||||
|
'remove_temp_files_after_crash': Bool(140000, None),
|
||||||
'replacement_sort_tuples': Integer(90600, 110000, 0, 2147483647, None),
|
'replacement_sort_tuples': Integer(90600, 110000, 0, 2147483647, None),
|
||||||
'restart_after_crash': Bool(90300, None),
|
'restart_after_crash': Bool(90300, None),
|
||||||
'row_security': Bool(90500, None),
|
'row_security': Bool(90500, None),
|
||||||
@@ -373,6 +384,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
'ssl_ca_file': String(90300, None),
|
'ssl_ca_file': String(90300, None),
|
||||||
'ssl_cert_file': String(90300, None),
|
'ssl_cert_file': String(90300, None),
|
||||||
'ssl_ciphers': String(90300, None),
|
'ssl_ciphers': String(90300, None),
|
||||||
|
'ssl_crl_dir': String(140000, None),
|
||||||
'ssl_crl_file': String(90300, None),
|
'ssl_crl_file': String(90300, None),
|
||||||
'ssl_dh_params_file': String(100000, None),
|
'ssl_dh_params_file': String(100000, None),
|
||||||
'ssl_ecdh_curve': String(90400, None),
|
'ssl_ecdh_curve': String(90400, None),
|
||||||
@@ -388,7 +400,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
'stats_temp_directory': String(90300, None),
|
'stats_temp_directory': String(90300, None),
|
||||||
'superuser_reserved_connections': (
|
'superuser_reserved_connections': (
|
||||||
Integer(90300, 90600, 0, 8388607, None),
|
Integer(90300, 90600, 0, 8388607, None),
|
||||||
Integer(90600, None, 0, 262143, None),
|
Integer(90600, None, 0, 262143, None)
|
||||||
),
|
),
|
||||||
'synchronize_seqscans': Bool(90300, None),
|
'synchronize_seqscans': Bool(90300, None),
|
||||||
'synchronous_commit': (
|
'synchronous_commit': (
|
||||||
@@ -424,6 +436,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
'track_counts': Bool(90300, None),
|
'track_counts': Bool(90300, None),
|
||||||
'track_functions': Enum(90300, None, ('none', 'pl', 'all')),
|
'track_functions': Enum(90300, None, ('none', 'pl', 'all')),
|
||||||
'track_io_timing': Bool(90300, None),
|
'track_io_timing': Bool(90300, None),
|
||||||
|
'track_wal_io_timing': Bool(140000, None),
|
||||||
'transaction_deferrable': Bool(90300, None),
|
'transaction_deferrable': Bool(90300, None),
|
||||||
'transaction_isolation': Enum(90300, None, ('serializable', 'repeatable read',
|
'transaction_isolation': Enum(90300, None, ('serializable', 'repeatable read',
|
||||||
'read committed', 'read uncommitted')),
|
'read committed', 'read uncommitted')),
|
||||||
@@ -433,7 +446,7 @@ parameters = CaseInsensitiveDict({
|
|||||||
'unix_socket_group': String(90300, None),
|
'unix_socket_group': String(90300, None),
|
||||||
'unix_socket_permissions': Integer(90300, None, 0, 511, None),
|
'unix_socket_permissions': Integer(90300, None, 0, 511, None),
|
||||||
'update_process_title': Bool(90300, None),
|
'update_process_title': Bool(90300, None),
|
||||||
'vacuum_cleanup_index_scale_factor': Real(110000, None, 0, 1e+10, None),
|
'vacuum_cleanup_index_scale_factor': Real(110000, 140000, 0, 1e+10, None),
|
||||||
'vacuum_cost_delay': (
|
'vacuum_cost_delay': (
|
||||||
Integer(90300, 120000, 0, 100, 'ms'),
|
Integer(90300, 120000, 0, 100, 'ms'),
|
||||||
Real(120000, None, 0, 100, 'ms')
|
Real(120000, None, 0, 100, 'ms')
|
||||||
@@ -443,8 +456,10 @@ parameters = CaseInsensitiveDict({
|
|||||||
'vacuum_cost_page_hit': Integer(90300, None, 0, 10000, None),
|
'vacuum_cost_page_hit': Integer(90300, None, 0, 10000, None),
|
||||||
'vacuum_cost_page_miss': Integer(90300, None, 0, 10000, None),
|
'vacuum_cost_page_miss': Integer(90300, None, 0, 10000, None),
|
||||||
'vacuum_defer_cleanup_age': Integer(90300, None, 0, 1000000, None),
|
'vacuum_defer_cleanup_age': Integer(90300, None, 0, 1000000, None),
|
||||||
|
'vacuum_failsafe_age': Integer(140000, None, 0, 2100000000, None),
|
||||||
'vacuum_freeze_min_age': Integer(90300, None, 0, 1000000000, None),
|
'vacuum_freeze_min_age': Integer(90300, None, 0, 1000000000, None),
|
||||||
'vacuum_freeze_table_age': Integer(90300, None, 0, 2000000000, None),
|
'vacuum_freeze_table_age': Integer(90300, None, 0, 2000000000, None),
|
||||||
|
'vacuum_multixact_failsafe_age': Integer(140000, None, 0, 2100000000, None),
|
||||||
'vacuum_multixact_freeze_min_age': Integer(90300, None, 0, 1000000000, None),
|
'vacuum_multixact_freeze_min_age': Integer(90300, None, 0, 1000000000, None),
|
||||||
'vacuum_multixact_freeze_table_age': Integer(90300, None, 0, 2000000000, None),
|
'vacuum_multixact_freeze_table_age': Integer(90300, None, 0, 2000000000, None),
|
||||||
'wal_buffers': Integer(90300, None, -1, 262143, '8kB'),
|
'wal_buffers': Integer(90300, None, -1, 262143, '8kB'),
|
||||||
@@ -511,6 +526,8 @@ def _transform_parameter_value(validators, version, name, value):
|
|||||||
def transform_postgresql_parameter_value(version, name, value):
|
def transform_postgresql_parameter_value(version, name, value):
|
||||||
if '.' in name:
|
if '.' in name:
|
||||||
return value
|
return value
|
||||||
|
if name in recovery_parameters:
|
||||||
|
return None
|
||||||
return _transform_parameter_value(parameters, version, name, value)
|
return _transform_parameter_value(parameters, version, name, value)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,40 @@
|
|||||||
|
__all__ = ['connect', 'quote_ident', 'quote_literal', 'DatabaseError', 'Error', 'OperationalError', 'ProgrammingError']
|
||||||
|
|
||||||
|
_legacy = False
|
||||||
|
try:
|
||||||
|
from psycopg2 import __version__
|
||||||
|
from . import MIN_PSYCOPG2, parse_version
|
||||||
|
if parse_version(__version__) < MIN_PSYCOPG2:
|
||||||
|
raise ImportError
|
||||||
|
from psycopg2 import connect, Error, DatabaseError, OperationalError, ProgrammingError
|
||||||
|
from psycopg2.extensions import adapt
|
||||||
|
|
||||||
|
try:
|
||||||
|
from psycopg2.extensions import quote_ident as _quote_ident
|
||||||
|
except ImportError:
|
||||||
|
_legacy = True
|
||||||
|
|
||||||
|
def quote_literal(value, conn=None):
|
||||||
|
value = adapt(value)
|
||||||
|
if conn:
|
||||||
|
value.prepare(conn)
|
||||||
|
return value.getquoted().decode('utf-8')
|
||||||
|
except ImportError:
|
||||||
|
from psycopg import connect as _connect, sql, Error, DatabaseError, OperationalError, ProgrammingError
|
||||||
|
|
||||||
|
def connect(*args, **kwargs):
|
||||||
|
ret = _connect(*args, **kwargs)
|
||||||
|
ret.server_version = ret.pgconn.server_version # compatibility with psycopg2
|
||||||
|
return ret
|
||||||
|
|
||||||
|
def _quote_ident(value, conn):
|
||||||
|
return sql.Identifier(value).as_string(conn)
|
||||||
|
|
||||||
|
def quote_literal(value, conn=None):
|
||||||
|
return sql.Literal(value).as_string(conn)
|
||||||
|
|
||||||
|
|
||||||
|
def quote_ident(value, conn=None):
|
||||||
|
if _legacy or conn is None:
|
||||||
|
return '"{0}"'.format(value.replace('"', '""'))
|
||||||
|
return _quote_ident(value, conn)
|
||||||
@@ -1,9 +1,7 @@
|
|||||||
import logging
|
import logging
|
||||||
import os
|
|
||||||
|
|
||||||
from patroni.daemon import AbstractPatroniDaemon, abstract_main
|
from .daemon import AbstractPatroniDaemon, abstract_main
|
||||||
from patroni.dcs.raft import KVStoreTTL
|
from .dcs.raft import KVStoreTTL
|
||||||
from pysyncobj import SyncObjConf
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -13,16 +11,13 @@ class RaftController(AbstractPatroniDaemon):
|
|||||||
def __init__(self, config):
|
def __init__(self, config):
|
||||||
super(RaftController, self).__init__(config)
|
super(RaftController, self).__init__(config)
|
||||||
|
|
||||||
raft_config = self.config.get('raft')
|
config = self.config.get('raft')
|
||||||
self_addr = raft_config['self_addr']
|
assert 'self_addr' in config
|
||||||
template = os.path.join(raft_config.get('data_dir', ''), self_addr)
|
self._raft = KVStoreTTL(None, None, None, **config)
|
||||||
self._syncobj_config = SyncObjConf(autoTick=False, appendEntriesUseBatch=False, dynamicMembershipChange=True,
|
|
||||||
journalFile=template + '.journal', fullDumpFile=template + '.dump')
|
|
||||||
self._raft = KVStoreTTL(self_addr, raft_config.get('partner_addrs', []), self._syncobj_config)
|
|
||||||
|
|
||||||
def _run_cycle(self):
|
def _run_cycle(self):
|
||||||
try:
|
try:
|
||||||
self._raft.doTick(self._syncobj_config.autoTickPeriod)
|
self._raft.doTick(self._raft.conf.autoTickPeriod)
|
||||||
except Exception:
|
except Exception:
|
||||||
logger.exception('doTick')
|
logger.exception('doTick')
|
||||||
|
|
||||||
|
|||||||
@@ -34,6 +34,9 @@ class PatroniRequest(object):
|
|||||||
|
|
||||||
if self._apply_ssl_file_param(config, 'cert'):
|
if self._apply_ssl_file_param(config, 'cert'):
|
||||||
self._apply_ssl_file_param(config, 'key')
|
self._apply_ssl_file_param(config, 'key')
|
||||||
|
|
||||||
|
password = self._get_cfg_value(config, 'keyfile_password')
|
||||||
|
self._apply_pool_param('key_password', password)
|
||||||
else:
|
else:
|
||||||
self._pool.connection_pool_kw.pop('key_file', None)
|
self._pool.connection_pool_kw.pop('key_file', None)
|
||||||
|
|
||||||
|
|||||||
@@ -27,13 +27,14 @@ import argparse
|
|||||||
import csv
|
import csv
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import psycopg2
|
|
||||||
import subprocess
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from collections import namedtuple
|
from collections import namedtuple
|
||||||
|
|
||||||
|
from .. import psycopg
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
RETRY_SLEEP_INTERVAL = 1
|
RETRY_SLEEP_INTERVAL = 1
|
||||||
@@ -215,7 +216,7 @@ class WALERestore(object):
|
|||||||
if self.master_connection:
|
if self.master_connection:
|
||||||
try:
|
try:
|
||||||
# get the difference in bytes between the current WAL location and the backup start offset
|
# get the difference in bytes between the current WAL location and the backup start offset
|
||||||
with psycopg2.connect(self.master_connection) as con:
|
with psycopg.connect(self.master_connection) as con:
|
||||||
if con.server_version >= 100000:
|
if con.server_version >= 100000:
|
||||||
wal_name = 'wal'
|
wal_name = 'wal'
|
||||||
lsn_name = 'lsn'
|
lsn_name = 'lsn'
|
||||||
@@ -233,7 +234,7 @@ class WALERestore(object):
|
|||||||
(backup_start_lsn, backup_start_lsn, backup_start_lsn))
|
(backup_start_lsn, backup_start_lsn, backup_start_lsn))
|
||||||
|
|
||||||
diff_in_bytes = int(cur.fetchone()[0])
|
diff_in_bytes = int(cur.fetchone()[0])
|
||||||
except psycopg2.Error:
|
except psycopg.Error:
|
||||||
logger.exception('could not determine difference with the master location')
|
logger.exception('could not determine difference with the master location')
|
||||||
if attempts_no < self.retries: # retry in case of a temporarily connection issue
|
if attempts_no < self.retries: # retry in case of a temporarily connection issue
|
||||||
attempts_no = attempts_no + 1
|
attempts_no = attempts_no + 1
|
||||||
@@ -250,7 +251,7 @@ class WALERestore(object):
|
|||||||
diff_in_bytes = 0
|
diff_in_bytes = 0
|
||||||
break
|
break
|
||||||
|
|
||||||
# if the size of the accumulated WAL segments is more than a certan percentage of the backup size
|
# if the size of the accumulated WAL segments is more than a certain percentage of the backup size
|
||||||
# or exceeds the pre-determined size - pg_basebackup is chosen instead.
|
# or exceeds the pre-determined size - pg_basebackup is chosen instead.
|
||||||
is_size_thresh_ok = diff_in_bytes < int(threshold_megabytes) * 1048576
|
is_size_thresh_ok = diff_in_bytes < int(threshold_megabytes) * 1048576
|
||||||
threshold_pct_bytes = backup_size * threshold_percent / 100.0
|
threshold_pct_bytes = backup_size * threshold_percent / 100.0
|
||||||
@@ -308,7 +309,7 @@ class WALERestore(object):
|
|||||||
try:
|
try:
|
||||||
os.mkdir(path)
|
os.mkdir(path)
|
||||||
except OSError:
|
except OSError:
|
||||||
logger.exception("coud not create missing %s directory path", dirname)
|
logger.exception("could not create missing %s directory path", dirname)
|
||||||
return False
|
return False
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
|||||||
+5
-5
@@ -405,7 +405,7 @@ def is_standby_cluster(config):
|
|||||||
|
|
||||||
def cluster_as_json(cluster):
|
def cluster_as_json(cluster):
|
||||||
leader_name = cluster.leader.name if cluster.leader else None
|
leader_name = cluster.leader.name if cluster.leader else None
|
||||||
xlog_location_cluster = cluster.last_leader_operation or 0
|
cluster_lsn = cluster.last_lsn or 0
|
||||||
|
|
||||||
ret = {'members': []}
|
ret = {'members': []}
|
||||||
for m in cluster.members:
|
for m in cluster.members:
|
||||||
@@ -427,11 +427,11 @@ def cluster_as_json(cluster):
|
|||||||
member.update({n: m.data[n] for n in optional_attributes if n in m.data})
|
member.update({n: m.data[n] for n in optional_attributes if n in m.data})
|
||||||
|
|
||||||
if m.name != leader_name:
|
if m.name != leader_name:
|
||||||
xlog_location = m.data.get('xlog_location')
|
lsn = m.data.get('xlog_location')
|
||||||
if xlog_location is None:
|
if lsn is None:
|
||||||
member['lag'] = 'unknown'
|
member['lag'] = 'unknown'
|
||||||
elif xlog_location_cluster >= xlog_location:
|
elif cluster_lsn >= lsn:
|
||||||
member['lag'] = xlog_location_cluster - xlog_location
|
member['lag'] = cluster_lsn - lsn
|
||||||
else:
|
else:
|
||||||
member['lag'] = 0
|
member['lag'] = 0
|
||||||
|
|
||||||
|
|||||||
@@ -302,10 +302,11 @@ validate_host_port_listen.expected_type = string_types
|
|||||||
validate_host_port_listen_multiple_hosts.expected_type = string_types
|
validate_host_port_listen_multiple_hosts.expected_type = string_types
|
||||||
validate_data_dir.expected_type = string_types
|
validate_data_dir.expected_type = string_types
|
||||||
validate_etcd = {
|
validate_etcd = {
|
||||||
Or("host", "hosts", "srv", "url", "proxy"): Case({
|
Or("host", "hosts", "srv", "srv_suffix", "url", "proxy"): Case({
|
||||||
"host": validate_host_port,
|
"host": validate_host_port,
|
||||||
"hosts": Or(comma_separated_host_port, [validate_host_port]),
|
"hosts": Or(comma_separated_host_port, [validate_host_port]),
|
||||||
"srv": str,
|
"srv": str,
|
||||||
|
"srv_suffix": str,
|
||||||
"url": str,
|
"url": str,
|
||||||
"proxy": str})
|
"proxy": str})
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-1
@@ -1 +1 @@
|
|||||||
__version__ = '2.0.2'
|
__version__ = '2.1.3'
|
||||||
|
|||||||
+6
-1
@@ -57,10 +57,15 @@ bootstrap:
|
|||||||
parameters:
|
parameters:
|
||||||
# wal_level: hot_standby
|
# wal_level: hot_standby
|
||||||
# hot_standby: "on"
|
# hot_standby: "on"
|
||||||
|
# max_connections: 100
|
||||||
|
# max_worker_processes: 8
|
||||||
# wal_keep_segments: 8
|
# wal_keep_segments: 8
|
||||||
# max_wal_senders: 10
|
# max_wal_senders: 10
|
||||||
# max_replication_slots: 10
|
# max_replication_slots: 10
|
||||||
|
# max_prepared_transactions: 0
|
||||||
|
# max_locks_per_transaction: 64
|
||||||
# wal_log_hints: "on"
|
# wal_log_hints: "on"
|
||||||
|
# track_commit_timestamp: "off"
|
||||||
# archive_mode: "on"
|
# archive_mode: "on"
|
||||||
# archive_timeout: 1800s
|
# archive_timeout: 1800s
|
||||||
# archive_command: mkdir -p ../wal_archive && test ! -f ../wal_archive/%f && cp %p ../wal_archive/%f
|
# archive_command: mkdir -p ../wal_archive && test ! -f ../wal_archive/%f && cp %p ../wal_archive/%f
|
||||||
@@ -86,7 +91,7 @@ bootstrap:
|
|||||||
# Some additional users users which needs to be created after initializing new cluster
|
# Some additional users users which needs to be created after initializing new cluster
|
||||||
users:
|
users:
|
||||||
admin:
|
admin:
|
||||||
password: admin
|
password: admin%
|
||||||
options:
|
options:
|
||||||
- createrole
|
- createrole
|
||||||
- createdb
|
- createdb
|
||||||
|
|||||||
+6
-1
@@ -51,10 +51,15 @@ bootstrap:
|
|||||||
parameters:
|
parameters:
|
||||||
# wal_level: hot_standby
|
# wal_level: hot_standby
|
||||||
# hot_standby: "on"
|
# hot_standby: "on"
|
||||||
|
# max_connections: 100
|
||||||
|
# max_worker_processes: 8
|
||||||
# wal_keep_segments: 8
|
# wal_keep_segments: 8
|
||||||
# max_wal_senders: 10
|
# max_wal_senders: 10
|
||||||
# max_replication_slots: 10
|
# max_replication_slots: 10
|
||||||
|
# max_prepared_transactions: 0
|
||||||
|
# max_locks_per_transaction: 64
|
||||||
# wal_log_hints: "on"
|
# wal_log_hints: "on"
|
||||||
|
# track_commit_timestamp: "off"
|
||||||
# archive_mode: "on"
|
# archive_mode: "on"
|
||||||
# archive_timeout: 1800s
|
# archive_timeout: 1800s
|
||||||
# archive_command: mkdir -p ../wal_archive && test ! -f ../wal_archive/%f && cp %p ../wal_archive/%f
|
# archive_command: mkdir -p ../wal_archive && test ! -f ../wal_archive/%f && cp %p ../wal_archive/%f
|
||||||
@@ -80,7 +85,7 @@ bootstrap:
|
|||||||
# Some additional users users which needs to be created after initializing new cluster
|
# Some additional users users which needs to be created after initializing new cluster
|
||||||
users:
|
users:
|
||||||
admin:
|
admin:
|
||||||
password: admin
|
password: admin%
|
||||||
options:
|
options:
|
||||||
- createrole
|
- createrole
|
||||||
- createdb
|
- createdb
|
||||||
|
|||||||
+6
-1
@@ -51,10 +51,15 @@ bootstrap:
|
|||||||
parameters:
|
parameters:
|
||||||
# wal_level: hot_standby
|
# wal_level: hot_standby
|
||||||
# hot_standby: "on"
|
# hot_standby: "on"
|
||||||
|
# max_connections: 100
|
||||||
|
# max_worker_processes: 8
|
||||||
# wal_keep_segments: 8
|
# wal_keep_segments: 8
|
||||||
# max_wal_senders: 10
|
# max_wal_senders: 10
|
||||||
# max_replication_slots: 10
|
# max_replication_slots: 10
|
||||||
|
# max_prepared_transactions: 0
|
||||||
|
# max_locks_per_transaction: 64
|
||||||
# wal_log_hints: "on"
|
# wal_log_hints: "on"
|
||||||
|
# track_commit_timestamp: "off"
|
||||||
# archive_mode: "on"
|
# archive_mode: "on"
|
||||||
# archive_timeout: 1800s
|
# archive_timeout: 1800s
|
||||||
# archive_command: mkdir -p ../wal_archive && test ! -f ../wal_archive/%f && cp %p ../wal_archive/%f
|
# archive_command: mkdir -p ../wal_archive && test ! -f ../wal_archive/%f && cp %p ../wal_archive/%f
|
||||||
@@ -77,7 +82,7 @@ bootstrap:
|
|||||||
# Some additional users users which needs to be created after initializing new cluster
|
# Some additional users users which needs to be created after initializing new cluster
|
||||||
users:
|
users:
|
||||||
admin:
|
admin:
|
||||||
password: admin
|
password: admin%
|
||||||
options:
|
options:
|
||||||
- createrole
|
- createrole
|
||||||
- createdb
|
- createdb
|
||||||
|
|||||||
+2
-1
@@ -9,6 +9,7 @@ python-consul>=0.7.1
|
|||||||
click>=4.1
|
click>=4.1
|
||||||
prettytable>=0.7
|
prettytable>=0.7
|
||||||
python-dateutil
|
python-dateutil
|
||||||
pysyncobj>=0.3.7
|
pysyncobj>=0.3.8
|
||||||
|
cryptography>=1.4
|
||||||
psutil>=2.0.0
|
psutil>=2.0.0
|
||||||
ydiff>=1.2.0
|
ydiff>=1.2.0
|
||||||
|
|||||||
@@ -5,6 +5,7 @@
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import inspect
|
import inspect
|
||||||
|
import logging
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
@@ -22,10 +23,10 @@ AUTHOR_EMAIL = '[email protected], [email protected], alexk
|
|||||||
KEYWORDS = 'etcd governor patroni postgresql postgres ha haproxy confd' +\
|
KEYWORDS = 'etcd governor patroni postgresql postgres ha haproxy confd' +\
|
||||||
' zookeeper exhibitor consul streaming replication kubernetes k8s'
|
' zookeeper exhibitor consul streaming replication kubernetes k8s'
|
||||||
|
|
||||||
EXTRAS_REQUIRE = {'aws': ['boto'], 'etcd': ['python-etcd'], 'etcd3': ['python-etcd'], 'consul': ['python-consul'],
|
EXTRAS_REQUIRE = {'aws': ['boto'], 'etcd': ['python-etcd'], 'etcd3': ['python-etcd'],
|
||||||
'exhibitor': ['kazoo'], 'zookeeper': ['kazoo'], 'kubernetes': ['ipaddress'], 'raft': ['pysyncobj']}
|
'consul': ['python-consul'], 'exhibitor': ['kazoo'], 'zookeeper': ['kazoo'],
|
||||||
|
'kubernetes': [], 'raft': ['pysyncobj', 'cryptography']}
|
||||||
COVERAGE_XML = True
|
COVERAGE_XML = True
|
||||||
COVERAGE_HTML = False
|
|
||||||
|
|
||||||
# Add here all kinds of additional classifiers as defined under
|
# Add here all kinds of additional classifiers as defined under
|
||||||
# https://pypi.python.org/pypi?%3Aaction=list_classifiers
|
# https://pypi.python.org/pypi?%3Aaction=list_classifiers
|
||||||
@@ -48,29 +49,29 @@ CLASSIFIERS = [
|
|||||||
'Programming Language :: Python :: 3.7',
|
'Programming Language :: Python :: 3.7',
|
||||||
'Programming Language :: Python :: 3.8',
|
'Programming Language :: Python :: 3.8',
|
||||||
'Programming Language :: Python :: 3.9',
|
'Programming Language :: Python :: 3.9',
|
||||||
|
'Programming Language :: Python :: 3.10',
|
||||||
'Programming Language :: Python :: Implementation :: CPython',
|
'Programming Language :: Python :: Implementation :: CPython',
|
||||||
]
|
]
|
||||||
|
|
||||||
CONSOLE_SCRIPTS = ['patroni = patroni:main',
|
CONSOLE_SCRIPTS = ['patroni = patroni.__main__:main',
|
||||||
'patronictl = patroni.ctl:ctl',
|
'patronictl = patroni.ctl:ctl',
|
||||||
'patroni_raft_controller = patroni.raft_controller:main',
|
'patroni_raft_controller = patroni.raft_controller:main',
|
||||||
"patroni_wale_restore = patroni.scripts.wale_restore:main",
|
"patroni_wale_restore = patroni.scripts.wale_restore:main",
|
||||||
"patroni_aws = patroni.scripts.aws:main"]
|
"patroni_aws = patroni.scripts.aws:main"]
|
||||||
|
|
||||||
|
|
||||||
class Flake8(Command):
|
class _Command(Command):
|
||||||
|
|
||||||
user_options = []
|
user_options = []
|
||||||
|
|
||||||
def initialize_options(self):
|
def initialize_options(self):
|
||||||
from flake8.main import application
|
pass
|
||||||
|
|
||||||
self.flake8 = application.Application()
|
|
||||||
self.flake8.initialize([])
|
|
||||||
|
|
||||||
def finalize_options(self):
|
def finalize_options(self):
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
class Flake8(_Command):
|
||||||
|
|
||||||
def package_files(self):
|
def package_files(self):
|
||||||
seen_package_directories = ()
|
seen_package_directories = ()
|
||||||
directories = self.distribution.package_dir or {}
|
directories = self.distribution.package_dir or {}
|
||||||
@@ -92,68 +93,31 @@ class Flake8(Command):
|
|||||||
return [package for package in self.package_files()] + ['tests', 'setup.py']
|
return [package for package in self.package_files()] + ['tests', 'setup.py']
|
||||||
|
|
||||||
def run(self):
|
def run(self):
|
||||||
self.flake8.run_checks(self.targets())
|
from flake8.main import application
|
||||||
self.flake8.formatter.start()
|
|
||||||
self.flake8.report_errors()
|
logging.getLogger().setLevel(logging.ERROR)
|
||||||
self.flake8.report_statistics()
|
flake8 = application.Application()
|
||||||
self.flake8.report_benchmarks()
|
flake8.run(self.targets())
|
||||||
self.flake8.formatter.stop()
|
flake8.exit()
|
||||||
try:
|
|
||||||
self.flake8.exit()
|
|
||||||
except SystemExit as e:
|
|
||||||
# Cause system exit only if exit code is not zero (terminates
|
|
||||||
# other possibly remaining/pending setuptools commands).
|
|
||||||
if e.code:
|
|
||||||
raise
|
|
||||||
|
|
||||||
|
|
||||||
class PyTest(Command):
|
class PyTest(_Command):
|
||||||
|
|
||||||
user_options = [('cov=', None, 'Run coverage'), ('cov-xml=', None, 'Generate junit xml report'),
|
def run(self):
|
||||||
('cov-html=', None, 'Generate junit html report')]
|
|
||||||
|
|
||||||
def initialize_options(self):
|
|
||||||
self.cov = []
|
|
||||||
self.cov_xml = False
|
|
||||||
self.cov_html = False
|
|
||||||
|
|
||||||
def finalize_options(self):
|
|
||||||
if self.cov_xml or self.cov_html:
|
|
||||||
self.cov = ['--cov', MAIN_PACKAGE, '--cov-report', 'term-missing']
|
|
||||||
if self.cov_xml:
|
|
||||||
self.cov.extend(['--cov-report', 'xml'])
|
|
||||||
if self.cov_html:
|
|
||||||
self.cov.extend(['--cov-report', 'html'])
|
|
||||||
|
|
||||||
def run_tests(self):
|
|
||||||
try:
|
try:
|
||||||
import pytest
|
import pytest
|
||||||
except Exception:
|
except Exception:
|
||||||
raise RuntimeError('py.test is not installed, run: pip install pytest')
|
raise RuntimeError('py.test is not installed, run: pip install pytest')
|
||||||
|
|
||||||
import logging
|
logging.getLogger().setLevel(logging.WARNING)
|
||||||
silence = logging.WARNING
|
|
||||||
logging.basicConfig(format='%(asctime)s %(levelname)s: %(message)s', level=os.getenv('LOGLEVEL', silence))
|
|
||||||
|
|
||||||
args = ['--verbose', 'tests', '--doctest-modules', MAIN_PACKAGE] +\
|
args = ['--verbose', 'tests', '--doctest-modules', MAIN_PACKAGE] +\
|
||||||
['-s' if logging.getLogger().getEffectiveLevel() < silence else '--capture=fd']
|
['-s' if logging.getLogger().getEffectiveLevel() < logging.WARNING else '--capture=fd'] +\
|
||||||
if self.cov:
|
['--cov', MAIN_PACKAGE, '--cov-report', 'term-missing', '--cov-report', 'xml']
|
||||||
args += self.cov
|
|
||||||
|
|
||||||
errno = pytest.main(args=args)
|
errno = pytest.main(args=args)
|
||||||
sys.exit(errno)
|
sys.exit(errno)
|
||||||
|
|
||||||
def run(self):
|
|
||||||
from pkg_resources import evaluate_marker
|
|
||||||
|
|
||||||
requirements = set(self.distribution.install_requires + ['mock>=2.0.0', 'pytest-cov', 'pytest'])
|
|
||||||
for k, v in self.distribution.extras_require.items():
|
|
||||||
if not k.startswith(':') or evaluate_marker(k[1:]):
|
|
||||||
requirements.update(v)
|
|
||||||
|
|
||||||
self.distribution.fetch_build_eggs(list(requirements))
|
|
||||||
self.run_tests()
|
|
||||||
|
|
||||||
|
|
||||||
def read(fname):
|
def read(fname):
|
||||||
with open(os.path.join(__location__, fname)) as fd:
|
with open(os.path.join(__location__, fname)) as fd:
|
||||||
@@ -161,6 +125,8 @@ def read(fname):
|
|||||||
|
|
||||||
|
|
||||||
def setup_package(version):
|
def setup_package(version):
|
||||||
|
logging.basicConfig(format='%(message)s', level=os.getenv('LOGLEVEL', logging.WARNING))
|
||||||
|
|
||||||
# Assemble additional setup commands
|
# Assemble additional setup commands
|
||||||
cmdclass = {'test': PyTest, 'flake8': Flake8}
|
cmdclass = {'test': PyTest, 'flake8': Flake8}
|
||||||
|
|
||||||
@@ -171,19 +137,18 @@ def setup_package(version):
|
|||||||
if r == '':
|
if r == '':
|
||||||
continue
|
continue
|
||||||
extra = False
|
extra = False
|
||||||
for e, v in EXTRAS_REQUIRE.items():
|
for e, deps in EXTRAS_REQUIRE.items():
|
||||||
if v and r.startswith(v[0]):
|
for i, v in enumerate(deps):
|
||||||
EXTRAS_REQUIRE[e] = [r] if e != 'kubernetes' or sys.version_info < (3, 0, 0) else []
|
if r.startswith(v):
|
||||||
extra = True
|
deps[i] = r
|
||||||
|
EXTRAS_REQUIRE[e] = deps
|
||||||
|
extra = True
|
||||||
|
break
|
||||||
|
if extra:
|
||||||
|
break
|
||||||
if not extra:
|
if not extra:
|
||||||
install_requires.append(r)
|
install_requires.append(r)
|
||||||
|
|
||||||
command_options = {'test': {}}
|
|
||||||
if COVERAGE_XML:
|
|
||||||
command_options['test']['cov_xml'] = 'setup.py', True
|
|
||||||
if COVERAGE_HTML:
|
|
||||||
command_options['test']['cov_html'] = 'setup.py', True
|
|
||||||
|
|
||||||
setup(
|
setup(
|
||||||
name=NAME,
|
name=NAME,
|
||||||
version=version,
|
version=version,
|
||||||
@@ -200,9 +165,7 @@ def setup_package(version):
|
|||||||
python_requires='>=2.7',
|
python_requires='>=2.7',
|
||||||
install_requires=install_requires,
|
install_requires=install_requires,
|
||||||
extras_require=EXTRAS_REQUIRE,
|
extras_require=EXTRAS_REQUIRE,
|
||||||
setup_requires='flake8',
|
|
||||||
cmdclass=cmdclass,
|
cmdclass=cmdclass,
|
||||||
command_options=command_options,
|
|
||||||
entry_points={'console_scripts': CONSOLE_SCRIPTS},
|
entry_points={'console_scripts': CONSOLE_SCRIPTS},
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -210,13 +173,14 @@ def setup_package(version):
|
|||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
old_modules = sys.modules.copy()
|
old_modules = sys.modules.copy()
|
||||||
try:
|
try:
|
||||||
from patroni import check_psycopg2, fatal, __version__
|
from patroni import check_psycopg, fatal
|
||||||
|
from patroni.version import __version__
|
||||||
finally:
|
finally:
|
||||||
sys.modules.clear()
|
sys.modules.clear()
|
||||||
sys.modules.update(old_modules)
|
sys.modules.update(old_modules)
|
||||||
|
|
||||||
if sys.version_info < (2, 7, 0):
|
if sys.version_info < (2, 7, 0):
|
||||||
fatal('Patroni needs to be run with Python 2.7+')
|
fatal('Patroni needs to be run with Python 2.7+')
|
||||||
check_psycopg2()
|
check_psycopg()
|
||||||
|
|
||||||
setup_package(__version__)
|
setup_package(__version__)
|
||||||
|
|||||||
+19
-11
@@ -1,16 +1,18 @@
|
|||||||
|
import datetime
|
||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from mock import Mock, patch
|
from mock import Mock, patch
|
||||||
|
|
||||||
import psycopg2
|
|
||||||
import urllib3
|
import urllib3
|
||||||
|
|
||||||
|
import patroni.psycopg as psycopg
|
||||||
|
|
||||||
from patroni.dcs import Leader, Member
|
from patroni.dcs import Leader, Member
|
||||||
from patroni.postgresql import Postgresql
|
from patroni.postgresql import Postgresql
|
||||||
from patroni.postgresql.config import ConfigHandler
|
from patroni.postgresql.config import ConfigHandler
|
||||||
from patroni.utils import RetryFailedError
|
from patroni.utils import RetryFailedError, tzutc
|
||||||
|
|
||||||
|
|
||||||
class SleepException(Exception):
|
class SleepException(Exception):
|
||||||
@@ -84,21 +86,26 @@ class MockCursor(object):
|
|||||||
|
|
||||||
def execute(self, sql, *params):
|
def execute(self, sql, *params):
|
||||||
if sql.startswith('blabla'):
|
if sql.startswith('blabla'):
|
||||||
raise psycopg2.ProgrammingError()
|
raise psycopg.ProgrammingError()
|
||||||
elif sql == 'CHECKPOINT' or sql.startswith('SELECT pg_catalog.pg_create_'):
|
elif sql == 'CHECKPOINT' or sql.startswith('SELECT pg_catalog.pg_create_'):
|
||||||
raise psycopg2.OperationalError()
|
raise psycopg.OperationalError()
|
||||||
elif sql.startswith('RetryFailedError'):
|
elif sql.startswith('RetryFailedError'):
|
||||||
raise RetryFailedError('retry')
|
raise RetryFailedError('retry')
|
||||||
|
elif sql.startswith('SELECT catalog_xmin'):
|
||||||
|
self.results = [(100, 501)]
|
||||||
|
elif sql.startswith('SELECT slot_name, catalog_xmin'):
|
||||||
|
self.results = [('ls', 100, 500, b'123456')]
|
||||||
elif sql.startswith('SELECT slot_name'):
|
elif sql.startswith('SELECT slot_name'):
|
||||||
self.results = [('blabla', 'physical'), ('foobar', 'physical'), ('ls', 'logical', 'a', 'b')]
|
self.results = [('blabla', 'physical'), ('foobar', 'physical'), ('ls', 'logical', 'a', 'b', 5, 100, 500)]
|
||||||
elif sql.startswith('SELECT CASE WHEN pg_catalog.pg_is_in_recovery()'):
|
elif sql.startswith('SELECT CASE WHEN pg_catalog.pg_is_in_recovery()'):
|
||||||
self.results = [(1, 2, 1, 0, False, 1, 1, None, None)]
|
self.results = [(1, 2, 1, 0, False, 1, 1, None, None, [{"slot_name": "ls", "confirmed_flush_lsn": 12345}])]
|
||||||
elif sql.startswith('SELECT pg_catalog.pg_is_in_recovery()'):
|
elif sql.startswith('SELECT pg_catalog.pg_is_in_recovery()'):
|
||||||
self.results = [(False, 2)]
|
self.results = [(False, 2)]
|
||||||
elif sql.startswith('SELECT pg_catalog.to_char'):
|
elif sql.startswith('SELECT pg_catalog.pg_postmaster_start_time'):
|
||||||
replication_info = '[{"application_name":"walreceiver","client_addr":"1.2.3.4",' +\
|
replication_info = '[{"application_name":"walreceiver","client_addr":"1.2.3.4",' +\
|
||||||
'"state":"streaming","sync_state":"async","sync_priority":0}]'
|
'"state":"streaming","sync_state":"async","sync_priority":0}]'
|
||||||
self.results = [('', 0, '', 0, '', '', False, replication_info)]
|
now = datetime.datetime.now(tzutc)
|
||||||
|
self.results = [(now, 0, '', 0, '', False, now, replication_info)]
|
||||||
elif sql.startswith('SELECT name, setting'):
|
elif sql.startswith('SELECT name, setting'):
|
||||||
self.results = [('wal_segment_size', '2048', '8kB', 'integer', 'internal'),
|
self.results = [('wal_segment_size', '2048', '8kB', 'integer', 'internal'),
|
||||||
('wal_block_size', '8192', None, 'integer', 'internal'),
|
('wal_block_size', '8192', None, 'integer', 'internal'),
|
||||||
@@ -156,7 +163,7 @@ class MockConnect(object):
|
|||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
||||||
def psycopg2_connect(*args, **kwargs):
|
def psycopg_connect(*args, **kwargs):
|
||||||
return MockConnect()
|
return MockConnect()
|
||||||
|
|
||||||
|
|
||||||
@@ -170,7 +177,7 @@ class PostgresInit(unittest.TestCase):
|
|||||||
'force_parallel_mode': '1', 'constraint_exclusion': '',
|
'force_parallel_mode': '1', 'constraint_exclusion': '',
|
||||||
'max_stack_depth': 'Z', 'vacuum_cost_limit': -1, 'vacuum_cost_delay': 200}
|
'max_stack_depth': 'Z', 'vacuum_cost_limit': -1, 'vacuum_cost_delay': 200}
|
||||||
|
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
@patch('patroni.postgresql.CallbackExecutor', Mock())
|
@patch('patroni.postgresql.CallbackExecutor', Mock())
|
||||||
@patch.object(ConfigHandler, 'write_postgresql_conf', Mock())
|
@patch.object(ConfigHandler, 'write_postgresql_conf', Mock())
|
||||||
@patch.object(ConfigHandler, 'replace_pg_hba', Mock())
|
@patch.object(ConfigHandler, 'replace_pg_hba', Mock())
|
||||||
@@ -183,7 +190,8 @@ class PostgresInit(unittest.TestCase):
|
|||||||
'krbsrvname': 'postgres', 'pgpass': os.path.join(data_dir, 'pgpass0'),
|
'krbsrvname': 'postgres', 'pgpass': os.path.join(data_dir, 'pgpass0'),
|
||||||
'listen': '127.0.0.2, 127.0.0.3:5432', 'connect_address': '127.0.0.2:5432',
|
'listen': '127.0.0.2, 127.0.0.3:5432', 'connect_address': '127.0.0.2:5432',
|
||||||
'authentication': {'superuser': {'username': 'foo', 'password': 'test'},
|
'authentication': {'superuser': {'username': 'foo', 'password': 'test'},
|
||||||
'replication': {'username': '', 'password': 'rep-pass'}},
|
'replication': {'username': '', 'password': 'rep-pass'},
|
||||||
|
'rewind': {'username': 'rewind', 'password': 'test'}},
|
||||||
'remove_data_directory_on_rewind_failure': True,
|
'remove_data_directory_on_rewind_failure': True,
|
||||||
'use_pg_rewind': True, 'pg_ctl_timeout': 'bla',
|
'use_pg_rewind': True, 'pg_ctl_timeout': 'bla',
|
||||||
'parameters': self._PARAMETERS,
|
'parameters': self._PARAMETERS,
|
||||||
|
|||||||
+117
-15
@@ -1,9 +1,10 @@
|
|||||||
import datetime
|
import datetime
|
||||||
import json
|
import json
|
||||||
import psycopg2
|
|
||||||
import unittest
|
import unittest
|
||||||
import socket
|
import socket
|
||||||
|
|
||||||
|
import patroni.psycopg as psycopg
|
||||||
|
|
||||||
from mock import Mock, PropertyMock, patch
|
from mock import Mock, PropertyMock, patch
|
||||||
from patroni.api import RestApiHandler, RestApiServer
|
from patroni.api import RestApiHandler, RestApiServer
|
||||||
from patroni.dcs import ClusterConfig, Member
|
from patroni.dcs import ClusterConfig, Member
|
||||||
@@ -11,7 +12,7 @@ from patroni.ha import _MemberStatus
|
|||||||
from patroni.utils import tzutc
|
from patroni.utils import tzutc
|
||||||
from six import BytesIO as IO
|
from six import BytesIO as IO
|
||||||
from six.moves import BaseHTTPServer
|
from six.moves import BaseHTTPServer
|
||||||
from . import psycopg2_connect, MockCursor
|
from . import psycopg_connect, MockCursor
|
||||||
from .test_ha import get_cluster_initialized_without_leader
|
from .test_ha import get_cluster_initialized_without_leader
|
||||||
|
|
||||||
|
|
||||||
@@ -30,16 +31,16 @@ class MockPostgresql(object):
|
|||||||
pending_restart = True
|
pending_restart = True
|
||||||
wal_name = 'wal'
|
wal_name = 'wal'
|
||||||
lsn_name = 'lsn'
|
lsn_name = 'lsn'
|
||||||
POSTMASTER_START_TIME = 'pg_catalog.to_char(pg_catalog.pg_postmaster_start_time'
|
POSTMASTER_START_TIME = 'pg_catalog.pg_postmaster_start_time()'
|
||||||
TL_LSN = 'CASE WHEN pg_catalog.pg_is_in_recovery()'
|
TL_LSN = 'CASE WHEN pg_catalog.pg_is_in_recovery()'
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def connection():
|
def connection():
|
||||||
return psycopg2_connect()
|
return psycopg_connect()
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def postmaster_start_time():
|
def postmaster_start_time():
|
||||||
return str(postmaster_start_time)
|
return postmaster_start_time
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def replica_cached_timeline(_):
|
def replica_cached_timeline(_):
|
||||||
@@ -77,7 +78,7 @@ class MockHa(object):
|
|||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def fetch_nodes_statuses(members):
|
def fetch_nodes_statuses(members):
|
||||||
return [_MemberStatus(None, True, None, 0, None, {}, False)]
|
return [_MemberStatus(None, True, None, 0, 0, None, {}, False)]
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def schedule_future_restart(data):
|
def schedule_future_restart(data):
|
||||||
@@ -118,7 +119,7 @@ class MockPatroni(object):
|
|||||||
postgresql = ha.state_handler
|
postgresql = ha.state_handler
|
||||||
dcs = Mock()
|
dcs = Mock()
|
||||||
logger = MockLogger()
|
logger = MockLogger()
|
||||||
tags = {}
|
tags = {"key1": True, "key2": False, "key3": 1, "key4": 1.4, "key5": "RandomTag"}
|
||||||
version = '0.00'
|
version = '0.00'
|
||||||
noloadbalance = PropertyMock(return_value=False)
|
noloadbalance = PropertyMock(return_value=False)
|
||||||
scheduled_restart = {'schedule': future_restart_time,
|
scheduled_restart = {'schedule': future_restart_time,
|
||||||
@@ -162,7 +163,7 @@ class TestRestApiHandler(unittest.TestCase):
|
|||||||
_authorization = '\nAuthorization: Basic dGVzdDp0ZXN0'
|
_authorization = '\nAuthorization: Basic dGVzdDp0ZXN0'
|
||||||
|
|
||||||
def test_do_GET(self):
|
def test_do_GET(self):
|
||||||
MockPatroni.dcs.cluster.last_leader_operation = 20
|
MockPatroni.dcs.cluster.last_lsn = 20
|
||||||
MockRestApiServer(RestApiHandler, 'GET /replica')
|
MockRestApiServer(RestApiHandler, 'GET /replica')
|
||||||
MockRestApiServer(RestApiHandler, 'GET /replica?lag=1M')
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=1M')
|
||||||
MockRestApiServer(RestApiHandler, 'GET /replica?lag=10MB')
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=10MB')
|
||||||
@@ -174,7 +175,7 @@ class TestRestApiHandler(unittest.TestCase):
|
|||||||
MockRestApiServer(RestApiHandler, 'GET /replica')
|
MockRestApiServer(RestApiHandler, 'GET /replica')
|
||||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'state': 'running'})):
|
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'state': 'running'})):
|
||||||
MockRestApiServer(RestApiHandler, 'GET /health')
|
MockRestApiServer(RestApiHandler, 'GET /health')
|
||||||
MockRestApiServer(RestApiHandler, 'GET /master')
|
MockRestApiServer(RestApiHandler, 'GET /leader')
|
||||||
MockPatroni.dcs.cluster.sync.members = [MockPostgresql.name]
|
MockPatroni.dcs.cluster.sync.members = [MockPostgresql.name]
|
||||||
MockPatroni.dcs.cluster.is_synchronous_mode = Mock(return_value=True)
|
MockPatroni.dcs.cluster.is_synchronous_mode = Mock(return_value=True)
|
||||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
|
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
|
||||||
@@ -197,6 +198,89 @@ class TestRestApiHandler(unittest.TestCase):
|
|||||||
with patch.object(MockHa, 'is_standby_cluster', Mock(return_value=True)):
|
with patch.object(MockHa, 'is_standby_cluster', Mock(return_value=True)):
|
||||||
MockRestApiServer(RestApiHandler, 'GET /standby_leader')
|
MockRestApiServer(RestApiHandler, 'GET /standby_leader')
|
||||||
|
|
||||||
|
# test tags
|
||||||
|
#
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /master?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /master?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=False&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /master?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1.0&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /master?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag&tag_key6=RandomTag2')
|
||||||
|
#
|
||||||
|
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'master'})):
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /master?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /master?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=False&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /master?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1.0&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /master?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag&tag_key6=RandomTag2')
|
||||||
|
#
|
||||||
|
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'standby_leader'})):
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /standby_leader?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /standby_leader?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=False&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /standby_leader?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1.0&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /standby_leader?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag&tag_key6=RandomTag2')
|
||||||
|
#
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=False&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1.0&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag&tag_key6=RandomTag2')
|
||||||
|
#
|
||||||
|
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'master'})):
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=False&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1.0&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /replica?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag&tag_key6=RandomTag2')
|
||||||
|
#
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /read-write?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /read-write?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=False&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /read-write?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1.0&tag_key4=1.4&tag_key5=RandomTag')
|
||||||
|
MockRestApiServer(RestApiHandler, 'GET /read-write?lag=1M&'
|
||||||
|
'tag_key1=true&tag_key2=false&'
|
||||||
|
'tag_key3=1&tag_key4=1.4&tag_key5=RandomTag&tag_key6=RandomTag2')
|
||||||
|
|
||||||
def test_do_OPTIONS(self):
|
def test_do_OPTIONS(self):
|
||||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'OPTIONS / HTTP/1.0'))
|
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'OPTIONS / HTTP/1.0'))
|
||||||
|
|
||||||
@@ -236,6 +320,10 @@ class TestRestApiHandler(unittest.TestCase):
|
|||||||
mock_dcs.cluster.config = None
|
mock_dcs.cluster.config = None
|
||||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /config'))
|
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /config'))
|
||||||
|
|
||||||
|
@patch.object(MockPatroni, 'dcs')
|
||||||
|
def test_do_GET_metrics(self, mock_dcs):
|
||||||
|
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /metrics'))
|
||||||
|
|
||||||
@patch.object(MockPatroni, 'dcs')
|
@patch.object(MockPatroni, 'dcs')
|
||||||
def test_do_PATCH_config(self, mock_dcs):
|
def test_do_PATCH_config(self, mock_dcs):
|
||||||
config = {'postgresql': {'use_slots': False, 'use_pg_rewind': True, 'parameters': {'wal_level': 'logical'}}}
|
config = {'postgresql': {'use_slots': False, 'use_pg_rewind': True, 'parameters': {'wal_level': 'logical'}}}
|
||||||
@@ -348,9 +436,9 @@ class TestRestApiHandler(unittest.TestCase):
|
|||||||
|
|
||||||
@patch('time.sleep', Mock())
|
@patch('time.sleep', Mock())
|
||||||
def test_RestApiServer_query(self):
|
def test_RestApiServer_query(self):
|
||||||
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg2.OperationalError)):
|
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)):
|
||||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /patroni'))
|
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /patroni'))
|
||||||
with patch.object(MockPostgresql, 'connection', Mock(side_effect=psycopg2.OperationalError)):
|
with patch.object(MockPostgresql, 'connection', Mock(side_effect=psycopg.OperationalError)):
|
||||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /patroni'))
|
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /patroni'))
|
||||||
|
|
||||||
@patch('time.sleep', Mock())
|
@patch('time.sleep', Mock())
|
||||||
@@ -451,7 +539,9 @@ class TestRestApiServer(unittest.TestCase):
|
|||||||
@patch.object(BaseHTTPServer.HTTPServer, '__init__', Mock())
|
@patch.object(BaseHTTPServer.HTTPServer, '__init__', Mock())
|
||||||
def setUp(self):
|
def setUp(self):
|
||||||
self.srv = MockRestApiServer(Mock(), '', {'listen': '*:8008', 'certfile': 'a', 'verify_client': 'required',
|
self.srv = MockRestApiServer(Mock(), '', {'listen': '*:8008', 'certfile': 'a', 'verify_client': 'required',
|
||||||
'ciphers': '!SSLv1:!SSLv2:!SSLv3:!TLSv1:!TLSv1.1'})
|
'ciphers': '!SSLv1:!SSLv2:!SSLv3:!TLSv1:!TLSv1.1',
|
||||||
|
'allowlist': ['127.0.0.1', '::1/128', '::1/zxc'],
|
||||||
|
'allowlist_include_members': True})
|
||||||
|
|
||||||
@patch.object(BaseHTTPServer.HTTPServer, '__init__', Mock())
|
@patch.object(BaseHTTPServer.HTTPServer, '__init__', Mock())
|
||||||
def test_reload_config(self):
|
def test_reload_config(self):
|
||||||
@@ -459,13 +549,21 @@ class TestRestApiServer(unittest.TestCase):
|
|||||||
self.assertRaises(ValueError, MockRestApiServer, None, '', bad_config)
|
self.assertRaises(ValueError, MockRestApiServer, None, '', bad_config)
|
||||||
self.assertRaises(ValueError, self.srv.reload_config, bad_config)
|
self.assertRaises(ValueError, self.srv.reload_config, bad_config)
|
||||||
self.assertRaises(ValueError, self.srv.reload_config, {})
|
self.assertRaises(ValueError, self.srv.reload_config, {})
|
||||||
with patch.object(socket.socket, 'setsockopt', Mock(side_effect=socket.error)):
|
with patch.object(socket.socket, 'setsockopt', Mock(side_effect=socket.error)), \
|
||||||
|
patch.object(MockRestApiServer, 'server_close', Mock()):
|
||||||
self.srv.reload_config({'listen': ':8008'})
|
self.srv.reload_config({'listen': ':8008'})
|
||||||
|
|
||||||
def test_check_auth(self):
|
@patch.object(MockPatroni, 'dcs')
|
||||||
|
def test_check_access(self, mock_dcs):
|
||||||
|
mock_dcs.cluster = get_cluster_initialized_without_leader()
|
||||||
|
mock_dcs.cluster.members[1].data['api_url'] = 'http://127.0.0.1z:8011/patroni'
|
||||||
|
mock_dcs.cluster.members.append(Member(0, 'bad-api-url', 30, {'api_url': 123}))
|
||||||
mock_rh = Mock()
|
mock_rh = Mock()
|
||||||
|
mock_rh.client_address = ('127.0.0.2',)
|
||||||
|
self.assertIsNot(self.srv.check_access(mock_rh), True)
|
||||||
|
mock_rh.client_address = ('127.0.0.1',)
|
||||||
mock_rh.request.getpeercert.return_value = None
|
mock_rh.request.getpeercert.return_value = None
|
||||||
self.assertIsNot(self.srv.check_auth(mock_rh), True)
|
self.assertIsNot(self.srv.check_access(mock_rh), True)
|
||||||
|
|
||||||
def test_handle_error(self):
|
def test_handle_error(self):
|
||||||
try:
|
try:
|
||||||
@@ -503,3 +601,7 @@ class TestRestApiServer(unittest.TestCase):
|
|||||||
Mock(return_value=(mock_request, mock_address))
|
Mock(return_value=(mock_request, mock_address))
|
||||||
):
|
):
|
||||||
self.srv._handle_request_noblock()
|
self.srv._handle_request_noblock()
|
||||||
|
|
||||||
|
@patch('ssl._ssl._test_decode_cert', Mock())
|
||||||
|
def test_reload_local_certificate(self):
|
||||||
|
self.assertTrue(self.srv.reload_local_certificate())
|
||||||
|
|||||||
@@ -8,11 +8,11 @@ from patroni.postgresql.bootstrap import Bootstrap
|
|||||||
from patroni.postgresql.cancellable import CancellableSubprocess
|
from patroni.postgresql.cancellable import CancellableSubprocess
|
||||||
from patroni.postgresql.config import ConfigHandler
|
from patroni.postgresql.config import ConfigHandler
|
||||||
|
|
||||||
from . import psycopg2_connect, BaseTestPostgresql
|
from . import psycopg_connect, BaseTestPostgresql
|
||||||
|
|
||||||
|
|
||||||
@patch('subprocess.call', Mock(return_value=0))
|
@patch('subprocess.call', Mock(return_value=0))
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
@patch('os.rename', Mock())
|
@patch('os.rename', Mock())
|
||||||
class TestBootstrap(BaseTestPostgresql):
|
class TestBootstrap(BaseTestPostgresql):
|
||||||
|
|
||||||
@@ -164,6 +164,7 @@ class TestBootstrap(BaseTestPostgresql):
|
|||||||
@patch('os.unlink', Mock())
|
@patch('os.unlink', Mock())
|
||||||
@patch('shutil.copy', Mock())
|
@patch('shutil.copy', Mock())
|
||||||
@patch('os.path.isfile', Mock(return_value=True))
|
@patch('os.path.isfile', Mock(return_value=True))
|
||||||
|
@patch('patroni.postgresql.bootstrap.quote_ident', Mock())
|
||||||
@patch.object(Bootstrap, 'call_post_bootstrap', Mock(return_value=True))
|
@patch.object(Bootstrap, 'call_post_bootstrap', Mock(return_value=True))
|
||||||
@patch.object(Bootstrap, '_custom_bootstrap', Mock(return_value=True))
|
@patch.object(Bootstrap, '_custom_bootstrap', Mock(return_value=True))
|
||||||
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
||||||
|
|||||||
@@ -27,8 +27,8 @@ class TestCancellableSubprocess(unittest.TestCase):
|
|||||||
def test_cancel(self):
|
def test_cancel(self):
|
||||||
self.c._process = Mock()
|
self.c._process = Mock()
|
||||||
self.c._process.is_running.return_value = True
|
self.c._process.is_running.return_value = True
|
||||||
self.c._process.children.side_effect = psutil.Error()
|
self.c._process.children.side_effect = psutil.NoSuchProcess(123)
|
||||||
self.c._process.suspend.side_effect = psutil.Error()
|
self.c._process.suspend.side_effect = psutil.AccessDenied()
|
||||||
self.c.cancel()
|
self.c.cancel()
|
||||||
self.c._process.is_running.side_effect = [True, False]
|
self.c._process.is_running.side_effect = [True, False]
|
||||||
self.c.cancel()
|
self.c.cancel()
|
||||||
|
|||||||
@@ -30,12 +30,14 @@ class TestConfig(unittest.TestCase):
|
|||||||
'PATRONI_SCOPE': 'batman2',
|
'PATRONI_SCOPE': 'batman2',
|
||||||
'PATRONI_LOGLEVEL': 'ERROR',
|
'PATRONI_LOGLEVEL': 'ERROR',
|
||||||
'PATRONI_LOG_LOGGERS': 'patroni.postmaster: WARNING, urllib3: DEBUG',
|
'PATRONI_LOG_LOGGERS': 'patroni.postmaster: WARNING, urllib3: DEBUG',
|
||||||
|
'PATRONI_LOG_FILE_NUM': '5',
|
||||||
'PATRONI_RESTAPI_USERNAME': 'username',
|
'PATRONI_RESTAPI_USERNAME': 'username',
|
||||||
'PATRONI_RESTAPI_PASSWORD': 'password',
|
'PATRONI_RESTAPI_PASSWORD': 'password',
|
||||||
'PATRONI_RESTAPI_LISTEN': '0.0.0.0:8008',
|
'PATRONI_RESTAPI_LISTEN': '0.0.0.0:8008',
|
||||||
'PATRONI_RESTAPI_CONNECT_ADDRESS': '127.0.0.1:8008',
|
'PATRONI_RESTAPI_CONNECT_ADDRESS': '127.0.0.1:8008',
|
||||||
'PATRONI_RESTAPI_CERTFILE': '/certfile',
|
'PATRONI_RESTAPI_CERTFILE': '/certfile',
|
||||||
'PATRONI_RESTAPI_KEYFILE': '/keyfile',
|
'PATRONI_RESTAPI_KEYFILE': '/keyfile',
|
||||||
|
'PATRONI_RESTAPI_ALLOWLIST_INCLUDE_MEMBERS': 'on',
|
||||||
'PATRONI_POSTGRESQL_LISTEN': '0.0.0.0:5432',
|
'PATRONI_POSTGRESQL_LISTEN': '0.0.0.0:5432',
|
||||||
'PATRONI_POSTGRESQL_CONNECT_ADDRESS': '127.0.0.1:5432',
|
'PATRONI_POSTGRESQL_CONNECT_ADDRESS': '127.0.0.1:5432',
|
||||||
'PATRONI_POSTGRESQL_DATA_DIR': 'data/postgres0',
|
'PATRONI_POSTGRESQL_DATA_DIR': 'data/postgres0',
|
||||||
|
|||||||
+70
-8
@@ -2,7 +2,7 @@ import consul
|
|||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from consul import ConsulException, NotFound
|
from consul import ConsulException, NotFound
|
||||||
from mock import Mock, patch
|
from mock import Mock, PropertyMock, patch
|
||||||
from patroni.dcs.consul import AbstractDCS, Cluster, Consul, ConsulInternalError, \
|
from patroni.dcs.consul import AbstractDCS, Cluster, Consul, ConsulInternalError, \
|
||||||
ConsulError, ConsulClient, HTTPClient, InvalidSessionTTL, InvalidSession
|
ConsulError, ConsulClient, HTTPClient, InvalidSessionTTL, InvalidSession
|
||||||
from . import SleepException
|
from . import SleepException
|
||||||
@@ -15,8 +15,7 @@ def kv_get(self, key, **kwargs):
|
|||||||
return None, None
|
return None, None
|
||||||
if key == 'service/good/leader':
|
if key == 'service/good/leader':
|
||||||
return '1', None
|
return '1', None
|
||||||
if key == 'service/good/':
|
good_cls = ('6429',
|
||||||
return ('6429',
|
|
||||||
[{'CreateIndex': 1334, 'Flags': 0, 'Key': key + 'failover', 'LockIndex': 0,
|
[{'CreateIndex': 1334, 'Flags': 0, 'Key': key + 'failover', 'LockIndex': 0,
|
||||||
'ModifyIndex': 1334, 'Value': b''},
|
'ModifyIndex': 1334, 'Value': b''},
|
||||||
{'CreateIndex': 1334, 'Flags': 0, 'Key': key + 'initialize', 'LockIndex': 0,
|
{'CreateIndex': 1334, 'Flags': 0, 'Key': key + 'initialize', 'LockIndex': 0,
|
||||||
@@ -34,7 +33,17 @@ def kv_get(self, key, **kwargs):
|
|||||||
{'CreateIndex': 1085, 'Flags': 0, 'Key': key + 'optime/leader', 'LockIndex': 0,
|
{'CreateIndex': 1085, 'Flags': 0, 'Key': key + 'optime/leader', 'LockIndex': 0,
|
||||||
'ModifyIndex': 6429, 'Value': b'4496294792'},
|
'ModifyIndex': 6429, 'Value': b'4496294792'},
|
||||||
{'CreateIndex': 1085, 'Flags': 0, 'Key': key + 'sync', 'LockIndex': 0,
|
{'CreateIndex': 1085, 'Flags': 0, 'Key': key + 'sync', 'LockIndex': 0,
|
||||||
'ModifyIndex': 6429, 'Value': b'{"leader": "leader", "sync_standby": null}'}])
|
'ModifyIndex': 6429, 'Value': b'{"leader": "leader", "sync_standby": null}'},
|
||||||
|
{'CreateIndex': 1085, 'Flags': 0, 'Key': key + 'status', 'LockIndex': 0,
|
||||||
|
'ModifyIndex': 6429, 'Value': b'{"optime":4496294792, "slots":{"ls":12345}}'}])
|
||||||
|
if key == 'service/good/':
|
||||||
|
return good_cls
|
||||||
|
if key == 'service/broken/':
|
||||||
|
good_cls[1][-1]['Value'] = b'{'
|
||||||
|
return good_cls
|
||||||
|
if key == 'service/legacy/':
|
||||||
|
good_cls[1].pop()
|
||||||
|
return good_cls
|
||||||
raise ConsulException
|
raise ConsulException
|
||||||
|
|
||||||
|
|
||||||
@@ -109,6 +118,10 @@ class TestConsul(unittest.TestCase):
|
|||||||
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
||||||
self.c._base_path = '/service/fail'
|
self.c._base_path = '/service/fail'
|
||||||
self.assertRaises(ConsulError, self.c.get_cluster)
|
self.assertRaises(ConsulError, self.c.get_cluster)
|
||||||
|
self.c._base_path = '/service/broken'
|
||||||
|
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
||||||
|
self.c._base_path = '/service/legacy'
|
||||||
|
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
||||||
self.c._base_path = '/service/good'
|
self.c._base_path = '/service/good'
|
||||||
self.c._session = 'fd4f44fe-2cac-bba5-a60b-304b51ff39b8'
|
self.c._session = 'fd4f44fe-2cac-bba5-a60b-304b51ff39b8'
|
||||||
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
self.assertIsInstance(self.c.get_cluster(), Cluster)
|
||||||
@@ -117,8 +130,9 @@ class TestConsul(unittest.TestCase):
|
|||||||
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=[True, ConsulException, InvalidSession]))
|
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=[True, ConsulException, InvalidSession]))
|
||||||
def test_touch_member(self):
|
def test_touch_member(self):
|
||||||
self.c.refresh_session = Mock(return_value=False)
|
self.c.refresh_session = Mock(return_value=False)
|
||||||
self.c.touch_member({'conn_url': 'postgres://replicator:[email protected]:5433/postgres',
|
with patch.object(Consul, 'update_service', Mock(side_effect=Exception)):
|
||||||
'api_url': 'http://127.0.0.1:8009/patroni'})
|
self.c.touch_member({'conn_url': 'postgres://replicator:[email protected]:5433/postgres',
|
||||||
|
'api_url': 'http://127.0.0.1:8009/patroni'})
|
||||||
self.c._register_service = True
|
self.c._register_service = True
|
||||||
self.c.refresh_session = Mock(return_value=True)
|
self.c.refresh_session = Mock(return_value=True)
|
||||||
for _ in range(0, 4):
|
for _ in range(0, 4):
|
||||||
@@ -140,13 +154,15 @@ class TestConsul(unittest.TestCase):
|
|||||||
def test_set_config_value(self):
|
def test_set_config_value(self):
|
||||||
self.c.set_config_value('')
|
self.c.set_config_value('')
|
||||||
|
|
||||||
|
@patch.object(Cluster, 'min_version', PropertyMock(return_value=(2, 0)))
|
||||||
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=ConsulException))
|
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=ConsulException))
|
||||||
def test_write_leader_optime(self):
|
def test_write_leader_optime(self):
|
||||||
|
self.c.get_cluster()
|
||||||
self.c.write_leader_optime('1')
|
self.c.write_leader_optime('1')
|
||||||
|
|
||||||
@patch.object(consul.Consul.Session, 'renew', Mock())
|
@patch.object(consul.Consul.Session, 'renew', Mock())
|
||||||
def test_update_leader(self):
|
def test_update_leader(self):
|
||||||
self.c.update_leader(None)
|
self.c.update_leader(12345)
|
||||||
|
|
||||||
@patch.object(consul.Consul.KV, 'delete', Mock(return_value=True))
|
@patch.object(consul.Consul.KV, 'delete', Mock(return_value=True))
|
||||||
def test_delete_leader(self):
|
def test_delete_leader(self):
|
||||||
@@ -202,4 +218,50 @@ class TestConsul(unittest.TestCase):
|
|||||||
self.assertIsNone(self.c.update_service({}, d))
|
self.assertIsNone(self.c.update_service({}, d))
|
||||||
|
|
||||||
def test_reload_config(self):
|
def test_reload_config(self):
|
||||||
self.c.reload_config({'consul': {'token': 'foo'}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
self.assertEqual([], self.c._service_tags)
|
||||||
|
self.c.reload_config({'consul': {'token': 'foo', 'register_service': True, 'service_tags': ['foo']},
|
||||||
|
'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||||
|
self.assertEqual(["foo"], self.c._service_tags)
|
||||||
|
|
||||||
|
self.c.refresh_session = Mock(return_value=False)
|
||||||
|
|
||||||
|
d = {'role': 'replica', 'api_url': 'http://a/t', 'conn_url': 'pg://c:1', 'state': 'running'}
|
||||||
|
|
||||||
|
# Changing register_service from True to False calls deregister()
|
||||||
|
self.c.reload_config({'consul': {'register_service': False}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||||
|
with patch('consul.Consul.Agent.Service.deregister') as mock_deregister:
|
||||||
|
self.c.touch_member(d)
|
||||||
|
mock_deregister.assert_called_once()
|
||||||
|
|
||||||
|
self.assertEqual([], self.c._service_tags)
|
||||||
|
|
||||||
|
# register_service staying False between reloads does not call deregister()
|
||||||
|
self.c.reload_config({'consul': {'register_service': False}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||||
|
with patch('consul.Consul.Agent.Service.deregister') as mock_deregister:
|
||||||
|
self.c.touch_member(d)
|
||||||
|
self.assertFalse(mock_deregister.called)
|
||||||
|
|
||||||
|
# Changing register_service from False to True calls register()
|
||||||
|
self.c.reload_config({'consul': {'register_service': True}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||||
|
with patch('consul.Consul.Agent.Service.register') as mock_register:
|
||||||
|
self.c.touch_member(d)
|
||||||
|
mock_register.assert_called_once()
|
||||||
|
|
||||||
|
# register_service staying True between reloads does not call register()
|
||||||
|
self.c.reload_config({'consul': {'register_service': True}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||||
|
with patch('consul.Consul.Agent.Service.register') as mock_register:
|
||||||
|
self.c.touch_member(d)
|
||||||
|
self.assertFalse(mock_deregister.called)
|
||||||
|
|
||||||
|
# register_service staying True between reloads does calls register() if other service data has changed
|
||||||
|
self.c.reload_config({'consul': {'register_service': True}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||||
|
with patch('consul.Consul.Agent.Service.register') as mock_register:
|
||||||
|
self.c.touch_member(d)
|
||||||
|
mock_register.assert_called_once()
|
||||||
|
|
||||||
|
# register_service staying True between reloads does calls register() if service_tags have changed
|
||||||
|
self.c.reload_config({'consul': {'register_service': True, 'service_tags': ['foo']}, 'loop_wait': 10,
|
||||||
|
'ttl': 30, 'retry_timeout': 10})
|
||||||
|
with patch('consul.Consul.Agent.Service.register') as mock_register:
|
||||||
|
self.c.touch_member(d)
|
||||||
|
mock_register.assert_called_once()
|
||||||
|
|||||||
+5
-5
@@ -9,11 +9,11 @@ from patroni.ctl import ctl, store_config, load_config, output_members, get_dcs,
|
|||||||
get_all_members, get_any_member, get_cursor, query_member, configure, PatroniCtlException, apply_config_changes, \
|
get_all_members, get_any_member, get_cursor, query_member, configure, PatroniCtlException, apply_config_changes, \
|
||||||
format_config_for_editing, show_diff, invoke_editor, format_pg_version, CONFIG_FILE_PATH
|
format_config_for_editing, show_diff, invoke_editor, format_pg_version, CONFIG_FILE_PATH
|
||||||
from patroni.dcs.etcd import AbstractEtcdClientWithFailover, Failover
|
from patroni.dcs.etcd import AbstractEtcdClientWithFailover, Failover
|
||||||
|
from patroni.psycopg import OperationalError
|
||||||
from patroni.utils import tzutc
|
from patroni.utils import tzutc
|
||||||
from psycopg2 import OperationalError
|
|
||||||
from urllib3 import PoolManager
|
from urllib3 import PoolManager
|
||||||
|
|
||||||
from . import MockConnect, MockCursor, MockResponse, psycopg2_connect
|
from . import MockConnect, MockCursor, MockResponse, psycopg_connect
|
||||||
from .test_etcd import etcd_read, socket_getaddrinfo
|
from .test_etcd import etcd_read, socket_getaddrinfo
|
||||||
from .test_ha import get_cluster_initialized_without_leader, get_cluster_initialized_with_leader, \
|
from .test_ha import get_cluster_initialized_without_leader, get_cluster_initialized_with_leader, \
|
||||||
get_cluster_initialized_with_only_leader, get_cluster_not_initialized_without_leader, get_cluster, Member
|
get_cluster_initialized_with_only_leader, get_cluster_not_initialized_without_leader, get_cluster, Member
|
||||||
@@ -48,7 +48,7 @@ class TestCtl(unittest.TestCase):
|
|||||||
self.assertRaises(PatroniCtlException, load_config, './non-existing-config-file', None)
|
self.assertRaises(PatroniCtlException, load_config, './non-existing-config-file', None)
|
||||||
self.assertRaises(PatroniCtlException, load_config, './non-existing-config-file', None)
|
self.assertRaises(PatroniCtlException, load_config, './non-existing-config-file', None)
|
||||||
|
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
def test_get_cursor(self):
|
def test_get_cursor(self):
|
||||||
self.assertIsNone(get_cursor(get_cluster_initialized_without_leader(), {}, role='master'))
|
self.assertIsNone(get_cursor(get_cluster_initialized_without_leader(), {}, role='master'))
|
||||||
|
|
||||||
@@ -57,7 +57,7 @@ class TestCtl(unittest.TestCase):
|
|||||||
# MockCursor returns pg_is_in_recovery as false
|
# MockCursor returns pg_is_in_recovery as false
|
||||||
self.assertIsNone(get_cursor(get_cluster_initialized_with_leader(), {}, role='replica'))
|
self.assertIsNone(get_cursor(get_cluster_initialized_with_leader(), {}, role='replica'))
|
||||||
|
|
||||||
self.assertIsNotNone(get_cursor(get_cluster_initialized_with_leader(), {'database': 'foo'}, role='any'))
|
self.assertIsNotNone(get_cursor(get_cluster_initialized_with_leader(), {'dbname': 'foo'}, role='any'))
|
||||||
|
|
||||||
def test_parse_dcs(self):
|
def test_parse_dcs(self):
|
||||||
assert parse_dcs(None) is None
|
assert parse_dcs(None) is None
|
||||||
@@ -165,7 +165,7 @@ class TestCtl(unittest.TestCase):
|
|||||||
def test_get_dcs(self):
|
def test_get_dcs(self):
|
||||||
self.assertRaises(PatroniCtlException, get_dcs, {'dummy': {}}, 'dummy')
|
self.assertRaises(PatroniCtlException, get_dcs, {'dummy': {}}, 'dummy')
|
||||||
|
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
@patch('patroni.ctl.query_member', Mock(return_value=([['mock column']], None)))
|
@patch('patroni.ctl.query_member', Mock(return_value=([['mock column']], None)))
|
||||||
@patch('patroni.ctl.get_dcs')
|
@patch('patroni.ctl.get_dcs')
|
||||||
@patch.object(etcd.Client, 'read', etcd_read)
|
@patch.object(etcd.Client, 'read', etcd_read)
|
||||||
|
|||||||
+20
-3
@@ -4,7 +4,7 @@ import socket
|
|||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from dns.exception import DNSException
|
from dns.exception import DNSException
|
||||||
from mock import Mock, patch
|
from mock import Mock, PropertyMock, patch
|
||||||
from patroni.dcs.etcd import AbstractDCS, EtcdClient, Cluster, Etcd, EtcdError, DnsCachingResolver
|
from patroni.dcs.etcd import AbstractDCS, EtcdClient, Cluster, Etcd, EtcdError, DnsCachingResolver
|
||||||
from patroni.exceptions import DCSError
|
from patroni.exceptions import DCSError
|
||||||
from patroni.utils import Retry
|
from patroni.utils import Retry
|
||||||
@@ -66,7 +66,13 @@ def etcd_read(self, key, **kwargs):
|
|||||||
"?application_name=http://127.0.0.1:8008/patroni",
|
"?application_name=http://127.0.0.1:8008/patroni",
|
||||||
"expiration": "2015-05-15T09:11:09.611860899Z", "ttl": 30,
|
"expiration": "2015-05-15T09:11:09.611860899Z", "ttl": 30,
|
||||||
"modifiedIndex": 20730, "createdIndex": 20730}],
|
"modifiedIndex": 20730, "createdIndex": 20730}],
|
||||||
"modifiedIndex": 1581, "createdIndex": 1581}], "modifiedIndex": 1581, "createdIndex": 1581}}
|
"modifiedIndex": 1581, "createdIndex": 1581},
|
||||||
|
{"key": "/service/batman5/status", "value": '{"optime":2164261704,"slots":{"ls":12345}}',
|
||||||
|
"modifiedIndex": 1582, "createdIndex": 1582}], "modifiedIndex": 1581, "createdIndex": 1581}}
|
||||||
|
if key == '/service/legacy/':
|
||||||
|
response['node']['nodes'].pop()
|
||||||
|
if key == '/service/broken/':
|
||||||
|
response['node']['nodes'][-1]['value'] = '{'
|
||||||
result = etcd.EtcdResult(**response)
|
result = etcd.EtcdResult(**response)
|
||||||
result.etcd_index = 0
|
result.etcd_index = 0
|
||||||
return result
|
return result
|
||||||
@@ -81,7 +87,8 @@ def dns_query(name, _):
|
|||||||
raise DNSException()
|
raise DNSException()
|
||||||
srv = Mock()
|
srv = Mock()
|
||||||
srv.port = 2380
|
srv.port = 2380
|
||||||
srv.target.to_text.return_value = 'localhost' if name == '_etcd-server._tcp.foobar' else '127.0.0.1'
|
srv.target.to_text.return_value = \
|
||||||
|
'localhost' if name in ['_etcd-server._tcp.foobar', '_etcd-server-baz._tcp.foobar'] else '127.0.0.1'
|
||||||
return [srv]
|
return [srv]
|
||||||
|
|
||||||
|
|
||||||
@@ -177,6 +184,7 @@ class TestClient(unittest.TestCase):
|
|||||||
|
|
||||||
def test__get_machines_cache_from_srv(self):
|
def test__get_machines_cache_from_srv(self):
|
||||||
self.client._get_machines_cache_from_srv('foobar')
|
self.client._get_machines_cache_from_srv('foobar')
|
||||||
|
self.client._get_machines_cache_from_srv('foobar', 'baz')
|
||||||
self.client.get_srv_record = Mock(return_value=[('localhost', 2380)])
|
self.client.get_srv_record = Mock(return_value=[('localhost', 2380)])
|
||||||
self.client._get_machines_cache_from_srv('blabla')
|
self.client._get_machines_cache_from_srv('blabla')
|
||||||
|
|
||||||
@@ -246,6 +254,10 @@ class TestEtcd(unittest.TestCase):
|
|||||||
cluster = self.etcd.get_cluster()
|
cluster = self.etcd.get_cluster()
|
||||||
self.assertIsInstance(cluster, Cluster)
|
self.assertIsInstance(cluster, Cluster)
|
||||||
self.assertFalse(cluster.is_synchronous_mode())
|
self.assertFalse(cluster.is_synchronous_mode())
|
||||||
|
self.etcd._base_path = '/service/legacy'
|
||||||
|
self.assertIsInstance(self.etcd.get_cluster(), Cluster)
|
||||||
|
self.etcd._base_path = '/service/broken'
|
||||||
|
self.assertIsInstance(self.etcd.get_cluster(), Cluster)
|
||||||
self.etcd._base_path = '/service/nocluster'
|
self.etcd._base_path = '/service/nocluster'
|
||||||
cluster = self.etcd.get_cluster()
|
cluster = self.etcd.get_cluster()
|
||||||
self.assertIsInstance(cluster, Cluster)
|
self.assertIsInstance(cluster, Cluster)
|
||||||
@@ -265,7 +277,9 @@ class TestEtcd(unittest.TestCase):
|
|||||||
self.etcd._base_path = '/service/failed'
|
self.etcd._base_path = '/service/failed'
|
||||||
self.assertFalse(self.etcd.attempt_to_acquire_leader())
|
self.assertFalse(self.etcd.attempt_to_acquire_leader())
|
||||||
|
|
||||||
|
@patch.object(Cluster, 'min_version', PropertyMock(return_value=(2, 0)))
|
||||||
def test_write_leader_optime(self):
|
def test_write_leader_optime(self):
|
||||||
|
self.etcd.get_cluster()
|
||||||
self.etcd.write_leader_optime('0')
|
self.etcd.write_leader_optime('0')
|
||||||
|
|
||||||
def test_update_leader(self):
|
def test_update_leader(self):
|
||||||
@@ -309,3 +323,6 @@ class TestEtcd(unittest.TestCase):
|
|||||||
|
|
||||||
def test_set_history_value(self):
|
def test_set_history_value(self):
|
||||||
self.assertFalse(self.etcd.set_history_value('{}'))
|
self.assertFalse(self.etcd.set_history_value('{}'))
|
||||||
|
|
||||||
|
def test_last_seen(self):
|
||||||
|
self.assertIsNotNone(self.etcd.last_seen)
|
||||||
|
|||||||
+31
-2
@@ -4,8 +4,9 @@ import unittest
|
|||||||
import urllib3
|
import urllib3
|
||||||
|
|
||||||
from mock import Mock, patch
|
from mock import Mock, patch
|
||||||
from patroni.dcs.etcd3 import PatroniEtcd3Client, Cluster, Etcd3, Etcd3Error, Etcd3ClientError, RetryFailedError,\
|
from patroni.dcs.etcd import DnsCachingResolver
|
||||||
InvalidAuthToken, Unavailable, Unknown, UnsupportedEtcdVersion, UserEmpty, AuthFailed, base64_encode
|
from patroni.dcs.etcd3 import PatroniEtcd3Client, Cluster, Etcd3Client, Etcd3Error, Etcd3ClientError, RetryFailedError,\
|
||||||
|
InvalidAuthToken, Unavailable, Unknown, UnsupportedEtcdVersion, UserEmpty, AuthFailed, base64_encode, Etcd3
|
||||||
from threading import Thread
|
from threading import Thread
|
||||||
|
|
||||||
from . import SleepException, MockResponse
|
from . import SleepException, MockResponse
|
||||||
@@ -33,6 +34,8 @@ def mock_urlopen(self, method, url, **kwargs):
|
|||||||
"value": base64_encode('foo'), "lease": "bla", "mod_revision": '1'},
|
"value": base64_encode('foo'), "lease": "bla", "mod_revision": '1'},
|
||||||
{"key": base64_encode('/patroni/test/members/foo'),
|
{"key": base64_encode('/patroni/test/members/foo'),
|
||||||
"value": base64_encode('{}'), "lease": "123", "mod_revision": '1'},
|
"value": base64_encode('{}'), "lease": "123", "mod_revision": '1'},
|
||||||
|
{"key": base64_encode('/patroni/test/members/bar'),
|
||||||
|
"value": base64_encode('{"version":"1.6.5"}'), "lease": "123", "mod_revision": '1'},
|
||||||
{"key": base64_encode('/patroni/test/failover'), "value": base64_encode('{}'), "mod_revision": '1'}
|
{"key": base64_encode('/patroni/test/failover'), "value": base64_encode('{}'), "mod_revision": '1'}
|
||||||
]
|
]
|
||||||
})
|
})
|
||||||
@@ -55,6 +58,16 @@ def mock_urlopen(self, method, url, **kwargs):
|
|||||||
return ret
|
return ret
|
||||||
|
|
||||||
|
|
||||||
|
class TestEtcd3Client(unittest.TestCase):
|
||||||
|
|
||||||
|
@patch.object(Thread, 'start', Mock())
|
||||||
|
@patch.object(urllib3.PoolManager, 'urlopen', mock_urlopen)
|
||||||
|
def test_authenticate(self):
|
||||||
|
etcd3 = Etcd3Client({'host': '127.0.0.1', 'port': 2379, 'use_proxies': True, 'retry_timeout': 10},
|
||||||
|
DnsCachingResolver())
|
||||||
|
self.assertIsNotNone(etcd3._cluster_version)
|
||||||
|
|
||||||
|
|
||||||
class BaseTestEtcd3(unittest.TestCase):
|
class BaseTestEtcd3(unittest.TestCase):
|
||||||
|
|
||||||
@patch.object(Thread, 'start', Mock())
|
@patch.object(Thread, 'start', Mock())
|
||||||
@@ -172,6 +185,22 @@ class TestEtcd3(BaseTestEtcd3):
|
|||||||
self.assertIsInstance(self.etcd3.get_cluster(), Cluster)
|
self.assertIsInstance(self.etcd3.get_cluster(), Cluster)
|
||||||
self.client._kv_cache = None
|
self.client._kv_cache = None
|
||||||
with patch.object(urllib3.PoolManager, 'urlopen') as mock_urlopen:
|
with patch.object(urllib3.PoolManager, 'urlopen') as mock_urlopen:
|
||||||
|
mock_urlopen.return_value = MockResponse()
|
||||||
|
mock_urlopen.return_value.content = json.dumps({
|
||||||
|
"header": {"revision": "1"},
|
||||||
|
"kvs": [
|
||||||
|
{"key": base64_encode('/patroni/test/status'),
|
||||||
|
"value": base64_encode('{"optime":1234567,"slots":{"ls":12345}}'), "mod_revision": '1'}
|
||||||
|
]
|
||||||
|
})
|
||||||
|
self.assertIsInstance(self.etcd3.get_cluster(), Cluster)
|
||||||
|
mock_urlopen.return_value.content = json.dumps({
|
||||||
|
"header": {"revision": "1"},
|
||||||
|
"kvs": [
|
||||||
|
{"key": base64_encode('/patroni/test/status'), "value": base64_encode('{'), "mod_revision": '1'}
|
||||||
|
]
|
||||||
|
})
|
||||||
|
self.assertIsInstance(self.etcd3.get_cluster(), Cluster)
|
||||||
mock_urlopen.side_effect = UnsupportedEtcdVersion('')
|
mock_urlopen.side_effect = UnsupportedEtcdVersion('')
|
||||||
self.assertRaises(UnsupportedEtcdVersion, self.etcd3.get_cluster)
|
self.assertRaises(UnsupportedEtcdVersion, self.etcd3.get_cluster)
|
||||||
mock_urlopen.side_effect = SleepException()
|
mock_urlopen.side_effect = SleepException()
|
||||||
|
|||||||
@@ -24,7 +24,7 @@ class TestExhibitor(unittest.TestCase):
|
|||||||
|
|
||||||
@patch('urllib3.PoolManager.request', Mock(return_value=urllib3.HTTPResponse(
|
@patch('urllib3.PoolManager.request', Mock(return_value=urllib3.HTTPResponse(
|
||||||
status=200, body=b'{"servers":["127.0.0.1","127.0.0.2","127.0.0.3"],"port":2181}')))
|
status=200, body=b'{"servers":["127.0.0.1","127.0.0.2","127.0.0.3"],"port":2181}')))
|
||||||
@patch('patroni.dcs.zookeeper.KazooClient', MockKazooClient)
|
@patch('patroni.dcs.zookeeper.PatroniKazooClient', MockKazooClient)
|
||||||
def setUp(self):
|
def setUp(self):
|
||||||
self.e = Exhibitor({'hosts': ['localhost', 'exhibitor'], 'port': 8181, 'scope': 'test',
|
self.e = Exhibitor({'hosts': ['localhost', 'exhibitor'], 'port': 8181, 'scope': 'test',
|
||||||
'name': 'foo', 'ttl': 30, 'retry_timeout': 10})
|
'name': 'foo', 'ttl': 30, 'retry_timeout': 10})
|
||||||
|
|||||||
+120
-48
@@ -19,7 +19,7 @@ from patroni.utils import tzutc
|
|||||||
from patroni.watchdog import Watchdog
|
from patroni.watchdog import Watchdog
|
||||||
from six.moves import builtins
|
from six.moves import builtins
|
||||||
|
|
||||||
from . import PostgresInit, MockPostmaster, psycopg2_connect, requests_get
|
from . import PostgresInit, MockPostmaster, psycopg_connect, requests_get
|
||||||
from .test_etcd import socket_getaddrinfo, etcd_read, etcd_write
|
from .test_etcd import socket_getaddrinfo, etcd_read, etcd_write
|
||||||
|
|
||||||
SYSID = '12345678901'
|
SYSID = '12345678901'
|
||||||
@@ -35,10 +35,10 @@ def false(*args, **kwargs):
|
|||||||
|
|
||||||
def get_cluster(initialize, leader, members, failover, sync, cluster_config=None):
|
def get_cluster(initialize, leader, members, failover, sync, cluster_config=None):
|
||||||
t = datetime.datetime.now().isoformat()
|
t = datetime.datetime.now().isoformat()
|
||||||
history = TimelineHistory(1, '[[1,67197376,"no recovery target specified","' + t + '"]]',
|
history = TimelineHistory(1, '[[1,67197376,"no recovery target specified","' + t + '","foo"]]',
|
||||||
[(1, 67197376, 'no recovery target specified', t)])
|
[(1, 67197376, 'no recovery target specified', t, 'foo')])
|
||||||
cluster_config = cluster_config or ClusterConfig(1, {'check_timeline': True}, 1)
|
cluster_config = cluster_config or ClusterConfig(1, {'check_timeline': True}, 1)
|
||||||
return Cluster(initialize, cluster_config, leader, 10, members, failover, sync, history)
|
return Cluster(initialize, cluster_config, leader, 10, members, failover, sync, history, None)
|
||||||
|
|
||||||
|
|
||||||
def get_cluster_not_initialized_without_leader(cluster_config=None):
|
def get_cluster_not_initialized_without_leader(cluster_config=None):
|
||||||
@@ -66,7 +66,7 @@ def get_cluster_initialized_with_leader(failover=None, sync=None):
|
|||||||
|
|
||||||
def get_cluster_initialized_with_only_leader(failover=None, cluster_config=None):
|
def get_cluster_initialized_with_only_leader(failover=None, cluster_config=None):
|
||||||
leader = get_cluster_initialized_without_leader(leader=True, failover=failover).leader
|
leader = get_cluster_initialized_without_leader(leader=True, failover=failover).leader
|
||||||
return get_cluster(True, leader, [leader], failover, None, cluster_config)
|
return get_cluster(True, leader, [leader.member], failover, None, cluster_config)
|
||||||
|
|
||||||
|
|
||||||
def get_standby_cluster_initialized_with_only_leader(failover=None, sync=None):
|
def get_standby_cluster_initialized_with_only_leader(failover=None, sync=None):
|
||||||
@@ -80,13 +80,14 @@ def get_standby_cluster_initialized_with_only_leader(failover=None, sync=None):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def get_node_status(reachable=True, in_recovery=True, timeline=2,
|
def get_node_status(reachable=True, in_recovery=True, dcs_last_seen=0,
|
||||||
wal_position=10, nofailover=False, watchdog_failed=False):
|
timeline=2, wal_position=10, nofailover=False,
|
||||||
|
watchdog_failed=False):
|
||||||
def fetch_node_status(e):
|
def fetch_node_status(e):
|
||||||
tags = {}
|
tags = {}
|
||||||
if nofailover:
|
if nofailover:
|
||||||
tags['nofailover'] = True
|
tags['nofailover'] = True
|
||||||
return _MemberStatus(e, reachable, in_recovery, timeline, wal_position, tags, watchdog_failed)
|
return _MemberStatus(e, reachable, in_recovery, dcs_last_seen, timeline, wal_position, tags, watchdog_failed)
|
||||||
return fetch_node_status
|
return fetch_node_status
|
||||||
|
|
||||||
|
|
||||||
@@ -152,13 +153,14 @@ def run_async(self, func, args=()):
|
|||||||
@patch.object(Postgresql, 'is_running', Mock(return_value=MockPostmaster()))
|
@patch.object(Postgresql, 'is_running', Mock(return_value=MockPostmaster()))
|
||||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=True))
|
@patch.object(Postgresql, 'is_leader', Mock(return_value=True))
|
||||||
@patch.object(Postgresql, 'timeline_wal_position', Mock(return_value=(1, 10, 1)))
|
@patch.object(Postgresql, 'timeline_wal_position', Mock(return_value=(1, 10, 1)))
|
||||||
@patch.object(Postgresql, '_cluster_info_state_get', Mock(return_value=3))
|
@patch.object(Postgresql, '_cluster_info_state_get', Mock(return_value=10))
|
||||||
@patch.object(Postgresql, 'data_directory_empty', Mock(return_value=False))
|
@patch.object(Postgresql, 'data_directory_empty', Mock(return_value=False))
|
||||||
@patch.object(Postgresql, 'controldata', Mock(return_value={
|
@patch.object(Postgresql, 'controldata', Mock(return_value={
|
||||||
'Database system identifier': SYSID,
|
'Database system identifier': SYSID,
|
||||||
'Database cluster state': 'shut down',
|
'Database cluster state': 'shut down',
|
||||||
'Latest checkpoint location': '0/12345678'}))
|
'Latest checkpoint location': '0/12345678',
|
||||||
@patch.object(SlotsHandler, 'sync_replication_slots', Mock())
|
"Latest checkpoint's TimeLineID": '2'}))
|
||||||
|
@patch.object(SlotsHandler, 'load_replication_slots', Mock(side_effect=Exception))
|
||||||
@patch.object(ConfigHandler, 'append_pg_hba', Mock())
|
@patch.object(ConfigHandler, 'append_pg_hba', Mock())
|
||||||
@patch.object(ConfigHandler, 'write_pgpass', Mock(return_value={}))
|
@patch.object(ConfigHandler, 'write_pgpass', Mock(return_value={}))
|
||||||
@patch.object(ConfigHandler, 'write_recovery_conf', Mock())
|
@patch.object(ConfigHandler, 'write_recovery_conf', Mock())
|
||||||
@@ -250,6 +252,15 @@ class TestHa(PostgresInit):
|
|||||||
self.assertEqual(self.ha.run_cycle(), 'starting as a secondary')
|
self.assertEqual(self.ha.run_cycle(), 'starting as a secondary')
|
||||||
self.assertEqual(self.ha.run_cycle(), 'failed to start postgres')
|
self.assertEqual(self.ha.run_cycle(), 'failed to start postgres')
|
||||||
|
|
||||||
|
def test_recover_raft(self):
|
||||||
|
self.p.controldata = lambda: {'Database cluster state': 'in recovery', 'Database system identifier': SYSID}
|
||||||
|
self.p.is_running = false
|
||||||
|
self.p.follow = true
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'starting as a secondary')
|
||||||
|
self.p.is_running = true
|
||||||
|
self.ha.dcs.__class__.__name__ = 'Raft'
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'started as a secondary')
|
||||||
|
|
||||||
def test_recover_former_master(self):
|
def test_recover_former_master(self):
|
||||||
self.p.follow = false
|
self.p.follow = false
|
||||||
self.p.is_running = false
|
self.p.is_running = false
|
||||||
@@ -278,7 +289,17 @@ class TestHa(PostgresInit):
|
|||||||
def test_recover_with_rewind(self):
|
def test_recover_with_rewind(self):
|
||||||
self.p.is_running = false
|
self.p.is_running = false
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||||
self.assertEqual(self.ha.run_cycle(), 'running pg_rewind from leader')
|
self.ha.cluster.leader.member.data.update(version='2.0.2', role='master')
|
||||||
|
self.ha._rewind.pg_rewind = true
|
||||||
|
self.ha._rewind.check_leader_is_not_in_recovery = true
|
||||||
|
with patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True)):
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'running pg_rewind from leader')
|
||||||
|
with patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=False)):
|
||||||
|
self.p.follow = true
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'starting as a secondary')
|
||||||
|
self.p.is_running = true
|
||||||
|
self.ha.follow = Mock(return_value='fake')
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'fake')
|
||||||
|
|
||||||
@patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True))
|
@patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True))
|
||||||
@patch.object(Bootstrap, 'create_replica', Mock(return_value=1))
|
@patch.object(Bootstrap, 'create_replica', Mock(return_value=1))
|
||||||
@@ -300,9 +321,9 @@ class TestHa(PostgresInit):
|
|||||||
self.p.is_healthy = true
|
self.p.is_healthy = true
|
||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.p.controldata = lambda: {'Database cluster state': 'in production', 'Database system identifier': SYSID}
|
self.p.controldata = lambda: {'Database cluster state': 'in production', 'Database system identifier': SYSID}
|
||||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader because i had the session lock')
|
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader because I had the session lock')
|
||||||
|
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
def test_acquire_lock_as_master(self):
|
def test_acquire_lock_as_master(self):
|
||||||
self.assertEqual(self.ha.run_cycle(), 'acquired session lock as a leader')
|
self.assertEqual(self.ha.run_cycle(), 'acquired session lock as a leader')
|
||||||
|
|
||||||
@@ -329,7 +350,7 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.p.is_leader = false
|
self.p.is_leader = false
|
||||||
self.p.set_role('master')
|
self.p.set_role('master')
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
|
|
||||||
def test_demote_after_failing_to_obtain_lock(self):
|
def test_demote_after_failing_to_obtain_lock(self):
|
||||||
self.ha.acquire_lock = false
|
self.ha.acquire_lock = false
|
||||||
@@ -354,7 +375,7 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.p.is_leader = false
|
self.p.is_leader = false
|
||||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader because i had the session lock')
|
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader because I had the session lock')
|
||||||
|
|
||||||
def test_promote_without_watchdog(self):
|
def test_promote_without_watchdog(self):
|
||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
@@ -369,35 +390,45 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
|
|
||||||
def test_demote_because_not_having_lock(self):
|
def test_demote_because_not_having_lock(self):
|
||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
with patch.object(Watchdog, 'is_running', PropertyMock(return_value=True)):
|
with patch.object(Watchdog, 'is_running', PropertyMock(return_value=True)):
|
||||||
self.assertEqual(self.ha.run_cycle(), 'demoting self because i do not have the lock and i was a leader')
|
self.assertEqual(self.ha.run_cycle(), 'demoting self because I do not have the lock and I was a leader')
|
||||||
|
|
||||||
def test_demote_because_update_lock_failed(self):
|
def test_demote_because_update_lock_failed(self):
|
||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.ha.update_lock = false
|
self.ha.update_lock = false
|
||||||
self.assertEqual(self.ha.run_cycle(), 'demoted self because failed to update leader lock in DCS')
|
self.assertEqual(self.ha.run_cycle(), 'demoted self because failed to update leader lock in DCS')
|
||||||
|
with patch.object(Ha, '_get_node_to_follow', Mock(side_effect=DCSError('foo'))):
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'demoted self because failed to update leader lock in DCS')
|
||||||
self.p.is_leader = false
|
self.p.is_leader = false
|
||||||
self.assertEqual(self.ha.run_cycle(), 'not promoting because failed to update leader lock in DCS')
|
self.assertEqual(self.ha.run_cycle(), 'not promoting because failed to update leader lock in DCS')
|
||||||
|
|
||||||
|
@patch.object(Postgresql, 'major_version', PropertyMock(return_value=130000))
|
||||||
def test_follow(self):
|
def test_follow(self):
|
||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
self.p.is_leader = false
|
self.p.is_leader = false
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am a secondary and i am following a leader')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), a secondary, and following a leader ()')
|
||||||
self.ha.patroni.replicatefrom = "foo"
|
self.ha.patroni.replicatefrom = "foo"
|
||||||
self.p.config.check_recovery_conf = Mock(return_value=(True, False))
|
self.p.config.check_recovery_conf = Mock(return_value=(True, False))
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am a secondary and i am following a leader')
|
self.ha.cluster.config.data.update({'slots': {'l': {'database': 'a', 'plugin': 'b'}}})
|
||||||
|
self.ha.cluster.members[1].data['tags']['replicatefrom'] = 'postgresql0'
|
||||||
|
self.ha.patroni.nofailover = True
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), a secondary, and following a leader ()')
|
||||||
|
del self.ha.cluster.config.data['slots']
|
||||||
|
self.ha.cluster.config.data.update({'postgresql': {'use_slots': False}})
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), a secondary, and following a leader ()')
|
||||||
|
del self.ha.cluster.config.data['postgresql']['use_slots']
|
||||||
|
|
||||||
def test_follow_in_pause(self):
|
def test_follow_in_pause(self):
|
||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
self.ha.is_paused = true
|
self.ha.is_paused = true
|
||||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: continue to run as master without lock')
|
self.assertEqual(self.ha.run_cycle(), 'PAUSE: continue to run as master without lock')
|
||||||
self.p.is_leader = false
|
self.p.is_leader = false
|
||||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: no action')
|
self.assertEqual(self.ha.run_cycle(), 'PAUSE: no action. I am (postgresql0)')
|
||||||
|
|
||||||
@patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True))
|
@patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True))
|
||||||
@patch.object(Rewind, 'can_rewind', PropertyMock(return_value=True))
|
@patch.object(Rewind, 'can_rewind', PropertyMock(return_value=True))
|
||||||
@@ -429,6 +460,8 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||||
self.assertEqual(self.ha.bootstrap(), 'failed to acquire initialize lock')
|
self.assertEqual(self.ha.bootstrap(), 'failed to acquire initialize lock')
|
||||||
|
|
||||||
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
|
@patch.object(Postgresql, 'connection', Mock(return_value=None))
|
||||||
def test_bootstrap_initialized_new_cluster(self):
|
def test_bootstrap_initialized_new_cluster(self):
|
||||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||||
self.e.initialize = true
|
self.e.initialize = true
|
||||||
@@ -446,6 +479,8 @@ class TestHa(PostgresInit):
|
|||||||
self.p.is_running = false
|
self.p.is_running = false
|
||||||
self.assertRaises(PatroniFatalException, self.ha.post_bootstrap)
|
self.assertRaises(PatroniFatalException, self.ha.post_bootstrap)
|
||||||
|
|
||||||
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
|
@patch.object(Postgresql, 'connection', Mock(return_value=None))
|
||||||
def test_bootstrap_release_initialize_key_on_watchdog_failure(self):
|
def test_bootstrap_release_initialize_key_on_watchdog_failure(self):
|
||||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||||
self.e.initialize = true
|
self.e.initialize = true
|
||||||
@@ -456,7 +491,7 @@ class TestHa(PostgresInit):
|
|||||||
self.assertEqual(self.ha.post_bootstrap(), 'running post_bootstrap')
|
self.assertEqual(self.ha.post_bootstrap(), 'running post_bootstrap')
|
||||||
self.assertRaises(PatroniFatalException, self.ha.post_bootstrap)
|
self.assertRaises(PatroniFatalException, self.ha.post_bootstrap)
|
||||||
|
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
def test_reinitialize(self):
|
def test_reinitialize(self):
|
||||||
self.assertIsNotNone(self.ha.reinitialize())
|
self.assertIsNotNone(self.ha.reinitialize())
|
||||||
|
|
||||||
@@ -508,27 +543,27 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.fetch_node_status = get_node_status()
|
self.ha.fetch_node_status = get_node_status()
|
||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', '', None))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', '', None))
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', self.p.name, None))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', self.p.name, None))
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'blabla', None))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'blabla', None))
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
f = Failover(0, self.p.name, '', None)
|
f = Failover(0, self.p.name, '', None)
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(f)
|
self.ha.cluster = get_cluster_initialized_with_leader(f)
|
||||||
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||||
self.ha._rewind.rewind_or_reinitialize_needed_and_possible = true
|
self.ha._rewind.rewind_or_reinitialize_needed_and_possible = true
|
||||||
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||||
self.ha.fetch_node_status = get_node_status(nofailover=True)
|
self.ha.fetch_node_status = get_node_status(nofailover=True)
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
self.ha.fetch_node_status = get_node_status(watchdog_failed=True)
|
self.ha.fetch_node_status = get_node_status(watchdog_failed=True)
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
self.ha.fetch_node_status = get_node_status(timeline=1)
|
self.ha.fetch_node_status = get_node_status(timeline=1)
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
# manual failover from the previous leader to us won't happen if we hold the nofailover flag
|
# manual failover from the previous leader to us won't happen if we hold the nofailover flag
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, None))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, None))
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
|
|
||||||
# Failover scheduled time must include timezone
|
# Failover scheduled time must include timezone
|
||||||
scheduled = datetime.datetime.now()
|
scheduled = datetime.datetime.now()
|
||||||
@@ -537,28 +572,28 @@ class TestHa(PostgresInit):
|
|||||||
|
|
||||||
scheduled = datetime.datetime.utcnow().replace(tzinfo=tzutc)
|
scheduled = datetime.datetime.utcnow().replace(tzinfo=tzutc)
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||||
self.assertEqual('no action. i am the leader with the lock', self.ha.run_cycle())
|
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||||
|
|
||||||
scheduled = scheduled + datetime.timedelta(seconds=30)
|
scheduled = scheduled + datetime.timedelta(seconds=30)
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||||
self.assertEqual('no action. i am the leader with the lock', self.ha.run_cycle())
|
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||||
|
|
||||||
scheduled = scheduled + datetime.timedelta(seconds=-600)
|
scheduled = scheduled + datetime.timedelta(seconds=-600)
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||||
self.assertEqual('no action. i am the leader with the lock', self.ha.run_cycle())
|
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||||
|
|
||||||
scheduled = None
|
scheduled = None
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||||
self.assertEqual('no action. i am the leader with the lock', self.ha.run_cycle())
|
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||||
|
|
||||||
def test_manual_failover_from_leader_in_pause(self):
|
def test_manual_failover_from_leader_in_pause(self):
|
||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.ha.is_paused = true
|
self.ha.is_paused = true
|
||||||
scheduled = datetime.datetime.now()
|
scheduled = datetime.datetime.now()
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||||
self.assertEqual('PAUSE: no action. i am the leader with the lock', self.ha.run_cycle())
|
self.assertEqual('PAUSE: no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, '', None))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, '', None))
|
||||||
self.assertEqual('PAUSE: no action. i am the leader with the lock', self.ha.run_cycle())
|
self.assertEqual('PAUSE: no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||||
|
|
||||||
def test_manual_failover_from_leader_in_synchronous_mode(self):
|
def test_manual_failover_from_leader_in_synchronous_mode(self):
|
||||||
self.p.is_leader = true
|
self.p.is_leader = true
|
||||||
@@ -567,7 +602,7 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.is_failover_possible = false
|
self.ha.is_failover_possible = false
|
||||||
self.ha.process_sync_replication = Mock()
|
self.ha.process_sync_replication = Mock()
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, None))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, None))
|
||||||
self.assertEqual('no action. i am the leader with the lock', self.ha.run_cycle())
|
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, 'a'))
|
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, 'a'))
|
||||||
self.ha.is_failover_possible = true
|
self.ha.is_failover_possible = true
|
||||||
self.assertEqual('manual failover: demoting myself', self.ha.run_cycle())
|
self.assertEqual('manual failover: demoting myself', self.ha.run_cycle())
|
||||||
@@ -595,6 +630,11 @@ class TestHa(PostgresInit):
|
|||||||
# same as previous, but set the current member to nofailover. In no case it should be elected as a leader
|
# same as previous, but set the current member to nofailover. In no case it should be elected as a leader
|
||||||
self.ha.patroni.nofailover = True
|
self.ha.patroni.nofailover = True
|
||||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
||||||
|
# in sync mode only the sync node is allowed to take over
|
||||||
|
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'other', None))
|
||||||
|
self.ha.patroni.nofailover = False
|
||||||
|
self.ha.is_synchronous_mode = true
|
||||||
|
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||||
|
|
||||||
def test_manual_failover_process_no_leader_in_pause(self):
|
def test_manual_failover_process_no_leader_in_pause(self):
|
||||||
self.ha.is_paused = true
|
self.ha.is_paused = true
|
||||||
@@ -634,7 +674,7 @@ class TestHa(PostgresInit):
|
|||||||
# in synchronous_mode consider itself healthy if the former leader is accessible in read-only and ahead of us
|
# in synchronous_mode consider itself healthy if the former leader is accessible in read-only and ahead of us
|
||||||
with patch.object(Ha, 'is_synchronous_mode', Mock(return_value=True)):
|
with patch.object(Ha, 'is_synchronous_mode', Mock(return_value=True)):
|
||||||
self.assertTrue(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
self.assertTrue(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
||||||
with patch('patroni.postgresql.Postgresql.timeline_wal_position', return_value=(1, 1, 1)):
|
with patch('patroni.postgresql.Postgresql.last_operation', return_value=1):
|
||||||
self.assertFalse(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
self.assertFalse(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
||||||
with patch('patroni.postgresql.Postgresql.replica_cached_timeline', return_value=1):
|
with patch('patroni.postgresql.Postgresql.replica_cached_timeline', return_value=1):
|
||||||
self.assertFalse(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
self.assertFalse(self.ha._is_healthiest_node(self.ha.old_cluster.members))
|
||||||
@@ -673,7 +713,7 @@ class TestHa(PostgresInit):
|
|||||||
|
|
||||||
def test_evaluate_scheduled_restart(self):
|
def test_evaluate_scheduled_restart(self):
|
||||||
self.p.postmaster_start_time = Mock(return_value=str(postmaster_start_time))
|
self.p.postmaster_start_time = Mock(return_value=str(postmaster_start_time))
|
||||||
# restart already in progres
|
# restart already in progress
|
||||||
with patch('patroni.async_executor.AsyncExecutor.busy', PropertyMock(return_value=True)):
|
with patch('patroni.async_executor.AsyncExecutor.busy', PropertyMock(return_value=True)):
|
||||||
self.assertIsNone(self.ha.evaluate_scheduled_restart())
|
self.assertIsNone(self.ha.evaluate_scheduled_restart())
|
||||||
# restart while the postmaster has been already restarted, fails
|
# restart while the postmaster has been already restarted, fails
|
||||||
@@ -728,7 +768,7 @@ class TestHa(PostgresInit):
|
|||||||
self.p.config.check_recovery_conf = Mock(return_value=(False, False))
|
self.p.config.check_recovery_conf = Mock(return_value=(False, False))
|
||||||
self.ha._leader_timeline = 1
|
self.ha._leader_timeline = 1
|
||||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to a standby leader because i had the session lock')
|
self.assertEqual(self.ha.run_cycle(), 'promoted self to a standby leader because i had the session lock')
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the standby leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (leader), the standby leader with the lock')
|
||||||
self.p.set_role('replica')
|
self.p.set_role('replica')
|
||||||
self.p.config.check_recovery_conf = Mock(return_value=(True, False))
|
self.p.config.check_recovery_conf = Mock(return_value=(True, False))
|
||||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to a standby leader because i had the session lock')
|
self.assertEqual(self.ha.run_cycle(), 'promoted self to a standby leader because i had the session lock')
|
||||||
@@ -737,7 +777,8 @@ class TestHa(PostgresInit):
|
|||||||
self.p.is_leader = false
|
self.p.is_leader = false
|
||||||
self.p.name = 'replica'
|
self.p.name = 'replica'
|
||||||
self.ha.cluster = get_standby_cluster_initialized_with_only_leader()
|
self.ha.cluster = get_standby_cluster_initialized_with_only_leader()
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am a secondary and i am following a standby leader')
|
self.assertEqual(self.ha.run_cycle(),
|
||||||
|
'no action. I am (replica), a secondary, and following a standby leader (leader)')
|
||||||
with patch.object(Leader, 'conn_url', PropertyMock(return_value='')):
|
with patch.object(Leader, 'conn_url', PropertyMock(return_value='')):
|
||||||
self.assertEqual(self.ha.run_cycle(), 'continue following the old known standby leader')
|
self.assertEqual(self.ha.run_cycle(), 'continue following the old known standby leader')
|
||||||
|
|
||||||
@@ -830,7 +871,8 @@ class TestHa(PostgresInit):
|
|||||||
|
|
||||||
self.ha.has_lock = false
|
self.ha.has_lock = false
|
||||||
self.p.is_leader = false
|
self.p.is_leader = false
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am a secondary and i am following a leader')
|
self.assertEqual(self.ha.run_cycle(),
|
||||||
|
'no action. I am (postgresql0), a secondary, and following a leader (leader)')
|
||||||
check_calls([(update_lock, False), (demote, False)])
|
check_calls([(update_lock, False), (demote, False)])
|
||||||
|
|
||||||
def test_manual_failover_while_starting(self):
|
def test_manual_failover_while_starting(self):
|
||||||
@@ -1055,7 +1097,7 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.cluster.config.data.clear()
|
self.ha.cluster.config.data.clear()
|
||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
|
|
||||||
def test_watch(self):
|
def test_watch(self):
|
||||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||||
@@ -1067,6 +1109,14 @@ class TestHa(PostgresInit):
|
|||||||
def test_shutdown(self):
|
def test_shutdown(self):
|
||||||
self.p.is_running = false
|
self.p.is_running = false
|
||||||
self.ha.is_leader = true
|
self.ha.is_leader = true
|
||||||
|
|
||||||
|
def stop(*args, **kwargs):
|
||||||
|
kwargs['on_shutdown'](123)
|
||||||
|
|
||||||
|
self.p.stop = stop
|
||||||
|
self.ha.shutdown()
|
||||||
|
|
||||||
|
self.ha.is_failover_possible = true
|
||||||
self.ha.shutdown()
|
self.ha.shutdown()
|
||||||
|
|
||||||
@patch('time.sleep', Mock())
|
@patch('time.sleep', Mock())
|
||||||
@@ -1090,7 +1140,7 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
for tl in (1, 3):
|
for tl in (1, 3):
|
||||||
self.p.get_master_timeline = Mock(return_value=tl)
|
self.p.get_master_timeline = Mock(return_value=tl)
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
|
|
||||||
@patch('sys.exit', return_value=1)
|
@patch('sys.exit', return_value=1)
|
||||||
def test_abort_join(self, exit_mock):
|
def test_abort_join(self, exit_mock):
|
||||||
@@ -1103,18 +1153,19 @@ class TestHa(PostgresInit):
|
|||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.ha.cluster.is_unlocked = false
|
self.ha.cluster.is_unlocked = false
|
||||||
self.ha.is_paused = true
|
self.ha.is_paused = true
|
||||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'PAUSE: no action. I am (postgresql0), the leader with the lock')
|
||||||
self.ha.is_paused = false
|
self.ha.is_paused = false
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||||
|
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
def test_permanent_logical_slots_after_promote(self):
|
def test_permanent_logical_slots_after_promote(self):
|
||||||
config = ClusterConfig(1, {'slots': {'l': {'database': 'postgres', 'plugin': 'test_decoding'}}}, 1)
|
config = ClusterConfig(1, {'slots': {'l': {'database': 'postgres', 'plugin': 'test_decoding'}}}, 1)
|
||||||
|
self.p.name = 'other'
|
||||||
self.ha.cluster = get_cluster_initialized_without_leader(cluster_config=config)
|
self.ha.cluster = get_cluster_initialized_without_leader(cluster_config=config)
|
||||||
self.assertEqual(self.ha.run_cycle(), 'acquired session lock as a leader')
|
self.assertEqual(self.ha.run_cycle(), 'acquired session lock as a leader')
|
||||||
self.ha.cluster = get_cluster_initialized_without_leader(leader=True, cluster_config=config)
|
self.ha.cluster = get_cluster_initialized_without_leader(leader=True, cluster_config=config)
|
||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.assertEqual(self.ha.run_cycle(), 'no action. i am the leader with the lock')
|
self.assertEqual(self.ha.run_cycle(), 'no action. I am (other), the leader with the lock')
|
||||||
|
|
||||||
@patch.object(Cluster, 'has_member', true)
|
@patch.object(Cluster, 'has_member', true)
|
||||||
def test_run_cycle(self):
|
def test_run_cycle(self):
|
||||||
@@ -1137,3 +1188,24 @@ class TestHa(PostgresInit):
|
|||||||
|
|
||||||
self.ha.has_lock = true
|
self.ha.has_lock = true
|
||||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: released leader key voluntarily due to the system ID mismatch')
|
self.assertEqual(self.ha.run_cycle(), 'PAUSE: released leader key voluntarily due to the system ID mismatch')
|
||||||
|
|
||||||
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
|
@patch('os.path.exists', Mock(return_value=True))
|
||||||
|
@patch('shutil.rmtree', Mock())
|
||||||
|
@patch('os.makedirs', Mock())
|
||||||
|
@patch('os.open', Mock())
|
||||||
|
@patch('os.fsync', Mock())
|
||||||
|
@patch('os.close', Mock())
|
||||||
|
@patch('os.rename', Mock())
|
||||||
|
@patch('patroni.postgresql.Postgresql.is_starting', Mock(return_value=False))
|
||||||
|
@patch.object(builtins, 'open', mock_open())
|
||||||
|
@patch.object(SlotsHandler, 'sync_replication_slots', Mock(return_value=['foo']))
|
||||||
|
def test_follow_copy(self):
|
||||||
|
self.ha.cluster.is_unlocked = false
|
||||||
|
self.p.is_leader = false
|
||||||
|
self.assertTrue(self.ha.run_cycle().startswith('Copying logical slots'))
|
||||||
|
|
||||||
|
def test_is_failover_possible(self):
|
||||||
|
self.ha.fetch_node_status = Mock(return_value=_MemberStatus(self.ha.cluster.members[0],
|
||||||
|
True, True, 0, 2, None, {}, False))
|
||||||
|
self.assertFalse(self.ha.is_failover_possible(self.ha.cluster.members))
|
||||||
|
|||||||
@@ -16,7 +16,8 @@ def mock_list_namespaced_config_map(*args, **kwargs):
|
|||||||
metadata = {'resource_version': '1', 'labels': {'f': 'b'}, 'name': 'test-config',
|
metadata = {'resource_version': '1', 'labels': {'f': 'b'}, 'name': 'test-config',
|
||||||
'annotations': {'initialize': '123', 'config': '{}'}}
|
'annotations': {'initialize': '123', 'config': '{}'}}
|
||||||
items = [k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata))]
|
items = [k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata))]
|
||||||
metadata.update({'name': 'test-leader', 'annotations': {'optime': '1234', 'leader': 'p-0', 'ttl': '30s'}})
|
metadata.update({'name': 'test-leader',
|
||||||
|
'annotations': {'optime': '1234x', 'leader': 'p-0', 'ttl': '30s', 'slots': '{'}})
|
||||||
items.append(k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata)))
|
items.append(k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata)))
|
||||||
metadata.update({'name': 'test-failover', 'annotations': {'leader': 'p-0'}})
|
metadata.update({'name': 'test-failover', 'annotations': {'leader': 'p-0'}})
|
||||||
items.append(k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata)))
|
items.append(k8s_client.V1ConfigMap(metadata=k8s_client.V1ObjectMeta(**metadata)))
|
||||||
@@ -260,10 +261,6 @@ class TestKubernetesEndpoints(BaseTestKubernetes):
|
|||||||
self.k._kinds._object_cache['test'].metadata.annotations['leader'] = 'p-1'
|
self.k._kinds._object_cache['test'].metadata.annotations['leader'] = 'p-1'
|
||||||
self.assertFalse(self.k.update_leader('123'))
|
self.assertFalse(self.k.update_leader('123'))
|
||||||
|
|
||||||
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_endpoints', mock_namespaced_kind, create=True)
|
|
||||||
def test_update_leader_with_restricted_access(self):
|
|
||||||
self.assertIsNotNone(self.k.update_leader('123', True))
|
|
||||||
|
|
||||||
@patch.object(k8s_client.CoreV1Api, 'read_namespaced_endpoints', create=True)
|
@patch.object(k8s_client.CoreV1Api, 'read_namespaced_endpoints', create=True)
|
||||||
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_endpoints', create=True)
|
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_endpoints', create=True)
|
||||||
def test__update_leader_with_retry(self, mock_patch, mock_read):
|
def test__update_leader_with_retry(self, mock_patch, mock_read):
|
||||||
|
|||||||
@@ -63,3 +63,12 @@ class TestPatroniLogger(unittest.TestCase):
|
|||||||
self.assertRaises(Exception, logger.shutdown)
|
self.assertRaises(Exception, logger.shutdown)
|
||||||
self.assertLessEqual(logger.queue_size, 2) # "Failed to close the old log handler" could be still in the queue
|
self.assertLessEqual(logger.queue_size, 2) # "Failed to close the old log handler" could be still in the queue
|
||||||
self.assertEqual(logger.records_lost, 0)
|
self.assertEqual(logger.records_lost, 0)
|
||||||
|
|
||||||
|
def test_interceptor(self):
|
||||||
|
logger = PatroniLogger()
|
||||||
|
logger.reload_config({'level': 'INFO'})
|
||||||
|
logger.start()
|
||||||
|
_LOG.info('Lock owner: ')
|
||||||
|
_LOG.info('blabla')
|
||||||
|
logger.shutdown()
|
||||||
|
self.assertEqual(logger.records_lost, 0)
|
||||||
|
|||||||
+17
-8
@@ -13,15 +13,24 @@ from patroni.dcs.etcd import AbstractEtcdClientWithFailover
|
|||||||
from patroni.exceptions import DCSError
|
from patroni.exceptions import DCSError
|
||||||
from patroni.postgresql import Postgresql
|
from patroni.postgresql import Postgresql
|
||||||
from patroni.postgresql.config import ConfigHandler
|
from patroni.postgresql.config import ConfigHandler
|
||||||
from patroni import Patroni, main as _main, patroni_main, check_psycopg2
|
from patroni import check_psycopg
|
||||||
|
from patroni.__main__ import Patroni, main as _main, patroni_main
|
||||||
from six.moves import BaseHTTPServer, builtins
|
from six.moves import BaseHTTPServer, builtins
|
||||||
from threading import Thread
|
from threading import Thread
|
||||||
|
|
||||||
from . import psycopg2_connect, SleepException
|
from . import psycopg_connect, SleepException
|
||||||
from .test_etcd import etcd_read, etcd_write
|
from .test_etcd import etcd_read, etcd_write
|
||||||
from .test_postgresql import MockPostmaster
|
from .test_postgresql import MockPostmaster
|
||||||
|
|
||||||
|
|
||||||
|
def mock_import(*args, **kwargs):
|
||||||
|
if args[0] == 'psycopg':
|
||||||
|
raise ImportError
|
||||||
|
ret = Mock()
|
||||||
|
ret.__version__ = '2.5.3.dev1 a b c'
|
||||||
|
return ret
|
||||||
|
|
||||||
|
|
||||||
class MockFrozenImporter(object):
|
class MockFrozenImporter(object):
|
||||||
|
|
||||||
toc = set(['patroni.dcs.etcd'])
|
toc = set(['patroni.dcs.etcd'])
|
||||||
@@ -29,7 +38,7 @@ class MockFrozenImporter(object):
|
|||||||
|
|
||||||
@patch('time.sleep', Mock())
|
@patch('time.sleep', Mock())
|
||||||
@patch('subprocess.call', Mock(return_value=0))
|
@patch('subprocess.call', Mock(return_value=0))
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
@patch.object(ConfigHandler, 'append_pg_hba', Mock())
|
@patch.object(ConfigHandler, 'append_pg_hba', Mock())
|
||||||
@patch.object(ConfigHandler, 'write_postgresql_conf', Mock())
|
@patch.object(ConfigHandler, 'write_postgresql_conf', Mock())
|
||||||
@patch.object(ConfigHandler, 'write_recovery_conf', Mock())
|
@patch.object(ConfigHandler, 'write_recovery_conf', Mock())
|
||||||
@@ -89,7 +98,7 @@ class TestPatroni(unittest.TestCase):
|
|||||||
|
|
||||||
@patch('os.getpid')
|
@patch('os.getpid')
|
||||||
@patch('multiprocessing.Process')
|
@patch('multiprocessing.Process')
|
||||||
@patch('patroni.patroni_main', Mock())
|
@patch('patroni.__main__.patroni_main', Mock())
|
||||||
def test_patroni_main(self, mock_process, mock_getpid):
|
def test_patroni_main(self, mock_process, mock_getpid):
|
||||||
mock_getpid.return_value = 2
|
mock_getpid.return_value = 2
|
||||||
_main()
|
_main()
|
||||||
@@ -181,8 +190,8 @@ class TestPatroni(unittest.TestCase):
|
|||||||
self.p.ha.shutdown = Mock(side_effect=Exception)
|
self.p.ha.shutdown = Mock(side_effect=Exception)
|
||||||
self.p.shutdown()
|
self.p.shutdown()
|
||||||
|
|
||||||
def test_check_psycopg2(self):
|
def test_check_psycopg(self):
|
||||||
with patch.object(builtins, '__import__', Mock(side_effect=ImportError)):
|
with patch.object(builtins, '__import__', Mock(side_effect=ImportError)):
|
||||||
self.assertRaises(SystemExit, check_psycopg2)
|
self.assertRaises(SystemExit, check_psycopg)
|
||||||
with patch('psycopg2.__version__', '2.5.3.dev1 a b c'):
|
with patch.object(builtins, '__import__', mock_import):
|
||||||
self.assertRaises(SystemExit, check_psycopg2)
|
self.assertRaises(SystemExit, check_psycopg)
|
||||||
|
|||||||
+91
-47
@@ -1,23 +1,25 @@
|
|||||||
import mock # for the mock.call method, importing it without a namespace breaks python3
|
import datetime
|
||||||
import os
|
import os
|
||||||
import psutil
|
import psutil
|
||||||
import psycopg2
|
|
||||||
import re
|
import re
|
||||||
import subprocess
|
import subprocess
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from mock import Mock, MagicMock, PropertyMock, patch, mock_open
|
from mock import Mock, MagicMock, PropertyMock, patch, mock_open
|
||||||
|
|
||||||
|
import patroni.psycopg as psycopg
|
||||||
|
|
||||||
from patroni.async_executor import CriticalTask
|
from patroni.async_executor import CriticalTask
|
||||||
from patroni.dcs import Cluster, ClusterConfig, Member, RemoteMember, SyncState
|
from patroni.dcs import Cluster, RemoteMember, SyncState
|
||||||
from patroni.exceptions import PostgresConnectionException, PatroniException
|
from patroni.exceptions import PostgresConnectionException, PatroniException
|
||||||
from patroni.postgresql import Postgresql, STATE_REJECT, STATE_NO_RESPONSE
|
from patroni.postgresql import Postgresql, STATE_REJECT, STATE_NO_RESPONSE
|
||||||
|
from patroni.postgresql.bootstrap import Bootstrap
|
||||||
from patroni.postgresql.postmaster import PostmasterProcess
|
from patroni.postgresql.postmaster import PostmasterProcess
|
||||||
from patroni.postgresql.slots import SlotsHandler
|
|
||||||
from patroni.utils import RetryFailedError
|
from patroni.utils import RetryFailedError
|
||||||
from six.moves import builtins
|
from six.moves import builtins
|
||||||
from threading import Thread, current_thread
|
from threading import Thread, current_thread
|
||||||
|
|
||||||
from . import BaseTestPostgresql, MockCursor, MockPostmaster, psycopg2_connect
|
from . import BaseTestPostgresql, MockCursor, MockPostmaster, psycopg_connect
|
||||||
|
|
||||||
|
|
||||||
mtime_ret = {}
|
mtime_ret = {}
|
||||||
@@ -87,7 +89,7 @@ Data page checksum version: 0
|
|||||||
|
|
||||||
|
|
||||||
@patch('subprocess.call', Mock(return_value=0))
|
@patch('subprocess.call', Mock(return_value=0))
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
class TestPostgresql(BaseTestPostgresql):
|
class TestPostgresql(BaseTestPostgresql):
|
||||||
|
|
||||||
@patch('subprocess.call', Mock(return_value=0))
|
@patch('subprocess.call', Mock(return_value=0))
|
||||||
@@ -203,6 +205,21 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
mock_postmaster.signal_stop.side_effect = [None, True]
|
mock_postmaster.signal_stop.side_effect = [None, True]
|
||||||
self.assertTrue(self.p.stop(on_safepoint=mock_callback, stop_timeout=30))
|
self.assertTrue(self.p.stop(on_safepoint=mock_callback, stop_timeout=30))
|
||||||
|
|
||||||
|
@patch('time.sleep', Mock())
|
||||||
|
@patch.object(Postgresql, 'is_running', MockPostmaster)
|
||||||
|
@patch.object(Postgresql, '_wait_for_connection_close', Mock())
|
||||||
|
@patch.object(Postgresql, 'latest_checkpoint_location', Mock(return_value='7'))
|
||||||
|
def test__do_stop(self):
|
||||||
|
mock_callback = Mock()
|
||||||
|
with patch.object(Postgresql, 'controldata', Mock(return_value={'Database cluster state': 'shut down'})):
|
||||||
|
self.assertTrue(self.p.stop(on_shutdown=mock_callback, stop_timeout=3))
|
||||||
|
mock_callback.assert_called()
|
||||||
|
with patch.object(Postgresql, 'controldata',
|
||||||
|
Mock(return_value={'Database cluster state': 'shut down in recovery'})):
|
||||||
|
self.assertTrue(self.p.stop(on_shutdown=mock_callback, stop_timeout=3))
|
||||||
|
with patch.object(Postgresql, 'controldata', Mock(return_value={'Database cluster state': 'shutting down'})):
|
||||||
|
self.assertTrue(self.p.stop(on_shutdown=mock_callback, stop_timeout=3))
|
||||||
|
|
||||||
def test_restart(self):
|
def test_restart(self):
|
||||||
self.p.start = Mock(return_value=False)
|
self.p.start = Mock(return_value=False)
|
||||||
self.assertFalse(self.p.restart())
|
self.assertFalse(self.p.restart())
|
||||||
@@ -224,6 +241,7 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
@patch('patroni.postgresql.config.mtime', mock_mtime)
|
@patch('patroni.postgresql.config.mtime', mock_mtime)
|
||||||
@patch('patroni.postgresql.config.ConfigHandler._get_pg_settings')
|
@patch('patroni.postgresql.config.ConfigHandler._get_pg_settings')
|
||||||
def test_check_recovery_conf(self, mock_get_pg_settings):
|
def test_check_recovery_conf(self, mock_get_pg_settings):
|
||||||
|
self.p.call_nowait('on_start')
|
||||||
mock_get_pg_settings.return_value = {
|
mock_get_pg_settings.return_value = {
|
||||||
'primary_conninfo': ['primary_conninfo', 'foo=', None, 'string', 'postmaster', self.p.config._auto_conf],
|
'primary_conninfo': ['primary_conninfo', 'foo=', None, 'string', 'postmaster', self.p.config._auto_conf],
|
||||||
'recovery_min_apply_delay': ['recovery_min_apply_delay', '0', 'ms', 'integer', 'sighup', 'foo']
|
'recovery_min_apply_delay': ['recovery_min_apply_delay', '0', 'ms', 'integer', 'sighup', 'foo']
|
||||||
@@ -242,8 +260,8 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
with patch('patroni.postgresql.config.ConfigHandler.primary_conninfo_params', Mock(return_value=conninfo)):
|
with patch('patroni.postgresql.config.ConfigHandler.primary_conninfo_params', Mock(return_value=conninfo)):
|
||||||
mock_get_pg_settings.return_value['recovery_min_apply_delay'][1] = '1'
|
mock_get_pg_settings.return_value['recovery_min_apply_delay'][1] = '1'
|
||||||
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
||||||
mock_get_pg_settings.return_value['primary_conninfo'][1] = 'host=1 passfile='\
|
mock_get_pg_settings.return_value['primary_conninfo'][1] = 'host=1 target_session_attrs=read-write'\
|
||||||
+ re.sub(r'([\'\\ ])', r'\\\1', self.p.config._pgpass)
|
+ ' passfile=' + re.sub(r'([\'\\ ])', r'\\\1', self.p.config._pgpass)
|
||||||
mock_get_pg_settings.return_value['recovery_min_apply_delay'][1] = '0'
|
mock_get_pg_settings.return_value['recovery_min_apply_delay'][1] = '0'
|
||||||
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
||||||
self.p.config.write_recovery_conf({'standby_mode': 'on', 'primary_conninfo': conninfo.copy()})
|
self.p.config.write_recovery_conf({'standby_mode': 'on', 'primary_conninfo': conninfo.copy()})
|
||||||
@@ -259,6 +277,7 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
@patch.object(MockPostmaster, 'create_time', Mock(return_value=1234567), create=True)
|
@patch.object(MockPostmaster, 'create_time', Mock(return_value=1234567), create=True)
|
||||||
@patch('patroni.postgresql.config.ConfigHandler._get_pg_settings')
|
@patch('patroni.postgresql.config.ConfigHandler._get_pg_settings')
|
||||||
def test__read_recovery_params(self, mock_get_pg_settings):
|
def test__read_recovery_params(self, mock_get_pg_settings):
|
||||||
|
self.p.call_nowait('on_start')
|
||||||
mock_get_pg_settings.return_value = {'primary_conninfo': ['primary_conninfo', '', None, 'string',
|
mock_get_pg_settings.return_value = {'primary_conninfo': ['primary_conninfo', '', None, 'string',
|
||||||
'postmaster', self.p.config._postgresql_conf]}
|
'postmaster', self.p.config._postgresql_conf]}
|
||||||
self.p.config.write_recovery_conf({'standby_mode': 'on', 'primary_conninfo': {'password': 'foo'}})
|
self.p.config.write_recovery_conf({'standby_mode': 'on', 'primary_conninfo': {'password': 'foo'}})
|
||||||
@@ -302,30 +321,7 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
m = RemoteMember('1', {'restore_command': '2', 'primary_slot_name': 'foo', 'conn_kwargs': {'host': 'bar'}})
|
m = RemoteMember('1', {'restore_command': '2', 'primary_slot_name': 'foo', 'conn_kwargs': {'host': 'bar'}})
|
||||||
self.p.follow(m)
|
self.p.follow(m)
|
||||||
|
|
||||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
@patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError))
|
||||||
def test_sync_replication_slots(self):
|
|
||||||
self.p.start()
|
|
||||||
config = ClusterConfig(1, {'slots': {'test_3': {'database': 'a', 'plugin': 'b'},
|
|
||||||
'A': 0, 'ls': 0, 'b': {'type': 'logical', 'plugin': '1'}},
|
|
||||||
'ignore_slots': [{'name': 'blabla'}]}, 1)
|
|
||||||
cluster = Cluster(True, config, self.leader, 0, [self.me, self.other, self.leadermem], None, None, None)
|
|
||||||
with mock.patch('patroni.postgresql.Postgresql._query', Mock(side_effect=psycopg2.OperationalError)):
|
|
||||||
self.p.slots_handler.sync_replication_slots(cluster)
|
|
||||||
self.p.slots_handler.sync_replication_slots(cluster)
|
|
||||||
with mock.patch('patroni.postgresql.Postgresql.role', new_callable=PropertyMock(return_value='replica')):
|
|
||||||
self.p.slots_handler.sync_replication_slots(cluster)
|
|
||||||
with patch.object(SlotsHandler, 'drop_replication_slot', Mock(return_value=True)),\
|
|
||||||
patch('patroni.dcs.logger.error', new_callable=Mock()) as errorlog_mock:
|
|
||||||
alias1 = Member(0, 'test-3', 28, {'conn_url': 'postgres://replicator:[email protected]:5436/postgres'})
|
|
||||||
alias2 = Member(0, 'test.3', 28, {'conn_url': 'postgres://replicator:[email protected]:5436/postgres'})
|
|
||||||
cluster.members.extend([alias1, alias2])
|
|
||||||
self.p.slots_handler.sync_replication_slots(cluster)
|
|
||||||
self.assertEqual(errorlog_mock.call_count, 5)
|
|
||||||
ca = errorlog_mock.call_args_list[0][0][1]
|
|
||||||
self.assertTrue("test-3" in ca, "non matching {0}".format(ca))
|
|
||||||
self.assertTrue("test.3" in ca, "non matching {0}".format(ca))
|
|
||||||
|
|
||||||
@patch.object(MockCursor, 'execute', Mock(side_effect=psycopg2.OperationalError))
|
|
||||||
def test__query(self):
|
def test__query(self):
|
||||||
self.assertRaises(PostgresConnectionException, self.p._query, 'blabla')
|
self.assertRaises(PostgresConnectionException, self.p._query, 'blabla')
|
||||||
self.p._state = 'restarting'
|
self.p._state = 'restarting'
|
||||||
@@ -334,19 +330,46 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
def test_query(self):
|
def test_query(self):
|
||||||
self.p.query('select 1')
|
self.p.query('select 1')
|
||||||
self.assertRaises(PostgresConnectionException, self.p.query, 'RetryFailedError')
|
self.assertRaises(PostgresConnectionException, self.p.query, 'RetryFailedError')
|
||||||
self.assertRaises(psycopg2.ProgrammingError, self.p.query, 'blabla')
|
self.assertRaises(psycopg.ProgrammingError, self.p.query, 'blabla')
|
||||||
|
|
||||||
@patch.object(Postgresql, 'pg_isready', Mock(return_value=STATE_REJECT))
|
@patch.object(Postgresql, 'pg_isready', Mock(return_value=STATE_REJECT))
|
||||||
def test_is_leader(self):
|
def test_is_leader(self):
|
||||||
self.assertTrue(self.p.is_leader())
|
self.assertTrue(self.p.is_leader())
|
||||||
self.p.reset_cluster_info_state()
|
self.p.reset_cluster_info_state(None)
|
||||||
with patch.object(Postgresql, '_query', Mock(side_effect=RetryFailedError(''))):
|
with patch.object(Postgresql, '_query', Mock(side_effect=RetryFailedError(''))):
|
||||||
self.assertRaises(PostgresConnectionException, self.p.is_leader)
|
self.assertFalse(self.p.is_leader())
|
||||||
|
|
||||||
@patch.object(Postgresql, 'controldata',
|
@patch.object(Postgresql, 'controldata', Mock(return_value={'Database cluster state': 'shut down',
|
||||||
Mock(return_value={'Database cluster state': 'shut down', 'Latest checkpoint location': 'X/678'}))
|
'Latest checkpoint location': '0/1ADBC18',
|
||||||
def test_latest_checkpoint_location(self):
|
"Latest checkpoint's TimeLineID": '1'}))
|
||||||
self.assertIsNone(self.p.latest_checkpoint_location())
|
@patch('subprocess.Popen')
|
||||||
|
def test_latest_checkpoint_location(self, mock_popen):
|
||||||
|
mock_popen.return_value.communicate.return_value = (None, None)
|
||||||
|
self.assertEqual(self.p.latest_checkpoint_location(), '28163096')
|
||||||
|
# 9.3 and 9.4 format
|
||||||
|
mock_popen.return_value.communicate.side_effect = [
|
||||||
|
(b'rmgr: XLOG len (rec/tot): 72/ 104, tx: 0, lsn: 0/01ADBC18, prev 0/01ADBBB8, ' +
|
||||||
|
b'bkp: 0000, desc: checkpoint: redo 0/1ADBC18; tli 1; prev tli 1; fpw true; xid 0/727; oid 16386; multi' +
|
||||||
|
b' 1; offset 0; oldest xid 715 in DB 1; oldest multi 1 in DB 1; oldest running xid 0; shutdown', None),
|
||||||
|
(b'rmgr: Transaction len (rec/tot): 64/ 96, tx: 726, lsn: 0/01ADBBB8, prev 0/01ADBB70, ' +
|
||||||
|
b'bkp: 0000, desc: commit: 2021-02-26 11:19:37.900918 CET; inval msgs: catcache 11 catcache 10', None)]
|
||||||
|
self.assertEqual(self.p.latest_checkpoint_location(), '28163096')
|
||||||
|
mock_popen.return_value.communicate.side_effect = [
|
||||||
|
(b'rmgr: XLOG len (rec/tot): 72/ 104, tx: 0, lsn: 0/01ADBC18, prev 0/01ADBBB8, ' +
|
||||||
|
b'bkp: 0000, desc: checkpoint: redo 0/1ADBC18; tli 1; prev tli 1; fpw true; xid 0/727; oid 16386; multi' +
|
||||||
|
b' 1; offset 0; oldest xid 715 in DB 1; oldest multi 1 in DB 1; oldest running xid 0; shutdown', None),
|
||||||
|
(b'rmgr: XLOG len (rec/tot): 0/ 32, tx: 0, lsn: 0/01ADBBB8, prev 0/01ADBBA0, ' +
|
||||||
|
b'bkp: 0000, desc: xlog switch ', None)]
|
||||||
|
self.assertEqual(self.p.latest_checkpoint_location(), '28163000')
|
||||||
|
# 9.5+ format
|
||||||
|
mock_popen.return_value.communicate.side_effect = [
|
||||||
|
(b'rmgr: XLOG len (rec/tot): 114/ 114, tx: 0, lsn: 0/01ADBC18, prev 0/018260F8, ' +
|
||||||
|
b'desc: CHECKPOINT_SHUTDOWN redo 0/1825ED8; tli 1; prev tli 1; fpw true; xid 0:494; oid 16387; multi 1' +
|
||||||
|
b'; offset 0; oldest xid 479 in DB 1; oldest multi 1 in DB 1; oldest/newest commit timestamp xid: 0/0;' +
|
||||||
|
b' oldest running xid 0; shutdown', None),
|
||||||
|
(b'rmgr: XLOG len (rec/tot): 24/ 24, tx: 0, lsn: 0/018260F8, prev 0/01826080, ' +
|
||||||
|
b'desc: SWITCH ', None)]
|
||||||
|
self.assertEqual(self.p.latest_checkpoint_location(), '25321720')
|
||||||
|
|
||||||
def test_reload(self):
|
def test_reload(self):
|
||||||
self.assertTrue(self.p.reload())
|
self.assertTrue(self.p.reload())
|
||||||
@@ -409,7 +432,7 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
@patch.object(Postgresql, 'is_running', Mock(return_value=MockPostmaster()))
|
@patch.object(Postgresql, 'is_running', Mock(return_value=MockPostmaster()))
|
||||||
def test_is_leader_exception(self):
|
def test_is_leader_exception(self):
|
||||||
self.p.start()
|
self.p.start()
|
||||||
self.p.query = Mock(side_effect=psycopg2.OperationalError("not supported"))
|
self.p.query = Mock(side_effect=psycopg.OperationalError("not supported"))
|
||||||
self.assertTrue(self.p.stop())
|
self.assertTrue(self.p.stop())
|
||||||
|
|
||||||
@patch('os.rename', Mock())
|
@patch('os.rename', Mock())
|
||||||
@@ -498,7 +521,8 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
parameters = self._PARAMETERS.copy()
|
parameters = self._PARAMETERS.copy()
|
||||||
parameters.pop('f.oo')
|
parameters.pop('f.oo')
|
||||||
parameters['wal_buffers'] = '512'
|
parameters['wal_buffers'] = '512'
|
||||||
config = {'pg_hba': [''], 'pg_ident': [''], 'use_unix_socket': True, 'authentication': {},
|
config = {'pg_hba': [''], 'pg_ident': [''], 'use_unix_socket': True, 'use_unix_socket_repl': True,
|
||||||
|
'authentication': {},
|
||||||
'retry_timeout': 10, 'listen': '*', 'krbsrvname': 'postgres', 'parameters': parameters}
|
'retry_timeout': 10, 'listen': '*', 'krbsrvname': 'postgres', 'parameters': parameters}
|
||||||
self.p.reload_config(config)
|
self.p.reload_config(config)
|
||||||
mock_fetchone.side_effect = Exception
|
mock_fetchone.side_effect = Exception
|
||||||
@@ -514,6 +538,14 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
self.p.reload_config(config)
|
self.p.reload_config(config)
|
||||||
self.p.config.resolve_connection_addresses()
|
self.p.config.resolve_connection_addresses()
|
||||||
|
|
||||||
|
def test_resolve_connection_addresses(self):
|
||||||
|
self.p.config._config['use_unix_socket'] = self.p.config._config['use_unix_socket_repl'] = True
|
||||||
|
self.p.config.resolve_connection_addresses()
|
||||||
|
self.assertEqual(self.p.config.local_replication_address, {'host': '/tmp', 'port': '5432'})
|
||||||
|
self.p.config._server_parameters.pop('unix_socket_directories')
|
||||||
|
self.p.config.resolve_connection_addresses()
|
||||||
|
self.assertEqual(self.p.config._local_address, {'port': '5432'})
|
||||||
|
|
||||||
@patch.object(Postgresql, '_version_file_exists', Mock(return_value=True))
|
@patch.object(Postgresql, '_version_file_exists', Mock(return_value=True))
|
||||||
def test_get_major_version(self):
|
def test_get_major_version(self):
|
||||||
with patch.object(builtins, 'open', mock_open(read_data='9.4')):
|
with patch.object(builtins, 'open', mock_open(read_data='9.4')):
|
||||||
@@ -522,13 +554,14 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
self.assertEqual(self.p.get_major_version(), 0)
|
self.assertEqual(self.p.get_major_version(), 0)
|
||||||
|
|
||||||
def test_postmaster_start_time(self):
|
def test_postmaster_start_time(self):
|
||||||
with patch.object(MockCursor, "fetchone", Mock(return_value=('foo', True, '', '', '', '', False))):
|
now = datetime.datetime.now()
|
||||||
self.assertEqual(self.p.postmaster_start_time(), 'foo')
|
with patch.object(MockCursor, "fetchone", Mock(return_value=(now, True, '', '', '', '', False))):
|
||||||
|
self.assertEqual(self.p.postmaster_start_time(), now.isoformat(sep=' '))
|
||||||
t = Thread(target=self.p.postmaster_start_time)
|
t = Thread(target=self.p.postmaster_start_time)
|
||||||
t.start()
|
t.start()
|
||||||
t.join()
|
t.join()
|
||||||
|
|
||||||
with patch.object(MockCursor, "execute", side_effect=psycopg2.Error):
|
with patch.object(MockCursor, "execute", side_effect=psycopg.Error):
|
||||||
self.assertIsNone(self.p.postmaster_start_time())
|
self.assertIsNone(self.p.postmaster_start_time())
|
||||||
|
|
||||||
def test_check_for_startup(self):
|
def test_check_for_startup(self):
|
||||||
@@ -609,7 +642,7 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
|
|
||||||
def test_pick_sync_standby(self):
|
def test_pick_sync_standby(self):
|
||||||
cluster = Cluster(True, None, self.leader, 0, [self.me, self.other, self.leadermem], None,
|
cluster = Cluster(True, None, self.leader, 0, [self.me, self.other, self.leadermem], None,
|
||||||
SyncState(0, self.me.name, self.leadermem.name), None)
|
SyncState(0, self.me.name, self.leadermem.name), None, None)
|
||||||
mock_cursor = Mock()
|
mock_cursor = Mock()
|
||||||
mock_cursor.fetchone.return_value = ('remote_apply',)
|
mock_cursor.fetchone.return_value = ('remote_apply',)
|
||||||
|
|
||||||
@@ -691,7 +724,7 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
self.p.stop(on_safepoint=mock_callback)
|
self.p.stop(on_safepoint=mock_callback)
|
||||||
|
|
||||||
mock_postmaster.is_running.side_effect = [True, False, False]
|
mock_postmaster.is_running.side_effect = [True, False, False]
|
||||||
with patch.object(MockCursor, "execute", Mock(side_effect=psycopg2.Error)):
|
with patch.object(MockCursor, "execute", Mock(side_effect=psycopg.Error)):
|
||||||
self.p.stop(on_safepoint=mock_callback)
|
self.p.stop(on_safepoint=mock_callback)
|
||||||
|
|
||||||
def test_terminate_starting_postmaster(self):
|
def test_terminate_starting_postmaster(self):
|
||||||
@@ -707,6 +740,8 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
self.assertEqual(self.p.get_master_timeline(), 1)
|
self.assertEqual(self.p.get_master_timeline(), 1)
|
||||||
|
|
||||||
@patch.object(Postgresql, 'get_postgres_role_from_data_directory', Mock(return_value='replica'))
|
@patch.object(Postgresql, 'get_postgres_role_from_data_directory', Mock(return_value='replica'))
|
||||||
|
@patch.object(Bootstrap, 'running_custom_bootstrap', PropertyMock(return_value=True))
|
||||||
|
@patch.object(Bootstrap, 'keep_existing_recovery_conf', PropertyMock(return_value=True))
|
||||||
def test__build_effective_configuration(self):
|
def test__build_effective_configuration(self):
|
||||||
with patch.object(Postgresql, 'controldata',
|
with patch.object(Postgresql, 'controldata',
|
||||||
Mock(return_value={'max_connections setting': '200',
|
Mock(return_value={'max_connections setting': '200',
|
||||||
@@ -725,10 +760,19 @@ class TestPostgresql(BaseTestPostgresql):
|
|||||||
@patch.object(Postgresql, '_query', Mock(side_effect=RetryFailedError('')))
|
@patch.object(Postgresql, '_query', Mock(side_effect=RetryFailedError('')))
|
||||||
def test_received_timeline(self):
|
def test_received_timeline(self):
|
||||||
self.p.set_role('standby_leader')
|
self.p.set_role('standby_leader')
|
||||||
self.p.reset_cluster_info_state()
|
self.p.reset_cluster_info_state(None)
|
||||||
self.assertRaises(PostgresConnectionException, self.p.received_timeline)
|
self.assertRaises(PostgresConnectionException, self.p.received_timeline)
|
||||||
|
|
||||||
def test__write_recovery_params(self):
|
def test__write_recovery_params(self):
|
||||||
self.p.config._write_recovery_params(Mock(), {'pause_at_recovery_target': 'false'})
|
self.p.config._write_recovery_params(Mock(), {'pause_at_recovery_target': 'false'})
|
||||||
with patch.object(Postgresql, 'major_version', PropertyMock(return_value=90400)):
|
with patch.object(Postgresql, 'major_version', PropertyMock(return_value=90400)):
|
||||||
self.p.config._write_recovery_params(Mock(), {'recovery_target_action': 'PROMOTE'})
|
self.p.config._write_recovery_params(Mock(), {'recovery_target_action': 'PROMOTE'})
|
||||||
|
|
||||||
|
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||||
|
def test_set_enforce_hot_standby_feedback(self):
|
||||||
|
self.p.set_enforce_hot_standby_feedback(True)
|
||||||
|
|
||||||
|
@patch.object(Postgresql, 'major_version', PropertyMock(return_value=140000))
|
||||||
|
@patch.object(Postgresql, '_cluster_info_state_get', Mock(return_value=True))
|
||||||
|
def test_handle_parameter_change(self):
|
||||||
|
self.p.handle_parameter_change()
|
||||||
|
|||||||
@@ -73,7 +73,7 @@ class TestPostmasterProcess(unittest.TestCase):
|
|||||||
|
|
||||||
# all processes successfully stopped
|
# all processes successfully stopped
|
||||||
mock_children.return_value = [Mock()]
|
mock_children.return_value = [Mock()]
|
||||||
mock_children.return_value[0].kill.side_effect = psutil.Error
|
mock_children.return_value[0].kill.side_effect = psutil.NoSuchProcess(123)
|
||||||
self.assertTrue(proc.signal_kill())
|
self.assertTrue(proc.signal_kill())
|
||||||
|
|
||||||
# postmaster has gone before suspend
|
# postmaster has gone before suspend
|
||||||
@@ -81,17 +81,17 @@ class TestPostmasterProcess(unittest.TestCase):
|
|||||||
self.assertTrue(proc.signal_kill())
|
self.assertTrue(proc.signal_kill())
|
||||||
|
|
||||||
# postmaster has gone before we got a list of children
|
# postmaster has gone before we got a list of children
|
||||||
mock_suspend.side_effect = psutil.Error()
|
mock_suspend.side_effect = psutil.AccessDenied()
|
||||||
mock_children.side_effect = psutil.NoSuchProcess(123)
|
mock_children.side_effect = psutil.NoSuchProcess(123)
|
||||||
self.assertTrue(proc.signal_kill())
|
self.assertTrue(proc.signal_kill())
|
||||||
|
|
||||||
# postmaster has gone after we got a list of children
|
# postmaster has gone after we got a list of children
|
||||||
mock_children.side_effect = psutil.Error()
|
mock_children.side_effect = psutil.AccessDenied()
|
||||||
mock_kill.side_effect = psutil.NoSuchProcess(123)
|
mock_kill.side_effect = psutil.NoSuchProcess(123)
|
||||||
self.assertTrue(proc.signal_kill())
|
self.assertTrue(proc.signal_kill())
|
||||||
|
|
||||||
# failed to kill postmaster
|
# failed to kill postmaster
|
||||||
mock_kill.side_effect = psutil.AccessDenied(123)
|
mock_kill.side_effect = psutil.AccessDenied()
|
||||||
self.assertFalse(proc.signal_kill())
|
self.assertFalse(proc.signal_kill())
|
||||||
|
|
||||||
@patch('psutil.Process.__init__', Mock())
|
@patch('psutil.Process.__init__', Mock())
|
||||||
|
|||||||
+37
-37
@@ -3,13 +3,13 @@ import unittest
|
|||||||
import tempfile
|
import tempfile
|
||||||
import time
|
import time
|
||||||
|
|
||||||
from mock import Mock, patch
|
from mock import Mock, PropertyMock, patch
|
||||||
from patroni.dcs.raft import DynMemberSyncObj, KVStoreTTL, Raft, SyncObjUtility
|
from patroni.dcs.raft import DynMemberSyncObj, KVStoreTTL, Raft, SyncObjUtility, TCPTransport, _TCPTransport
|
||||||
from pysyncobj import SyncObjConf, FAIL_REASON
|
from pysyncobj import SyncObjConf, FAIL_REASON
|
||||||
|
|
||||||
|
|
||||||
def remove_files(prefix):
|
def remove_files(prefix):
|
||||||
for f in ('journal', 'dump'):
|
for f in ('journal', 'journal.meta', 'dump'):
|
||||||
f = prefix + f
|
f = prefix + f
|
||||||
if os.path.isfile(f):
|
if os.path.isfile(f):
|
||||||
for i in range(0, 15):
|
for i in range(0, 15):
|
||||||
@@ -23,6 +23,16 @@ def remove_files(prefix):
|
|||||||
time.sleep(1.0)
|
time.sleep(1.0)
|
||||||
|
|
||||||
|
|
||||||
|
class TestTCPTransport(unittest.TestCase):
|
||||||
|
|
||||||
|
@patch.object(TCPTransport, '__init__', Mock())
|
||||||
|
@patch.object(TCPTransport, 'setOnUtilityMessageCallback', Mock())
|
||||||
|
@patch.object(TCPTransport, '_connectIfNecessarySingle', Mock(side_effect=Exception))
|
||||||
|
def test__connectIfNecessarySingle(self):
|
||||||
|
t = _TCPTransport(Mock(), None, [])
|
||||||
|
self.assertFalse(t._connectIfNecessarySingle(None))
|
||||||
|
|
||||||
|
|
||||||
@patch('pysyncobj.tcp_server.TcpServer.bind', Mock())
|
@patch('pysyncobj.tcp_server.TcpServer.bind', Mock())
|
||||||
class TestDynMemberSyncObj(unittest.TestCase):
|
class TestDynMemberSyncObj(unittest.TestCase):
|
||||||
|
|
||||||
@@ -31,50 +41,38 @@ class TestDynMemberSyncObj(unittest.TestCase):
|
|||||||
self.conf = SyncObjConf(appendEntriesUseBatch=False, dynamicMembershipChange=True, autoTick=False)
|
self.conf = SyncObjConf(appendEntriesUseBatch=False, dynamicMembershipChange=True, autoTick=False)
|
||||||
self.so = DynMemberSyncObj('127.0.0.1:1234', ['127.0.0.1:1235'], self.conf)
|
self.so = DynMemberSyncObj('127.0.0.1:1234', ['127.0.0.1:1235'], self.conf)
|
||||||
|
|
||||||
@patch.object(SyncObjUtility, 'sendMessage')
|
@patch.object(SyncObjUtility, 'executeCommand')
|
||||||
def test_add_member(self, mock_send_message):
|
def test_add_member(self, mock_execute_command):
|
||||||
mock_send_message.return_value = [{'addr': '127.0.0.1:1235'}, {'addr': '127.0.0.1:1236'}]
|
mock_execute_command.return_value = [{'addr': '127.0.0.1:1235'}, {'addr': '127.0.0.1:1236'}]
|
||||||
mock_send_message.ver = 0
|
mock_execute_command.ver = 0
|
||||||
DynMemberSyncObj('127.0.0.1:1234', ['127.0.0.1:1235'], self.conf)
|
DynMemberSyncObj('127.0.0.1:1234', ['127.0.0.1:1235'], self.conf)
|
||||||
self.conf.dynamicMembershipChange = False
|
self.conf.dynamicMembershipChange = False
|
||||||
DynMemberSyncObj('127.0.0.1:1234', ['127.0.0.1:1235'], self.conf)
|
DynMemberSyncObj('127.0.0.1:1234', ['127.0.0.1:1235'], self.conf)
|
||||||
|
|
||||||
def test___onUtilityMessage(self):
|
def test_getMembers(self):
|
||||||
self.so._SyncObj__encryptor = Mock()
|
|
||||||
mock_conn = Mock()
|
mock_conn = Mock()
|
||||||
mock_conn.sendRandKey = None
|
|
||||||
self.so._SyncObj__transport._onIncomingMessageReceived(mock_conn, 'randkey')
|
|
||||||
self.so._SyncObj__transport._onIncomingMessageReceived(mock_conn, ['members'])
|
self.so._SyncObj__transport._onIncomingMessageReceived(mock_conn, ['members'])
|
||||||
self.so._SyncObj__transport._onIncomingMessageReceived(mock_conn, ['status'])
|
|
||||||
|
|
||||||
def test__SyncObj__doChangeCluster(self):
|
def test__SyncObj__doChangeCluster(self):
|
||||||
self.so._SyncObj__doChangeCluster(['add', '127.0.0.1:1236'])
|
self.so._SyncObj__doChangeCluster(['add', '127.0.0.1:1236'])
|
||||||
|
|
||||||
def test_utility(self):
|
|
||||||
utility = SyncObjUtility(['127.0.0.1:1235'], self.conf)
|
|
||||||
utility.setPartnerNode(list(utility._SyncObj__otherNodes)[0])
|
|
||||||
utility.sendMessage(['members'])
|
|
||||||
utility._onMessageReceived(0, '')
|
|
||||||
|
|
||||||
|
|
||||||
|
@patch.object(SyncObjConf, 'fullDumpFile', PropertyMock(return_value=None), create=True)
|
||||||
|
@patch.object(SyncObjConf, 'journalFile', PropertyMock(return_value=None), create=True)
|
||||||
class TestKVStoreTTL(unittest.TestCase):
|
class TestKVStoreTTL(unittest.TestCase):
|
||||||
|
|
||||||
|
@patch.object(SyncObjConf, 'fullDumpFile', PropertyMock(return_value=None), create=True)
|
||||||
|
@patch.object(SyncObjConf, 'journalFile', PropertyMock(return_value=None), create=True)
|
||||||
def setUp(self):
|
def setUp(self):
|
||||||
self.conf = SyncObjConf(appendEntriesUseBatch=False, appendEntriesPeriod=0.001,
|
|
||||||
raftMinTimeout=0.004, raftMaxTimeout=0.005, autoTickPeriod=0.001)
|
|
||||||
callback = Mock()
|
callback = Mock()
|
||||||
callback.replicated = False
|
callback.replicated = False
|
||||||
self.so = KVStoreTTL('127.0.0.1:1234', [], self.conf, on_set=callback, on_delete=callback)
|
self.so = KVStoreTTL(None, callback, callback, self_addr='127.0.0.1:1234')
|
||||||
|
self.so.startAutoTick()
|
||||||
self.so.set_retry_timeout(10)
|
self.so.set_retry_timeout(10)
|
||||||
|
|
||||||
@staticmethod
|
|
||||||
def destroy(so):
|
|
||||||
so.destroy()
|
|
||||||
so._SyncObj__thread.join()
|
|
||||||
|
|
||||||
def tearDown(self):
|
def tearDown(self):
|
||||||
if self.so:
|
if self.so:
|
||||||
self.destroy(self.so)
|
self.so.destroy()
|
||||||
|
|
||||||
def test_set(self):
|
def test_set(self):
|
||||||
self.assertTrue(self.so.set('foo', 'bar', prevExist=False, ttl=30))
|
self.assertTrue(self.so.set('foo', 'bar', prevExist=False, ttl=30))
|
||||||
@@ -83,7 +81,7 @@ class TestKVStoreTTL(unittest.TestCase):
|
|||||||
self.assertTrue(self.so.retry(self.so._set, 'foo', {'value': 'buz', 'created': 1, 'updated': 1}))
|
self.assertTrue(self.so.retry(self.so._set, 'foo', {'value': 'buz', 'created': 1, 'updated': 1}))
|
||||||
|
|
||||||
def test_delete(self):
|
def test_delete(self):
|
||||||
self.conf.autoTickPeriod = 0.1
|
self.so.autoTickPeriod = 0.2
|
||||||
self.so.set('foo', 'bar')
|
self.so.set('foo', 'bar')
|
||||||
self.so.set('fooo', 'bar')
|
self.so.set('fooo', 'bar')
|
||||||
self.assertFalse(self.so.delete('foo', prevValue='buz'))
|
self.assertFalse(self.so.delete('foo', prevValue='buz'))
|
||||||
@@ -111,11 +109,10 @@ class TestKVStoreTTL(unittest.TestCase):
|
|||||||
|
|
||||||
def test_on_ready_override(self):
|
def test_on_ready_override(self):
|
||||||
self.assertTrue(self.so.set('foo', 'bar'))
|
self.assertTrue(self.so.set('foo', 'bar'))
|
||||||
self.destroy(self.so)
|
self.so.destroy()
|
||||||
self.so = None
|
self.so = None
|
||||||
self.conf.onReady = Mock()
|
so = KVStoreTTL(Mock(), None, None, self_addr='127.0.0.1:1234',
|
||||||
self.conf.autoTick = False
|
partner_addrs=['127.0.0.1:1235'], patronictl=True)
|
||||||
so = KVStoreTTL('127.0.0.1:1234', ['127.0.0.1:1235'], self.conf)
|
|
||||||
so.doTick(0)
|
so.doTick(0)
|
||||||
so.destroy()
|
so.destroy()
|
||||||
|
|
||||||
@@ -127,16 +124,20 @@ class TestRaft(unittest.TestCase):
|
|||||||
def test_raft(self):
|
def test_raft(self):
|
||||||
raft = Raft({'ttl': 30, 'scope': 'test', 'name': 'pg', 'self_addr': '127.0.0.1:1234',
|
raft = Raft({'ttl': 30, 'scope': 'test', 'name': 'pg', 'self_addr': '127.0.0.1:1234',
|
||||||
'retry_timeout': 10, 'data_dir': self._TMP})
|
'retry_timeout': 10, 'data_dir': self._TMP})
|
||||||
raft.set_retry_timeout(20)
|
raft.reload_config({'retry_timeout': 20, 'ttl': 60, 'loop_wait': 10})
|
||||||
raft.set_ttl(60)
|
self.assertTrue(raft._sync_obj.set(raft.members_path + 'legacy', '{"version":"2.0.0"}'))
|
||||||
self.assertTrue(raft.touch_member(''))
|
self.assertTrue(raft.touch_member(''))
|
||||||
self.assertTrue(raft.initialize())
|
self.assertTrue(raft.initialize())
|
||||||
self.assertTrue(raft.cancel_initialization())
|
self.assertTrue(raft.cancel_initialization())
|
||||||
self.assertTrue(raft.set_config_value('{}'))
|
self.assertTrue(raft.set_config_value('{}'))
|
||||||
self.assertTrue(raft.write_sync_state('foo', 'bar'))
|
self.assertTrue(raft.write_sync_state('foo', 'bar'))
|
||||||
self.assertTrue(raft.update_leader('1'))
|
|
||||||
self.assertTrue(raft.manual_failover('foo', 'bar'))
|
self.assertTrue(raft.manual_failover('foo', 'bar'))
|
||||||
raft.get_cluster()
|
raft.get_cluster()
|
||||||
|
self.assertTrue(raft._sync_obj.set(raft.status_path, '{"optime":1234567,"slots":{"ls":12345}}'))
|
||||||
|
raft.get_cluster()
|
||||||
|
self.assertTrue(raft.update_leader('1'))
|
||||||
|
self.assertTrue(raft._sync_obj.set(raft.status_path, '{'))
|
||||||
|
raft.get_cluster()
|
||||||
self.assertTrue(raft.delete_sync_state())
|
self.assertTrue(raft.delete_sync_state())
|
||||||
self.assertTrue(raft.delete_leader())
|
self.assertTrue(raft.delete_leader())
|
||||||
self.assertTrue(raft.set_history_value(''))
|
self.assertTrue(raft.set_history_value(''))
|
||||||
@@ -145,7 +146,6 @@ class TestRaft(unittest.TestCase):
|
|||||||
self.assertTrue(raft.take_leader())
|
self.assertTrue(raft.take_leader())
|
||||||
raft.watch(None, 0.001)
|
raft.watch(None, 0.001)
|
||||||
raft._sync_obj.destroy()
|
raft._sync_obj.destroy()
|
||||||
raft._sync_obj._SyncObj__thread.join()
|
|
||||||
|
|
||||||
def tearDown(self):
|
def tearDown(self):
|
||||||
remove_files(os.path.join(self._TMP, '127.0.0.1:1234.'))
|
remove_files(os.path.join(self._TMP, '127.0.0.1:1234.'))
|
||||||
@@ -157,6 +157,6 @@ class TestRaft(unittest.TestCase):
|
|||||||
@patch('threading.Event')
|
@patch('threading.Event')
|
||||||
def test_init(self, mock_event, mock_kvstore):
|
def test_init(self, mock_event, mock_kvstore):
|
||||||
mock_kvstore.return_value.applied_local_log = False
|
mock_kvstore.return_value.applied_local_log = False
|
||||||
mock_event.return_value.isSet.side_effect = [False, True]
|
mock_event.return_value.is_set.side_effect = [False, True]
|
||||||
self.assertIsNotNone(Raft({'ttl': 30, 'scope': 'test', 'name': 'pg', 'patronictl': True,
|
self.assertIsNotNone(Raft({'ttl': 30, 'scope': 'test', 'name': 'pg', 'patronictl': True,
|
||||||
'self_addr': '1', 'data_dir': self._TMP}))
|
'self_addr': '1', 'data_dir': self._TMP}))
|
||||||
|
|||||||
+27
-15
@@ -5,7 +5,7 @@ from patroni.postgresql.cancellable import CancellableSubprocess
|
|||||||
from patroni.postgresql.rewind import Rewind
|
from patroni.postgresql.rewind import Rewind
|
||||||
from six.moves import builtins
|
from six.moves import builtins
|
||||||
|
|
||||||
from . import BaseTestPostgresql, MockCursor, psycopg2_connect
|
from . import BaseTestPostgresql, MockCursor, psycopg_connect
|
||||||
|
|
||||||
|
|
||||||
class MockThread(object):
|
class MockThread(object):
|
||||||
@@ -47,7 +47,7 @@ def mock_single_user_mode(self, communicate, options):
|
|||||||
|
|
||||||
|
|
||||||
@patch('subprocess.call', Mock(return_value=0))
|
@patch('subprocess.call', Mock(return_value=0))
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
class TestRewind(BaseTestPostgresql):
|
class TestRewind(BaseTestPostgresql):
|
||||||
|
|
||||||
def setUp(self):
|
def setUp(self):
|
||||||
@@ -93,7 +93,7 @@ class TestRewind(BaseTestPostgresql):
|
|||||||
self.r.rewind_or_reinitialize_needed_and_possible(self.leader)
|
self.r.rewind_or_reinitialize_needed_and_possible(self.leader)
|
||||||
|
|
||||||
with patch.object(Postgresql, 'is_running', Mock(return_value=True)):
|
with patch.object(Postgresql, 'is_running', Mock(return_value=True)):
|
||||||
with patch.object(MockCursor, 'fetchone', Mock(side_effect=[(0, 0, 1, 1,), Exception])):
|
with patch.object(MockCursor, 'fetchone', Mock(side_effect=[(0, 0, 1, 1, 0, 0, 0, 0, 0, None), Exception])):
|
||||||
self.r.rewind_or_reinitialize_needed_and_possible(self.leader)
|
self.r.rewind_or_reinitialize_needed_and_possible(self.leader)
|
||||||
|
|
||||||
@patch.object(CancellableSubprocess, 'call', mock_cancellable_call)
|
@patch.object(CancellableSubprocess, 'call', mock_cancellable_call)
|
||||||
@@ -102,6 +102,11 @@ class TestRewind(BaseTestPostgresql):
|
|||||||
@patch.object(Postgresql, 'start', Mock())
|
@patch.object(Postgresql, 'start', Mock())
|
||||||
def test_execute(self, mock_checkpoint):
|
def test_execute(self, mock_checkpoint):
|
||||||
self.r.execute(self.leader)
|
self.r.execute(self.leader)
|
||||||
|
with patch.object(Postgresql, 'major_version', PropertyMock(return_value=130000)):
|
||||||
|
self.r.execute(self.leader)
|
||||||
|
with patch.object(MockCursor, 'fetchone', Mock(side_effect=Exception)):
|
||||||
|
self.r.execute(self.leader)
|
||||||
|
|
||||||
with patch.object(Rewind, 'pg_rewind', Mock(return_value=False)):
|
with patch.object(Rewind, 'pg_rewind', Mock(return_value=False)):
|
||||||
mock_checkpoint.side_effect = ['1', '', '', '']
|
mock_checkpoint.side_effect = ['1', '', '', '']
|
||||||
self.r.execute(self.leader)
|
self.r.execute(self.leader)
|
||||||
@@ -143,11 +148,16 @@ class TestRewind(BaseTestPostgresql):
|
|||||||
mock_check_leader_is_not_in_recovery.return_value = True
|
mock_check_leader_is_not_in_recovery.return_value = True
|
||||||
self.assertFalse(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
self.assertFalse(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
||||||
self.r.trigger_check_diverged_lsn()
|
self.r.trigger_check_diverged_lsn()
|
||||||
with patch('psycopg2.connect', Mock(side_effect=Exception)):
|
with patch.object(MockCursor, 'fetchone', Mock(side_effect=[('', 3, '0/0'), ('', b'4\t0/40159C0\tn\n')])):
|
||||||
|
self.assertTrue(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
||||||
|
self.r.reset_state()
|
||||||
|
self.r.trigger_check_diverged_lsn()
|
||||||
|
with patch('patroni.psycopg.connect', Mock(side_effect=Exception)):
|
||||||
self.assertFalse(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
self.assertFalse(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
||||||
self.r.trigger_check_diverged_lsn()
|
self.r.trigger_check_diverged_lsn()
|
||||||
with patch.object(MockCursor, 'fetchone', Mock(side_effect=[('', 3, '0/0'), ('', b'3\t0/40159C0\tn\n')])):
|
with patch.object(MockCursor, 'fetchone', Mock(side_effect=[('', 3, '0/0'), ('', b'1\t0/40159C0\tn\n')])):
|
||||||
self.assertFalse(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
self.assertTrue(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
||||||
|
self.r.reset_state()
|
||||||
self.r.trigger_check_diverged_lsn()
|
self.r.trigger_check_diverged_lsn()
|
||||||
with patch.object(MockCursor, 'fetchone', Mock(return_value=('', 1, '0/0'))):
|
with patch.object(MockCursor, 'fetchone', Mock(return_value=('', 1, '0/0'))):
|
||||||
with patch.object(Rewind, '_get_local_timeline_lsn', Mock(return_value=(True, 1, '0/0'))):
|
with patch.object(Rewind, '_get_local_timeline_lsn', Mock(return_value=(True, 1, '0/0'))):
|
||||||
@@ -219,18 +229,20 @@ class TestRewind(BaseTestPostgresql):
|
|||||||
@patch('patroni.postgresql.rewind.Thread', MockThread)
|
@patch('patroni.postgresql.rewind.Thread', MockThread)
|
||||||
@patch.object(Postgresql, 'controldata')
|
@patch.object(Postgresql, 'controldata')
|
||||||
@patch.object(Postgresql, 'checkpoint')
|
@patch.object(Postgresql, 'checkpoint')
|
||||||
def test_ensure_checkpoint_after_promote(self, mock_checkpoint, mock_controldata):
|
@patch.object(Postgresql, 'get_master_timeline')
|
||||||
mock_checkpoint.return_value = None
|
def test_ensure_checkpoint_after_promote(self, mock_get_master_timeline, mock_checkpoint, mock_controldata):
|
||||||
|
mock_controldata.return_value = {"Latest checkpoint's TimeLineID": 1}
|
||||||
|
mock_get_master_timeline.return_value = 1
|
||||||
|
self.r.ensure_checkpoint_after_promote(Mock())
|
||||||
|
|
||||||
|
self.r.reset_state()
|
||||||
|
mock_get_master_timeline.return_value = 2
|
||||||
|
mock_checkpoint.return_value = 0
|
||||||
self.r.ensure_checkpoint_after_promote(Mock())
|
self.r.ensure_checkpoint_after_promote(Mock())
|
||||||
self.r.ensure_checkpoint_after_promote(Mock())
|
self.r.ensure_checkpoint_after_promote(Mock())
|
||||||
|
|
||||||
self.r.reset_state()
|
self.r.reset_state()
|
||||||
mock_controldata.return_value = {"Latest checkpoint's TimeLineID": 1}
|
|
||||||
|
mock_controldata.side_effect = TypeError
|
||||||
mock_checkpoint.side_effect = Exception
|
mock_checkpoint.side_effect = Exception
|
||||||
self.r.ensure_checkpoint_after_promote(Mock())
|
self.r.ensure_checkpoint_after_promote(Mock())
|
||||||
self.r.ensure_checkpoint_after_promote(Mock())
|
|
||||||
|
|
||||||
self.r.reset_state()
|
|
||||||
mock_controldata.side_effect = TypeError
|
|
||||||
self.r.ensure_checkpoint_after_promote(Mock())
|
|
||||||
self.r.ensure_checkpoint_after_promote(Mock())
|
|
||||||
|
|||||||
@@ -0,0 +1,126 @@
|
|||||||
|
import mock
|
||||||
|
import os
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
|
||||||
|
from mock import Mock, PropertyMock, patch
|
||||||
|
|
||||||
|
from patroni import psycopg
|
||||||
|
from patroni.dcs import Cluster, ClusterConfig, Member
|
||||||
|
from patroni.postgresql import Postgresql
|
||||||
|
from patroni.postgresql.slots import SlotsHandler, fsync_dir
|
||||||
|
|
||||||
|
from . import BaseTestPostgresql, psycopg_connect, MockCursor
|
||||||
|
|
||||||
|
|
||||||
|
@patch('subprocess.call', Mock(return_value=0))
|
||||||
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
|
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||||
|
class TestSlotsHandler(BaseTestPostgresql):
|
||||||
|
|
||||||
|
@patch('subprocess.call', Mock(return_value=0))
|
||||||
|
@patch('os.rename', Mock())
|
||||||
|
@patch('patroni.postgresql.CallbackExecutor', Mock())
|
||||||
|
@patch.object(Postgresql, 'get_major_version', Mock(return_value=130000))
|
||||||
|
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||||
|
def setUp(self):
|
||||||
|
super(TestSlotsHandler, self).setUp()
|
||||||
|
self.s = self.p.slots_handler
|
||||||
|
self.p.start()
|
||||||
|
|
||||||
|
def test_sync_replication_slots(self):
|
||||||
|
config = ClusterConfig(1, {'slots': {'test_3': {'database': 'a', 'plugin': 'b'},
|
||||||
|
'A': 0, 'ls': 0, 'b': {'type': 'logical', 'plugin': '1'}},
|
||||||
|
'ignore_slots': [{'name': 'blabla'}]}, 1)
|
||||||
|
cluster = Cluster(True, config, self.leader, 0,
|
||||||
|
[self.me, self.other, self.leadermem], None, None, None, {'test_3': 10})
|
||||||
|
with mock.patch('patroni.postgresql.Postgresql._query', Mock(side_effect=psycopg.OperationalError)):
|
||||||
|
self.s.sync_replication_slots(cluster, False)
|
||||||
|
self.p.set_role('standby_leader')
|
||||||
|
self.s.sync_replication_slots(cluster, False)
|
||||||
|
self.p.set_role('replica')
|
||||||
|
with patch.object(Postgresql, 'is_leader', Mock(return_value=False)):
|
||||||
|
self.s.sync_replication_slots(cluster, False)
|
||||||
|
self.p.set_role('master')
|
||||||
|
with mock.patch('patroni.postgresql.Postgresql.role', new_callable=PropertyMock(return_value='replica')):
|
||||||
|
self.s.sync_replication_slots(cluster, False)
|
||||||
|
with patch.object(SlotsHandler, 'drop_replication_slot', Mock(return_value=True)),\
|
||||||
|
patch('patroni.dcs.logger.error', new_callable=Mock()) as errorlog_mock:
|
||||||
|
alias1 = Member(0, 'test-3', 28, {'conn_url': 'postgres://replicator:[email protected]:5436/postgres'})
|
||||||
|
alias2 = Member(0, 'test.3', 28, {'conn_url': 'postgres://replicator:[email protected]:5436/postgres'})
|
||||||
|
cluster.members.extend([alias1, alias2])
|
||||||
|
self.s.sync_replication_slots(cluster, False)
|
||||||
|
self.assertEqual(errorlog_mock.call_count, 5)
|
||||||
|
ca = errorlog_mock.call_args_list[0][0][1]
|
||||||
|
self.assertTrue("test-3" in ca, "non matching {0}".format(ca))
|
||||||
|
self.assertTrue("test.3" in ca, "non matching {0}".format(ca))
|
||||||
|
with patch.object(Postgresql, 'major_version', PropertyMock(return_value=90618)):
|
||||||
|
self.s.sync_replication_slots(cluster, False)
|
||||||
|
|
||||||
|
def test_process_permanent_slots(self):
|
||||||
|
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}},
|
||||||
|
'ignore_slots': [{'name': 'blabla'}]}, 1)
|
||||||
|
cluster = Cluster(True, config, self.leader, 0, [self.me, self.other, self.leadermem], None, None, None, None)
|
||||||
|
|
||||||
|
self.s.sync_replication_slots(cluster, False)
|
||||||
|
with patch.object(Postgresql, '_query') as mock_query:
|
||||||
|
self.p.reset_cluster_info_state(None)
|
||||||
|
mock_query.return_value.fetchone.return_value = (
|
||||||
|
1, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||||
|
[{"slot_name": "ls", "type": "logical", "datoid": 5, "plugin": "b",
|
||||||
|
"confirmed_flush_lsn": 12345, "catalog_xmin": 105}])
|
||||||
|
self.assertEqual(self.p.slots(), {'ls': 12345})
|
||||||
|
|
||||||
|
self.p.reset_cluster_info_state(None)
|
||||||
|
mock_query.return_value.fetchone.return_value = (
|
||||||
|
1, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||||
|
[{"slot_name": "ls", "type": "logical", "datoid": 6, "plugin": "b",
|
||||||
|
"confirmed_flush_lsn": 12345, "catalog_xmin": 105}])
|
||||||
|
self.assertEqual(self.p.slots(), {})
|
||||||
|
|
||||||
|
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||||
|
def test__ensure_logical_slots_replica(self):
|
||||||
|
self.p.set_role('replica')
|
||||||
|
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}}}, 1)
|
||||||
|
cluster = Cluster(True, config, self.leader, 0,
|
||||||
|
[self.me, self.other, self.leadermem], None, None, None, {'ls': 12346})
|
||||||
|
self.assertEqual(self.s.sync_replication_slots(cluster, False), [])
|
||||||
|
self.s._schedule_load_slots = False
|
||||||
|
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)),\
|
||||||
|
patch.object(psycopg.OperationalError, 'diag') as mock_diag:
|
||||||
|
type(mock_diag).sqlstate = PropertyMock(return_value='58P01')
|
||||||
|
self.assertEqual(self.s.sync_replication_slots(cluster, False), ['ls'])
|
||||||
|
cluster.slots['ls'] = 'a'
|
||||||
|
self.assertEqual(self.s.sync_replication_slots(cluster, False), [])
|
||||||
|
with patch.object(MockCursor, 'rowcount', PropertyMock(return_value=1), create=True):
|
||||||
|
self.assertEqual(self.s.sync_replication_slots(cluster, False), ['ls'])
|
||||||
|
|
||||||
|
@patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError))
|
||||||
|
def test_copy_logical_slots(self):
|
||||||
|
self.s.copy_logical_slots(self.leader, ['foo'])
|
||||||
|
|
||||||
|
@patch.object(Postgresql, 'stop', Mock(return_value=True))
|
||||||
|
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
||||||
|
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||||
|
def test_check_logical_slots_readiness(self):
|
||||||
|
self.s.copy_logical_slots(self.leader, ['ls'])
|
||||||
|
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}}}, 1)
|
||||||
|
cluster = Cluster(True, config, self.leader, 0,
|
||||||
|
[self.me, self.other, self.leadermem], None, None, None, {'ls': 12345})
|
||||||
|
self.assertEqual(self.s.sync_replication_slots(cluster, False), [])
|
||||||
|
with patch.object(MockCursor, 'rowcount', PropertyMock(return_value=1), create=True):
|
||||||
|
self.s.check_logical_slots_readiness(cluster, False, None)
|
||||||
|
|
||||||
|
@patch.object(Postgresql, 'stop', Mock(return_value=True))
|
||||||
|
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
||||||
|
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||||
|
def test_on_promote(self):
|
||||||
|
self.s.copy_logical_slots(self.leader, ['ls'])
|
||||||
|
self.s.on_promote()
|
||||||
|
|
||||||
|
@unittest.skipIf(os.name == 'nt', "Windows not supported")
|
||||||
|
@patch('os.open', Mock())
|
||||||
|
@patch('os.close', Mock())
|
||||||
|
@patch('os.fsync', Mock(side_effect=OSError))
|
||||||
|
def test_fsync_dir(self):
|
||||||
|
self.assertRaises(OSError, fsync_dir, 'foo')
|
||||||
@@ -1,14 +1,15 @@
|
|||||||
import psycopg2
|
|
||||||
import subprocess
|
import subprocess
|
||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
|
import patroni.psycopg as psycopg
|
||||||
|
|
||||||
from mock import Mock, PropertyMock, patch, mock_open
|
from mock import Mock, PropertyMock, patch, mock_open
|
||||||
from patroni.scripts import wale_restore
|
from patroni.scripts import wale_restore
|
||||||
from patroni.scripts.wale_restore import WALERestore, main as _main, get_major_version
|
from patroni.scripts.wale_restore import WALERestore, main as _main, get_major_version
|
||||||
from six.moves import builtins
|
from six.moves import builtins
|
||||||
from threading import current_thread
|
from threading import current_thread
|
||||||
|
|
||||||
from . import MockConnect, psycopg2_connect
|
from . import MockConnect, psycopg_connect
|
||||||
|
|
||||||
wale_output_header = (
|
wale_output_header = (
|
||||||
b'name\tlast_modified\t'
|
b'name\tlast_modified\t'
|
||||||
@@ -34,7 +35,7 @@ WALE_TEST_RETRIES = 2
|
|||||||
@patch('os.makedirs', Mock(return_value=True))
|
@patch('os.makedirs', Mock(return_value=True))
|
||||||
@patch('os.path.exists', Mock(return_value=True))
|
@patch('os.path.exists', Mock(return_value=True))
|
||||||
@patch('os.path.isdir', Mock(return_value=True))
|
@patch('os.path.isdir', Mock(return_value=True))
|
||||||
@patch('psycopg2.connect', psycopg2_connect)
|
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||||
@patch('subprocess.check_output', Mock(return_value=wale_output))
|
@patch('subprocess.check_output', Mock(return_value=wale_output))
|
||||||
class TestWALERestore(unittest.TestCase):
|
class TestWALERestore(unittest.TestCase):
|
||||||
|
|
||||||
@@ -57,7 +58,7 @@ class TestWALERestore(unittest.TestCase):
|
|||||||
with patch('subprocess.check_output', Mock(return_value=wale_output.replace(b'167772160', b'1'))):
|
with patch('subprocess.check_output', Mock(return_value=wale_output.replace(b'167772160', b'1'))):
|
||||||
self.assertFalse(self.wale_restore.should_use_s3_to_create_replica())
|
self.assertFalse(self.wale_restore.should_use_s3_to_create_replica())
|
||||||
|
|
||||||
with patch('psycopg2.connect', Mock(side_effect=psycopg2.Error("foo"))):
|
with patch('patroni.psycopg.connect', Mock(side_effect=psycopg.Error("foo"))):
|
||||||
save_no_master = self.wale_restore.no_master
|
save_no_master = self.wale_restore.no_master
|
||||||
save_master_connection = self.wale_restore.master_connection
|
save_master_connection = self.wale_restore.master_connection
|
||||||
|
|
||||||
|
|||||||
+38
-13
@@ -2,12 +2,13 @@ import select
|
|||||||
import six
|
import six
|
||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from kazoo.client import KazooState
|
from kazoo.client import KazooClient, KazooState
|
||||||
from kazoo.exceptions import NoNodeError, NodeExistsError
|
from kazoo.exceptions import NoNodeError, NodeExistsError
|
||||||
from kazoo.handlers.threading import SequentialThreadingHandler
|
from kazoo.handlers.threading import SequentialThreadingHandler
|
||||||
from kazoo.protocol.states import ZnodeStat
|
from kazoo.protocol.states import KeeperState, ZnodeStat
|
||||||
from mock import Mock, patch
|
from mock import Mock, PropertyMock, patch
|
||||||
from patroni.dcs.zookeeper import Leader, PatroniSequentialThreadingHandler, ZooKeeper, ZooKeeperError
|
from patroni.dcs.zookeeper import Cluster, Leader, PatroniKazooClient,\
|
||||||
|
PatroniSequentialThreadingHandler, ZooKeeper, ZooKeeperError
|
||||||
|
|
||||||
|
|
||||||
class MockKazooClient(Mock):
|
class MockKazooClient(Mock):
|
||||||
@@ -30,7 +31,9 @@ class MockKazooClient(Mock):
|
|||||||
def get(self, path, watch=None):
|
def get(self, path, watch=None):
|
||||||
if not isinstance(path, six.string_types):
|
if not isinstance(path, six.string_types):
|
||||||
raise TypeError("Invalid type for 'path' (string expected)")
|
raise TypeError("Invalid type for 'path' (string expected)")
|
||||||
if path == '/no_node':
|
if path == '/broken/status':
|
||||||
|
return (b'{', ZnodeStat(0, 0, 0, 0, 0, 0, 0, -1, 0, 0, 0))
|
||||||
|
elif path in ('/no_node', '/legacy/status'):
|
||||||
raise NoNodeError
|
raise NoNodeError
|
||||||
elif '/members/' in path:
|
elif '/members/' in path:
|
||||||
return (
|
return (
|
||||||
@@ -45,6 +48,8 @@ class MockKazooClient(Mock):
|
|||||||
return (b'foo', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0))
|
return (b'foo', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0))
|
||||||
elif path.endswith('/initialize'):
|
elif path.endswith('/initialize'):
|
||||||
return (b'foo', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0))
|
return (b'foo', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0))
|
||||||
|
elif path.endswith('/status'):
|
||||||
|
return (b'{"optime":500,"slots":{"ls":1234567}}', ZnodeStat(0, 0, 0, 0, 0, 0, 0, -1, 0, 0, 0))
|
||||||
return (b'', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0))
|
return (b'', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0))
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
@@ -124,12 +129,23 @@ class TestPatroniSequentialThreadingHandler(unittest.TestCase):
|
|||||||
self.assertRaises(select.error, self.handler.select)
|
self.assertRaises(select.error, self.handler.select)
|
||||||
|
|
||||||
|
|
||||||
|
class TestPatroniKazooClient(unittest.TestCase):
|
||||||
|
|
||||||
|
def test__call(self):
|
||||||
|
c = PatroniKazooClient()
|
||||||
|
with patch.object(KazooClient, '_call', Mock()):
|
||||||
|
self.assertIsNotNone(c._call(None, Mock()))
|
||||||
|
c._state = KeeperState.CONNECTING
|
||||||
|
self.assertFalse(c._call(None, Mock()))
|
||||||
|
|
||||||
|
|
||||||
class TestZooKeeper(unittest.TestCase):
|
class TestZooKeeper(unittest.TestCase):
|
||||||
|
|
||||||
@patch('patroni.dcs.zookeeper.KazooClient', MockKazooClient)
|
@patch('patroni.dcs.zookeeper.PatroniKazooClient', MockKazooClient)
|
||||||
def setUp(self):
|
def setUp(self):
|
||||||
self.zk = ZooKeeper({'hosts': ['localhost:2181'], 'scope': 'test',
|
self.zk = ZooKeeper({'hosts': ['localhost:2181'], 'scope': 'test',
|
||||||
'name': 'foo', 'ttl': 30, 'retry_timeout': 10, 'loop_wait': 10})
|
'name': 'foo', 'ttl': 30, 'retry_timeout': 10, 'loop_wait': 10,
|
||||||
|
'set_acls': {'CN=principal2': ['ALL']}})
|
||||||
|
|
||||||
def test_session_listener(self):
|
def test_session_listener(self):
|
||||||
self.zk.session_listener(KazooState.SUSPENDED)
|
self.zk.session_listener(KazooState.SUSPENDED)
|
||||||
@@ -147,6 +163,10 @@ class TestZooKeeper(unittest.TestCase):
|
|||||||
def test__inner_load_cluster(self):
|
def test__inner_load_cluster(self):
|
||||||
self.zk._base_path = self.zk._base_path.replace('test', 'bla')
|
self.zk._base_path = self.zk._base_path.replace('test', 'bla')
|
||||||
self.zk._inner_load_cluster()
|
self.zk._inner_load_cluster()
|
||||||
|
self.zk._base_path = self.zk._base_path = '/broken'
|
||||||
|
self.zk._inner_load_cluster()
|
||||||
|
self.zk._base_path = self.zk._base_path = '/legacy'
|
||||||
|
self.zk._inner_load_cluster()
|
||||||
self.zk._base_path = self.zk._base_path = '/no_node'
|
self.zk._base_path = self.zk._base_path = '/no_node'
|
||||||
self.zk._inner_load_cluster()
|
self.zk._inner_load_cluster()
|
||||||
|
|
||||||
@@ -154,13 +174,15 @@ class TestZooKeeper(unittest.TestCase):
|
|||||||
self.assertRaises(ZooKeeperError, self.zk.get_cluster)
|
self.assertRaises(ZooKeeperError, self.zk.get_cluster)
|
||||||
cluster = self.zk.get_cluster(True)
|
cluster = self.zk.get_cluster(True)
|
||||||
self.assertIsInstance(cluster.leader, Leader)
|
self.assertIsInstance(cluster.leader, Leader)
|
||||||
|
self.zk.status_watcher(None)
|
||||||
|
self.zk.get_cluster()
|
||||||
self.zk.touch_member({'foo': 'foo'})
|
self.zk.touch_member({'foo': 'foo'})
|
||||||
self.zk._name = 'bar'
|
self.zk._name = 'bar'
|
||||||
self.zk.optime_watcher(None)
|
self.zk.status_watcher(None)
|
||||||
with patch.object(ZooKeeper, 'get_node', Mock(side_effect=Exception)):
|
with patch.object(ZooKeeper, 'get_node', Mock(side_effect=Exception)):
|
||||||
self.zk.get_cluster()
|
self.zk.get_cluster()
|
||||||
cluster = self.zk.get_cluster()
|
cluster = self.zk.get_cluster()
|
||||||
self.assertEqual(cluster.last_leader_operation, 500)
|
self.assertEqual(cluster.last_lsn, 500)
|
||||||
|
|
||||||
def test_delete_leader(self):
|
def test_delete_leader(self):
|
||||||
self.assertTrue(self.zk.delete_leader())
|
self.assertTrue(self.zk.delete_leader())
|
||||||
@@ -194,6 +216,7 @@ class TestZooKeeper(unittest.TestCase):
|
|||||||
self.zk.touch_member({'retry': 'retry'})
|
self.zk.touch_member({'retry': 'retry'})
|
||||||
self.zk._fetch_cluster = True
|
self.zk._fetch_cluster = True
|
||||||
self.zk.get_cluster()
|
self.zk.get_cluster()
|
||||||
|
self.zk.touch_member({'retry': 'retry'})
|
||||||
self.zk.touch_member({'conn_url': 'postgres://repuser:rep-pass@localhost:5434/postgres',
|
self.zk.touch_member({'conn_url': 'postgres://repuser:rep-pass@localhost:5434/postgres',
|
||||||
'api_url': 'http://127.0.0.1:8009/patroni'})
|
'api_url': 'http://127.0.0.1:8009/patroni'})
|
||||||
|
|
||||||
@@ -203,16 +226,18 @@ class TestZooKeeper(unittest.TestCase):
|
|||||||
self.zk.take_leader()
|
self.zk.take_leader()
|
||||||
|
|
||||||
def test_update_leader(self):
|
def test_update_leader(self):
|
||||||
self.assertTrue(self.zk.update_leader(None))
|
self.assertTrue(self.zk.update_leader(12345))
|
||||||
|
|
||||||
|
@patch.object(Cluster, 'min_version', PropertyMock(return_value=(2, 0)))
|
||||||
def test_write_leader_optime(self):
|
def test_write_leader_optime(self):
|
||||||
self.zk.last_leader_operation = '0'
|
self.zk.last_lsn = '0'
|
||||||
self.zk.write_leader_optime('1')
|
self.zk.write_leader_optime('1')
|
||||||
with patch.object(MockKazooClient, 'create_async', Mock()):
|
with patch.object(MockKazooClient, 'create_async', Mock()):
|
||||||
self.zk.write_leader_optime('1')
|
self.zk.write_leader_optime('1')
|
||||||
with patch.object(MockKazooClient, 'set_async', Mock()):
|
with patch.object(MockKazooClient, 'set_async', Mock()):
|
||||||
self.zk.write_leader_optime('2')
|
self.zk.write_leader_optime('2')
|
||||||
self.zk._base_path = self.zk._base_path.replace('test', 'bla')
|
self.zk._base_path = self.zk._base_path.replace('test', 'bla')
|
||||||
|
self.zk.get_cluster()
|
||||||
self.zk.write_leader_optime('3')
|
self.zk.write_leader_optime('3')
|
||||||
|
|
||||||
def test_delete_cluster(self):
|
def test_delete_cluster(self):
|
||||||
@@ -220,8 +245,8 @@ class TestZooKeeper(unittest.TestCase):
|
|||||||
|
|
||||||
def test_watch(self):
|
def test_watch(self):
|
||||||
self.zk.watch(None, 0)
|
self.zk.watch(None, 0)
|
||||||
self.zk.event.isSet = Mock(return_value=True)
|
self.zk.event.is_set = Mock(return_value=True)
|
||||||
self.zk._fetch_optime = False
|
self.zk._fetch_status = False
|
||||||
self.zk.watch(None, 0)
|
self.zk.watch(None, 0)
|
||||||
|
|
||||||
def test__kazoo_connect(self):
|
def test__kazoo_connect(self):
|
||||||
|
|||||||
Reference in New Issue
Block a user