mirror of
https://github.com/outbackdingo/patroni.git
synced 2026-08-28 00:20:17 +00:00
Compare commits
@@ -18,7 +18,10 @@ def install_requirements(what):
|
||||
finally:
|
||||
sys.path = old_path
|
||||
requirements = ['mock>=2.0.0', 'flake8', 'pytest', 'pytest-cov'] if what == 'all' else ['behave']
|
||||
requirements += ['psycopg2-binary', 'coverage']
|
||||
requirements += ['coverage']
|
||||
# try to split tests between psycopg2 and psycopg3
|
||||
requirements += ['psycopg[binary]'] if sys.version_info >= (3, 6, 0) and\
|
||||
(sys.platform != 'darwin' or what == 'etcd3') else ['psycopg2-binary']
|
||||
for r in read('requirements.txt').split('\n'):
|
||||
r = r.strip()
|
||||
if r != '':
|
||||
@@ -107,7 +110,7 @@ def install_etcd():
|
||||
|
||||
|
||||
def install_postgres():
|
||||
version = os.environ.get('PGVERSION', '12.1-1')
|
||||
version = os.environ.get('PGVERSION', '14.1-1')
|
||||
platform = {'darwin': 'osx', 'win32': 'windows-x64', 'cygwin': 'windows-x64'}[sys.platform]
|
||||
name = 'postgresql-{0}-{1}-binaries.zip'.format(version, platform)
|
||||
get_file('http://get.enterprisedb.com/postgresql/' + name, name)
|
||||
|
||||
@@ -27,7 +27,7 @@ def main():
|
||||
|
||||
version = versions.get(what)
|
||||
path = '/usr/lib/postgresql/{0}/bin:.'.format(version)
|
||||
unbuffer = ['timeout', '600', 'unbuffer']
|
||||
unbuffer = ['timeout', '900', 'unbuffer']
|
||||
args = ['--tags=-skip'] if what == 'etcd' else []
|
||||
else:
|
||||
path = os.path.abspath(os.path.join('pgsql', 'bin'))
|
||||
|
||||
@@ -30,15 +30,6 @@ jobs:
|
||||
run: python .github/workflows/run_tests.py
|
||||
if: matrix.os != 'windows'
|
||||
|
||||
- name: Set up Python 3.5
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: 3.5
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
- name: Run tests and flake8
|
||||
run: python .github/workflows/run_tests.py
|
||||
|
||||
- name: Set up Python 3.6
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
@@ -75,6 +66,15 @@ jobs:
|
||||
- name: Run tests and flake8
|
||||
run: python .github/workflows/run_tests.py
|
||||
|
||||
- name: Set up Python 3.10
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: '3.10'
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
- name: Run tests and flake8
|
||||
run: python .github/workflows/run_tests.py
|
||||
|
||||
- name: Combine coverage
|
||||
run: python .github/workflows/run_tests.py combine
|
||||
|
||||
@@ -88,26 +88,31 @@ jobs:
|
||||
GITHUB_TOKEN: ${{ secrets.github_token }}
|
||||
run: python -m coveralls --service=github
|
||||
|
||||
- name: Run codacy-coverage-reporter
|
||||
uses: codacy/codacy-coverage-reporter-action@master
|
||||
env:
|
||||
SECRETS_AVAILABLE: ${{ secrets.CODACY_PROJECT_TOKEN != '' }}
|
||||
with:
|
||||
project-token: ${{ secrets.CODACY_PROJECT_TOKEN }}
|
||||
coverage-reports: coverage.xml
|
||||
if: ${{ matrix.os == 'ubuntu' && env.SECRETS_AVAILABLE == 'true' }}
|
||||
|
||||
behave:
|
||||
runs-on: ${{ matrix.os }}-latest
|
||||
env:
|
||||
DCS: ${{ matrix.dcs }}
|
||||
ETCDVERSION: 3.3.13
|
||||
PGVERSION: 12.1-1 # for windows and macos
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu]
|
||||
python-version: [2.7, 3.5, 3.8]
|
||||
python-version: [2.7, 3.6, 3.9]
|
||||
dcs: [etcd, etcd3, consul, exhibitor, kubernetes, raft]
|
||||
exclude:
|
||||
- dcs: kubernetes
|
||||
python-version: 2.7
|
||||
include:
|
||||
- os: macos
|
||||
python-version: 3.7
|
||||
dcs: raft
|
||||
- os: macos
|
||||
python-version: 3.8
|
||||
dcs: etcd
|
||||
- os: macos
|
||||
python-version: '3.10'
|
||||
dcs: etcd3
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v1
|
||||
@@ -117,45 +122,14 @@ jobs:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
- name: Add postgresql apt repo
|
||||
run: sudo sh -c 'echo "deb http://apt.postgresql.org/pub/repos/apt $(lsb_release -cs)-pgdg main" > /etc/apt/sources.list.d/pgdg.list'
|
||||
if: matrix.os == 'ubuntu'
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
- name: Run behave tests
|
||||
run: python .github/workflows/run_tests.py
|
||||
- uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: 3.9
|
||||
- name: Install coveralls
|
||||
run: python -m pip install coveralls
|
||||
- name: Upload Coverage
|
||||
env:
|
||||
COVERALLS_FLAG_NAME: behave-${{ matrix.os }}-${{ matrix.dcs }}-${{ matrix.python-version }}
|
||||
COVERALLS_PARALLEL: 'true'
|
||||
GITHUB_TOKEN: ${{ secrets.github_token }}
|
||||
run: python -m coveralls --service=github
|
||||
|
||||
behavem:
|
||||
runs-on: ${{ matrix.os }}-latest
|
||||
env:
|
||||
DCS: ${{ matrix.dcs }}
|
||||
ETCDVERSION: 3.3.13
|
||||
PGVERSION: 12.1-1 # for windows and macos
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [macos] #, windows]
|
||||
python-version: [3.7]
|
||||
dcs: [etcd, etcd3, raft]
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v1
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v2
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
- name: Install dependencies
|
||||
run: python .github/workflows/install_deps.py
|
||||
- name: Run behave tests
|
||||
run: python .github/workflows/run_tests.py
|
||||
python-version: '3.10'
|
||||
- name: Install coveralls
|
||||
run: python -m pip install coveralls
|
||||
- name: Upload Coverage
|
||||
@@ -167,7 +141,7 @@ jobs:
|
||||
|
||||
coveralls-finish:
|
||||
name: Finalize coveralls.io
|
||||
needs: [unit, behave, behavem]
|
||||
needs: [unit, behave]
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/setup-python@v2
|
||||
|
||||
+8
-2
@@ -12,7 +12,7 @@ Patroni is a template for you to create your own customized, high-availability s
|
||||
|
||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely.
|
||||
|
||||
Currently supported PostgreSQL versions: 9.3 to 13.
|
||||
Currently supported PostgreSQL versions: 9.3 to 14.
|
||||
|
||||
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the `Kubernetes <https://github.com/zalando/patroni/blob/master/docs/kubernetes.rst>`__ chapter of the Patroni documentation.
|
||||
|
||||
@@ -61,7 +61,7 @@ To install requirements on a Mac, run the following:
|
||||
|
||||
brew install postgresql etcd haproxy libyaml python
|
||||
|
||||
**Psycopg2**
|
||||
**Psycopg**
|
||||
|
||||
Starting from `psycopg2-2.8 <http://initd.org/psycopg/articles/2019/04/04/psycopg-28-released/>`__ the binary version of psycopg2 will no longer be installed by default. Installing it from the source code requires C compiler and postgres+python dev packages.
|
||||
Since in the python world it is not possible to specify dependency as ``psycopg2 OR psycopg2-binary`` you will have to decide how to install it.
|
||||
@@ -88,6 +88,12 @@ There are a few options available:
|
||||
|
||||
pip install psycopg2>=2.5.4
|
||||
|
||||
4. Use psycopg 3.0 instead of psycopg2
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg[binary]
|
||||
|
||||
**General installation for pip**
|
||||
|
||||
Patroni can be installed with pip:
|
||||
|
||||
@@ -49,6 +49,7 @@ Consul
|
||||
- **PATRONI\_CONSUL\_CHECKS**: (optional) list of Consul health checks used for the session. By default an empty list is used.
|
||||
- **PATRONI\_CONSUL\_REGISTER\_SERVICE**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, replica or standby-leader depending on the node's role. Defaults to **false**
|
||||
- **PATRONI\_CONSUL\_SERVICE\_CHECK\_INTERVAL**: (optional) how often to perform health check against registered url
|
||||
- **PATRONI\_CONSUL\_SERVICE\_CHECK\_TLS\_SERVER\_NAME**: (optional) overide SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
|
||||
|
||||
Etcd
|
||||
----
|
||||
@@ -59,7 +60,8 @@ Etcd
|
||||
- **PATRONI\_ETCD\_USE\_PROXIES**: If this parameter is set to true, Patroni will consider **hosts** as a list of proxies and will not perform a topology discovery of etcd cluster but stick to a fixed list of **hosts**.
|
||||
- **PATRONI\_ETCD\_PROTOCOL**: http or https, if not specified http is used. If the **url** or **proxy** is specified - will take protocol from them.
|
||||
- **PATRONI\_ETCD\_HOST**: the host:port for the etcd endpoint.
|
||||
- **PATRONI\_ETCD\_SRV**: Domain to search the SRV record(s) for cluster autodiscovery.
|
||||
- **PATRONI\_ETCD\_SRV**: Domain to search the SRV record(s) for cluster autodiscovery. Patroni will try to query these SRV service names for specified domain (in that order until first success): ``_etcd-client-ssl``, ``_etcd-client``, ``_etcd-ssl``, ``_etcd``, ``_etcd-server-ssl``, ``_etcd-server``. If SRV records for ``_etcd-server-ssl`` or ``_etcd-server`` are retrieved then ETCD peer protocol is used do query ETCD for available members. Otherwise hosts from SRV records will be used.
|
||||
- **PATRONI\_ETCD\_SRV\_SUFFIX**: Configures a suffix to the SRV name that is queried during discovery. Use this flag to differentiate between multiple etcd clusters under the same domain. Works only with conjunction with **PATRONI\_ETCD\_SRV**. For example, if ``PATRONI_ETCD_SRV_SUFFIX=foo`` and ``PATRONI_ETCD_SRV=example.org`` are set, the following DNS SRV query is made:``_etcd-client-ssl-foo._tcp.example.com`` (and so on for every possible ETCD SRV service name).
|
||||
- **PATRONI\_ETCD\_USERNAME**: username for etcd authentication.
|
||||
- **PATRONI\_ETCD\_PASSWORD**: password for etcd authentication.
|
||||
- **PATRONI\_ETCD\_CACERT**: The ca certificate. If present it will enable validation.
|
||||
@@ -83,6 +85,7 @@ ZooKeeper
|
||||
- **PATRONI\_ZOOKEEPER\_KEY**: (optional) File with the client key.
|
||||
- **PATRONI\_ZOOKEEPER\_KEY\_PASSWORD**: (optional) The client key password.
|
||||
- **PATRONI\_ZOOKEEPER\_VERIFY**: (optional) Whether to verify certificate or not. Defaults to ``true``.
|
||||
- **PATRONI\_ZOOKEEPER\_SET\_ACLS**: (optional) If set, configure Kazoo to apply a default ACL to each ZNode that it creates. ACLs will assume 'x509' schema and should be specified as a dictionary with the principal as the key and one or more permissions as a list in the value. Permissions may be one of ``CREATE``, ``READ``, ``WRITE``, ``DELETE`` or ``ADMIN``. For example, ``set_acls: {CN=principal1: [CREATE, READ], CN=principal2: [ALL]}``.
|
||||
|
||||
.. note::
|
||||
It is required to install ``kazoo>=2.6.0`` to support SSL.
|
||||
@@ -105,6 +108,7 @@ Kubernetes
|
||||
- **PATRONI\_KUBERNETES\_USE\_ENDPOINTS**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||
- **PATRONI\_KUBERNETES\_POD\_IP**: (optional) IP address of the pod Patroni is running in. This value is required when `PATRONI_KUBERNETES_USE_ENDPOINTS` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||
- **PATRONI\_KUBERNETES\_PORTS**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``PATRONI_KUBERNETES_PORTS='[{"name": "postgresql", "port": 5432}]'`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `PATRONI_KUBERNETES_USE_ENDPOINTS` is set.
|
||||
- **PATRONI\_KUBERNETES\_CACERT**: (optional) Specifies the file with the CA_BUNDLE file with certificates of trusted CAs to use while verifying Kubernetes API SSL certs. If not provided, patroni will use the value provided by the ServiceAccount secret.
|
||||
|
||||
Raft
|
||||
----
|
||||
@@ -131,6 +135,7 @@ PostgreSQL
|
||||
- **PATRONI\_REPLICATION\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **PATRONI\_REPLICATION\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **PATRONI\_REPLICATION\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_REPLICATION\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_REPLICATION\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **PATRONI\_REPLICATION\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
- **PATRONI\_SUPERUSER\_USERNAME**: name for the superuser, set during initialization (initdb) and later used by Patroni to connect to the postgres. Also this user is used by pg_rewind.
|
||||
@@ -141,6 +146,7 @@ PostgreSQL
|
||||
- **PATRONI\_SUPERUSER\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **PATRONI\_SUPERUSER\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **PATRONI\_SUPERUSER\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_SUPERUSER\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_SUPERUSER\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **PATRONI\_SUPERUSER\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
- **PATRONI\_REWIND\_USERNAME**: name for the user for ``pg_rewind``; the user will be created during initialization of postgres 11+ and all necessary `permissions <https://www.postgresql.org/docs/11/app-pgrewind.html#id-1.9.5.8.8>`__ will be granted.
|
||||
@@ -151,6 +157,7 @@ PostgreSQL
|
||||
- **PATRONI\_REWIND\_SSLCERT**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **PATRONI\_REWIND\_SSLROOTCERT**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **PATRONI\_REWIND\_SSLCRL**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_REWIND\_SSLCRLDIR**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **PATRONI\_REWIND\_GSSENCMODE**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **PATRONI\_REWIND\_CHANNEL\_BINDING**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
|
||||
|
||||
+8
-2
@@ -35,7 +35,7 @@ To install requirements on a Mac, run the following:
|
||||
|
||||
.. _psycopg2_install_options:
|
||||
|
||||
**Psycopg2**
|
||||
**Psycopg**
|
||||
|
||||
Starting from `psycopg2-2.8 <http://initd.org/psycopg/articles/2019/04/04/psycopg-28-released/>`__ the binary version of psycopg2 will no longer be installed by default. Installing it from the source code requires C compiler and postgres+python dev packages.
|
||||
Since in the python world it is not possible to specify dependency as ``psycopg2 OR psycopg2-binary`` you will have to decide how to install it.
|
||||
@@ -62,6 +62,12 @@ There are a few options available:
|
||||
|
||||
pip install psycopg2>=2.5.4
|
||||
|
||||
4. Use psycopg 3.0 instead of psycopg2
|
||||
|
||||
::
|
||||
|
||||
pip install psycopg[binary]>=3.0.0
|
||||
|
||||
**General installation for pip**
|
||||
|
||||
Patroni can be installed with pip:
|
||||
@@ -73,7 +79,7 @@ Patroni can be installed with pip:
|
||||
where dependencies can be either empty, or consist of one or more of the following:
|
||||
|
||||
etcd or etcd3
|
||||
`python-etcd` module in order to use Etcd as DCS
|
||||
`python-etcd` module in order to use Etcd as Distributed Configuration Store (DCS)
|
||||
consul
|
||||
`python-consul` module in order to use Consul as DCS
|
||||
zookeeper
|
||||
|
||||
+11
-3
@@ -132,6 +132,7 @@ Most of the parameters are optional, but you have to specify one of the **host**
|
||||
- **register\_service**: (optional) whether or not to register a service with the name defined by the scope parameter and the tag master, replica or standby-leader depending on the node's role. Defaults to **false**.
|
||||
- **service\_tags**: (optional) additional static tags to add to the Consul service apart from the role (``master``/``replica``/``standby-leader``). By default an empty list is used.
|
||||
- **service\_check\_interval**: (optional) how often to perform health check against registered url.
|
||||
- **service\_check\_tls\_server\_name**: (optional) overide SNI host when connecting via TLS, see also `consul agent check API reference <https://www.consul.io/api-docs/agent/check#tlsservername>`__.
|
||||
|
||||
The ``token`` needs to have the following ACL permissions:
|
||||
|
||||
@@ -156,7 +157,8 @@ Most of the parameters are optional, but you have to specify one of the **host**
|
||||
- **use\_proxies**: If this parameter is set to true, Patroni will consider **hosts** as a list of proxies and will not perform a topology discovery of etcd cluster.
|
||||
- **url**: url for the etcd.
|
||||
- **proxy**: proxy url for the etcd. If you are connecting to the etcd using proxy, use this parameter instead of **url**.
|
||||
- **srv**: Domain to search the SRV record(s) for cluster autodiscovery.
|
||||
- **srv**: Domain to search the SRV record(s) for cluster autodiscovery. Patroni will try to query these SRV service names for specified domain (in that order until first success): ``_etcd-client-ssl``, ``_etcd-client``, ``_etcd-ssl``, ``_etcd``, ``_etcd-server-ssl``, ``_etcd-server``. If SRV records for ``_etcd-server-ssl`` or ``_etcd-server`` are retrieved then ETCD peer protocol is used do query ETCD for available members. Otherwise hosts from SRV records will be used.
|
||||
- **srv\_suffix**: Configures a suffix to the SRV name that is queried during discovery. Use this flag to differentiate between multiple etcd clusters under the same domain. Works only with conjunction with **srv**. For example, if ``srv_suffix: foo`` and ``srv: example.org`` are set, the following DNS SRV query is made:``_etcd-client-ssl-foo._tcp.example.com`` (and so on for every possible ETCD SRV service name).
|
||||
- **protocol**: (optional) http or https, if not specified http is used. If the **url** or **proxy** is specified - will take protocol from them.
|
||||
- **username**: (optional) username for etcd authentication.
|
||||
- **password**: (optional) password for etcd authentication.
|
||||
@@ -181,6 +183,7 @@ ZooKeeper
|
||||
- **key**: (optional) File with the client key.
|
||||
- **key_password**: (optional) The client key password.
|
||||
- **verify**: (optional) Whether to verify certificate or not. Defaults to ``true``.
|
||||
- **set_acls**: (optional) If set, configure Kazoo to apply a default ACL to each ZNode that it creates. ACLs will assume 'x509' schema and should be specified as a dictionary with the principal as the key and one or more permissions as a list in the value. Permissions may be one of ``CREATE``, ``READ``, ``WRITE``, ``DELETE`` or ``ADMIN``. For example, ``set_acls: {CN=principal1: [CREATE, READ], CN=principal2: [ALL]}``.
|
||||
|
||||
.. note::
|
||||
It is required to install ``kazoo>=2.6.0`` to support SSL.
|
||||
@@ -204,6 +207,7 @@ Kubernetes
|
||||
- **use\_endpoints**: (optional) if set to true, Patroni will use Endpoints instead of ConfigMaps to run leader elections and keep cluster state.
|
||||
- **pod\_ip**: (optional) IP address of the pod Patroni is running in. This value is required when `use_endpoints` is enabled and is used to populate the leader endpoint subsets when the pod's PostgreSQL is promoted.
|
||||
- **ports**: (optional) if the Service object has the name for the port, the same name must appear in the Endpoint object, otherwise service won't work. For example, if your service is defined as ``{Kind: Service, spec: {ports: [{name: postgresql, port: 5432, targetPort: 5432}]}}``, then you have to set ``kubernetes.ports: [{"name": "postgresql", "port": 5432}]`` and Patroni will use it for updating subsets of the leader Endpoint. This parameter is used only if `kubernetes.use_endpoints` is set.
|
||||
- **cacert**: (optional) Specifies the file with the CA_BUNDLE file with certificates of trusted CAs to use while verifying Kubernetes API SSL certs. If not provided, patroni will use the value provided by the ServiceAccount secret.
|
||||
|
||||
|
||||
.. _raft_settings:
|
||||
@@ -254,6 +258,7 @@ PostgreSQL
|
||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
- **replication**:
|
||||
@@ -265,6 +270,7 @@ PostgreSQL
|
||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
- **rewind**:
|
||||
@@ -276,6 +282,7 @@ PostgreSQL
|
||||
- **sslcert**: (optional) maps to the `sslcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCERT>`__ connection parameter, which specifies the location of the client certificate.
|
||||
- **sslrootcert**: (optional) maps to the `sslrootcert <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLROOTCERT>`__ connection parameter, which specifies the location of a file containing one ore more certificate authorities (CA) certificates that the client will use to verify a server's certificate.
|
||||
- **sslcrl**: (optional) maps to the `sslcrl <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRL>`__ connection parameter, which specifies the location of a file containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **sslcrldir**: (optional) maps to the `sslcrldir <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-SSLCRLDIR>`__ connection parameter, which specifies the location of a directory with files containing a certificate revocation list. A client will reject connecting to any server that has a certificate present in this list.
|
||||
- **gssencmode**: (optional) maps to the `gssencmode <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-GSSENCMODE>`__ connection parameter, which determines whether or with what priority a secure GSS TCP/IP connection will be negotiated with the server
|
||||
- **channel_binding**: (optional) maps to the `channel_binding <https://www.postgresql.org/docs/current/libpq-connect.html#LIBPQ-CONNECT-CHANNEL-BINDING>`__ connection parameter, which controls the client's use of channel binding.
|
||||
- **callbacks**: callback scripts to run on certain actions. Patroni will pass the action, role and cluster name. (See scripts/aws.py as an example of how to write them.)
|
||||
@@ -307,7 +314,7 @@ PostgreSQL
|
||||
- **pg\_ctl\_timeout**: How long should pg_ctl wait when doing ``start``, ``stop`` or ``restart``. Default value is 60 seconds.
|
||||
- **use\_pg\_rewind**: try to use pg\_rewind on the former leader when it joins cluster as a replica.
|
||||
- **remove\_data\_directory\_on\_rewind\_failure**: If this option is enabled, Patroni will remove the PostgreSQL data directory and recreate the replica. Otherwise it will try to follow the new leader. Default value is **false**.
|
||||
- **remove\_data\_directory\_on\_diverged\_timelines**: Patroni will remove the PostgreSQL data directory and recreate the replica if it notices that timelines are diverging and the former master can not start streaming from the new master. This option is useful when ``pg_rewind`` can not be used. Default value is **false**.
|
||||
- **remove\_data\_directory\_on\_diverged\_timelines**: Patroni will remove the PostgreSQL data directory and recreate the replica if it notices that timelines are diverging and the former master can not start streaming from the new master. This option is useful when ``pg_rewind`` can not be used. While performing timelines divergence check on PostgreSQL v10 and older Patroni will try to connect with replication credential to the "postgres" database. Hence, such access should be allowed in the pg_hba.conf. Default value is **false**.
|
||||
- **replica\_method**: for each create_replica_methods other than basebackup, you would add a configuration section of the same name. At a minimum, this should include "command" with a full path to the actual script to be executed. Other configuration parameters will be passed along to the script in the form "parameter=value".
|
||||
- **pre\_promote**: a fencing script that executes during a failover after acquiring the leader lock but before promoting the replica. If the script exits with a non-zero code, Patroni does not promote the replica and removes the leader key from DCS.
|
||||
|
||||
@@ -323,7 +330,7 @@ REST API
|
||||
- **password**: Basic-auth password to protect unsafe REST API endpoints.
|
||||
- **certfile**: (optional): Specifies the file with the certificate in the PEM format. If the certfile is not specified or is left empty, the API server will work without SSL.
|
||||
- **keyfile**: (optional): Specifies the file with the secret key in the PEM format.
|
||||
- **keyfile_password**: (optional): Specifies a password for decrypting the keyfile.
|
||||
- **keyfile\_password**: (optional): Specifies a password for decrypting the keyfile.
|
||||
- **cafile**: (optional): Specifies the file with the CA_BUNDLE with certificates of trusted CAs to use while verifying client certs.
|
||||
- **ciphers**: (optional): Specifies the permitted cipher suites (e.g. "ECDHE-RSA-AES256-GCM-SHA384:DHE-RSA-AES256-GCM-SHA384:ECDHE-RSA-AES128-GCM-SHA256:DHE-RSA-AES128-GCM-SHA256:!SSLv1:!SSLv2:!SSLv3:!TLSv1:!TLSv1.1")
|
||||
- **verify\_client**: (optional): ``none`` (default), ``optional`` or ``required``. When ``none`` REST API will not check client certificates. When ``required`` client certificates are required for all REST API calls. When ``optional`` client certificates are required for all unsafe REST API endpoints. When ``required`` is used, then client authentication succeeds, if the certificate signature verification succeeds. For ``optional`` the client cert will only be checked for ``PUT``, ``POST``, ``PATCH``, and ``DELETE`` requests.
|
||||
@@ -361,6 +368,7 @@ CTL
|
||||
- **cacert**: Specifies the file with the CA_BUNDLE file or directory with certificates of trusted CAs to use while verifying REST API SSL certs. If not provided patronictl will use the value provided for REST API "cafile" parameter.
|
||||
- **certfile**: Specifies the file with the client certificate in the PEM format. If not provided patronictl will use the value provided for REST API "certfile" parameter.
|
||||
- **keyfile**: Specifies the file with the client secret key in the PEM format. If not provided patronictl will use the value provided for REST API "keyfile" parameter.
|
||||
- **keyfile\_password**: Specifies a password for decrypting the keyfile. If not provided patronictl will use the value provided for REST API "keyfile\_password" parameter.
|
||||
|
||||
Watchdog
|
||||
--------
|
||||
|
||||
+4
-1
@@ -194,4 +194,7 @@ intersphinx_mapping = {'https://docs.python.org/': None}
|
||||
# A possibility to have an own stylesheet, to add new rules or override existing ones
|
||||
# For the latter case, the CSS specificity of the rules should be higher than the default ones
|
||||
def setup(app):
|
||||
app.add_stylesheet("custom.css")
|
||||
if hasattr(app, 'add_css_file'):
|
||||
app.add_css_file('custom.css')
|
||||
else:
|
||||
app.add_stylesheet('custom.css')
|
||||
|
||||
+1
-1
@@ -10,7 +10,7 @@ Patroni is a template for you to create your own customized, high-availability s
|
||||
|
||||
We call Patroni a "template" because it is far from being a one-size-fits-all or plug-and-play replication system. It will have its own caveats. Use wisely. There are many ways to run high availability with PostgreSQL; for a list, see the `PostgreSQL Documentation <https://wiki.postgresql.org/wiki/Replication,_Clustering,_and_Connection_Pooling>`__.
|
||||
|
||||
Currently supported PostgreSQL versions: 9.3 to 13.
|
||||
Currently supported PostgreSQL versions: 9.3 to 14.
|
||||
|
||||
**Note to Kubernetes users**: Patroni can run natively on top of Kubernetes. Take a look at the :ref:`Kubernetes <kubernetes>` chapter of the Patroni documentation.
|
||||
|
||||
|
||||
+280
-2
@@ -3,6 +3,284 @@
|
||||
Release notes
|
||||
=============
|
||||
|
||||
Version 2.1.4
|
||||
-------------
|
||||
|
||||
**New features**
|
||||
|
||||
- Improve ``pg_rewind`` behavior on typical Debian/Ubuntu systems (Gunnar "Nick" Bluth)
|
||||
|
||||
On Postgres setups that keep `postgresql.conf` outside of the data directory (e.g. Ubuntu/Debian packages), ``pg_rewind --restore-target-wal`` fails to figure out the value of the ``restore_command``.
|
||||
|
||||
- Allow setting ``TLSServerName`` on Consul service checks (Michael Gmelin)
|
||||
|
||||
Useful when checks are performed by IP and the Consul ``node_name`` is not a FQDN.
|
||||
|
||||
- Added ``ppc64le`` support in watchdog (Jean-Michel Scheiwiler)
|
||||
|
||||
And fixed watchdog support on some non-x86 platforms.
|
||||
|
||||
- Switched aws.py callback from ``boto`` to ``boto3`` (Alexander Kukushkin)
|
||||
|
||||
``boto`` 2.x is abandoned since 2018 and fails with python 3.9.
|
||||
|
||||
- Periodically refresh service account token on K8s (Haitao Li)
|
||||
|
||||
Since Kubernetes v1.21 service account tokens expire in 1 hour.
|
||||
|
||||
- Added ``/read-only-sync`` monitoring endpoint (Dennis4b)
|
||||
|
||||
It is similar to the ``/read-only`` but includes only synchronous replicas.
|
||||
|
||||
|
||||
**Stability improvements**
|
||||
|
||||
- Don't copy the logical replication slot to a replica if there is a configuration mismatch in the logical decoding setup with the primary (Alexander)
|
||||
|
||||
A replica won't copy a logical replication slot from the primary anymore if the slot doesn't match the ``plugin`` or ``database`` configuration options. Previously, the check for whether the slot matches those configuration options was not performed until after the replica copied the slot and started with it, resulting in unnecessary and repeated restarts.
|
||||
|
||||
- Special handling of recovery configuration parameters for PostgreSQL v12+ (Alexander)
|
||||
|
||||
While starting as replica Patroni should be able to update ``postgresql.conf`` and restart/reload if the leader address has changed by caching current parameters values instead of querying them from ``pg_settings``.
|
||||
|
||||
- Better handling of IPv6 addresses in the ``postgresql.listen`` parameters (Alexander)
|
||||
|
||||
Since the ``listen`` parameter has a port, people try to put IPv6 addresses into square brackets, which were not correctly stripped when there is more than one IP in the list.
|
||||
|
||||
- Use ``replication`` credentials when performing divergence check only on PostgreSQL v10 and older (Alexander)
|
||||
|
||||
If ``rewind`` is enabled, Patroni will again use either ``superuser`` or ``rewind`` credentials on newer Postgres versions.
|
||||
|
||||
|
||||
**Bugfixes**
|
||||
|
||||
- Fixed missing import of ``dateutil.parser`` (Wesley Mendes)
|
||||
|
||||
Tests weren't failing only because it was also imported from other modules.
|
||||
|
||||
- Ensure that ``optime`` annotation is a string (Sebastian Hasler)
|
||||
|
||||
In certain cases Patroni was trying to pass it as numeric.
|
||||
|
||||
- Better handling of failed ``pg_rewind`` attempt (Alexander)
|
||||
|
||||
If the primary becomes unavailable during ``pg_rewind``, ``$PGDATA`` will be left in a broken state. Following that, Patroni will remove the data directory even if this is not allowed by the configuration.
|
||||
|
||||
- Don't remove ``slots`` annotations from the leader ``ConfigMap``/``Endpoint`` when PostgreSQL isn't ready (Alexander)
|
||||
|
||||
If ``slots`` value isn't passed the annotation will keep the current value.
|
||||
|
||||
- Handle concurrency problem with K8s API watchers (Alexander)
|
||||
|
||||
Under certain (unknown) conditions watchers might become stale; as a result, ``attempt_to_acquire_leader()`` method could fail due to the HTTP status code 409. In that case we reset watchers connections and restart from scratch.
|
||||
|
||||
|
||||
Version 2.1.3
|
||||
-------------
|
||||
|
||||
**New features**
|
||||
|
||||
- Added support for encrypted TLS keys for ``patronictl`` (Alexander Kukushkin)
|
||||
|
||||
It could be configured via ``ctl.keyfile_password`` or the ``PATRONI_CTL_KEYFILE_PASSWORD`` environment variable.
|
||||
|
||||
- Added more metrics to the /metrics endpoint (Alexandre Pereira)
|
||||
|
||||
Specifically, ``patroni_pending_restart`` and ``patroni_is_paused``.
|
||||
|
||||
- Make it possible to specify multiple hosts in the standby cluster configuration (Michael Banck)
|
||||
|
||||
If the standby cluster is replicating from the Patroni cluster it might be nice to rely on client-side failover which is available in ``libpq`` since PostgreSQL v10. That is, the ``primary_conninfo`` on the standby leader and ``pg_rewind`` setting ``target_session_attrs=read-write`` in the connection string. The ``pgpass`` file will be generated with multiple lines (one line per host), and instead of calling ``CHECKPOINT`` on the primary cluster nodes the standby cluster will wait for ``pg_control`` to be updated.
|
||||
|
||||
**Stability improvements**
|
||||
|
||||
- Compatibility with legacy ``psycopg2`` (Alexander)
|
||||
|
||||
For example, the ``psycopg2`` installed from Ubuntu 18.04 packages doesn't have the ``UndefinedFile`` exception yet.
|
||||
|
||||
- Restart ``etcd3`` watcher if all Etcd nodes don't respond (Alexander)
|
||||
|
||||
If the watcher is alive the ``get_cluster()`` method continues returning stale information even if all Etcd nodes are failing.
|
||||
|
||||
- Don't remove the leader lock in the standby cluster while paused (Alexander)
|
||||
|
||||
Previously the lock was maintained only by the node that was running as a primary and not a standby leader.
|
||||
|
||||
**Bugfixes**
|
||||
|
||||
- Fixed bug in the standby-leader bootstrap (Alexander)
|
||||
|
||||
Patroni was considering bootstrap as failed if Postgres didn't start accepting connections after 60 seconds. The bug was introduced in the 2.1.2 release.
|
||||
|
||||
- Fixed bug with failover to a cascading standby (Alexander)
|
||||
|
||||
When figuring out which slots should be created on cascading standby we forgot to take into account that the leader might be absent.
|
||||
|
||||
- Fixed small issues in Postgres config validator (Alexander)
|
||||
|
||||
Integer parameters introduced in PostgreSQL v14 were failing to validate because min and max values were quoted in the validator.py
|
||||
|
||||
- Use replication credentials when checking leader status (Alexander)
|
||||
|
||||
It could be that the ``remove_data_directory_on_diverged_timelines`` is set, but there is no ``rewind_credentials`` defined and superuser access between nodes is not allowed.
|
||||
|
||||
- Fixed "port in use" error on REST API certificate replacement (Ants Aasma)
|
||||
|
||||
When switching certificates there was a race condition with a concurrent API request. If there is one active during the replacement period then the replacement will error out with a port in use error and Patroni gets stuck in a state without an active API server.
|
||||
|
||||
- Fixed a bug in cluster bootstrap if passwords contain ``%`` characters (Bastien Wirtz)
|
||||
|
||||
The bootstrap method executes the ``DO`` block, with all parameters properly quoted, but the ``cursor.execute()`` method didn't like an empty list with parameters passed.
|
||||
|
||||
- Fixed the "AttributeError: no attribute 'leader'" exception (Hrvoje Milković)
|
||||
|
||||
It could happen if the synchronous mode is enabled and the DCS content was wiped out.
|
||||
|
||||
- Fix bug in divergence timeline check (Alexander)
|
||||
|
||||
Patroni was falsely assuming that timelines have diverged. For pg_rewind it didn't create any problem, but if pg_rewind is not allowed and the ``remove_data_directory_on_diverged_timelines`` is set, it resulted in reinitializing the former leader.
|
||||
|
||||
|
||||
Version 2.1.2
|
||||
-------------
|
||||
|
||||
**New features**
|
||||
|
||||
- Compatibility with ``psycopg>=3.0`` (Alexander Kukushkin)
|
||||
|
||||
By default ``psycopg2`` is preferred. `psycopg>=3.0` will be used only if ``psycopg2`` is not available or its version is too old.
|
||||
|
||||
- Add ``dcs_last_seen`` field to the REST API (Michael Banck)
|
||||
|
||||
This field notes the last time (as unix epoch) a cluster member has successfully communicated with the DCS. This is useful to identify and/or analyze network partitions.
|
||||
|
||||
- Release the leader lock when ``pg_controldata`` reports "shut down" (Alexander)
|
||||
|
||||
To solve the problem of slow switchover/shutdown in case ``archive_command`` is slow/failing, Patroni will remove the leader key immediately after ``pg_controldata`` started reporting PGDATA as ``shut down`` cleanly and it verified that there is at least one replica that received all changes. If there are no replicas that fulfill this condition the leader key is not removed and the old behavior is retained, i.e. Patroni will keep updating the lock.
|
||||
|
||||
- Add ``sslcrldir`` connection parameter support (Kostiantyn Nemchenko)
|
||||
|
||||
The new connection parameter was introduced in the PostgreSQL v14.
|
||||
|
||||
- Allow setting ACLs for ZNodes in Zookeeper (Alwyn Davis)
|
||||
|
||||
Introduce a new configuration option ``zookeeper.set_acls`` so that Kazoo will apply a default ACL for each ZNode that it creates.
|
||||
|
||||
|
||||
**Stability improvements**
|
||||
|
||||
- Delay the next attempt of recovery till next HA loop (Alexander)
|
||||
|
||||
If Postgres crashed due to out of disk space (for example) and fails to start because of that Patroni is too eagerly trying to recover it flooding logs.
|
||||
|
||||
- Add log before demoting, which can take some time (Michael)
|
||||
|
||||
It can take some time for the demote to finish and it might not be obvious from looking at the logs what exactly is going on.
|
||||
|
||||
- Improve "I am" status messages (Michael)
|
||||
|
||||
``no action. I am a secondary ({0})`` vs ``no action. I am ({0}), a secondary``
|
||||
|
||||
- Cast to int ``wal_keep_segments`` when converting to ``wal_keep_size`` (Jorge Solórzano)
|
||||
|
||||
It is possible to specify ``wal_keep_segments`` as a string in the global :ref:`dynamic configuration <dynamic_configuration>` and due to Python being a dynamically typed language the string was simply multiplied. Example: ``wal_keep_segments: "100"`` was converted to ``100100100100100100100100100100100100100100100100MB``.
|
||||
|
||||
- Allow switchover only to sync nodes when synchronous replication is enabled (Alexander)
|
||||
|
||||
In addition to that do the leader race only against known synchronous nodes.
|
||||
|
||||
- Use cached role as a fallback when Postgres is slow (Alexander)
|
||||
|
||||
In some extreme cases Postgres could be so slow that the normal monitoring query does not finish in a few seconds. The ``statement_timeout`` exception not being properly handled could lead to the situation where Postgres was not demoted on time when the leader key expired or the update failed. In case of such exception Patroni will use the cached ``role`` to determine whether Postgres is running as a primary.
|
||||
|
||||
- Avoid unnecessary updates of the member ZNode (Alexander)
|
||||
|
||||
If no values have changed in the members data, the update should not happen.
|
||||
|
||||
- Optimize checkpoint after promote (Alexander)
|
||||
|
||||
Avoid doing ``CHECKPOINT`` if the latest timeline is already stored in ``pg_control``. It helps to avoid unnecessary ``CHECKPOINT`` right after initializing the new cluster with ``initdb``.
|
||||
|
||||
- Prefer members without ``nofailover`` when picking sync nodes (Alexander)
|
||||
|
||||
Previously sync nodes were selected only based on the replication lag, hence the node with ``nofailover`` tag had the same chances to become synchronous as any other node. That behavior was confusing and dangerous at the same time because in case of a failed primary the failover could not happen automatically.
|
||||
|
||||
- Remove duplicate hosts from the etcd machine cache (Michael)
|
||||
|
||||
Advertised client URLs in the etcd cluster could be misconfigured. Removing duplicates in Patroni in this case is a low-hanging fruit.
|
||||
|
||||
|
||||
**Bugfixes**
|
||||
|
||||
- Skip temporary replication slots while doing slot management (Alexander)
|
||||
|
||||
Starting from v10 ``pg_basebackup`` creates a temporary replication slot for WAL streaming and Patroni was trying to drop it because the slot name looks unknown. In order to fix it, we skip all temporary slots when querying ``pg_stat_replication_slots`` view.
|
||||
|
||||
- Ensure ``pg_replication_slot_advance()`` doesn't timeout (Alexander)
|
||||
|
||||
Patroni was using the default ``statement_timeout`` in this case and once the call failed there are very high chances that it will never recover, resulting in increased size of ``pg_wal`` and ``pg_catalog`` bloat.
|
||||
|
||||
- The ``/status`` wasn't updated on demote (Alexander)
|
||||
|
||||
After demoting PostgreSQL the old leader updates the last LSN in DCS. Starting from ``2.1.0`` the new ``/status`` key was introduced, but the optime was still written to the ``/optime/leader``.
|
||||
|
||||
- Handle DCS exceptions when demoting (Alexander)
|
||||
|
||||
While demoting the master due to failure to update the leader lock it could happen that DCS goes completely down and the ``get_cluster()`` call raises an exception. Not being handled properly it results in Postgres remaining stopped until DCS recovers.
|
||||
|
||||
- The ``use_unix_socket_repl`` didn't work is some cases (Alexander)
|
||||
|
||||
Specifically, if ``postgresql.unix_socket_directories`` is not set. In this case Patroni is supposed to use the default value from ``libpq``.
|
||||
|
||||
- Fix a few issues with Patroni REST API (Alexander)
|
||||
|
||||
The ``clusters_unlocked`` sometimes could be not defined, what resulted in exceptions in the ``GET /metrics`` endpoint. In addition to that the error handling method was assuming that the ``connect_address`` tuple always has two elements, while in fact there could be more in case of IPv6.
|
||||
|
||||
- Wait for newly promoted node to finish recovery before deciding to rewind (Alexander)
|
||||
|
||||
It could take some time before the actual promote happens and the new timeline is created. Without waiting replicas could come to the conclusion that rewind isn't required.
|
||||
|
||||
- Handle missing timelines in a history file when deciding to rewind (Alexander)
|
||||
|
||||
If the current replica timeline is missing in the history file on the primary the replica was falsely assuming that rewind isn't required.
|
||||
|
||||
|
||||
Version 2.1.1
|
||||
-------------
|
||||
|
||||
**New features**
|
||||
|
||||
- Support for ETCD SRV name suffix (David Pavlicek)
|
||||
|
||||
Etcd allows to differentiate between multiple Etcd clusters under the same domain and from now on Patroni also supports it.
|
||||
|
||||
- Enrich history with the new leader (huiyalin525)
|
||||
|
||||
It adds the new column to the ``patronictl history`` output.
|
||||
|
||||
- Make the CA bundle configurable for in-cluster Kubernetes config (Aron Parsons)
|
||||
|
||||
By default Patroni is using ``/var/run/secrets/kubernetes.io/serviceaccount/ca.crt`` and this new feature allows specifying the custom ``kubernetes.cacert``.
|
||||
|
||||
- Support dynamically registering/deregistering as a Consul service and changing tags (Tommy Li)
|
||||
|
||||
Previously it required Patroni restart.
|
||||
|
||||
**Bugfixes**
|
||||
|
||||
- Avoid unnecessary reload of REST API (Alexander Kukushkin)
|
||||
|
||||
The previous release added a feature of reloading REST API certificates if changed on disk. Unfortunately, the reload was happening unconditionally right after the start.
|
||||
|
||||
- Don't resolve cluster members when ``etcd.use_proxies`` is set (Alexander)
|
||||
|
||||
When starting up Patroni checks the healthiness of Etcd cluster by querying the list of members. In addition to that, it also tried to resolve their hostnames, which is not necessary when working with Etcd via proxy and was causing unnecessary warnings.
|
||||
|
||||
- Skip rows with NULL values in the ``pg_stat_replication`` (Alexander)
|
||||
|
||||
It seems that the ``pg_stat_replication`` view could contain NULL values in the ``replay_lsn``, ``flush_lsn``, or ``write_lsn`` fields even when ``state = 'streaming'``.
|
||||
|
||||
|
||||
Version 2.1.0
|
||||
-------------
|
||||
|
||||
@@ -39,13 +317,13 @@ This version adds compatibility with PostgreSQL v14, makes logical replication s
|
||||
When everything goes normal, only one line will be written for every run of HA loop.
|
||||
|
||||
|
||||
**Breaking chances**
|
||||
**Breaking changes**
|
||||
|
||||
- The old ``permanent logical replication slots`` feature will no longer work with PostgreSQL v10 and older (Alexander)
|
||||
|
||||
The strategy of creating the logical slots after performing a promotion can't guaranty that no logical events are lost and therefore disabled.
|
||||
|
||||
- The ``/leader`` always endpoint returns 200 if the node holds the lock (Alexander)
|
||||
- The ``/leader`` endpoint always returns 200 if the node holds the lock (Alexander)
|
||||
|
||||
Promoting the standby cluster requires updating load-balancer health checks, which is not very convenient and easy to forget. To solve it, we change the behavior of the ``/leader`` health check endpoint. It will return 200 without taking into account whether the cluster is normal or the ``standby_cluster``.
|
||||
|
||||
|
||||
@@ -44,8 +44,11 @@ For all health check ``GET`` requests Patroni returns a JSON document with the s
|
||||
|
||||
- ``GET /synchronous`` or ``GET /sync``: returns HTTP status code **200** only when the Patroni node is running as a synchronous standby.
|
||||
|
||||
- ``GET /read-only-sync``: like the above endpoint, but also includes the primary.
|
||||
|
||||
- ``GET /asynchronous`` or ``GET /async``: returns HTTP status code **200** only when the Patroni node is running as an asynchronous standby.
|
||||
|
||||
|
||||
- ``GET /asynchronous?lag=<max-lag>`` or ``GET /async?lag=<max-lag>``: asynchronous standby check endpoint. In addition to checks from ``asynchronous`` or ``async``, it also checks replication latency and returns status code **200** only when it is below specified value. The key cluster.last_leader_operation from DCS is used for Leader wal position and compute latency on replica for performance reasons. max-lag can be specified in bytes (integer) or in human readable values, for e.g. 16kB, 64MB, 1GB.
|
||||
|
||||
- ``GET /async?lag=1048576``
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
import abc
|
||||
import datetime
|
||||
import os
|
||||
import psycopg2
|
||||
import json
|
||||
import shutil
|
||||
import signal
|
||||
@@ -13,6 +12,8 @@ import threading
|
||||
import time
|
||||
import yaml
|
||||
|
||||
import patroni.psycopg as psycopg
|
||||
|
||||
from six.moves.BaseHTTPServer import BaseHTTPRequestHandler, HTTPServer
|
||||
|
||||
|
||||
@@ -185,7 +186,7 @@ class PatroniController(AbstractController):
|
||||
config['postgresql']['parameters'].update({
|
||||
'logging_collector': 'on', 'log_destination': 'csvlog', 'log_directory': self._output_dir,
|
||||
'log_filename': name + '.log', 'log_statement': 'all', 'log_min_messages': 'debug1',
|
||||
'unix_socket_directories': self._data_dir})
|
||||
'unix_socket_directories': tempfile.gettempdir()})
|
||||
|
||||
if 'bootstrap' in config:
|
||||
config['bootstrap']['post_bootstrap'] = 'psql -w -c "SELECT 1"'
|
||||
@@ -205,16 +206,16 @@ class PatroniController(AbstractController):
|
||||
|
||||
user = config['postgresql'].get('authentication', config['postgresql']).get('superuser', {})
|
||||
self._connkwargs = {k: user[n] for n, k in [('username', 'user'), ('password', 'password')] if n in user}
|
||||
self._connkwargs.update({'host': host, 'port': self.__PORT, 'database': 'postgres'})
|
||||
self._connkwargs.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres'})
|
||||
|
||||
self._replication = config['postgresql'].get('authentication', config['postgresql']).get('replication', {})
|
||||
self._replication.update({'host': host, 'port': self.__PORT, 'database': 'postgres'})
|
||||
self._replication.update({'host': host, 'port': self.__PORT, 'dbname': 'postgres'})
|
||||
|
||||
return patroni_config_path
|
||||
|
||||
def _connection(self):
|
||||
if not self._conn or self._conn.closed != 0:
|
||||
self._conn = psycopg2.connect(**self._connkwargs)
|
||||
self._conn = psycopg.connect(**self._connkwargs)
|
||||
self._conn.autocommit = True
|
||||
return self._conn
|
||||
|
||||
@@ -228,7 +229,7 @@ class PatroniController(AbstractController):
|
||||
cursor = self._cursor()
|
||||
cursor.execute(query)
|
||||
return cursor
|
||||
except psycopg2.Error:
|
||||
except psycopg.Error:
|
||||
if not fail_ok:
|
||||
raise
|
||||
|
||||
@@ -268,7 +269,7 @@ class PatroniController(AbstractController):
|
||||
|
||||
@property
|
||||
def backup_source(self):
|
||||
return 'postgres://{username}:{password}@{host}:{port}/{database}'.format(**self._replication)
|
||||
return 'postgres://{username}:{password}@{host}:{port}/{dbname}'.format(**self._replication)
|
||||
|
||||
def backup(self, dest=os.path.join('data', 'basebackup')):
|
||||
subprocess.call(PatroniPoolController.BACKUP_SCRIPT + ['--walmethod=none',
|
||||
@@ -659,7 +660,7 @@ class PatroniPoolController(object):
|
||||
def output_dir(self):
|
||||
return self._output_dir
|
||||
|
||||
def start(self, name, max_wait_limit=20, custom_config=None):
|
||||
def start(self, name, max_wait_limit=40, custom_config=None):
|
||||
if name not in self._processes:
|
||||
self._processes[name] = PatroniController(self._context, name, self.patroni_path,
|
||||
self._output_dir, custom_config)
|
||||
|
||||
@@ -51,11 +51,11 @@ Scenario: check the scheduled restart
|
||||
Given I issue a PATCH request to http://127.0.0.1:8008/config with {"postgresql": {"parameters": {"superuser_reserved_connections": "6"}}}
|
||||
Then I receive a response code 200
|
||||
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 5 seconds
|
||||
Given I issue a scheduled restart at http://127.0.0.1:8008 in 3 seconds with {"role": "replica"}
|
||||
Given I issue a scheduled restart at http://127.0.0.1:8008 in 5 seconds with {"role": "replica"}
|
||||
Then I receive a response code 202
|
||||
And I sleep for 4 seconds
|
||||
And I sleep for 8 seconds
|
||||
And Response on GET http://127.0.0.1:8008/patroni contains pending_restart after 10 seconds
|
||||
Given I issue a scheduled restart at http://127.0.0.1:8008 in 3 seconds with {"restart_pending": "True"}
|
||||
Given I issue a scheduled restart at http://127.0.0.1:8008 in 5 seconds with {"restart_pending": "True"}
|
||||
Then I receive a response code 202
|
||||
And Response on GET http://127.0.0.1:8008/patroni does not contain pending_restart after 10 seconds
|
||||
And postgres0 role is the primary after 10 seconds
|
||||
@@ -71,6 +71,7 @@ Scenario: check API requests for the primary-replica pair in the pause mode
|
||||
When I run patronictl.py restart batman postgres1 --force
|
||||
Then I receive a response returncode 0
|
||||
Then replication works from postgres0 to postgres1 after 20 seconds
|
||||
And I sleep for 2 seconds
|
||||
When I issue a GET request to http://127.0.0.1:8009/replica
|
||||
Then I receive a response code 200
|
||||
And I receive a response state running
|
||||
@@ -103,12 +104,12 @@ Scenario: check the switchover via the API in the pause mode
|
||||
Then I receive a response code 503
|
||||
|
||||
Scenario: check the scheduled switchover
|
||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 3 seconds
|
||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 10 seconds
|
||||
Then I receive a response returncode 1
|
||||
And I receive a response output "Can't schedule switchover in the paused state"
|
||||
When I run patronictl.py resume batman
|
||||
Then I receive a response returncode 0
|
||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 3 seconds
|
||||
Given I issue a scheduled switchover from postgres1 to postgres0 in 5 seconds
|
||||
Then I receive a response returncode 0
|
||||
And postgres0 is a leader after 20 seconds
|
||||
And postgres0 role is the primary after 10 seconds
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import psycopg2 as pg
|
||||
import patroni.psycopg as pg
|
||||
|
||||
from behave import step, then
|
||||
from time import sleep, time
|
||||
|
||||
@@ -76,6 +76,8 @@ def do_request(context, request_method, url, data):
|
||||
data = data and json.loads(data)
|
||||
try:
|
||||
r = request_executor.request(request_method, url, data)
|
||||
if request_method == 'PATCH' and r.status == 409:
|
||||
r = request_executor.request(request_method, url, data)
|
||||
except Exception:
|
||||
context.status_code = context.response = None
|
||||
else:
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import time
|
||||
import psycopg2
|
||||
|
||||
from behave import step, then
|
||||
import patroni.psycopg as pg
|
||||
|
||||
|
||||
@step('I create a logical replication slot {slot_name} on {pg_name:w} with the {plugin:w} plugin')
|
||||
@@ -10,7 +10,7 @@ def create_logical_replication_slot(context, slot_name, pg_name, plugin):
|
||||
output = context.pctl.query(pg_name, ("SELECT pg_create_logical_replication_slot('{0}', '{1}'),"
|
||||
" current_database()").format(slot_name, plugin))
|
||||
print(output.fetchone())
|
||||
except psycopg2.Error as e:
|
||||
except pg.Error as e:
|
||||
print(e)
|
||||
assert False, "Error creating slot {0} on {1} with plugin {2}".format(slot_name, pg_name, plugin)
|
||||
|
||||
@@ -24,7 +24,7 @@ def has_logical_replication_slot(context, pg_name, slot_name, plugin):
|
||||
assert row[0] == "logical", "Found replication slot named {0} but wasn't a logical slot".format(slot_name)
|
||||
assert row[1] == plugin, ("Found replication slot named {0} but was using plugin "
|
||||
"{1} rather than {2}").format(slot_name, row[1], plugin)
|
||||
except psycopg2.Error:
|
||||
except pg.Error:
|
||||
assert False, "Error looking for slot {0} on {1} with plugin {2}".format(slot_name, pg_name, plugin)
|
||||
|
||||
|
||||
@@ -34,7 +34,7 @@ def does_not_have_logical_replication_slot(context, pg_name, slot_name):
|
||||
row = context.pctl.query(pg_name, ("SELECT 1 FROM pg_replication_slots"
|
||||
" WHERE slot_name = '{0}'").format(slot_name)).fetchone()
|
||||
assert not row, "Found unexpected replication slot named {0}".format(slot_name)
|
||||
except psycopg2.Error:
|
||||
except pg.Error:
|
||||
assert False, "Error looking for slot {0} on {1}".format(slot_name, pg_name)
|
||||
|
||||
|
||||
|
||||
@@ -57,7 +57,7 @@ def start_patroni_standby_cluster(context, name, cluster_name, name2):
|
||||
|
||||
@step('{pg_name1:w} is replicating from {pg_name2:w} after {timeout:d} seconds')
|
||||
def check_replication_status(context, pg_name1, pg_name2, timeout):
|
||||
bound_time = time.time() + timeout
|
||||
bound_time = time.time() + timeout * context.timeout_multiplier
|
||||
|
||||
while time.time() < bound_time:
|
||||
cur = context.pctl.query(
|
||||
|
||||
@@ -20,6 +20,10 @@ metadata:
|
||||
spec:
|
||||
replicas: 3
|
||||
serviceName: *cluster_name
|
||||
selector:
|
||||
matchLabels:
|
||||
application: patroni
|
||||
cluster-name: *cluster_name
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
|
||||
+1
-1
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/env python
|
||||
from patroni import main
|
||||
from patroni.__main__ import main
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
+21
-186
@@ -1,142 +1,8 @@
|
||||
import logging
|
||||
import os
|
||||
import signal
|
||||
import sys
|
||||
import time
|
||||
|
||||
from .daemon import AbstractPatroniDaemon, abstract_main
|
||||
from .version import __version__
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
PATRONI_ENV_PREFIX = 'PATRONI_'
|
||||
KUBERNETES_ENV_PREFIX = 'KUBERNETES_'
|
||||
|
||||
|
||||
class Patroni(AbstractPatroniDaemon):
|
||||
|
||||
def __init__(self, config):
|
||||
from patroni.api import RestApiServer
|
||||
from patroni.dcs import get_dcs
|
||||
from patroni.ha import Ha
|
||||
from patroni.postgresql import Postgresql
|
||||
from patroni.request import PatroniRequest
|
||||
from patroni.watchdog import Watchdog
|
||||
|
||||
super(Patroni, self).__init__(config)
|
||||
|
||||
self.version = __version__
|
||||
self.dcs = get_dcs(self.config)
|
||||
self.watchdog = Watchdog(self.config)
|
||||
self.load_dynamic_configuration()
|
||||
|
||||
self.postgresql = Postgresql(self.config['postgresql'])
|
||||
self.api = RestApiServer(self, self.config['restapi'])
|
||||
self.request = PatroniRequest(self.config, True)
|
||||
self.ha = Ha(self)
|
||||
|
||||
self.tags = self.get_tags()
|
||||
self.next_run = time.time()
|
||||
self.scheduled_restart = {}
|
||||
|
||||
def load_dynamic_configuration(self):
|
||||
from patroni.exceptions import DCSError
|
||||
while True:
|
||||
try:
|
||||
cluster = self.dcs.get_cluster()
|
||||
if cluster and cluster.config and cluster.config.data:
|
||||
if self.config.set_dynamic_configuration(cluster.config):
|
||||
self.dcs.reload_config(self.config)
|
||||
self.watchdog.reload_config(self.config)
|
||||
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
||||
if self.config.set_dynamic_configuration(self.config['bootstrap']['dcs']):
|
||||
self.dcs.reload_config(self.config)
|
||||
break
|
||||
except DCSError:
|
||||
logger.warning('Can not get cluster from dcs')
|
||||
time.sleep(5)
|
||||
|
||||
def get_tags(self):
|
||||
return {tag: value for tag, value in self.config.get('tags', {}).items()
|
||||
if tag not in ('clonefrom', 'nofailover', 'noloadbalance', 'nosync') or value}
|
||||
|
||||
@property
|
||||
def nofailover(self):
|
||||
return bool(self.tags.get('nofailover', False))
|
||||
|
||||
@property
|
||||
def nosync(self):
|
||||
return bool(self.tags.get('nosync', False))
|
||||
|
||||
def reload_config(self, sighup=False, local=False):
|
||||
try:
|
||||
super(Patroni, self).reload_config(sighup, local)
|
||||
if local:
|
||||
self.tags = self.get_tags()
|
||||
self.request.reload_config(self.config)
|
||||
if local or self.api.reload_local_certificate():
|
||||
self.api.reload_config(self.config['restapi'])
|
||||
self.watchdog.reload_config(self.config)
|
||||
self.postgresql.reload_config(self.config['postgresql'], sighup)
|
||||
self.dcs.reload_config(self.config)
|
||||
except Exception:
|
||||
logger.exception('Failed to reload config_file=%s', self.config.config_file)
|
||||
|
||||
@property
|
||||
def replicatefrom(self):
|
||||
return self.tags.get('replicatefrom')
|
||||
|
||||
@property
|
||||
def noloadbalance(self):
|
||||
return bool(self.tags.get('noloadbalance', False))
|
||||
|
||||
def schedule_next_run(self):
|
||||
self.next_run += self.dcs.loop_wait
|
||||
current_time = time.time()
|
||||
nap_time = self.next_run - current_time
|
||||
if nap_time <= 0:
|
||||
self.next_run = current_time
|
||||
# Release the GIL so we don't starve anyone waiting on async_executor lock
|
||||
time.sleep(0.001)
|
||||
# Warn user that Patroni is not keeping up
|
||||
logger.warning("Loop time exceeded, rescheduling immediately.")
|
||||
elif self.ha.watch(nap_time):
|
||||
self.next_run = time.time()
|
||||
|
||||
def run(self):
|
||||
self.api.start()
|
||||
self.next_run = time.time()
|
||||
super(Patroni, self).run()
|
||||
|
||||
def _run_cycle(self):
|
||||
logger.info(self.ha.run_cycle())
|
||||
|
||||
if self.dcs.cluster and self.dcs.cluster.config and self.dcs.cluster.config.data \
|
||||
and self.config.set_dynamic_configuration(self.dcs.cluster.config):
|
||||
self.reload_config()
|
||||
|
||||
if self.postgresql.role != 'uninitialized':
|
||||
self.config.save_cache()
|
||||
|
||||
self.schedule_next_run()
|
||||
|
||||
def _shutdown(self):
|
||||
try:
|
||||
self.api.shutdown()
|
||||
except Exception:
|
||||
logger.exception('Exception during RestApi.shutdown')
|
||||
try:
|
||||
self.ha.shutdown()
|
||||
except Exception:
|
||||
logger.exception('Exception during Ha.shutdown')
|
||||
|
||||
|
||||
def patroni_main():
|
||||
from multiprocessing import freeze_support
|
||||
from patroni.validator import schema
|
||||
|
||||
freeze_support()
|
||||
abstract_main(Patroni, schema)
|
||||
MIN_PSYCOPG2 = (2, 5, 4)
|
||||
|
||||
|
||||
def fatal(string, *args):
|
||||
@@ -144,63 +10,32 @@ def fatal(string, *args):
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def check_psycopg2():
|
||||
min_psycopg2 = (2, 5, 4)
|
||||
min_psycopg2_str = '.'.join(map(str, min_psycopg2))
|
||||
|
||||
def parse_version(version):
|
||||
def parse_version(version):
|
||||
def _parse_version(version):
|
||||
for e in version.split('.'):
|
||||
try:
|
||||
yield int(e)
|
||||
except ValueError:
|
||||
break
|
||||
return tuple(_parse_version(version.split(' ')[0]))
|
||||
|
||||
|
||||
# We pass MIN_PSYCOPG2 and parse_version as arguments to simplify usage of check_psycopg from the setup.py
|
||||
def check_psycopg(_min_psycopg2=MIN_PSYCOPG2, _parse_version=parse_version):
|
||||
min_psycopg2_str = '.'.join(map(str, _min_psycopg2))
|
||||
|
||||
try:
|
||||
import psycopg2
|
||||
version_str = psycopg2.__version__.split(' ')[0]
|
||||
version = tuple(parse_version(version_str))
|
||||
if version < min_psycopg2:
|
||||
fatal('Patroni requires psycopg2>={0}, but only {1} is available', min_psycopg2_str, version_str)
|
||||
from psycopg2 import __version__
|
||||
if _parse_version(__version__) >= _min_psycopg2:
|
||||
return
|
||||
version_str = __version__.split(' ')[0]
|
||||
except ImportError:
|
||||
fatal('Patroni requires psycopg2>={0} or psycopg2-binary', min_psycopg2_str)
|
||||
version_str = None
|
||||
|
||||
|
||||
def main():
|
||||
if os.getpid() != 1:
|
||||
check_psycopg2()
|
||||
return patroni_main()
|
||||
|
||||
# Patroni started with PID=1, it looks like we are in the container
|
||||
pid = 0
|
||||
|
||||
# Looks like we are in a docker, so we will act like init
|
||||
def sigchld_handler(signo, stack_frame):
|
||||
try:
|
||||
while True:
|
||||
ret = os.waitpid(-1, os.WNOHANG)
|
||||
if ret == (0, 0):
|
||||
break
|
||||
elif ret[0] != pid:
|
||||
logger.info('Reaped pid=%s, exit status=%s', *ret)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
def passtochild(signo, stack_frame):
|
||||
if pid:
|
||||
os.kill(pid, signo)
|
||||
|
||||
if os.name != 'nt':
|
||||
signal.signal(signal.SIGCHLD, sigchld_handler)
|
||||
signal.signal(signal.SIGHUP, passtochild)
|
||||
signal.signal(signal.SIGQUIT, passtochild)
|
||||
signal.signal(signal.SIGUSR1, passtochild)
|
||||
signal.signal(signal.SIGUSR2, passtochild)
|
||||
signal.signal(signal.SIGINT, passtochild)
|
||||
signal.signal(signal.SIGABRT, passtochild)
|
||||
signal.signal(signal.SIGTERM, passtochild)
|
||||
|
||||
import multiprocessing
|
||||
patroni = multiprocessing.Process(target=patroni_main)
|
||||
patroni.start()
|
||||
pid = patroni.pid
|
||||
patroni.join()
|
||||
try:
|
||||
from psycopg import __version__
|
||||
except ImportError:
|
||||
error = 'Patroni requires psycopg2>={0}, psycopg2-binary, or psycopg>=3.0'.format(min_psycopg2_str)
|
||||
if version_str:
|
||||
error += ', but only psycopg2=={0} is available'.format(version_str)
|
||||
fatal(error)
|
||||
|
||||
+178
-1
@@ -1,4 +1,181 @@
|
||||
from patroni import main
|
||||
import logging
|
||||
import os
|
||||
import signal
|
||||
import time
|
||||
|
||||
from .daemon import AbstractPatroniDaemon, abstract_main
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Patroni(AbstractPatroniDaemon):
|
||||
|
||||
def __init__(self, config):
|
||||
from .api import RestApiServer
|
||||
from .dcs import get_dcs
|
||||
from .ha import Ha
|
||||
from .postgresql import Postgresql
|
||||
from .request import PatroniRequest
|
||||
from .version import __version__
|
||||
from .watchdog import Watchdog
|
||||
|
||||
super(Patroni, self).__init__(config)
|
||||
|
||||
self.version = __version__
|
||||
self.dcs = get_dcs(self.config)
|
||||
self.watchdog = Watchdog(self.config)
|
||||
self.load_dynamic_configuration()
|
||||
|
||||
self.postgresql = Postgresql(self.config['postgresql'])
|
||||
self.api = RestApiServer(self, self.config['restapi'])
|
||||
self.request = PatroniRequest(self.config, True)
|
||||
self.ha = Ha(self)
|
||||
|
||||
self.tags = self.get_tags()
|
||||
self.next_run = time.time()
|
||||
self.scheduled_restart = {}
|
||||
|
||||
def load_dynamic_configuration(self):
|
||||
from patroni.exceptions import DCSError
|
||||
while True:
|
||||
try:
|
||||
cluster = self.dcs.get_cluster()
|
||||
if cluster and cluster.config and cluster.config.data:
|
||||
if self.config.set_dynamic_configuration(cluster.config):
|
||||
self.dcs.reload_config(self.config)
|
||||
self.watchdog.reload_config(self.config)
|
||||
elif not self.config.dynamic_configuration and 'bootstrap' in self.config:
|
||||
if self.config.set_dynamic_configuration(self.config['bootstrap']['dcs']):
|
||||
self.dcs.reload_config(self.config)
|
||||
break
|
||||
except DCSError:
|
||||
logger.warning('Can not get cluster from dcs')
|
||||
time.sleep(5)
|
||||
|
||||
def get_tags(self):
|
||||
return {tag: value for tag, value in self.config.get('tags', {}).items()
|
||||
if tag not in ('clonefrom', 'nofailover', 'noloadbalance', 'nosync') or value}
|
||||
|
||||
@property
|
||||
def nofailover(self):
|
||||
return bool(self.tags.get('nofailover', False))
|
||||
|
||||
@property
|
||||
def nosync(self):
|
||||
return bool(self.tags.get('nosync', False))
|
||||
|
||||
def reload_config(self, sighup=False, local=False):
|
||||
try:
|
||||
super(Patroni, self).reload_config(sighup, local)
|
||||
if local:
|
||||
self.tags = self.get_tags()
|
||||
self.request.reload_config(self.config)
|
||||
if local or sighup and self.api.reload_local_certificate():
|
||||
self.api.reload_config(self.config['restapi'])
|
||||
self.watchdog.reload_config(self.config)
|
||||
self.postgresql.reload_config(self.config['postgresql'], sighup)
|
||||
self.dcs.reload_config(self.config)
|
||||
except Exception:
|
||||
logger.exception('Failed to reload config_file=%s', self.config.config_file)
|
||||
|
||||
@property
|
||||
def replicatefrom(self):
|
||||
return self.tags.get('replicatefrom')
|
||||
|
||||
@property
|
||||
def noloadbalance(self):
|
||||
return bool(self.tags.get('noloadbalance', False))
|
||||
|
||||
def schedule_next_run(self):
|
||||
self.next_run += self.dcs.loop_wait
|
||||
current_time = time.time()
|
||||
nap_time = self.next_run - current_time
|
||||
if nap_time <= 0:
|
||||
self.next_run = current_time
|
||||
# Release the GIL so we don't starve anyone waiting on async_executor lock
|
||||
time.sleep(0.001)
|
||||
# Warn user that Patroni is not keeping up
|
||||
logger.warning("Loop time exceeded, rescheduling immediately.")
|
||||
elif self.ha.watch(nap_time):
|
||||
self.next_run = time.time()
|
||||
|
||||
def run(self):
|
||||
self.api.start()
|
||||
self.next_run = time.time()
|
||||
super(Patroni, self).run()
|
||||
|
||||
def _run_cycle(self):
|
||||
logger.info(self.ha.run_cycle())
|
||||
|
||||
if self.dcs.cluster and self.dcs.cluster.config and self.dcs.cluster.config.data \
|
||||
and self.config.set_dynamic_configuration(self.dcs.cluster.config):
|
||||
self.reload_config()
|
||||
|
||||
if self.postgresql.role != 'uninitialized':
|
||||
self.config.save_cache()
|
||||
|
||||
self.schedule_next_run()
|
||||
|
||||
def _shutdown(self):
|
||||
try:
|
||||
self.api.shutdown()
|
||||
except Exception:
|
||||
logger.exception('Exception during RestApi.shutdown')
|
||||
try:
|
||||
self.ha.shutdown()
|
||||
except Exception:
|
||||
logger.exception('Exception during Ha.shutdown')
|
||||
|
||||
|
||||
def patroni_main():
|
||||
from multiprocessing import freeze_support
|
||||
from patroni.validator import schema
|
||||
|
||||
freeze_support()
|
||||
abstract_main(Patroni, schema)
|
||||
|
||||
|
||||
def main():
|
||||
if os.getpid() != 1:
|
||||
from . import check_psycopg
|
||||
|
||||
check_psycopg()
|
||||
return patroni_main()
|
||||
|
||||
# Patroni started with PID=1, it looks like we are in the container
|
||||
pid = 0
|
||||
|
||||
# Looks like we are in a docker, so we will act like init
|
||||
def sigchld_handler(signo, stack_frame):
|
||||
try:
|
||||
while True:
|
||||
ret = os.waitpid(-1, os.WNOHANG)
|
||||
if ret == (0, 0):
|
||||
break
|
||||
elif ret[0] != pid:
|
||||
logger.info('Reaped pid=%s, exit status=%s', *ret)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
def passtochild(signo, stack_frame):
|
||||
if pid:
|
||||
os.kill(pid, signo)
|
||||
|
||||
if os.name != 'nt':
|
||||
signal.signal(signal.SIGCHLD, sigchld_handler)
|
||||
signal.signal(signal.SIGHUP, passtochild)
|
||||
signal.signal(signal.SIGQUIT, passtochild)
|
||||
signal.signal(signal.SIGUSR1, passtochild)
|
||||
signal.signal(signal.SIGUSR2, passtochild)
|
||||
signal.signal(signal.SIGINT, passtochild)
|
||||
signal.signal(signal.SIGABRT, passtochild)
|
||||
signal.signal(signal.SIGTERM, passtochild)
|
||||
|
||||
import multiprocessing
|
||||
patroni = multiprocessing.Process(target=patroni_main)
|
||||
patroni.start()
|
||||
pid = patroni.pid
|
||||
patroni.join()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
+36
-11
@@ -2,7 +2,6 @@ import base64
|
||||
import hmac
|
||||
import json
|
||||
import logging
|
||||
import psycopg2
|
||||
import time
|
||||
import traceback
|
||||
import dateutil.parser
|
||||
@@ -18,6 +17,7 @@ from six.moves.socketserver import ThreadingMixIn
|
||||
from six.moves.urllib_parse import urlparse, parse_qs
|
||||
from threading import Thread
|
||||
|
||||
from . import psycopg
|
||||
from .exceptions import PostgresConnectionException, PostgresException
|
||||
from .postgresql.misc import postgres_version_to_int
|
||||
from .utils import deep_compare, enable_keepalive, parse_bool, patch_config, Retry, \
|
||||
@@ -152,6 +152,11 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
status_code = replica_status_code
|
||||
elif path in ('/async', '/asynchronous') and not is_synchronous:
|
||||
status_code = replica_status_code
|
||||
elif path in ('/read-only-sync', '/read-only-synchronous'):
|
||||
if 200 in (primary_status_code, standby_leader_status_code):
|
||||
status_code = 200
|
||||
elif is_synchronous:
|
||||
status_code = replica_status_code
|
||||
|
||||
# check for user defined tags in query params
|
||||
if not ignore_tags and status_code == 200:
|
||||
@@ -282,12 +287,27 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
|
||||
metrics.append("# HELP patroni_cluster_unlocked Value is 1 if the cluster is unlocked, 0 if locked.")
|
||||
metrics.append("# TYPE patroni_cluster_unlocked gauge")
|
||||
metrics.append("patroni_cluster_unlocked{0} {1}".format(scope_label, int(postgres['cluster_unlocked'])))
|
||||
metrics.append("patroni_cluster_unlocked{0} {1}".format(scope_label, int(postgres.get('cluster_unlocked', 0))))
|
||||
|
||||
metrics.append("# HELP patroni_postgres_timeline Postgres timeline of this node (if running), 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_postgres_timeline counter")
|
||||
metrics.append("patroni_postgres_timeline{0} {1}".format(scope_label, postgres.get('timeline', 0)))
|
||||
|
||||
metrics.append("# HELP patroni_dcs_last_seen Epoch timestamp when DCS was last contacted successfully"
|
||||
" by Patroni.")
|
||||
metrics.append("# TYPE patroni_dcs_last_seen gauge")
|
||||
metrics.append("patroni_dcs_last_seen{0} {1}".format(scope_label, postgres.get('dcs_last_seen', 0)))
|
||||
|
||||
metrics.append("# HELP patroni_pending_restart Value is 1 if the node needs a restart, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_pending_restart gauge")
|
||||
metrics.append("patroni_pending_restart{0} {1}"
|
||||
.format(scope_label, int(patroni.postgresql.pending_restart)))
|
||||
|
||||
metrics.append("# HELP patroni_is_paused Value is 1 if auto failover is disabled, 0 otherwise.")
|
||||
metrics.append("# TYPE patroni_is_paused gauge")
|
||||
metrics.append("patroni_is_paused{0} {1}"
|
||||
.format(scope_label, int(patroni.ha.is_paused())))
|
||||
|
||||
self._write_response(200, '\n'.join(metrics)+'\n', content_type='text/plain')
|
||||
|
||||
def _read_json_content(self, body_is_optional=False):
|
||||
@@ -599,7 +619,6 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
'postmaster_start_time': row[0],
|
||||
'role': 'replica' if row[1] == 0 else 'master',
|
||||
'server_version': postgresql.server_version,
|
||||
'cluster_unlocked': bool(not cluster or cluster.is_unlocked()),
|
||||
'xlog': ({
|
||||
'received_location': row[4] or row[3],
|
||||
'replayed_location': row[3],
|
||||
@@ -621,13 +640,17 @@ class RestApiHandler(BaseHTTPRequestHandler):
|
||||
if row[7]:
|
||||
result['replication'] = row[7]
|
||||
|
||||
return result
|
||||
except (psycopg2.Error, RetryFailedError, PostgresConnectionException):
|
||||
except (psycopg.Error, RetryFailedError, PostgresConnectionException):
|
||||
state = postgresql.state
|
||||
if state == 'running':
|
||||
logger.exception('get_postgresql_status')
|
||||
state = 'unknown'
|
||||
return {'state': state, 'role': postgresql.role}
|
||||
result = {'state': state, 'role': postgresql.role}
|
||||
|
||||
if not cluster or cluster.is_unlocked():
|
||||
result['cluster_unlocked'] = True
|
||||
result['dcs_last_seen'] = self.server.patroni.dcs.last_seen
|
||||
return result
|
||||
|
||||
def handle_one_request(self):
|
||||
self.__start_time = time.time()
|
||||
@@ -646,10 +669,10 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
||||
self.patroni = patroni
|
||||
self.__listen = None
|
||||
self.__ssl_options = None
|
||||
self.reload_config(config)
|
||||
self.daemon = True
|
||||
self.__ssl_serial_number = None
|
||||
self._received_new_cert = False
|
||||
self.reload_config(config)
|
||||
self.daemon = True
|
||||
|
||||
def query(self, sql, *params):
|
||||
cursor = None
|
||||
@@ -657,7 +680,7 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
||||
with self.patroni.postgresql.connection().cursor() as cursor:
|
||||
cursor.execute(sql, params)
|
||||
return [r for r in cursor]
|
||||
except psycopg2.Error as e:
|
||||
except psycopg.Error as e:
|
||||
if cursor and cursor.connection.closed == 0:
|
||||
raise e
|
||||
raise PostgresConnectionException('connection problems')
|
||||
@@ -760,6 +783,8 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
||||
reloading_config = self.__listen is not None # changing config in runtime
|
||||
if reloading_config:
|
||||
self.shutdown()
|
||||
# Rely on ThreadingMixIn.server_close() to have all requests terminate before we continue
|
||||
self.server_close()
|
||||
|
||||
self.__listen = listen
|
||||
self.__ssl_options = ssl_options
|
||||
@@ -867,6 +892,6 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread):
|
||||
|
||||
@staticmethod
|
||||
def handle_error(request, client_address):
|
||||
address, port = client_address
|
||||
logger.warning('Exception happened during processing of request from {}:{}'.format(address, port))
|
||||
logger.warning('Exception happened during processing of request from %s:%s',
|
||||
client_address[0], client_address[1])
|
||||
logger.warning(traceback.format_exc())
|
||||
|
||||
+7
-6
@@ -24,6 +24,7 @@ _AUTH_ALLOWED_PARAMETERS = (
|
||||
'sslpassword',
|
||||
'sslrootcert',
|
||||
'sslcrl',
|
||||
'sslcrldir',
|
||||
'gssencmode',
|
||||
'channel_binding'
|
||||
)
|
||||
@@ -269,7 +270,7 @@ class Config(object):
|
||||
_set_section_values('restapi', ['listen', 'connect_address', 'certfile', 'keyfile', 'keyfile_password',
|
||||
'cafile', 'ciphers', 'verify_client', 'http_extra_headers',
|
||||
'https_extra_headers', 'allowlist', 'allowlist_include_members'])
|
||||
_set_section_values('ctl', ['insecure', 'cacert', 'certfile', 'keyfile'])
|
||||
_set_section_values('ctl', ['insecure', 'cacert', 'certfile', 'keyfile', 'keyfile_password'])
|
||||
_set_section_values('postgresql', ['listen', 'connect_address', 'config_dir', 'data_dir', 'pgpass', 'bin_dir'])
|
||||
_set_section_values('log', ['level', 'traceback_level', 'format', 'dateformat', 'max_queue_size',
|
||||
'dir', 'file_size', 'file_num', 'loggers'])
|
||||
@@ -347,17 +348,17 @@ class Config(object):
|
||||
if param.startswith(PATRONI_ENV_PREFIX):
|
||||
# PATRONI_(ETCD|CONSUL|ZOOKEEPER|EXHIBITOR|...)_(HOSTS?|PORT|..)
|
||||
name, suffix = (param[8:].split('_', 1) + [''])[:2]
|
||||
if suffix in ('HOST', 'HOSTS', 'PORT', 'USE_PROXIES', 'PROTOCOL', 'SRV', 'URL', 'PROXY',
|
||||
if suffix in ('HOST', 'HOSTS', 'PORT', 'USE_PROXIES', 'PROTOCOL', 'SRV', 'SRV_SUFFIX', 'URL', 'PROXY',
|
||||
'CACERT', 'CERT', 'KEY', 'VERIFY', 'TOKEN', 'CHECKS', 'DC', 'CONSISTENCY',
|
||||
'REGISTER_SERVICE', 'SERVICE_CHECK_INTERVAL', 'NAMESPACE', 'CONTEXT',
|
||||
'USE_ENDPOINTS', 'SCOPE_LABEL', 'ROLE_LABEL', 'POD_IP', 'PORTS', 'LABELS',
|
||||
'BYPASS_API_SERVICE', 'KEY_PASSWORD', 'USE_SSL') and name:
|
||||
'REGISTER_SERVICE', 'SERVICE_CHECK_INTERVAL', 'SERVICE_CHECK_TLS_SERVER_NAME',
|
||||
'NAMESPACE', 'CONTEXT', 'USE_ENDPOINTS', 'SCOPE_LABEL', 'ROLE_LABEL', 'POD_IP',
|
||||
'PORTS', 'LABELS', 'BYPASS_API_SERVICE', 'KEY_PASSWORD', 'USE_SSL', 'SET_ACLS') and name:
|
||||
value = os.environ.pop(param)
|
||||
if suffix == 'PORT':
|
||||
value = value and parse_int(value)
|
||||
elif suffix in ('HOSTS', 'PORTS', 'CHECKS'):
|
||||
value = value and _parse_list(value)
|
||||
elif suffix == 'LABELS':
|
||||
elif suffix in ('LABELS', 'SET_ACLS'):
|
||||
value = _parse_dict(value)
|
||||
elif suffix in ('USE_PROXIES', 'REGISTER_SERVICE', 'USE_ENDPOINTS', 'BYPASS_API_SERVICE', 'VERIFY'):
|
||||
value = parse_bool(value)
|
||||
|
||||
+16
-13
@@ -264,13 +264,13 @@ def get_cursor(cluster, connect_parameters, role='master', member=None):
|
||||
|
||||
params = member.conn_kwargs(connect_parameters)
|
||||
params.update({'fallback_application_name': 'Patroni ctl', 'connect_timeout': '5'})
|
||||
if 'database' in connect_parameters:
|
||||
params['database'] = connect_parameters['database']
|
||||
if 'dbname' in connect_parameters:
|
||||
params['dbname'] = connect_parameters['dbname']
|
||||
else:
|
||||
params.pop('database')
|
||||
params.pop('dbname')
|
||||
|
||||
import psycopg2
|
||||
conn = psycopg2.connect(**params)
|
||||
from . import psycopg
|
||||
conn = psycopg.connect(**params)
|
||||
conn.autocommit = True
|
||||
cursor = conn.cursor()
|
||||
if role == 'any':
|
||||
@@ -401,7 +401,7 @@ def query(
|
||||
if password:
|
||||
connect_parameters['password'] = click.prompt('Password', hide_input=True, type=str)
|
||||
if dbname:
|
||||
connect_parameters['database'] = dbname
|
||||
connect_parameters['dbname'] = dbname
|
||||
|
||||
if p_file is not None:
|
||||
command = p_file.read()
|
||||
@@ -418,7 +418,7 @@ def query(
|
||||
|
||||
|
||||
def query_member(cluster, cursor, member, role, command, connect_parameters):
|
||||
import psycopg2
|
||||
from . import psycopg
|
||||
try:
|
||||
if cursor is None:
|
||||
cursor = get_cursor(cluster, connect_parameters, role=role, member=member)
|
||||
@@ -433,11 +433,11 @@ def query_member(cluster, cursor, member, role, command, connect_parameters):
|
||||
|
||||
cursor.execute(command)
|
||||
return cursor.fetchall(), [d.name for d in cursor.description]
|
||||
except (psycopg2.OperationalError, psycopg2.DatabaseError) as oe:
|
||||
logging.debug(oe)
|
||||
except psycopg.DatabaseError as de:
|
||||
logging.debug(de)
|
||||
if cursor is not None and not cursor.connection.closed:
|
||||
cursor.connection.close()
|
||||
message = oe.pgcode or oe.pgerror or str(oe)
|
||||
message = de.diag.sqlstate or str(de)
|
||||
message = message.replace('\n', ' ')
|
||||
return [[timestamp(0), 'ERROR, SQLSTATE: {0}'.format(message)]], None
|
||||
|
||||
@@ -1299,10 +1299,13 @@ def version(obj, cluster_name, member_names):
|
||||
def history(obj, cluster_name, fmt):
|
||||
cluster = get_dcs(obj, cluster_name).get_cluster()
|
||||
history = cluster.history and cluster.history.lines or []
|
||||
table_header_row = ['TL', 'LSN', 'Reason', 'Timestamp', 'New Leader']
|
||||
for line in history:
|
||||
if len(line) < 4:
|
||||
line.append('')
|
||||
print_output(['TL', 'LSN', 'Reason', 'Timestamp'], history, {'TL': 'r', 'LSN': 'r'}, fmt)
|
||||
if len(line) < len(table_header_row):
|
||||
add_column_num = len(table_header_row) - len(line)
|
||||
for _ in range(add_column_num):
|
||||
line.append('')
|
||||
print_output(table_header_row, history, {'TL': 'r', 'LSN': 'r'}, fmt)
|
||||
|
||||
|
||||
def format_pg_version(version):
|
||||
|
||||
+23
-12
@@ -1,5 +1,5 @@
|
||||
import abc
|
||||
import dateutil
|
||||
import dateutil.parser
|
||||
import importlib
|
||||
import inspect
|
||||
import json
|
||||
@@ -160,7 +160,7 @@ class Member(namedtuple('Member', 'index,name,session,data')):
|
||||
defaults = {
|
||||
"host": None,
|
||||
"port": None,
|
||||
"database": None
|
||||
"dbname": None
|
||||
}
|
||||
ret = self.data.get('conn_kwargs')
|
||||
if ret:
|
||||
@@ -174,7 +174,7 @@ class Member(namedtuple('Member', 'index,name,session,data')):
|
||||
ret = {
|
||||
'host': r.hostname,
|
||||
'port': r.port or 5432,
|
||||
'database': r.path[1:]
|
||||
'dbname': r.path[1:]
|
||||
}
|
||||
self.data['conn_kwargs'] = ret.copy()
|
||||
|
||||
@@ -460,8 +460,12 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_lsn,members,f
|
||||
:param slots: state of permanent logical replication slots on the primary in the format: {"slot_name": int}
|
||||
"""
|
||||
|
||||
@property
|
||||
def leader_name(self):
|
||||
return self.leader and self.leader.name
|
||||
|
||||
def is_unlocked(self):
|
||||
return not (self.leader and self.leader.name)
|
||||
return not self.leader_name
|
||||
|
||||
def has_member(self, member_name):
|
||||
return any(m for m in self.members if m.name == member_name)
|
||||
@@ -499,7 +503,7 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_lsn,members,f
|
||||
|
||||
@property
|
||||
def use_slots(self):
|
||||
return self.config and self.config.data.get('postgresql', {}).get('use_slots', True)
|
||||
return self.config and (self.config.data.get('postgresql') or {}).get('use_slots', True)
|
||||
|
||||
def get_replication_slots(self, my_name, role, nofailover, major_version, show_error=False):
|
||||
# if the replicatefrom tag is set on the member - we should not create the replication slot for it on
|
||||
@@ -516,7 +520,7 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_lsn,members,f
|
||||
else:
|
||||
# only manage slots for replicas that replicate from this one, except for the leader among them
|
||||
slot_members = [m.name for m in self.members if use_slots and
|
||||
m.replicatefrom == my_name and m.name != self.leader.name]
|
||||
m.replicatefrom == my_name and m.name != self.leader_name]
|
||||
permanent_slots = self.__permanent_logical_slots if use_slots and not nofailover else {}
|
||||
|
||||
slots = {slot_name_from_member_name(name): {'type': 'physical'} for name in slot_members}
|
||||
@@ -585,7 +589,7 @@ class Cluster(namedtuple('Cluster', 'initialize,config,leader,last_lsn,members,f
|
||||
return True
|
||||
|
||||
if self.use_slots:
|
||||
members = [m for m in self.members if m.replicatefrom == my_name and m.name != self.leader.name]
|
||||
members = [m for m in self.members if m.replicatefrom == my_name and m.name != self.leader_name]
|
||||
return any(self.should_enforce_hot_standby_feedback(m.name, m.nofailover, major_version) for m in members)
|
||||
return False
|
||||
|
||||
@@ -652,6 +656,7 @@ class AbstractDCS(object):
|
||||
self._cluster_valid_till = 0
|
||||
self._cluster_thread_lock = Lock()
|
||||
self._last_lsn = ''
|
||||
self._last_seen = 0
|
||||
self._last_status = {}
|
||||
self.event = Event()
|
||||
|
||||
@@ -722,6 +727,10 @@ class AbstractDCS(object):
|
||||
def loop_wait(self):
|
||||
return self._loop_wait
|
||||
|
||||
@property
|
||||
def last_seen(self):
|
||||
return self._last_seen
|
||||
|
||||
@abc.abstractmethod
|
||||
def _load_cluster(self):
|
||||
"""Internally this method should build `Cluster` object which
|
||||
@@ -744,6 +753,8 @@ class AbstractDCS(object):
|
||||
self.reset_cluster()
|
||||
raise
|
||||
|
||||
self._last_seen = int(time.time())
|
||||
|
||||
with self._cluster_thread_lock:
|
||||
self._cluster = cluster
|
||||
self._cluster_valid_till = time.time() + self.ttl
|
||||
@@ -767,8 +778,7 @@ class AbstractDCS(object):
|
||||
:returns: `!True` on success."""
|
||||
|
||||
def write_leader_optime(self, last_lsn):
|
||||
if self._last_lsn != last_lsn and self._write_leader_optime(last_lsn):
|
||||
self._last_lsn = last_lsn
|
||||
self.write_status({self._OPTIME: last_lsn})
|
||||
|
||||
@abc.abstractmethod
|
||||
def _write_status(self, value):
|
||||
@@ -782,7 +792,8 @@ class AbstractDCS(object):
|
||||
self._last_status = value
|
||||
cluster = self.cluster
|
||||
min_version = cluster and cluster.min_version
|
||||
if min_version and min_version < (2, 1, 0):
|
||||
if min_version and min_version < (2, 1, 0) and self._last_lsn != value[self._OPTIME]:
|
||||
self._last_lsn = value[self._OPTIME]
|
||||
self._write_leader_optime(str(value[self._OPTIME]))
|
||||
|
||||
@abc.abstractmethod
|
||||
@@ -883,7 +894,7 @@ class AbstractDCS(object):
|
||||
:param last_lsn: latest checkpoint location in bytes"""
|
||||
|
||||
if last_lsn:
|
||||
self.write_leader_optime(last_lsn)
|
||||
self.write_status({self._OPTIME: last_lsn})
|
||||
return self._delete_leader()
|
||||
|
||||
@abc.abstractmethod
|
||||
@@ -926,4 +937,4 @@ class AbstractDCS(object):
|
||||
:returns: `!True` if you would like to reschedule the next run of ha cycle"""
|
||||
|
||||
self.event.wait(timeout)
|
||||
return self.event.isSet()
|
||||
return self.event.is_set()
|
||||
|
||||
+41
-11
@@ -227,13 +227,13 @@ class Consul(AbstractDCS):
|
||||
self._last_session_refresh = 0
|
||||
self.__session_checks = config.get('checks', [])
|
||||
self._register_service = config.get('register_service', False)
|
||||
self._previous_loop_register_service = self._register_service
|
||||
self._service_tags = sorted(config.get('service_tags', []))
|
||||
self._previous_loop_service_tags = self._service_tags
|
||||
if self._register_service:
|
||||
self._service_tags = config.get('service_tags', [])
|
||||
self._service_name = service_name_from_scope_name(self._scope)
|
||||
if self._scope != self._service_name:
|
||||
logger.warning('Using %s as consul service name instead of scope name %s', self._service_name,
|
||||
self._scope)
|
||||
self._set_service_name()
|
||||
self._service_check_interval = config.get('service_check_interval', '5s')
|
||||
self._service_check_tls_server_name = config.get('service_check_tls_server_name', None)
|
||||
if not self._ctl:
|
||||
self.create_session()
|
||||
|
||||
@@ -250,7 +250,18 @@ class Consul(AbstractDCS):
|
||||
|
||||
def reload_config(self, config):
|
||||
super(Consul, self).reload_config(config)
|
||||
self._client.reload_config(config.get('consul', {}))
|
||||
|
||||
consul_config = config.get('consul', {})
|
||||
self._client.reload_config(consul_config)
|
||||
self._previous_loop_service_tags = self._service_tags
|
||||
self._service_tags = sorted(consul_config.get('service_tags', []))
|
||||
|
||||
should_register_service = consul_config.get('register_service', False)
|
||||
if should_register_service and not self._register_service:
|
||||
self._set_service_name()
|
||||
|
||||
self._previous_loop_register_service = self._register_service
|
||||
self._register_service = should_register_service
|
||||
|
||||
def set_ttl(self, ttl):
|
||||
if self._client.http.set_ttl(ttl/2.0): # Consul multiplies the TTL by 2x
|
||||
@@ -402,14 +413,18 @@ class Consul(AbstractDCS):
|
||||
self._client.kv.delete(self.member_path)
|
||||
create_member = True
|
||||
|
||||
if self._register_service or self._previous_loop_register_service:
|
||||
try:
|
||||
self.update_service(not create_member and member and member.data or {}, data)
|
||||
except Exception:
|
||||
logger.exception('update_service')
|
||||
|
||||
if not create_member and member and deep_compare(data, member.data):
|
||||
return True
|
||||
|
||||
try:
|
||||
args = {} if permanent else {'acquire': self._session}
|
||||
self._client.kv.put(self.member_path, json.dumps(data, separators=(',', ':')), **args)
|
||||
if self._register_service:
|
||||
self.update_service(not create_member and member and member.data or {}, data)
|
||||
return True
|
||||
except InvalidSession:
|
||||
self._session = None
|
||||
@@ -418,6 +433,11 @@ class Consul(AbstractDCS):
|
||||
logger.exception('touch_member')
|
||||
return False
|
||||
|
||||
def _set_service_name(self):
|
||||
self._service_name = service_name_from_scope_name(self._scope)
|
||||
if self._scope != self._service_name:
|
||||
logger.warning('Using %s as consul service name instead of scope name %s', self._service_name, self._scope)
|
||||
|
||||
@catch_consul_errors
|
||||
def register_service(self, service_name, **kwargs):
|
||||
logger.info('Register service %s, params %s', service_name, kwargs)
|
||||
@@ -439,19 +459,26 @@ class Consul(AbstractDCS):
|
||||
conn_parts = urlparse(data['conn_url'])
|
||||
check = base.Check.http(api_parts.geturl(), self._service_check_interval,
|
||||
deregister='{0}s'.format(self._client.http.ttl * 10))
|
||||
if self._service_check_tls_server_name is not None:
|
||||
check['TLSServerName'] = self._service_check_tls_server_name
|
||||
tags = self._service_tags[:]
|
||||
tags.append(role)
|
||||
self._previous_loop_service_tags = self._service_tags
|
||||
|
||||
params = {
|
||||
'service_id': '{0}/{1}'.format(self._scope, self._name),
|
||||
'address': conn_parts.hostname,
|
||||
'port': conn_parts.port,
|
||||
'check': check,
|
||||
'tags': tags
|
||||
'tags': tags,
|
||||
'enable_tag_override': True,
|
||||
}
|
||||
|
||||
if state == 'stopped':
|
||||
if state == 'stopped' or (not self._register_service and self._previous_loop_register_service):
|
||||
self._previous_loop_register_service = self._register_service
|
||||
return self.deregister_service(params['service_id'])
|
||||
|
||||
self._previous_loop_register_service = self._register_service
|
||||
if role in ['master', 'replica', 'standby-leader']:
|
||||
if state != 'running':
|
||||
return
|
||||
@@ -470,7 +497,10 @@ class Consul(AbstractDCS):
|
||||
if old_data.get(key) != new_data[key]:
|
||||
update = True
|
||||
|
||||
if force or update:
|
||||
if (
|
||||
force or update or self._register_service != self._previous_loop_register_service
|
||||
or self._service_tags != self._previous_loop_service_tags
|
||||
):
|
||||
return self._update_service(new_data)
|
||||
|
||||
@catch_consul_errors
|
||||
|
||||
+10
-6
@@ -184,11 +184,12 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
||||
|
||||
for base_uri in machines_cache:
|
||||
try:
|
||||
machines = list(self._get_members(base_uri, **kwargs))
|
||||
machines = list(set(self._get_members(base_uri, **kwargs)))
|
||||
logger.debug("Retrieved list of machines: %s", machines)
|
||||
if machines:
|
||||
random.shuffle(machines)
|
||||
self._update_dns_cache(self._dns_resolver.resolve_async, machines)
|
||||
if not self._use_proxies:
|
||||
self._update_dns_cache(self._dns_resolver.resolve_async, machines)
|
||||
return machines
|
||||
except Exception as e:
|
||||
self.http.clear()
|
||||
@@ -268,6 +269,7 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
||||
nodes, timeout, retries = self._calculate_timeouts(etcd_nodes, remaining_time)
|
||||
if nodes == 0:
|
||||
self._update_machines_cache = True
|
||||
self.set_base_uri(self._base_uri) # trigger Etcd3 watcher restart
|
||||
raise ex
|
||||
retry.sleep_func(sleeptime)
|
||||
retry.update_delay()
|
||||
@@ -282,13 +284,14 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
||||
except DNSException:
|
||||
return []
|
||||
|
||||
def _get_machines_cache_from_srv(self, srv):
|
||||
def _get_machines_cache_from_srv(self, srv, srv_suffix=None):
|
||||
"""Fetch list of etcd-cluster member by resolving _etcd-server._tcp. SRV record.
|
||||
This record should contain list of host and peer ports which could be used to run
|
||||
'GET http://{host}:{port}/members' request (peer protocol)"""
|
||||
|
||||
ret = []
|
||||
for r in ['-client-ssl', '-client', '-ssl', '', '-server-ssl', '-server']:
|
||||
r = '{0}-{1}'.format(r, srv_suffix) if srv_suffix else r
|
||||
protocol = 'https' if '-ssl' in r else 'http'
|
||||
endpoint = '/members' if '-server' in r else ''
|
||||
for host, port in self.get_srv_record('_etcd{0}._tcp.{1}'.format(r, srv)):
|
||||
@@ -325,7 +328,7 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
||||
|
||||
machines_cache = []
|
||||
if 'srv' in self._config:
|
||||
machines_cache = self._get_machines_cache_from_srv(self._config['srv'])
|
||||
machines_cache = self._get_machines_cache_from_srv(self._config['srv'], self._config.get('srv_suffix'))
|
||||
|
||||
if not machines_cache and 'hosts' in self._config:
|
||||
machines_cache = list(self._config['hosts'])
|
||||
@@ -392,8 +395,9 @@ class AbstractEtcdClientWithFailover(etcd.Client):
|
||||
self._machines_cache_updated = time.time()
|
||||
|
||||
def set_base_uri(self, value):
|
||||
logger.info('Selected new etcd server %s', value)
|
||||
self._base_uri = value
|
||||
if self._base_uri != value:
|
||||
logger.info('Selected new etcd server %s', value)
|
||||
self._base_uri = value
|
||||
|
||||
|
||||
class EtcdClient(AbstractEtcdClientWithFailover):
|
||||
|
||||
@@ -617,7 +617,7 @@ class Etcd3(AbstractEtcd):
|
||||
return self.retry(self._do_refresh_lease)
|
||||
except (Etcd3ClientError, RetryFailedError):
|
||||
logger.exception('refresh_lease')
|
||||
raise Etcd3Error('Failed ro keepalive/grant lease')
|
||||
raise Etcd3Error('Failed to keepalive/grant lease')
|
||||
|
||||
def create_lease(self):
|
||||
while not self._lease:
|
||||
|
||||
+124
-61
@@ -48,31 +48,43 @@ class K8sConfig(object):
|
||||
|
||||
def __init__(self):
|
||||
self.pool_config = {'maxsize': 10, 'num_pools': 10} # configuration for urllib3.PoolManager
|
||||
self._token_expires_at = datetime.datetime.max
|
||||
self._make_headers()
|
||||
|
||||
def _set_token(self, token):
|
||||
self._headers['authorization'] = 'Bearer ' + token
|
||||
|
||||
def _make_headers(self, token=None, **kwargs):
|
||||
self._headers = urllib3.make_headers(user_agent=USER_AGENT, **kwargs)
|
||||
if token:
|
||||
self._headers['authorization'] = 'Bearer ' + token
|
||||
self._set_token(token)
|
||||
|
||||
def load_incluster_config(self):
|
||||
if SERVICE_HOST_ENV_NAME not in os.environ or SERVICE_PORT_ENV_NAME not in os.environ:
|
||||
raise self.ConfigException('Service host/port is not set.')
|
||||
if not os.environ[SERVICE_HOST_ENV_NAME] or not os.environ[SERVICE_PORT_ENV_NAME]:
|
||||
raise self.ConfigException('Service host/port is set but empty.')
|
||||
if not os.path.isfile(SERVICE_CERT_FILENAME):
|
||||
raise self.ConfigException('Service certificate file does not exists.')
|
||||
with open(SERVICE_CERT_FILENAME) as f:
|
||||
if not f.read():
|
||||
raise self.ConfigException('Cert file exists but empty.')
|
||||
def _read_token_file(self):
|
||||
if not os.path.isfile(SERVICE_TOKEN_FILENAME):
|
||||
raise self.ConfigException('Service token file does not exists.')
|
||||
with open(SERVICE_TOKEN_FILENAME) as f:
|
||||
token = f.read()
|
||||
if not token:
|
||||
raise self.ConfigException('Token file exists but empty.')
|
||||
self._make_headers(token=token)
|
||||
self.pool_config['ca_certs'] = SERVICE_CERT_FILENAME
|
||||
self._token_expires_at = datetime.datetime.now() + self._token_refresh_interval
|
||||
return token
|
||||
|
||||
def load_incluster_config(self, ca_certs=SERVICE_CERT_FILENAME,
|
||||
token_refresh_interval=datetime.timedelta(minutes=1)):
|
||||
if SERVICE_HOST_ENV_NAME not in os.environ or SERVICE_PORT_ENV_NAME not in os.environ:
|
||||
raise self.ConfigException('Service host/port is not set.')
|
||||
if not os.environ[SERVICE_HOST_ENV_NAME] or not os.environ[SERVICE_PORT_ENV_NAME]:
|
||||
raise self.ConfigException('Service host/port is set but empty.')
|
||||
|
||||
if not os.path.isfile(ca_certs):
|
||||
raise self.ConfigException('Service certificate file does not exists.')
|
||||
with open(ca_certs) as f:
|
||||
if not f.read():
|
||||
raise self.ConfigException('Cert file exists but empty.')
|
||||
self.pool_config['ca_certs'] = ca_certs
|
||||
self._token_refresh_interval = token_refresh_interval
|
||||
token = self._read_token_file()
|
||||
self._make_headers(token=token)
|
||||
self._server = uri('https', (os.environ[SERVICE_HOST_ENV_NAME], os.environ[SERVICE_PORT_ENV_NAME]))
|
||||
|
||||
@staticmethod
|
||||
@@ -107,6 +119,11 @@ class K8sConfig(object):
|
||||
|
||||
@property
|
||||
def headers(self):
|
||||
if self._token_expires_at <= datetime.datetime.now():
|
||||
try:
|
||||
self._set_token(self._read_token_file())
|
||||
except Exception as e:
|
||||
logger.error('Failed to refresh service account token: %r', e)
|
||||
return self._headers.copy()
|
||||
|
||||
|
||||
@@ -343,12 +360,12 @@ class K8sClient(object):
|
||||
try:
|
||||
self._load_api_servers_cache()
|
||||
api_servers_cache = self.api_servers_cache
|
||||
api_servers = len(api_servers)
|
||||
api_servers = len(api_servers_cache)
|
||||
except Exception as e:
|
||||
logger.debug('Failed to update list of K8s master nodes: %r', e)
|
||||
|
||||
sleeptime = retry.sleeptime
|
||||
remaining_time = retry.stoptime - sleeptime - time.time()
|
||||
remaining_time = (retry.stoptime or time.time()) - sleeptime - time.time()
|
||||
nodes, timeout, retries = self._calculate_timeouts(api_servers, remaining_time)
|
||||
if nodes == 0:
|
||||
self._update_api_servers_cache = True
|
||||
@@ -501,6 +518,8 @@ class ObjectCache(Thread):
|
||||
self._condition = condition
|
||||
self._name = name # name of this pod
|
||||
self._is_ready = False
|
||||
self._response = None # needs to be accessible from the `kill_stream()` method
|
||||
self._response_lock = Lock() # protect the `self._response` from concurrent access
|
||||
self._object_cache = {}
|
||||
self._object_cache_lock = Lock()
|
||||
self._annotations_map = {self._dcs.leader_path: self._dcs._LEADER, self._dcs.config_path: self._dcs._CONFIG}
|
||||
@@ -541,63 +560,99 @@ class ObjectCache(Thread):
|
||||
with self._object_cache_lock:
|
||||
return self._object_cache.get(name)
|
||||
|
||||
def _process_event(self, event):
|
||||
ev_type = event['type']
|
||||
obj = event['object']
|
||||
name = obj['metadata']['name']
|
||||
|
||||
if ev_type in ('ADDED', 'MODIFIED'):
|
||||
obj = K8sObject(obj)
|
||||
success, old_value = self.set(name, obj)
|
||||
if success:
|
||||
new_value = (obj.metadata.annotations or {}).get(self._annotations_map.get(name))
|
||||
elif ev_type == 'DELETED':
|
||||
success, old_value = self.delete(name, obj['metadata']['resourceVersion'])
|
||||
new_value = None
|
||||
else:
|
||||
return logger.warning('Unexpected event type: %s', ev_type)
|
||||
|
||||
if success and obj.get('kind') != 'Pod':
|
||||
if old_value:
|
||||
old_value = (old_value.metadata.annotations or {}).get(self._annotations_map.get(name))
|
||||
|
||||
value_changed = old_value != new_value and \
|
||||
(name != self._dcs.config_path or old_value is not None and new_value is not None)
|
||||
|
||||
if value_changed:
|
||||
logger.debug('%s changed from %s to %s', name, old_value, new_value)
|
||||
|
||||
# Do not wake up HA loop if we run as leader and received leader object update event
|
||||
if value_changed or name == self._dcs.leader_path and self._name != new_value:
|
||||
self._dcs.event.set()
|
||||
|
||||
@staticmethod
|
||||
def _finish_response(response):
|
||||
try:
|
||||
response.close()
|
||||
finally:
|
||||
response.release_conn()
|
||||
|
||||
def _do_watch(self, resource_version):
|
||||
with self._response_lock:
|
||||
self._response = None
|
||||
response = self._watch(resource_version)
|
||||
with self._response_lock:
|
||||
if self._response is None:
|
||||
self._response = response
|
||||
|
||||
if not self._response:
|
||||
return self._finish_response(response)
|
||||
|
||||
for event in iter_response_objects(response):
|
||||
if event['object'].get('code') == 410:
|
||||
break
|
||||
self._process_event(event)
|
||||
|
||||
def _build_cache(self):
|
||||
objects = self._list()
|
||||
return_type = 'V1' + objects.kind[:-4]
|
||||
with self._object_cache_lock:
|
||||
self._object_cache = {item.metadata.name: item for item in objects.items}
|
||||
with self._condition:
|
||||
self._is_ready = True
|
||||
self._condition.notify()
|
||||
|
||||
response = self._watch(objects.metadata.resource_version)
|
||||
try:
|
||||
for event in iter_response_objects(response):
|
||||
obj = event['object']
|
||||
if obj.get('code') == 410:
|
||||
break
|
||||
|
||||
ev_type = event['type']
|
||||
name = obj['metadata']['name']
|
||||
|
||||
if ev_type in ('ADDED', 'MODIFIED'):
|
||||
obj = K8sObject(obj)
|
||||
success, old_value = self.set(name, obj)
|
||||
if success:
|
||||
new_value = (obj.metadata.annotations or {}).get(self._annotations_map.get(name))
|
||||
elif ev_type == 'DELETED':
|
||||
success, old_value = self.delete(name, obj['metadata']['resourceVersion'])
|
||||
new_value = None
|
||||
else:
|
||||
logger.warning('Unexpected event type: %s', ev_type)
|
||||
continue
|
||||
|
||||
if success and return_type != 'V1Pod':
|
||||
if old_value:
|
||||
old_value = (old_value.metadata.annotations or {}).get(self._annotations_map.get(name))
|
||||
|
||||
value_changed = old_value != new_value and \
|
||||
(name != self._dcs.config_path or old_value is not None and new_value is not None)
|
||||
|
||||
if value_changed:
|
||||
logger.debug('%s changed from %s to %s', name, old_value, new_value)
|
||||
|
||||
# Do not wake up HA loop if we run as leader and received leader object update event
|
||||
if value_changed or name == self._dcs.leader_path and self._name != new_value:
|
||||
self._dcs.event.set()
|
||||
self._do_watch(objects.metadata.resource_version)
|
||||
finally:
|
||||
with self._condition:
|
||||
self._is_ready = False
|
||||
response.close()
|
||||
response.release_conn()
|
||||
with self._response_lock:
|
||||
response, self._response = self._response, None
|
||||
if response:
|
||||
self._finish_response(response)
|
||||
|
||||
def kill_stream(self):
|
||||
sock = None
|
||||
with self._response_lock:
|
||||
if self._response:
|
||||
try:
|
||||
sock = self._response.connection.sock
|
||||
except Exception:
|
||||
sock = None
|
||||
else:
|
||||
self._response = False
|
||||
if sock:
|
||||
try:
|
||||
sock.shutdown(socket.SHUT_RDWR)
|
||||
sock.close()
|
||||
except Exception as e:
|
||||
logger.debug('Error on socket.shutdown: %r', e)
|
||||
|
||||
def run(self):
|
||||
while True:
|
||||
try:
|
||||
self._build_cache()
|
||||
except Exception as e:
|
||||
with self._condition:
|
||||
self._is_ready = False
|
||||
logger.error('ObjectCache.run %r', e)
|
||||
|
||||
def is_ready(self):
|
||||
@@ -613,13 +668,14 @@ class Kubernetes(AbstractDCS):
|
||||
self._label_selector = ','.join('{0}={1}'.format(k, v) for k, v in self._labels.items())
|
||||
self._namespace = config.get('namespace') or 'default'
|
||||
self._role_label = config.get('role_label', 'role')
|
||||
self._ca_certs = os.environ.get('PATRONI_KUBERNETES_CACERT', config.get('cacert')) or SERVICE_CERT_FILENAME
|
||||
config['namespace'] = ''
|
||||
super(Kubernetes, self).__init__(config)
|
||||
self._retry = Retry(deadline=config['retry_timeout'], max_delay=1, max_tries=-1,
|
||||
retry_exceptions=KubernetesRetriableException)
|
||||
self._ttl = None
|
||||
try:
|
||||
k8s_config.load_incluster_config()
|
||||
k8s_config.load_incluster_config(ca_certs=self._ca_certs)
|
||||
except k8s_config.ConfigException:
|
||||
k8s_config.load_kube_config(context=config.get('context', 'local'))
|
||||
|
||||
@@ -871,7 +927,14 @@ class Kubernetes(AbstractDCS):
|
||||
def patch_or_create(self, name, annotations, resource_version=None, patch=False, retry=True, ips=None):
|
||||
if retry is True:
|
||||
retry = self.retry
|
||||
return self._patch_or_create(name, annotations, resource_version, patch, retry, ips)
|
||||
try:
|
||||
return self._patch_or_create(name, annotations, resource_version, patch, retry, ips)
|
||||
except k8s_client.rest.ApiException as e:
|
||||
if e.status == 409 and resource_version: # Conflict in resource_version
|
||||
# Terminate watchers, it could be a sign that K8s API is in a failed state
|
||||
self._kinds.kill_stream()
|
||||
self._pods.kill_stream()
|
||||
raise e
|
||||
|
||||
def patch_or_create_config(self, annotations, resource_version=None, patch=False, retry=True):
|
||||
# SCOPE-config endpoint requires corresponding service otherwise it might be "cleaned" by k8s master
|
||||
@@ -924,7 +987,7 @@ class Kubernetes(AbstractDCS):
|
||||
|
||||
# Try to get the latest version directly from K8s API instead of relying on async cache
|
||||
try:
|
||||
kind = retry(self._api.read_namespaced_kind, self.leader_path, self._namespace)
|
||||
kind = _retry(self._api.read_namespaced_kind, self.leader_path, self._namespace)
|
||||
except Exception as e:
|
||||
logger.error('Failed to get the leader object "%s": %r', self.leader_path, e)
|
||||
return False
|
||||
@@ -958,7 +1021,7 @@ class Kubernetes(AbstractDCS):
|
||||
'transitions': leader_observed_record.get('transitions') or '0'}
|
||||
if last_lsn:
|
||||
annotations[self._OPTIME] = str(last_lsn)
|
||||
annotations['slots'] = json.dumps(slots) if slots else None
|
||||
annotations['slots'] = json.dumps(slots) if slots else None
|
||||
|
||||
resource_version = kind and kind.metadata.resource_version
|
||||
return self._update_leader_with_retry(annotations, resource_version, self.__ips)
|
||||
@@ -1008,7 +1071,7 @@ class Kubernetes(AbstractDCS):
|
||||
def touch_member(self, data, permanent=False):
|
||||
cluster = self.cluster
|
||||
if cluster and cluster.leader and cluster.leader.name == self._name:
|
||||
role = 'promoted' if data['role'] in ('replica', 'promoted') else 'master'
|
||||
role = 'master'
|
||||
elif data['state'] == 'running' and data['role'] != 'master':
|
||||
role = data['role']
|
||||
else:
|
||||
@@ -1042,12 +1105,12 @@ class Kubernetes(AbstractDCS):
|
||||
if kind and (kind.metadata.annotations or {}).get(self._LEADER) == self._name:
|
||||
annotations = {self._LEADER: None}
|
||||
if last_lsn:
|
||||
annotations[self._OPTIME] = last_lsn
|
||||
annotations[self._OPTIME] = str(last_lsn)
|
||||
self.patch_or_create(self.leader_path, annotations, kind.metadata.resource_version, True, False, [])
|
||||
self.reset_cluster()
|
||||
|
||||
def cancel_initialization(self):
|
||||
self.patch_or_create_config({self._INITIALIZE: None}, self._config_resource_version, True)
|
||||
return self.patch_or_create_config({self._INITIALIZE: None}, None, True)
|
||||
|
||||
@catch_kubernetes_errors
|
||||
def delete_cluster(self):
|
||||
|
||||
+1
-1
@@ -271,7 +271,7 @@ class Raft(AbstractDCS):
|
||||
|
||||
while True:
|
||||
ready_event.wait(5)
|
||||
if ready_event.isSet() or self._sync_obj.applied_local_log:
|
||||
if ready_event.is_set() or self._sync_obj.applied_local_log:
|
||||
break
|
||||
else:
|
||||
logger.info('waiting on raft')
|
||||
|
||||
+36
-15
@@ -7,6 +7,7 @@ from kazoo.client import KazooClient, KazooState, KazooRetry
|
||||
from kazoo.exceptions import NoNodeError, NodeExistsError, SessionExpiredError
|
||||
from kazoo.handlers.threading import SequentialThreadingHandler
|
||||
from kazoo.protocol.states import KeeperState
|
||||
from kazoo.security import make_acl
|
||||
|
||||
from . import AbstractDCS, ClusterConfig, Cluster, Failover, Leader, Member, SyncState, TimelineHistory
|
||||
from ..exceptions import DCSError
|
||||
@@ -83,6 +84,19 @@ class ZooKeeper(AbstractDCS):
|
||||
'cert': 'certfile', 'key': 'keyfile', 'key_password': 'keyfile_password'}
|
||||
kwargs = {v: config[k] for k, v in mapping.items() if k in config}
|
||||
|
||||
if 'set_acls' in config:
|
||||
kwargs['default_acl'] = []
|
||||
for principal, permissions in config['set_acls'].items():
|
||||
normalizedPermissions = [p.upper() for p in permissions]
|
||||
kwargs['default_acl'].append(make_acl(scheme='x509',
|
||||
credential=principal,
|
||||
read='READ' in normalizedPermissions,
|
||||
write='WRITE' in normalizedPermissions,
|
||||
create='CREATE' in normalizedPermissions,
|
||||
delete='DELETE' in normalizedPermissions,
|
||||
admin='ADMIN' in normalizedPermissions,
|
||||
all='ALL' in normalizedPermissions))
|
||||
|
||||
self._client = PatroniKazooClient(hosts, handler=PatroniSequentialThreadingHandler(config['retry_timeout']),
|
||||
timeout=config['ttl'], connection_retry=KazooRetry(max_delay=1, max_tries=-1,
|
||||
sleep_func=time.sleep), command_retry=KazooRetry(max_delay=1, max_tries=-1,
|
||||
@@ -91,6 +105,7 @@ class ZooKeeper(AbstractDCS):
|
||||
|
||||
self._fetch_cluster = True
|
||||
self._fetch_status = True
|
||||
self.__last_member_data = None
|
||||
|
||||
self._orig_kazoo_connect = self._client._connection._connect
|
||||
self._client._connection._connect = self._kazoo_connect
|
||||
@@ -268,17 +283,20 @@ class ZooKeeper(AbstractDCS):
|
||||
logger.exception('get_cluster')
|
||||
self.cluster_watcher(None)
|
||||
raise ZooKeeperError('ZooKeeper in not responding properly')
|
||||
# The /status ZNode was updated or doesn't exist and we are not leader
|
||||
elif (self._fetch_status and not self._fetch_cluster or not cluster.last_lsn
|
||||
or cluster.has_permanent_logical_slots(self._name, False) and not cluster.slots) and\
|
||||
not (cluster.leader and cluster.leader.name == self._name):
|
||||
try:
|
||||
last_lsn, slots = self.get_status(cluster.leader)
|
||||
# The /status ZNode was updated or doesn't exist
|
||||
elif self._fetch_status and not self._fetch_cluster or not cluster.last_lsn \
|
||||
or cluster.has_permanent_logical_slots(self._name, False) and not cluster.slots:
|
||||
# If current node is the leader just clear the event without fetching anything (we are updating the /status)
|
||||
if cluster.leader and cluster.leader.name == self._name:
|
||||
self.event.clear()
|
||||
cluster = Cluster(cluster.initialize, cluster.config, cluster.leader, last_lsn,
|
||||
cluster.members, cluster.failover, cluster.sync, cluster.history, slots)
|
||||
except Exception:
|
||||
pass
|
||||
else:
|
||||
try:
|
||||
last_lsn, slots = self.get_status(cluster.leader)
|
||||
self.event.clear()
|
||||
cluster = Cluster(cluster.initialize, cluster.config, cluster.leader, last_lsn,
|
||||
cluster.members, cluster.failover, cluster.sync, cluster.history, slots)
|
||||
except Exception:
|
||||
pass
|
||||
return cluster
|
||||
|
||||
def _bypass_caches(self):
|
||||
@@ -334,11 +352,11 @@ class ZooKeeper(AbstractDCS):
|
||||
def touch_member(self, data, permanent=False):
|
||||
cluster = self.cluster
|
||||
member = cluster and cluster.get_member(self._name, fallback_to_leader=False)
|
||||
encoded_data = json.dumps(data, separators=(',', ':')).encode('utf-8')
|
||||
member_data = self.__last_member_data or member and member.data
|
||||
if member and (self._client.client_id is not None and member.session != self._client.client_id[0] or
|
||||
not (deep_compare(member.data.get('tags', {}), data.get('tags', {})) and
|
||||
member.data.get('version') == data.get('version') and
|
||||
member.data.get('checkpoint_after_promote') == data.get('checkpoint_after_promote'))):
|
||||
not (deep_compare(member_data.get('tags', {}), data.get('tags', {})) and
|
||||
member_data.get('version') == data.get('version') and
|
||||
member_data.get('checkpoint_after_promote') == data.get('checkpoint_after_promote'))):
|
||||
try:
|
||||
self._client.delete_async(self.member_path).get(timeout=1)
|
||||
except NoNodeError:
|
||||
@@ -347,13 +365,15 @@ class ZooKeeper(AbstractDCS):
|
||||
return False
|
||||
member = None
|
||||
|
||||
encoded_data = json.dumps(data, separators=(',', ':')).encode('utf-8')
|
||||
if member:
|
||||
if deep_compare(data, member.data):
|
||||
if deep_compare(data, member_data):
|
||||
return True
|
||||
else:
|
||||
try:
|
||||
self._client.create_async(self.member_path, encoded_data, makepath=True,
|
||||
ephemeral=not permanent).get(timeout=1)
|
||||
self.__last_member_data = data
|
||||
return True
|
||||
except Exception as e:
|
||||
if not isinstance(e, NodeExistsError):
|
||||
@@ -361,6 +381,7 @@ class ZooKeeper(AbstractDCS):
|
||||
return False
|
||||
try:
|
||||
self._client.set_async(self.member_path, encoded_data).get(timeout=1)
|
||||
self.__last_member_data = data
|
||||
return True
|
||||
except Exception:
|
||||
logger.exception('touch_member')
|
||||
|
||||
+114
-52
@@ -2,32 +2,36 @@ import datetime
|
||||
import functools
|
||||
import json
|
||||
import logging
|
||||
import psycopg2
|
||||
import six
|
||||
import sys
|
||||
import time
|
||||
import uuid
|
||||
|
||||
from collections import namedtuple
|
||||
from multiprocessing.pool import ThreadPool
|
||||
from patroni.async_executor import AsyncExecutor, CriticalTask
|
||||
from patroni.exceptions import DCSError, PostgresConnectionException, PatroniFatalException
|
||||
from patroni.postgresql import ACTION_ON_START, ACTION_ON_ROLE_CHANGE
|
||||
from patroni.postgresql.misc import postgres_version_to_int
|
||||
from patroni.postgresql.rewind import Rewind
|
||||
from patroni.utils import polling_loop, tzutc, is_standby_cluster as _is_standby_cluster, parse_int
|
||||
from patroni.dcs import RemoteMember
|
||||
from threading import RLock
|
||||
|
||||
from . import psycopg
|
||||
from .async_executor import AsyncExecutor, CriticalTask
|
||||
from .exceptions import DCSError, PostgresConnectionException, PatroniFatalException
|
||||
from .postgresql import ACTION_ON_START, ACTION_ON_ROLE_CHANGE
|
||||
from .postgresql.misc import postgres_version_to_int
|
||||
from .postgresql.rewind import Rewind
|
||||
from .utils import polling_loop, tzutc, is_standby_cluster as _is_standby_cluster, parse_int
|
||||
from .dcs import RemoteMember
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class _MemberStatus(namedtuple('_MemberStatus', ['member', 'reachable', 'in_recovery', 'timeline',
|
||||
'wal_position', 'tags', 'watchdog_failed'])):
|
||||
class _MemberStatus(namedtuple('_MemberStatus', ['member', 'reachable', 'in_recovery',
|
||||
'dcs_last_seen', 'timeline', 'wal_position',
|
||||
'tags', 'watchdog_failed'])):
|
||||
"""Node status distilled from API response:
|
||||
|
||||
member - dcs.Member object of the node
|
||||
reachable - `!False` if the node is not reachable or is not responding with correct JSON
|
||||
in_recovery - `!True` if pg_is_in_recovery() == true
|
||||
dcs_last_seen - timestamp from JSON of last succesful communication with DCS
|
||||
timeline - timeline value from JSON
|
||||
wal_position - maximum value of `replayed_location` or `received_location` from JSON
|
||||
tags - dictionary with values of different tags (i.e. nofailover)
|
||||
@@ -37,12 +41,14 @@ class _MemberStatus(namedtuple('_MemberStatus', ['member', 'reachable', 'in_reco
|
||||
def from_api_response(cls, member, json):
|
||||
is_master = json['role'] == 'master'
|
||||
timeline = json.get('timeline', 0)
|
||||
dcs_last_seen = json.get('dcs_last_seen', 0)
|
||||
wal = not is_master and max(json['xlog'].get('received_location', 0), json['xlog'].get('replayed_location', 0))
|
||||
return cls(member, True, not is_master, timeline, wal, json.get('tags', {}), json.get('watchdog_failed', False))
|
||||
return cls(member, True, not is_master, dcs_last_seen, timeline, wal,
|
||||
json.get('tags', {}), json.get('watchdog_failed', False))
|
||||
|
||||
@classmethod
|
||||
def unknown(cls, member):
|
||||
return cls(member, False, None, 0, 0, {}, False)
|
||||
return cls(member, False, None, 0, 0, 0, {}, False)
|
||||
|
||||
def failover_limitation(self):
|
||||
"""Returns reason why this node can't promote or None if everything is ok."""
|
||||
@@ -186,9 +192,6 @@ class Ha(object):
|
||||
'version': self.patroni.version
|
||||
}
|
||||
|
||||
# following two lines are mainly necessary for consul, to avoid creation of master service
|
||||
if data['role'] == 'master' and not self.is_leader():
|
||||
data['role'] = 'promoted'
|
||||
if self.is_leader() and not self._rewind.checkpoint_after_promote():
|
||||
data['checkpoint_after_promote'] = False
|
||||
tags = self.get_effective_tags()
|
||||
@@ -238,7 +241,7 @@ class Ha(object):
|
||||
logger.info('bootstrapped %s', msg)
|
||||
cluster = self.dcs.get_cluster()
|
||||
node_to_follow = self._get_node_to_follow(cluster)
|
||||
return self.state_handler.follow(node_to_follow)
|
||||
return self.state_handler.follow(node_to_follow) is not False
|
||||
else:
|
||||
logger.error('failed to bootstrap %s', msg)
|
||||
self.state_handler.remove_data_directory()
|
||||
@@ -290,17 +293,30 @@ class Ha(object):
|
||||
|
||||
return result
|
||||
|
||||
def _handle_crash_recovery(self):
|
||||
if not self._crash_recovery_executed and (self.cluster.is_unlocked() or self._rewind.can_rewind):
|
||||
self._crash_recovery_executed = True
|
||||
self._crash_recovery_started = time.time()
|
||||
msg = 'doing crash recovery in a single user mode'
|
||||
return self._async_executor.try_run_async(msg, self._rewind.ensure_clean_shutdown) or msg
|
||||
|
||||
def _handle_rewind_or_reinitialize(self):
|
||||
leader = self.get_remote_master() if self.is_standby_cluster() else self.cluster.leader
|
||||
if not self._rewind.rewind_or_reinitialize_needed_and_possible(leader):
|
||||
return None
|
||||
|
||||
if self._rewind.can_rewind:
|
||||
# rewind is required, but postgres wasn't shut down cleanly.
|
||||
if not self.state_handler.is_running() and \
|
||||
self.state_handler.controldata().get('Database cluster state') == 'in archive recovery':
|
||||
msg = self._handle_crash_recovery()
|
||||
if msg:
|
||||
return msg
|
||||
|
||||
msg = 'running pg_rewind from ' + leader.name
|
||||
return self._async_executor.try_run_async(msg, self._rewind.execute, args=(leader,)) or msg
|
||||
|
||||
# remove_data_directory_on_diverged_timelines is set
|
||||
if not self.is_standby_cluster():
|
||||
if self._rewind.should_remove_data_directory_on_diverged_timelines and not self.is_standby_cluster():
|
||||
msg = 'reinitializing due to diverged timelines'
|
||||
return self._async_executor.try_run_async(msg, self._do_reinitialize, args=(self.cluster,)) or msg
|
||||
|
||||
@@ -313,10 +329,7 @@ class Ha(object):
|
||||
if timeout == 0:
|
||||
# We are requested to prefer failing over to restarting master. But see first if there
|
||||
# is anyone to fail over to.
|
||||
members = self.cluster.members
|
||||
if self.is_synchronous_mode():
|
||||
members = [m for m in members if self.cluster.sync.matches(m.name)]
|
||||
if self.is_failover_possible(members):
|
||||
if self.is_failover_possible(self.cluster.members):
|
||||
logger.info("Master crashed. Failing over.")
|
||||
self.demote('immediate')
|
||||
return 'stopped PostgreSQL to fail over after a crash'
|
||||
@@ -325,13 +338,10 @@ class Ha(object):
|
||||
|
||||
data = self.state_handler.controldata()
|
||||
logger.info('pg_controldata:\n%s\n', '\n'.join(' {0}: {1}'.format(k, v) for k, v in data.items()))
|
||||
if data.get('Database cluster state') in ('in production', 'shutting down', 'in crash recovery') \
|
||||
and not self._crash_recovery_executed and \
|
||||
(self.cluster.is_unlocked() or self._rewind.can_rewind):
|
||||
self._crash_recovery_executed = True
|
||||
self._crash_recovery_started = time.time()
|
||||
msg = 'doing crash recovery in a single user mode'
|
||||
return self._async_executor.try_run_async(msg, self._rewind.ensure_clean_shutdown) or msg
|
||||
if data.get('Database cluster state') in ('in production', 'shutting down', 'in crash recovery'):
|
||||
msg = self._handle_crash_recovery()
|
||||
if msg:
|
||||
return msg
|
||||
|
||||
self.load_cluster_from_dcs()
|
||||
|
||||
@@ -413,9 +423,10 @@ class Ha(object):
|
||||
self.state_handler.get_history(self._leader_timeline + 1):
|
||||
self._rewind.trigger_check_diverged_lsn()
|
||||
|
||||
msg = self._handle_rewind_or_reinitialize()
|
||||
if msg:
|
||||
return msg
|
||||
if not self.state_handler.is_starting():
|
||||
msg = self._handle_rewind_or_reinitialize()
|
||||
if msg:
|
||||
return msg
|
||||
|
||||
if not self.is_paused():
|
||||
self.state_handler.handle_parameter_change()
|
||||
@@ -551,7 +562,7 @@ class Ha(object):
|
||||
if master_timeline == 1:
|
||||
if cluster_history:
|
||||
self.dcs.set_history_value('[]')
|
||||
elif not cluster_history or cluster_history[-1][0] != master_timeline - 1 or len(cluster_history[-1]) != 4:
|
||||
elif not cluster_history or cluster_history[-1][0] != master_timeline - 1 or len(cluster_history[-1]) != 5:
|
||||
cluster_history = {line[0]: line for line in cluster_history or []}
|
||||
history = self.state_handler.get_history(master_timeline)
|
||||
if history and self.cluster.config:
|
||||
@@ -559,9 +570,11 @@ class Ha(object):
|
||||
for line in history:
|
||||
# enrich current history with promotion timestamps stored in DCS
|
||||
if len(line) == 3 and line[0] in cluster_history \
|
||||
and len(cluster_history[line[0]]) == 4 \
|
||||
and len(cluster_history[line[0]]) >= 4 \
|
||||
and cluster_history[line[0]][1] == line[1]:
|
||||
line.append(cluster_history[line[0]][3])
|
||||
if len(cluster_history[line[0]]) == 5:
|
||||
line.append(cluster_history[line[0]][4])
|
||||
self.dcs.set_history_value(json.dumps(history, separators=(',', ':')))
|
||||
|
||||
def enforce_follow_remote_master(self, message):
|
||||
@@ -684,16 +697,21 @@ class Ha(object):
|
||||
logger.info('Ignoring the former leader being ahead of us')
|
||||
return True
|
||||
|
||||
def is_failover_possible(self, members):
|
||||
def is_failover_possible(self, members, check_synchronous=True, cluster_lsn=None):
|
||||
ret = False
|
||||
cluster_timeline = self.cluster.timeline
|
||||
members = [m for m in members if m.name != self.state_handler.name and not m.nofailover and m.api_url]
|
||||
if check_synchronous and self.is_synchronous_mode():
|
||||
members = [m for m in members if self.cluster.sync.matches(m.name)]
|
||||
if members:
|
||||
for st in self.fetch_nodes_statuses(members):
|
||||
not_allowed_reason = st.failover_limitation()
|
||||
if not_allowed_reason:
|
||||
logger.info('Member %s is %s', st.member.name, not_allowed_reason)
|
||||
elif self.is_lagging(st.wal_position):
|
||||
elif not isinstance(st.wal_position, six.integer_types):
|
||||
logger.info('Member %s does not report wal_position', st.member.name)
|
||||
elif cluster_lsn and st.wal_position < cluster_lsn or\
|
||||
not cluster_lsn and self.is_lagging(st.wal_position):
|
||||
logger.info('Member %s exceeds maximum replication lag', st.member.name)
|
||||
elif self.check_timeline() and (not st.timeline or st.timeline < cluster_timeline):
|
||||
logger.info('Timeline %s of member %s is behind the cluster timeline %s',
|
||||
@@ -777,6 +795,10 @@ class Ha(object):
|
||||
return False
|
||||
|
||||
if self.cluster.failover:
|
||||
# When doing a switchover in synchronous mode only synchronous nodes and former leader are allowed to race
|
||||
if self.is_synchronous_mode() and self.cluster.failover.leader and \
|
||||
self.cluster.failover.candidate and not self.cluster.sync.matches(self.state_handler.name):
|
||||
return False
|
||||
return self.manual_failover_process_no_leader()
|
||||
|
||||
if not self.watchdog.is_healthy:
|
||||
@@ -785,7 +807,7 @@ class Ha(object):
|
||||
|
||||
# When in sync mode, only last known master and sync standby are allowed to promote automatically.
|
||||
all_known_members = self.cluster.members + self.old_cluster.members
|
||||
if self.is_synchronous_mode() and self.cluster.sync.leader:
|
||||
if self.is_synchronous_mode() and self.cluster.sync and self.cluster.sync.leader:
|
||||
if not self.cluster.sync.matches(self.state_handler.name):
|
||||
return False
|
||||
# pick between synchronous candidates so we minimize unnecessary failovers/demotions
|
||||
@@ -824,23 +846,44 @@ class Ha(object):
|
||||
'immediate-nolock': dict(stop='immediate', checkpoint=False, release=False, offline=False, async_req=True),
|
||||
}[mode]
|
||||
|
||||
logger.info('Demoting self (%s)', mode)
|
||||
|
||||
self._rewind.trigger_check_diverged_lsn()
|
||||
|
||||
status = {'released': False}
|
||||
|
||||
def on_shutdown(checkpoint_location):
|
||||
# Postmaster is still running, but pg_control already reports clean "shut down".
|
||||
# It could happen if Postgres is still archiving the backlog of WAL files.
|
||||
# If we know that there are replicas that received the shutdown checkpoint
|
||||
# location, we can remove the leader key and allow them to start leader race.
|
||||
if self.is_failover_possible(self.cluster.members, cluster_lsn=checkpoint_location):
|
||||
self.state_handler.set_role('demoted')
|
||||
with self._async_executor:
|
||||
self.release_leader_key_voluntarily(checkpoint_location)
|
||||
status['released'] = True
|
||||
|
||||
self.state_handler.stop(mode_control['stop'], checkpoint=mode_control['checkpoint'],
|
||||
on_safepoint=self.watchdog.disable if self.watchdog.is_running else None,
|
||||
on_shutdown=on_shutdown if mode_control['release'] else None,
|
||||
stop_timeout=self.master_stop_timeout())
|
||||
self.state_handler.set_role('demoted')
|
||||
self.set_is_leader(False)
|
||||
|
||||
if mode_control['release']:
|
||||
checkpoint_location = self.state_handler.latest_checkpoint_location() if mode == 'graceful' else None
|
||||
with self._async_executor:
|
||||
self.release_leader_key_voluntarily(checkpoint_location)
|
||||
if not status['released']:
|
||||
checkpoint_location = self.state_handler.latest_checkpoint_location() if mode == 'graceful' else None
|
||||
with self._async_executor:
|
||||
self.release_leader_key_voluntarily(checkpoint_location)
|
||||
time.sleep(2) # Give a time to somebody to take the leader lock
|
||||
if mode_control['offline']:
|
||||
node_to_follow, leader = None, None
|
||||
else:
|
||||
cluster = self.dcs.get_cluster()
|
||||
node_to_follow, leader = self._get_node_to_follow(cluster), cluster.leader
|
||||
try:
|
||||
cluster = self.dcs.get_cluster()
|
||||
node_to_follow, leader = self._get_node_to_follow(cluster), cluster.leader
|
||||
except Exception:
|
||||
node_to_follow, leader = None, None
|
||||
|
||||
# FIXME: with mode offline called from DCS exception handler and handle_long_action_in_progress
|
||||
# there could be an async action already running, calling follow from here will lead
|
||||
@@ -918,7 +961,7 @@ class Ha(object):
|
||||
else:
|
||||
members = [m for m in self.cluster.members
|
||||
if not failover.candidate or m.name == failover.candidate]
|
||||
if self.is_failover_possible(members): # check that there are healthy members
|
||||
if self.is_failover_possible(members, False): # check that there are healthy members
|
||||
ret = self._async_executor.try_run_async('manual failover: demote', self.demote, ('graceful',))
|
||||
return ret or 'manual failover: demoting myself'
|
||||
else:
|
||||
@@ -980,8 +1023,9 @@ class Ha(object):
|
||||
if self.cluster.failover and self.cluster.failover.candidate == self.state_handler.name:
|
||||
return 'waiting to become master after promote...'
|
||||
|
||||
self._delete_leader()
|
||||
return 'removed leader lock because postgres is not running as master'
|
||||
if not self.is_standby_cluster():
|
||||
self._delete_leader()
|
||||
return 'removed leader lock because postgres is not running as master'
|
||||
|
||||
if self.update_lock(True):
|
||||
msg = self.process_manual_failover_from_leader()
|
||||
@@ -995,13 +1039,13 @@ class Ha(object):
|
||||
# in case of standby cluster we don't really need to
|
||||
# enforce anything, since the leader is not a master.
|
||||
# So just remind the role.
|
||||
msg = 'no action. I am ({0}) the standby leader with the lock'.format(self.state_handler.name) \
|
||||
msg = 'no action. I am ({0}), the standby leader with the lock'.format(self.state_handler.name) \
|
||||
if self.state_handler.role == 'standby_leader' else \
|
||||
'promoted self to a standby leader because i had the session lock'
|
||||
return self.enforce_follow_remote_master(msg)
|
||||
else:
|
||||
return self.enforce_master_role(
|
||||
'no action. I am ({0}) the leader with the lock'.format(self.state_handler.name),
|
||||
'no action. I am ({0}), the leader with the lock'.format(self.state_handler.name),
|
||||
'promoted self to leader because I had the session lock'
|
||||
)
|
||||
else:
|
||||
@@ -1019,10 +1063,10 @@ class Ha(object):
|
||||
lock_owner = self.cluster.leader and self.cluster.leader.name
|
||||
if self.is_standby_cluster():
|
||||
return self.follow('cannot be a real primary in a standby cluster',
|
||||
'no action. I am a secondary ({0}) and following a standby leader ({1})'.format(
|
||||
'no action. I am ({0}), a secondary, and following a standby leader ({1})'.format(
|
||||
self.state_handler.name, lock_owner), refresh=False)
|
||||
return self.follow('demoting self because I do not have the lock and I was a leader',
|
||||
'no action. I am a secondary ({0}) and following a leader ({1})'.format(
|
||||
'no action. I am ({0}), a secondary, and following a leader ({1})'.format(
|
||||
self.state_handler.name, lock_owner), refresh=False)
|
||||
|
||||
def evaluate_scheduled_restart(self):
|
||||
@@ -1248,6 +1292,7 @@ class Ha(object):
|
||||
if not self.watchdog.activate():
|
||||
logger.error('Cancelling bootstrap because watchdog activation failed')
|
||||
self.cancel_initialization()
|
||||
self._rewind.ensure_checkpoint_after_promote(self.wakeup)
|
||||
self.dcs.initialize(create_new=(self.cluster.initialize is None), sysid=self.state_handler.sysid)
|
||||
self.dcs.set_config_value(json.dumps(self.patroni.config.dynamic_configuration, separators=(',', ':')))
|
||||
self.dcs.take_leader()
|
||||
@@ -1446,7 +1491,7 @@ class Ha(object):
|
||||
if create_slots and self.cluster.leader:
|
||||
err = self._async_executor.try_run_async('copy_logical_slots',
|
||||
self.state_handler.slots_handler.copy_logical_slots,
|
||||
args=(self.cluster.leader, create_slots))
|
||||
args=(self.cluster, create_slots))
|
||||
if not err:
|
||||
ret = 'Copying logical slots {0} from the primary'.format(create_slots)
|
||||
return ret
|
||||
@@ -1457,7 +1502,7 @@ class Ha(object):
|
||||
self.demote('offline')
|
||||
return 'demoted self because DCS is not accessible and i was a leader'
|
||||
return 'DCS is not accessible'
|
||||
except (psycopg2.Error, PostgresConnectionException):
|
||||
except (psycopg.Error, PostgresConnectionException):
|
||||
return 'Error communicating with PostgreSQL. Will try again later'
|
||||
finally:
|
||||
if not dcs_failed:
|
||||
@@ -1484,10 +1529,27 @@ class Ha(object):
|
||||
# This might not be the desired behavior of users, as a graceful shutdown of the host can mean lost data.
|
||||
# We probably need to something smarter here.
|
||||
disable_wd = self.watchdog.disable if self.watchdog.is_running else None
|
||||
|
||||
status = {'deleted': False}
|
||||
|
||||
def _on_shutdown(checkpoint_location):
|
||||
if self.is_leader():
|
||||
# Postmaster is still running, but pg_control already reports clean "shut down".
|
||||
# It could happen if Postgres is still archiving the backlog of WAL files.
|
||||
# If we know that there are replicas that received the shutdown checkpoint
|
||||
# location, we can remove the leader key and allow them to start leader race.
|
||||
if self.is_failover_possible(self.cluster.members, cluster_lsn=checkpoint_location):
|
||||
self.dcs.delete_leader(checkpoint_location)
|
||||
status['deleted'] = True
|
||||
else:
|
||||
self.dcs.write_leader_optime(checkpoint_location)
|
||||
|
||||
on_shutdown = _on_shutdown if self.is_leader() else None
|
||||
self.while_not_sync_standby(lambda: self.state_handler.stop(checkpoint=False, on_safepoint=disable_wd,
|
||||
on_shutdown=on_shutdown,
|
||||
stop_timeout=self.master_stop_timeout()))
|
||||
if not self.state_handler.is_running():
|
||||
if self.is_leader():
|
||||
if self.is_leader() and not status['deleted']:
|
||||
checkpoint_location = self.state_handler.latest_checkpoint_location()
|
||||
self.dcs.delete_leader(checkpoint_location)
|
||||
self.touch_member()
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
import logging
|
||||
import os
|
||||
import psycopg2
|
||||
import re
|
||||
import shlex
|
||||
import shutil
|
||||
@@ -23,6 +22,7 @@ from .connection import Connection, get_connection_cursor
|
||||
from .misc import parse_history, parse_lsn, postgres_major_version_to_int
|
||||
from .postmaster import PostmasterProcess
|
||||
from .slots import SlotsHandler
|
||||
from .. import psycopg
|
||||
from ..exceptions import PostgresConnectionException
|
||||
from ..utils import Retry, RetryFailedError, polling_loop, data_directory_is_empty, parse_int
|
||||
|
||||
@@ -264,15 +264,15 @@ class Postgresql(object):
|
||||
cursor = None
|
||||
try:
|
||||
cursor = self._connection.cursor()
|
||||
cursor.execute(sql, params)
|
||||
cursor.execute(sql, params or None)
|
||||
return cursor
|
||||
except psycopg2.Error as e:
|
||||
except psycopg.Error as e:
|
||||
if cursor and cursor.connection.closed == 0:
|
||||
# When connected via unix socket, psycopg2 can't recoginze 'connection lost'
|
||||
# and leaves `_cursor_holder.connection.closed == 0`, but psycopg2.OperationalError
|
||||
# is still raised (what is correct). It doesn't make sense to continiue with existing
|
||||
# connection and we will close it, to avoid its reuse by the `cursor` method.
|
||||
if isinstance(e, psycopg2.OperationalError):
|
||||
if isinstance(e, psycopg.OperationalError):
|
||||
self._connection.close()
|
||||
else:
|
||||
raise e
|
||||
@@ -327,6 +327,9 @@ class Postgresql(object):
|
||||
if cluster and cluster.config and cluster.config.modify_index:
|
||||
self._has_permanent_logical_slots =\
|
||||
cluster.has_permanent_logical_slots(self.name, nofailover, self.major_version)
|
||||
|
||||
# We want to enable hot_standby_feedback if the replica is supposed
|
||||
# to have a logical slot or in case if it is the cascading replica.
|
||||
self.set_enforce_hot_standby_feedback(
|
||||
self._has_permanent_logical_slots or
|
||||
cluster.should_enforce_hot_standby_feedback(self.name, nofailover, self.major_version))
|
||||
@@ -338,7 +341,9 @@ class Postgresql(object):
|
||||
cluster_info_state = dict(zip(['timeline', 'wal_position', 'replayed_location',
|
||||
'received_location', 'replay_paused', 'pg_control_timeline',
|
||||
'received_tli', 'slot_name', 'conninfo', 'slots'], result))
|
||||
cluster_info_state['slots'] = self.slots_handler.process_permanent_slots(cluster_info_state['slots'])
|
||||
if self._has_permanent_logical_slots:
|
||||
cluster_info_state['slots'] =\
|
||||
self.slots_handler.process_permanent_slots(cluster_info_state['slots'])
|
||||
self._cluster_info_state = cluster_info_state
|
||||
except RetryFailedError as e: # SELECT failed two times
|
||||
self._cluster_info_state = {'error': str(e)}
|
||||
@@ -369,7 +374,11 @@ class Postgresql(object):
|
||||
return self._cluster_info_state_get('received_tli')
|
||||
|
||||
def is_leader(self):
|
||||
return bool(self._cluster_info_state_get('timeline'))
|
||||
try:
|
||||
return bool(self._cluster_info_state_get('timeline'))
|
||||
except PostgresConnectionException:
|
||||
logger.warning('Failed to determine PostgreSQL state from the connection, falling back to cached role')
|
||||
return bool(self.is_running() and self.role == 'master')
|
||||
|
||||
def replay_paused(self):
|
||||
return self._cluster_info_state_get('replay_paused')
|
||||
@@ -378,12 +387,13 @@ class Postgresql(object):
|
||||
self._query('SELECT pg_catalog.pg_{0}_replay_resume()'.format(self.wal_name))
|
||||
|
||||
def handle_parameter_change(self):
|
||||
if self.major_version >= 140000 and self.replay_paused():
|
||||
if self.major_version >= 140000 and not self.is_starting() and self.replay_paused():
|
||||
logger.info('Resuming paused WAL replay for PostgreSQL 14+')
|
||||
self.resume_wal_replay()
|
||||
|
||||
def pg_control_timeline(self):
|
||||
try:
|
||||
|
||||
return int(self.controldata().get("Latest checkpoint's TimeLineID"))
|
||||
except (TypeError, ValueError):
|
||||
logger.exception('Failed to parse timeline from pg_controldata output')
|
||||
@@ -593,12 +603,13 @@ class Postgresql(object):
|
||||
cur.execute('SELECT pg_catalog.pg_is_in_recovery()')
|
||||
if cur.fetchone()[0]:
|
||||
return 'is_in_recovery=true'
|
||||
return cur.execute('CHECKPOINT')
|
||||
except psycopg2.Error:
|
||||
cur.execute('CHECKPOINT')
|
||||
except psycopg.Error:
|
||||
logger.exception('Exception during CHECKPOINT')
|
||||
return 'not accessible or not healty'
|
||||
|
||||
def stop(self, mode='fast', block_callbacks=False, checkpoint=None, on_safepoint=None, stop_timeout=None):
|
||||
def stop(self, mode='fast', block_callbacks=False, checkpoint=None,
|
||||
on_safepoint=None, on_shutdown=None, stop_timeout=None):
|
||||
"""Stop PostgreSQL
|
||||
|
||||
Supports a callback when a safepoint is reached. A safepoint is when no user backend can return a successful
|
||||
@@ -606,11 +617,12 @@ class Postgresql(object):
|
||||
could be added.
|
||||
|
||||
:param on_safepoint: This callback is called when no user backends are running.
|
||||
:param on_shutdown: is called when pg_controldata starts reporting `Database cluster state: shut down`
|
||||
"""
|
||||
if checkpoint is None:
|
||||
checkpoint = False if mode == 'immediate' else True
|
||||
|
||||
success, pg_signaled = self._do_stop(mode, block_callbacks, checkpoint, on_safepoint, stop_timeout)
|
||||
success, pg_signaled = self._do_stop(mode, block_callbacks, checkpoint, on_safepoint, on_shutdown, stop_timeout)
|
||||
if success:
|
||||
# block_callbacks is used during restart to avoid
|
||||
# running start/stop callbacks in addition to restart ones
|
||||
@@ -623,7 +635,7 @@ class Postgresql(object):
|
||||
self.set_state('stop failed')
|
||||
return success
|
||||
|
||||
def _do_stop(self, mode, block_callbacks, checkpoint, on_safepoint, stop_timeout):
|
||||
def _do_stop(self, mode, block_callbacks, checkpoint, on_safepoint, on_shutdown, stop_timeout):
|
||||
postmaster = self.is_running()
|
||||
if not postmaster:
|
||||
if on_safepoint:
|
||||
@@ -650,6 +662,22 @@ class Postgresql(object):
|
||||
postmaster.wait_for_user_backends_to_close()
|
||||
on_safepoint()
|
||||
|
||||
if on_shutdown and mode in ('fast', 'smart'):
|
||||
i = 0
|
||||
# Wait for pg_controldata `Database cluster state:` to change to "shut down"
|
||||
while postmaster.is_running():
|
||||
data = self.controldata()
|
||||
if data.get('Database cluster state', '') == 'shut down':
|
||||
on_shutdown(int(self.latest_checkpoint_location()))
|
||||
break
|
||||
elif data.get('Database cluster state', '').startswith('shut down'): # shut down in recovery
|
||||
break
|
||||
elif stop_timeout and i >= stop_timeout:
|
||||
stop_timeout = 0
|
||||
break
|
||||
time.sleep(STOP_POLLING_INTERVAL)
|
||||
i += STOP_POLLING_INTERVAL
|
||||
|
||||
try:
|
||||
postmaster.wait(timeout=stop_timeout)
|
||||
except TimeoutExpired:
|
||||
@@ -684,7 +712,7 @@ class Postgresql(object):
|
||||
while postmaster.is_running(): # Need a timeout here?
|
||||
cur.execute("SELECT 1")
|
||||
time.sleep(STOP_POLLING_INTERVAL)
|
||||
except psycopg2.Error:
|
||||
except psycopg.Error:
|
||||
pass
|
||||
|
||||
def reload(self, block_callbacks=False):
|
||||
@@ -770,7 +798,8 @@ class Postgresql(object):
|
||||
return True
|
||||
|
||||
def get_guc_value(self, name):
|
||||
cmd = [self.pgcommand('postgres'), '-D', self._data_dir, '-C', name]
|
||||
cmd = [self.pgcommand('postgres'), '-D', self._data_dir, '-C', name,
|
||||
'--config-file={}'.format(self.config.postgresql_conf)]
|
||||
try:
|
||||
data = subprocess.check_output(cmd)
|
||||
if data:
|
||||
@@ -809,7 +838,7 @@ class Postgresql(object):
|
||||
return None, None
|
||||
|
||||
@contextmanager
|
||||
def get_replication_connection_cursor(self, host='localhost', port=5432, **kwargs):
|
||||
def get_replication_connection_cursor(self, host=None, port=5432, **kwargs):
|
||||
conn_kwargs = self.config.replication.copy()
|
||||
conn_kwargs.update(host=host, port=int(port) if port else None, user=conn_kwargs.pop('username'),
|
||||
connect_timeout=3, replication=1, options='-c statement_timeout=2000')
|
||||
@@ -843,6 +872,7 @@ class Postgresql(object):
|
||||
if history[-1][0] == timeline - 1:
|
||||
history_mtime = datetime.fromtimestamp(history_mtime).replace(tzinfo=tz.tzlocal())
|
||||
history[-1].append(history_mtime.isoformat())
|
||||
history[-1].append(self.name)
|
||||
return history
|
||||
except Exception:
|
||||
logger.exception('Failed to read and parse %s', (history_path,))
|
||||
@@ -860,20 +890,22 @@ class Postgresql(object):
|
||||
if change_role:
|
||||
self.__cb_pending = ACTION_NOOP
|
||||
|
||||
ret = True
|
||||
if self.is_running():
|
||||
if do_reload:
|
||||
self.config.write_postgresql_conf()
|
||||
if self.reload(block_callbacks=change_role) and change_role:
|
||||
ret = self.reload(block_callbacks=change_role)
|
||||
if ret and change_role:
|
||||
self.set_role(role)
|
||||
else:
|
||||
self.restart(block_callbacks=change_role, role=role)
|
||||
ret = self.restart(block_callbacks=change_role, role=role)
|
||||
else:
|
||||
self.start(timeout=timeout, block_callbacks=change_role, role=role)
|
||||
ret = self.start(timeout=timeout, block_callbacks=change_role, role=role) or None
|
||||
|
||||
if change_role:
|
||||
# TODO: postpone this until start completes, or maybe do even earlier
|
||||
self.call_nowait(ACTION_ON_ROLE_CHANGE)
|
||||
return True
|
||||
return ret
|
||||
|
||||
def _wait_promote(self, wait_seconds):
|
||||
for _ in polling_loop(wait_seconds):
|
||||
@@ -954,7 +986,7 @@ class Postgresql(object):
|
||||
with self.connection().cursor() as cursor:
|
||||
cursor.execute(query)
|
||||
return cursor.fetchone()[0].isoformat(sep=' ')
|
||||
except psycopg2.Error:
|
||||
except psycopg.Error:
|
||||
return None
|
||||
|
||||
def last_operation(self):
|
||||
@@ -1077,15 +1109,16 @@ class Postgresql(object):
|
||||
for app_name, sync_state, replica_lsn in self.query(
|
||||
"SELECT pg_catalog.lower(application_name), sync_state, pg_{2}_{1}_diff({0}_{1}, '0/0')::bigint"
|
||||
" FROM pg_catalog.pg_stat_replication"
|
||||
" WHERE state = 'streaming'"
|
||||
" WHERE state = 'streaming' AND {0}_{1} IS NOT NULL"
|
||||
" ORDER BY sync_state DESC, {0}_{1} DESC".format(sort_col, self.lsn_name, self.wal_name)):
|
||||
member = members.get(app_name)
|
||||
if member and not member.tags.get('nosync', False):
|
||||
replica_list.append((member.name, sync_state, replica_lsn))
|
||||
replica_list.append((member.name, sync_state, replica_lsn, bool(member.nofailover)))
|
||||
|
||||
max_lsn = max(replica_list, key=lambda x: x[2])[2] if len(replica_list) > 1 else int(str(self.last_operation()))
|
||||
|
||||
for app_name, sync_state, replica_lsn in replica_list:
|
||||
# Prefer members without nofailover tag. We are relying on the fact that sorts are guaranteed to be stable.
|
||||
for app_name, sync_state, replica_lsn, _ in sorted(replica_list, key=lambda x: x[3]):
|
||||
if sync_node_maxlag <= 0 or max_lsn - replica_lsn <= sync_node_maxlag:
|
||||
candidates.append(app_name)
|
||||
if sync_state == 'sync':
|
||||
|
||||
@@ -4,10 +4,12 @@ import shlex
|
||||
import tempfile
|
||||
import time
|
||||
|
||||
from patroni.dcs import RemoteMember
|
||||
from patroni.utils import deep_compare
|
||||
from six import string_types
|
||||
|
||||
from ..dcs import RemoteMember
|
||||
from ..psycopg import quote_ident, quote_literal
|
||||
from ..utils import deep_compare
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -297,26 +299,24 @@ class Bootstrap(object):
|
||||
if 'NOLOGIN' not in options and 'LOGIN' not in options:
|
||||
options.append('LOGIN')
|
||||
|
||||
params = [name]
|
||||
if password:
|
||||
options.extend(['PASSWORD', '%s'])
|
||||
params.extend([password, password])
|
||||
options.extend(['PASSWORD', quote_literal(password)])
|
||||
|
||||
sql = """DO $$
|
||||
BEGIN
|
||||
SET local synchronous_commit = 'local';
|
||||
PERFORM * FROM pg_authid WHERE rolname = %s;
|
||||
PERFORM * FROM pg_catalog.pg_authid WHERE rolname = {0};
|
||||
IF FOUND THEN
|
||||
ALTER ROLE "{0}" WITH {1};
|
||||
ALTER ROLE {1} WITH {2};
|
||||
ELSE
|
||||
CREATE ROLE "{0}" WITH {1};
|
||||
CREATE ROLE {1} WITH {2};
|
||||
END IF;
|
||||
END;$$""".format(name, ' '.join(options))
|
||||
END;$$""".format(quote_literal(name), quote_ident(name, self._postgresql.connection()), ' '.join(options))
|
||||
self._postgresql.query('SET log_statement TO none')
|
||||
self._postgresql.query('SET log_min_duration_statement TO -1')
|
||||
self._postgresql.query("SET log_min_error_statement TO 'log'")
|
||||
try:
|
||||
self._postgresql.query(sql, *params)
|
||||
self._postgresql.query(sql)
|
||||
finally:
|
||||
self._postgresql.query('RESET log_min_error_statement')
|
||||
self._postgresql.query('RESET log_min_duration_statement')
|
||||
@@ -342,8 +342,8 @@ END;$$""".format(name, ' '.join(options))
|
||||
sql = """DO $$
|
||||
BEGIN
|
||||
SET local synchronous_commit = 'local';
|
||||
GRANT EXECUTE ON function pg_catalog.{0} TO "{1}";
|
||||
END;$$""".format(f, rewind['username'])
|
||||
GRANT EXECUTE ON function pg_catalog.{0} TO {1};
|
||||
END;$$""".format(f, quote_ident(rewind['username'], self._postgresql.connection()))
|
||||
postgresql.query(sql)
|
||||
|
||||
for name, value in (config.get('users') or {}).items():
|
||||
|
||||
@@ -12,6 +12,7 @@ from .validator import CaseInsensitiveDict, recovery_parameters,\
|
||||
transform_postgresql_parameter_value, transform_recovery_parameter_value
|
||||
from ..dcs import slot_name_from_member_name, RemoteMember
|
||||
from ..exceptions import PatroniFatalException
|
||||
from ..psycopg import quote_ident as _quote_ident
|
||||
from ..utils import compare_values, parse_bool, parse_int, split_host_port, uri, \
|
||||
validate_directory, is_subpath
|
||||
|
||||
@@ -23,7 +24,7 @@ PARAMETER_RE = re.compile(r'([a-z_]+)\s*=\s*')
|
||||
|
||||
def quote_ident(value):
|
||||
"""Very simplified version of quote_ident"""
|
||||
return value if SYNC_STANDBY_NAME_RE.match(value) else '"' + value + '"'
|
||||
return value if SYNC_STANDBY_NAME_RE.match(value) else _quote_ident(value)
|
||||
|
||||
|
||||
def conninfo_uri_parse(dsn):
|
||||
@@ -477,18 +478,20 @@ class ConfigHandler(object):
|
||||
ret.setdefault('channel_binding', 'prefer')
|
||||
if self._krbsrvname:
|
||||
ret['krbsrvname'] = self._krbsrvname
|
||||
if 'database' in ret:
|
||||
del ret['database']
|
||||
if 'dbname' in ret:
|
||||
del ret['dbname']
|
||||
return ret
|
||||
|
||||
def format_dsn(self, params, include_dbname=False):
|
||||
# A list of keywords that can be found in a conninfo string. Follows what is acceptable by libpq
|
||||
keywords = ('dbname', 'user', 'passfile' if params.get('passfile') else 'password', 'host', 'port',
|
||||
'sslmode', 'sslcompression', 'sslcert', 'sslkey', 'sslpassword', 'sslrootcert', 'sslcrl',
|
||||
'application_name', 'krbsrvname', 'gssencmode', 'channel_binding')
|
||||
'sslcrldir', 'application_name', 'krbsrvname', 'gssencmode', 'channel_binding',
|
||||
'target_session_attrs')
|
||||
if include_dbname:
|
||||
params = params.copy()
|
||||
params['dbname'] = params.get('database') or self._postgresql.database
|
||||
if 'dbname' not in params:
|
||||
params['dbname'] = self._postgresql.database
|
||||
# we are abusing information about the necessity of dbname
|
||||
# dsn should contain passfile or password only if there is no dbname in it (it is used in recovery.conf)
|
||||
skip = {'passfile', 'password'}
|
||||
@@ -540,6 +543,12 @@ class ConfigHandler(object):
|
||||
if use_slots and not (is_remote_master and member.no_replication_slot):
|
||||
primary_slot_name = member.primary_slot_name if is_remote_master else self._postgresql.name
|
||||
recovery_params['primary_slot_name'] = slot_name_from_member_name(primary_slot_name)
|
||||
# We are a standby leader and are using a replication slot. Make sure we connect to
|
||||
# the leader of the main cluster (in case more than one host is specified in the
|
||||
# connstr) by adding 'target_session_attrs=read-write' to primary_conninfo.
|
||||
if is_remote_master and 'target_sesions_attrs' not in primary_conninfo and\
|
||||
self._postgresql.major_version >= 100000:
|
||||
primary_conninfo['target_session_attrs'] = 'read-write'
|
||||
recovery_params['primary_conninfo'] = primary_conninfo
|
||||
|
||||
# standby_cluster config might have different parameters, we want to override them
|
||||
@@ -568,6 +577,9 @@ class ConfigHandler(object):
|
||||
return self._RECOVERY_PARAMETERS - skip_params
|
||||
|
||||
def _read_recovery_params(self):
|
||||
if self._postgresql.is_starting():
|
||||
return None, False
|
||||
|
||||
pg_conf_mtime = mtime(self._postgresql_conf)
|
||||
auto_conf_mtime = mtime(self._auto_conf)
|
||||
passfile_mtime = mtime(self._passfile) if self._passfile else False
|
||||
@@ -616,19 +628,19 @@ class ConfigHandler(object):
|
||||
|
||||
def _check_passfile(self, passfile, wanted_primary_conninfo):
|
||||
# If there is a passfile in the primary_conninfo try to figure out that
|
||||
# the passfile contains the line allowing connection to the given node.
|
||||
# the passfile contains the line(s) allowing connection to the given node.
|
||||
# We assume that the passfile was created by Patroni and therefore doing
|
||||
# the full match and not covering cases when host, port or user are set to '*'
|
||||
passfile_mtime = mtime(passfile)
|
||||
if passfile_mtime:
|
||||
try:
|
||||
with open(passfile) as f:
|
||||
wanted_line = self._pgpass_line(wanted_primary_conninfo).strip()
|
||||
for raw_line in f:
|
||||
if raw_line.strip() == wanted_line:
|
||||
self._passfile = passfile
|
||||
self._passfile_mtime = passfile_mtime
|
||||
return True
|
||||
wanted_lines = self._pgpass_line(wanted_primary_conninfo).splitlines()
|
||||
file_lines = f.read().splitlines()
|
||||
if set(wanted_lines) == set(file_lines):
|
||||
self._passfile = passfile
|
||||
self._passfile_mtime = passfile_mtime
|
||||
return True
|
||||
except Exception:
|
||||
logger.info('Failed to read %s', passfile)
|
||||
return False
|
||||
@@ -641,16 +653,17 @@ class ConfigHandler(object):
|
||||
elif not primary_conninfo:
|
||||
return False
|
||||
|
||||
wal_receiver_primary_conninfo = self._postgresql.primary_conninfo()
|
||||
if wal_receiver_primary_conninfo:
|
||||
wal_receiver_primary_conninfo = parse_dsn(wal_receiver_primary_conninfo)
|
||||
# when wal receiver is alive use primary_conninfo from pg_stat_wal_receiver for comparison
|
||||
if not self._postgresql.is_starting():
|
||||
wal_receiver_primary_conninfo = self._postgresql.primary_conninfo()
|
||||
if wal_receiver_primary_conninfo:
|
||||
primary_conninfo = wal_receiver_primary_conninfo
|
||||
# There could be no password in the primary_conninfo or it is masked.
|
||||
# Just copy the "desired" value in order to make comparison succeed.
|
||||
if 'password' in wanted_primary_conninfo:
|
||||
primary_conninfo['password'] = wanted_primary_conninfo['password']
|
||||
wal_receiver_primary_conninfo = parse_dsn(wal_receiver_primary_conninfo)
|
||||
# when wal receiver is alive use primary_conninfo from pg_stat_wal_receiver for comparison
|
||||
if wal_receiver_primary_conninfo:
|
||||
primary_conninfo = wal_receiver_primary_conninfo
|
||||
# There could be no password in the primary_conninfo or it is masked.
|
||||
# Just copy the "desired" value in order to make comparison succeed.
|
||||
if 'password' in wanted_primary_conninfo:
|
||||
primary_conninfo['password'] = wanted_primary_conninfo['password']
|
||||
|
||||
if 'passfile' in primary_conninfo and 'password' not in primary_conninfo \
|
||||
and 'password' in wanted_primary_conninfo:
|
||||
@@ -659,7 +672,7 @@ class ConfigHandler(object):
|
||||
else:
|
||||
return False
|
||||
|
||||
return all(primary_conninfo.get(p) == str(v) for p, v in wanted_primary_conninfo.items() if v is not None)
|
||||
return all(str(primary_conninfo.get(p)) == str(v) for p, v in wanted_primary_conninfo.items() if v is not None)
|
||||
|
||||
def check_recovery_conf(self, member):
|
||||
"""Returns a tuple. The first boolean element indicates that recovery params don't match
|
||||
@@ -695,16 +708,19 @@ class ConfigHandler(object):
|
||||
else: # empty string, primary_conninfo is not in the config
|
||||
primary_conninfo[0] = {}
|
||||
|
||||
# when wal receiver is alive take primary_slot_name from pg_stat_wal_receiver
|
||||
wal_receiver_primary_slot_name = self._postgresql.primary_slot_name()
|
||||
if not wal_receiver_primary_slot_name and self._postgresql.primary_conninfo():
|
||||
wal_receiver_primary_slot_name = ''
|
||||
if wal_receiver_primary_slot_name is not None:
|
||||
self._current_recovery_params['primary_slot_name'][0] = wal_receiver_primary_slot_name
|
||||
if not self._postgresql.is_starting():
|
||||
# when wal receiver is alive take primary_slot_name from pg_stat_wal_receiver
|
||||
wal_receiver_primary_slot_name = self._postgresql.primary_slot_name()
|
||||
if not wal_receiver_primary_slot_name and self._postgresql.primary_conninfo():
|
||||
wal_receiver_primary_slot_name = ''
|
||||
if wal_receiver_primary_slot_name is not None:
|
||||
self._current_recovery_params['primary_slot_name'][0] = wal_receiver_primary_slot_name
|
||||
|
||||
# Increment the 'reload' to enforce write of postgresql.conf when joining the running postgres
|
||||
required = {'restart': 0,
|
||||
'reload': int(not self._postgresql.cb_called and self._postgresql.major_version >= 120000)}
|
||||
'reload': int(self._postgresql.major_version >= 120000
|
||||
and not self._postgresql.cb_called
|
||||
and not self._postgresql.is_starting())}
|
||||
|
||||
def record_missmatch(mtype):
|
||||
required['restart' if mtype else 'reload'] += 1
|
||||
@@ -743,7 +759,12 @@ class ConfigHandler(object):
|
||||
return re.sub(r'([:\\])', r'\\\1', str(value))
|
||||
|
||||
record = {n: escape(record.get(n) or '*') for n in ('host', 'port', 'user', 'password')}
|
||||
return '{host}:{port}:*:{user}:{password}'.format(**record)
|
||||
# 'host' could be several comma-separated hostnames, in this case
|
||||
# we need to write on pgpass line per host
|
||||
line = ''
|
||||
for hostname in record.get('host').split(','):
|
||||
line += hostname + ':{port}:*:{user}:{password}'.format(**record) + '\n'
|
||||
return line.rstrip()
|
||||
|
||||
def write_pgpass(self, record):
|
||||
line = self._pgpass_line(record)
|
||||
@@ -766,6 +787,15 @@ class ConfigHandler(object):
|
||||
else:
|
||||
self._remove_file_if_exists(self._standby_signal)
|
||||
open(self._recovery_signal, 'w').close()
|
||||
|
||||
def restart_required(name):
|
||||
if self._postgresql.major_version >= 140000:
|
||||
return False
|
||||
return name == 'restore_command' or (self._postgresql.major_version < 130000
|
||||
and name in ('primary_conninfo', 'primary_slot_name'))
|
||||
|
||||
self._current_recovery_params = {n: [v, restart_required(n), self._postgresql_conf]
|
||||
for n, v in recovery_params.items()}
|
||||
else:
|
||||
with ConfigWriter(self._recovery_conf) as f:
|
||||
os.chmod(self._recovery_conf, stat.S_IWRITE | stat.S_IREAD)
|
||||
@@ -775,6 +805,7 @@ class ConfigHandler(object):
|
||||
for name in (self._recovery_conf, self._standby_signal, self._recovery_signal):
|
||||
self._remove_file_if_exists(name)
|
||||
self._recovery_params = {}
|
||||
self._current_recovery_params = None
|
||||
|
||||
def _sanitize_auto_conf(self):
|
||||
overwrite = False
|
||||
@@ -834,7 +865,7 @@ class ConfigHandler(object):
|
||||
# this exercise is improving cross version compatibility and user must set the correct parameter in the config.
|
||||
if self._postgresql.major_version >= 130000:
|
||||
wal_keep_segments = parameters.pop('wal_keep_segments', self.CMDLINE_OPTIONS['wal_keep_segments'][0])
|
||||
parameters.setdefault('wal_keep_size', str(wal_keep_segments * 16) + 'MB')
|
||||
parameters.setdefault('wal_keep_size', str(int(wal_keep_segments) * 16) + 'MB')
|
||||
elif self._postgresql.major_version:
|
||||
wal_keep_size = parse_int(parameters.pop('wal_keep_size', self.CMDLINE_OPTIONS['wal_keep_size'][0]), 'MB')
|
||||
parameters.setdefault('wal_keep_segments', int((wal_keep_size + 8) / 16))
|
||||
@@ -870,7 +901,7 @@ class ConfigHandler(object):
|
||||
ret['user'] = self._superuser['username']
|
||||
del ret['username']
|
||||
# ensure certain Patroni configurations are available
|
||||
ret.update({'database': self._postgresql.database,
|
||||
ret.update({'dbname': self._postgresql.database,
|
||||
'fallback_application_name': 'Patroni',
|
||||
'connect_timeout': 3,
|
||||
'options': '-c statement_timeout=2000'})
|
||||
@@ -1001,8 +1032,9 @@ class ConfigHandler(object):
|
||||
if self._postgresql.major_version >= 90500:
|
||||
time.sleep(1)
|
||||
try:
|
||||
pending_restart = self._postgresql.query('SELECT COUNT(*) FROM pg_catalog.pg_settings'
|
||||
' WHERE pending_restart').fetchone()[0] > 0
|
||||
pending_restart = self._postgresql.query(
|
||||
'SELECT COUNT(*) FROM pg_catalog.pg_settings WHERE pg_catalog.lower(name) != ALL(%s)'
|
||||
' AND pending_restart', [n.lower() for n in self._RECOVERY_PARAMETERS]).fetchone()[0] > 0
|
||||
self._postgresql.set_pending_restart(pending_restart)
|
||||
except Exception as e:
|
||||
logger.warning('Exception %r when running query', e)
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
import logging
|
||||
import psycopg2
|
||||
|
||||
from contextlib import contextmanager
|
||||
from threading import Lock
|
||||
|
||||
from .. import psycopg
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -20,7 +21,7 @@ class Connection(object):
|
||||
def get(self):
|
||||
with self._lock:
|
||||
if not self._connection or self._connection.closed != 0:
|
||||
self._connection = psycopg2.connect(**self._conn_kwargs)
|
||||
self._connection = psycopg.connect(**self._conn_kwargs)
|
||||
self._connection.autocommit = True
|
||||
self.server_version = self._connection.server_version
|
||||
return self._connection
|
||||
@@ -40,7 +41,7 @@ class Connection(object):
|
||||
|
||||
@contextmanager
|
||||
def get_connection_cursor(**kwargs):
|
||||
conn = psycopg2.connect(**kwargs)
|
||||
conn = psycopg.connect(**kwargs)
|
||||
conn.autocommit = True
|
||||
with conn.cursor() as cur:
|
||||
yield cur
|
||||
|
||||
@@ -28,13 +28,17 @@ class Rewind(object):
|
||||
def configuration_allows_rewind(data):
|
||||
return data.get('wal_log_hints setting', 'off') == 'on' or data.get('Data page checksum version', '0') != '0'
|
||||
|
||||
@property
|
||||
def enabled(self):
|
||||
return self._postgresql.config.get('use_pg_rewind')
|
||||
|
||||
@property
|
||||
def can_rewind(self):
|
||||
""" check if pg_rewind executable is there and that pg_controldata indicates
|
||||
we have either wal_log_hints or checksums turned on
|
||||
"""
|
||||
# low-hanging fruit: check if pg_rewind configuration is there
|
||||
if not self._postgresql.config.get('use_pg_rewind'):
|
||||
if not self.enabled:
|
||||
return False
|
||||
|
||||
cmd = [self._postgresql.pgcommand('pg_rewind'), '--help']
|
||||
@@ -46,9 +50,13 @@ class Rewind(object):
|
||||
return False
|
||||
return self.configuration_allows_rewind(self._postgresql.controldata())
|
||||
|
||||
@property
|
||||
def should_remove_data_directory_on_diverged_timelines(self):
|
||||
return self._postgresql.config.get('remove_data_directory_on_diverged_timelines')
|
||||
|
||||
@property
|
||||
def can_rewind_or_reinitialize_allowed(self):
|
||||
return self._postgresql.config.get('remove_data_directory_on_diverged_timelines') or self.can_rewind
|
||||
return self.should_remove_data_directory_on_diverged_timelines or self.can_rewind
|
||||
|
||||
def trigger_check_diverged_lsn(self):
|
||||
if self.can_rewind_or_reinitialize_allowed and self._state != REWIND_STATUS.NEED:
|
||||
@@ -65,6 +73,20 @@ class Rewind(object):
|
||||
except Exception:
|
||||
return logger.exception('Exception when working with leader')
|
||||
|
||||
@staticmethod
|
||||
def check_leader_has_run_checkpoint(conn_kwargs):
|
||||
try:
|
||||
with get_connection_cursor(connect_timeout=3, options='-c statement_timeout=2000', **conn_kwargs) as cur:
|
||||
cur.execute("SELECT NOT pg_catalog.pg_is_in_recovery()" +
|
||||
" AND ('x' || pg_catalog.substr(pg_catalog.pg_walfile_name(" +
|
||||
" pg_catalog.pg_current_wal_lsn()), 1, 8))::bit(32)::int = timeline_id" +
|
||||
" FROM pg_catalog.pg_control_checkpoint()")
|
||||
if not cur.fetchone()[0]:
|
||||
return 'leader has not run a checkpoint yet'
|
||||
except Exception:
|
||||
logger.exception('Exception when working with leader')
|
||||
return 'not accessible or not healty'
|
||||
|
||||
def _get_checkpoint_end(self, timeline, lsn):
|
||||
"""The checkpoint record size in WAL depends on postgres major version and platform (memory alignment).
|
||||
Hence, the only reliable way to figure out where it ends, read the record from file with the help of pg_waldump
|
||||
@@ -98,7 +120,7 @@ class Rewind(object):
|
||||
in_recovery = timeline = lsn = None
|
||||
data = self._postgresql.controldata()
|
||||
try:
|
||||
if data.get('Database cluster state') == 'shut down in recovery':
|
||||
if data.get('Database cluster state') in ('shut down in recovery', 'in archive recovery'):
|
||||
in_recovery = True
|
||||
lsn = data.get('Minimum recovery ending location')
|
||||
timeline = int(data.get("Min recovery ending loc's timeline"))
|
||||
@@ -152,8 +174,12 @@ class Rewind(object):
|
||||
|
||||
def _conn_kwargs(self, member, auth):
|
||||
ret = member.conn_kwargs(auth)
|
||||
if not ret.get('database'):
|
||||
ret['database'] = self._postgresql.database
|
||||
if not ret.get('dbname'):
|
||||
ret['dbname'] = self._postgresql.database
|
||||
# Add target_session_attrs in case more than one hostname is specified
|
||||
# (libpq client-side failover) making sure we hit the primary
|
||||
if 'target_session_attrs' not in ret and self._postgresql.major_version >= 100000:
|
||||
ret['target_session_attrs'] = 'read-write'
|
||||
return ret
|
||||
|
||||
def _check_timeline_and_lsn(self, leader):
|
||||
@@ -161,11 +187,17 @@ class Rewind(object):
|
||||
if local_timeline is None or local_lsn is None:
|
||||
return
|
||||
|
||||
if isinstance(leader, Leader):
|
||||
if leader.member.data.get('role') != 'master':
|
||||
return
|
||||
# standby cluster
|
||||
elif not self.check_leader_is_not_in_recovery(self._conn_kwargs(leader, self._postgresql.config.replication)):
|
||||
if isinstance(leader, Leader) and leader.member.data.get('role') != 'master':
|
||||
return
|
||||
|
||||
# We want to use replication credentials when connecting to the "postgres" database in case if
|
||||
# `use_pg_rewind` isn't enabled and only `remove_data_directory_on_diverged_timelines` is set
|
||||
# for Postgresql older than v11 (where Patroni can't use a dedicated user for rewind).
|
||||
# In all other cases we will use rewind or superuser credentials.
|
||||
check_credentials = self._postgresql.config.replication if not self.enabled and\
|
||||
self.should_remove_data_directory_on_diverged_timelines and\
|
||||
self._postgresql.major_version < 110000 else self._postgresql.config.rewind_credentials
|
||||
if not self.check_leader_is_not_in_recovery(self._conn_kwargs(leader, check_credentials)):
|
||||
return
|
||||
|
||||
history = need_rewind = None
|
||||
@@ -179,7 +211,7 @@ class Rewind(object):
|
||||
elif local_timeline == master_timeline:
|
||||
need_rewind = False
|
||||
elif master_timeline > 1:
|
||||
cur.execute('TIMELINE_HISTORY %s', (master_timeline,))
|
||||
cur.execute('TIMELINE_HISTORY {0}'.format(master_timeline))
|
||||
history = cur.fetchone()[1]
|
||||
if not isinstance(history, six.string_types):
|
||||
history = bytes(history).decode('utf-8')
|
||||
@@ -202,7 +234,10 @@ class Rewind(object):
|
||||
need_rewind = switchpoint != self._get_checkpoint_end(local_timeline, local_lsn)
|
||||
break
|
||||
elif parent_timeline > local_timeline:
|
||||
need_rewind = True
|
||||
break
|
||||
else:
|
||||
need_rewind = True
|
||||
self._log_master_history(history, i)
|
||||
|
||||
self._state = need_rewind and REWIND_STATUS.NEED or REWIND_STATUS.NOT_NEED
|
||||
@@ -230,16 +265,14 @@ class Rewind(object):
|
||||
with self._checkpoint_task_lock:
|
||||
if self._checkpoint_task:
|
||||
with self._checkpoint_task:
|
||||
if self._checkpoint_task.result:
|
||||
if self._checkpoint_task.result is not None:
|
||||
self._state = REWIND_STATUS.CHECKPOINT
|
||||
if self._checkpoint_task.result is not False:
|
||||
return
|
||||
self._checkpoint_task = None
|
||||
elif self._postgresql.get_master_timeline() == self._postgresql.pg_control_timeline():
|
||||
self._state = REWIND_STATUS.CHECKPOINT
|
||||
else:
|
||||
self._checkpoint_task = CriticalTask()
|
||||
return Thread(target=self.__checkpoint, args=(self._checkpoint_task, wakeup)).start()
|
||||
|
||||
if self._postgresql.get_master_timeline() == self._postgresql.pg_control_timeline():
|
||||
self._state = REWIND_STATUS.CHECKPOINT
|
||||
Thread(target=self.__checkpoint, args=(self._checkpoint_task, wakeup)).start()
|
||||
|
||||
def checkpoint_after_promote(self):
|
||||
return self._state == REWIND_STATUS.CHECKPOINT
|
||||
@@ -292,9 +325,18 @@ class Rewind(object):
|
||||
restore_command = self._postgresql.config.get('recovery_conf', {}).get('restore_command') \
|
||||
if self._postgresql.major_version < 120000 else self._postgresql.get_guc_value('restore_command')
|
||||
|
||||
# Until v15 pg_rewind expected postgresql.conf to be inside $PGDATA, which is not the case on e.g. Debian
|
||||
pg_rewind_can_restore = restore_command and (self._postgresql.major_version >= 150000 or
|
||||
(self._postgresql.major_version >= 130000 and
|
||||
self._postgresql.config._config_dir == self._postgresql.data_dir))
|
||||
|
||||
cmd = [self._postgresql.pgcommand('pg_rewind')]
|
||||
if self._postgresql.major_version >= 130000 and restore_command:
|
||||
if pg_rewind_can_restore:
|
||||
cmd.append('--restore-target-wal')
|
||||
if self._postgresql.major_version >= 150000 and\
|
||||
self._postgresql.config._config_dir != self._postgresql.data_dir:
|
||||
cmd.append('--config-file={0}'.format(self._postgresql.config.postgresql_conf))
|
||||
|
||||
cmd.extend(['-D', self._postgresql.data_dir, '--source-server', dsn])
|
||||
|
||||
while True:
|
||||
@@ -310,7 +352,7 @@ class Rewind(object):
|
||||
if ret == 0:
|
||||
return True
|
||||
|
||||
if not restore_command or self._postgresql.major_version >= 130000:
|
||||
if not restore_command or pg_rewind_can_restore:
|
||||
return False
|
||||
|
||||
missing_wal = self._find_missing_wal(results['stderr']) or self._find_missing_wal(results['stdout'])
|
||||
@@ -333,9 +375,14 @@ class Rewind(object):
|
||||
# running a checkpoint or
|
||||
# waiting until Patroni on the master will expose checkpoint_after_promote=True
|
||||
checkpoint_status = leader.checkpoint_after_promote if isinstance(leader, Leader) else None
|
||||
if checkpoint_status is None: # master still runs the old Patroni
|
||||
leader_status = self._postgresql.checkpoint(self._conn_kwargs(leader, self._postgresql.config.superuser))
|
||||
if leader_status:
|
||||
if checkpoint_status is None: # we are the standby-cluster leader or master still runs the old Patroni
|
||||
# superuser credentials match rewind_credentials if the latter are not provided or we run 10 or older
|
||||
if self._postgresql.config.superuser == self._postgresql.config.rewind_credentials:
|
||||
leader_status = self._postgresql.checkpoint(
|
||||
self._conn_kwargs(leader, self._postgresql.config.superuser))
|
||||
else: # we run 11+ and have a dedicated pg_rewind user
|
||||
leader_status = self.check_leader_has_run_checkpoint(r)
|
||||
if leader_status: # we tried to run/check for a checkpoint on the remote leader, but it failed
|
||||
return logger.warning('Can not use %s for rewind: %s', leader.name, leader_status)
|
||||
elif not checkpoint_status:
|
||||
return logger.info('Waiting for checkpoint on %s before rewind', leader.name)
|
||||
@@ -344,19 +391,22 @@ class Rewind(object):
|
||||
|
||||
if self.pg_rewind(r):
|
||||
self._state = REWIND_STATUS.SUCCESS
|
||||
elif not self.check_leader_is_not_in_recovery(r):
|
||||
logger.warning('Failed to rewind because master %s become unreachable', leader.name)
|
||||
else:
|
||||
logger.error('Failed to rewind from healty master: %s', leader.name)
|
||||
|
||||
for name in ('remove_data_directory_on_rewind_failure', 'remove_data_directory_on_diverged_timelines'):
|
||||
if self._postgresql.config.get(name):
|
||||
logger.warning('%s is set. removing...', name)
|
||||
self._postgresql.remove_data_directory()
|
||||
self._state = REWIND_STATUS.INITIAL
|
||||
break
|
||||
if not self.check_leader_is_not_in_recovery(r):
|
||||
logger.warning('Failed to rewind because master %s become unreachable', leader.name)
|
||||
if not self.can_rewind: # It is possible that the previous attempt damaged pg_control file!
|
||||
self._state = REWIND_STATUS.FAILED
|
||||
else:
|
||||
logger.error('Failed to rewind from healty master: %s', leader.name)
|
||||
self._state = REWIND_STATUS.FAILED
|
||||
|
||||
if self.failed:
|
||||
for name in ('remove_data_directory_on_rewind_failure', 'remove_data_directory_on_diverged_timelines'):
|
||||
if self._postgresql.config.get(name):
|
||||
logger.warning('%s is set. removing...', name)
|
||||
self._postgresql.remove_data_directory()
|
||||
self._state = REWIND_STATUS.INITIAL
|
||||
break
|
||||
return False
|
||||
|
||||
def reset_state(self):
|
||||
|
||||
+59
-25
@@ -5,10 +5,10 @@ import shutil
|
||||
|
||||
from collections import defaultdict
|
||||
from contextlib import contextmanager
|
||||
from psycopg2.errors import UndefinedFile
|
||||
|
||||
from .connection import get_connection_cursor
|
||||
from .misc import format_lsn
|
||||
from ..psycopg import OperationalError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -36,7 +36,7 @@ class SlotsHandler(object):
|
||||
def __init__(self, postgresql):
|
||||
self._postgresql = postgresql
|
||||
self._replication_slots = {} # already existing replication slots
|
||||
self._unready_logical_slots = set()
|
||||
self._unready_logical_slots = {}
|
||||
self.schedule()
|
||||
|
||||
def _query(self, sql, *params):
|
||||
@@ -82,8 +82,9 @@ class SlotsHandler(object):
|
||||
replication_slots = {}
|
||||
extra = ", catalog_xmin, pg_catalog.pg_wal_lsn_diff(confirmed_flush_lsn, '0/0')::bigint"\
|
||||
if self._postgresql.major_version >= 100000 else ""
|
||||
skip_temp_slots = ' WHERE NOT temporary' if self._postgresql.major_version >= 100000 else ''
|
||||
cursor = self._query('SELECT slot_name, slot_type, plugin, database, datoid'
|
||||
'{0} FROM pg_catalog.pg_replication_slots'.format(extra))
|
||||
'{0} FROM pg_catalog.pg_replication_slots{1}'.format(extra, skip_temp_slots))
|
||||
for r in cursor:
|
||||
value = {'type': r[1]}
|
||||
if r[1] == 'logical':
|
||||
@@ -94,7 +95,7 @@ class SlotsHandler(object):
|
||||
self._replication_slots = replication_slots
|
||||
self._schedule_load_slots = False
|
||||
if self._force_readiness_check:
|
||||
self._unready_logical_slots = set(n for n, v in replication_slots.items() if v['type'] == 'logical')
|
||||
self._unready_logical_slots = {n: None for n, v in replication_slots.items() if v['type'] == 'logical'}
|
||||
self._force_readiness_check = False
|
||||
|
||||
def ignore_replication_slot(self, cluster, name):
|
||||
@@ -142,9 +143,9 @@ class SlotsHandler(object):
|
||||
self._schedule_load_slots = True
|
||||
|
||||
@contextmanager
|
||||
def _get_local_connection_cursor(self, database):
|
||||
def _get_local_connection_cursor(self, **kwargs):
|
||||
conn_kwargs = self._postgresql.config.local_connect_kwargs
|
||||
conn_kwargs['database'] = database
|
||||
conn_kwargs.update(kwargs)
|
||||
with get_connection_cursor(**conn_kwargs) as cur:
|
||||
yield cur
|
||||
|
||||
@@ -161,7 +162,7 @@ class SlotsHandler(object):
|
||||
|
||||
# Create new logical slots
|
||||
for database, values in logical_slots.items():
|
||||
with self._get_local_connection_cursor(database) as cur:
|
||||
with self._get_local_connection_cursor(dbname=database) as cur:
|
||||
for name, value in values.items():
|
||||
try:
|
||||
cur.execute("SELECT pg_catalog.pg_create_logical_replication_slot(%s, %s)" +
|
||||
@@ -193,14 +194,14 @@ class SlotsHandler(object):
|
||||
|
||||
# Advance logical slots
|
||||
for database, values in advance_slots.items():
|
||||
with self._get_local_connection_cursor(database) as cur:
|
||||
with self._get_local_connection_cursor(dbname=database, options='-c statement_timeout=0') as cur:
|
||||
for name, value in values.items():
|
||||
try:
|
||||
cur.execute("SELECT pg_catalog.pg_replication_slot_advance(%s, %s)",
|
||||
(name, format_lsn(int(cluster.slots[name]))))
|
||||
except Exception as e:
|
||||
logger.error("Failed to advance logical replication slot '%s': %r", name, e)
|
||||
if isinstance(e, UndefinedFile):
|
||||
if isinstance(e, OperationalError) and e.diag.sqlstate == '58P01': # WAL file is gone
|
||||
create_slots.append(name)
|
||||
self._schedule_load_slots = True
|
||||
return create_slots
|
||||
@@ -235,7 +236,7 @@ class SlotsHandler(object):
|
||||
@contextmanager
|
||||
def _get_leader_connection_cursor(self, leader):
|
||||
conn_kwargs = leader.conn_kwargs(self._postgresql.config.rewind_credentials)
|
||||
conn_kwargs['database'] = self._postgresql.database
|
||||
conn_kwargs['dbname'] = self._postgresql.database
|
||||
with get_connection_cursor(connect_timeout=3, options="-c statement_timeout=2000", **conn_kwargs) as cur:
|
||||
yield cur
|
||||
|
||||
@@ -244,35 +245,68 @@ class SlotsHandler(object):
|
||||
slot_name = cluster.get_my_slot_name_on_primary(self._postgresql.name, replicatefrom)
|
||||
try:
|
||||
with self._get_leader_connection_cursor(cluster.leader) as cur:
|
||||
cur.execute("SELECT catalog_xmin FROM pg_catalog.pg_get_replication_slots()"
|
||||
" WHERE NOT pg_catalog.pg_is_in_recovery() AND slot_name = %s", (slot_name,))
|
||||
if cur.rowcount < 1:
|
||||
cur.execute("SELECT slot_name, catalog_xmin FROM pg_catalog.pg_get_replication_slots()"
|
||||
" WHERE NOT pg_catalog.pg_is_in_recovery() AND slot_name = ANY(%s)",
|
||||
([n for n, v in self._unready_logical_slots.items() if v is None] + [slot_name],))
|
||||
slots = {row[0]: row[1] for row in cur}
|
||||
if slot_name not in slots:
|
||||
return logger.warning('Physical slot %s does not exist on the primary', slot_name)
|
||||
catalog_xmin = cur.fetchone()[0]
|
||||
catalog_xmin = slots.pop(slot_name)
|
||||
except Exception as e:
|
||||
return logger.error("Failed to check %s physical slot on the primary: %r", slot_name, e)
|
||||
# Remember catalog_xmin of logical slots on the primary when catalog_xmin of
|
||||
# the physical slot became valid. Logical slots on replica will be safe to use after
|
||||
# promote when catalog_xmin of the physical slot overtakes these values.
|
||||
if catalog_xmin:
|
||||
for name, value in slots.items():
|
||||
self._unready_logical_slots[name] = value
|
||||
else: # Replica isn't streaming or the hot_standby_feedback isn't enabled
|
||||
try:
|
||||
cur = self._query("SELECT pg_catalog.current_setting('hot_standby_feedback')::boolean")
|
||||
if not cur.fetchone()[0]:
|
||||
return logger.error('Logical slot failover requires "hot_standby_feedback".'
|
||||
' Please check postgresql.auto.conf')
|
||||
except Exception as e:
|
||||
return logger.error('Failed to check the hot_standby_feedback setting: %r', e)
|
||||
|
||||
for name in list(self._unready_logical_slots):
|
||||
value = self._replication_slots.get(name)
|
||||
if not value or catalog_xmin <= value['catalog_xmin']:
|
||||
self._unready_logical_slots.remove(name)
|
||||
# The logical slot on a replica is safe to use when the physical replica slot on the primary:
|
||||
# 1. has a nonzero/non-null catalog_xmin
|
||||
# 2. has a catalog_xmin that is not newer (greater) than the catalog_xmin of any slot on the standby
|
||||
# 3. overtook the catalog_xmin of remembered values of logical slots on the primary.
|
||||
if not value or self._unready_logical_slots[name] <= catalog_xmin <= value['catalog_xmin']:
|
||||
del self._unready_logical_slots[name]
|
||||
if value:
|
||||
logger.info('Logical slot %s is safe to be used after a failover', name)
|
||||
|
||||
def copy_logical_slots(self, leader, slots):
|
||||
def copy_logical_slots(self, cluster, create_slots):
|
||||
leader = cluster.leader
|
||||
slots = cluster.get_replication_slots(self._postgresql.name, 'replica', False, self._postgresql.major_version)
|
||||
with self._get_leader_connection_cursor(leader) as cur:
|
||||
try:
|
||||
cur.execute("SELECT slot_name, catalog_xmin, "
|
||||
cur.execute("SELECT slot_name, slot_type, datname, plugin, catalog_xmin, "
|
||||
"pg_catalog.pg_wal_lsn_diff(confirmed_flush_lsn, '0/0')::bigint, "
|
||||
"pg_catalog.pg_read_binary_file('pg_replslot/' || slot_name || '/state')"
|
||||
" FROM pg_catalog.pg_get_replication_slots() WHERE NOT pg_catalog.pg_is_in_recovery()"
|
||||
" AND slot_name = ANY(%s)", (slots,))
|
||||
slots = {r[0]: {'catalog_xmin': r[1], 'confirmed_flush_lsn': r[2], 'data': r[3]} for r in cur}
|
||||
" FROM pg_catalog.pg_get_replication_slots() JOIN pg_catalog.pg_database ON datoid = oid"
|
||||
" WHERE NOT pg_catalog.pg_is_in_recovery() AND slot_name = ANY(%s)", (create_slots,))
|
||||
|
||||
create_slots = {}
|
||||
for r in cur:
|
||||
if r[0] in slots: # slot_name is defined in the global configuration
|
||||
slot = {'type': r[1], 'database': r[2], 'plugin': r[3],
|
||||
'catalog_xmin': r[4], 'confirmed_flush_lsn': r[5], 'data': r[6]}
|
||||
if compare_slots(slot, slots[r[0]]):
|
||||
create_slots[r[0]] = slot
|
||||
else:
|
||||
logger.warning('Will not copy the logical slot "%s" due to the configuration mismatch: ' +
|
||||
'configuration=%s, slot on the primary=%s', r[0], slots[r[0]], slot)
|
||||
except Exception as e:
|
||||
logger.error("Failed to copy logical slots from the %s via postgresql connection: %r", leader.name, e)
|
||||
|
||||
if isinstance(slots, dict) and self._postgresql.stop():
|
||||
if isinstance(create_slots, dict) and create_slots and self._postgresql.stop():
|
||||
pg_replslot_dir = os.path.join(self._postgresql.data_dir, 'pg_replslot')
|
||||
for name, value in slots.items():
|
||||
for name, value in create_slots.items():
|
||||
slot_dir = os.path.join(pg_replslot_dir, name)
|
||||
slot_tmp_dir = slot_dir + '.tmp'
|
||||
if os.path.exists(slot_tmp_dir):
|
||||
@@ -287,7 +321,7 @@ class SlotsHandler(object):
|
||||
shutil.rmtree(slot_dir)
|
||||
os.rename(slot_tmp_dir, slot_dir)
|
||||
fsync_dir(slot_dir)
|
||||
self._unready_logical_slots.add(name)
|
||||
self._unready_logical_slots[name] = None
|
||||
fsync_dir(pg_replslot_dir)
|
||||
self._postgresql.start()
|
||||
|
||||
@@ -299,4 +333,4 @@ class SlotsHandler(object):
|
||||
def on_promote(self):
|
||||
if self._unready_logical_slots:
|
||||
logger.warning('Logical replication slots that might be unsafe to use after promote: %s',
|
||||
self._unready_logical_slots)
|
||||
set(self._unready_logical_slots))
|
||||
|
||||
@@ -99,9 +99,11 @@ class String(namedtuple('String', 'version_from,version_till')):
|
||||
# key - parameter name
|
||||
# value - tuple or multiple tuples if something was changing in GUC across postgres versions
|
||||
parameters = CaseInsensitiveDict({
|
||||
'allow_in_place_tablespaces': Bool(150000, None),
|
||||
'allow_system_table_mods': Bool(90300, None),
|
||||
'application_name': String(90300, None),
|
||||
'archive_command': String(90300, None),
|
||||
'archive_library': String(150000, None),
|
||||
'archive_mode': (
|
||||
Bool(90300, 90500),
|
||||
EnumBool(90500, None, ('always',))
|
||||
@@ -151,14 +153,17 @@ parameters = CaseInsensitiveDict({
|
||||
Integer(90600, None, 30, 86400, 's')
|
||||
),
|
||||
'checkpoint_warning': Integer(90300, None, 0, 2147483647, 's'),
|
||||
'client_connection_check_interval': Integer(140000, None, '0', '2147483647', 'ms'),
|
||||
'client_connection_check_interval': Integer(140000, None, 0, 2147483647, 'ms'),
|
||||
'client_encoding': String(90300, None),
|
||||
'client_min_messages': Enum(90300, None, ('debug5', 'debug4', 'debug3', 'debug2',
|
||||
'debug1', 'log', 'notice', 'warning', 'error')),
|
||||
'cluster_name': String(90500, None),
|
||||
'commit_delay': Integer(90300, None, 0, 100000, None),
|
||||
'commit_siblings': Integer(90300, None, 0, 1000, None),
|
||||
'compute_query_id': EnumBool(140000, None, ('auto',)),
|
||||
'compute_query_id': (
|
||||
EnumBool(140000, 150000, ('auto',)),
|
||||
EnumBool(150000, None, ('auto', 'regress'))
|
||||
),
|
||||
'config_file': String(90300, None),
|
||||
'constraint_exclusion': EnumBool(90300, None, ('partition',)),
|
||||
'cpu_index_tuple_cost': Real(90300, None, 0, 1.79769e+308, None),
|
||||
@@ -170,7 +175,7 @@ parameters = CaseInsensitiveDict({
|
||||
'DateStyle': String(90300, None),
|
||||
'db_user_namespace': Bool(90300, None),
|
||||
'deadlock_timeout': Integer(90300, None, 1, 2147483647, 'ms'),
|
||||
'debug_invalidate_system_caches_always': Integer(140000, None, '0', '0', None),
|
||||
'debug_discard_caches': Integer(150000, None, 0, 0, None),
|
||||
'debug_pretty_print': Bool(90300, None),
|
||||
'debug_print_parse': Bool(90300, None),
|
||||
'debug_print_plan': Bool(90300, None),
|
||||
@@ -195,12 +200,14 @@ parameters = CaseInsensitiveDict({
|
||||
'enable_async_append': Bool(140000, None),
|
||||
'enable_bitmapscan': Bool(90300, None),
|
||||
'enable_gathermerge': Bool(100000, None),
|
||||
'enable_group_by_reordering': Bool(150000, None),
|
||||
'enable_hashagg': Bool(90300, None),
|
||||
'enable_hashjoin': Bool(90300, None),
|
||||
'enable_incremental_sort': Bool(130000, None),
|
||||
'enable_indexonlyscan': Bool(90300, None),
|
||||
'enable_indexscan': Bool(90300, None),
|
||||
'enable_material': Bool(90300, None),
|
||||
'enable_memoize': Bool(150000, None),
|
||||
'enable_mergejoin': Bool(90300, None),
|
||||
'enable_nestloop': Bool(90300, None),
|
||||
'enable_parallel_append': Bool(110000, None),
|
||||
@@ -208,7 +215,6 @@ parameters = CaseInsensitiveDict({
|
||||
'enable_partition_pruning': Bool(110000, None),
|
||||
'enable_partitionwise_aggregate': Bool(110000, None),
|
||||
'enable_partitionwise_join': Bool(110000, None),
|
||||
'enable_resultcache': Bool(140000, None),
|
||||
'enable_seqscan': Bool(90300, None),
|
||||
'enable_sort': Bool(90300, None),
|
||||
'enable_tidscan': Bool(90300, None),
|
||||
@@ -236,10 +242,10 @@ parameters = CaseInsensitiveDict({
|
||||
'hot_standby': Bool(90300, None),
|
||||
'hot_standby_feedback': Bool(90300, None),
|
||||
'huge_pages': EnumBool(90400, None, ('try',)),
|
||||
'huge_page_size': Integer(140000, None, '0', '2147483647', 'kB'),
|
||||
'huge_page_size': Integer(140000, None, 0, 2147483647, 'kB'),
|
||||
'ident_file': String(90300, None),
|
||||
'idle_in_transaction_session_timeout': Integer(90600, None, 0, 2147483647, 'ms'),
|
||||
'idle_session_timeout': Integer(140000, None, '0', '2147483647', 'ms'),
|
||||
'idle_session_timeout': Integer(140000, None, 0, 2147483647, 'ms'),
|
||||
'ignore_checksum_failure': Bool(90300, None),
|
||||
'ignore_invalid_pages': Bool(130000, None),
|
||||
'ignore_system_indexes': Bool(90300, None),
|
||||
@@ -296,6 +302,7 @@ parameters = CaseInsensitiveDict({
|
||||
'log_replication_commands': Bool(90500, None),
|
||||
'log_rotation_age': Integer(90300, None, 0, 35791394, 'min'),
|
||||
'log_rotation_size': Integer(90300, None, 0, 2097151, 'kB'),
|
||||
'log_startup_progress_interval': Integer(150000, None, 0, 2147483647, 'ms'),
|
||||
'log_statement': Enum(90300, None, ('none', 'ddl', 'mod', 'all')),
|
||||
'log_statement_sample_rate': Real(130000, None, 0, 1, None),
|
||||
'log_statement_stats': Bool(90300, None),
|
||||
@@ -346,7 +353,7 @@ parameters = CaseInsensitiveDict({
|
||||
Integer(90400, 90600, 1, 8388607, None),
|
||||
Integer(90600, None, 0, 262143, None)
|
||||
),
|
||||
'min_dynamic_shared_memory': Integer(140000, None, '0', '2147483647', 'MB'),
|
||||
'min_dynamic_shared_memory': Integer(140000, None, 0, 2147483647, 'MB'),
|
||||
'min_parallel_index_scan_size': Integer(100000, None, 0, 715827882, '8kB'),
|
||||
'min_parallel_relation_size': Integer(90600, 100000, 0, 715827882, '8kB'),
|
||||
'min_parallel_table_scan_size': Integer(100000, None, 0, 715827882, '8kB'),
|
||||
@@ -370,6 +377,8 @@ parameters = CaseInsensitiveDict({
|
||||
'quote_all_identifiers': Bool(90300, None),
|
||||
'random_page_cost': Real(90300, None, 0, 1.79769e+308, None),
|
||||
'recovery_init_sync_method': Enum(140000, None, ('fsync', 'syncfs')),
|
||||
'recovery_prefetch': EnumBool(150000, None, ('try',)),
|
||||
'recursive_worktable_factor': Real(150000, None, 0.001, 1e+06, None),
|
||||
'remove_temp_files_after_crash': Bool(140000, None),
|
||||
'replacement_sort_tuples': Integer(90600, 110000, 0, 2147483647, None),
|
||||
'restart_after_crash': Bool(90300, None),
|
||||
@@ -399,7 +408,8 @@ parameters = CaseInsensitiveDict({
|
||||
'ssl_renegotiation_limit': Integer(90300, 90500, 0, 2147483647, 'kB'),
|
||||
'standard_conforming_strings': Bool(90300, None),
|
||||
'statement_timeout': Integer(90300, None, 0, 2147483647, 'ms'),
|
||||
'stats_temp_directory': String(90300, None),
|
||||
'stats_fetch_consistency': Enum(150000, None, ('none', 'cache', 'snapshot')),
|
||||
'stats_temp_directory': String(90300, 150000),
|
||||
'superuser_reserved_connections': (
|
||||
Integer(90300, 90600, 0, 8388607, None),
|
||||
Integer(90600, None, 0, 262143, None)
|
||||
@@ -458,15 +468,19 @@ parameters = CaseInsensitiveDict({
|
||||
'vacuum_cost_page_hit': Integer(90300, None, 0, 10000, None),
|
||||
'vacuum_cost_page_miss': Integer(90300, None, 0, 10000, None),
|
||||
'vacuum_defer_cleanup_age': Integer(90300, None, 0, 1000000, None),
|
||||
'vacuum_failsafe_age': Integer(140000, None, '0', '2100000000', None),
|
||||
'vacuum_failsafe_age': Integer(140000, None, 0, 2100000000, None),
|
||||
'vacuum_freeze_min_age': Integer(90300, None, 0, 1000000000, None),
|
||||
'vacuum_freeze_table_age': Integer(90300, None, 0, 2000000000, None),
|
||||
'vacuum_multixact_failsafe_age': Integer(140000, None, '0', '2100000000', None),
|
||||
'vacuum_multixact_failsafe_age': Integer(140000, None, 0, 2100000000, None),
|
||||
'vacuum_multixact_freeze_min_age': Integer(90300, None, 0, 1000000000, None),
|
||||
'vacuum_multixact_freeze_table_age': Integer(90300, None, 0, 2000000000, None),
|
||||
'wal_buffers': Integer(90300, None, -1, 262143, '8kB'),
|
||||
'wal_compression': Bool(90500, None),
|
||||
'wal_compression': (
|
||||
Bool(90500, 150000),
|
||||
EnumBool(150000, None, ('pglz', 'lz4', 'zstd'))
|
||||
),
|
||||
'wal_consistency_checking': String(100000, None),
|
||||
'wal_decode_buffer_size': Integer(150000, None, 65536, 1073741823, 'B'),
|
||||
'wal_init_zero': Bool(120000, None),
|
||||
'wal_keep_segments': Integer(90300, 130000, 0, 2147483647, None),
|
||||
'wal_keep_size': Integer(130000, None, 0, 2147483647, 'MB'),
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
__all__ = ['connect', 'quote_ident', 'quote_literal', 'DatabaseError', 'Error', 'OperationalError', 'ProgrammingError']
|
||||
|
||||
_legacy = False
|
||||
try:
|
||||
from psycopg2 import __version__
|
||||
from . import MIN_PSYCOPG2, parse_version
|
||||
if parse_version(__version__) < MIN_PSYCOPG2:
|
||||
raise ImportError
|
||||
from psycopg2 import connect, Error, DatabaseError, OperationalError, ProgrammingError
|
||||
from psycopg2.extensions import adapt
|
||||
|
||||
try:
|
||||
from psycopg2.extensions import quote_ident as _quote_ident
|
||||
except ImportError:
|
||||
_legacy = True
|
||||
|
||||
def quote_literal(value, conn=None):
|
||||
value = adapt(value)
|
||||
if conn:
|
||||
value.prepare(conn)
|
||||
return value.getquoted().decode('utf-8')
|
||||
except ImportError:
|
||||
from psycopg import connect as _connect, sql, Error, DatabaseError, OperationalError, ProgrammingError
|
||||
|
||||
def connect(*args, **kwargs):
|
||||
ret = _connect(*args, **kwargs)
|
||||
ret.server_version = ret.pgconn.server_version # compatibility with psycopg2
|
||||
return ret
|
||||
|
||||
def _quote_ident(value, conn):
|
||||
return sql.Identifier(value).as_string(conn)
|
||||
|
||||
def quote_literal(value, conn=None):
|
||||
return sql.Literal(value).as_string(conn)
|
||||
|
||||
|
||||
def quote_ident(value, conn=None):
|
||||
if _legacy or conn is None:
|
||||
return '"{0}"'.format(value.replace('"', '""'))
|
||||
return _quote_ident(value, conn)
|
||||
@@ -34,6 +34,9 @@ class PatroniRequest(object):
|
||||
|
||||
if self._apply_ssl_file_param(config, 'cert'):
|
||||
self._apply_ssl_file_param(config, 'key')
|
||||
|
||||
password = self._get_cfg_value(config, 'keyfile_password')
|
||||
self._apply_pool_param('key_password', password)
|
||||
else:
|
||||
self._pool.connection_pool_kw.pop('key_file', None)
|
||||
|
||||
|
||||
+14
-10
@@ -3,10 +3,12 @@
|
||||
import json
|
||||
import logging
|
||||
import sys
|
||||
import boto.ec2
|
||||
import boto3
|
||||
|
||||
from patroni.utils import Retry, RetryFailedError
|
||||
from patroni.request import get as requests_get
|
||||
from ..utils import Retry, RetryFailedError
|
||||
from ..request import get as requests_get
|
||||
|
||||
from botocore.exceptions import ClientError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -16,7 +18,7 @@ class AWSConnection(object):
|
||||
def __init__(self, cluster_name):
|
||||
self.available = False
|
||||
self.cluster_name = cluster_name if cluster_name is not None else 'unknown'
|
||||
self._retry = Retry(deadline=300, max_delay=30, max_tries=-1, retry_exceptions=(boto.exception.StandardError,))
|
||||
self._retry = Retry(deadline=300, max_delay=30, max_tries=-1, retry_exceptions=(ClientError,))
|
||||
try:
|
||||
# get the instance id
|
||||
r = requests_get('http://169.254.169.254/latest/dynamic/instance-identity/document', timeout=2.1)
|
||||
@@ -42,20 +44,22 @@ class AWSConnection(object):
|
||||
|
||||
def _tag_ebs(self, conn, role):
|
||||
""" set tags, carrying the cluster name, instance role and instance id for the EBS storage """
|
||||
tags = {'Name': 'spilo_' + self.cluster_name, 'Role': role, 'Instance': self.instance_id}
|
||||
volumes = conn.get_all_volumes(filters={'attachment.instance-id': self.instance_id})
|
||||
conn.create_tags([v.id for v in volumes], tags)
|
||||
tags = [{'Key': 'Name', 'Value': 'spilo_' + self.cluster_name},
|
||||
{'Key': 'Role', 'Value': role},
|
||||
{'Key': 'Instance', 'Value': self.instance_id}]
|
||||
volumes = conn.volumes.filter(Filters=[{'Name': 'attachment.instance-id', 'Values': [self.instance_id]}])
|
||||
conn.create_tags(Resources=[v.id for v in volumes], Tags=tags)
|
||||
|
||||
def _tag_ec2(self, conn, role):
|
||||
""" tag the current EC2 instance with a cluster role """
|
||||
tags = {'Role': role}
|
||||
conn.create_tags([self.instance_id], tags)
|
||||
tags = [{'Key': 'Role', 'Value': role}]
|
||||
conn.create_tags(Resources=[self.instance_id], Tags=tags)
|
||||
|
||||
def on_role_change(self, new_role):
|
||||
if not self.available:
|
||||
return False
|
||||
try:
|
||||
conn = self.retry(boto.ec2.connect_to_region, self.region)
|
||||
conn = boto3.resource('ec2', region_name=self.region)
|
||||
self.retry(self._tag_ec2, conn, new_role)
|
||||
self.retry(self._tag_ebs, conn, new_role)
|
||||
except RetryFailedError:
|
||||
|
||||
@@ -27,13 +27,14 @@ import argparse
|
||||
import csv
|
||||
import logging
|
||||
import os
|
||||
import psycopg2
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
from collections import namedtuple
|
||||
|
||||
from .. import psycopg
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
RETRY_SLEEP_INTERVAL = 1
|
||||
@@ -215,7 +216,7 @@ class WALERestore(object):
|
||||
if self.master_connection:
|
||||
try:
|
||||
# get the difference in bytes between the current WAL location and the backup start offset
|
||||
with psycopg2.connect(self.master_connection) as con:
|
||||
with psycopg.connect(self.master_connection) as con:
|
||||
if con.server_version >= 100000:
|
||||
wal_name = 'wal'
|
||||
lsn_name = 'lsn'
|
||||
@@ -233,7 +234,7 @@ class WALERestore(object):
|
||||
(backup_start_lsn, backup_start_lsn, backup_start_lsn))
|
||||
|
||||
diff_in_bytes = int(cur.fetchone()[0])
|
||||
except psycopg2.Error:
|
||||
except psycopg.Error:
|
||||
logger.exception('could not determine difference with the master location')
|
||||
if attempts_no < self.retries: # retry in case of a temporarily connection issue
|
||||
attempts_no = attempts_no + 1
|
||||
|
||||
+1
-1
@@ -362,7 +362,7 @@ def polling_loop(timeout, interval=1):
|
||||
def split_host_port(value, default_port):
|
||||
t = value.rsplit(':', 1)
|
||||
if ':' in t[0]:
|
||||
t[0] = t[0].strip('[]')
|
||||
t[0] = ','.join([h.strip().strip('[]') for h in t[0].split(',')])
|
||||
t.append(default_port)
|
||||
return t[0], int(t[1])
|
||||
|
||||
|
||||
@@ -302,10 +302,11 @@ validate_host_port_listen.expected_type = string_types
|
||||
validate_host_port_listen_multiple_hosts.expected_type = string_types
|
||||
validate_data_dir.expected_type = string_types
|
||||
validate_etcd = {
|
||||
Or("host", "hosts", "srv", "url", "proxy"): Case({
|
||||
Or("host", "hosts", "srv", "srv_suffix", "url", "proxy"): Case({
|
||||
"host": validate_host_port,
|
||||
"hosts": Or(comma_separated_host_port, [validate_host_port]),
|
||||
"srv": str,
|
||||
"srv_suffix": str,
|
||||
"url": str,
|
||||
"proxy": str})
|
||||
}
|
||||
|
||||
+1
-1
@@ -1 +1 @@
|
||||
__version__ = '2.1.0'
|
||||
__version__ = '2.1.4'
|
||||
|
||||
@@ -16,10 +16,10 @@ IOC_DIRBITS = 2
|
||||
|
||||
# Non-generic platform special cases
|
||||
machine = platform.machine()
|
||||
if machine in ['mips', 'sparc', 'powerpc', 'ppc64']: # pragma: no cover
|
||||
if machine in ['mips', 'sparc', 'powerpc', 'ppc64', 'ppc64le']: # pragma: no cover
|
||||
IOC_SIZEBITS = 13
|
||||
IOC_DIRBITS = 3
|
||||
IOC_NONE, IOC_WRITE, IOC_READ = 1, 2, 4
|
||||
IOC_NONE, IOC_WRITE, IOC_READ = 1, 4, 2
|
||||
elif machine == 'parisc': # pragma: no cover
|
||||
IOC_WRITE, IOC_READ = 2, 1
|
||||
|
||||
|
||||
+2
-2
@@ -91,7 +91,7 @@ bootstrap:
|
||||
# Some additional users users which needs to be created after initializing new cluster
|
||||
users:
|
||||
admin:
|
||||
password: admin
|
||||
password: admin%
|
||||
options:
|
||||
- createrole
|
||||
- createdb
|
||||
@@ -119,7 +119,7 @@ postgresql:
|
||||
# Fully qualified kerberos ticket file for the running user
|
||||
# same as KRB5CCNAME used by the GSS
|
||||
# krb_server_keyfile: /var/spool/keytabs/postgres
|
||||
unix_socket_directories: '.'
|
||||
unix_socket_directories: '..' # parent directory of data_dir
|
||||
# Additional fencing script executed after acquiring the leader lock but before promoting the replica
|
||||
#pre_promote: /path/to/pre_promote.sh
|
||||
|
||||
|
||||
+2
-2
@@ -85,7 +85,7 @@ bootstrap:
|
||||
# Some additional users users which needs to be created after initializing new cluster
|
||||
users:
|
||||
admin:
|
||||
password: admin
|
||||
password: admin%
|
||||
options:
|
||||
- createrole
|
||||
- createdb
|
||||
@@ -113,7 +113,7 @@ postgresql:
|
||||
# Fully qualified kerberos ticket file for the running user
|
||||
# same as KRB5CCNAME used by the GSS
|
||||
# krb_server_keyfile: /var/spool/keytabs/postgres
|
||||
unix_socket_directories: '.'
|
||||
unix_socket_directories: '..' # parent directory of data_dir
|
||||
basebackup:
|
||||
- verbose
|
||||
- max-rate: 100M
|
||||
|
||||
+2
-2
@@ -82,7 +82,7 @@ bootstrap:
|
||||
# Some additional users users which needs to be created after initializing new cluster
|
||||
users:
|
||||
admin:
|
||||
password: admin
|
||||
password: admin%
|
||||
options:
|
||||
- createrole
|
||||
- createdb
|
||||
@@ -110,7 +110,7 @@ postgresql:
|
||||
# Fully qualified kerberos ticket file for the running user
|
||||
# same as KRB5CCNAME used by the GSS
|
||||
# krb_server_keyfile: /var/spool/keytabs/postgres
|
||||
unix_socket_directories: '.'
|
||||
unix_socket_directories: '..' # parent directory of data_dir
|
||||
tags:
|
||||
nofailover: false
|
||||
noloadbalance: false
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
urllib3>=1.19.1,!=1.21
|
||||
ipaddress; python_version=="2.7"
|
||||
boto
|
||||
boto3
|
||||
PyYAML
|
||||
six >= 1.7
|
||||
kazoo>=1.3.1
|
||||
|
||||
@@ -5,6 +5,7 @@
|
||||
"""
|
||||
|
||||
import inspect
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
|
||||
@@ -22,11 +23,10 @@ AUTHOR_EMAIL = '[email protected], [email protected], alexk
|
||||
KEYWORDS = 'etcd governor patroni postgresql postgres ha haproxy confd' +\
|
||||
' zookeeper exhibitor consul streaming replication kubernetes k8s'
|
||||
|
||||
EXTRAS_REQUIRE = {'aws': ['boto'], 'etcd': ['python-etcd'], 'etcd3': ['python-etcd'],
|
||||
EXTRAS_REQUIRE = {'aws': ['boto3'], 'etcd': ['python-etcd'], 'etcd3': ['python-etcd'],
|
||||
'consul': ['python-consul'], 'exhibitor': ['kazoo'], 'zookeeper': ['kazoo'],
|
||||
'kubernetes': [], 'raft': ['pysyncobj', 'cryptography']}
|
||||
COVERAGE_XML = True
|
||||
COVERAGE_HTML = False
|
||||
|
||||
# Add here all kinds of additional classifiers as defined under
|
||||
# https://pypi.python.org/pypi?%3Aaction=list_classifiers
|
||||
@@ -49,29 +49,29 @@ CLASSIFIERS = [
|
||||
'Programming Language :: Python :: 3.7',
|
||||
'Programming Language :: Python :: 3.8',
|
||||
'Programming Language :: Python :: 3.9',
|
||||
'Programming Language :: Python :: 3.10',
|
||||
'Programming Language :: Python :: Implementation :: CPython',
|
||||
]
|
||||
|
||||
CONSOLE_SCRIPTS = ['patroni = patroni:main',
|
||||
CONSOLE_SCRIPTS = ['patroni = patroni.__main__:main',
|
||||
'patronictl = patroni.ctl:ctl',
|
||||
'patroni_raft_controller = patroni.raft_controller:main',
|
||||
"patroni_wale_restore = patroni.scripts.wale_restore:main",
|
||||
"patroni_aws = patroni.scripts.aws:main"]
|
||||
|
||||
|
||||
class Flake8(Command):
|
||||
|
||||
class _Command(Command):
|
||||
user_options = []
|
||||
|
||||
def initialize_options(self):
|
||||
from flake8.main import application
|
||||
|
||||
self.flake8 = application.Application()
|
||||
self.flake8.initialize([])
|
||||
pass
|
||||
|
||||
def finalize_options(self):
|
||||
pass
|
||||
|
||||
|
||||
class Flake8(_Command):
|
||||
|
||||
def package_files(self):
|
||||
seen_package_directories = ()
|
||||
directories = self.distribution.package_dir or {}
|
||||
@@ -93,68 +93,31 @@ class Flake8(Command):
|
||||
return [package for package in self.package_files()] + ['tests', 'setup.py']
|
||||
|
||||
def run(self):
|
||||
self.flake8.run_checks(self.targets())
|
||||
self.flake8.formatter.start()
|
||||
self.flake8.report_errors()
|
||||
self.flake8.report_statistics()
|
||||
self.flake8.report_benchmarks()
|
||||
self.flake8.formatter.stop()
|
||||
try:
|
||||
self.flake8.exit()
|
||||
except SystemExit as e:
|
||||
# Cause system exit only if exit code is not zero (terminates
|
||||
# other possibly remaining/pending setuptools commands).
|
||||
if e.code:
|
||||
raise
|
||||
from flake8.main import application
|
||||
|
||||
logging.getLogger().setLevel(logging.ERROR)
|
||||
flake8 = application.Application()
|
||||
flake8.run(self.targets())
|
||||
flake8.exit()
|
||||
|
||||
|
||||
class PyTest(Command):
|
||||
class PyTest(_Command):
|
||||
|
||||
user_options = [('cov=', None, 'Run coverage'), ('cov-xml=', None, 'Generate junit xml report'),
|
||||
('cov-html=', None, 'Generate junit html report')]
|
||||
|
||||
def initialize_options(self):
|
||||
self.cov = []
|
||||
self.cov_xml = False
|
||||
self.cov_html = False
|
||||
|
||||
def finalize_options(self):
|
||||
if self.cov_xml or self.cov_html:
|
||||
self.cov = ['--cov', MAIN_PACKAGE, '--cov-report', 'term-missing']
|
||||
if self.cov_xml:
|
||||
self.cov.extend(['--cov-report', 'xml'])
|
||||
if self.cov_html:
|
||||
self.cov.extend(['--cov-report', 'html'])
|
||||
|
||||
def run_tests(self):
|
||||
def run(self):
|
||||
try:
|
||||
import pytest
|
||||
except Exception:
|
||||
raise RuntimeError('py.test is not installed, run: pip install pytest')
|
||||
|
||||
import logging
|
||||
silence = logging.WARNING
|
||||
logging.basicConfig(format='%(asctime)s %(levelname)s: %(message)s', level=os.getenv('LOGLEVEL', silence))
|
||||
logging.getLogger().setLevel(logging.WARNING)
|
||||
|
||||
args = ['--verbose', 'tests', '--doctest-modules', MAIN_PACKAGE] +\
|
||||
['-s' if logging.getLogger().getEffectiveLevel() < silence else '--capture=fd']
|
||||
if self.cov:
|
||||
args += self.cov
|
||||
['-s' if logging.getLogger().getEffectiveLevel() < logging.WARNING else '--capture=fd'] +\
|
||||
['--cov', MAIN_PACKAGE, '--cov-report', 'term-missing', '--cov-report', 'xml']
|
||||
|
||||
errno = pytest.main(args=args)
|
||||
sys.exit(errno)
|
||||
|
||||
def run(self):
|
||||
from pkg_resources import evaluate_marker
|
||||
|
||||
requirements = set(self.distribution.install_requires + ['mock>=2.0.0', 'pytest-cov', 'pytest'])
|
||||
for k, v in self.distribution.extras_require.items():
|
||||
if not k.startswith(':') or evaluate_marker(k[1:]):
|
||||
requirements.update(v)
|
||||
|
||||
self.distribution.fetch_build_eggs(list(requirements))
|
||||
self.run_tests()
|
||||
|
||||
|
||||
def read(fname):
|
||||
with open(os.path.join(__location__, fname)) as fd:
|
||||
@@ -162,6 +125,8 @@ def read(fname):
|
||||
|
||||
|
||||
def setup_package(version):
|
||||
logging.basicConfig(format='%(message)s', level=os.getenv('LOGLEVEL', logging.WARNING))
|
||||
|
||||
# Assemble additional setup commands
|
||||
cmdclass = {'test': PyTest, 'flake8': Flake8}
|
||||
|
||||
@@ -184,12 +149,6 @@ def setup_package(version):
|
||||
if not extra:
|
||||
install_requires.append(r)
|
||||
|
||||
command_options = {'test': {}}
|
||||
if COVERAGE_XML:
|
||||
command_options['test']['cov_xml'] = 'setup.py', True
|
||||
if COVERAGE_HTML:
|
||||
command_options['test']['cov_html'] = 'setup.py', True
|
||||
|
||||
setup(
|
||||
name=NAME,
|
||||
version=version,
|
||||
@@ -206,9 +165,7 @@ def setup_package(version):
|
||||
python_requires='>=2.7',
|
||||
install_requires=install_requires,
|
||||
extras_require=EXTRAS_REQUIRE,
|
||||
setup_requires='flake8',
|
||||
cmdclass=cmdclass,
|
||||
command_options=command_options,
|
||||
entry_points={'console_scripts': CONSOLE_SCRIPTS},
|
||||
)
|
||||
|
||||
@@ -216,13 +173,14 @@ def setup_package(version):
|
||||
if __name__ == '__main__':
|
||||
old_modules = sys.modules.copy()
|
||||
try:
|
||||
from patroni import check_psycopg2, fatal, __version__
|
||||
from patroni import check_psycopg, fatal
|
||||
from patroni.version import __version__
|
||||
finally:
|
||||
sys.modules.clear()
|
||||
sys.modules.update(old_modules)
|
||||
|
||||
if sys.version_info < (2, 7, 0):
|
||||
fatal('Patroni needs to be run with Python 2.7+')
|
||||
check_psycopg2()
|
||||
check_psycopg()
|
||||
|
||||
setup_package(__version__)
|
||||
|
||||
+11
-9
@@ -5,9 +5,10 @@ import unittest
|
||||
|
||||
from mock import Mock, patch
|
||||
|
||||
import psycopg2
|
||||
import urllib3
|
||||
|
||||
import patroni.psycopg as psycopg
|
||||
|
||||
from patroni.dcs import Leader, Member
|
||||
from patroni.postgresql import Postgresql
|
||||
from patroni.postgresql.config import ConfigHandler
|
||||
@@ -85,15 +86,15 @@ class MockCursor(object):
|
||||
|
||||
def execute(self, sql, *params):
|
||||
if sql.startswith('blabla'):
|
||||
raise psycopg2.ProgrammingError()
|
||||
raise psycopg.ProgrammingError()
|
||||
elif sql == 'CHECKPOINT' or sql.startswith('SELECT pg_catalog.pg_create_'):
|
||||
raise psycopg2.OperationalError()
|
||||
raise psycopg.OperationalError()
|
||||
elif sql.startswith('RetryFailedError'):
|
||||
raise RetryFailedError('retry')
|
||||
elif sql.startswith('SELECT catalog_xmin'):
|
||||
self.results = [(100, 501)]
|
||||
elif sql.startswith('SELECT slot_name, catalog_xmin'):
|
||||
self.results = [('ls', 100, 500, b'123456')]
|
||||
self.results = [('postgresql0', 100), ('ls', 100)]
|
||||
elif sql.startswith('SELECT slot_name, slot_type, datname, plugin, catalog_xmin'):
|
||||
self.results = [('ls', 'logical', 'a', 'b', 100, 500, b'123456')]
|
||||
elif sql.startswith('SELECT slot_name'):
|
||||
self.results = [('blabla', 'physical'), ('foobar', 'physical'), ('ls', 'logical', 'a', 'b', 5, 100, 500)]
|
||||
elif sql.startswith('SELECT CASE WHEN pg_catalog.pg_is_in_recovery()'):
|
||||
@@ -162,7 +163,7 @@ class MockConnect(object):
|
||||
pass
|
||||
|
||||
|
||||
def psycopg2_connect(*args, **kwargs):
|
||||
def psycopg_connect(*args, **kwargs):
|
||||
return MockConnect()
|
||||
|
||||
|
||||
@@ -176,7 +177,7 @@ class PostgresInit(unittest.TestCase):
|
||||
'force_parallel_mode': '1', 'constraint_exclusion': '',
|
||||
'max_stack_depth': 'Z', 'vacuum_cost_limit': -1, 'vacuum_cost_delay': 200}
|
||||
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch('patroni.postgresql.CallbackExecutor', Mock())
|
||||
@patch.object(ConfigHandler, 'write_postgresql_conf', Mock())
|
||||
@patch.object(ConfigHandler, 'replace_pg_hba', Mock())
|
||||
@@ -189,7 +190,8 @@ class PostgresInit(unittest.TestCase):
|
||||
'krbsrvname': 'postgres', 'pgpass': os.path.join(data_dir, 'pgpass0'),
|
||||
'listen': '127.0.0.2, 127.0.0.3:5432', 'connect_address': '127.0.0.2:5432',
|
||||
'authentication': {'superuser': {'username': 'foo', 'password': 'test'},
|
||||
'replication': {'username': '', 'password': 'rep-pass'}},
|
||||
'replication': {'username': '', 'password': 'rep-pass'},
|
||||
'rewind': {'username': 'rewind', 'password': 'test'}},
|
||||
'remove_data_directory_on_rewind_failure': True,
|
||||
'use_pg_rewind': True, 'pg_ctl_timeout': 'bla',
|
||||
'parameters': self._PARAMETERS,
|
||||
|
||||
+12
-7
@@ -1,9 +1,10 @@
|
||||
import datetime
|
||||
import json
|
||||
import psycopg2
|
||||
import unittest
|
||||
import socket
|
||||
|
||||
import patroni.psycopg as psycopg
|
||||
|
||||
from mock import Mock, PropertyMock, patch
|
||||
from patroni.api import RestApiHandler, RestApiServer
|
||||
from patroni.dcs import ClusterConfig, Member
|
||||
@@ -11,7 +12,7 @@ from patroni.ha import _MemberStatus
|
||||
from patroni.utils import tzutc
|
||||
from six import BytesIO as IO
|
||||
from six.moves import BaseHTTPServer
|
||||
from . import psycopg2_connect, MockCursor
|
||||
from . import psycopg_connect, MockCursor
|
||||
from .test_ha import get_cluster_initialized_without_leader
|
||||
|
||||
|
||||
@@ -35,7 +36,7 @@ class MockPostgresql(object):
|
||||
|
||||
@staticmethod
|
||||
def connection():
|
||||
return psycopg2_connect()
|
||||
return psycopg_connect()
|
||||
|
||||
@staticmethod
|
||||
def postmaster_start_time():
|
||||
@@ -77,7 +78,7 @@ class MockHa(object):
|
||||
|
||||
@staticmethod
|
||||
def fetch_nodes_statuses(members):
|
||||
return [_MemberStatus(None, True, None, 0, None, {}, False)]
|
||||
return [_MemberStatus(None, True, None, 0, 0, None, {}, False)]
|
||||
|
||||
@staticmethod
|
||||
def schedule_future_restart(data):
|
||||
@@ -168,6 +169,7 @@ class TestRestApiHandler(unittest.TestCase):
|
||||
MockRestApiServer(RestApiHandler, 'GET /replica?lag=10MB')
|
||||
MockRestApiServer(RestApiHandler, 'GET /replica?lag=10485760')
|
||||
MockRestApiServer(RestApiHandler, 'GET /read-only')
|
||||
MockRestApiServer(RestApiHandler, 'GET /read-only-sync')
|
||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={})):
|
||||
MockRestApiServer(RestApiHandler, 'GET /replica')
|
||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'master'})):
|
||||
@@ -179,6 +181,8 @@ class TestRestApiHandler(unittest.TestCase):
|
||||
MockPatroni.dcs.cluster.is_synchronous_mode = Mock(return_value=True)
|
||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
|
||||
MockRestApiServer(RestApiHandler, 'GET /synchronous')
|
||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
|
||||
MockRestApiServer(RestApiHandler, 'GET /read-only-sync')
|
||||
with patch.object(RestApiHandler, 'get_postgresql_status', Mock(return_value={'role': 'replica'})):
|
||||
MockPatroni.dcs.cluster.sync.members = []
|
||||
MockRestApiServer(RestApiHandler, 'GET /asynchronous')
|
||||
@@ -435,9 +439,9 @@ class TestRestApiHandler(unittest.TestCase):
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
def test_RestApiServer_query(self):
|
||||
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg2.OperationalError)):
|
||||
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)):
|
||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /patroni'))
|
||||
with patch.object(MockPostgresql, 'connection', Mock(side_effect=psycopg2.OperationalError)):
|
||||
with patch.object(MockPostgresql, 'connection', Mock(side_effect=psycopg.OperationalError)):
|
||||
self.assertIsNotNone(MockRestApiServer(RestApiHandler, 'GET /patroni'))
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
@@ -548,7 +552,8 @@ class TestRestApiServer(unittest.TestCase):
|
||||
self.assertRaises(ValueError, MockRestApiServer, None, '', bad_config)
|
||||
self.assertRaises(ValueError, self.srv.reload_config, bad_config)
|
||||
self.assertRaises(ValueError, self.srv.reload_config, {})
|
||||
with patch.object(socket.socket, 'setsockopt', Mock(side_effect=socket.error)):
|
||||
with patch.object(socket.socket, 'setsockopt', Mock(side_effect=socket.error)), \
|
||||
patch.object(MockRestApiServer, 'server_close', Mock()):
|
||||
self.srv.reload_config({'listen': ':8008'})
|
||||
|
||||
@patch.object(MockPatroni, 'dcs')
|
||||
|
||||
+15
-10
@@ -1,4 +1,4 @@
|
||||
import boto.ec2
|
||||
import botocore
|
||||
import sys
|
||||
import unittest
|
||||
import urllib3
|
||||
@@ -8,21 +8,27 @@ from collections import namedtuple
|
||||
from patroni.scripts.aws import AWSConnection, main as _main
|
||||
|
||||
|
||||
class MockEc2Connection(object):
|
||||
class MockVolumes(object):
|
||||
|
||||
@staticmethod
|
||||
def get_all_volumes(*args, **kwargs):
|
||||
def filter(*args, **kwargs):
|
||||
oid = namedtuple('Volume', 'id')
|
||||
return [oid(id='a'), oid(id='b')]
|
||||
|
||||
|
||||
class MockEc2Connection(object):
|
||||
|
||||
volumes = MockVolumes()
|
||||
|
||||
@staticmethod
|
||||
def create_tags(objects, *args, **kwargs):
|
||||
if len(objects) == 0:
|
||||
raise boto.exception.BotoServerError(503, 'Service Unavailable', 'Request limit exceeded')
|
||||
def create_tags(Resources, **kwargs):
|
||||
if len(Resources) == 0:
|
||||
raise botocore.exceptions.ClientError({'Error': {'Code': 503, 'Message': 'Request limit exceeded'}},
|
||||
'create_tags')
|
||||
return True
|
||||
|
||||
|
||||
@patch('boto.ec2.connect_to_region', Mock(return_value=MockEc2Connection()))
|
||||
@patch('boto3.resource', Mock(return_value=MockEc2Connection()))
|
||||
class TestAWSConnection(unittest.TestCase):
|
||||
|
||||
@patch('patroni.scripts.aws.requests_get', Mock(return_value=urllib3.HTTPResponse(
|
||||
@@ -32,7 +38,7 @@ class TestAWSConnection(unittest.TestCase):
|
||||
|
||||
def test_on_role_change(self):
|
||||
self.assertTrue(self.conn.on_role_change('master'))
|
||||
with patch.object(MockEc2Connection, 'get_all_volumes', Mock(return_value=[])):
|
||||
with patch.object(MockVolumes, 'filter', Mock(return_value=[])):
|
||||
self.conn._retry.max_tries = 1
|
||||
self.assertFalse(self.conn.on_role_change('master'))
|
||||
|
||||
@@ -46,8 +52,7 @@ class TestAWSConnection(unittest.TestCase):
|
||||
conn = AWSConnection('test')
|
||||
self.assertFalse(conn.aws_available())
|
||||
|
||||
@patch('patroni.scripts.aws.requests_get', Mock(return_value=urllib3.HTTPResponse(
|
||||
status=200, body=b'{"instanceId": "012345", "region": "eu-west-1"}')))
|
||||
@patch('patroni.scripts.aws.requests_get', Mock(return_value=urllib3.HTTPResponse(status=503, body=b'Error')))
|
||||
@patch('sys.exit', Mock())
|
||||
def test_main(self):
|
||||
self.assertIsNone(_main())
|
||||
|
||||
@@ -8,11 +8,11 @@ from patroni.postgresql.bootstrap import Bootstrap
|
||||
from patroni.postgresql.cancellable import CancellableSubprocess
|
||||
from patroni.postgresql.config import ConfigHandler
|
||||
|
||||
from . import psycopg2_connect, BaseTestPostgresql
|
||||
from . import psycopg_connect, BaseTestPostgresql
|
||||
|
||||
|
||||
@patch('subprocess.call', Mock(return_value=0))
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch('os.rename', Mock())
|
||||
class TestBootstrap(BaseTestPostgresql):
|
||||
|
||||
@@ -164,6 +164,7 @@ class TestBootstrap(BaseTestPostgresql):
|
||||
@patch('os.unlink', Mock())
|
||||
@patch('shutil.copy', Mock())
|
||||
@patch('os.path.isfile', Mock(return_value=True))
|
||||
@patch('patroni.postgresql.bootstrap.quote_ident', Mock())
|
||||
@patch.object(Bootstrap, 'call_post_bootstrap', Mock(return_value=True))
|
||||
@patch.object(Bootstrap, '_custom_bootstrap', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
||||
|
||||
@@ -27,8 +27,8 @@ class TestCancellableSubprocess(unittest.TestCase):
|
||||
def test_cancel(self):
|
||||
self.c._process = Mock()
|
||||
self.c._process.is_running.return_value = True
|
||||
self.c._process.children.side_effect = psutil.Error()
|
||||
self.c._process.suspend.side_effect = psutil.Error()
|
||||
self.c._process.children.side_effect = psutil.NoSuchProcess(123)
|
||||
self.c._process.suspend.side_effect = psutil.AccessDenied()
|
||||
self.c.cancel()
|
||||
self.c._process.is_running.side_effect = [True, False]
|
||||
self.c.cancel()
|
||||
|
||||
+54
-5
@@ -2,7 +2,7 @@ import consul
|
||||
import unittest
|
||||
|
||||
from consul import ConsulException, NotFound
|
||||
from mock import Mock, patch
|
||||
from mock import Mock, PropertyMock, patch
|
||||
from patroni.dcs.consul import AbstractDCS, Cluster, Consul, ConsulInternalError, \
|
||||
ConsulError, ConsulClient, HTTPClient, InvalidSessionTTL, InvalidSession
|
||||
from . import SleepException
|
||||
@@ -91,7 +91,7 @@ class TestConsul(unittest.TestCase):
|
||||
Consul({'ttl': 30, 'scope': 't_', 'name': 'p', 'url': 'https://l:1', 'retry_timeout': 10,
|
||||
'verify': 'on', 'cert': 'bar', 'cacert': 'buz', 'register_service': True})
|
||||
self.c = Consul({'ttl': 30, 'scope': 'test', 'name': 'postgresql1', 'host': 'localhost:1', 'retry_timeout': 10,
|
||||
'register_service': True})
|
||||
'register_service': True, 'service_check_tls_server_name': True})
|
||||
self.c._base_path = '/service/good'
|
||||
self.c.get_cluster()
|
||||
|
||||
@@ -130,8 +130,9 @@ class TestConsul(unittest.TestCase):
|
||||
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=[True, ConsulException, InvalidSession]))
|
||||
def test_touch_member(self):
|
||||
self.c.refresh_session = Mock(return_value=False)
|
||||
self.c.touch_member({'conn_url': 'postgres://replicator:[email protected]:5433/postgres',
|
||||
'api_url': 'http://127.0.0.1:8009/patroni'})
|
||||
with patch.object(Consul, 'update_service', Mock(side_effect=Exception)):
|
||||
self.c.touch_member({'conn_url': 'postgres://replicator:[email protected]:5433/postgres',
|
||||
'api_url': 'http://127.0.0.1:8009/patroni'})
|
||||
self.c._register_service = True
|
||||
self.c.refresh_session = Mock(return_value=True)
|
||||
for _ in range(0, 4):
|
||||
@@ -153,8 +154,10 @@ class TestConsul(unittest.TestCase):
|
||||
def test_set_config_value(self):
|
||||
self.c.set_config_value('')
|
||||
|
||||
@patch.object(Cluster, 'min_version', PropertyMock(return_value=(2, 0)))
|
||||
@patch.object(consul.Consul.KV, 'put', Mock(side_effect=ConsulException))
|
||||
def test_write_leader_optime(self):
|
||||
self.c.get_cluster()
|
||||
self.c.write_leader_optime('1')
|
||||
|
||||
@patch.object(consul.Consul.Session, 'renew', Mock())
|
||||
@@ -215,4 +218,50 @@ class TestConsul(unittest.TestCase):
|
||||
self.assertIsNone(self.c.update_service({}, d))
|
||||
|
||||
def test_reload_config(self):
|
||||
self.c.reload_config({'consul': {'token': 'foo'}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||
self.assertEqual([], self.c._service_tags)
|
||||
self.c.reload_config({'consul': {'token': 'foo', 'register_service': True, 'service_tags': ['foo']},
|
||||
'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||
self.assertEqual(["foo"], self.c._service_tags)
|
||||
|
||||
self.c.refresh_session = Mock(return_value=False)
|
||||
|
||||
d = {'role': 'replica', 'api_url': 'http://a/t', 'conn_url': 'pg://c:1', 'state': 'running'}
|
||||
|
||||
# Changing register_service from True to False calls deregister()
|
||||
self.c.reload_config({'consul': {'register_service': False}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||
with patch('consul.Consul.Agent.Service.deregister') as mock_deregister:
|
||||
self.c.touch_member(d)
|
||||
mock_deregister.assert_called_once()
|
||||
|
||||
self.assertEqual([], self.c._service_tags)
|
||||
|
||||
# register_service staying False between reloads does not call deregister()
|
||||
self.c.reload_config({'consul': {'register_service': False}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||
with patch('consul.Consul.Agent.Service.deregister') as mock_deregister:
|
||||
self.c.touch_member(d)
|
||||
self.assertFalse(mock_deregister.called)
|
||||
|
||||
# Changing register_service from False to True calls register()
|
||||
self.c.reload_config({'consul': {'register_service': True}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||
with patch('consul.Consul.Agent.Service.register') as mock_register:
|
||||
self.c.touch_member(d)
|
||||
mock_register.assert_called_once()
|
||||
|
||||
# register_service staying True between reloads does not call register()
|
||||
self.c.reload_config({'consul': {'register_service': True}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||
with patch('consul.Consul.Agent.Service.register') as mock_register:
|
||||
self.c.touch_member(d)
|
||||
self.assertFalse(mock_deregister.called)
|
||||
|
||||
# register_service staying True between reloads does calls register() if other service data has changed
|
||||
self.c.reload_config({'consul': {'register_service': True}, 'loop_wait': 10, 'ttl': 30, 'retry_timeout': 10})
|
||||
with patch('consul.Consul.Agent.Service.register') as mock_register:
|
||||
self.c.touch_member(d)
|
||||
mock_register.assert_called_once()
|
||||
|
||||
# register_service staying True between reloads does calls register() if service_tags have changed
|
||||
self.c.reload_config({'consul': {'register_service': True, 'service_tags': ['foo']}, 'loop_wait': 10,
|
||||
'ttl': 30, 'retry_timeout': 10})
|
||||
with patch('consul.Consul.Agent.Service.register') as mock_register:
|
||||
self.c.touch_member(d)
|
||||
mock_register.assert_called_once()
|
||||
|
||||
+5
-5
@@ -9,11 +9,11 @@ from patroni.ctl import ctl, store_config, load_config, output_members, get_dcs,
|
||||
get_all_members, get_any_member, get_cursor, query_member, configure, PatroniCtlException, apply_config_changes, \
|
||||
format_config_for_editing, show_diff, invoke_editor, format_pg_version, CONFIG_FILE_PATH
|
||||
from patroni.dcs.etcd import AbstractEtcdClientWithFailover, Failover
|
||||
from patroni.psycopg import OperationalError
|
||||
from patroni.utils import tzutc
|
||||
from psycopg2 import OperationalError
|
||||
from urllib3 import PoolManager
|
||||
|
||||
from . import MockConnect, MockCursor, MockResponse, psycopg2_connect
|
||||
from . import MockConnect, MockCursor, MockResponse, psycopg_connect
|
||||
from .test_etcd import etcd_read, socket_getaddrinfo
|
||||
from .test_ha import get_cluster_initialized_without_leader, get_cluster_initialized_with_leader, \
|
||||
get_cluster_initialized_with_only_leader, get_cluster_not_initialized_without_leader, get_cluster, Member
|
||||
@@ -48,7 +48,7 @@ class TestCtl(unittest.TestCase):
|
||||
self.assertRaises(PatroniCtlException, load_config, './non-existing-config-file', None)
|
||||
self.assertRaises(PatroniCtlException, load_config, './non-existing-config-file', None)
|
||||
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
def test_get_cursor(self):
|
||||
self.assertIsNone(get_cursor(get_cluster_initialized_without_leader(), {}, role='master'))
|
||||
|
||||
@@ -57,7 +57,7 @@ class TestCtl(unittest.TestCase):
|
||||
# MockCursor returns pg_is_in_recovery as false
|
||||
self.assertIsNone(get_cursor(get_cluster_initialized_with_leader(), {}, role='replica'))
|
||||
|
||||
self.assertIsNotNone(get_cursor(get_cluster_initialized_with_leader(), {'database': 'foo'}, role='any'))
|
||||
self.assertIsNotNone(get_cursor(get_cluster_initialized_with_leader(), {'dbname': 'foo'}, role='any'))
|
||||
|
||||
def test_parse_dcs(self):
|
||||
assert parse_dcs(None) is None
|
||||
@@ -165,7 +165,7 @@ class TestCtl(unittest.TestCase):
|
||||
def test_get_dcs(self):
|
||||
self.assertRaises(PatroniCtlException, get_dcs, {'dummy': {}}, 'dummy')
|
||||
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch('patroni.ctl.query_member', Mock(return_value=([['mock column']], None)))
|
||||
@patch('patroni.ctl.get_dcs')
|
||||
@patch.object(etcd.Client, 'read', etcd_read)
|
||||
|
||||
+9
-2
@@ -4,7 +4,7 @@ import socket
|
||||
import unittest
|
||||
|
||||
from dns.exception import DNSException
|
||||
from mock import Mock, patch
|
||||
from mock import Mock, PropertyMock, patch
|
||||
from patroni.dcs.etcd import AbstractDCS, EtcdClient, Cluster, Etcd, EtcdError, DnsCachingResolver
|
||||
from patroni.exceptions import DCSError
|
||||
from patroni.utils import Retry
|
||||
@@ -87,7 +87,8 @@ def dns_query(name, _):
|
||||
raise DNSException()
|
||||
srv = Mock()
|
||||
srv.port = 2380
|
||||
srv.target.to_text.return_value = 'localhost' if name == '_etcd-server._tcp.foobar' else '127.0.0.1'
|
||||
srv.target.to_text.return_value = \
|
||||
'localhost' if name in ['_etcd-server._tcp.foobar', '_etcd-server-baz._tcp.foobar'] else '127.0.0.1'
|
||||
return [srv]
|
||||
|
||||
|
||||
@@ -183,6 +184,7 @@ class TestClient(unittest.TestCase):
|
||||
|
||||
def test__get_machines_cache_from_srv(self):
|
||||
self.client._get_machines_cache_from_srv('foobar')
|
||||
self.client._get_machines_cache_from_srv('foobar', 'baz')
|
||||
self.client.get_srv_record = Mock(return_value=[('localhost', 2380)])
|
||||
self.client._get_machines_cache_from_srv('blabla')
|
||||
|
||||
@@ -275,7 +277,9 @@ class TestEtcd(unittest.TestCase):
|
||||
self.etcd._base_path = '/service/failed'
|
||||
self.assertFalse(self.etcd.attempt_to_acquire_leader())
|
||||
|
||||
@patch.object(Cluster, 'min_version', PropertyMock(return_value=(2, 0)))
|
||||
def test_write_leader_optime(self):
|
||||
self.etcd.get_cluster()
|
||||
self.etcd.write_leader_optime('0')
|
||||
|
||||
def test_update_leader(self):
|
||||
@@ -319,3 +323,6 @@ class TestEtcd(unittest.TestCase):
|
||||
|
||||
def test_set_history_value(self):
|
||||
self.assertFalse(self.etcd.set_history_value('{}'))
|
||||
|
||||
def test_last_seen(self):
|
||||
self.assertIsNotNone(self.etcd.last_seen)
|
||||
|
||||
+80
-39
@@ -19,7 +19,7 @@ from patroni.utils import tzutc
|
||||
from patroni.watchdog import Watchdog
|
||||
from six.moves import builtins
|
||||
|
||||
from . import PostgresInit, MockPostmaster, psycopg2_connect, requests_get
|
||||
from . import PostgresInit, MockPostmaster, psycopg_connect, requests_get
|
||||
from .test_etcd import socket_getaddrinfo, etcd_read, etcd_write
|
||||
|
||||
SYSID = '12345678901'
|
||||
@@ -35,8 +35,8 @@ def false(*args, **kwargs):
|
||||
|
||||
def get_cluster(initialize, leader, members, failover, sync, cluster_config=None):
|
||||
t = datetime.datetime.now().isoformat()
|
||||
history = TimelineHistory(1, '[[1,67197376,"no recovery target specified","' + t + '"]]',
|
||||
[(1, 67197376, 'no recovery target specified', t)])
|
||||
history = TimelineHistory(1, '[[1,67197376,"no recovery target specified","' + t + '","foo"]]',
|
||||
[(1, 67197376, 'no recovery target specified', t, 'foo')])
|
||||
cluster_config = cluster_config or ClusterConfig(1, {'check_timeline': True}, 1)
|
||||
return Cluster(initialize, cluster_config, leader, 10, members, failover, sync, history, None)
|
||||
|
||||
@@ -80,13 +80,14 @@ def get_standby_cluster_initialized_with_only_leader(failover=None, sync=None):
|
||||
)
|
||||
|
||||
|
||||
def get_node_status(reachable=True, in_recovery=True, timeline=2,
|
||||
wal_position=10, nofailover=False, watchdog_failed=False):
|
||||
def get_node_status(reachable=True, in_recovery=True, dcs_last_seen=0,
|
||||
timeline=2, wal_position=10, nofailover=False,
|
||||
watchdog_failed=False):
|
||||
def fetch_node_status(e):
|
||||
tags = {}
|
||||
if nofailover:
|
||||
tags['nofailover'] = True
|
||||
return _MemberStatus(e, reachable, in_recovery, timeline, wal_position, tags, watchdog_failed)
|
||||
return _MemberStatus(e, reachable, in_recovery, dcs_last_seen, timeline, wal_position, tags, watchdog_failed)
|
||||
return fetch_node_status
|
||||
|
||||
|
||||
@@ -283,6 +284,18 @@ class TestHa(PostgresInit):
|
||||
self.ha.patroni.config.set_dynamic_configuration({'maximum_lag_on_failover': 10})
|
||||
self.assertEqual(self.ha.run_cycle(), 'terminated crash recovery because of startup timeout')
|
||||
|
||||
@patch.object(Rewind, 'ensure_clean_shutdown', Mock())
|
||||
@patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True))
|
||||
@patch.object(Rewind, 'can_rewind', PropertyMock(return_value=True))
|
||||
def test_crash_recovery_before_rewind(self):
|
||||
self.p.is_leader = false
|
||||
self.p.is_running = false
|
||||
self.p.controldata = lambda: {'Database cluster state': 'in archive recovery',
|
||||
'Database system identifier': SYSID}
|
||||
self.ha._rewind.trigger_check_diverged_lsn()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.assertEqual(self.ha.run_cycle(), 'doing crash recovery in a single user mode')
|
||||
|
||||
@patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True))
|
||||
@patch.object(Rewind, 'can_rewind', PropertyMock(return_value=True))
|
||||
def test_recover_with_rewind(self):
|
||||
@@ -301,6 +314,7 @@ class TestHa(PostgresInit):
|
||||
self.assertEqual(self.ha.run_cycle(), 'fake')
|
||||
|
||||
@patch.object(Rewind, 'rewind_or_reinitialize_needed_and_possible', Mock(return_value=True))
|
||||
@patch.object(Rewind, 'should_remove_data_directory_on_diverged_timelines', PropertyMock(return_value=True))
|
||||
@patch.object(Bootstrap, 'create_replica', Mock(return_value=1))
|
||||
def test_recover_with_reinitialize(self):
|
||||
self.p.is_running = false
|
||||
@@ -322,7 +336,7 @@ class TestHa(PostgresInit):
|
||||
self.p.controldata = lambda: {'Database cluster state': 'in production', 'Database system identifier': SYSID}
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to leader because I had the session lock')
|
||||
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
def test_acquire_lock_as_master(self):
|
||||
self.assertEqual(self.ha.run_cycle(), 'acquired session lock as a leader')
|
||||
|
||||
@@ -349,7 +363,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.has_lock = true
|
||||
self.p.is_leader = false
|
||||
self.p.set_role('master')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
|
||||
def test_demote_after_failing_to_obtain_lock(self):
|
||||
self.ha.acquire_lock = false
|
||||
@@ -389,7 +403,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
self.ha.cluster.is_unlocked = false
|
||||
self.ha.has_lock = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
|
||||
def test_demote_because_not_having_lock(self):
|
||||
self.ha.cluster.is_unlocked = false
|
||||
@@ -401,6 +415,8 @@ class TestHa(PostgresInit):
|
||||
self.ha.has_lock = true
|
||||
self.ha.update_lock = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'demoted self because failed to update leader lock in DCS')
|
||||
with patch.object(Ha, '_get_node_to_follow', Mock(side_effect=DCSError('foo'))):
|
||||
self.assertEqual(self.ha.run_cycle(), 'demoted self because failed to update leader lock in DCS')
|
||||
self.p.is_leader = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'not promoting because failed to update leader lock in DCS')
|
||||
|
||||
@@ -408,16 +424,16 @@ class TestHa(PostgresInit):
|
||||
def test_follow(self):
|
||||
self.ha.cluster.is_unlocked = false
|
||||
self.p.is_leader = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am a secondary (postgresql0) and following a leader ()')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), a secondary, and following a leader ()')
|
||||
self.ha.patroni.replicatefrom = "foo"
|
||||
self.p.config.check_recovery_conf = Mock(return_value=(True, False))
|
||||
self.ha.cluster.config.data.update({'slots': {'l': {'database': 'a', 'plugin': 'b'}}})
|
||||
self.ha.cluster.members[1].data['tags']['replicatefrom'] = 'postgresql0'
|
||||
self.ha.patroni.nofailover = True
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am a secondary (postgresql0) and following a leader ()')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), a secondary, and following a leader ()')
|
||||
del self.ha.cluster.config.data['slots']
|
||||
self.ha.cluster.config.data.update({'postgresql': {'use_slots': False}})
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am a secondary (postgresql0) and following a leader ()')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), a secondary, and following a leader ()')
|
||||
del self.ha.cluster.config.data['postgresql']['use_slots']
|
||||
|
||||
def test_follow_in_pause(self):
|
||||
@@ -457,6 +473,8 @@ class TestHa(PostgresInit):
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.assertEqual(self.ha.bootstrap(), 'failed to acquire initialize lock')
|
||||
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch.object(Postgresql, 'connection', Mock(return_value=None))
|
||||
def test_bootstrap_initialized_new_cluster(self):
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.e.initialize = true
|
||||
@@ -474,6 +492,8 @@ class TestHa(PostgresInit):
|
||||
self.p.is_running = false
|
||||
self.assertRaises(PatroniFatalException, self.ha.post_bootstrap)
|
||||
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch.object(Postgresql, 'connection', Mock(return_value=None))
|
||||
def test_bootstrap_release_initialize_key_on_watchdog_failure(self):
|
||||
self.ha.cluster = get_cluster_not_initialized_without_leader()
|
||||
self.e.initialize = true
|
||||
@@ -484,7 +504,7 @@ class TestHa(PostgresInit):
|
||||
self.assertEqual(self.ha.post_bootstrap(), 'running post_bootstrap')
|
||||
self.assertRaises(PatroniFatalException, self.ha.post_bootstrap)
|
||||
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
def test_reinitialize(self):
|
||||
self.assertIsNotNone(self.ha.reinitialize())
|
||||
|
||||
@@ -536,27 +556,27 @@ class TestHa(PostgresInit):
|
||||
self.ha.fetch_node_status = get_node_status()
|
||||
self.ha.has_lock = true
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', '', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', self.p.name, None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'blabla', None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
f = Failover(0, self.p.name, '', None)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(f)
|
||||
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||
self.ha._rewind.rewind_or_reinitialize_needed_and_possible = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'manual failover: demoting myself')
|
||||
self.ha.fetch_node_status = get_node_status(nofailover=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(watchdog_failed=True)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(timeline=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.fetch_node_status = get_node_status(wal_position=1)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
# manual failover from the previous leader to us won't happen if we hold the nofailover flag
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, None))
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
|
||||
# Failover scheduled time must include timezone
|
||||
scheduled = datetime.datetime.now()
|
||||
@@ -565,28 +585,28 @@ class TestHa(PostgresInit):
|
||||
|
||||
scheduled = datetime.datetime.utcnow().replace(tzinfo=tzutc)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0) the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
|
||||
scheduled = scheduled + datetime.timedelta(seconds=30)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0) the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
|
||||
scheduled = scheduled + datetime.timedelta(seconds=-600)
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0) the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
|
||||
scheduled = None
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('no action. I am (postgresql0) the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
|
||||
def test_manual_failover_from_leader_in_pause(self):
|
||||
self.ha.has_lock = true
|
||||
self.ha.is_paused = true
|
||||
scheduled = datetime.datetime.now()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled))
|
||||
self.assertEqual('PAUSE: no action. I am (postgresql0) the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual('PAUSE: no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, '', None))
|
||||
self.assertEqual('PAUSE: no action. I am (postgresql0) the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual('PAUSE: no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
|
||||
def test_manual_failover_from_leader_in_synchronous_mode(self):
|
||||
self.p.is_leader = true
|
||||
@@ -595,7 +615,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.is_failover_possible = false
|
||||
self.ha.process_sync_replication = Mock()
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, None))
|
||||
self.assertEqual('no action. I am (postgresql0) the leader with the lock', self.ha.run_cycle())
|
||||
self.assertEqual('no action. I am (postgresql0), the leader with the lock', self.ha.run_cycle())
|
||||
self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, self.p.name, 'a', None), (self.p.name, 'a'))
|
||||
self.ha.is_failover_possible = true
|
||||
self.assertEqual('manual failover: demoting myself', self.ha.run_cycle())
|
||||
@@ -623,6 +643,11 @@ class TestHa(PostgresInit):
|
||||
# same as previous, but set the current member to nofailover. In no case it should be elected as a leader
|
||||
self.ha.patroni.nofailover = True
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because I am not allowed to promote')
|
||||
# in sync mode only the sync node is allowed to take over
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, 'leader', 'other', None))
|
||||
self.ha.patroni.nofailover = False
|
||||
self.ha.is_synchronous_mode = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node')
|
||||
|
||||
def test_manual_failover_process_no_leader_in_pause(self):
|
||||
self.ha.is_paused = true
|
||||
@@ -756,7 +781,7 @@ class TestHa(PostgresInit):
|
||||
self.p.config.check_recovery_conf = Mock(return_value=(False, False))
|
||||
self.ha._leader_timeline = 1
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to a standby leader because i had the session lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (leader) the standby leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (leader), the standby leader with the lock')
|
||||
self.p.set_role('replica')
|
||||
self.p.config.check_recovery_conf = Mock(return_value=(True, False))
|
||||
self.assertEqual(self.ha.run_cycle(), 'promoted self to a standby leader because i had the session lock')
|
||||
@@ -766,7 +791,7 @@ class TestHa(PostgresInit):
|
||||
self.p.name = 'replica'
|
||||
self.ha.cluster = get_standby_cluster_initialized_with_only_leader()
|
||||
self.assertEqual(self.ha.run_cycle(),
|
||||
'no action. I am a secondary (replica) and following a standby leader (leader)')
|
||||
'no action. I am (replica), a secondary, and following a standby leader (leader)')
|
||||
with patch.object(Leader, 'conn_url', PropertyMock(return_value='')):
|
||||
self.assertEqual(self.ha.run_cycle(), 'continue following the old known standby leader')
|
||||
|
||||
@@ -860,7 +885,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.has_lock = false
|
||||
self.p.is_leader = false
|
||||
self.assertEqual(self.ha.run_cycle(),
|
||||
'no action. I am a secondary (postgresql0) and following a leader (leader)')
|
||||
'no action. I am (postgresql0), a secondary, and following a leader (leader)')
|
||||
check_calls([(update_lock, False), (demote, False)])
|
||||
|
||||
def test_manual_failover_while_starting(self):
|
||||
@@ -1085,7 +1110,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.cluster.config.data.clear()
|
||||
self.ha.has_lock = true
|
||||
self.ha.cluster.is_unlocked = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
|
||||
def test_watch(self):
|
||||
self.ha.cluster = get_cluster_initialized_with_leader()
|
||||
@@ -1097,6 +1122,14 @@ class TestHa(PostgresInit):
|
||||
def test_shutdown(self):
|
||||
self.p.is_running = false
|
||||
self.ha.is_leader = true
|
||||
|
||||
def stop(*args, **kwargs):
|
||||
kwargs['on_shutdown'](123)
|
||||
|
||||
self.p.stop = stop
|
||||
self.ha.shutdown()
|
||||
|
||||
self.ha.is_failover_possible = true
|
||||
self.ha.shutdown()
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
@@ -1120,7 +1153,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.cluster.is_unlocked = false
|
||||
for tl in (1, 3):
|
||||
self.p.get_master_timeline = Mock(return_value=tl)
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
|
||||
@patch('sys.exit', return_value=1)
|
||||
def test_abort_join(self, exit_mock):
|
||||
@@ -1133,11 +1166,11 @@ class TestHa(PostgresInit):
|
||||
self.ha.has_lock = true
|
||||
self.ha.cluster.is_unlocked = false
|
||||
self.ha.is_paused = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: no action. I am (postgresql0), the leader with the lock')
|
||||
self.ha.is_paused = false
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (postgresql0), the leader with the lock')
|
||||
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
def test_permanent_logical_slots_after_promote(self):
|
||||
config = ClusterConfig(1, {'slots': {'l': {'database': 'postgres', 'plugin': 'test_decoding'}}}, 1)
|
||||
self.p.name = 'other'
|
||||
@@ -1145,7 +1178,7 @@ class TestHa(PostgresInit):
|
||||
self.assertEqual(self.ha.run_cycle(), 'acquired session lock as a leader')
|
||||
self.ha.cluster = get_cluster_initialized_without_leader(leader=True, cluster_config=config)
|
||||
self.ha.has_lock = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (other) the leader with the lock')
|
||||
self.assertEqual(self.ha.run_cycle(), 'no action. I am (other), the leader with the lock')
|
||||
|
||||
@patch.object(Cluster, 'has_member', true)
|
||||
def test_run_cycle(self):
|
||||
@@ -1169,7 +1202,7 @@ class TestHa(PostgresInit):
|
||||
self.ha.has_lock = true
|
||||
self.assertEqual(self.ha.run_cycle(), 'PAUSE: released leader key voluntarily due to the system ID mismatch')
|
||||
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch('os.path.exists', Mock(return_value=True))
|
||||
@patch('shutil.rmtree', Mock())
|
||||
@patch('os.makedirs', Mock())
|
||||
@@ -1179,8 +1212,16 @@ class TestHa(PostgresInit):
|
||||
@patch('os.rename', Mock())
|
||||
@patch('patroni.postgresql.Postgresql.is_starting', Mock(return_value=False))
|
||||
@patch.object(builtins, 'open', mock_open())
|
||||
@patch.object(SlotsHandler, 'sync_replication_slots', Mock(return_value=['foo']))
|
||||
@patch.object(ConfigHandler, 'check_recovery_conf', Mock(return_value=(False, False)))
|
||||
@patch.object(Postgresql, 'major_version', PropertyMock(return_value=130000))
|
||||
@patch.object(SlotsHandler, 'sync_replication_slots', Mock(return_value=['ls']))
|
||||
def test_follow_copy(self):
|
||||
self.ha.cluster.is_unlocked = false
|
||||
self.ha.cluster.config.data['slots'] = {'ls': {'database': 'a', 'plugin': 'b'}}
|
||||
self.p.is_leader = false
|
||||
self.assertTrue(self.ha.run_cycle().startswith('Copying logical slots'))
|
||||
|
||||
def test_is_failover_possible(self):
|
||||
self.ha.fetch_node_status = Mock(return_value=_MemberStatus(self.ha.cluster.members[0],
|
||||
True, True, 0, 2, None, {}, False))
|
||||
self.assertFalse(self.ha.is_failover_possible(self.ha.cluster.members))
|
||||
|
||||
@@ -1,9 +1,10 @@
|
||||
import datetime
|
||||
import json
|
||||
import socket
|
||||
import time
|
||||
import unittest
|
||||
|
||||
from mock import Mock, mock_open, patch
|
||||
from mock import Mock, PropertyMock, mock_open, patch
|
||||
from patroni.dcs.kubernetes import k8s_client, k8s_config, K8sConfig, K8sConnectionFailed,\
|
||||
K8sException, K8sObject, Kubernetes, KubernetesError, KubernetesRetriableException,\
|
||||
Retry, RetryFailedError, SERVICE_HOST_ENV_NAME, SERVICE_PORT_ENV_NAME
|
||||
@@ -79,6 +80,28 @@ class TestK8sConfig(unittest.TestCase):
|
||||
self.assertRaises(k8s_config.ConfigException, k8s_config.load_incluster_config)
|
||||
k8s_config.load_incluster_config()
|
||||
self.assertEqual(k8s_config.server, 'https://a:1')
|
||||
self.assertEqual(k8s_config.headers.get('authorization'), 'Bearer a')
|
||||
|
||||
def test_refresh_token(self):
|
||||
with patch('os.environ', {SERVICE_HOST_ENV_NAME: 'a', SERVICE_PORT_ENV_NAME: '1'}),\
|
||||
patch('os.path.isfile', Mock(side_effect=[True, True, False, True, True, True])),\
|
||||
patch.object(builtins, 'open', Mock(side_effect=[
|
||||
mock_open(read_data='cert')(), mock_open(read_data='a')(),
|
||||
mock_open()(), mock_open(read_data='b')(), mock_open(read_data='c')()])):
|
||||
k8s_config.load_incluster_config(token_refresh_interval=datetime.timedelta(milliseconds=100))
|
||||
self.assertEqual(k8s_config.headers.get('authorization'), 'Bearer a')
|
||||
time.sleep(0.1)
|
||||
# token file doesn't exist
|
||||
self.assertEqual(k8s_config.headers.get('authorization'), 'Bearer a')
|
||||
# token file is empty
|
||||
self.assertEqual(k8s_config.headers.get('authorization'), 'Bearer a')
|
||||
# token refreshed
|
||||
self.assertEqual(k8s_config.headers.get('authorization'), 'Bearer b')
|
||||
time.sleep(0.1)
|
||||
# token refreshed
|
||||
self.assertEqual(k8s_config.headers.get('authorization'), 'Bearer c')
|
||||
# no need to refresh token
|
||||
self.assertEqual(k8s_config.headers.get('authorization'), 'Bearer c')
|
||||
|
||||
def test_load_kube_config(self):
|
||||
config = {
|
||||
@@ -212,7 +235,9 @@ class TestKubernetesConfigMaps(BaseTestKubernetes):
|
||||
self.k.manual_failover('foo', 'bar')
|
||||
|
||||
def test_set_config_value(self):
|
||||
self.k.set_config_value('{}')
|
||||
with patch.object(k8s_client.CoreV1Api, 'patch_namespaced_config_map',
|
||||
Mock(side_effect=k8s_client.rest.ApiException(409, '')), create=True):
|
||||
self.k.set_config_value('{}', 1)
|
||||
|
||||
@patch.object(k8s_client.CoreV1Api, 'patch_namespaced_pod', create=True)
|
||||
def test_touch_member(self, mock_patch_namespaced_pod):
|
||||
@@ -322,3 +347,18 @@ class TestCacheBuilder(BaseTestKubernetes):
|
||||
def test__list(self):
|
||||
self.k._pods._func = Mock(side_effect=Exception)
|
||||
self.assertRaises(Exception, self.k._pods._list)
|
||||
|
||||
@patch('patroni.dcs.kubernetes.ObjectCache._watch', Mock(return_value=None))
|
||||
def test__do_watch(self):
|
||||
self.assertRaises(AttributeError, self.k._kinds._do_watch, '1')
|
||||
|
||||
@patch.object(k8s_client.CoreV1Api, 'list_namespaced_config_map', mock_list_namespaced_config_map, create=True)
|
||||
@patch('patroni.dcs.kubernetes.ObjectCache._watch')
|
||||
def test_kill_stream(self, mock_watch):
|
||||
self.k._kinds.kill_stream()
|
||||
mock_watch.return_value.read_chunked.return_value = []
|
||||
mock_watch.return_value.connection.sock.close.side_effect = Exception
|
||||
self.k._kinds._do_watch('1')
|
||||
self.k._kinds.kill_stream()
|
||||
type(mock_watch.return_value).connection = PropertyMock(side_effect=Exception)
|
||||
self.k._kinds.kill_stream()
|
||||
|
||||
+17
-8
@@ -13,15 +13,24 @@ from patroni.dcs.etcd import AbstractEtcdClientWithFailover
|
||||
from patroni.exceptions import DCSError
|
||||
from patroni.postgresql import Postgresql
|
||||
from patroni.postgresql.config import ConfigHandler
|
||||
from patroni import Patroni, main as _main, patroni_main, check_psycopg2
|
||||
from patroni import check_psycopg
|
||||
from patroni.__main__ import Patroni, main as _main, patroni_main
|
||||
from six.moves import BaseHTTPServer, builtins
|
||||
from threading import Thread
|
||||
|
||||
from . import psycopg2_connect, SleepException
|
||||
from . import psycopg_connect, SleepException
|
||||
from .test_etcd import etcd_read, etcd_write
|
||||
from .test_postgresql import MockPostmaster
|
||||
|
||||
|
||||
def mock_import(*args, **kwargs):
|
||||
if args[0] == 'psycopg':
|
||||
raise ImportError
|
||||
ret = Mock()
|
||||
ret.__version__ = '2.5.3.dev1 a b c'
|
||||
return ret
|
||||
|
||||
|
||||
class MockFrozenImporter(object):
|
||||
|
||||
toc = set(['patroni.dcs.etcd'])
|
||||
@@ -29,7 +38,7 @@ class MockFrozenImporter(object):
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
@patch('subprocess.call', Mock(return_value=0))
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch.object(ConfigHandler, 'append_pg_hba', Mock())
|
||||
@patch.object(ConfigHandler, 'write_postgresql_conf', Mock())
|
||||
@patch.object(ConfigHandler, 'write_recovery_conf', Mock())
|
||||
@@ -89,7 +98,7 @@ class TestPatroni(unittest.TestCase):
|
||||
|
||||
@patch('os.getpid')
|
||||
@patch('multiprocessing.Process')
|
||||
@patch('patroni.patroni_main', Mock())
|
||||
@patch('patroni.__main__.patroni_main', Mock())
|
||||
def test_patroni_main(self, mock_process, mock_getpid):
|
||||
mock_getpid.return_value = 2
|
||||
_main()
|
||||
@@ -181,8 +190,8 @@ class TestPatroni(unittest.TestCase):
|
||||
self.p.ha.shutdown = Mock(side_effect=Exception)
|
||||
self.p.shutdown()
|
||||
|
||||
def test_check_psycopg2(self):
|
||||
def test_check_psycopg(self):
|
||||
with patch.object(builtins, '__import__', Mock(side_effect=ImportError)):
|
||||
self.assertRaises(SystemExit, check_psycopg2)
|
||||
with patch('psycopg2.__version__', '2.5.3.dev1 a b c'):
|
||||
self.assertRaises(SystemExit, check_psycopg2)
|
||||
self.assertRaises(SystemExit, check_psycopg)
|
||||
with patch.object(builtins, '__import__', mock_import):
|
||||
self.assertRaises(SystemExit, check_psycopg)
|
||||
|
||||
+31
-12
@@ -1,12 +1,14 @@
|
||||
import datetime
|
||||
import os
|
||||
import psutil
|
||||
import psycopg2
|
||||
import re
|
||||
import subprocess
|
||||
import time
|
||||
|
||||
from mock import Mock, MagicMock, PropertyMock, patch, mock_open
|
||||
|
||||
import patroni.psycopg as psycopg
|
||||
|
||||
from patroni.async_executor import CriticalTask
|
||||
from patroni.dcs import Cluster, RemoteMember, SyncState
|
||||
from patroni.exceptions import PostgresConnectionException, PatroniException
|
||||
@@ -17,7 +19,7 @@ from patroni.utils import RetryFailedError
|
||||
from six.moves import builtins
|
||||
from threading import Thread, current_thread
|
||||
|
||||
from . import BaseTestPostgresql, MockCursor, MockPostmaster, psycopg2_connect
|
||||
from . import BaseTestPostgresql, MockCursor, MockPostmaster, psycopg_connect
|
||||
|
||||
|
||||
mtime_ret = {}
|
||||
@@ -87,13 +89,13 @@ Data page checksum version: 0
|
||||
|
||||
|
||||
@patch('subprocess.call', Mock(return_value=0))
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
class TestPostgresql(BaseTestPostgresql):
|
||||
|
||||
@patch('subprocess.call', Mock(return_value=0))
|
||||
@patch('os.rename', Mock())
|
||||
@patch('patroni.postgresql.CallbackExecutor', Mock())
|
||||
@patch.object(Postgresql, 'get_major_version', Mock(return_value=130000))
|
||||
@patch.object(Postgresql, 'get_major_version', Mock(return_value=140000))
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||
def setUp(self):
|
||||
super(TestPostgresql, self).setUp()
|
||||
@@ -203,6 +205,21 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
mock_postmaster.signal_stop.side_effect = [None, True]
|
||||
self.assertTrue(self.p.stop(on_safepoint=mock_callback, stop_timeout=30))
|
||||
|
||||
@patch('time.sleep', Mock())
|
||||
@patch.object(Postgresql, 'is_running', MockPostmaster)
|
||||
@patch.object(Postgresql, '_wait_for_connection_close', Mock())
|
||||
@patch.object(Postgresql, 'latest_checkpoint_location', Mock(return_value='7'))
|
||||
def test__do_stop(self):
|
||||
mock_callback = Mock()
|
||||
with patch.object(Postgresql, 'controldata', Mock(return_value={'Database cluster state': 'shut down'})):
|
||||
self.assertTrue(self.p.stop(on_shutdown=mock_callback, stop_timeout=3))
|
||||
mock_callback.assert_called()
|
||||
with patch.object(Postgresql, 'controldata',
|
||||
Mock(return_value={'Database cluster state': 'shut down in recovery'})):
|
||||
self.assertTrue(self.p.stop(on_shutdown=mock_callback, stop_timeout=3))
|
||||
with patch.object(Postgresql, 'controldata', Mock(return_value={'Database cluster state': 'shutting down'})):
|
||||
self.assertTrue(self.p.stop(on_shutdown=mock_callback, stop_timeout=3))
|
||||
|
||||
def test_restart(self):
|
||||
self.p.start = Mock(return_value=False)
|
||||
self.assertFalse(self.p.restart())
|
||||
@@ -243,8 +260,8 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
with patch('patroni.postgresql.config.ConfigHandler.primary_conninfo_params', Mock(return_value=conninfo)):
|
||||
mock_get_pg_settings.return_value['recovery_min_apply_delay'][1] = '1'
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
||||
mock_get_pg_settings.return_value['primary_conninfo'][1] = 'host=1 passfile='\
|
||||
+ re.sub(r'([\'\\ ])', r'\\\1', self.p.config._pgpass)
|
||||
mock_get_pg_settings.return_value['primary_conninfo'][1] = 'host=1 target_session_attrs=read-write'\
|
||||
+ ' passfile=' + re.sub(r'([\'\\ ])', r'\\\1', self.p.config._pgpass)
|
||||
mock_get_pg_settings.return_value['recovery_min_apply_delay'][1] = '0'
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
||||
self.p.config.write_recovery_conf({'standby_mode': 'on', 'primary_conninfo': conninfo.copy()})
|
||||
@@ -270,6 +287,8 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
mock_get_pg_settings.side_effect = Exception
|
||||
with patch('patroni.postgresql.config.mtime', mock_mtime):
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (True, True))
|
||||
with patch.object(Postgresql, 'is_starting', Mock(return_value=True)):
|
||||
self.assertEqual(self.p.config.check_recovery_conf(None), (False, False))
|
||||
|
||||
@patch.object(Postgresql, 'major_version', PropertyMock(return_value=100000))
|
||||
@patch.object(Postgresql, 'primary_conninfo', Mock(return_value='host=1'))
|
||||
@@ -304,7 +323,7 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
m = RemoteMember('1', {'restore_command': '2', 'primary_slot_name': 'foo', 'conn_kwargs': {'host': 'bar'}})
|
||||
self.p.follow(m)
|
||||
|
||||
@patch.object(MockCursor, 'execute', Mock(side_effect=psycopg2.OperationalError))
|
||||
@patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError))
|
||||
def test__query(self):
|
||||
self.assertRaises(PostgresConnectionException, self.p._query, 'blabla')
|
||||
self.p._state = 'restarting'
|
||||
@@ -313,14 +332,14 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
def test_query(self):
|
||||
self.p.query('select 1')
|
||||
self.assertRaises(PostgresConnectionException, self.p.query, 'RetryFailedError')
|
||||
self.assertRaises(psycopg2.ProgrammingError, self.p.query, 'blabla')
|
||||
self.assertRaises(psycopg.ProgrammingError, self.p.query, 'blabla')
|
||||
|
||||
@patch.object(Postgresql, 'pg_isready', Mock(return_value=STATE_REJECT))
|
||||
def test_is_leader(self):
|
||||
self.assertTrue(self.p.is_leader())
|
||||
self.p.reset_cluster_info_state(None)
|
||||
with patch.object(Postgresql, '_query', Mock(side_effect=RetryFailedError(''))):
|
||||
self.assertRaises(PostgresConnectionException, self.p.is_leader)
|
||||
self.assertFalse(self.p.is_leader())
|
||||
|
||||
@patch.object(Postgresql, 'controldata', Mock(return_value={'Database cluster state': 'shut down',
|
||||
'Latest checkpoint location': '0/1ADBC18',
|
||||
@@ -415,7 +434,7 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=MockPostmaster()))
|
||||
def test_is_leader_exception(self):
|
||||
self.p.start()
|
||||
self.p.query = Mock(side_effect=psycopg2.OperationalError("not supported"))
|
||||
self.p.query = Mock(side_effect=psycopg.OperationalError("not supported"))
|
||||
self.assertTrue(self.p.stop())
|
||||
|
||||
@patch('os.rename', Mock())
|
||||
@@ -544,7 +563,7 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
t.start()
|
||||
t.join()
|
||||
|
||||
with patch.object(MockCursor, "execute", side_effect=psycopg2.Error):
|
||||
with patch.object(MockCursor, "execute", side_effect=psycopg.Error):
|
||||
self.assertIsNone(self.p.postmaster_start_time())
|
||||
|
||||
def test_check_for_startup(self):
|
||||
@@ -707,7 +726,7 @@ class TestPostgresql(BaseTestPostgresql):
|
||||
self.p.stop(on_safepoint=mock_callback)
|
||||
|
||||
mock_postmaster.is_running.side_effect = [True, False, False]
|
||||
with patch.object(MockCursor, "execute", Mock(side_effect=psycopg2.Error)):
|
||||
with patch.object(MockCursor, "execute", Mock(side_effect=psycopg.Error)):
|
||||
self.p.stop(on_safepoint=mock_callback)
|
||||
|
||||
def test_terminate_starting_postmaster(self):
|
||||
|
||||
@@ -73,7 +73,7 @@ class TestPostmasterProcess(unittest.TestCase):
|
||||
|
||||
# all processes successfully stopped
|
||||
mock_children.return_value = [Mock()]
|
||||
mock_children.return_value[0].kill.side_effect = psutil.Error
|
||||
mock_children.return_value[0].kill.side_effect = psutil.NoSuchProcess(123)
|
||||
self.assertTrue(proc.signal_kill())
|
||||
|
||||
# postmaster has gone before suspend
|
||||
@@ -81,17 +81,17 @@ class TestPostmasterProcess(unittest.TestCase):
|
||||
self.assertTrue(proc.signal_kill())
|
||||
|
||||
# postmaster has gone before we got a list of children
|
||||
mock_suspend.side_effect = psutil.Error()
|
||||
mock_suspend.side_effect = psutil.AccessDenied()
|
||||
mock_children.side_effect = psutil.NoSuchProcess(123)
|
||||
self.assertTrue(proc.signal_kill())
|
||||
|
||||
# postmaster has gone after we got a list of children
|
||||
mock_children.side_effect = psutil.Error()
|
||||
mock_children.side_effect = psutil.AccessDenied()
|
||||
mock_kill.side_effect = psutil.NoSuchProcess(123)
|
||||
self.assertTrue(proc.signal_kill())
|
||||
|
||||
# failed to kill postmaster
|
||||
mock_kill.side_effect = psutil.AccessDenied(123)
|
||||
mock_kill.side_effect = psutil.AccessDenied()
|
||||
self.assertFalse(proc.signal_kill())
|
||||
|
||||
@patch('psutil.Process.__init__', Mock())
|
||||
|
||||
+1
-1
@@ -157,6 +157,6 @@ class TestRaft(unittest.TestCase):
|
||||
@patch('threading.Event')
|
||||
def test_init(self, mock_event, mock_kvstore):
|
||||
mock_kvstore.return_value.applied_local_log = False
|
||||
mock_event.return_value.isSet.side_effect = [False, True]
|
||||
mock_event.return_value.is_set.side_effect = [False, True]
|
||||
self.assertIsNotNone(Raft({'ttl': 30, 'scope': 'test', 'name': 'pg', 'patronictl': True,
|
||||
'self_addr': '1', 'data_dir': self._TMP}))
|
||||
|
||||
+27
-15
@@ -5,7 +5,7 @@ from patroni.postgresql.cancellable import CancellableSubprocess
|
||||
from patroni.postgresql.rewind import Rewind
|
||||
from six.moves import builtins
|
||||
|
||||
from . import BaseTestPostgresql, MockCursor, psycopg2_connect
|
||||
from . import BaseTestPostgresql, MockCursor, psycopg_connect
|
||||
|
||||
|
||||
class MockThread(object):
|
||||
@@ -47,7 +47,7 @@ def mock_single_user_mode(self, communicate, options):
|
||||
|
||||
|
||||
@patch('subprocess.call', Mock(return_value=0))
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
class TestRewind(BaseTestPostgresql):
|
||||
|
||||
def setUp(self):
|
||||
@@ -66,7 +66,7 @@ class TestRewind(BaseTestPostgresql):
|
||||
|
||||
def test_pg_rewind(self):
|
||||
r = {'user': '', 'host': '', 'port': '', 'database': '', 'password': ''}
|
||||
with patch.object(Postgresql, 'major_version', PropertyMock(return_value=130000)),\
|
||||
with patch.object(Postgresql, 'major_version', PropertyMock(return_value=150000)),\
|
||||
patch.object(CancellableSubprocess, 'call', Mock(return_value=None)):
|
||||
with patch('subprocess.check_output', Mock(return_value=b'boo')):
|
||||
self.assertFalse(self.r.pg_rewind(r))
|
||||
@@ -102,6 +102,11 @@ class TestRewind(BaseTestPostgresql):
|
||||
@patch.object(Postgresql, 'start', Mock())
|
||||
def test_execute(self, mock_checkpoint):
|
||||
self.r.execute(self.leader)
|
||||
with patch.object(Postgresql, 'major_version', PropertyMock(return_value=130000)):
|
||||
self.r.execute(self.leader)
|
||||
with patch.object(MockCursor, 'fetchone', Mock(side_effect=Exception)):
|
||||
self.r.execute(self.leader)
|
||||
|
||||
with patch.object(Rewind, 'pg_rewind', Mock(return_value=False)):
|
||||
mock_checkpoint.side_effect = ['1', '', '', '']
|
||||
self.r.execute(self.leader)
|
||||
@@ -143,11 +148,16 @@ class TestRewind(BaseTestPostgresql):
|
||||
mock_check_leader_is_not_in_recovery.return_value = True
|
||||
self.assertFalse(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
||||
self.r.trigger_check_diverged_lsn()
|
||||
with patch('psycopg2.connect', Mock(side_effect=Exception)):
|
||||
with patch.object(MockCursor, 'fetchone', Mock(side_effect=[('', 3, '0/0'), ('', b'4\t0/40159C0\tn\n')])):
|
||||
self.assertTrue(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
||||
self.r.reset_state()
|
||||
self.r.trigger_check_diverged_lsn()
|
||||
with patch('patroni.psycopg.connect', Mock(side_effect=Exception)):
|
||||
self.assertFalse(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
||||
self.r.trigger_check_diverged_lsn()
|
||||
with patch.object(MockCursor, 'fetchone', Mock(side_effect=[('', 3, '0/0'), ('', b'3\t0/40159C0\tn\n')])):
|
||||
self.assertFalse(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
||||
with patch.object(MockCursor, 'fetchone', Mock(side_effect=[('', 3, '0/0'), ('', b'1\t0/40159C0\tn\n')])):
|
||||
self.assertTrue(self.r.rewind_or_reinitialize_needed_and_possible(self.leader))
|
||||
self.r.reset_state()
|
||||
self.r.trigger_check_diverged_lsn()
|
||||
with patch.object(MockCursor, 'fetchone', Mock(return_value=('', 1, '0/0'))):
|
||||
with patch.object(Rewind, '_get_local_timeline_lsn', Mock(return_value=(True, 1, '0/0'))):
|
||||
@@ -219,18 +229,20 @@ class TestRewind(BaseTestPostgresql):
|
||||
@patch('patroni.postgresql.rewind.Thread', MockThread)
|
||||
@patch.object(Postgresql, 'controldata')
|
||||
@patch.object(Postgresql, 'checkpoint')
|
||||
def test_ensure_checkpoint_after_promote(self, mock_checkpoint, mock_controldata):
|
||||
mock_checkpoint.return_value = None
|
||||
@patch.object(Postgresql, 'get_master_timeline')
|
||||
def test_ensure_checkpoint_after_promote(self, mock_get_master_timeline, mock_checkpoint, mock_controldata):
|
||||
mock_controldata.return_value = {"Latest checkpoint's TimeLineID": 1}
|
||||
mock_get_master_timeline.return_value = 1
|
||||
self.r.ensure_checkpoint_after_promote(Mock())
|
||||
|
||||
self.r.reset_state()
|
||||
mock_get_master_timeline.return_value = 2
|
||||
mock_checkpoint.return_value = 0
|
||||
self.r.ensure_checkpoint_after_promote(Mock())
|
||||
self.r.ensure_checkpoint_after_promote(Mock())
|
||||
|
||||
self.r.reset_state()
|
||||
mock_controldata.return_value = {"Latest checkpoint's TimeLineID": 1}
|
||||
|
||||
mock_controldata.side_effect = TypeError
|
||||
mock_checkpoint.side_effect = Exception
|
||||
self.r.ensure_checkpoint_after_promote(Mock())
|
||||
self.r.ensure_checkpoint_after_promote(Mock())
|
||||
|
||||
self.r.reset_state()
|
||||
mock_controldata.side_effect = TypeError
|
||||
self.r.ensure_checkpoint_after_promote(Mock())
|
||||
self.r.ensure_checkpoint_after_promote(Mock())
|
||||
|
||||
+31
-23
@@ -1,20 +1,20 @@
|
||||
import mock
|
||||
import os
|
||||
import psycopg2
|
||||
import unittest
|
||||
|
||||
|
||||
from mock import Mock, PropertyMock, patch
|
||||
|
||||
from patroni import psycopg
|
||||
from patroni.dcs import Cluster, ClusterConfig, Member
|
||||
from patroni.postgresql import Postgresql
|
||||
from patroni.postgresql.slots import SlotsHandler, fsync_dir
|
||||
|
||||
from . import BaseTestPostgresql, psycopg2_connect, MockCursor
|
||||
from . import BaseTestPostgresql, psycopg_connect, MockCursor
|
||||
|
||||
|
||||
@patch('subprocess.call', Mock(return_value=0))
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch.object(Postgresql, 'is_running', Mock(return_value=True))
|
||||
class TestSlotsHandler(BaseTestPostgresql):
|
||||
|
||||
@@ -27,6 +27,9 @@ class TestSlotsHandler(BaseTestPostgresql):
|
||||
super(TestSlotsHandler, self).setUp()
|
||||
self.s = self.p.slots_handler
|
||||
self.p.start()
|
||||
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}}}, 1)
|
||||
self.cluster = Cluster(True, config, self.leader, 0,
|
||||
[self.me, self.other, self.leadermem], None, None, None, {'ls': 12345})
|
||||
|
||||
def test_sync_replication_slots(self):
|
||||
config = ClusterConfig(1, {'slots': {'test_3': {'database': 'a', 'plugin': 'b'},
|
||||
@@ -34,7 +37,7 @@ class TestSlotsHandler(BaseTestPostgresql):
|
||||
'ignore_slots': [{'name': 'blabla'}]}, 1)
|
||||
cluster = Cluster(True, config, self.leader, 0,
|
||||
[self.me, self.other, self.leadermem], None, None, None, {'test_3': 10})
|
||||
with mock.patch('patroni.postgresql.Postgresql._query', Mock(side_effect=psycopg2.OperationalError)):
|
||||
with mock.patch('patroni.postgresql.Postgresql._query', Mock(side_effect=psycopg.OperationalError)):
|
||||
self.s.sync_replication_slots(cluster, False)
|
||||
self.p.set_role('standby_leader')
|
||||
self.s.sync_replication_slots(cluster, False)
|
||||
@@ -81,39 +84,44 @@ class TestSlotsHandler(BaseTestPostgresql):
|
||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||
def test__ensure_logical_slots_replica(self):
|
||||
self.p.set_role('replica')
|
||||
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}}}, 1)
|
||||
cluster = Cluster(True, config, self.leader, 0,
|
||||
[self.me, self.other, self.leadermem], None, None, None, {'ls': 12346})
|
||||
self.assertEqual(self.s.sync_replication_slots(cluster, False), [])
|
||||
self.cluster.slots['ls'] = 12346
|
||||
with patch.object(SlotsHandler, 'check_logical_slots_readiness', Mock()):
|
||||
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), [])
|
||||
self.s._schedule_load_slots = False
|
||||
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg2.errors.UndefinedFile)):
|
||||
self.assertEqual(self.s.sync_replication_slots(cluster, False), ['ls'])
|
||||
cluster.slots['ls'] = 'a'
|
||||
self.assertEqual(self.s.sync_replication_slots(cluster, False), [])
|
||||
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)),\
|
||||
patch.object(psycopg.OperationalError, 'diag') as mock_diag:
|
||||
type(mock_diag).sqlstate = PropertyMock(return_value='58P01')
|
||||
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), ['ls'])
|
||||
self.cluster.slots['ls'] = 'a'
|
||||
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), [])
|
||||
with patch.object(MockCursor, 'rowcount', PropertyMock(return_value=1), create=True):
|
||||
self.assertEqual(self.s.sync_replication_slots(cluster, False), ['ls'])
|
||||
self.assertEqual(self.s.sync_replication_slots(self.cluster, False), ['ls'])
|
||||
|
||||
@patch.object(MockCursor, 'execute', Mock(side_effect=psycopg2.OperationalError))
|
||||
def test_copy_logical_slots(self):
|
||||
self.s.copy_logical_slots(self.leader, ['foo'])
|
||||
self.cluster.config.data['slots']['ls']['database'] = 'b'
|
||||
self.s.copy_logical_slots(self.cluster, ['ls'])
|
||||
with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg.OperationalError)):
|
||||
self.s.copy_logical_slots(self.cluster, ['foo'])
|
||||
|
||||
@patch.object(Postgresql, 'stop', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||
def test_check_logical_slots_readiness(self):
|
||||
self.s.copy_logical_slots(self.leader, ['ls'])
|
||||
config = ClusterConfig(1, {'slots': {'ls': {'database': 'a', 'plugin': 'b'}}}, 1)
|
||||
cluster = Cluster(True, config, self.leader, 0,
|
||||
[self.me, self.other, self.leadermem], None, None, None, {'ls': 12345})
|
||||
self.assertEqual(self.s.sync_replication_slots(cluster, False), [])
|
||||
with patch.object(MockCursor, 'rowcount', PropertyMock(return_value=1), create=True):
|
||||
self.s.check_logical_slots_readiness(cluster, False, None)
|
||||
self.s.copy_logical_slots(self.cluster, ['ls'])
|
||||
with patch.object(MockCursor, '__iter__', Mock(return_value=iter([('postgresql0', None)]))),\
|
||||
patch.object(MockCursor, 'fetchone', Mock(side_effect=Exception)):
|
||||
self.assertIsNone(self.s.check_logical_slots_readiness(self.cluster, False, None))
|
||||
with patch.object(MockCursor, '__iter__', Mock(return_value=iter([('postgresql0', None)]))),\
|
||||
patch.object(MockCursor, 'fetchone', Mock(return_value=(False,))):
|
||||
self.assertIsNone(self.s.check_logical_slots_readiness(self.cluster, False, None))
|
||||
with patch.object(MockCursor, '__iter__', Mock(return_value=iter([('ls', 100)]))):
|
||||
self.s.check_logical_slots_readiness(self.cluster, False, None)
|
||||
|
||||
@patch.object(Postgresql, 'stop', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'start', Mock(return_value=True))
|
||||
@patch.object(Postgresql, 'is_leader', Mock(return_value=False))
|
||||
def test_on_promote(self):
|
||||
self.s.copy_logical_slots(self.leader, ['ls'])
|
||||
self.s.copy_logical_slots(self.cluster, ['ls'])
|
||||
self.s.on_promote()
|
||||
|
||||
@unittest.skipIf(os.name == 'nt', "Windows not supported")
|
||||
|
||||
@@ -1,14 +1,15 @@
|
||||
import psycopg2
|
||||
import subprocess
|
||||
import unittest
|
||||
|
||||
import patroni.psycopg as psycopg
|
||||
|
||||
from mock import Mock, PropertyMock, patch, mock_open
|
||||
from patroni.scripts import wale_restore
|
||||
from patroni.scripts.wale_restore import WALERestore, main as _main, get_major_version
|
||||
from six.moves import builtins
|
||||
from threading import current_thread
|
||||
|
||||
from . import MockConnect, psycopg2_connect
|
||||
from . import MockConnect, psycopg_connect
|
||||
|
||||
wale_output_header = (
|
||||
b'name\tlast_modified\t'
|
||||
@@ -34,7 +35,7 @@ WALE_TEST_RETRIES = 2
|
||||
@patch('os.makedirs', Mock(return_value=True))
|
||||
@patch('os.path.exists', Mock(return_value=True))
|
||||
@patch('os.path.isdir', Mock(return_value=True))
|
||||
@patch('psycopg2.connect', psycopg2_connect)
|
||||
@patch('patroni.psycopg.connect', psycopg_connect)
|
||||
@patch('subprocess.check_output', Mock(return_value=wale_output))
|
||||
class TestWALERestore(unittest.TestCase):
|
||||
|
||||
@@ -57,7 +58,7 @@ class TestWALERestore(unittest.TestCase):
|
||||
with patch('subprocess.check_output', Mock(return_value=wale_output.replace(b'167772160', b'1'))):
|
||||
self.assertFalse(self.wale_restore.should_use_s3_to_create_replica())
|
||||
|
||||
with patch('psycopg2.connect', Mock(side_effect=psycopg2.Error("foo"))):
|
||||
with patch('patroni.psycopg.connect', Mock(side_effect=psycopg.Error("foo"))):
|
||||
save_no_master = self.wale_restore.no_master
|
||||
save_master_connection = self.wale_restore.master_connection
|
||||
|
||||
|
||||
+10
-4
@@ -6,8 +6,8 @@ from kazoo.client import KazooClient, KazooState
|
||||
from kazoo.exceptions import NoNodeError, NodeExistsError
|
||||
from kazoo.handlers.threading import SequentialThreadingHandler
|
||||
from kazoo.protocol.states import KeeperState, ZnodeStat
|
||||
from mock import Mock, patch
|
||||
from patroni.dcs.zookeeper import Leader, PatroniKazooClient,\
|
||||
from mock import Mock, PropertyMock, patch
|
||||
from patroni.dcs.zookeeper import Cluster, Leader, PatroniKazooClient,\
|
||||
PatroniSequentialThreadingHandler, ZooKeeper, ZooKeeperError
|
||||
|
||||
|
||||
@@ -144,7 +144,8 @@ class TestZooKeeper(unittest.TestCase):
|
||||
@patch('patroni.dcs.zookeeper.PatroniKazooClient', MockKazooClient)
|
||||
def setUp(self):
|
||||
self.zk = ZooKeeper({'hosts': ['localhost:2181'], 'scope': 'test',
|
||||
'name': 'foo', 'ttl': 30, 'retry_timeout': 10, 'loop_wait': 10})
|
||||
'name': 'foo', 'ttl': 30, 'retry_timeout': 10, 'loop_wait': 10,
|
||||
'set_acls': {'CN=principal2': ['ALL']}})
|
||||
|
||||
def test_session_listener(self):
|
||||
self.zk.session_listener(KazooState.SUSPENDED)
|
||||
@@ -173,6 +174,8 @@ class TestZooKeeper(unittest.TestCase):
|
||||
self.assertRaises(ZooKeeperError, self.zk.get_cluster)
|
||||
cluster = self.zk.get_cluster(True)
|
||||
self.assertIsInstance(cluster.leader, Leader)
|
||||
self.zk.status_watcher(None)
|
||||
self.zk.get_cluster()
|
||||
self.zk.touch_member({'foo': 'foo'})
|
||||
self.zk._name = 'bar'
|
||||
self.zk.status_watcher(None)
|
||||
@@ -213,6 +216,7 @@ class TestZooKeeper(unittest.TestCase):
|
||||
self.zk.touch_member({'retry': 'retry'})
|
||||
self.zk._fetch_cluster = True
|
||||
self.zk.get_cluster()
|
||||
self.zk.touch_member({'retry': 'retry'})
|
||||
self.zk.touch_member({'conn_url': 'postgres://repuser:rep-pass@localhost:5434/postgres',
|
||||
'api_url': 'http://127.0.0.1:8009/patroni'})
|
||||
|
||||
@@ -224,6 +228,7 @@ class TestZooKeeper(unittest.TestCase):
|
||||
def test_update_leader(self):
|
||||
self.assertTrue(self.zk.update_leader(12345))
|
||||
|
||||
@patch.object(Cluster, 'min_version', PropertyMock(return_value=(2, 0)))
|
||||
def test_write_leader_optime(self):
|
||||
self.zk.last_lsn = '0'
|
||||
self.zk.write_leader_optime('1')
|
||||
@@ -232,6 +237,7 @@ class TestZooKeeper(unittest.TestCase):
|
||||
with patch.object(MockKazooClient, 'set_async', Mock()):
|
||||
self.zk.write_leader_optime('2')
|
||||
self.zk._base_path = self.zk._base_path.replace('test', 'bla')
|
||||
self.zk.get_cluster()
|
||||
self.zk.write_leader_optime('3')
|
||||
|
||||
def test_delete_cluster(self):
|
||||
@@ -239,7 +245,7 @@ class TestZooKeeper(unittest.TestCase):
|
||||
|
||||
def test_watch(self):
|
||||
self.zk.watch(None, 0)
|
||||
self.zk.event.isSet = Mock(return_value=True)
|
||||
self.zk.event.is_set = Mock(return_value=True)
|
||||
self.zk._fetch_status = False
|
||||
self.zk.watch(None, 0)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user