diff --git a/.travis.yml b/.travis.yml index 696295f3..23c69358 100644 --- a/.travis.yml +++ b/.travis.yml @@ -6,9 +6,10 @@ python: install: - if [[ $TRAVIS_PYTHON_VERSION == 2* ]]; then pip install -r requirements-py2.txt --use-mirrors; fi - if [[ $TRAVIS_PYTHON_VERSION == 3* ]]; then pip install -r requirements-py3.txt; fi - - pip install coveralls + - pip install coveralls codacy-coverage script: - python setup.py test - python setup.py flake8 after_success: - coveralls + - python-codacy-coverage -r coverage.xml diff --git a/Dockerfile b/Dockerfile index 84b9cdf6..12f6157c 100644 --- a/Dockerfile +++ b/Dockerfile @@ -12,17 +12,22 @@ RUN curl https://www.postgresql.org/media/keys/ACCC4CF8.asc | apt-key add - RUN apt-get update -y RUN apt-get upgrade -y -ENV PGVERSION 9.4 -RUN apt-get install python python-yaml python-requests python-boto postgresql-${PGVERSION} python-dnspython python-kazoo python-pip -y -RUN apt-get install python-dev postgresql-server-dev-${PGVERSION} -y -RUN pip install python-etcd psycopg2 +ENV PGVERSION 9.5 +RUN apt-get install postgresql-${PGVERSION} postgresql-server-dev-${PGVERSION} -y +RUN apt-get install python python-dev python-pip -y +ADD requirements-py2.txt /requirements-py2.txt +RUN pip install -r /requirements-py2.txt ENV PATH /usr/lib/postgresql/${PGVERSION}/bin:$PATH ADD patroni.py /patroni.py +ADD patronictl.py /patronictl.py ADD patroni/ /patroni -ENV ETCDVERSION 2.0.13 +RUN ln -s /patroni.py /usr/local/bin/patroni +RUN ln -s /patronictl.py /usr/local/bin/patronictl + +ENV ETCDVERSION 2.2.5 RUN curl -L https://github.com/coreos/etcd/releases/download/v${ETCDVERSION}/etcd-v${ETCDVERSION}-linux-amd64.tar.gz | tar xz -C /bin --strip=1 --wildcards --no-anchored etcd etcdctl ### Setting up a simple script that will serve as an entrypoint diff --git a/LICENCE b/LICENSE similarity index 100% rename from LICENCE rename to LICENSE diff --git a/MAINTAINERS b/MAINTAINERS new file mode 100644 index 00000000..fbe0251f --- /dev/null +++ b/MAINTAINERS @@ -0,0 +1,3 @@ +Alexander Kukushkin +Feike Steenbergen +Oleksii Kliukin diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh index 9edb2120..9bf176bb 100755 --- a/docker/entrypoint.sh +++ b/docker/entrypoint.sh @@ -79,7 +79,12 @@ then ETCD_CLUSTER="127.0.0.1:4001" fi -cat > /patroni/postgres.yml <<__EOF__ +mkdir -p ~postgres/.config/patroni +cat > ~postgres/.config/patroni/patronictl.yaml <<__EOF__ +{dcs_api: 'etcd://${ETCD_CLUSTER}', namespace: /service/} +__EOF__ + +cat > /patroni/postgres.yaml <<__EOF__ ttl: &ttl 30 loop_wait: &loop_wait 10 @@ -119,14 +124,15 @@ postgresql: archive_command: 'true' max_wal_senders: 20 listen_addresses: 0.0.0.0 - checkpoint_segments: 64 + max_wal_size: 1GB + min_wal_size: 128MB wal_keep_segments: 64 archive_timeout: 1800s max_replication_slots: 20 hot_standby: "on" __EOF__ -cat /patroni/postgres.yml +cat /patroni/postgres.yaml if [ ! -z $CHEAT ] then @@ -135,5 +141,5 @@ then sleep 60 done else - exec python /patroni.py /patroni/postgres.yml + exec python /patroni.py /patroni/postgres.yaml fi diff --git a/extras/startup-scripts/README.md b/extras/startup-scripts/README.md index 244ee65e..7cc0d445 100644 --- a/extras/startup-scripts/README.md +++ b/extras/startup-scripts/README.md @@ -8,3 +8,6 @@ Scripts supplied: ### patroni.upstart.conf Upstart job for Ubuntu 12.04 or 14.04. Requires Upstart > 1.4. Intended for systems where Patroni has been installed on a base system, rather than in Docker. + +### patroni.service +Systemd service file, to be copied to /etc/systemd/system/patroni.service, tested on Centos 7.1 with Patroni installed from pip. diff --git a/extras/startup-scripts/patroni.service b/extras/startup-scripts/patroni.service new file mode 100644 index 00000000..fdd7558b --- /dev/null +++ b/extras/startup-scripts/patroni.service @@ -0,0 +1,28 @@ +# This is an example systemd config file for Patroni +# You can copy it to "/etc/systemd/system/patroni.service", + +[Unit] +Description=Runners to orchestrate a high-availability PostgreSQL +After=syslog.target network.target + +[Service] +Type=simple + +User=postgres +Group=postgres + +# Where to send early-startup messages from the server +# This is normally controlled by the global default set by systemd +# StandardOutput=syslog + +ExecStart=/bin/patroni /etc/patroni.yml + +# Give a reasonable amount of time for the server to start up/shut down +TimeoutSec=10 + +# Do not restart the service if it crashes, we want to manually inspect database on failure +Restart=no + +[Install] +WantedBy=multi-user.target + diff --git a/patroni/__init__.py b/patroni/__init__.py index 1ded41aa..36c8feac 100644 --- a/patroni/__init__.py +++ b/patroni/__init__.py @@ -15,7 +15,7 @@ from .version import __version__ logger = logging.getLogger(__name__) -class Patroni: +class Patroni(object): def __init__(self, config): self.nap_time = config['loop_wait'] @@ -31,6 +31,10 @@ class Patroni: def nofailover(self): return self.tags.get('nofailover', False) + @property + def replicatefrom(self): + return self.tags.get('replicatefrom') + @staticmethod def get_dcs(name, config): if 'etcd' in config: @@ -64,7 +68,7 @@ def main(): setup_signal_handlers() if len(sys.argv) < 2 or not os.path.isfile(sys.argv[1]): - print('Usage: {} config.yml'.format(sys.argv[0])) + print('Usage: {0} config.yml'.format(sys.argv[0])) return with open(sys.argv[1], 'r') as f: diff --git a/patroni/api.py b/patroni/api.py index 2318fbf0..ba18b333 100644 --- a/patroni/api.py +++ b/patroni/api.py @@ -5,6 +5,9 @@ import logging import psycopg2 import socket import time +import dateutil +import datetime +import pytz from patroni.exceptions import PostgresConnectionException from patroni.utils import Retry, RetryFailedError @@ -101,7 +104,7 @@ class RestApiHandler(BaseHTTPRequestHandler): @check_auth def do_POST_restart(self): - status_code = 503 + status_code = 500 data = b'restart failed' try: status, msg = self.server.patroni.ha.restart() @@ -172,17 +175,37 @@ class RestApiHandler(BaseHTTPRequestHandler): def do_POST_failover(self): content_length = int(self.headers.get('content-length', 0)) request = json.loads(self.rfile.read(content_length).decode('utf-8')) - leader = request.get('leader', None) - member = request.get('member', None) + leader = request.get('leader') + member = request.get('member') cluster = self.server.patroni.ha.dcs.get_cluster() - status_code = 503 - data = self.is_failover_possible(cluster, leader, member) - if not data: - if not self.server.patroni.dcs.manual_failover(leader, member): - data = b'failed to write failover key into DCS' - else: - self.server.patroni.dcs.event.set() - status_code, data = self.poll_failover_result(cluster.leader and cluster.leader.name, member) + status_code = 500 + + data = b'' + if request.get('scheduled_at'): + try: + scheduled_at = dateutil.parser.parse(request['scheduled_at']) + if scheduled_at.tzinfo is None: + data = b'Timezone information is mandatory for scheduled_at' + status_code = 400 + elif scheduled_at < datetime.datetime.now(pytz.utc): + data = b'Cannot schedule failover in the past' + status_code = 422 + elif self.server.patroni.dcs.manual_failover(leader, member, scheduled_at): + data = b'Failover scheduled' + status_code = 200 + except (ValueError, TypeError): + logger.exception('Invalid scheduled failover time: {}'.format(request['scheduled_at'])) + data = b'Unable to parse scheduled timestamp. It should be in an unambiguous format, e.g. ISO 8601' + status_code = 422 + else: + data = self.is_failover_possible(cluster, leader, member) + if not data: + if not self.server.patroni.dcs.manual_failover(leader, member): + data = b'failed to write failover key into DCS' + status_code = 503 + else: + self.server.patroni.dcs.event.set() + status_code, data = self.poll_failover_result(cluster.leader and cluster.leader.name, member) self.send_response(status_code) self.send_header('Content-Type', 'text/html') @@ -252,8 +275,8 @@ class RestApiHandler(BaseHTTPRequestHandler): def get_tags(self): return {'tags': self.server.patroni.tags} - def log_message(self, format, *args): - logger.debug("API thread: " + format % args) + def log_message(self, fmt, *args): + logger.debug("API thread: %s - - [%s] %s", self.client_address[0], self.log_date_time_string(), fmt % args) class RestApiServer(ThreadingMixIn, HTTPServer, Thread): @@ -270,12 +293,12 @@ class RestApiServer(ThreadingMixIn, HTTPServer, Thread): # wrap socket with ssl if 'certfile' is defined in a config.yaml # Sometime it's also needed to pass reference to a 'keyfile'. options = {option: config[option] for option in ['certfile', 'keyfile'] if option in config} - if options.get('certfile', None): + if options.get('certfile'): import ssl self.socket = ssl.wrap_socket(self.socket, server_side=True, **options) protocol = 'https' - self.connection_string = '{}://{}/patroni'.format(protocol, config.get('connect_address', config['listen'])) + self.connection_string = '{0}://{1}/patroni'.format(protocol, config.get('connect_address', config['listen'])) self.patroni = patroni self.daemon = True diff --git a/patroni/async_executor.py b/patroni/async_executor.py index fc222202..e009ab19 100644 --- a/patroni/async_executor.py +++ b/patroni/async_executor.py @@ -4,10 +4,9 @@ from threading import Lock, Thread logger = logging.getLogger(__name__) -class AsyncExecutor: +class AsyncExecutor(object): def __init__(self): - Lock.__init__(self) self._busy = False self._thread_lock = Lock() self._scheduled_action = None @@ -51,5 +50,5 @@ class AsyncExecutor: def __enter__(self): self._thread_lock.acquire() - def __exit__(self, type, value, traceback): + def __exit__(self, *args): self._thread_lock.release() diff --git a/patroni/ctl.py b/patroni/ctl.py index 3d457c9f..c4cc9fe0 100644 --- a/patroni/ctl.py +++ b/patroni/ctl.py @@ -14,6 +14,8 @@ import datetime from prettytable import PrettyTable from six.moves.urllib_parse import urlparse import logging +import dateutil +import tzlocal from .etcd import Etcd from .exceptions import PatroniCtlException @@ -56,12 +58,12 @@ def parse_dcs(dcs): def load_config(path, dcs): - logging.debug('Loading configuration from file {}'.format(path)) + logging.debug('Loading configuration from file %s', path) config = dict() try: with open(path, 'rb') as fd: config = yaml.safe_load(fd) - except: + except (IOError, yaml.YAMLError): logging.exception('Could not load configuration file') if dcs: @@ -74,15 +76,14 @@ def load_config(path, dcs): def store_config(config, path): dir_path = os.path.dirname(path) - if dir_path: - if not os.path.isdir(dir_path): - os.makedirs(dir_path) + if dir_path and not os.path.isdir(dir_path): + os.makedirs(dir_path) with open(path, 'w') as fd: yaml.dump(config, fd) option_config_file = click.option('--config-file', '-c', help='Configuration file', default=CONFIG_FILE_PATH) -option_format = click.option('--format', '-f', help='Output format (pretty, json)', default='pretty') +option_format = click.option('--format', '-f', 'fmt', help='Output format (pretty, json)', default='pretty') option_dcs = click.option('--dcs', '-d', help='Use this DCS', envvar='DCS') option_watchrefresh = click.option('-w', '--watch', type=float, help='Auto update the screen every X seconds') option_watch = click.option('-W', is_flag=True, help='Auto update the screen every 2 seconds') @@ -102,20 +103,22 @@ def get_dcs(config, scope): scheme, hostname, port = map(config.get('dcs', {}).get, ('scheme', 'hostname', 'port')) if scheme == 'etcd': - return Etcd(name=scope, config={'scope': scope, 'host': '{}:{}'.format(hostname, port)}) + return Etcd(name=scope, config={'scope': scope, 'host': '{0}:{1}'.format(hostname, port)}) raise PatroniCtlException('Can not find suitable configuration of distributed configuration store') -def post_patroni(member, endpoint, content, headers={'Content-Type': 'application/json'}): +def post_patroni(member, endpoint, content, headers=None): url = urlparse(member.api_url) logging.debug(url) - return requests.post('{}://{}/{}'.format(url.scheme, url.netloc, endpoint), headers=headers, + return requests.post('{0}://{1}/{2}'.format(url.scheme, url.netloc, endpoint), + headers=headers or {'Content-Type': 'application/json'}, data=json.dumps(content), timeout=60) -def print_output(columns, rows=[], alignment=None, format='pretty', header=True, delimiter='\t'): - if format == 'pretty': +def print_output(columns, rows=None, alignment=None, fmt='pretty', header=True, delimiter='\t'): + rows = rows or [] + if fmt == 'pretty': t = PrettyTable(columns) for k, v in (alignment or {}).items(): t.align[k] = v @@ -124,18 +127,18 @@ def print_output(columns, rows=[], alignment=None, format='pretty', header=True, click.echo(t) return - if format == 'json': + if fmt == 'json': elements = list() for r in rows: elements.append(dict(zip(columns, r))) click.echo(json.dumps(elements)) - if format == 'tsv': + if fmt == 'tsv': if columns is not None and header: click.echo(delimiter.join(columns) + '\n') - for r in rows or []: + for r in rows: c = [str(c) for c in r] click.echo(delimiter.join(c)) @@ -168,8 +171,8 @@ def watching(w, watch, max_count=None, clear=True): yield 0 -def build_connect_parameters(conn_url, connect_parameters={}): - params = connect_parameters.copy() +def build_connect_parameters(conn_url, connect_parameters=None): + params = (connect_parameters or {}).copy() parsed = parseurl(conn_url) params['host'] = parsed['host'] params['port'] = parsed['port'] @@ -200,12 +203,12 @@ def get_any_member(cluster, role='master', member=None): return None -def get_cursor(cluster, role='master', member=None, connect_parameters={}): +def get_cursor(cluster, role='master', member=None, connect_parameters=None): member = get_any_member(cluster=cluster, role=role, member=member) if member is None: return None - params = build_connect_parameters(member.conn_url, connect_parameters=connect_parameters) + params = build_connect_parameters(member.conn_url, connect_parameters) conn = psycopg2.connect(**params) conn.autocommit = True @@ -243,15 +246,17 @@ def dsn(cluster_name, config_file, dcs, role, member): raise PatroniCtlException('Can not find a suitable member') params = build_connect_parameters(m.conn_url) - click.echo('host={} port={}'.format(params['host'], params['port'])) + click.echo('host={host} port={port}'.format(**params)) @ctl.command('query', help='Query a Patroni PostgreSQL member') @click.argument('cluster_name') @option_config_file @option_format -@click.option('--format', help='Output format (pretty, json)', default='tsv') -@click.option('--file', '-f', help='Execute the SQL commands from this file', type=click.File('rb')) +@click.option('--format', 'fmt', help='Output format (pretty, json)', default='tsv') +@click.option('--file', '-f', 'p_file', help='Execute the SQL commands from this file', type=click.File('rb')) +@click.option('--password', help='force password prompt', is_flag=True) +@click.option('-U', '--username', help='database user name', type=str) @option_dcs @option_watch @option_watchrefresh @@ -260,6 +265,7 @@ def dsn(cluster_name, config_file, dcs, role, member): @click.option('--member', '-m', help='Query a specific member', type=str) @click.option('--delimiter', help='The column delimiter', default='\t') @click.option('--command', '-c', help='The SQL commands to execute') +@click.option('-d', '--dbname', help='database name to connect to', type=str) def query( cluster_name, config_file, @@ -270,42 +276,57 @@ def query( watch, delimiter, command, - file, - format='tsv', + p_file, + password, + username, + dbname, + fmt='tsv', ): if role is not None and member is not None: raise PatroniCtlException('--role and --member are mutually exclusive options') if member is None and role is None: role = 'master' - if file is not None and command is not None: + if p_file is not None and command is not None: raise PatroniCtlException('--file and --command are mutually exclusive options') - if file is not None: - command = file.read() + if p_file is None and command is None: + raise PatroniCtlException('You need to specify either --command or --file') + + connect_parameters = dict() + if username: + connect_parameters['user'] = username + if password: + connect_parameters['password'] = click.prompt('Password', hide_input=True, type=str) + if dbname: + connect_parameters['database'] = dbname + + if p_file is not None: + command = p_file.read() config, dcs, cluster = ctl_load_config(cluster_name, config_file, dcs) cursor = None for _ in watching(w, watch, clear=False): - output, cursor = query_member(cluster=cluster, cursor=cursor, member=member, role=role, command=command) - print_output(None, output, format=format, delimiter=delimiter) + output, cursor = query_member(cluster=cluster, cursor=cursor, member=member, role=role, command=command, + connect_parameters=connect_parameters) + print_output(None, output, fmt=fmt, delimiter=delimiter) if cursor is None: cluster = dcs.get_cluster() -def query_member(cluster, cursor, member, role, command): +def query_member(cluster, cursor, member, role, command, connect_parameters=None): try: if cursor is None: - cursor = get_cursor(cluster, role=role, member=member) + cursor = get_cursor(cluster, role=role, member=member, connect_parameters=connect_parameters) if cursor is None: if role is None: - message = 'No connection to member {} is available'.format(member) + message = 'No connection to member {0} is available'.format(member) else: - message = 'No connection to role={} is available'.format(role) + message = 'No connection to role={0} is available'.format(role) logging.debug(message) return [[timestamp(0), message]], None @@ -324,7 +345,7 @@ def query_member(cluster, cursor, member, role, command): cursor.connection.close() message = oe.pgcode or oe.pgerror or str(oe) message = message.replace('\n', ' ') - return [[timestamp(0), 'ERROR, SQLSTATE: {}'.format(message)]], None + return [[timestamp(0), 'ERROR, SQLSTATE: {0}'.format(message)]], None @ctl.command('remove', help='Remove cluster from DCS') @@ -332,13 +353,13 @@ def query_member(cluster, cursor, member, role, command): @option_config_file @option_format @option_dcs -def remove(config_file, cluster_name, format, dcs): +def remove(config_file, cluster_name, fmt, dcs): config, dcs, cluster = ctl_load_config(cluster_name, config_file, dcs) if not isinstance(dcs, Etcd): - raise PatroniCtlException('We have not implemented this for DCS of type {}'.format(type(dcs))) + raise PatroniCtlException('We have not implemented this for DCS of type {0}'.format(type(dcs))) - output_members(cluster, format=format) + output_members(cluster, fmt=fmt) confirm = click.prompt('Please confirm the cluster name to remove', type=str) if confirm != cluster_name: @@ -346,17 +367,17 @@ def remove(config_file, cluster_name, format, dcs): message = 'Yes I am aware' confirm = \ - click.prompt('You are about to remove all information in DCS for {}, please type: "{}"'.format(cluster_name, + click.prompt('You are about to remove all information in DCS for {0}, please type: "{1}"'.format(cluster_name, message), type=str) if message != confirm: - raise PatroniCtlException('You did not exactly type "{}"'.format(message)) + raise PatroniCtlException('You did not exactly type "{0}"'.format(message)) if cluster.leader: confirm = click.prompt('This cluster currently is healthy. Please specify the master name to continue') if confirm != cluster.leader.name: raise PatroniCtlException('You did not specify the current master of the cluster') - dcs.client.delete(dcs._base_path, recursive=True) + dcs.client.delete(dcs.client_path(''), recursive=True) def wait_for_leader(dcs, timeout=30): @@ -378,25 +399,25 @@ def empty_post_to_members(cluster, member_names, force, endpoint): for m in cluster.members: candidates[m.name] = m - if len(member_names) == 0: - member_names = [click.prompt('Which member do you want to {} [{}]?'.format(endpoint, + if not member_names: + member_names = [click.prompt('Which member do you want to {0} [{1}]?'.format(endpoint, ', '.join(candidates.keys())), type=str, default='')] for mn in member_names: if mn not in candidates.keys(): - raise PatroniCtlException('{} is not a member of cluster'.format(mn)) + raise PatroniCtlException('{0} is not a member of cluster'.format(mn)) if not force: - confirm = click.confirm('Are you sure you want to {} members {}?'.format(endpoint, ', '.join(member_names))) + confirm = click.confirm('Are you sure you want to {0} members {1}?'.format(endpoint, ', '.join(member_names))) if not confirm: - raise PatroniCtlException('Aborted {}'.format(endpoint)) + raise PatroniCtlException('Aborted {0}'.format(endpoint)) for mn in member_names: r = post_patroni(candidates[mn], endpoint, '') if r.status_code != 200: - click.echo('{} failed for member {}, status code={}, ({})'.format(endpoint, mn, r.status_code, r.text)) + click.echo('{0} failed for member {1}, status code={2}, ({3})'.format(endpoint, mn, r.status_code, r.text)) else: - click.echo('Succesful {} on member {}'.format(endpoint, mn)) + click.echo('Succesful {0} on member {1}'.format(endpoint, mn)) def ctl_load_config(cluster_name, config_file, dcs): @@ -412,21 +433,21 @@ def ctl_load_config(cluster_name, config_file, dcs): @click.argument('member_names', nargs=-1) @click.option('--role', '-r', help='Restart only members with this role', default='any', type=click.Choice(['master', 'replica', 'any'])) -@click.option('--any', help='Restart a single member only', is_flag=True) +@click.option('--any', 'p_any', help='Restart a single member only', is_flag=True) @option_config_file @option_force @option_dcs -def restart(cluster_name, member_names, config_file, dcs, force, role, any): +def restart(cluster_name, member_names, config_file, dcs, force, role, p_any): config, dcs, cluster = ctl_load_config(cluster_name, config_file, dcs) role_names = [m.name for m in get_all_members(cluster=cluster, role=role)] - if len(member_names) > 0: + if member_names: member_names = list(set(member_names) & set(role_names)) else: member_names = role_names - if any: + if p_any: random.shuffle(member_names) member_names = member_names[:1] @@ -449,10 +470,12 @@ def reinit(cluster_name, member_names, config_file, dcs, force): @click.argument('cluster_name') @click.option('--master', help='The name of the current master', default=None) @click.option('--candidate', help='The name of the candidate', default=None) +@click.option('--scheduled', help='Timestamp of a scheduled failover in unambiguous format (e.g. ISO 8601)', + default=None) @click.option('--force', is_flag=True) @option_config_file @option_dcs -def failover(config_file, cluster_name, master, candidate, force, dcs): +def failover(config_file, cluster_name, master, candidate, force, dcs, scheduled): """ We want to trigger a failover for the specified cluster name. @@ -472,13 +495,13 @@ def failover(config_file, cluster_name, master, candidate, force, dcs): master = click.prompt('Master', type=str, default=cluster.leader.member.name) if cluster.leader.member.name != master: - raise PatroniCtlException('Member {} is not the leader of cluster {}'.format(master, cluster_name)) + raise PatroniCtlException('Member {0} is not the leader of cluster {1}'.format(master, cluster_name)) candidate_names = [str(m.name) for m in cluster.members if m.name != master] # We sort the names for consistent output to the client candidate_names.sort() - if len(candidate_names) == 0: + if not candidate_names: raise PatroniCtlException('No candidates found to failover to') if candidate is None and not force: @@ -488,7 +511,26 @@ def failover(config_file, cluster_name, master, candidate, force, dcs): raise PatroniCtlException('Failover target and source are the same.') if candidate and candidate not in candidate_names: - raise PatroniCtlException('Member {} does not exist in cluster {}'.format(candidate, cluster_name)) + raise PatroniCtlException('Member {0} does not exist in cluster {1}'.format(candidate, cluster_name)) + + if scheduled is None and not force: + scheduled = click.prompt('When should the failover take place (e.g. 2015-10-01T14:30) ', type=str, + default='now') + + if (scheduled or 'now') == 'now': + scheduled_at = None + else: + try: + scheduled_at = dateutil.parser.parse(scheduled) + if scheduled_at.tzinfo is None: + scheduled_at = tzlocal.get_localzone().localize(scheduled_at) + except (ValueError, TypeError): + message = 'Unable to parse scheduled timestamp ({}). It should be in an unambiguous format (e.g. ISO 8601)' + raise PatroniCtlException(message.format(scheduled)) + scheduled_at = scheduled_at.isoformat() + + failover_value = {'leader': master, 'member': candidate, 'scheduled_at': scheduled_at} + logging.debug(failover_value) # By now we have established that the leader exists and the candidate exists click.echo('Current cluster topology') @@ -496,44 +538,33 @@ def failover(config_file, cluster_name, master, candidate, force, dcs): if not force: a = \ - click.confirm('Are you sure you want to failover cluster {}, demoting current master {}?'.format( + click.confirm('Are you sure you want to failover cluster {0}, demoting current master {1}?'.format( cluster_name, master)) if not a: raise PatroniCtlException('Aborting failover') - failover_value = '{}:{}'.format(master, candidate or '') - - t_started = time.time() r = None try: - r = post_patroni(cluster.leader.member, 'failover', {'leader': master, 'member': candidate or ''}) + r = post_patroni(cluster.leader.member, 'failover', failover_value) if r.status_code == 200: logging.debug(r) - logging.debug(r.text) cluster = dcs.get_cluster() - click.echo(timestamp() + ' Failing over to new leader: {}'.format(cluster.leader.member.name)) + logging.debug(cluster) + click.echo('{0} {1}'.format(timestamp(), r.text)) else: - click.echo('Failover failed, details: {}, {}'.format(r.status_code, r.text)) + click.echo('Failover failed, details: {0}, {1}'.format(r.status_code, r.text)) return except: logging.exception(r) logging.warning('Failing over to DCS') click.echo(timestamp() + ' Could not failover using Patroni api, falling back to DCS') - dcs.set_failover_value(failover_value) - click.echo(timestamp() + ' Initialized failover from master {}'.format(master)) - # The failover process should within a minute update the failover key, we will keep watching it until it changes - # or we timeout - cluster = wait_for_leader(dcs, timeout=60) - if cluster.leader.member.name == master: - click.echo('Failover failed, master did not change after {:0.1f} seconds'.format(time.time() - t_started)) - return + click.echo(timestamp() + ' Initializing failover from master {0}'.format(master)) + dcs.manual_failover(leader=master, member=candidate, scheduled_at=failover_value) - click.echo(timestamp() + ' Failover completed in {:0.1f} seconds, new leader is {}'.format(time.time() - t_started, - str(cluster.leader.member.name))) output_members(cluster, name=cluster_name) -def output_members(cluster, name=None, format='pretty'): +def output_members(cluster, name=None, fmt='pretty'): rows = [] logging.debug(cluster) leader_name = None @@ -553,10 +584,9 @@ def output_members(cluster, name=None, format='pretty'): host = build_connect_parameters(m.conn_url)['host'] - xlog_location = m.data.get('xlog_location') - if xlog_location is None or (xlog_location_cluster < xlog_location): - lag = '' - else: + xlog_location = m.data.get('xlog_location') or 0 + lag = '' + if (xlog_location_cluster >= xlog_location): lag = round((xlog_location_cluster - xlog_location)/1024/1024) rows.append([ @@ -578,7 +608,7 @@ def output_members(cluster, name=None, format='pretty'): ] alignment = {'Cluster': 'l', 'Member': 'l', 'Host': 'l', 'Lag in MB': 'r'} - print_output(columns, rows, alignment, format) + print_output(columns, rows, alignment, fmt) @ctl.command('list', help='List the Patroni members for a given Patroni') @@ -588,8 +618,8 @@ def output_members(cluster, name=None, format='pretty'): @option_watch @option_watchrefresh @option_dcs -def members(config_file, cluster_names, format, watch, w, dcs): - if len(cluster_names) == 0: +def members(config_file, cluster_names, fmt, watch, w, dcs): + if not cluster_names: logging.warning('Listing members: No cluster names were provided') return @@ -598,7 +628,7 @@ def members(config_file, cluster_names, format, watch, w, dcs): dcs = get_dcs(config, cn) for _ in watching(w, watch): - output_members(dcs.get_cluster(), name=cn, format=format) + output_members(dcs.get_cluster(), name=cn, fmt=fmt) def timestamp(precision=6): diff --git a/patroni/dcs.py b/patroni/dcs.py index f640787c..4c2dddad 100644 --- a/patroni/dcs.py +++ b/patroni/dcs.py @@ -1,5 +1,6 @@ import abc import json +import dateutil from collections import namedtuple from patroni.exceptions import DCSError @@ -51,22 +52,26 @@ class Member(namedtuple('Member', 'index,name,session,data')): else: try: data = json.loads(data) - except: + except (TypeError, ValueError): data = {} return Member(index, name, session, data) @property def conn_url(self): - return self.data.get('conn_url', None) + return self.data.get('conn_url') @property def api_url(self): - return self.data.get('api_url', None) + return self.data.get('api_url') @property def nofailover(self): return self.data.get('tags', {}).get('nofailover', False) + @property + def replicatefrom(self): + return self.data.get('tags', {}).get('replicatefrom') + class Leader(namedtuple('Leader', 'index,session,member')): @@ -85,12 +90,44 @@ class Leader(namedtuple('Leader', 'index,session,member')): return self.member.conn_url -class Failover(namedtuple('Failover', 'index,leader,member')): +class Failover(namedtuple('Failover', 'index,leader,member,scheduled_at')): + """ + >>> 'Failover' in str(Failover.from_node(1, '{"leader": "cluster_leader"}')) + True + >>> 'Failover' in str(Failover.from_node(1, '{"leader": "cluster_leader", "member": "cluster:member"}')) + True + >>> Failover.from_node(1, 'null') is None + True + >>> n = '{"leader": "cluster_leader", "member": "cluster:member", "scheduled_at": "2016-01-14T10:09:57.1394Z"}' + >>> 'tzinfo=' in str(Failover.from_node(1, n)) + True + >>> Failover.from_node(1, None) is None + True + >>> Failover.from_node(1, '{}') is None + True + >>> 'abc' in Failover.from_node(1, 'abc:def') + True + """ @staticmethod def from_node(index, value): - t = [a.strip() for a in value.split(':')] + [''] - return Failover(index, t[0], t[1]) if t[0] or t[1] else None + if not value: + return None + + try: + data = json.loads(value) + if not data: + return None + except ValueError: + t = [a.strip() for a in value.split(':')] + leader = t[0] + candidate = t[1] if len(t) > 1 else None + return Failover(index, leader, candidate, None) if leader or candidate else None + + if data.get('scheduled_at'): + data['scheduled_at'] = dateutil.parser.parse(data['scheduled_at']) + + return Failover(index, data.get('leader'), data.get('member'), data.get('scheduled_at')) class Cluster(namedtuple('Cluster', 'initialize,leader,last_leader_operation,members,failover')): @@ -107,8 +144,11 @@ class Cluster(namedtuple('Cluster', 'initialize,leader,last_leader_operation,mem def is_unlocked(self): return not (self.leader and self.leader.name) + def has_member(self, member_name): + return any(m for m in self.members if m.name == member_name) -class AbstractDCS: + +class AbstractDCS(object): __metaclass__ = abc.ABCMeta @@ -126,7 +166,7 @@ class AbstractDCS: i.e.: `zookeeper` for zookeeper, `etcd` for etcd, etc... """ self._name = name - self._namespace = '/{}'.format(config.get('namespace', '/service/').strip('/')) + self._namespace = '/{0}'.format(config.get('namespace', '/service/').strip('/')) self._base_path = '/'.join([self._namespace, config['scope']]) self._cluster = None @@ -216,8 +256,18 @@ class AbstractDCS: def set_failover_value(self, value, index=None): """Create or update `/failover` key""" - def manual_failover(self, leader, member, index=None): - return self.set_failover_value(leader + (':' + member if member else ''), index) + def manual_failover(self, leader, member, scheduled_at=None, index=None): + failover_value = dict() + if leader: + failover_value['leader'] = leader + + if member: + failover_value['member'] = member + + if scheduled_at: + failover_value['scheduled_at'] = scheduled_at.isoformat() + + return self.set_failover_value(json.dumps(failover_value), index) def current_leader(self): try: diff --git a/patroni/etcd.py b/patroni/etcd.py index 5f4fa0fb..79842ca4 100644 --- a/patroni/etcd.py +++ b/patroni/etcd.py @@ -51,7 +51,8 @@ class Client(etcd.Client): def api_execute(self, path, method, **kwargs): # Update machines_cache if previous attempt of update has failed - self._update_machines_cache and self._load_machines_cache() + if self._update_machines_cache: + self._load_machines_cache() try: return super(Client, self).api_execute(path, method, **kwargs) except etcd.EtcdConnectionFailed: @@ -73,7 +74,7 @@ class Client(etcd.Client): except urllib3.exceptions.TimeoutError: raise except Exception as e: - raise etcd.EtcdException('Unable to decode server response: %s' % e) + raise etcd.EtcdException('Unable to decode server response: {0}'.format(e)) return super(Client, self)._result_from_response(response) def _get_machines_cache_from_srv(self, discovery_srv): @@ -83,7 +84,7 @@ class Client(etcd.Client): ret = [] for host, port in self.get_srv_record(discovery_srv): - url = '{}://{}:{}/members'.format(self._protocol, host, port) + url = '{0}://{1}:{2}/members'.format(self._protocol, host, port) try: response = requests.get(url, timeout=5) if response.ok: @@ -101,10 +102,10 @@ class Client(etcd.Client): host, port = addr.split(':') try: for r in set(socket.getaddrinfo(host, port, socket.AF_INET, socket.SOCK_STREAM, socket.IPPROTO_TCP)): - ret.append('{}://{}:{}'.format(self._protocol, r[4][0], r[4][1])) + ret.append('{0}://{1}:{2}'.format(self._protocol, r[4][0], r[4][1])) except socket.error: logger.exception('Can not resolve %s', host) - return list(set(ret)) if ret else ['{}://{}:{}'.format(self._protocol, host, port)] + return list(set(ret)) if ret else ['{0}://{1}:{2}'.format(self._protocol, host, port)] def _load_machines_cache(self): """This method should fill up `_machines_cache` from scratch. @@ -132,7 +133,9 @@ class Client(etcd.Client): # After filling up initial list of machines_cache we should ask etcd-cluster about actual list self._base_uri = self._machines_cache.pop(0) self._machines_cache = self.machines - self._base_uri in self._machines_cache and self._machines_cache.remove(self._base_uri) + + if self._base_uri in self._machines_cache: + self._machines_cache.remove(self._base_uri) self._update_machines_cache = False @@ -140,7 +143,7 @@ class Client(etcd.Client): def catch_etcd_errors(func): def wrapper(*args, **kwargs): try: - return not func(*args, **kwargs) is None + return func(*args, **kwargs) is not None except (RetryFailedError, etcd.EtcdException): return False except: @@ -165,7 +168,8 @@ class Etcd(AbstractDCS): def retry(self, *args, **kwargs): return self._retry.copy()(*args, **kwargs) - def get_etcd_client(self, config): + @staticmethod + def get_etcd_client(config): client = None while not client: try: @@ -185,25 +189,25 @@ class Etcd(AbstractDCS): nodes = {os.path.relpath(node.key, result.key): node for node in result.leaves} # get initialize flag - initialize = nodes.get(self._INITIALIZE, None) + initialize = nodes.get(self._INITIALIZE) initialize = initialize and initialize.value # get last leader operation - last_leader_operation = nodes.get(self._LEADER_OPTIME, None) + last_leader_operation = nodes.get(self._LEADER_OPTIME) last_leader_operation = 0 if last_leader_operation is None else int(last_leader_operation.value) # get list of members members = [self.member(n) for k, n in nodes.items() if k.startswith(self._MEMBERS) and k.count('/') == 1] # get leader - leader = nodes.get(self._LEADER, None) + leader = nodes.get(self._LEADER) if leader: member = Member(-1, leader.value, None, {}) member = ([m for m in members if m.name == leader.value] or [member])[0] leader = Leader(leader.modifiedIndex, leader.ttl, member) # failover key - failover = nodes.get(self._FAILOVER, None) + failover = nodes.get(self._FAILOVER) if failover: failover = Failover.from_node(failover.modifiedIndex, failover.value) diff --git a/patroni/exceptions.py b/patroni/exceptions.py index 43f54e7f..d07e6426 100644 --- a/patroni/exceptions.py +++ b/patroni/exceptions.py @@ -1,3 +1,6 @@ +from click import ClickException + + class PatroniException(Exception): """Parent class for all kind of exceptions related to selected distributed configuration store""" @@ -13,7 +16,7 @@ class PatroniException(Exception): return repr(self.value) -class PatroniCtlException(Exception): +class PatroniCtlException(ClickException): pass diff --git a/patroni/ha.py b/patroni/ha.py index 3a8d46d7..8f5d772d 100644 --- a/patroni/ha.py +++ b/patroni/ha.py @@ -3,15 +3,18 @@ import logging import psycopg2 import requests import sys +import datetime +import pytz +from multiprocessing.pool import ThreadPool from patroni.async_executor import AsyncExecutor from patroni.exceptions import DCSError, PostgresConnectionException -from multiprocessing.pool import ThreadPool +from patroni.utils import sleep logger = logging.getLogger(__name__) -class Ha: +class Ha(object): def __init__(self, patroni): self.patroni = patroni @@ -19,12 +22,13 @@ class Ha: self.dcs = patroni.dcs self.cluster = None self.old_cluster = None + self.recovering = False self._async_executor = AsyncExecutor() def load_cluster_from_dcs(self): cluster = self.dcs.get_cluster() - # We want to keep the state of cluster when it was healhy + # We want to keep the state of cluster when it was healthy if not cluster.is_unlocked() or not self.old_cluster: self.old_cluster = cluster self.cluster = cluster @@ -61,18 +65,18 @@ class Ha: pass self.dcs.touch_member(json.dumps(data, separators=(',', ':'))) - def copy_backup_from_leader(self, leader): - if self.state_handler.bootstrap(leader): - logger.info('bootstrapped from leader') + def clone(self, leader): + if self.state_handler.bootstrap(cluster_initialized=True, current_leader=leader): + logger.info('bootstrapped from leader' if leader else 'bootstrapped without leader') else: self.state_handler.stop('immediate') self.state_handler.remove_data_directory() - logger.error('failed to bootstrap from leader') + logger.error('failed to bootstrap from leader' if leader else 'failed to bootstrap (without leader)') def bootstrap(self): if not self.cluster.is_unlocked(): # cluster already has leader self._async_executor.schedule('bootstrap from leader') - self._async_executor.run_async(self.copy_backup_from_leader, args=(self.cluster.leader, )) + self._async_executor.run_async(self.clone, args=(self.cluster.leader, )) return 'trying to bootstrap from leader' elif not self.cluster.initialize and not self.patroni.nofailover: # no initialize key if self.dcs.initialize(create_new=True): # race for initialization @@ -87,45 +91,48 @@ class Ha: self.state_handler.move_data_directory() raise self.dcs.take_leader() + self.load_cluster_from_dcs() return 'initialized a new cluster' else: return 'failed to acquire initialize lock' else: + if self.state_handler.can_create_replica_without_leader(): + self._async_executor.run_async(self.clone, args=(None, )) + return "trying to bootstrap without leader" return 'waiting for leader to bootstrap' def recover(self): - has_lock = self.has_lock() - # try to see if we are the former master that crashed. If so - we likely need to run pg_rewind # in order to join the former standby being promoted. pg_controldata = self.state_handler.controldata() - if not has_lock and pg_controldata and\ + if (self.state_handler.role == 'master') and pg_controldata and\ pg_controldata.get('Database cluster state', '') == 'in production': # crashed master self.state_handler.require_rewind() + self.recovering = True + return self.follow("started as readonly because i had the session lock", + "started as a secondary", + refresh=True, recovery=True) - # XXX: follow the leader calls stop, which might take quite some time. - # perhaps we should run sync asynchronously - # (we still need the exit code from follow_the_leader) - ret = self.state_handler.follow_the_leader(None if has_lock else self.cluster.leader, recovery=True) - if not ret: - if not has_lock: - return 'failed to start postgres' - self.dcs.delete_leader() - self.dcs.reset_cluster() - return 'removed leader key after trying and failing to start postgres' - if not has_lock: - return 'started as a secondary' - logger.info('started as readonly because i had the session lock') - self.load_cluster_from_dcs() + def follow(self, demote_reason, follow_reason, refresh=True, recovery=False): + if refresh: + self.load_cluster_from_dcs() - def follow_the_leader(self, demote_reason, follow_reason, refresh=True): - refresh and self.load_cluster_from_dcs() - ret = demote_reason if self.state_handler.is_leader() else follow_reason - leader = self.cluster.leader - leader = None if (leader and leader.name) == self.state_handler.name else leader - if not self.state_handler.check_recovery_conf(leader): + if not recovery and self.state_handler.is_leader() or recovery and self.state_handler.role == 'master': + ret = demote_reason + else: + ret = follow_reason + + # determine the node to follow. If replicatefrom tag is set, + # try to follow the node mentioned there, otherwise, follow the leader. + if self.patroni.replicatefrom: + node_to_follow = [m for m in self.cluster.members if m.name == self.patroni.replicatefrom] + node_to_follow = node_to_follow[0] if node_to_follow else self.cluster.leader + else: + node_to_follow = self.cluster.leader + node_to_follow = None if node_to_follow and node_to_follow.name == self.state_handler.name else node_to_follow + if not self.state_handler.check_recovery_conf(node_to_follow) or recovery: self._async_executor.schedule('changing primary_conninfo and restarting') - self._async_executor.run_async(self.state_handler.follow_the_leader, (leader, )) + self._async_executor.run_async(self.state_handler.follow, (node_to_follow, recovery)) return ret def enforce_master_role(self, message, promote_message): @@ -266,10 +273,34 @@ class Ha: self.dcs.delete_leader() self.touch_member() self.dcs.reset_cluster() - self.state_handler.follow_the_leader(None) + self.state_handler.follow(None) def process_manual_failover_from_leader(self): failover = self.cluster.failover + + if failover.scheduled_at: + # If the failover is in the far future, we shouldn't do anything and just return. + # If the failover is in the past, we consider the value to be stale and we remove + # the value. + # If the value is close to now, we initiate the failover + now = datetime.datetime.now(pytz.utc) + try: + delta = (failover.scheduled_at - now).total_seconds() + + if delta > 10: + logging.info('Awaiting failover at %s (in %.0f seconds)', failover.scheduled_at.isoformat(), delta) + return + elif delta < -15: + logger.warning('Found a stale failover value, cleaning up: %s', failover.scheduled_at) + self.dcs.manual_failover('', '', self.cluster.failover.index) + return + + # The value is very close to now + sleep(max(delta, 0)) + logger.info('Manual scheduled failover at {}'.format(failover.scheduled_at.isoformat())) + except TypeError: + logger.warning('Incorrect value in of scheduled_at: %s', failover.scheduled_at) + if not failover.leader or failover.leader == self.state_handler.name: if not failover.member or failover.member != self.state_handler.name: members = [m for m in self.cluster.members if not failover.member or m.name == failover.member] @@ -298,14 +329,14 @@ class Ha: return self.enforce_master_role('acquired session lock as a leader', 'promoted self to leader by acquiring session lock') else: - return self.follow_the_leader('demoted self due after trying and failing to obtain lock', - 'following new leader after trying and failing to obtain lock') + return self.follow('demoted self after trying and failing to obtain lock', + 'following new leader after trying and failing to obtain lock') else: if self.patroni.nofailover: - return self.follow_the_leader('demoting self because I am not allowed to become master', - 'following a different leader because I am not allowed to promote') - return self.follow_the_leader('demoting self because i am not the healthiest node', - 'following a different leader because i am not the healthiest node') + return self.follow('demoting self because I am not allowed to become master', + 'following a different leader because I am not allowed to promote') + return self.follow('demoting self because i am not the healthiest node', + 'following a different leader because i am not the healthiest node') def process_healthy_cluster(self): if self.has_lock(): @@ -323,8 +354,8 @@ class Ha: self.load_cluster_from_dcs() else: logger.info('does not have lock') - return self.follow_the_leader('demoting self because i do not have the lock and i was a leader', - 'no action. i am a secondary and i am following a leader', False) + return self.follow('demoting self because i do not have the lock and i was a leader', + 'no action. i am a secondary and i am following a leader', False) def schedule(self, action): with self._async_executor: @@ -352,7 +383,7 @@ class Ha: def reinitialize(self, cluster): self.state_handler.stop('immediate') self.state_handler.remove_data_directory() - self.copy_backup_from_leader(cluster.leader) + self.clone(cluster.leader) def process_scheduled_action(self): if self.reinitialize_scheduled(): @@ -377,11 +408,21 @@ class Ha: else: return self._async_executor.scheduled_action + ' in progress' - def sysid_valid(self, sysid): + @staticmethod + def sysid_valid(sysid): # sysid does tv_sec << 32, where tv_sec is the number of seconds sine 1970, # so even 1 << 32 would have 10 digits. return str(sysid) and len(str(sysid)) >= 10 and str(sysid).isdigit() + def post_recover(self): + if not self.state_handler.is_running(): + if self.has_lock(): + self.dcs.delete_leader() + self.dcs.reset_cluster() + return 'removed leader key after trying and failing to start postgres' + return 'failed to start postgres' + return None + def _run_cycle(self): try: self.load_cluster_from_dcs() @@ -395,6 +436,13 @@ class Ha: if self._async_executor.busy: return self.handle_long_action_in_progress() + # we've got here, so any async action has finished. Check if we tried to recover and failed + if self.recovering: + self.recovering = False + msg = self.post_recover() + if msg is not None: + return msg + # currently it can trigger only reinitialize msg = self.process_scheduled_action() if msg is not None: @@ -425,14 +473,18 @@ class Ha: else: return self.process_healthy_cluster() finally: - self.state_handler.sync_replication_slots(self.cluster) + # we might not have a valid PostgreSQL connection here if another thread + # stops PostgreSQL, therefore, we only reload replication slots if no + # asynchronous processes are running (should be always the case for the master) + if not self._async_executor.busy: + self.state_handler.sync_replication_slots(self.cluster) except DCSError: logger.error('Error communicating with DCS') if self.state_handler.is_running() and self.state_handler.is_leader(): self.demote(delete_leader=False) return 'demoted self because DCS is not accessible and i was a leader' except (psycopg2.Error, PostgresConnectionException): - logger.exception('Error communicating with Postgresql. Will try again later') + logger.exception('Error communicating with PostgreSQL. Will try again later') def run_cycle(self): with self._async_executor: diff --git a/patroni/postgresql.py b/patroni/postgresql.py index 6e4ecddf..a1cedb58 100644 --- a/patroni/postgresql.py +++ b/patroni/postgresql.py @@ -39,7 +39,7 @@ def parseurl(url): return ret -class Postgresql: +class Postgresql(object): def __init__(self, config): self.config = config @@ -52,7 +52,7 @@ class Postgresql: self.superuser = config['superuser'] self.admin = config['admin'] self.initdb_options = config.get('initdb', []) - self.pgpass = config.get('pgpass', None) or os.path.join(os.path.expanduser('~'), 'pgpass') + self.pgpass = config.get('pgpass') or os.path.join(os.path.expanduser('~'), 'pgpass') self.pg_rewind = config.get('pg_rewind', {}) self.callback = config.get('callbacks', {}) self.use_slots = config.get('use_slots', True) @@ -61,13 +61,13 @@ class Postgresql: self.configuration_to_save = (os.path.join(self.data_dir, 'pg_hba.conf'), os.path.join(self.data_dir, 'postgresql.conf')) self.postmaster_pid = os.path.join(self.data_dir, 'postmaster.pid') - self.trigger_file = config.get('recovery_conf', {}).get('trigger_file', None) or 'promote' + self.trigger_file = config.get('recovery_conf', {}).get('trigger_file') or 'promote' self.trigger_file = os.path.abspath(os.path.join(self.data_dir, self.trigger_file)) self._pg_ctl = ['pg_ctl', '-w', '-D', self.data_dir] self.local_address = self.get_local_address() - connect_address = config.get('connect_address', None) or self.local_address + connect_address = config.get('connect_address') or self.local_address self.connection_string = 'postgres://{username}:{password}@{connect_address}/postgres'.format( connect_address=connect_address, **self.replication) @@ -128,17 +128,25 @@ class Postgresql: break return local_address + ':' + self.port + @property + def _connect_kwargs(self): + r = parseurl('postgres://{0}/postgres'.format(self.local_address)) + if 'username' in self.superuser: + r['user'] = self.superuser['username'] + if 'password' in self.superuser: + r['password'] = self.superuser['password'] + return r + def connection(self): if not self._connection or self._connection.closed != 0: - r = parseurl('postgres://{}/postgres'.format(self.local_address)) - self._connection = psycopg2.connect(**r) + self._connection = psycopg2.connect(**self._connect_kwargs) self._connection.autocommit = True self.server_version = self._connection.server_version return self._connection def _cursor(self): if not self._cursor_holder or self._cursor_holder.closed or self._cursor_holder.connection.closed != 0: - logger.info("established a new patroni connection to the postgres cluster") + logger.info("establishing a new patroni connection to the postgres cluster") self._cursor_holder = self.connection().cursor() return self._cursor_holder @@ -172,32 +180,36 @@ class Postgresql: @staticmethod def initdb_allowed_option(name): if name in ['pgdata', 'nosync', 'pwfile', 'sync-only']: - raise Exception('{} option for initdb is not allowed'.format(name)) + raise Exception('{0} option for initdb is not allowed'.format(name)) return True def get_initdb_options(self): options = [] for o in self.initdb_options: if isinstance(o, string_types) and self.initdb_allowed_option(o): - options.append('--{}'.format(o)) + options.append('--{0}'.format(o)) elif isinstance(o, dict): keys = list(o.keys()) if len(keys) != 1 or not isinstance(keys[0], string_types) or not self.initdb_allowed_option(keys[0]): - raise Exception('Invalid option: {}'.format(o)) - options.append('--{}={}'.format(keys[0], o[keys[0]])) + raise Exception('Invalid option: {0}'.format(o)) + options.append('--{0}={1}'.format(keys[0], o[keys[0]])) else: - raise Exception('Unknown type of initdb option: {}'.format(o)) + raise Exception('Unknown type of initdb option: {0}'.format(o)) return options def initialize(self): self.set_state('initalizing new cluster') options = self.get_initdb_options() pwfile = None - if self.superuser and 'username' not in self.superuser and 'password' in self.superuser: - (fd, pwfile) = tempfile.mkstemp() - os.write(fd, self.superuser['password'].encode()) - os.close(fd) - options.append('--pwfile={}'.format(pwfile)) + + if self.superuser: + if 'username' in self.superuser: + options.append('--username={0}'.format(self.superuser['username'])) + if 'password' in self.superuser: + (fd, pwfile) = tempfile.mkstemp() + os.write(fd, self.superuser['password'].encode()) + os.close(fd) + options.append('--pwfile={0}'.format(pwfile)) ret = subprocess.call(self._pg_ctl + ['initdb'] + (['-o', ' '.join(options)] if options else [])) == 0 if pwfile: @@ -209,7 +221,8 @@ class Postgresql: return ret def delete_trigger_file(self): - os.path.exists(self.trigger_file) and os.unlink(self.trigger_file) + if os.path.exists(self.trigger_file): + os.unlink(self.trigger_file) def write_pgpass(self, record): with open(self.pgpass, 'w') as f: @@ -220,13 +233,12 @@ class Postgresql: env['PGPASSFILE'] = self.pgpass return env - def sync_from_leader(self, leader): - r = parseurl(leader.conn_url) - - env = self.write_pgpass(r) - ret = self.create_replica(leader, env) == 0 - ret and self.delete_trigger_file() - return ret + def sync_replica(self, leader): + env = self.write_pgpass(parseurl(leader.conn_url)) if leader else os.environ.copy() + if self.create_replica(leader, env) == 0: + self.delete_trigger_file() + return True + return False @staticmethod def build_connstring(conn): @@ -234,16 +246,29 @@ class Postgresql: >>> Postgresql.build_connstring({'host': '127.0.0.1', 'port': '5432'}) == 'host=127.0.0.1 port=5432' True """ - return ' '.join('{}={}'.format(param, val) for param, val in sorted(conn.items())) + return ' '.join('{0}={1}'.format(param, val) for param, val in sorted(conn.items())) + + def replica_method_can_work_without_leader(self, method): + return method != 'basebackup' and self.config and self.config.get(method, {}).get('no_master') + + def can_create_replica_without_leader(self): + """ go through the replication methods to see if there are ones + that does not require a running leader to create the replica. + """ + replica_methods = self.config.get('create_replica_method', []) + return any(self.replica_method_can_work_without_leader(replica_method) for replica_method in replica_methods) def create_replica(self, leader, env): # create the replica according to the replica_method # defined by the user. this is a list, so we need to # loop through all methods the user supplies - connstring = leader.conn_url + connstring = leader.conn_url if leader else "" # get list of replica methods from config. # If there is no configuration key, or no value is specified, use basebackup replica_methods = self.config.get('create_replica_method') or ['basebackup'] + # if we don't have any leader, leave only replica methods that work without it + replica_methods = [r for r in replica_methods if self.replica_method_can_work_without_leader(r)] if not leader \ + else replica_methods # go through them in priority order ret = 1 for replica_method in replica_methods: @@ -296,7 +321,7 @@ class Postgresql: cmd = self.callback[cb_name] try: subprocess.Popen(shlex.split(cmd) + [cb_name, self.role, self.scope]) - except: + except OSError: logger.exception('callback %s %s %s %s failed', cmd, cb_name, self.role, self.scope) return False return True @@ -332,7 +357,10 @@ class Postgresql: if not block_callbacks: self.set_state('starting') - ret = subprocess.call(self._pg_ctl + ['start', '-o', self.server_options()]) == 0 + env = os.environ.copy() + if 'username' in self.superuser: + env['PGUSER'] = self.superuser['username'] + ret = subprocess.call(self._pg_ctl + ['start', '-o', self.server_options()], env=env, preexec_fn=os.setsid) == 0 self.set_state('running' if ret else 'start failed') @@ -340,18 +368,21 @@ class Postgresql: self.save_configuration_files() # block_callbacks is used during restart to avoid # running start/stop callbacks in addition to restart ones - ret and not block_callbacks and self.call_nowait(ACTION_ON_START) + if ret and not block_callbacks: + self.call_nowait(ACTION_ON_START) return ret - def checkpoint(self, connstring=None): + def checkpoint(self, connect_kwargs=None): + connect_kwargs = connect_kwargs or self._connect_kwargs + for p in ['connect_timeout', 'options']: + connect_kwargs.pop(p, None) try: - connstring = connstring or 'postgres://{}/postgres'.format(self.local_address) - with psycopg2.connect(connstring) as conn: + with psycopg2.connect(**connect_kwargs) as conn: conn.autocommit = True with conn.cursor() as cur: cur.execute("SET statement_timeout = 0") cur.execute('CHECKPOINT') - except: + except psycopg2.Error: logging.exception('Exception during CHECKPOINT') def stop(self, mode='fast', block_callbacks=False): @@ -383,7 +414,8 @@ class Postgresql: def reload(self): ret = subprocess.call(self._pg_ctl + ['reload']) == 0 - ret and self.call_nowait(ACTION_ON_RELOAD) + if ret: + self.call_nowait(ACTION_ON_RELOAD) return ret def restart(self): @@ -392,13 +424,13 @@ class Postgresql: if ret: self.call_nowait(ACTION_ON_RESTART) else: - self.set_state('restart failed ({})'.format(self.state)) + self.set_state('restart failed ({0})'.format(self.state)) return ret def server_options(self): - options = "--listen_addresses='{}' --port={}".format(self.listen_addresses, self.port) + options = "--listen_addresses='{0}' --port={1}".format(self.listen_addresses, self.port) for setting, value in self.server_parameters.items(): - options += " --{}='{}'".format(setting, value) + options += " --{0}='{1}'".format(setting, value) return options def is_healthy(self): @@ -408,8 +440,7 @@ class Postgresql: return True def check_replication_lag(self, last_leader_operation): - return (last_leader_operation if last_leader_operation else 0) - self.xlog_position() <=\ - self.config.get('maximum_lag_on_failover', 0) + return (last_leader_operation or 0) - self.xlog_position() <= self.config.get('maximum_lag_on_failover', 0) def write_pg_hba(self): with open(os.path.join(self.data_dir, 'pg_hba.conf'), 'a') as f: @@ -436,33 +467,34 @@ class Postgresql: return pattern and (pattern in line) return not pattern - def write_recovery_conf(self, leader): + def write_recovery_conf(self, leader, bootstrap=False): with open(self.recovery_conf, 'w') as f: f.write("""standby_mode = 'on' recovery_target_timeline = 'latest' """) if leader and leader.conn_url: - f.write("""primary_conninfo = '{}'\n""".format(self.primary_conninfo(leader.conn_url))) + f.write("""primary_conninfo = '{0}'\n""".format(self.primary_conninfo(leader.conn_url))) if self.use_slots: - f.write("""primary_slot_name = '{}'\n""".format(self.name)) + f.write("""primary_slot_name = '{0}'\n""".format(self.name)) + if (leader and leader.conn_url) or bootstrap: for name, value in self.config.get('recovery_conf', {}).items(): - f.write("{} = '{}'\n".format(name, value)) + f.write("{0} = '{1}'\n".format(name, value)) def rewind(self, leader): # prepare pg_rewind connection r = parseurl(leader.conn_url) r.update(self.pg_rewind) - r['user'] = r['username'] + r['user'] = r.pop('username') env = self.write_pgpass(r) pc = "user={user} host={host} port={port} dbname=postgres sslmode=prefer sslcompression=1".format(**r) # first run a checkpoint on a promoted master in order # to make it store the new timeline (5540277D.8020309@iki.fi) - self.checkpoint(pc) - logger.info("running pg_rewind from {}".format(pc)) + self.checkpoint(r) + logger.info("running pg_rewind from %s", pc) pg_rewind = ['pg_rewind', '-D', self.data_dir, '--source-server', pc] try: - ret = (subprocess.call(pg_rewind, env=env) == 0) - except: + ret = subprocess.call(pg_rewind, env=env) == 0 + except OSError: ret = False if ret: self.write_recovery_conf(leader) @@ -478,8 +510,7 @@ recovery_target_timeline = 'latest' result = {l.split(':')[0].replace('Current ', '', 1): l.split(':')[1].strip() for l in data if l} except subprocess.CalledProcessError: logger.exception("Error when calling pg_controldata") - finally: - return result + return result def read_postmaster_opts(self): """ returns the list of option names/values from postgres.opts, Empty dict if read failed or no file """ @@ -495,26 +526,26 @@ recovery_target_timeline = 'latest' result[name] = val except IOError: logger.exception('Error when reading postmaster.opts') - finally: - return result + return result - def single_user_mode(self, command=None, options={}): + def single_user_mode(self, command=None, options=None): """ run a given command in a single-user mode. If the command is empty - then just start and stop """ cmd = ['postgres', '--single', '-D', self.data_dir] - for opt in sorted(options): - cmd.extend(['-c', '{0}={1}'.format(opt, options[opt])]) + for opt, val in sorted((options or {}).items()): + cmd.extend(['-c', '{0}={1}'.format(opt, val)]) # need a database name to connect cmd.append('postgres') p = subprocess.Popen(cmd, stdin=subprocess.PIPE, stdout=open(os.devnull, 'w'), stderr=subprocess.STDOUT) if p: - command and p.communicate('{}\n'.format(command)) + if command: + p.communicate('{0}\n'.format(command)) p.stdin.close() return p.wait() return 1 def cleanup_archive_status(self): status_dir = os.path.join(self.data_dir, 'pg_xlog', 'archive_status') - if os.path.isdir(status_dir): + try: for f in os.listdir(status_dir): path = os.path.join(status_dir, f) try: @@ -522,50 +553,51 @@ recovery_target_timeline = 'latest' os.unlink(path) elif os.path.isfile(path): os.remove(path) - except: - logger.exception("Unable to remove {}".format(path)) + except OSError: + logger.exception("Unable to remove %s", path) + except OSError: + logger.exception("Unable to list %s", status_dir) - def follow_the_leader(self, leader, recovery=False): - if not self.check_recovery_conf(leader) or recovery: - change_role = (self.role == 'master') - - self._need_rewind = (self._need_rewind or change_role) and self.can_rewind - if self._need_rewind: - logger.info("set the rewind flag after demote") - self.write_recovery_conf(leader) - if not leader or not self._need_rewind: # do not rewind until the leader becomes available - ret = self.restart() - else: # we have a leader and need to rewind - if self.is_running(): - self.stop() - # at present, pg_rewind only runs when the cluster is shut down cleanly - # and not shutdown in recovery. We have to remove the recovery.conf if present - # and start/shutdown in a single user mode to emulate this. - # XXX: if recovery.conf is linked, it will be written anew as a normal file. - if os.path.islink(self.recovery_conf): - os.unlink(self.recovery_conf) - else: - os.remove(self.recovery_conf) - # Archived segments might be useful to pg_rewind, - # clean the flags that tell we should remove them. - self.cleanup_archive_status() - # Start in a single user mode and stop to produce a clean shutdown - opts = self.read_postmaster_opts() - opts['archive_mode'] = 'on' - opts['archive_command'] = 'false' - self.single_user_mode(options=opts) - if self.rewind(leader): - ret = self.start() - else: - logger.error("unable to rewind the former master") - self.remove_data_directory() - ret = True - self._need_rewind = False - change_role and ret and self.call_nowait(ACTION_ON_ROLE_CHANGE) - return ret - else: + def follow(self, leader, recovery=False): + if self.check_recovery_conf(leader) and not recovery: return True + change_role = self.role == 'master' + self._need_rewind = (self._need_rewind or change_role) and self.can_rewind + if self._need_rewind: + logger.info("set the rewind flag after demote") + self.write_recovery_conf(leader) + if leader and self._need_rewind: # we have a leader and need to rewind + if self.is_running(): + self.stop() + # at present, pg_rewind only runs when the cluster is shut down cleanly + # and not shutdown in recovery. We have to remove the recovery.conf if present + # and start/shutdown in a single user mode to emulate this. + # XXX: if recovery.conf is linked, it will be written anew as a normal file. + if os.path.islink(self.recovery_conf): + os.unlink(self.recovery_conf) + else: + os.remove(self.recovery_conf) + # Archived segments might be useful to pg_rewind, + # clean the flags that tell we should remove them. + self.cleanup_archive_status() + # Start in a single user mode and stop to produce a clean shutdown + opts = self.read_postmaster_opts() + opts.update({'archive_mode': 'on', 'archive_command': 'false'}) + self.single_user_mode(options=opts) + if self.rewind(leader): + ret = self.start() + else: + logger.error("unable to rewind the former master") + self.remove_data_directory() + ret = True + self._need_rewind = False + else: # do not rewind until the leader becomes available + ret = self.restart() + if change_role and ret: + self.call_nowait(ACTION_ON_ROLE_CHANGE) + return ret + def save_configuration_files(self): """ copy postgresql.conf to postgresql.conf.backup to be able to retrive configuration files @@ -574,16 +606,18 @@ recovery_target_timeline = 'latest' """ try: for f in self.configuration_to_save: - os.path.isfile(f) and shutil.copy(f, f + '.backup') - except: + if os.path.isfile(f): + shutil.copy(f, f + '.backup') + except IOError: logger.exception('unable to create backup copies of configuration files') def restore_configuration_files(self): """ restore a previously saved postgresql.conf """ try: for f in self.configuration_to_save: - not os.path.isfile(f) and os.path.isfile(f + '.backup') and shutil.copy(f + '.backup', f) - except: + if not os.path.isfile(f) and os.path.isfile(f + '.backup'): + shutil.copy(f + '.backup', f) + except IOError: logger.exception('unable to restore configuration files from backup') def promote(self): @@ -597,9 +631,6 @@ recovery_target_timeline = 'latest' self.call_nowait(ACTION_ON_ROLE_CHANGE) return ret - def demote(self): - self.follow_the_leader(None) - def create_or_update_role(self, name, password, options): self.query("""DO $$ BEGIN @@ -616,9 +647,7 @@ $$""".format(name, options), name, password, password) def create_replication_user(self): self.create_or_update_role(self.replication['username'], self.replication['password'], 'REPLICATION') - def create_connection_users(self): - if 'username' in self.superuser: - self.create_or_update_role(self.superuser['username'], self.superuser['password'], 'SUPERUSER') + def create_connection_user(self): if self.admin: self.create_or_update_role(self.admin['username'], self.admin['password'], 'CREATEDB CREATEROLE') @@ -638,7 +667,18 @@ $$""".format(name, options), name, password, password) if self.use_slots: try: self.load_replication_slots() - slots = [m.name for m in cluster.members if m.name != self.name] if self.role == 'master' else [] + # if the replicatefrom tag is set on the member - we should not create the replication slot for it on + # the current master, because that member would replicate from elsewhere. We still create the slot if + # the replicatefrom destination member is currently not a member of the cluster (fallback to the + # master), or if replicatefrom destination member happens to be the current master + if self.role == 'master': + slots = [m.name for m in cluster.members if m.name != self.name and + (m.replicatefrom is None or m.replicatefrom == self.name or + not cluster.has_member(m.replicatefrom))] + else: + # only manage slots for replicas that replicate from this one, except for the leader among them + slots = [m.name for m in cluster.members if m.replicatefrom == self.name and + m.name != cluster.leader.name] # drop unused slots for slot in set(self.replication_slots) - set(slots): self.query("""SELECT pg_drop_replication_slot(%s) @@ -652,34 +692,44 @@ $$""".format(name, options), name, password, password) WHERE slot_name = %s)""", slot, slot) self.replication_slots = slots - except: + except psycopg2.Error: logger.exception('Exception when changing replication slots') def last_operation(self): return str(self.xlog_position()) - def bootstrap(self, current_leader=None): + def bootstrap(self, cluster_initialized=False, current_leader=None): """ - Initially bootstrap PostgreSQL, either by creating a data - directory with initdb, or by initalizing a replica from an - exiting leader. Failure in the first case always leads to - exception, since there is no point in continuing if initdb failed. - In the second case, however, a False is returned on failure, since - it is normal for the replica to retry a failed attempt to initialize - from the master. + Populate PostgreSQL data directory by doing one of the following: + - create with initdb if there is no master. + - initialize the replica from an existing master + - initialize the replica using the replica creation method that + works without the master (i.e. restore from on-disk base backup) + + The choice between the last 2 is triggered by the initialize flag. + We should never try to initdb an already initialized cluster, nor + try to bootstrap the cluster that lacks the initialize key from from + the master-less replica creation method (in the latter case, there is + no clear inidicator of the moment we should abandon our attempts and + swich to initdb). + + Failure during initdb always leads to an exception, since there is + no point in continuing if initdb fails. For the rest of the cases, + the function returns False in order to inidicate a failed attempt + that should be retried in the future. """ ret = False - if not current_leader: + if not (cluster_initialized or current_leader): ret = self.initialize() and self.start() if ret: self.create_replication_user() - self.create_connection_users() + self.create_connection_user() else: raise PostgresException("Could not bootstrap master PostgreSQL") else: - if self.sync_from_leader(current_leader): + if self.sync_replica(current_leader): self.restore_configuration_files() - self.write_recovery_conf(current_leader) + self.write_recovery_conf(current_leader, True) ret = self.start() return ret @@ -689,7 +739,7 @@ $$""".format(name, options), name, password, password) new_name = '{0}_{1}'.format(self.data_dir, time.strftime('%Y-%m-%d-%H-%M-%S')) logger.info('renaming data directory to %s', new_name) os.rename(self.data_dir, new_name) - except: + except OSError: logger.exception("Could not rename data directory %s", self.data_dir) def remove_data_directory(self): @@ -703,7 +753,7 @@ $$""".format(name, options), name, password, password) os.remove(self.data_dir) elif os.path.isdir(self.data_dir): shutil.rmtree(self.data_dir) - except: + except (IOError, OSError): logger.exception('Could not remove data directory %s', self.data_dir) self.move_data_directory() diff --git a/patroni/scripts/aws.py b/patroni/scripts/aws.py index bdd1bb45..34a756fd 100755 --- a/patroni/scripts/aws.py +++ b/patroni/scripts/aws.py @@ -9,7 +9,7 @@ import boto.ec2 logger = logging.getLogger(__name__) -class AWSConnection: +class AWSConnection(object): def __init__(self, cluster_name): self.available = False self.cluster_name = cluster_name if cluster_name is not None else 'unknown' @@ -56,7 +56,7 @@ class AWSConnection: conn = boto.ec2.connect_to_region(self.region) conn.create_tags([self.instance_id], tags) except Exception as e: - logger.info("could not set tags for EC2 instance {}: {}".format(self.instance_id, e)) + logger.info("could not set tags for EC2 instance %s: %s", self.instance_id, e) return False return True @@ -66,6 +66,7 @@ class AWSConnection: def main(): + logging.basicConfig(format='%(asctime)s %(levelname)s: %(message)s', level=logging.INFO) if len(sys.argv) == 4 and sys.argv[1] in ('on_start', 'on_stop', 'on_role_change'): AWSConnection(cluster_name=sys.argv[3]).on_role_change(sys.argv[2]) else: diff --git a/patroni/scripts/wale_restore.py b/patroni/scripts/wale_restore.py index f54c37a3..c80cdfae 100755 --- a/patroni/scripts/wale_restore.py +++ b/patroni/scripts/wale_restore.py @@ -1,4 +1,4 @@ -#!/usr/bin/python +#!/usr/bin/env python # sample script to clone new replicas using WAL-E restore # falls back to pg_basebackup if WAL-E restore fails, or if @@ -36,13 +36,12 @@ import argparse if sys.hexversion >= 0x03000000: long = int -logging.basicConfig(format='%(asctime)s %(levelname)s: %(message)s', level=logging.INFO) logger = logging.getLogger(__name__) class WALERestore(object): - def __init__(self, scope, datadir, connstring, env_dir, threshold_mb, threshold_pct, use_iam): + def __init__(self, scope, datadir, connstring, env_dir, threshold_mb, threshold_pct, use_iam, no_master): self.scope = scope self.master_connection = connstring self.data_dir = datadir @@ -51,6 +50,7 @@ class WALERestore(object): self.wal_e.threshold_mb = threshold_mb self.wal_e.threshold_pct = threshold_pct self.wal_e.iam_string = ' --aws-instance-profile ' if use_iam == 1 else '' + self.no_master = no_master self.wal_e.cmd = 'envdir {0} wal-e {1} '.format(self.wal_e.dir, self.wal_e.iam_string) self.init_error = (not os.path.exists(self.wal_e.dir)) @@ -104,24 +104,23 @@ class WALERestore(object): lsn_offset = hex((long(backup_start_segment[16:32], 16) << 24) + long(backup_start_offset))[2:-1] # construct the LSN from the segment and offset - backup_start_lsn = '{}/{}'.format(lsn_segment, lsn_offset) + backup_start_lsn = '{0}/{1}'.format(lsn_segment, lsn_offset) - conn = None - cursor = None diff_in_bytes = long(backup_size) - try: - # get the difference in bytes between the current WAL location and the backup start offset - conn = psycopg2.connect(self.master_connection) - conn.autocommit = True - cursor = conn.cursor() - cursor.execute("SELECT pg_xlog_location_diff(pg_current_xlog_location(), %s)", (backup_start_lsn,)) - diff_in_bytes = long(cursor.fetchone()[0]) - except psycopg2.Error as e: - logger.error('could not determine difference with the master location: {}'.format(e)) - return False - finally: - cursor and cursor.close() - conn and conn.close() + if not self.no_master: + try: + # get the difference in bytes between the current WAL location and the backup start offset + with psycopg2.connect(self.master_connection) as con: + con.autocommit = True + with con.cursor() as cur: + cur.execute("SELECT pg_xlog_location_diff(pg_current_xlog_location(), %s)", (backup_start_lsn,)) + diff_in_bytes = long(cur.fetchone()[0]) + except psycopg2.Error as e: + logger.error('could not determine difference with the master location: %s', e) + return False + else: + # always try to use WAL-E if base backup is available + diff_in_bytes = 0 # if the size of the accumulated WAL segments is more than a certan percentage of the backup size # or exceeds the pre-determined size - pg_basebackup is chosen instead. @@ -140,6 +139,7 @@ class WALERestore(object): def main(): + logging.basicConfig(format='%(asctime)s %(levelname)s: %(message)s', level=logging.INFO) parser = argparse.ArgumentParser(description='Script to image replicas using WAL-E') parser.add_argument('--scope', required=True) parser.add_argument('--role', required=False) @@ -150,13 +150,15 @@ def main(): parser.add_argument('--threshold_megabytes', type=int, default=10240) parser.add_argument('--threshold_backup_size_percentage', type=int, default=30) parser.add_argument('--use_iam', type=int, default=0) + parser.add_argument('--no_master', type=int, default=0) args = parser.parse_args() # retry cloning in a loop for retry in range(0, args.retries + 1): restore = WALERestore(scope=args.scope, datadir=args.datadir, connstring=args.connstring, env_dir=args.envdir, threshold_mb=args.threshold_megabytes, - threshold_pct=args.threshold_backup_size_percentage, use_iam=args.use_iam) + threshold_pct=args.threshold_backup_size_percentage, use_iam=args.use_iam, + no_master=args.no_master) ret = restore.run() if ret == 0: break diff --git a/patroni/utils.py b/patroni/utils.py index 9b040294..cb101758 100644 --- a/patroni/utils.py +++ b/patroni/utils.py @@ -1,76 +1,61 @@ import datetime import os import random -import re import signal import sys import time +import pytz +import dateutil.parser from patroni.exceptions import PatroniException -ignore_sigterm = False -interrupted_sleep = False -reap_children = False - -_DATE_TIME_RE = re.compile(r'''^ -(?P\d{4})\-(?P\d{2})\-(?P\d{2}) # date -T -(?P\d{2}):(?P\d{2}):(?P\d{2})\.(?P\d{6}) # time -\d*Z$''', re.X) - - -def parse_datetime(time_str): - """ - >>> parse_datetime('2015-06-10T12:56:30.552539016Z') - datetime.datetime(2015, 6, 10, 12, 56, 30, 552539) - >>> parse_datetime('2015-06-10 12:56:30.552539016Z') - """ - m = _DATE_TIME_RE.match(time_str) - if not m: - return None - p = dict((n, int(m.group(n))) for n in 'year month day hour minute second microsecond'.split(' ')) - return datetime.datetime(**p) +__ignore_sigterm = False +__interrupted_sleep = False +__reap_children = False def calculate_ttl(expiration): """ >>> calculate_ttl(None) - >>> calculate_ttl('2015-06-10 12:56:30.552539016Z') + >>> calculate_ttl('2015-06-10 12:56:30.552539016Z') < 0 + True >>> calculate_ttl('2015-06-10T12:56:30.552539016Z') < 0 True + >>> calculate_ttl('fail-06-10T12:56:30.552539016Z') """ if not expiration: return None - expiration = parse_datetime(expiration) - if not expiration: + try: + expiration = dateutil.parser.parse(expiration) + except (ValueError, TypeError): return None - now = datetime.datetime.utcnow() + now = datetime.datetime.now(pytz.utc) return int((expiration - now).total_seconds()) def sigterm_handler(signo, stack_frame): - global ignore_sigterm - if not ignore_sigterm: - ignore_sigterm = True + global __ignore_sigterm + if not __ignore_sigterm: + __ignore_sigterm = True sys.exit() def sigchld_handler(signo, stack_frame): - global interrupted_sleep, reap_children - reap_children = interrupted_sleep = True + global __interrupted_sleep, __reap_children + __reap_children = __interrupted_sleep = True def sleep(interval): - global interrupted_sleep + global __interrupted_sleep current_time = time.time() end_time = current_time + interval while current_time < end_time: - interrupted_sleep = False + __interrupted_sleep = False time.sleep(end_time - current_time) - if not interrupted_sleep: # we will ignore only sigchld + if not __interrupted_sleep: # we will ignore only sigchld break current_time = time.time() - interrupted_sleep = False + __interrupted_sleep = False def setup_signal_handlers(): @@ -79,8 +64,8 @@ def setup_signal_handlers(): def reap_children(): - global reap_children - if reap_children: + global __reap_children + if __reap_children: try: while True: ret = os.waitpid(-1, os.WNOHANG) @@ -89,7 +74,7 @@ def reap_children(): except OSError: pass finally: - reap_children = False + __reap_children = False class RetryFailedError(PatroniException): @@ -97,7 +82,7 @@ class RetryFailedError(PatroniException): """Raised when retrying an operation ultimately failed, after retrying the maximum number of attempts.""" -class Retry: +class Retry(object): """Helper for retrying a method in the face of retry-able exceptions""" diff --git a/patroni/zookeeper.py b/patroni/zookeeper.py index bc9b83c4..79e8a72a 100644 --- a/patroni/zookeeper.py +++ b/patroni/zookeeper.py @@ -17,7 +17,7 @@ class ZooKeeperError(DCSError): pass -class ExhibitorEnsembleProvider: +class ExhibitorEnsembleProvider(object): TIMEOUT = 3.1 @@ -54,7 +54,7 @@ class ExhibitorEnsembleProvider: def _query_exhibitors(self, exhibitors): random.shuffle(exhibitors) for host in exhibitors: - uri = 'http://{}:{}{}'.format(host, self._exhibitor_port, self._uri_path) + uri = 'http://{0}:{1}{2}'.format(host, self._exhibitor_port, self._uri_path) try: response = requests.get(uri, timeout=self.TIMEOUT) return response.json() @@ -84,9 +84,9 @@ class ZooKeeper(AbstractDCS): hosts = self.exhibitor.zookeeper_hosts self.client = KazooClient(hosts=hosts, - timeout=(config.get('session_timeout', None) or 30), + timeout=(config.get('session_timeout') or 30), command_retry={ - 'deadline': (config.get('reconnect_timeout', None) or 10), + 'deadline': (config.get('reconnect_timeout') or 10), 'max_delay': 1, 'max_tries': -1}, connection_retry={'max_delay': 1, 'max_tries': -1}) @@ -190,7 +190,8 @@ class ZooKeeper(AbstractDCS): def attempt_to_acquire_leader(self): ret = self._create(self.leader_path, self._name, makepath=True, ephemeral=True) - ret or logger.info('Could not take out TTL lock') + if ret: + logger.info('Could not take out TTL lock') return ret def set_failover_value(self, value, index=None): diff --git a/patronictl.py b/patronictl.py index 5b06c153..50e65c87 100755 --- a/patronictl.py +++ b/patronictl.py @@ -2,4 +2,4 @@ from patroni.ctl import ctl if __name__ == '__main__': - ctl() + ctl(None) diff --git a/postgres0.yml b/postgres0.yml index 244a0f19..8b77259e 100644 --- a/postgres0.yml +++ b/postgres0.yml @@ -87,14 +87,13 @@ postgresql: archive_mode: "on" wal_level: hot_standby archive_command: mkdir -p ../wal_archive && test ! -f ../wal_archive/%f && cp %p ../wal_archive/%f - max_wal_senders: 5 + max_wal_senders: 10 wal_keep_segments: 8 archive_timeout: 1800s - max_replication_slots: 5 + max_replication_slots: 10 hot_standby: "on" wal_log_hints: "on" tags: nofailover: False noloadbalance: False clonefrom: False - replicatefrom: 127.0.0.1 diff --git a/postgres1.yml b/postgres1.yml index 58fc308c..c2d84ea0 100644 --- a/postgres1.yml +++ b/postgres1.yml @@ -63,7 +63,7 @@ postgresql: password: rep-pass network: 127.0.0.1/32 superuser: - user: postgres + username: postgres password: zalando admin: username: admin @@ -88,14 +88,13 @@ postgresql: archive_mode: "on" wal_level: hot_standby archive_command: mkdir -p ../wal_archive && test ! -f ../wal_archive/%f && cp %p ../wal_archive/%f - max_wal_senders: 5 + max_wal_senders: 10 wal_keep_segments: 8 archive_timeout: 1800s - max_replication_slots: 5 + max_replication_slots: 10 hot_standby: "on" wal_log_hints: "on" tags: nofailover: False noloadbalance: False clonefrom: False - replicatefrom: 127.0.0.1 diff --git a/postgres2.yml b/postgres2.yml new file mode 100644 index 00000000..33620823 --- /dev/null +++ b/postgres2.yml @@ -0,0 +1,101 @@ +ttl: &ttl 30 +loop_wait: &loop_wait 10 +scope: &scope batman +restapi: + listen: 127.0.0.1:8010 + connect_address: 127.0.0.1:8010 + auth: 'username:password' +# certfile: /etc/ssl/certs/ssl-cert-snakeoil.pem +# keyfile: /etc/ssl/private/ssl-cert-snakeoil.key +etcd: + scope: *scope + ttl: *ttl + host: 127.0.0.1:4001 + #discovery_srv: my-etcd.domain +#zookeeper: +# scope: *scope +# session_timeout: *ttl +# reconnect_timeout: *loop_wait +# hosts: +# - 127.0.0.1:2181 +# - 127.0.0.2:2181 +# exhibitor: +# poll_interval: 300 +# port: 8181 +# hosts: +# - host1 +# - host2 +# - host3 +postgresql: + name: postgresql2 + scope: *scope + listen: 127.0.0.1:5434 + connect_address: 127.0.0.1:5434 + data_dir: data/postgresql2 + maximum_lag_on_failover: 1048576 # 1 megabyte in bytes + use_slots: True + pgpass: /tmp/pgpass2 + initdb: ## We allow the following options to be passed on to initdb + # - auth: authmethod + # - auth-host: authmethod + # - auth-local: authmethod + - encoding: UTF8 + # - data-checksums # When pg_rewind is needed on 9.3, this needs to be enabled + # - locale: locale + # - lc-collate: locale + # - lc-ctype: locale + # - lc-messages: locale + # - lc-monetary: locale + # - lc-numeric: locale + # - lc-time: locale + # - text-search-config: CFG + # - xlogdir: directory + # - debug + # - noclean + pg_rewind: + username: postgres + password: zalando + pg_hba: + - host all all 0.0.0.0/0 md5 + - hostssl all all 0.0.0.0/0 md5 + replication: + username: replicator + password: rep-pass + network: 127.0.0.1/32 + superuser: + username: postgres + password: zalando + admin: + username: admin + password: admin +# commented-out example for wal-e provisioning + create_replica_method: + - basebackup +# - wal_e +# commented-out example for wal-e provisioning + #wal_e: + #command: /patroni/scripts/wale_restore.py + #env_dir: /home/postgres/etc/wal-e.d/env + #threshold_megabytes: 10240 + #threshold_backup_size_percentage: 30 + #retries: 2 + #use_iam: 1 + #recovery_conf: + #restore_command: envdir /etc/wal-e.d/env wal-e wal-fetch "%f" "%p" -p 1 + recovery_conf: + restore_command: cp ../wal_archive/%f %p + parameters: + archive_mode: "on" + wal_level: hot_standby + archive_command: mkdir -p ../wal_archive && test ! -f ../wal_archive/%f && cp %p ../wal_archive/%f + max_wal_senders: 10 + wal_keep_segments: 8 + archive_timeout: 1800s + max_replication_slots: 10 + hot_standby: "on" + wal_log_hints: "on" +tags: + nofailover: False + noloadbalance: False + clonefrom: False + replicatefrom: postgresql1 diff --git a/requirements-py2.txt b/requirements-py2.txt index 1e194bc0..26c0892d 100644 --- a/requirements-py2.txt +++ b/requirements-py2.txt @@ -9,3 +9,5 @@ kazoo>=2.2.1 python-etcd==0.4.2 click>=4.1 prettytable>=0.7 +tzlocal +python-dateutil diff --git a/requirements-py3.txt b/requirements-py3.txt index 13c3010c..a827efb3 100644 --- a/requirements-py3.txt +++ b/requirements-py3.txt @@ -9,3 +9,5 @@ kazoo>=2.2.1 python-etcd==0.4.2 click>=4.1 prettytable>=0.7 +tzlocal +python-dateutil diff --git a/tests/test_api.py b/tests/test_api.py index 4b24704d..1be0f96c 100644 --- a/tests/test_api.py +++ b/tests/test_api.py @@ -19,10 +19,12 @@ class MockPostgresql(Mock): server_version = '999999' scope = 'dummy' - def connection(self): + @staticmethod + def connection(): return psycopg2_connect() - def is_running(self): + @staticmethod + def is_running(): return True @@ -31,23 +33,28 @@ class MockHa(Mock): dcs = Mock() state_handler = MockPostgresql() - def schedule_restart(self): + @staticmethod + def schedule_restart(): return 'restart' - def schedule_reinitialize(self): + @staticmethod + def schedule_reinitialize(): return 'reinitialize' - def restart(self): + @staticmethod + def restart(): return (True, '') - def restart_scheduled(self): + @staticmethod + def restart_scheduled(): return False - def fetch_nodes_statuses(self, members): + @staticmethod + def fetch_nodes_statuses(members): return [[None, True, None, None, {}]] -class MockPatroni: +class MockPatroni(Mock): postgresql = MockPostgresql() ha = MockHa() @@ -56,7 +63,7 @@ class MockPatroni: version = '0.00' -class MockRequest: +class MockRequest(object): def __init__(self, path): self.path = path @@ -94,10 +101,10 @@ class TestRestApiHandler(unittest.TestCase): MockRestApiServer(RestApiHandler, b'GET /master') with patch.object(MockHa, 'restart_scheduled', Mock(return_value=True)): MockRestApiServer(RestApiHandler, b'GET /master') - MockRestApiServer(RestApiHandler, b'GET /master') + self.assertIsNotNone(MockRestApiServer(RestApiHandler, b'GET /master')) def test_do_OPTIONS(self): - MockRestApiServer(RestApiHandler, b'OPTIONS / HTTP/1.0') + self.assertIsNotNone(MockRestApiServer(RestApiHandler, b'OPTIONS / HTTP/1.0')) with patch.object(BaseHTTPRequestHandler, 'handle_one_request') as mock_handle_request: mock_handle_request.side_effect = socket.error("foo") @@ -111,15 +118,15 @@ class TestRestApiHandler(unittest.TestCase): MockRestApiServer(RestApiHandler, b'OPTIONS / HTTP/1.0') def test_do_GET_patroni(self): - MockRestApiServer(RestApiHandler, b'GET /patroni') + self.assertIsNotNone(MockRestApiServer(RestApiHandler, b'GET /patroni')) def test_basicauth(self): - MockRestApiServer(RestApiHandler, b'POST /restart HTTP/1.0') + self.assertIsNotNone(MockRestApiServer(RestApiHandler, b'POST /restart HTTP/1.0')) MockRestApiServer(RestApiHandler, b'POST /restart HTTP/1.0\nAuthorization:') def test_do_POST_restart(self): request = b'POST /restart HTTP/1.0\nAuthorization: Basic dGVzdDp0ZXN0' - MockRestApiServer(RestApiHandler, request) + self.assertIsNotNone(MockRestApiServer(RestApiHandler, request)) with patch.object(MockHa, 'restart', Mock(side_effect=Exception)): MockRestApiServer(RestApiHandler, request) @@ -133,14 +140,14 @@ class TestRestApiHandler(unittest.TestCase): with patch.object(MockHa, 'schedule_reinitialize', Mock(return_value=None)): MockRestApiServer(RestApiHandler, request) cluster.leader.name = 'test' - MockRestApiServer(RestApiHandler, request) + self.assertIsNotNone(MockRestApiServer(RestApiHandler, request)) @patch('time.sleep', Mock()) def test_RestApiServer_query(self): with patch.object(MockCursor, 'execute', Mock(side_effect=psycopg2.OperationalError)): - MockRestApiServer(RestApiHandler, b'GET /patroni') + self.assertIsNotNone(MockRestApiServer(RestApiHandler, b'GET /patroni')) with patch.object(MockPostgresql, 'connection', Mock(side_effect=psycopg2.OperationalError)): - MockRestApiServer(RestApiHandler, b'GET /patroni') + self.assertIsNotNone(MockRestApiServer(RestApiHandler, b'GET /patroni')) @patch('time.sleep', Mock()) @patch.object(MockHa, 'dcs') @@ -169,3 +176,23 @@ class TestRestApiHandler(unittest.TestCase): request = b'POST /failover HTTP/1.0\nAuthorization: Basic dGVzdDp0ZXN0\n' +\ b'Content-Length: 50\n\n{"leader": "postgresql1", "member": "postgresql2"}' MockRestApiServer(RestApiHandler, request) + + # Valid future date + request = b'POST /failover HTTP/1.0\nAuthorization: Basic dGVzdDp0ZXN0\nContent-Length: 103\n\n{"leader": ' +\ + b'"postgresql1", "member": "postgresql2", "scheduled_at": "6016-02-15T18:13:30.568224+01:00"}' + MockRestApiServer(RestApiHandler, request) + + # Exception: No timezone specified + request = b'POST /failover HTTP/1.0\nAuthorization: Basic dGVzdDp0ZXN0\nContent-Length: 97\n\n{"leader": ' +\ + b'"postgresql1", "member": "postgresql2", "scheduled_at": "6016-02-15T18:13:30.568224"}' + MockRestApiServer(RestApiHandler, request) + + # Exception: Scheduled in the past + request = b'POST /failover HTTP/1.0\nAuthorization: Basic dGVzdDp0ZXN0\nContent-Length: 103\n\n{"leader": ' +\ + b'"postgresql1", "member": "postgresql2", "scheduled_at": "1016-02-15T18:13:30.568224+01:00"}' + MockRestApiServer(RestApiHandler, request) + + # Invalid date + request = b'POST /failover HTTP/1.0\nAuthorization: Basic dGVzdDp0ZXN0\nContent-Length: 103\n\n{"leader": ' +\ + b'"postgresql1", "member": "postgresql2", "scheduled_at": "2010-02-29T18:13:30.568224+01:00"}' + self.assertIsNotNone(MockRestApiServer(RestApiHandler, request)) diff --git a/tests/test_aws.py b/tests/test_aws.py index 09c357f4..be4b918d 100644 --- a/tests/test_aws.py +++ b/tests/test_aws.py @@ -1,12 +1,15 @@ -import unittest -import requests import boto.ec2 +import requests +import sys +import unittest + +from mock import Mock, patch from collections import namedtuple -from patroni.scripts.aws import AWSConnection +from patroni.scripts.aws import AWSConnection, main as _main from requests.exceptions import RequestException -class MockEc2Connection: +class MockEc2Connection(object): def __init__(self, error=False): self.error = error @@ -23,7 +26,7 @@ class MockEc2Connection: return True -class MockResponse: +class MockResponse(object): def __init__(self, content): self.content = content @@ -35,15 +38,6 @@ class MockResponse: class TestAWSConnection(unittest.TestCase): - def __init__(self, method_name='runTest'): - super(TestAWSConnection, self).__init__(method_name) - - def set_error(self): - self.error = True - - def set_json_error(self): - self.json_error = True - def boto_ec2_connect_to_region(self, region): return MockEc2Connection(self.error) @@ -74,21 +68,27 @@ class TestAWSConnection(unittest.TestCase): self.assertTrue(self.conn.on_role_change('master')) def test_non_aws(self): - self.set_error() + self.error = True conn = AWSConnection('test') self.assertFalse(conn.aws_available()) self.assertFalse(conn._tag_ebs('master')) self.assertFalse(conn._tag_ec2('master')) def test_aws_bizare_response(self): - self.set_json_error() + self.json_error = True conn = AWSConnection('test') self.assertFalse(conn.aws_available()) def test_aws_tag_ebs_error(self): - self.set_error() + self.error = True self.assertFalse(self.conn._tag_ebs("master")) def test_aws_tag_ec2_error(self): - self.set_error() + self.error = True self.assertFalse(self.conn._tag_ec2("master")) + + @patch('sys.exit', Mock()) + def test_main(self): + self.assertIsNone(_main()) + sys.argv = ['aws.py', 'on_start', 'replica', 'foo'] + self.assertIsNone(_main()) diff --git a/tests/test_ctl.py b/tests/test_ctl.py index 093ead32..3987619d 100644 --- a/tests/test_ctl.py +++ b/tests/test_ctl.py @@ -1,31 +1,27 @@ -#!/usr/bin/env python -# -*- coding: utf-8 -*- - import os import pytest +import requests.exceptions import unittest -import psycopg2 -import requests -import patroni.exceptions -import etcd -from mock import patch, Mock from click.testing import CliRunner +from etcd import EtcdException +from mock import patch, Mock, MagicMock from patroni.ctl import ctl, members, store_config, load_config, output_members, post_patroni, get_dcs, \ wait_for_leader, get_all_members, get_any_member, get_cursor, query_member, configure -from patroni.ha import Ha from patroni.etcd import Etcd, Client -from test_ha import get_cluster_initialized_without_leader, get_cluster_initialized_with_leader, \ - get_cluster_initialized_with_only_leader, MockPostgresql, MockPatroni, run_async, \ - get_cluster_not_initialized_without_leader +from patroni.exceptions import PatroniCtlException +from psycopg2 import OperationalError from test_etcd import etcd_read, etcd_write, requests_get, socket_getaddrinfo, MockResponse +from test_ha import get_cluster_initialized_without_leader, get_cluster_initialized_with_leader, \ + get_cluster_initialized_with_only_leader from test_postgresql import MockConnect, psycopg2_connect CONFIG_FILE_PATH = './test-ctl.yaml' + def test_rw_config(): runner = CliRunner() - config = {'a':'b'} + config = {'a': 'b'} with runner.isolated_filesystem(): store_config(config, CONFIG_FILE_PATH + '/dummy') os.remove(CONFIG_FILE_PATH + '/dummy') @@ -45,44 +41,36 @@ def test_rw_config(): load_config(CONFIG_FILE_PATH, None) load_config(CONFIG_FILE_PATH, '0.0.0.0') + @patch('patroni.ctl.load_config', Mock(return_value={'dcs': {'scheme': 'etcd', 'hostname': 'localhost', 'port': 4001}})) class TestCtl(unittest.TestCase): @patch('socket.getaddrinfo', socket_getaddrinfo) - @patch.object(Client, 'machines') - def setUp(self, mock_machines): - mock_machines.__get__ = Mock(return_value=['http://remotehost:2379']) - self.p = MockPostgresql() - self.e = Etcd('foo', {'ttl': 30, 'host': 'ok:2379', 'scope': 'test'}) - self.e.client.read = etcd_read - self.e.client.write = etcd_write - self.e.client.delete = Mock(side_effect=etcd.EtcdException()) - self.ha = Ha(MockPatroni(self.p, self.e)) - self.ha._async_executor.run_async = run_async - self.ha.old_cluster = self.e.get_cluster() - self.ha.cluster = get_cluster_not_initialized_without_leader() - self.ha.load_cluster_from_dcs = Mock() + def setUp(self): + self.runner = CliRunner() + with patch.object(Client, 'machines') as mock_machines: + mock_machines.__get__ = Mock(return_value=['http://remotehost:2379']) + self.e = Etcd('foo', {'ttl': 30, 'host': 'ok:2379', 'scope': 'test'}) + self.e.client.read = etcd_read + self.e.client.write = etcd_write + self.e.client.delete = Mock(side_effect=EtcdException) @patch('psycopg2.connect', psycopg2_connect) def test_get_cursor(self): - c = get_cursor(get_cluster_initialized_without_leader(), role='master') - assert c is None + self.assertIsNone(get_cursor(get_cluster_initialized_without_leader(), role='master')) - c = get_cursor(get_cluster_initialized_with_leader(), role='master') - assert c is not None + self.assertIsNotNone(get_cursor(get_cluster_initialized_with_leader(), role='master')) - c = get_cursor(get_cluster_initialized_with_leader(), role='replica') - # # MockCursor returns pg_is_in_recovery as false - assert c is None + # MockCursor returns pg_is_in_recovery as false + self.assertIsNone(get_cursor(get_cluster_initialized_with_leader(), role='replica')) - c = get_cursor(get_cluster_initialized_with_leader(), role='any') - assert c is not None + self.assertIsNotNone(get_cursor(get_cluster_initialized_with_leader(), role='any')) def test_output_members(self): cluster = get_cluster_initialized_with_leader() - output_members(cluster, name='abc', format='pretty') - output_members(cluster, name='abc', format='json') - output_members(cluster, name='abc', format='tsv') + self.assertIsNone(output_members(cluster, name='abc', fmt='pretty')) + self.assertIsNone(output_members(cluster, name='abc', fmt='json')) + self.assertIsNone(output_members(cluster, name='abc', fmt='tsv')) @patch('patroni.etcd.Etcd.get_cluster', Mock(return_value=get_cluster_initialized_with_leader())) @patch('patroni.etcd.Etcd.get_etcd_client', Mock(return_value=None)) @@ -92,89 +80,108 @@ class TestCtl(unittest.TestCase): @patch('requests.post', requests_get) @patch('patroni.ctl.post_patroni', Mock(return_value=MockResponse())) def test_failover(self): - runner = CliRunner() - with patch('patroni.etcd.Etcd.get_cluster', Mock(return_value=get_cluster_initialized_with_leader())): - result = runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader other -y''') - assert 'Failing over to new leader' in result.output - result = runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader +y''') + assert 'leader' in result.output + + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader other +2100-01-01T12:23:00 +y''') + assert result.exit_code == 0 + + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader +other +2030-01-01T12:23:00 +y''') + assert result.exit_code == 0 + + # Aborting failover,as we anser NO to the confirmation + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader +other + N''') - assert 'Aborting failover' in str(result.exception) + assert result.exit_code == 1 - result = runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader + # Target and source are equal + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader leader -y''') - assert 'target and source are the same' in str(result.exception) - result = runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader +y''') + assert result.exit_code == 1 + + # Reality is not part of this cluster + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader Reality + y''') - assert 'Reality does not exist' in str(result.exception) + assert result.exit_code == 1 - result = runner.invoke(ctl, ['failover', 'dummy', '--force']) - assert 'Failing over to new leader' in result.output + result = self.runner.invoke(ctl, ['failover', 'dummy', '--force']) + assert 'Member' in result.output - result = runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='dummy') - assert 'is not the leader of cluster' in str(result.exception) + result = self.runner.invoke(ctl, ['failover', 'dummy', '--force', + '--scheduled', '2015-01-01T12:00:00+01:00']) + assert result.exit_code == 0 + + # Invalid timestamp + result = self.runner.invoke(ctl, ['failover', 'dummy', '--force', '--scheduled', 'invalid']) + assert result.exit_code != 0 + + # Invalid timestamp + result = self.runner.invoke(ctl, ['failover', 'dummy', '--force', + '--scheduled', '2115-02-30T12:00:00+01:00']) + assert result.exit_code != 0 + + # Specifying wrong leader + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='dummy') + assert result.exit_code == 1 with patch('patroni.etcd.Etcd.get_cluster', Mock(return_value=get_cluster_initialized_with_only_leader())): - result = runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader + # No members available + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader other + y''') - assert 'No candidates found to failover to' in str(result.exception) + assert result.exit_code == 1 with patch('patroni.etcd.Etcd.get_cluster', Mock(return_value=get_cluster_initialized_without_leader())): - result = runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader + # No master available + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader other + y''') - assert 'This cluster has no master' in str(result.exception) + assert result.exit_code == 1 with patch('patroni.ctl.post_patroni', Mock(side_effect=Exception())): - result = runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader + # Non-responding patroni + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader other + y''') assert 'falling back to DCS' in result.output - assert 'Failover failed' in result.output mocked = Mock() mocked.return_value.status_code = 500 with patch('patroni.ctl.post_patroni', Mock(return_value=mocked)): - result = runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader + result = self.runner.invoke(ctl, ['failover', 'dummy', '--dcs', '8.8.8.8'], input='''leader other + y''') - assert 'Failover failed, details' in result.output - -# with patch('patroni.dcs.AbstractDCS.get_cluster', Mock(return_value=get_cluster_initialized_with_leader())): -# result = runner.invoke(ctl, ['failover', 'alpha', '--dcs', '8.8.8.8'], input='nonsense') -# assert 'is not the leader of cluster' in str(result.exception) - - # result = runner.invoke(ctl, ['failover', 'alpha', '--dcs', '8.8.8.8', '--master', 'nonsense']) - # assert 'is not the leader of cluster' in str(result.exception) - - # result = runner.invoke(ctl, ['failover', 'alpha', '--dcs', '8.8.8.8'], input='leader\nother\nn') - # assert 'Aborting failover' in str(result.exception) - - # with patch('patroni.ctl.wait_for_leader', Mock(return_value = get_cluster_initialized_with_leader())): - # result = runner.invoke(ctl, ['failover', 'alpha', '--dcs', '8.8.8.8'], input='leader\nother\nY') - # assert 'master did not change after' in result.output - - # result = runner.invoke(ctl, ['failover', 'alpha', '--dcs', '8.8.8.8'], input='leader\nother\nY') - # assert 'Failover failed' in result.output + assert 'Failover failed' in result.output def test_(self): - self.assertRaises(patroni.exceptions.PatroniCtlException, get_dcs, {'scheme': 'dummy'}, 'dummy') + self.assertRaises(PatroniCtlException, get_dcs, {'scheme': 'dummy'}, 'dummy') @patch('psycopg2.connect', psycopg2_connect) @patch('patroni.ctl.query_member', Mock(return_value=([['mock column']], None))) def test_query(self): - runner = CliRunner() - with patch('patroni.ctl.get_dcs', Mock(return_value=self.e)): - result = runner.invoke(ctl, [ + # Mutually exclusive + result = self.runner.invoke(ctl, [ 'query', 'alpha', '--member', @@ -182,14 +189,14 @@ y''') '--role', 'master', ]) - assert 'mutually exclusive' in str(result.exception) + assert result.exit_code == 1 - with runner.isolated_filesystem(): - dummy_file = open('dummy', 'w') - dummy_file.write('SELECT 1') - dummy_file.close() + with self.runner.isolated_filesystem(): + with open('dummy', 'w') as dummy_file: + dummy_file.write('SELECT 1') - result = runner.invoke(ctl, [ + # Mutually exclusive + result = self.runner.invoke(ctl, [ 'query', 'alpha', '--file', @@ -197,45 +204,52 @@ y''') '--command', 'dummy', ]) - assert 'mutually exclusive' in str(result.exception) + assert result.exit_code == 1 - result = runner.invoke(ctl, ['query', 'alpha', '--file', 'dummy']) + result = self.runner.invoke(ctl, ['query', 'alpha', '--file', 'dummy']) os.remove('dummy') - result = runner.invoke(ctl, ['query', 'alpha', '--command', 'SELECT 1']) + result = self.runner.invoke(ctl, ['query', 'alpha', '--command', 'SELECT 1']) + assert 'mock column' in result.output + + # --command or --file is mandatory + result = self.runner.invoke(ctl, ['query', 'alpha']) + assert result.exit_code == 1 + + result = self.runner.invoke(ctl, ['query', 'alpha', '--command', 'SELECT 1', '--username', 'root', + '--password', '--dbname', 'postgres'], input='ab\nab') assert 'mock column' in result.output @patch('patroni.ctl.get_cursor', Mock(return_value=MockConnect().cursor())) def test_query_member(self): rows = query_member(None, None, None, 'master', 'SELECT pg_is_in_recovery()') - assert 'False' in str(rows) + self.assertTrue('False' in str(rows)) rows = query_member(None, None, None, 'replica', 'SELECT pg_is_in_recovery()') - assert rows == (None, None) + self.assertEquals(rows, (None, None)) with patch('patroni.ctl.get_cursor', Mock(return_value=None)): rows = query_member(None, None, None, None, 'SELECT pg_is_in_recovery()') - assert 'No connection to' in str(rows) + self.assertTrue('No connection to' in str(rows)) rows = query_member(None, None, None, 'replica', 'SELECT pg_is_in_recovery()') - assert 'No connection to' in str(rows) + self.assertTrue('No connection to' in str(rows)) - with patch('patroni.ctl.get_cursor', Mock(side_effect=psycopg2.OperationalError('bla'))): + with patch('patroni.ctl.get_cursor', Mock(side_effect=OperationalError('bla'))): rows = query_member(None, None, None, 'replica', 'SELECT pg_is_in_recovery()') - with patch('test_postgresql.MockCursor.execute', Mock(side_effect=psycopg2.OperationalError('bla'))): + with patch('test_postgresql.MockCursor.execute', Mock(side_effect=OperationalError('bla'))): rows = query_member(None, None, None, 'replica', 'SELECT pg_is_in_recovery()') @patch('patroni.dcs.AbstractDCS.get_cluster', Mock(return_value=get_cluster_initialized_with_leader())) def test_dsn(self): - runner = CliRunner() - with patch('patroni.ctl.get_dcs', Mock(return_value=self.e)): - result = runner.invoke(ctl, ['dsn', 'alpha', '--dcs', '8.8.8.8']) + result = self.runner.invoke(ctl, ['dsn', 'alpha', '--dcs', '8.8.8.8']) assert 'host=127.0.0.1 port=5435' in result.output - result = runner.invoke(ctl, [ + # Mutually exclusive options + result = self.runner.invoke(ctl, [ 'dsn', 'alpha', '--role', @@ -243,26 +257,29 @@ y''') '--member', 'dummy', ]) - assert 'mutually exclusive' in str(result.exception) + assert result.exit_code == 1 - result = runner.invoke(ctl, ['dsn', 'alpha', '--member', 'dummy']) - assert 'Can not find' in str(result.exception) - - # result = runner.invoke(ctl, ['dsn', 'alpha', '--dcs', '8.8.8.8', '--role', 'replica']) - # assert 'host=127.0.0.1 port=5436' in result.output + # Non-existing member + result = self.runner.invoke(ctl, ['dsn', 'alpha', '--member', 'dummy']) + assert result.exit_code == 1 @patch('patroni.etcd.Etcd.get_cluster', Mock(return_value=get_cluster_initialized_with_leader())) @patch('patroni.etcd.Etcd.get_etcd_client', Mock(return_value=None)) @patch('requests.get', requests_get) @patch('requests.post', requests_get) def test_restart_reinit(self): - runner = CliRunner() + result = self.runner.invoke(ctl, ['restart', 'alpha', '--dcs', '8.8.8.8'], input='y') + assert result.exit_code == 0 - result = runner.invoke(ctl, ['restart', 'alpha', '--dcs', '8.8.8.8'], input='y') - result = runner.invoke(ctl, ['reinit', 'alpha', '--dcs', '8.8.8.8'], input='y') + result = self.runner.invoke(ctl, ['reinit', 'alpha', '--dcs', '8.8.8.8'], input='y') + assert result.exit_code == 1 - result = runner.invoke(ctl, ['restart', 'alpha', '--dcs', '8.8.8.8'], input='N') - result = runner.invoke(ctl, [ + # Aborted restart + result = self.runner.invoke(ctl, ['restart', 'alpha', '--dcs', '8.8.8.8'], input='N') + assert result.exit_code == 1 + + # Not a member + result = self.runner.invoke(ctl, [ 'restart', 'alpha', '--dcs', @@ -270,100 +287,93 @@ y''') 'dummy', '--any', ], input='y') - assert 'not a member' in str(result.exception) + assert result.exit_code == 1 with patch('requests.post', Mock(return_value=MockResponse())): - result = runner.invoke(ctl, ['restart', 'alpha', '--dcs', '8.8.8.8'], input='y') + result = self.runner.invoke(ctl, ['restart', 'alpha', '--dcs', '8.8.8.8'], input='y') @patch('patroni.etcd.Etcd.get_cluster', Mock(return_value=get_cluster_initialized_with_leader())) @patch('patroni.etcd.Etcd.get_etcd_client', Mock(return_value=None)) def test_remove(self): - runner = CliRunner() - - result = runner.invoke(ctl, ['remove', 'alpha', '--dcs', '8.8.8.8'], input='alpha\nslave') + result = self.runner.invoke(ctl, ['remove', 'alpha', '--dcs', '8.8.8.8'], input='alpha\nslave') assert 'Please confirm' in result.output assert 'You are about to remove all' in result.output - assert 'You did not exactly type' in str(result.exception) + # Not typing an exact confirmation + assert result.exit_code == 1 - result = runner.invoke(ctl, ['remove', 'alpha', '--dcs', '8.8.8.8'], input='''alpha + # master specified does not match master of cluster + result = self.runner.invoke(ctl, ['remove', 'alpha', '--dcs', '8.8.8.8'], input='''alpha Yes I am aware slave''') - assert 'You did not specify the current master of the cluster' in str(result.exception) + assert result.exit_code == 1 - result = runner.invoke(ctl, ['remove', 'alpha', '--dcs', '8.8.8.8'], input='beta\nleader') - assert 'Cluster names specified do not match' in str(result.exception) + # cluster specified on cmdline does not match verification prompt + result = self.runner.invoke(ctl, ['remove', 'alpha', '--dcs', '8.8.8.8'], input='beta\nleader') + assert result.exit_code == 1 with patch('patroni.etcd.Etcd.get_cluster', get_cluster_initialized_with_leader): - result = runner.invoke(ctl, ['remove', 'alpha', '--dcs', '8.8.8.8'], - input='''alpha + result = self.runner.invoke(ctl, ['remove', 'alpha', '--dcs', '8.8.8.8'], + input='''alpha Yes I am aware leader''') assert 'object has no attribute' in str(result.exception) with patch('patroni.ctl.get_dcs', Mock(return_value=Mock())): - result = runner.invoke(ctl, ['remove', 'alpha', '--dcs', '8.8.8.8'], - input='''alpha + # Not implemented DCS + result = self.runner.invoke(ctl, ['remove', 'alpha', '--dcs', '8.8.8.8'], input='''alpha Yes I am aware leader''') - assert 'We have not implemented this for DCS of type' in str(result.exception) + assert result.exit_code == 1 @patch('patroni.etcd.Etcd.watch', Mock(return_value=None)) @patch('patroni.etcd.Etcd.get_cluster', Mock(return_value=get_cluster_initialized_with_leader())) def test_wait_for_leader(self): dcs = self.e - self.assertRaises(patroni.exceptions.PatroniCtlException, wait_for_leader, dcs, 0) + self.assertRaises(PatroniCtlException, wait_for_leader, dcs, 0) cluster = wait_for_leader(dcs=dcs, timeout=2) assert cluster.leader.member.name == 'leader' def test_post_patroni(self): - member = get_cluster_initialized_with_leader().leader.member - self.assertRaises(requests.exceptions.ConnectionError, post_patroni, member, 'dummy', {}) + with patch('requests.post', MagicMock(side_effect=requests.exceptions.ConnectionError('foo'))): + member = get_cluster_initialized_with_leader().leader.member + self.assertRaises(requests.exceptions.ConnectionError, post_patroni, member, 'dummy', {}) def test_ctl(self): - runner = CliRunner() + self.runner.invoke(ctl, ['list']) - runner.invoke(ctl, ['list']) - - result = runner.invoke(ctl, ['--help']) + result = self.runner.invoke(ctl, ['--help']) assert 'Usage:' in result.output def test_get_any_member(self): - m = get_any_member(get_cluster_initialized_without_leader(), role='master') - assert m is None + self.assertIsNone(get_any_member(get_cluster_initialized_without_leader(), role='master')) m = get_any_member(get_cluster_initialized_with_leader(), role='master') - assert m.name == 'leader' + self.assertEquals(m.name, 'leader') def test_get_all_members(self): - r = list(get_all_members(get_cluster_initialized_without_leader(), role='master')) - assert len(r) == 0 + self.assertEquals(list(get_all_members(get_cluster_initialized_without_leader(), role='master')), []) r = list(get_all_members(get_cluster_initialized_with_leader(), role='master')) - assert len(r) == 1 - assert r[0].name == 'leader' + self.assertEquals(len(r), 1) + self.assertEquals(r[0].name, 'leader') r = list(get_all_members(get_cluster_initialized_with_leader(), role='replica')) - assert len(r) == 1 - assert r[0].name == 'other' + self.assertEquals(len(r), 1) + self.assertEquals(r[0].name, 'other') - r = list(get_all_members(get_cluster_initialized_without_leader(), role='replica')) - assert len(r) == 2 + self.assertEquals(len(list(get_all_members(get_cluster_initialized_without_leader(), role='replica'))), 2) @patch('patroni.etcd.Etcd.get_cluster', Mock(return_value=get_cluster_initialized_with_leader())) @patch('patroni.etcd.Etcd.get_etcd_client', Mock(return_value=None)) @patch('requests.get', requests_get) @patch('requests.post', requests_get) def test_members(self): - runner = CliRunner() - - result = runner.invoke(members, ['alpha']) + result = self.runner.invoke(members, ['alpha']) assert result.exit_code == 0 def test_configure(self): - runner = CliRunner() - - result = runner.invoke(configure, [ + result = self.runner.invoke(configure, [ '--dcs', 'abc', '-c', @@ -373,5 +383,3 @@ leader''') ]) assert result.exit_code == 0 - - diff --git a/tests/test_etcd.py b/tests/test_etcd.py index 443dc2d8..601a0cd6 100644 --- a/tests/test_etcd.py +++ b/tests/test_etcd.py @@ -11,7 +11,7 @@ from patroni.dcs import Cluster, DCSError, Leader from patroni.etcd import Client, Etcd, EtcdError -class MockResponse: +class MockResponse(object): def __init__(self): self.status_code = 200 @@ -34,6 +34,7 @@ class MockResponse: def status(self): return self.status_code + @staticmethod def getheader(*args): return '' @@ -43,7 +44,8 @@ class MockPostgresql(Mock): server_version = '999999' scope = 'dummy' - def last_operation(self): + @staticmethod + def last_operation(): return '0' @@ -56,10 +58,7 @@ def requests_get(url, **kwargs): elif ':8011/patroni' in url: response.content = '{"role": "replica", "xlog": {"replayed_location": 0}, "tags": {}}' elif url.endswith('/members'): - if url.startswith('http://error'): - response.content = '[{}]' - else: - response.content = members + response.content = '[{}]' if url.startswith('http://error') else members elif url.startswith('http://exhibitor'): response.content = '{"servers":["127.0.0.1","127.0.0.2","127.0.0.3"],"port":2181}' else: @@ -84,9 +83,9 @@ def etcd_watch(key, index=None, timeout=None, recursive=None): def etcd_write(key, value, **kwargs): if key == '/service/exists/leader': raise etcd.EtcdAlreadyExist - if key == '/service/test/leader' or key == '/patroni/test/leader': - if kwargs.get('prevValue', None) == 'foo' or not kwargs.get('prevExist', True): - return True + if key in ['/service/test/leader', '/patroni/test/leader'] and \ + (kwargs.get('prevValue') == 'foo' or not kwargs.get('prevExist', True)): + return True raise etcd.EtcdException @@ -127,12 +126,12 @@ class SleepException(Exception): pass -class MockSRV: +class MockSRV(object): port = 2380 target = '127.0.0.1' -def dns_query(name, type): +def dns_query(name, _): if name == '_etcd-server._tcp.blabla': return [] elif name == '_etcd-server._tcp.exception': @@ -174,6 +173,7 @@ class TestClient(unittest.TestCase): self.client._machines_cache = [] self.assertRaises(etcd.EtcdConnectionFailed, self.client.api_execute, '/', 'GET') self.assertTrue(self.client._update_machines_cache) + self.assertRaises(etcd.EtcdException, self.client.api_execute, '/', 'GET') def test_get_srv_record(self): self.assertEquals(self.client.get_srv_record('blabla'), []) diff --git a/tests/test_ha.py b/tests/test_ha.py index 9ff6547d..a65a444a 100644 --- a/tests/test_ha.py +++ b/tests/test_ha.py @@ -1,11 +1,14 @@ -import etcd import unittest +import datetime +import pytz +from etcd import EtcdException from mock import Mock, MagicMock, patch from patroni.dcs import Cluster, Failover, Leader, Member from patroni.etcd import Client, Etcd from patroni.exceptions import DCSError, PostgresException from patroni.ha import Ha +from patroni.postgresql import Postgresql from test_etcd import socket_getaddrinfo, etcd_read, etcd_write, requests_get @@ -18,7 +21,7 @@ def false(*args, **kwargs): def get_cluster(initialize, leader, members, failover): - return Cluster(initialize, leader, None, members, failover) + return Cluster(initialize, leader, 10, members, failover) def get_cluster_not_initialized_without_leader(): @@ -27,7 +30,7 @@ def get_cluster_not_initialized_without_leader(): def get_cluster_initialized_without_leader(leader=False, failover=None): m = Member(0, 'leader', 28, {'conn_url': 'postgres://replicator:rep-pass@127.0.0.1:5435/postgres', - 'api_url': 'http://127.0.0.1:8008/patroni', 'xlog_location':4}) + 'api_url': 'http://127.0.0.1:8008/patroni', 'xlog_location': 4}) l = Leader(0, 0, m) if leader else None o = Member(0, 'other', 28, {'conn_url': 'postgres://replicator:rep-pass@127.0.0.1:5436/postgres', 'api_url': 'http://127.0.0.1:8011/patroni'}) @@ -37,52 +40,13 @@ def get_cluster_initialized_without_leader(leader=False, failover=None): def get_cluster_initialized_with_leader(failover=None): return get_cluster_initialized_without_leader(leader=True, failover=failover) + def get_cluster_initialized_with_only_leader(failover=None): l = get_cluster_initialized_without_leader(leader=True, failover=failover).leader return get_cluster(True, l, [l], failover) -class MockPostgresql(Mock): - - name = 'postgresql0' - role = 'replica' - state = 'running' - connection_string = 'postgres://foo@bar/postgres' - server_version = '999999' - scope = 'dummy' - - def is_healthy(self): - return True - - def start(self): - return True - - def is_healthiest_node(self, members): - return True - - def is_leader(self): - return True - - def xlog_position(self): - return 0 - - def last_operation(self): - return 0 - - def data_directory_empty(self): - return False - - def bootstrap(self, *args, **kwargs): - return True - - def check_replication_lag(self, last_leader_operation): - return True - - def check_recovery_conf(self, leader): - return False - - -class MockPatroni: +class MockPatroni(object): def __init__(self, p, d): self.postgresql = p @@ -90,29 +54,48 @@ class MockPatroni: self.api = Mock() self.tags = {} self.nofailover = None + self.replicatefrom = None self.api.connection_string = 'http://127.0.0.1:8008' def run_async(func, args=()): - func(*args) if args else func() + return func(*args) if args else func() +@patch.object(Postgresql, 'is_running', Mock(return_value=True)) +@patch.object(Postgresql, 'is_leader', Mock(return_value=True)) +@patch.object(Postgresql, 'xlog_position', Mock(return_value=0)) +@patch.object(Postgresql, 'call_nowait', Mock(return_value=True)) +@patch.object(Postgresql, 'data_directory_empty', Mock(return_value=False)) +@patch.object(Postgresql, 'controldata', Mock(return_value={'Database system identifier': '1234567890'})) +@patch.object(Postgresql, 'sync_replication_slots', Mock()) +@patch.object(Postgresql, 'write_pg_hba', Mock()) +@patch.object(Postgresql, 'write_pgpass', Mock()) +@patch.object(Postgresql, 'write_recovery_conf', Mock()) +@patch.object(Postgresql, 'query', Mock()) +@patch.object(Postgresql, 'checkpoint', Mock()) +@patch('subprocess.call', Mock(return_value=0)) class TestHa(unittest.TestCase): @patch('socket.getaddrinfo', socket_getaddrinfo) - @patch.object(Client, 'machines') - def setUp(self, mock_machines): - mock_machines.__get__ = Mock(return_value=['http://remotehost:2379']) - self.p = MockPostgresql() - self.e = Etcd('foo', {'ttl': 30, 'host': 'ok:2379', 'scope': 'test'}) - self.e.client.read = etcd_read - self.e.client.write = etcd_write - self.e.client.delete = Mock(side_effect=etcd.EtcdException()) - self.ha = Ha(MockPatroni(self.p, self.e)) - self.ha._async_executor.run_async = run_async - self.ha.old_cluster = self.e.get_cluster() - self.ha.cluster = get_cluster_not_initialized_without_leader() - self.ha.load_cluster_from_dcs = Mock() + def setUp(self): + with patch.object(Client, 'machines') as mock_machines: + mock_machines.__get__ = Mock(return_value=['http://remotehost:2379']) + self.p = Postgresql({'name': 'postgresql0', 'scope': 'dummy', 'listen': '127.0.0.1:5432', + 'data_dir': 'data/postgresql0', 'superuser': {}, 'admin': {}, + 'replication': {'username': '', 'password': '', 'network': ''}}) + self.p.set_state('running') + self.p.check_replication_lag = true + self.p.can_create_replica_without_leader = MagicMock(return_value=False) + self.e = Etcd('foo', {'ttl': 30, 'host': 'ok:2379', 'scope': 'test'}) + self.e.client.read = etcd_read + self.e.client.write = etcd_write + self.e.client.delete = Mock(side_effect=EtcdException()) + self.ha = Ha(MockPatroni(self.p, self.e)) + self.ha._async_executor.run_async = run_async + self.ha.old_cluster = self.e.get_cluster() + self.ha.cluster = get_cluster_not_initialized_without_leader() + self.ha.load_cluster_from_dcs = Mock() def test_update_lock(self): self.p.last_operation = Mock(side_effect=PostgresException('')) @@ -129,13 +112,19 @@ class TestHa(unittest.TestCase): def test_recover_replica_failed(self): self.p.controldata = lambda: {'Database cluster state': 'in production'} self.p.is_healthy = false - self.p.follow_the_leader = false + self.p.is_running = false + self.p.follow = false + self.assertEquals(self.ha.run_cycle(), 'started as a secondary') self.assertEquals(self.ha.run_cycle(), 'failed to start postgres') def test_recover_master_failed(self): - self.p.follow_the_leader = false + self.p.follow = false self.p.is_healthy = false + self.p.is_running = false self.ha.has_lock = true + self.p.set_role('master') + self.p.controldata = lambda: {'Database cluster state': 'in production'} + self.assertEquals(self.ha.run_cycle(), 'started as readonly because i had the session lock') self.assertEquals(self.ha.run_cycle(), 'removed leader key after trying and failing to start postgres') @patch('sys.exit', return_value=1) @@ -146,7 +135,8 @@ class TestHa(unittest.TestCase): @patch.object(Cluster, 'is_unlocked', Mock(return_value=False)) def test_start_as_readonly(self): - self.p.is_leader = self.p.is_healthy = false + self.p.is_leader = false + self.p.is_healthy = true self.ha.has_lock = true self.assertEquals(self.ha.run_cycle(), 'promoted self to leader because i had the session lock') @@ -160,7 +150,7 @@ class TestHa(unittest.TestCase): def test_demote_after_failing_to_obtain_lock(self): self.ha.acquire_lock = false - self.assertEquals(self.ha.run_cycle(), 'demoted self due after trying and failing to obtain lock') + self.assertEquals(self.ha.run_cycle(), 'demoted self after trying and failing to obtain lock') def test_follow_new_leader_after_failing_to_obtain_lock(self): self.ha.is_healthiest_node = true @@ -198,10 +188,12 @@ class TestHa(unittest.TestCase): self.ha.update_lock = false self.assertEquals(self.ha.run_cycle(), 'demoting self because i do not have the lock and i was a leader') - def test_follow_the_leader(self): + def test_follow(self): self.ha.cluster.is_unlocked = false self.p.is_leader = false self.assertEquals(self.ha.run_cycle(), 'no action. i am a secondary and i am following a leader') + self.ha.patroni.replicatefrom = "foo" + self.assertEquals(self.ha.run_cycle(), 'no action. i am a secondary and i am following a leader') def test_no_etcd_connection_master_demote(self): self.ha.load_cluster_from_dcs = Mock(side_effect=DCSError('Etcd is not responding properly')) @@ -216,6 +208,11 @@ class TestHa(unittest.TestCase): self.ha.cluster = get_cluster_initialized_without_leader() self.assertEquals(self.ha.bootstrap(), 'waiting for leader to bootstrap') + def test_bootstrap_without_leader(self): + self.ha.cluster = get_cluster_initialized_without_leader() + self.p.can_create_replica_without_leader = MagicMock(return_value=True) + self.assertEquals(self.ha.bootstrap(), "trying to bootstrap without leader") + def test_bootstrap_initialize_lock_failed(self): self.ha.cluster = get_cluster_not_initialized_without_leader() self.assertEquals(self.ha.bootstrap(), 'failed to acquire initialize lock') @@ -271,38 +268,62 @@ class TestHa(unittest.TestCase): @patch('requests.get', requests_get) def test_manual_failover_from_leader(self): self.ha.has_lock = true - self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', '')) + self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', '', None)) self.assertEquals(self.ha.run_cycle(), 'no action. i am the leader with the lock') - self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', MockPostgresql.name)) + self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', self.p.name, None)) self.assertEquals(self.ha.run_cycle(), 'no action. i am the leader with the lock') - self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'blabla')) + self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, '', 'blabla', None)) self.assertEquals(self.ha.run_cycle(), 'no action. i am the leader with the lock') - f = Failover(0, MockPostgresql.name, '') + f = Failover(0, self.p.name, '', None) self.ha.cluster = get_cluster_initialized_with_leader(f) self.assertEquals(self.ha.run_cycle(), 'manual failover: demoting myself') self.ha.fetch_node_status = lambda e: (e, True, True, 0, {'nofailover': 'True'}) self.assertEquals(self.ha.run_cycle(), 'no action. i am the leader with the lock') # manual failover from the previous leader to us won't happen if we hold the nofailover flag - self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', MockPostgresql.name)) + self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, None)) self.assertEquals(self.ha.run_cycle(), 'no action. i am the leader with the lock') + # Failover scheduled time must include timezone + scheduled = datetime.datetime.now() + self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled)) + self.ha.run_cycle() + + scheduled = datetime.datetime.utcnow().replace(tzinfo=pytz.UTC) + self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled)) + self.assertEquals('no action. i am the leader with the lock', self.ha.run_cycle()) + + scheduled = scheduled + datetime.timedelta(seconds=30) + self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled)) + self.assertEquals('no action. i am the leader with the lock', self.ha.run_cycle()) + + scheduled = scheduled + datetime.timedelta(seconds=-600) + self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled)) + self.assertEquals('no action. i am the leader with the lock', self.ha.run_cycle()) + + scheduled = None + self.ha.cluster = get_cluster_initialized_with_leader(Failover(0, 'blabla', self.p.name, scheduled)) + self.assertEquals('no action. i am the leader with the lock', self.ha.run_cycle()) + @patch('requests.get', requests_get) def test_manual_failover_process_no_leader(self): self.p.is_leader = false - self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', MockPostgresql.name)) + self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', self.p.name, None)) self.assertEquals(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock') - self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'leader')) + self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'leader', None)) + self.p.set_role('replica') self.assertEquals(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock') self.ha.fetch_node_status = lambda e: (e, True, True, 0, {}) # accessible, in_recovery self.assertEquals(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node') - self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, MockPostgresql.name, '')) + self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, self.p.name, '', None)) self.assertEquals(self.ha.run_cycle(), 'following a different leader because i am not the healthiest node') self.ha.fetch_node_status = lambda e: (e, False, True, 0, {}) # inaccessible, in_recovery + self.p.set_role('replica') self.assertEquals(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock') # set failover flag to True for all members of the cluster # this should elect the current member, as we are not going to call the API for it. - self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other')) + self.ha.cluster = get_cluster_initialized_without_leader(failover=Failover(0, '', 'other', None)) self.ha.fetch_node_status = lambda e: (e, True, True, 0, {'nofailover': 'True'}) # accessible, in_recovery + self.p.set_role('replica') self.assertEquals(self.ha.run_cycle(), 'promoted self to leader by acquiring session lock') # same as previous, but set the current member to nofailover. In no case it should be elected as a leader self.ha.patroni.nofailover = True @@ -335,3 +356,12 @@ class TestHa(unittest.TestCase): self.ha.fetch_node_status(member) member = Member(0, 'test', 1, {'api_url': 'http://localhost:8011/patroni'}) self.ha.fetch_node_status(member) + + def test_post_recover(self): + self.p.is_running = false + self.ha.has_lock = true + self.assertEqual(self.ha.post_recover(), 'removed leader key after trying and failing to start postgres') + self.ha.has_lock = false + self.assertEqual(self.ha.post_recover(), 'failed to start postgres') + self.p.is_running = true + self.assertIsNone(self.ha.post_recover()) diff --git a/tests/test_patroni.py b/tests/test_patroni.py index 18f5c14b..2698de58 100644 --- a/tests/test_patroni.py +++ b/tests/test_patroni.py @@ -7,7 +7,7 @@ from mock import Mock, patch from patroni.api import RestApiServer from patroni.async_executor import AsyncExecutor from patroni.etcd import Etcd -from patroni import Patroni, main +from patroni import Patroni, main as _main from patroni.zookeeper import ZooKeeper from six.moves import BaseHTTPServer from test_etcd import Client, SleepException, etcd_read, etcd_write @@ -15,10 +15,6 @@ from test_postgresql import Postgresql, psycopg2_connect from test_zookeeper import MockKazooClient -def time_sleep(*args): - raise SleepException() - - @patch('time.sleep', Mock()) @patch('subprocess.call', Mock(return_value=0)) @patch('psycopg2.connect', psycopg2_connect) @@ -28,19 +24,19 @@ def time_sleep(*args): @patch.object(AsyncExecutor, 'run', Mock()) class TestPatroni(unittest.TestCase): - @patch.object(Client, 'machines') - def setUp(self, mock_machines): - mock_machines.__get__ = Mock(return_value=['http://remotehost:2379']) - self.touched = False - self.init_cancelled = False - RestApiServer._BaseServer__is_shut_down = Mock() - RestApiServer._BaseServer__shutdown_request = True - RestApiServer.socket = 0 - with open('postgres0.yml', 'r') as f: - config = yaml.load(f) - self.p = Patroni(config) - self.p.ha.dcs.client.write = etcd_write - self.p.ha.dcs.client.read = etcd_read + def setUp(self): + with patch.object(Client, 'machines') as mock_machines: + mock_machines.__get__ = Mock(return_value=['http://remotehost:2379']) + self.touched = False + self.init_cancelled = False + RestApiServer._BaseServer__is_shut_down = Mock() + RestApiServer._BaseServer__shutdown_request = True + RestApiServer.socket = 0 + with open('postgres0.yml', 'r') as f: + config = yaml.load(f) + self.p = Patroni(config) + self.p.ha.dcs.client.write = etcd_write + self.p.ha.dcs.client.read = etcd_read @patch('patroni.zookeeper.KazooClient', MockKazooClient()) def test_get_dcs(self): @@ -51,18 +47,18 @@ class TestPatroni(unittest.TestCase): @patch.object(Etcd, 'delete_leader', Mock()) @patch.object(Client, 'machines') def test_patroni_main(self, mock_machines): - main() + _main() sys.argv = ['patroni.py', 'postgres0.yml'] mock_machines.__get__ = Mock(return_value=['http://remotehost:2379']) with patch.object(Patroni, 'run', Mock(side_effect=SleepException())): - self.assertRaises(SleepException, main) + self.assertRaises(SleepException, _main) with patch.object(Patroni, 'run', Mock(side_effect=KeyboardInterrupt())): - main() + _main() @patch('time.sleep', Mock(side_effect=SleepException())) def test_run(self): - self.p.ha.dcs.watch = time_sleep + self.p.ha.dcs.watch = Mock(side_effect=SleepException()) self.assertRaises(SleepException, self.p.run) self.p.ha.state_handler.is_leader = Mock(return_value=False) @@ -80,3 +76,8 @@ class TestPatroni(unittest.TestCase): self.assertTrue(self.p.nofailover) self.p.tags['nofailover'] = None self.assertFalse(self.p.nofailover) + + def test_replicatefrom(self): + self.assertIsNone(self.p.replicatefrom) + self.p.tags['replicatefrom'] = 'foo' + self.assertEqual(self.p.replicatefrom, 'foo') diff --git a/tests/test_postgresql.py b/tests/test_postgresql.py index 2097bcb7..5558437b 100644 --- a/tests/test_postgresql.py +++ b/tests/test_postgresql.py @@ -2,24 +2,19 @@ import mock # for the mock.call method, importing it without a namespace breaks import os import psycopg2 import shutil +import subprocess import unittest -from six.moves import builtins from mock import Mock, MagicMock, PropertyMock, patch, mock_open from patroni.dcs import Cluster, Leader, Member from patroni.exceptions import PostgresException, PostgresConnectionException from patroni.postgresql import Postgresql from patroni.utils import RetryFailedError +from six.moves import builtins from test_ha import false -import subprocess -def is_file_raise_on_backup(*args, **kwargs): - if args[0].endswith('.backup'): - raise Exception("foo") - - -class MockCursor: +class MockCursor(object): def __init__(self, connection): self.connection = connection @@ -59,7 +54,8 @@ class MockCursor: def fetchall(self): return self.results - def close(self): + @staticmethod + def close(): pass def __iter__(self): @@ -154,9 +150,12 @@ def psycopg2_connect(*args, **kwargs): return MockConnect() +def fake_listdir(path): + return ["a", "b", "c"] if path.endswith('pg_xlog/archive_status') else [] + + @patch('subprocess.call', Mock(return_value=0)) @patch('psycopg2.connect', psycopg2_connect) -@patch('shutil.copy', Mock()) class TestPostgresql(unittest.TestCase): @patch('subprocess.call', Mock(return_value=0)) @@ -165,7 +164,7 @@ class TestPostgresql(unittest.TestCase): self.p = Postgresql({'name': 'test0', 'scope': 'batman', 'data_dir': 'data/test0', 'listen': '127.0.0.1, *:5432', 'connect_address': '127.0.0.2:5432', 'pg_hba': ['hostssl all all 0.0.0.0/0 md5', 'host all all 0.0.0.0/0 md5'], - 'superuser': {'password': 'test'}, + 'superuser': {'username': 'test', 'password': 'test'}, 'admin': {'username': 'admin', 'password': 'admin'}, 'pg_rewind': {'username': 'admin', 'password': 'admin'}, 'replication': {'username': 'replicator', @@ -181,7 +180,8 @@ class TestPostgresql(unittest.TestCase): os.makedirs(self.p.data_dir) self.leadermem = Member(0, 'leader', 28, {'conn_url': 'postgres://replicator:rep-pass@127.0.0.1:5435/postgres'}) self.leader = Leader(-1, 28, self.leadermem) - self.other = Member(0, 'test1', 28, {'conn_url': 'postgres://replicator:rep-pass@127.0.0.1:5433/postgres'}) + self.other = Member(0, 'test1', 28, {'conn_url': 'postgres://replicator:rep-pass@127.0.0.1:5433/postgres', + 'tags': {'replicatefrom': 'leader'}}) self.me = Member(0, 'test0', 28, {'conn_url': 'postgres://replicator:rep-pass@127.0.0.1:5434/postgres'}) def tearDown(self): @@ -204,6 +204,11 @@ class TestPostgresql(unittest.TestCase): self.assertTrue(self.p.initialize()) self.assertTrue(os.path.exists(os.path.join(self.p.data_dir, 'pg_hba.conf'))) + @patch('os.path.exists', Mock(return_value=True)) + @patch('os.unlink', Mock()) + def test_delete_trigger_file(self): + self.p.delete_trigger_file() + def test_start(self): self.assertTrue(self.p.start()) self.p.is_running = false @@ -228,10 +233,12 @@ class TestPostgresql(unittest.TestCase): self.p.write_pgpass({'host': 'localhost', 'port': '5432', 'user': 'foo', 'password': 'bar'}) @patch('patroni.postgresql.Postgresql.write_pgpass', MagicMock(return_value=dict())) - def test_sync_from_leader(self): - self.assertTrue(self.p.sync_from_leader(self.leader)) + def test_sync_replica(self): + self.assertTrue(self.p.sync_replica(self.leader)) + self.p.create_replica = Mock(return_value=1) + self.assertFalse(self.p.sync_replica(self.leader)) - @patch('subprocess.call', side_effect=Exception("Test")) + @patch('subprocess.call', side_effect=OSError) @patch('patroni.postgresql.Postgresql.write_pgpass', MagicMock(return_value=dict())) def test_pg_rewind(self, mock_call): self.assertTrue(self.p.rewind(self.leader)) @@ -242,26 +249,28 @@ class TestPostgresql(unittest.TestCase): @patch('patroni.postgresql.Postgresql.remove_data_directory', MagicMock(return_value=True)) @patch('patroni.postgresql.Postgresql.single_user_mode', MagicMock(return_value=1)) @patch('patroni.postgresql.Postgresql.write_pgpass', MagicMock(return_value=dict())) - def test_follow_the_leader(self, mock_pg_rewind): - self.p.demote() - self.p.follow_the_leader(None) - self.p.demote() - self.p.follow_the_leader(self.leader) - self.p.follow_the_leader(Leader(-1, 28, self.other)) + @patch('subprocess.check_output', Mock(return_value=0, side_effect=pg_controldata_string)) + def test_follow(self, mock_pg_rewind): + self.p.follow(None) + self.p.follow(self.leader) + self.p.follow(Leader(-1, 28, self.other)) self.p.rewind = mock_pg_rewind - self.p.follow_the_leader(self.leader) + self.p.follow(self.leader) self.p.require_rewind() with mock.patch('os.path.islink', MagicMock(return_value=True)): with mock.patch('patroni.postgresql.Postgresql.can_rewind', new_callable=PropertyMock(return_value=True)): with mock.patch('os.unlink', MagicMock(return_value=True)): - self.p.follow_the_leader(self.leader, recovery=True) + self.p.follow(self.leader, recovery=True) self.p.require_rewind() with mock.patch('patroni.postgresql.Postgresql.can_rewind', new_callable=PropertyMock(return_value=True)): self.p.rewind.return_value = True - self.p.follow_the_leader(self.leader, recovery=True) + self.p.follow(self.leader, recovery=True) self.p.rewind.return_value = False - self.p.follow_the_leader(self.leader, recovery=True) + self.p.follow(self.leader, recovery=True) + with mock.patch('patroni.postgresql.Postgresql.check_recovery_conf', MagicMock(return_value=True)): + self.assertTrue(self.p.follow(None)) + @patch('subprocess.check_output', Mock(return_value=0, side_effect=pg_controldata_string)) def test_can_rewind(self): tmp = self.p.pg_rewind self.p.pg_rewind = None @@ -269,16 +278,16 @@ class TestPostgresql(unittest.TestCase): self.p.pg_rewind = tmp with mock.patch('subprocess.call', MagicMock(return_value=1)): self.assertFalse(self.p.can_rewind) - with mock.patch('subprocess.call', side_effect=OSError("foo")): + with mock.patch('subprocess.call', side_effect=OSError): self.assertFalse(self.p.can_rewind) - tmp = self.p.controldata() + tmp = self.p.controldata self.p.controldata = lambda: {'wal_log_hints setting': 'on'} self.assertTrue(self.p.can_rewind) self.p.controldata = tmp @patch('time.sleep', Mock()) def test_create_replica(self): - self.p.delete_trigger_file = Mock(side_effect=OSError()) + self.p.delete_trigger_file = Mock(side_effect=OSError) with patch('subprocess.call', Mock(side_effect=[1, 0])): self.assertEquals(self.p.create_replica(self.leader, ''), 0) with patch('subprocess.call', Mock(side_effect=[Exception(), 0])): @@ -294,12 +303,6 @@ class TestPostgresql(unittest.TestCase): with patch('subprocess.call', Mock(side_effect=Exception("foo"))): self.assertEquals(self.p.create_replica(self.leader, ''), 1) - def test_create_connection_users(self): - cfg = self.p.config - cfg['superuser']['username'] = 'test' - p = Postgresql(cfg) - p.create_connection_users() - def test_sync_replication_slots(self): self.p.start() cluster = Cluster(True, self.leader, 0, [self.me, self.other, self.leadermem], None) @@ -307,6 +310,9 @@ class TestPostgresql(unittest.TestCase): self.p.query = Mock(side_effect=psycopg2.OperationalError) self.p.schedule_load_slots = True self.p.sync_replication_slots(cluster) + self.p.schedule_load_slots = False + with mock.patch('patroni.postgresql.Postgresql.role', new_callable=PropertyMock(return_value='replica')): + self.p.sync_replication_slots(cluster) @patch.object(MockConnect, 'closed', 2) def test__query(self): @@ -338,7 +344,7 @@ class TestPostgresql(unittest.TestCase): def test_last_operation(self): self.assertEquals(self.p.last_operation(), '0') - @patch('subprocess.Popen', Mock(side_effect=OSError())) + @patch('subprocess.Popen', Mock(side_effect=OSError)) def test_call_nowait(self): self.assertFalse(self.p.call_nowait('on_start')) @@ -358,7 +364,7 @@ class TestPostgresql(unittest.TestCase): def test_move_data_directory(self): self.p.is_running = false self.p.move_data_directory() - with patch('os.rename', Mock(side_effect=OSError())): + with patch('os.rename', Mock(side_effect=OSError)): self.p.move_data_directory() @patch('patroni.postgresql.Postgresql.write_pgpass', MagicMock(return_value=dict())) @@ -366,7 +372,8 @@ class TestPostgresql(unittest.TestCase): with patch('subprocess.call', Mock(return_value=1)): self.assertRaises(PostgresException, self.p.bootstrap) self.p.bootstrap() - self.p.bootstrap(self.leader) + with patch('patroni.postgresql.Postgresql.sync_replica', MagicMock(return_value=True)): + self.p.bootstrap(self.leader) def test_remove_data_directory(self): self.p.data_dir = 'data_dir' @@ -376,26 +383,20 @@ class TestPostgresql(unittest.TestCase): open(self.p.data_dir, 'w').close() self.p.remove_data_directory() os.symlink('unexisting', self.p.data_dir) - with patch('os.unlink', Mock(side_effect=Exception)): + with patch('os.unlink', Mock(side_effect=OSError)): self.p.remove_data_directory() self.p.remove_data_directory() - @patch('subprocess.check_output', MagicMock(return_value=0, side_effect=pg_controldata_string)) - @patch('subprocess.check_output', side_effect=subprocess.CalledProcessError) - @patch('subprocess.check_output', side_effect=Exception('Failed')) - def test_controldata(self, check_output_call_error, check_output_generic_exception): - data = self.p.controldata() - self.assertEquals(len(data), 50) - self.assertEquals(data['Database cluster state'], 'shut down in recovery') - self.assertEquals(data['wal_log_hints setting'], 'on') - self.assertEquals(int(data['Database block size']), 8192) + def test_controldata(self): + with patch('subprocess.check_output', Mock(return_value=0, side_effect=pg_controldata_string)): + data = self.p.controldata() + self.assertEquals(len(data), 50) + self.assertEquals(data['Database cluster state'], 'shut down in recovery') + self.assertEquals(data['wal_log_hints setting'], 'on') + self.assertEquals(int(data['Database block size']), 8192) - subprocess.check_output = check_output_call_error - data = self.p.controldata() - self.assertEquals(data, dict()) - - subprocess.check_output = check_output_generic_exception - self.assertRaises(Exception, self.p.controldata()) + with patch('subprocess.check_output', Mock(side_effect=subprocess.CalledProcessError(1, ''))): + self.assertEquals(self.p.controldata(), {}) def test_read_postmaster_opts(self): m = mock_open(read_data=postmaster_opts_string()) @@ -405,13 +406,10 @@ class TestPostgresql(unittest.TestCase): self.assertEquals(int(data['max_replication_slots']), 5) self.assertEqual(data.get('D'), None) - m.side_effect = IOError("foo") + m.side_effect = IOError data = self.p.read_postmaster_opts() self.assertEqual(data, dict()) - m.side_effect = Exception("foo") - self.assertRaises(Exception, self.p.read_postmaster_opts()) - @patch('subprocess.Popen') @patch.object(builtins, 'open', MagicMock(return_value=42)) def test_single_user_mode(self, subprocess_popen_mock): @@ -431,13 +429,7 @@ class TestPostgresql(unittest.TestCase): subprocess_popen_mock.return_value = None self.assertEquals(self.p.single_user_mode(), 1) - def fake_listdir(path): - if path.endswith(os.path.join('pg_xlog', 'archive_status')): - return ["a", "b", "c"] - return [] - @patch('os.listdir', MagicMock(side_effect=fake_listdir)) - @patch('os.path.isdir', MagicMock(return_value=True)) @patch('os.unlink', return_value=True) @patch('os.remove', return_value=True) @patch('os.path.islink', return_value=False) @@ -459,8 +451,8 @@ class TestPostgresql(unittest.TestCase): mock_unlink.reset_mock() mock_remove.reset_mock() - mock_file.side_effect = Exception("foo") - mock_link.side_effect = Exception("foo") + mock_file.side_effect = OSError + mock_link.side_effect = OSError self.p.cleanup_archive_status() mock_unlink.assert_not_called() mock_remove.assert_not_called() @@ -469,14 +461,27 @@ class TestPostgresql(unittest.TestCase): def test_sysid(self): self.assertEqual(self.p.sysid, "6200971513092291716") - @patch('os.path.isfile', MagicMock(return_value=True)) - @patch('shutil.copy', side_effect=Exception) - def test_save_configuration_files(self, mock_copy): - shutil.copy = mock_copy + @patch('os.path.isfile', Mock(return_value=True)) + @patch('shutil.copy', Mock(side_effect=IOError)) + def test_save_configuration_files(self): self.p.save_configuration_files() - @patch('os.path.isfile', MagicMock(side_effect=is_file_raise_on_backup)) - @patch('shutil.copy', side_effect=Exception) - def test_restore_configuration_files(self, mock_copy): - shutil.copy = mock_copy + @patch('os.path.isfile', Mock(side_effect=[False, True])) + @patch('shutil.copy', Mock(side_effect=IOError)) + def test_restore_configuration_files(self): self.p.restore_configuration_files() + + def test_can_create_replica_without_leader(self): + self.p.config['create_replica_method'] = [] + self.assertFalse(self.p.can_create_replica_without_leader()) + self.p.config['create_replica_method'] = ['wale', 'basebackup'] + self.p.config['wale'] = {'command': 'foo', 'no_master': 1} + self.assertTrue(self.p.can_create_replica_without_leader()) + + def test_replica_method_can_work_without_leader(self): + self.assertFalse(self.p.replica_method_can_work_without_leader('basebackup')) + self.assertFalse(self.p.replica_method_can_work_without_leader('foobar')) + self.p.config['foo'] = {'command': 'bar', 'no_master': 1} + self.assertTrue(self.p.replica_method_can_work_without_leader('foo')) + self.p.config['foo'] = {'command': 'bar'} + self.assertFalse(self.p.replica_method_can_work_without_leader('foo')) diff --git a/tests/test_utils.py b/tests/test_utils.py index 265f98ab..740fef03 100644 --- a/tests/test_utils.py +++ b/tests/test_utils.py @@ -16,20 +16,21 @@ class TestUtils(unittest.TestCase): @patch('time.sleep', Mock()) def test_reap_children(self): - reap_children() + self.assertIsNone(reap_children()) with patch('os.waitpid', Mock(return_value=(0, 0))): sigchld_handler(None, None) - reap_children() + self.assertIsNone(reap_children()) @patch('time.sleep', time_sleep) def test_sleep(self): - sleep(0.01) + self.assertIsNone(sleep(0.01)) @patch('time.sleep', Mock()) class TestRetrySleeper(unittest.TestCase): - def _fail(self, times=1): + @staticmethod + def _fail(times=1): scope = dict(times=0) def inner(): @@ -40,36 +41,33 @@ class TestRetrySleeper(unittest.TestCase): raise PatroniException('Failed!') return inner - def _makeOne(self, *args, **kwargs): - return Retry(*args, **kwargs) - def test_reset(self): - retry = self._makeOne(delay=0, max_tries=2) + retry = Retry(delay=0, max_tries=2) retry(self._fail()) self.assertEquals(retry._attempts, 1) retry.reset() self.assertEquals(retry._attempts, 0) def test_too_many_tries(self): - retry = self._makeOne(delay=0) + retry = Retry(delay=0) self.assertRaises(RetryFailedError, retry, self._fail(times=999)) self.assertEquals(retry._attempts, 1) def test_maximum_delay(self): - retry = self._makeOne(delay=10, max_tries=100) + retry = Retry(delay=10, max_tries=100) retry(self._fail(times=10)) self.assertTrue(retry._cur_delay < 4000, retry._cur_delay) # gevent's sleep function is picky about the type self.assertEquals(type(retry._cur_delay), float) def test_deadline(self): - retry = self._makeOne(deadline=0.0001) + retry = Retry(deadline=0.0001) self.assertRaises(RetryFailedError, retry, self._fail(times=100)) def test_copy(self): def _sleep(t): - None + pass - retry = self._makeOne(sleep_func=_sleep) + retry = Retry(sleep_func=_sleep) rcopy = retry.copy() self.assertTrue(rcopy.sleep_func is _sleep) diff --git a/tests/test_wale_restore.py b/tests/test_wale_restore.py index 05f34187..40761051 100644 --- a/tests/test_wale_restore.py +++ b/tests/test_wale_restore.py @@ -1,9 +1,9 @@ -import unittest -from mock import MagicMock, patch, PropertyMock -import os import psycopg2 import subprocess -from patroni.scripts.wale_restore import WALERestore +import unittest + +from mock import MagicMock, patch, PropertyMock +from patroni.scripts.wale_restore import WALERestore, main as _main def fake_cursor_fetchone(*args, **kwargs): @@ -28,16 +28,19 @@ def fake_backup_data(self, *args, **kwargs): base_00000001000000000000007F_00000040 2015-05-18T10:13:25.000Z 167772160 00000001000000000000007F 00000040 00000001000000000000007F 00000240 """ + def fake_backup_data_2(self, *args, **kwargs): """ return the fake result of WAL-E backup-list""" return """name last_modified expanded_size_bytes wal_segment_backup_start wal_segment_offset_backup_start wal_segment_backup_stop wal_segment_offset_backup_stop """ + def fake_backup_data_3(self, *args, **kwargs): """ return the fake result of WAL-E backup-list""" return """name last_modified expanded_size_bytes wal_segment_backup_start wal_segment_offset_backup_start wal_segment_backup_stop base_00000001000000000000007F_00000040 2015-05-18T10:13:25.000Z 167772160 00000001000000000000007F 00000040 00000001000000000000007F 00000240 """ + def fake_backup_data_4(self, *args, **kwargs): """ return the fake result of WAL-E backup-list""" return """name last_modified expanded_size_foo wal_segment_backup_start wal_segment_offset_backup_start wal_segment_backup_stop wal_segment_offset_backup_stop @@ -58,7 +61,7 @@ class TestWALERestore(unittest.TestCase): def setUp(self): self.wale_restore = WALERestore("batman", "/data", - "host=batman port=5432 user=batman", "/etc", 100, 100, 1) + "host=batman port=5432 user=batman", "/etc", 100, 100, 1, 0) def tearDown(self): pass @@ -76,6 +79,8 @@ class TestWALERestore(unittest.TestCase): self.assertFalse(self.wale_restore.should_use_s3_to_create_replica()) self.wale_restore.should_use_s3_to_create_replica() + self.wale_restore.no_master = 1 + self.assertTrue(self.wale_restore.should_use_s3_to_create_replica()) def test_create_replica_with_s3(self): with patch('subprocess.call', MagicMock(return_value=0)): @@ -89,3 +94,8 @@ class TestWALERestore(unittest.TestCase): with patch.object(self.wale_restore, 'should_use_s3_to_create_replica', MagicMock(return_value=True)): with patch.object(self.wale_restore, 'create_replica_with_s3', MagicMock(return_value=0)): self.assertEqual(self.wale_restore.run(), 0) + + @patch('sys.exit', MagicMock()) + @patch.object(WALERestore, 'run', MagicMock(return_value=0)) + def test_main(self): + self.assertEqual(_main(), None) diff --git a/tests/test_zookeeper.py b/tests/test_zookeeper.py index 84807270..f5cbdf13 100644 --- a/tests/test_zookeeper.py +++ b/tests/test_zookeeper.py @@ -20,7 +20,8 @@ class MockKazooClient(Mock): def client_id(self): return (-1, '') - def retry(self, func, *args, **kwargs): + @staticmethod + def retry(func, *args, **kwargs): func(*args, **kwargs) def get(self, path, watch=None): @@ -43,7 +44,8 @@ class MockKazooClient(Mock): return (b'foo', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0)) return (b'', ZnodeStat(0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0)) - def get_children(self, path, watch=None, include_data=False): + @staticmethod + def get_children(path, watch=None, include_data=False): if not isinstance(path, six.string_types): raise TypeError("Invalid type for 'path' (string expected)") if path.startswith('/no_node'): @@ -62,16 +64,16 @@ class MockKazooClient(Mock): elif value == b'retry' or (value == b'exists' and self.exists): raise NodeExistsError - def set(self, path, value, version=-1): + @staticmethod + def set(path, value, version=-1): if not isinstance(path, six.string_types): raise TypeError("Invalid type for 'path' (string expected)") if not isinstance(value, (six.binary_type,)): raise TypeError("Invalid type for 'value' (must be a byte string)") if path == '/service/bla/optime/leader': raise Exception - if path == '/service/test/members/bar': - if value == b'retry': - return + if path == '/service/test/members/bar' and value == b'retry': + return if path == '/service/test/failover': if value == b'Exception': raise Exception