allow multiple hosts for healthcheck
Build and Push Docker Images / build-and-push (push) Successful in 23s
Build and Push Docker Images / build-and-push (push) Successful in 23s
This commit is contained in:
@@ -1,7 +1,7 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Healthcheck script for Puppeteer API
|
||||
Monitors the API endpoint and restarts the container if it fails
|
||||
Monitors multiple API endpoints and restarts containers if they fail
|
||||
"""
|
||||
|
||||
import os
|
||||
@@ -26,24 +26,46 @@ logger = logging.getLogger(__name__)
|
||||
|
||||
class PuppeteerHealthcheck:
|
||||
def __init__(self):
|
||||
self.base_url = os.getenv('BASE_URL', 'https://puppeteer.workwithkora.com')
|
||||
self.hosts = self._parse_hosts()
|
||||
self.test_url = os.getenv('TEST_URL', 'https://www.google.com')
|
||||
self.api_key = os.getenv('API_KEY', 'Q7Sd#hhFkyHy*T')
|
||||
self.target_container = os.getenv('TARGET_CONTAINER', 'puppeteer-api')
|
||||
self.check_interval = int(os.getenv('CHECK_INTERVAL', '60'))
|
||||
self.max_consecutive_failures = int(os.getenv('MAX_CONSECUTIVE_FAILURES', '3'))
|
||||
self.timeout = int(os.getenv('TIMEOUT', '20'))
|
||||
|
||||
# Track consecutive failures for each host
|
||||
self.failure_counters = {host: 0 for host in self.hosts}
|
||||
|
||||
# Initialize Docker client with proper error handling
|
||||
self.docker_client = self._initialize_docker_client()
|
||||
|
||||
logger.info(f"Healthcheck initialized with:")
|
||||
logger.info(f" Base URL: {self.base_url}")
|
||||
logger.info(f" Hosts: {', '.join(self.hosts)}")
|
||||
logger.info(f" Test URL: {self.test_url}")
|
||||
logger.info(f" Target Container: {self.target_container}")
|
||||
logger.info(f" Check Interval: {self.check_interval} seconds")
|
||||
logger.info(f" Timeout: {self.timeout} seconds")
|
||||
|
||||
def _parse_hosts(self):
|
||||
"""Parse hosts from environment variable or use default"""
|
||||
hosts_env = os.getenv('HOSTS', '')
|
||||
if hosts_env:
|
||||
# Split by comma and strip whitespace
|
||||
hosts = [host.strip() for host in hosts_env.split(',') if host.strip()]
|
||||
if hosts:
|
||||
return hosts
|
||||
|
||||
# Fallback to legacy BASE_URL for backward compatibility
|
||||
base_url = os.getenv('BASE_URL', 'https://puppeteer.workwithkora.com')
|
||||
if base_url.startswith('http://'):
|
||||
# Extract host from http://host:port format
|
||||
host_part = base_url.replace('http://', '').split('/')[0]
|
||||
if ':' in host_part:
|
||||
host = host_part.split(':')[0]
|
||||
return [host]
|
||||
|
||||
# Default fallback
|
||||
return ['puppeteer-api']
|
||||
|
||||
def _initialize_docker_client(self):
|
||||
"""Initialize Docker client with proper error handling"""
|
||||
try:
|
||||
@@ -66,19 +88,20 @@ class PuppeteerHealthcheck:
|
||||
logger.error(f"Unexpected error initializing Docker client: {e}")
|
||||
raise
|
||||
|
||||
def perform_health_check(self):
|
||||
"""Perform the health check by making a request to the puppeteer API"""
|
||||
def perform_health_check(self, host):
|
||||
"""Perform the health check for a specific host"""
|
||||
try:
|
||||
# Construct the URL with the test URL as a parameter
|
||||
base_url = f"http://{host}:8000"
|
||||
encoded_test_url = quote(self.test_url, safe='')
|
||||
url = f"{self.base_url}/?url={encoded_test_url}&skipCache=true"
|
||||
url = f"{base_url}/?url={encoded_test_url}"
|
||||
|
||||
headers = {
|
||||
'x-api-key': self.api_key,
|
||||
'User-Agent': 'Puppeteer-Healthcheck/1.0'
|
||||
}
|
||||
|
||||
logger.info(f"Performing health check: {url}")
|
||||
logger.info(f"Performing health check for {host}: {url}")
|
||||
|
||||
response = requests.get(
|
||||
url,
|
||||
@@ -92,77 +115,81 @@ class PuppeteerHealthcheck:
|
||||
try:
|
||||
response_data = response.json()
|
||||
if response_data.get('status') == 'error':
|
||||
logger.warning(f"Health check failed - API returned status: error")
|
||||
logger.warning(f"Health check failed for {host} - API returned status: error")
|
||||
logger.warning(f"Response: {response.text[:200]}...")
|
||||
return False
|
||||
else:
|
||||
logger.info("Health check passed - API is responding correctly")
|
||||
logger.info(f"Health check passed for {host} - API is responding correctly")
|
||||
return True
|
||||
except json.JSONDecodeError:
|
||||
# If response is not JSON, treat as success (backward compatibility)
|
||||
logger.info("Health check passed - API is responding correctly (non-JSON response)")
|
||||
logger.info(f"Health check passed for {host} - API is responding correctly (non-JSON response)")
|
||||
return True
|
||||
|
||||
else:
|
||||
logger.warning(f"Health check failed - Status code: {response.status_code}")
|
||||
logger.warning(f"Health check failed for {host} - Status code: {response.status_code}")
|
||||
logger.warning(f"Response: {response.text[:200]}...")
|
||||
return False
|
||||
|
||||
except requests.exceptions.RequestException as e:
|
||||
logger.error(f"Health check failed - Request error: {e}")
|
||||
logger.error(f"Health check failed for {host} - Request error: {e}")
|
||||
return False
|
||||
except Exception as e:
|
||||
logger.error(f"Health check failed - Unexpected error: {e}")
|
||||
logger.error(f"Health check failed for {host} - Unexpected error: {e}")
|
||||
return False
|
||||
|
||||
def restart_container(self):
|
||||
"""Restart the target puppeteer-api container"""
|
||||
def restart_container(self, host):
|
||||
"""Restart the container for a specific host"""
|
||||
try:
|
||||
logger.info(f"Attempting to restart container: {self.target_container}")
|
||||
logger.info(f"Attempting to restart container: {host}")
|
||||
|
||||
# Get the container
|
||||
container = self.docker_client.containers.get(self.target_container)
|
||||
# Get the container (container name matches host name)
|
||||
container = self.docker_client.containers.get(host)
|
||||
|
||||
# Restart the container
|
||||
container.restart(timeout=30)
|
||||
|
||||
logger.info(f"Successfully restarted container: {self.target_container}")
|
||||
logger.info(f"Successfully restarted container: {host}")
|
||||
return True
|
||||
|
||||
except docker.errors.NotFound:
|
||||
logger.error(f"Container not found: {self.target_container}")
|
||||
logger.error(f"Container not found: {host}")
|
||||
return False
|
||||
except docker.errors.APIError as e:
|
||||
logger.error(f"Docker API error while restarting container: {e}")
|
||||
logger.error(f"Docker API error while restarting container {host}: {e}")
|
||||
return False
|
||||
except Exception as e:
|
||||
logger.error(f"Unexpected error while restarting container: {e}")
|
||||
logger.error(f"Unexpected error while restarting container {host}: {e}")
|
||||
return False
|
||||
|
||||
def check_host(self, host):
|
||||
"""Check a single host and handle failures"""
|
||||
if self.perform_health_check(host):
|
||||
if self.failure_counters[host] > 0:
|
||||
logger.info(f"Health check passed for {host} - resetting failure counter")
|
||||
self.failure_counters[host] = 0
|
||||
else:
|
||||
self.failure_counters[host] += 1
|
||||
logger.warning(f"Health check failed for {host} - consecutive failures: {self.failure_counters[host]}")
|
||||
|
||||
# Restart container if we've had too many consecutive failures
|
||||
if self.failure_counters[host] >= self.max_consecutive_failures:
|
||||
logger.error(f"Health check failed {self.failure_counters[host]} times consecutively for {host} - restarting container")
|
||||
if self.restart_container(host):
|
||||
self.failure_counters[host] = 0
|
||||
logger.info(f"Container {host} restarted successfully - resetting failure counter")
|
||||
else:
|
||||
logger.error(f"Failed to restart container {host}")
|
||||
|
||||
def run(self):
|
||||
"""Main loop for the healthcheck"""
|
||||
logger.info("Starting Puppeteer API healthcheck service")
|
||||
|
||||
consecutive_failures = 0
|
||||
|
||||
while True:
|
||||
try:
|
||||
# Perform health check
|
||||
if self.perform_health_check():
|
||||
consecutive_failures = 0
|
||||
logger.info("Health check passed - resetting failure counter")
|
||||
else:
|
||||
consecutive_failures += 1
|
||||
logger.warning(f"Health check failed - consecutive failures: {consecutive_failures}")
|
||||
|
||||
# Restart container if we've had too many consecutive failures
|
||||
if consecutive_failures >= self.max_consecutive_failures:
|
||||
logger.error(f"Health check failed {consecutive_failures} times consecutively - restarting container")
|
||||
if self.restart_container():
|
||||
consecutive_failures = 0
|
||||
logger.info("Container restarted successfully - resetting failure counter")
|
||||
else:
|
||||
logger.error("Failed to restart container")
|
||||
# Check each host
|
||||
for host in self.hosts:
|
||||
self.check_host(host)
|
||||
|
||||
# Wait before next check
|
||||
logger.info(f"Waiting {self.check_interval} seconds before next health check...")
|
||||
|
||||
Reference in New Issue
Block a user