diff --git a/Dockers/puppeteer-api/Dockerfile b/Dockers/puppeteer-api/Dockerfile index 1503a96..1600f8e 100644 --- a/Dockers/puppeteer-api/Dockerfile +++ b/Dockers/puppeteer-api/Dockerfile @@ -1,4 +1,4 @@ -FROM python:3.9-slim +FROM python:3.11-slim # Install required system dependencies RUN apt-get update && apt-get install -y \ @@ -6,11 +6,6 @@ RUN apt-get update && apt-get install -y \ gnupg2 \ procps \ sqlite3 \ - && wget -q -O - https://dl-ssl.google.com/linux/linux_signing_key.pub | gpg --dearmor -o /usr/share/keyrings/google-chrome-keyring.gpg \ - && echo "deb [arch=amd64 signed-by=/usr/share/keyrings/google-chrome-keyring.gpg] http://dl.google.com/linux/chrome/deb/ stable main" | tee /etc/apt/sources.list.d/google-chrome.list \ - && apt-get update \ - && apt-get install -y \ - google-chrome-stable \ fonts-ipafont-gothic \ fonts-wqy-zenhei \ fonts-thai-tlwg \ @@ -19,9 +14,7 @@ RUN apt-get update && apt-get install -y \ libxss1 \ xvfb \ --no-install-recommends \ - && rm -rf /var/lib/apt/lists/* \ - # Verify Chrome installation - && google-chrome --version + && rm -rf /var/lib/apt/lists/* # Set working directory WORKDIR /app @@ -30,6 +23,23 @@ WORKDIR /app COPY requirements.txt . RUN pip install --no-cache-dir -r requirements.txt +# Install Playwright system dependencies as root +RUN playwright install-deps chromium + +# Create a non-root user +RUN useradd -m playwright + +# Set HOME for playwright user and ensure cache directory exists +ENV HOME=/home/playwright +RUN mkdir -p /home/playwright/.cache/ms-playwright && chown -R playwright:playwright /home/playwright + +# Install Playwright browsers as the playwright user +USER playwright +RUN playwright install chromium + +# Switch back to root for file operations +USER root + # Copy the rest of the application COPY . . @@ -39,17 +49,19 @@ ENV API_KEY="" ENV CACHE_EXPIRY_HOURS=36 ENV CLEANUP_CRON="0 3 * * *" -# Create a non-root user and switch to it -RUN useradd -m puppeteer && chown -R puppeteer:puppeteer /app +# Set ownership of the application to playwright user +RUN chown -R playwright:playwright /app + # Create the database directory and set permissions -RUN mkdir -p /db && chmod 777 /db && chown -R puppeteer:puppeteer /app +RUN mkdir -p /db && chmod 777 /db && chown -R playwright:playwright /app # Add system limits configuration RUN echo "* soft nofile 65535" >> /etc/security/limits.conf && \ echo "* hard nofile 65535" >> /etc/security/limits.conf && \ echo "session required pam_limits.so" >> /etc/pam.d/common-session -USER puppeteer +# Switch to playwright user for running the application +USER playwright # Expose port EXPOSE 8000 diff --git a/Dockers/puppeteer-api/MIGRATION_GUIDE.md b/Dockers/puppeteer-api/MIGRATION_GUIDE.md new file mode 100644 index 0000000..49d2e23 --- /dev/null +++ b/Dockers/puppeteer-api/MIGRATION_GUIDE.md @@ -0,0 +1,344 @@ +# Migration Guide: Puppeteer to Playwright + +This guide will help you migrate from Puppeteer (pyppeteer) to Playwright in your project. + +## Overview + +The migration from Puppeteer to Playwright provides several benefits: + +- Better performance and stability +- Enhanced features for modern web applications +- Active development and community support +- Multi-browser support (Chromium, Firefox, WebKit) + +## Pre-Migration Checklist + +Before starting the migration, ensure you have: + +- [ ] Backed up your current codebase +- [ ] Documented any custom Puppeteer configurations +- [ ] Identified all Puppeteer-specific code in your project +- [ ] Tested your current application thoroughly + +## Step-by-Step Migration + +### 1. Update Dependencies + +Replace Puppeteer dependencies with Playwright: + +**Before (requirements.txt):** + +``` +pyppeteer==1.0.2 +``` + +**After (requirements.txt):** + +``` +playwright==1.40.0 +``` + +### 2. Install Playwright + +```bash +# Install Playwright +pip install playwright + +# Install browsers (Chromium is recommended for compatibility) +playwright install chromium + +# Install system dependencies (Linux) +playwright install-deps chromium +``` + +### 3. Update Imports + +**Before:** + +```python +from pyppeteer import launch +``` + +**After:** + +```python +from playwright.async_api import async_playwright +``` + +### 4. Update Browser Launch + +**Before (Puppeteer):** + +```python +browser = await launch( + headless=True, + executablePath='/usr/bin/google-chrome', + args=['--no-sandbox', '--disable-setuid-sandbox'], + handleSIGINT=False, + handleSIGTERM=False, + handleSIGHUP=False, + ignoreHTTPSErrors=True, + autoClose=True, +) +``` + +**After (Playwright):** + +```python +playwright = await async_playwright().start() +browser = await playwright.chromium.launch( + headless=True, + args=['--no-sandbox', '--disable-setuid-sandbox'], + ignore_default_args=['--enable-automation'], +) +``` + +### 5. Update Page Creation + +**Before (Puppeteer):** + +```python +page = await browser.newPage() +await page.setUserAgent(CUSTOM_USER_AGENT) +await page.setViewport({'width': 1280, 'height': 800}) +page.setDefaultNavigationTimeout(30000) +``` + +**After (Playwright):** + +```python +context = await browser.new_context( + user_agent=CUSTOM_USER_AGENT, + viewport={'width': 1920, 'height': 1080}, + ignore_https_errors=True, +) +page = await context.new_page() +page.set_default_timeout(30000) +``` + +### 6. Update Navigation + +**Before (Puppeteer):** + +```python +response = await page.goto(url, waitUntil='networkidle2', timeout=30000) +``` + +**After (Playwright):** + +```python +response = await page.goto(url, wait_until='networkidle', timeout=30000) +``` + +### 7. Update Wait Methods + +**Before (Puppeteer):** + +```python +await page.waitForNetworkIdle(idleTime=500, timeout=30000) +await page.waitForNavigation(waitUntil='networkidle2', timeout=10000) +``` + +**After (Playwright):** + +```python +await page.wait_for_load_state('networkidle', timeout=30000) +await page.wait_for_load_state('networkidle', timeout=10000) +``` + +### 8. Update Request Interception + +**Before (Puppeteer):** + +```python +await page.setRequestInterception(True) + +async def intercept(request): + if request.resourceType in ['image', 'media', 'font', 'stylesheet']: + await request.abort() + else: + await request.continue_() + +page.on('request', lambda req: asyncio.ensure_future(intercept(req))) +``` + +**After (Playwright):** + +```python +await page.route("**/*", lambda route: route.abort() + if route.request.resource_type in ['image', 'stylesheet', 'font', 'media'] + else route.continue_()) +``` + +### 9. Update Browser Cleanup + +**Before (Puppeteer):** + +```python +pages = await browser.pages() +for page in pages: + await page.close() +await browser.close() +``` + +**After (Playwright):** + +```python +pages = browser.contexts[0].pages if browser.contexts else [] +for page in pages: + await page.close() +await browser.close() +``` + +## Docker Migration + +### Update Dockerfile + +**Before:** + +```dockerfile +# Install Chrome +RUN wget -q -O - https://dl-ssl.google.com/linux/linux_signing_key.pub | gpg --dearmor -o /usr/share/keyrings/google-chrome-keyring.gpg \ + && echo "deb [arch=amd64 signed-by=/usr/share/keyrings/google-chrome-keyring.gpg] http://dl.google.com/linux/chrome/deb/ stable main" | tee /etc/apt/sources.list.d/google-chrome.list \ + && apt-get update \ + && apt-get install -y google-chrome-stable +``` + +**After:** + +```dockerfile +# Install Playwright browsers +RUN playwright install chromium +RUN playwright install-deps chromium +``` + +## Testing the Migration + +### 1. Run the Test Script + +```bash +python test_playwright_migration.py +``` + +This script tests: + +- Playwright installation +- Browser utilities +- Browser service functionality + +### 2. Manual Testing + +Test your key functionality: + +- Basic page navigation +- JavaScript execution +- Screenshot capture +- PDF generation (if used) + +### 3. Performance Testing + +Compare performance metrics: + +- Page load times +- Memory usage +- CPU usage +- Browser startup time + +## Common Issues and Solutions + +### Issue: Browser Not Starting + +**Solution:** Ensure Playwright browsers are installed: + +```bash +playwright install chromium +``` + +### Issue: Permission Errors + +**Solution:** Install system dependencies: + +```bash +playwright install-deps chromium +``` + +### Issue: Page Navigation Fails + +**Solution:** Check timeout settings and network conditions: + +```python +# Increase timeout if needed +await page.goto(url, wait_until='networkidle', timeout=60000) +``` + +### Issue: Memory Leaks + +**Solution:** Ensure proper cleanup: + +```python +# Always close context and browser +await context.close() +await browser.close() +``` + +## Performance Optimizations + +### 1. Browser Pool Management + +```python +# Use browser pooling for better performance +browser_pool = Queue(maxsize=MAX_BROWSERS) +``` + +### 2. Context Reuse + +```python +# Reuse contexts when possible +context = await browser.new_context() +# Use context for multiple pages +``` + +### 3. Resource Blocking + +```python +# Block unnecessary resources +await page.route("**/*", lambda route: route.abort() + if route.request.resource_type in ['image', 'stylesheet', 'font', 'media'] + else route.continue_()) +``` + +## Rollback Plan + +If you need to rollback to Puppeteer: + +1. Restore your backup +2. Revert requirements.txt changes +3. Revert Dockerfile changes +4. Test thoroughly + +## Post-Migration Checklist + +After completing the migration: + +- [ ] All tests pass +- [ ] Performance is acceptable +- [ ] No memory leaks detected +- [ ] Error handling works correctly +- [ ] Documentation is updated +- [ ] Team is trained on new API + +## Support + +If you encounter issues during migration: + +1. Check the [Playwright documentation](https://playwright.dev/python/) +2. Review the [migration guide](https://playwright.dev/python/docs/migrate-from-puppeteer) +3. Search existing issues on GitHub +4. Create a new issue with detailed information + +## Additional Resources + +- [Playwright Python Documentation](https://playwright.dev/python/) +- [Migration Guide from Puppeteer](https://playwright.dev/python/docs/migrate-from-puppeteer) +- [Playwright GitHub Repository](https://github.com/microsoft/playwright) +- [Community Discord](https://discord.gg/playwright) diff --git a/Dockers/puppeteer-api/README.md b/Dockers/puppeteer-api/README.md index b870742..f0d0975 100644 --- a/Dockers/puppeteer-api/README.md +++ b/Dockers/puppeteer-api/README.md @@ -1,6 +1,27 @@ -# Puppeteer API Server +# Playwright API Server -A FastAPI-based server that uses Puppeteer (via pyppeteer) to scrape websites and extract various types of data. +A FastAPI-based server that uses Playwright to scrape websites and extract various types of data. + +## Recent Migration (v3.0) + +### Migration from Puppeteer to Playwright + +This project has been successfully migrated from Puppeteer (pyppeteer) to Playwright for better performance, reliability, and maintainability. + +#### Key Benefits of Playwright + +- **Better performance**: More efficient browser management and faster page loads +- **Improved stability**: Better handling of modern web applications +- **Enhanced features**: Better support for modern web standards +- **Active development**: More frequent updates and better community support +- **Multi-browser support**: Can easily switch between Chromium, Firefox, and WebKit + +#### Migration Changes + +- **Dependencies**: Updated from `pyppeteer` to `playwright` +- **Browser management**: Improved browser pool with better resource management +- **API compatibility**: All existing endpoints remain the same +- **Docker image**: Updated to use Playwright browsers instead of Chrome ## Recent Fixes (v2.0) @@ -102,19 +123,34 @@ python monitor.py your-api-key 30 http://localhost:8000 cleanup ```bash # Build the image -docker build -t puppeteer-api . +docker build -t playwright-api . # Run with environment variables docker run -d \ - --name puppeteer-api \ + --name playwright-api \ -p 8000:8000 \ -e API_KEY=your-api-key \ -e MAX_BROWSERS=3 \ -e BROWSER_TTL=1800 \ -v /path/to/cache:/db \ - puppeteer-api + playwright-api ``` +## Testing the Migration + +To verify that the Playwright migration works correctly: + +```bash +# Run the migration test script +python test_playwright_migration.py +``` + +This will test: + +- Playwright installation and basic functionality +- Browser utilities module +- Browser service functionality + ## Troubleshooting ### High CPU Usage @@ -135,6 +171,12 @@ docker run -d \ 2. Trigger emergency cleanup 3. Restart the container if needed +### Playwright Installation Issues + +1. Ensure Playwright browsers are installed: `playwright install chromium` +2. Check Docker build logs for browser installation +3. Verify system dependencies are installed + ## Performance Tips 1. **Use caching**: The API caches results for 36 hours by default diff --git a/Dockers/puppeteer-api/app/services/browser.py b/Dockers/puppeteer-api/app/services/browser.py index 2a2ff6c..704dc74 100644 --- a/Dockers/puppeteer-api/app/services/browser.py +++ b/Dockers/puppeteer-api/app/services/browser.py @@ -10,7 +10,7 @@ async def visit_url_service(decoded_url): # Define the operation to perform with the browser async def visit_operation(page): try: - response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) + response = await page.goto(decoded_url, wait_until='networkidle', timeout=30000) if not response: print(f"Warning: No response object returned for {decoded_url}") @@ -36,7 +36,7 @@ async def extract_seo_service(decoded_url): # Define the operation to perform with the browser async def seo_operation(page): try: - response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) + response = await page.goto(decoded_url, wait_until='networkidle', timeout=30000) # Extract SEO information seo_data = await page.evaluate('''() => { @@ -154,7 +154,7 @@ async def detect_pagination_service(decoded_url): try: # Navigate to the URL try: - await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) + await page.goto(decoded_url, wait_until='networkidle', timeout=30000) except Exception as e: print(f"Error navigating to URL: {e}") # Try to get the current URL even if navigation failed @@ -433,7 +433,7 @@ async def detect_pagination_service(decoded_url): if clicked: print("Successfully clicked on pagination element") # Wait for navigation to complete - await page.waitForNavigation(waitUntil='networkidle2', timeout=10000) + await page.wait_for_load_state('networkidle', timeout=10000) next_page_url = page.url else: print("No clickable pagination element found") @@ -447,8 +447,8 @@ async def detect_pagination_service(decoded_url): print('Searching for pagination parameter') # Parse both URLs try: - original_parsed = await page.evaluate(f'''(originalUrl) => {{ - try {{ + original_parsed = await page.evaluate('''([originalUrl]) => { + try { const original = new URL(originalUrl); const current = new URL(window.location.href); @@ -458,99 +458,99 @@ async def detect_pagination_service(decoded_url): // Common pagination parameters to check const paginationParams = ['page', 'p', 'pg', 'offset', 'o', 'from', 'start', 'limit', 'currentPage', 'current_page', 'currentpage', 'pagenum', 'pageNumber', 'paged']; - for (const param of paginationParams) {{ + for (const param of paginationParams) { const originalValue = original.searchParams.get(param); const currentValue = current.searchParams.get(param); - if (originalValue !== currentValue && currentValue !== null) {{ - paramDiff = {{ + if (originalValue !== currentValue && currentValue !== null) { + paramDiff = { name: param, originalValue: originalValue, currentValue: currentValue - }}; + }; break; - }} - }} + } + } // Check for path differences (like /page/1 vs /page/2 or /vacatures vs /vacatures/page/2) const originalPath = original.pathname; const currentPath = current.pathname; let pathDiff = null; - if (originalPath !== currentPath) {{ + if (originalPath !== currentPath) { const originalSegments = originalPath.split('/').filter(s => s); const currentSegments = currentPath.split('/').filter(s => s); // Case 1: Same number of segments - find the one that changed - if (originalSegments.length === currentSegments.length) {{ - for (let i = 0; i < originalSegments.length; i++) {{ - if (originalSegments[i] !== currentSegments[i]) {{ + if (originalSegments.length === currentSegments.length) { + for (let i = 0; i < originalSegments.length; i++) { + if (originalSegments[i] !== currentSegments[i]) { // Check if the difference is numeric - if (!isNaN(originalSegments[i]) && !isNaN(currentSegments[i])) {{ - pathDiff = {{ + if (!isNaN(originalSegments[i]) && !isNaN(currentSegments[i])) { + pathDiff = { type: 'replace', index: i, originalValue: originalSegments[i], currentValue: currentSegments[i] - }}; - }} - }} - }} - }} + }; + } + } + } + } // Case 2: Current path has more segments - check for added pagination segments - else if (currentSegments.length > originalSegments.length) {{ + else if (currentSegments.length > originalSegments.length) { // Look for patterns like /page/NUMBER or /p/NUMBER at the end const pagePattern = /^(page|p)\/(\d+)$/i; // Check the last two segments of the current path - if (currentSegments.length >= 2) {{ + if (currentSegments.length >= 2) { const lastTwoSegments = currentSegments.slice(-2).join('/'); const match = lastTwoSegments.match(pagePattern); - if (match) {{ - pathDiff = {{ + if (match) { + pathDiff = { type: 'append', pageSegment: match[1], // 'page' or 'p' pageNumber: match[2], // the actual number originalSegments: originalSegments, currentSegments: currentSegments - }}; - }} - }} + }; + } + } // If no pattern match, check if the last segment is numeric - if (!pathDiff && currentSegments.length > 0) {{ + if (!pathDiff && currentSegments.length > 0) { const lastSegment = currentSegments[currentSegments.length - 1]; - if (!isNaN(lastSegment)) {{ - pathDiff = {{ + if (!isNaN(lastSegment)) { + pathDiff = { type: 'append', pageSegment: null, pageNumber: lastSegment, originalSegments: originalSegments, currentSegments: currentSegments - }}; - }} - }} - }} - }} + }; + } + } + } + } - return {{ + return { paramDiff, pathDiff, originalUrl: originalUrl, currentUrl: window.location.href - }}; - }} catch (error) {{ + }; + } catch (error) { console.error("Error during URL analysis:", error); - return {{ + return { paramDiff: null, pathDiff: null, originalUrl: originalUrl, currentUrl: window.location.href, error: error.message - }}; - }} - }}''', original_url) + }; + } + }''', [original_url]) except Exception as e: print(f"Error during URL analysis: {e}") original_parsed = { @@ -585,16 +585,16 @@ async def detect_pagination_service(decoded_url): # Create URL template for query parameter try: - url_obj = await page.evaluate(f'''(url, paramName) => {{ - try {{ + url_obj = await page.evaluate('''([url, paramName]) => { + try { const urlObj = new URL(url); - urlObj.searchParams.set(paramName, "{{PAGE_NUMBER}}"); + urlObj.searchParams.set(paramName, "{PAGE_NUMBER}"); return urlObj.toString(); - }} catch (error) {{ + } catch (error) { console.error("Error creating URL template:", error); return null; - }} - }}''', original_url, param_name) + } + }''', [original_url, param_name]) url_template = url_obj except Exception as e: @@ -626,18 +626,18 @@ async def detect_pagination_service(decoded_url): # Create URL template for path parameter replacement try: - url_template = await page.evaluate(f'''(url, pathIndex) => {{ - try {{ + url_template = await page.evaluate('''([url, pathIndex]) => { + try { const urlObj = new URL(url); const pathSegments = urlObj.pathname.split('/').filter(s => s); - pathSegments[pathIndex] = "{{PAGE_NUMBER}}"; + pathSegments[pathIndex] = "{PAGE_NUMBER}"; urlObj.pathname = '/' + pathSegments.join('/'); return urlObj.toString(); - }} catch (error) {{ + } catch (error) { console.error("Error creating path URL template:", error); return null; - }} - }}''', original_url, path_index) + } + }''', [original_url, path_index]) except Exception as e: print(f"Error creating URL template for path parameter: {e}") url_template = None @@ -662,30 +662,30 @@ async def detect_pagination_service(decoded_url): # Create URL template for appended path parameter try: - url_template = await page.evaluate(f'''(url, pageSegment) => {{ - try {{ + url_template = await page.evaluate('''([url, pageSegment]) => { + try { const urlObj = new URL(url); let newPath = urlObj.pathname; // Remove trailing slash if present - if (newPath.endsWith('/')) {{ + if (newPath.endsWith('/')) { newPath = newPath.slice(0, -1); - }} + } // Append the pagination segment - if (pageSegment) {{ - newPath += '/' + pageSegment + '/{{PAGE_NUMBER}}'; - }} else {{ - newPath += '/{{PAGE_NUMBER}}'; - }} + if (pageSegment) { + newPath += '/' + pageSegment + '/{PAGE_NUMBER}'; + } else { + newPath += '/{PAGE_NUMBER}'; + } urlObj.pathname = newPath; return urlObj.toString(); - }} catch (error) {{ + } catch (error) { console.error("Error creating appended path URL template:", error); return null; - }} - }}''', original_url, path_diff.get('pageSegment')) + } + }''', [original_url, path_diff.get('pageSegment')]) except Exception as e: print(f"Error creating URL template for appended path parameter: {e}") url_template = None @@ -699,16 +699,16 @@ async def detect_pagination_service(decoded_url): if not url_template and pagination_data['detectedParameter']: param_name = pagination_data['detectedParameter']['name'] try: - url_template = await page.evaluate(f'''(url, paramName) => {{ - try {{ + url_template = await page.evaluate('''([url, paramName]) => { + try { const urlObj = new URL(url); - urlObj.searchParams.set(paramName, "{{PAGE_NUMBER}}"); + urlObj.searchParams.set(paramName, "{PAGE_NUMBER}"); return urlObj.toString(); - }} catch (error) {{ + } catch (error) { console.error("Error creating inferred URL template:", error); return null; - }} - }}''', original_url, param_name) + } + }''', [original_url, param_name]) except Exception as e: print(f"Error creating inferred URL template: {e}") url_template = None @@ -716,7 +716,7 @@ async def detect_pagination_service(decoded_url): # If still no template and we have pagination elements, try to infer from the current URL structure if not url_template and pagination_data['hasPagination']: try: - url_template = await page.evaluate('''(originalUrl) => { + url_template = await page.evaluate('''([originalUrl]) => { try { const urlObj = new URL(originalUrl); let path = urlObj.pathname; @@ -730,10 +730,10 @@ async def detect_pagination_service(decoded_url): const pagePattern = /\/(page|p)\/\d+$/i; if (pagePattern.test(path)) { // Replace the existing page number with placeholder - path = path.replace(/\/(page|p)\/\d+$/i, '/$1/{{PAGE_NUMBER}}'); + path = path.replace(/\/(page|p)\/\d+$/i, '/$1/{PAGE_NUMBER}'); } else { // Add pagination pattern - path += '/page/{{PAGE_NUMBER}}'; + path += '/page/{PAGE_NUMBER}'; } urlObj.pathname = path; @@ -742,7 +742,7 @@ async def detect_pagination_service(decoded_url): console.error("Error creating fallback URL template:", error); return null; } - }''', original_url) + }''', [original_url]) except Exception as e: print(f"Error creating fallback URL template: {e}") url_template = None diff --git a/Dockers/puppeteer-api/app/utils/browser_utils.py b/Dockers/puppeteer-api/app/utils/browser_utils.py index bdb287a..4e7d8b1 100644 --- a/Dockers/puppeteer-api/app/utils/browser_utils.py +++ b/Dockers/puppeteer-api/app/utils/browser_utils.py @@ -1,31 +1,34 @@ -from pyppeteer import launch +from playwright.async_api import async_playwright from app.config import CUSTOM_USER_AGENT import asyncio async def wait_for_network_idle(page): """Wait until no network requests are in flight""" - await page.waitForNetworkIdle(idleTime=500, timeout=30000) + await page.wait_for_load_state('networkidle') async def safe_browser_operation(url, operation_func): """Safely perform browser operations with proper cleanup""" browser = None + context = None page = None try: - browser = await launch( + # Get or create playwright instance + playwright = await async_playwright().start() + + browser = await playwright.chromium.launch( headless=True, - executablePath='/usr/bin/google-chrome', args=['--no-sandbox', '--disable-setuid-sandbox'], - handleSIGINT=False, - handleSIGTERM=False, - handleSIGHUP=False ) - # Create new page with timeout - page = await browser.newPage() - page.setDefaultNavigationTimeout(30000) + # Create context and page + context = await browser.new_context( + user_agent=CUSTOM_USER_AGENT, + viewport={'width': 1920, 'height': 1080}, + ignore_https_errors=True, + ) - # Set custom user agent - await page.setUserAgent(CUSTOM_USER_AGENT) + page = await context.new_page() + page.set_default_timeout(30000) # Call the operation function that uses the page result = await operation_func(page) @@ -48,6 +51,13 @@ async def safe_browser_operation(url, operation_func): except Exception as e: print(f"Error closing page: {e}") + # Ensure context is closed properly + if context: + try: + await context.close() + except Exception as e: + print(f"Error closing context: {e}") + # Ensure browser is closed properly if browser: try: diff --git a/Dockers/puppeteer-api/install_playwright.py b/Dockers/puppeteer-api/install_playwright.py new file mode 100644 index 0000000..9869411 --- /dev/null +++ b/Dockers/puppeteer-api/install_playwright.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +""" +Installation script for Playwright migration +""" + +import subprocess +import sys +import os + +def run_command(command, description): + """Run a command and handle errors""" + print(f"๐Ÿ”„ {description}...") + try: + result = subprocess.run(command, shell=True, check=True, capture_output=True, text=True) + print(f"โœ… {description} completed successfully") + return True + except subprocess.CalledProcessError as e: + print(f"โŒ {description} failed:") + print(f" Error: {e.stderr}") + return False + +def main(): + """Main installation process""" + print("๐Ÿš€ Playwright Migration Installation Script") + print("=" * 50) + + # Check if we're in a virtual environment + if not hasattr(sys, 'real_prefix') and not (hasattr(sys, 'base_prefix') and sys.base_prefix != sys.prefix): + print("โš ๏ธ Warning: It's recommended to run this in a virtual environment") + response = input("Continue anyway? (y/N): ") + if response.lower() != 'y': + print("Installation cancelled.") + return 1 + + # Install Playwright + if not run_command("pip install playwright", "Installing Playwright"): + return 1 + + # Install Playwright browsers + if not run_command("playwright install chromium", "Installing Chromium browser"): + return 1 + + # Install system dependencies (for Linux) + if os.name == 'posix' and os.uname().sysname == 'Linux': + if not run_command("playwright install-deps chromium", "Installing system dependencies"): + print("โš ๏ธ System dependencies installation failed. You may need to install them manually.") + print(" On Ubuntu/Debian: sudo apt-get install -y libnss3 libatk-bridge2.0-0 libdrm2 libxkbcommon0 libxcomposite1 libxdamage1 libxrandr2 libgbm1 libasound2") + + # Test the installation + print("\n๐Ÿงช Testing Playwright installation...") + test_script = """ +import asyncio +from playwright.async_api import async_playwright + +async def test(): + async with async_playwright() as p: + browser = await p.chromium.launch(headless=True) + context = await browser.new_context() + page = await context.new_page() + await page.goto('https://example.com') + title = await page.title() + await context.close() + await browser.close() + return title + +result = asyncio.run(test()) +print(f"โœ… Playwright test successful! Page title: {result}") +""" + + try: + result = subprocess.run([sys.executable, '-c', test_script], + capture_output=True, text=True, check=True) + print(result.stdout.strip()) + except subprocess.CalledProcessError as e: + print(f"โŒ Playwright test failed: {e.stderr}") + return 1 + + print("\n๐ŸŽ‰ Playwright installation completed successfully!") + print("\nNext steps:") + print("1. Update your requirements.txt to include 'playwright==1.40.0'") + print("2. Update your Dockerfile to install Playwright browsers") + print("3. Test your application with: python test_playwright_migration.py") + + return 0 + +if __name__ == "__main__": + sys.exit(main()) \ No newline at end of file diff --git a/Dockers/puppeteer-api/main.py b/Dockers/puppeteer-api/main.py index d3deb41..9711407 100644 --- a/Dockers/puppeteer-api/main.py +++ b/Dockers/puppeteer-api/main.py @@ -1,5 +1,5 @@ from fastapi import FastAPI, HTTPException, Header, Request -from pyppeteer import launch +from playwright.async_api import async_playwright import os import asyncio import json @@ -46,6 +46,7 @@ browser_lock = Lock() browser_creation_times = {} active_browsers = set() # Track active browsers operation_semaphore = Semaphore(MAX_CONCURRENT_OPERATIONS) # Limit concurrent operations +playwright_instance = None # Global playwright instance # Rate limiting configuration RATE_LIMIT_MINUTE = int(os.getenv('RATE_LIMIT_MINUTE', '60')) # requests per minute @@ -95,10 +96,14 @@ app.add_middleware(RateLimitMiddleware) async def create_browser(): """Create a new browser instance with improved resource management""" + global playwright_instance + try: - browser = await launch( + if playwright_instance is None: + playwright_instance = await async_playwright().start() + + browser = await playwright_instance.chromium.launch( headless=True, - executablePath='/usr/bin/google-chrome', args=[ '--no-sandbox', '--disable-setuid-sandbox', @@ -127,11 +132,7 @@ async def create_browser(): '--disable-web-security', '--disable-features=VizDisplayCompositor', ], - handleSIGINT=False, - handleSIGTERM=False, - handleSIGHUP=False, - ignoreHTTPSErrors=True, - autoClose=True, # Ensure browser closes automatically + ignore_default_args=['--enable-automation'], ) browser_creation_times[browser] = time.time() @@ -152,7 +153,7 @@ async def cleanup_browser(browser): del browser_creation_times[browser] # Close all pages first - pages = await browser.pages() + pages = browser.contexts[0].pages if browser.contexts else [] for page in pages: try: await page.close() @@ -191,7 +192,9 @@ async def check_browser_health(): browser = await create_browser() else: # Quick health check - await browser.pages() + contexts = browser.contexts + if contexts: + pages = contexts[0].pages # Put back in pool if healthy if not browser_pool.full(): @@ -200,208 +203,190 @@ async def check_browser_health(): # Pool is full, cleanup this browser await cleanup_browser(browser) except Exception as e: - print(f"Browser health check failed: {e}") - # If unhealthy, close and create new - await cleanup_browser(browser) - if not browser_pool.full(): - new_browser = await create_browser() - await browser_pool.put(new_browser) + print(f"Error checking browser health: {e}") + # Cleanup the problematic browser + try: + await cleanup_browser(browser) + except: + pass + + # Create new browsers if pool is empty + while browser_pool.qsize() < MAX_BROWSERS: + try: + browser = await create_browser() + await browser_pool.put(browser) + except Exception as e: + print(f"Error creating browser for pool: {e}") + break + except Exception as e: - print(f"Error in browser health check: {str(e)}") + print(f"Error in browser health check: {e}") + await asyncio.sleep(60) # Wait before retrying async def force_cleanup_all_browsers(): - """Force cleanup all browsers in emergency situations""" + """Force cleanup all browsers - useful for emergency situations""" print("Force cleaning up all browsers...") - # Clean up pool - while not browser_pool.empty(): - try: - browser = await browser_pool.get_nowait() - await cleanup_browser(browser) - except asyncio.QueueEmpty: - break + async with browser_lock: + # Clean up browsers in pool + while not browser_pool.empty(): + try: + browser = await browser_pool.get_nowait() + await cleanup_browser(browser) + except asyncio.QueueEmpty: + break - # Clean up active browsers - for browser in list(active_browsers): - await cleanup_browser(browser) + # Clean up any remaining active browsers + for browser in list(active_browsers): + try: + await cleanup_browser(browser) + except Exception as e: + print(f"Error force cleaning browser: {e}") + + print("Force cleanup completed") -# Initialize browser pool @app.on_event("startup") async def init_browser_pool(): - """Initialize the browser pool with some browsers""" - try: - for _ in range(min(2, MAX_BROWSERS)): # Start with 2 browsers instead of 3 + """Initialize the browser pool on startup""" + print("Initializing browser pool...") + + # Start browser health check task + asyncio.create_task(check_browser_health()) + + # Pre-populate pool with initial browsers + for _ in range(min(2, MAX_BROWSERS)): + try: browser = await create_browser() await browser_pool.put(browser) - print(f"Browser pool initialized with {browser_pool.qsize()} browsers") + except Exception as e: + print(f"Error creating initial browser: {e}") - # Start browser health check task - asyncio.create_task(check_browser_health()) - except Exception as e: - print(f"Error initializing browser pool: {e}") + print(f"Browser pool initialized with {browser_pool.qsize()} browsers") @app.on_event("shutdown") async def cleanup_browser_pool(): - """Clean up all browsers in the pool""" + """Cleanup browser pool on shutdown""" + print("Cleaning up browser pool...") await force_cleanup_all_browsers() -# Signal handlers for graceful shutdown + # Stop playwright instance + global playwright_instance + if playwright_instance: + await playwright_instance.stop() + playwright_instance = None + + print("Browser pool cleanup completed") + def signal_handler(signum, frame): + """Handle shutdown signals""" print(f"Received signal {signum}, shutting down gracefully...") - asyncio.create_task(force_cleanup_all_browsers()) + asyncio.create_task(cleanup_browser_pool()) + exit(0) -signal.signal(signal.SIGTERM, signal_handler) +# Register signal handlers signal.signal(signal.SIGINT, signal_handler) +signal.signal(signal.SIGTERM, signal_handler) -# Initialize SQLite database def init_db(): - global DB_PATH + """Initialize the SQLite database""" + conn = sqlite3.connect('/db/cache.db') + cursor = conn.cursor() - # Try to use the mounted volume first - db_path = '/db/cache.db' - db_dir = os.path.dirname(db_path) - - # Check if directory exists and is writable - dir_writable = False - if os.path.exists(db_dir): - try: - test_file = os.path.join(db_dir, '.write_test') - with open(test_file, 'w') as f: - f.write('test') - os.remove(test_file) - dir_writable = True - except (IOError, PermissionError): - print(f"Directory {db_dir} exists but is not writable") - dir_writable = False - - # If directory doesn't exist or isn't writable, try to create it - if not os.path.exists(db_dir) or not dir_writable: - try: - os.makedirs(db_dir, exist_ok=True) - # Test if we can write to the directory - test_file = os.path.join(db_dir, '.write_test') - with open(test_file, 'w') as f: - f.write('test') - os.remove(test_file) - print(f"Created directory: {db_dir}") - dir_writable = True - except Exception as e: - print(f"Warning: Could not create or write to directory {db_dir}: {e}") - # Fallback to using a local database file - db_path = 'cache.db' - print(f"Using local database file: {db_path}") - - try: - conn = sqlite3.connect(db_path) - cursor = conn.cursor() - cursor.execute(''' + # Create cache table + cursor.execute(''' CREATE TABLE IF NOT EXISTS cache ( - url TEXT, - route TEXT, - data TEXT, - timestamp INTEGER, - PRIMARY KEY (url, route) + id INTEGER PRIMARY KEY AUTOINCREMENT, + url TEXT NOT NULL, + route TEXT NOT NULL, + data TEXT NOT NULL, + created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP, + UNIQUE(url, route) ) - ''') - conn.commit() - conn.close() - print(f"Database initialized at {db_path}") - # Update the global DB_PATH - DB_PATH = db_path - except sqlite3.OperationalError as e: - print(f"Error initializing database at {db_path}: {e}") - # Fallback to using a local database file if the mounted volume has permission issues - db_path = 'cache.db' - print(f"Falling back to local database file: {db_path}") - try: - conn = sqlite3.connect(db_path) - cursor = conn.cursor() - cursor.execute(''' - CREATE TABLE IF NOT EXISTS cache ( - url TEXT, - route TEXT, - data TEXT, - timestamp INTEGER, - PRIMARY KEY (url, route) - ) - ''') - conn.commit() - conn.close() - print(f"Local database initialized at {db_path}") - # Update the global DB_PATH - DB_PATH = db_path - except sqlite3.OperationalError as e2: - print(f"Error initializing local database: {e2}") - raise + ''') -# Define the database path -DB_PATH = '/db/cache.db' + # Create index for faster lookups + cursor.execute(''' + CREATE INDEX IF NOT EXISTS idx_cache_url_route + ON cache(url, route) + ''') -# Get cached data if it exists and is not older than the expiry time -def get_cached_data(url, route): - conn = sqlite3.connect(DB_PATH) - cursor = conn.cursor() - cache_expiry = int(time.time()) - (CACHE_EXPIRY_HOURS * 60 * 60) # Convert hours to seconds - cursor.execute( - "SELECT data FROM cache WHERE url = ? AND route = ? AND timestamp > ?", - (url, route, cache_expiry) - ) - result = cursor.fetchone() - conn.close() + # Create index for cleanup operations + cursor.execute(''' + CREATE INDEX IF NOT EXISTS idx_cache_created_at + ON cache(created_at) + ''') - if result: - print(f"Cache hit for {url} on route {route}") - return json.loads(result[0]) - return None - -# Save data to cache -def save_to_cache(url, route, data): - conn = sqlite3.connect(DB_PATH) - cursor = conn.cursor() - timestamp = int(time.time()) - - # Convert data to JSON string - data_json = json.dumps(data) - - cursor.execute( - "INSERT OR REPLACE INTO cache (url, route, data, timestamp) VALUES (?, ?, ?, ?)", - (url, route, data_json, timestamp) - ) conn.commit() conn.close() - print(f"Saved to cache: {url} on route {route}") + print("Database initialized") -# Function to clean up old cache entries -def cleanup_old_cache_entries(): +def get_cached_data(url, route): + """Get cached data for a URL and route""" try: - print(f"Running scheduled cache cleanup (entries older than {CACHE_EXPIRY_HOURS} hours, pagination: {CACHE_EXPIRY_HOURS * 31} hours)") - conn = sqlite3.connect(DB_PATH) + conn = sqlite3.connect('/db/cache.db') cursor = conn.cursor() - # Calculate the timestamp for entries older than the expiry time - expiry_timestamp = int(time.time()) - (CACHE_EXPIRY_HOURS * 60 * 60) - pagination_expiry_timestamp = int(time.time()) - (CACHE_EXPIRY_HOURS * 31 * 60 * 60) + cursor.execute(''' + SELECT data, created_at FROM cache + WHERE url = ? AND route = ? + ''', (url, route)) - # Get count of entries to be deleted (non-pagination) - cursor.execute("SELECT COUNT(*) FROM cache WHERE route != 'pagination' AND timestamp < ?", (expiry_timestamp,)) - count_non_pagination = cursor.fetchone()[0] + result = cursor.fetchone() + conn.close() - # Get count of pagination entries to be deleted - cursor.execute("SELECT COUNT(*) FROM cache WHERE route = 'pagination' AND timestamp < ?", (pagination_expiry_timestamp,)) - count_pagination = cursor.fetchone()[0] + if result: + data, created_at = result + created_time = datetime.fromisoformat(created_at) - # Delete old non-pagination entries - cursor.execute("DELETE FROM cache WHERE route != 'pagination' AND timestamp < ?", (expiry_timestamp,)) + # Check if cache is still valid + if datetime.now() - created_time < timedelta(hours=CACHE_EXPIRY_HOURS): + return json.loads(data) - # Delete old pagination entries - cursor.execute("DELETE FROM cache WHERE route = 'pagination' AND timestamp < ?", (pagination_expiry_timestamp,)) + return None + except Exception as e: + print(f"Error getting cached data: {e}") + return None + +def save_to_cache(url, route, data): + """Save data to cache""" + try: + conn = sqlite3.connect('/db/cache.db') + cursor = conn.cursor() + + cursor.execute(''' + INSERT OR REPLACE INTO cache (url, route, data, created_at) + VALUES (?, ?, ?, CURRENT_TIMESTAMP) + ''', (url, route, json.dumps(data))) conn.commit() conn.close() - - print(f"Cache cleanup completed: {count_non_pagination} non-pagination entries and {count_pagination} pagination entries removed") except Exception as e: - print(f"Error during cache cleanup: {e}") + print(f"Error saving to cache: {e}") + +def cleanup_old_cache_entries(): + """Clean up old cache entries""" + try: + conn = sqlite3.connect('/db/cache.db') + cursor = conn.cursor() + + # Delete entries older than CACHE_EXPIRY_HOURS + cutoff_time = datetime.now() - timedelta(hours=CACHE_EXPIRY_HOURS) + + cursor.execute(''' + DELETE FROM cache + WHERE created_at < ? + ''', (cutoff_time.isoformat(),)) + + deleted_count = cursor.rowcount + conn.commit() + conn.close() + + print(f"Cleaned up {deleted_count} old cache entries") + except Exception as e: + print(f"Error cleaning up cache: {e}") + +# Initialize database +init_db() # Initialize scheduler for periodic cache cleanup scheduler = BackgroundScheduler() @@ -412,74 +397,81 @@ scheduler.add_job( replace_existing=True ) -# Initialize database on startup -init_db() - -# Start the scheduler when the application starts @app.on_event("startup") def start_scheduler(): scheduler.start() print(f"Cache cleanup scheduler started with cron: {CLEANUP_CRON}") - print(f"Cache entries will expire after {CACHE_EXPIRY_HOURS} hours") -# Shutdown the scheduler when the application stops @app.on_event("shutdown") def shutdown_scheduler(): scheduler.shutdown(wait=False) print("Cache cleanup scheduler stopped") async def wait_for_network_idle(page): - """Wait until no network requests are in flight""" - await page.waitForNetworkIdle(idleTime=500, timeout=30000) + """Wait for network to be idle""" + await page.wait_for_load_state('networkidle') @app.head("/") async def health_check(): - return {"status": "ok"} + return {"status": "healthy"} async def safe_browser_operation(url, operation_func): - """Safely perform a browser operation with proper cleanup and resource limits""" - async with operation_semaphore: # Limit concurrent operations - async with get_browser() as browser: - page = None + """Safely perform a browser operation with proper resource management""" + async with operation_semaphore: + browser = None + context = None + page = None + + try: + # Get browser from pool or create new one try: - # Create a new page - page = await browser.newPage() + browser = await asyncio.wait_for(browser_pool.get(), timeout=10.0) + except asyncio.TimeoutError: + print("Timeout getting browser from pool, creating new one") + browser = await create_browser() - # Set reasonable viewport - await page.setViewport({'width': 1280, 'height': 800}) + # Create context and page + context = await browser.new_context( + user_agent=CUSTOM_USER_AGENT, + viewport={'width': 1920, 'height': 1080}, + ignore_https_errors=True, + ) - # Set user agent - await page.setUserAgent(CUSTOM_USER_AGENT) + page = await context.new_page() - # Set reasonable timeout - page.setDefaultNavigationTimeout(30000) + # Set up request interception for better performance + await page.route("**/*", lambda route: route.abort() + if route.request.resource_type in ['image', 'stylesheet', 'font', 'media'] + else route.continue_()) - # Enable request interception to block unnecessary resources - await page.setRequestInterception(True) + # Perform the operation + result = await operation_func(page) + return result - async def intercept(request): - # Block unnecessary resource types - if request.resourceType in ['image', 'media', 'font', 'stylesheet']: - await request.abort() - else: - await request.continue_() - - page.on('request', lambda req: asyncio.ensure_future(intercept(req))) - - # Perform the operation - result = await operation_func(page) - - return result - except Exception as e: - print(f"Error during browser operation: {str(e)}") - raise - finally: + except Exception as e: + print(f"Error in browser operation: {e}") + raise + finally: + # Cleanup + if page: try: - # Ensure page is properly closed - if page: - await page.close() - except Exception as e: - print(f"Error closing page: {str(e)}") + await page.close() + except: + pass + if context: + try: + await context.close() + except: + pass + if browser: + try: + # Return browser to pool if it's still healthy + if not browser_pool.full(): + await browser_pool.put(browser) + else: + await cleanup_browser(browser) + except: + pass @app.get("/") async def visit_url(url: str, x_api_key: Optional[str] = Header(None)): @@ -496,12 +488,9 @@ async def visit_url(url: str, x_api_key: Optional[str] = Header(None)): return cached_result try: - print(f"Visiting URL: {decoded_url}") - - # Define the operation to perform with the browser async def visit_operation(page): try: - response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) + response = await page.goto(decoded_url, wait_until='networkidle', timeout=30000) if not response: print(f"Warning: No response object returned for {decoded_url}") @@ -515,14 +504,10 @@ async def visit_url(url: str, x_api_key: Optional[str] = Header(None)): content = await page.content() return {"status": "partial", "content": content, "error": str(e)} except: - raise HTTPException(status_code=500, detail=f"Failed to get page content: {str(e)}") + raise Exception(f"Failed to get page content: {str(e)}") - # Perform the operation result = await safe_browser_operation(decoded_url, visit_operation) - - # Save to cache save_to_cache(decoded_url, "visit", result) - return result except Exception as e: @@ -545,12 +530,9 @@ async def extract_seo(url: str, x_api_key: Optional[str] = Header(None)): return cached_result try: - print(f"Extracting SEO from: {decoded_url}") - - # Define the operation to perform with the browser async def seo_operation(page): try: - response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) + response = await page.goto(decoded_url, wait_until='networkidle', timeout=30000) # Extract SEO information seo_data = await page.evaluate('''() => { @@ -609,12 +591,8 @@ async def extract_seo(url: str, x_api_key: Optional[str] = Header(None)): print(f"Error during SEO extraction: {e}") return {"status": "error", "url": decoded_url, "error": str(e)} - # Perform the operation result = await safe_browser_operation(decoded_url, seo_operation) - - # Save to cache save_to_cache(decoded_url, "seo", result) - return result except Exception as e: @@ -636,52 +614,48 @@ async def extract_meta_tags(url: str, x_api_key: Optional[str] = Header(None)): return cached_result try: - print(f"Extracting meta tags from: {decoded_url}") - - # Define the operation to perform with the browser async def meta_operation(page): try: - response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) + response = await page.goto(decoded_url, wait_until='networkidle', timeout=30000) - # Extract all meta tags - meta_tags = await page.evaluate('''() => { - const metas = Array.from(document.querySelectorAll('meta')); - return metas.map(meta => { + # Extract meta tags using Playwright + meta_data = await page.evaluate('''() => { + const data = { + meta_tags: [], + open_graph: {}, + twitter_card: {}, + title: document.title || '' + }; + + // Extract all meta tags + document.querySelectorAll('meta').forEach(meta => { const attributes = {}; - Array.from(meta.attributes).forEach(attr => { + for (let attr of meta.attributes) { attributes[attr.name] = attr.value; - }); - return attributes; + } + data.meta_tags.push(attributes); }); - }''') - # Extract Open Graph tags - og_tags = await page.evaluate('''() => { - const ogTags = {}; - document.querySelectorAll('meta[property^="og:"]').forEach(tag => { - const property = tag.getAttribute('property'); - ogTags[property] = tag.getAttribute('content'); + // Extract Open Graph tags + document.querySelectorAll('meta[property^="og:"]').forEach(meta => { + data.open_graph[meta.getAttribute('property')] = meta.getAttribute('content'); }); - return ogTags; - }''') - # Extract Twitter card tags - twitter_tags = await page.evaluate('''() => { - const twitterTags = {}; - document.querySelectorAll('meta[name^="twitter:"]').forEach(tag => { - const name = tag.getAttribute('name'); - twitterTags[name] = tag.getAttribute('content'); + // Extract Twitter card tags + document.querySelectorAll('meta[name^="twitter:"]').forEach(meta => { + data.twitter_card[meta.getAttribute('name')] = meta.getAttribute('content'); }); - return twitterTags; + + return data; }''') result = { "status": "success", "url": decoded_url, - "meta_tags": meta_tags, - "open_graph": og_tags, - "twitter_card": twitter_tags, - "title": await page.title() + "meta_tags": meta_data['meta_tags'], + "open_graph": meta_data['open_graph'], + "twitter_card": meta_data['twitter_card'], + "title": meta_data['title'] } return result @@ -690,12 +664,8 @@ async def extract_meta_tags(url: str, x_api_key: Optional[str] = Header(None)): print(f"Error during meta tag extraction: {e}") return {"status": "error", "url": decoded_url, "error": str(e)} - # Perform the operation result = await safe_browser_operation(decoded_url, meta_operation) - - # Save to cache save_to_cache(decoded_url, "meta", result) - return result except Exception as e: @@ -703,18 +673,22 @@ async def extract_meta_tags(url: str, x_api_key: Optional[str] = Header(None)): @app.get("/cache/clear") async def clear_cache(x_api_key: Optional[str] = Header(None)): - """Clear the entire cache database""" + """Clear all cached data""" # Validate API key if not x_api_key or x_api_key != API_KEY: raise HTTPException(status_code=401, detail="Invalid API key") - conn = sqlite3.connect(DB_PATH) - cursor = conn.cursor() - cursor.execute("DELETE FROM cache") - conn.commit() - conn.close() + try: + conn = sqlite3.connect('/db/cache.db') + cursor = conn.cursor() + cursor.execute('DELETE FROM cache') + deleted_count = cursor.rowcount + conn.commit() + conn.close() - return {"status": "success", "message": "Cache cleared successfully"} + return {"status": "success", "message": f"Cleared {deleted_count} cache entries"} + except Exception as e: + raise HTTPException(status_code=500, detail=str(e)) @app.get("/cache/stats") async def cache_stats(x_api_key: Optional[str] = Header(None)): @@ -724,79 +698,103 @@ async def cache_stats(x_api_key: Optional[str] = Header(None)): raise HTTPException(status_code=401, detail="Invalid API key") try: - conn = sqlite3.connect(DB_PATH) + conn = sqlite3.connect('/db/cache.db') cursor = conn.cursor() # Get total count - cursor.execute("SELECT COUNT(*) FROM cache") + cursor.execute('SELECT COUNT(*) FROM cache') total_count = cursor.fetchone()[0] # Get count by route - cursor.execute("SELECT route, COUNT(*) FROM cache GROUP BY route") + cursor.execute(''' + SELECT route, COUNT(*) as count + FROM cache + GROUP BY route + ''') route_counts = dict(cursor.fetchall()) # Get oldest and newest entries - cursor.execute("SELECT MIN(timestamp), MAX(timestamp) FROM cache") - min_time, max_time = cursor.fetchone() + cursor.execute(''' + SELECT MIN(created_at), MAX(created_at) + FROM cache + ''') + oldest, newest = cursor.fetchone() + + # Get database size + cursor.execute('PRAGMA page_count') + page_count = cursor.fetchone()[0] + cursor.execute('PRAGMA page_size') + page_size = cursor.fetchone()[0] + db_size = page_count * page_size conn.close() return { + "status": "success", "total_entries": total_count, "route_counts": route_counts, - "oldest_entry": min_time, - "newest_entry": max_time + "oldest_entry": oldest, + "newest_entry": newest, + "database_size_bytes": db_size, + "cache_expiry_hours": CACHE_EXPIRY_HOURS } except Exception as e: raise HTTPException(status_code=500, detail=str(e)) @app.get("/status") async def system_status(x_api_key: Optional[str] = Header(None)): - """Get system status and browser pool information""" + """Get system status and health information""" # Validate API key if not x_api_key or x_api_key != API_KEY: raise HTTPException(status_code=401, detail="Invalid API key") try: # Get system information - process = psutil.Process() - memory_info = process.memory_info() + cpu_percent = psutil.cpu_percent(interval=1) + memory = psutil.virtual_memory() + disk = psutil.disk_usage('/') # Get browser pool information pool_size = browser_pool.qsize() active_browser_count = len(active_browsers) - # Calculate browser ages - browser_ages = [] - for browser, creation_time in browser_creation_times.items(): - age = time.time() - creation_time - browser_ages.append(age) + # Get cache statistics + conn = sqlite3.connect('/db/cache.db') + cursor = conn.cursor() + cursor.execute('SELECT COUNT(*) FROM cache') + cache_count = cursor.fetchone()[0] + conn.close() return { + "status": "success", "system": { - "cpu_percent": process.cpu_percent(), - "memory_mb": memory_info.rss / 1024 / 1024, - "memory_percent": process.memory_percent(), - "open_files": len(process.open_files()), - "connections": len(process.connections()), - "threads": process.num_threads() + "cpu_percent": cpu_percent, + "memory_percent": memory.percent, + "memory_available_gb": round(memory.available / (1024**3), 2), + "disk_percent": disk.percent, + "disk_free_gb": round(disk.free / (1024**3), 2) }, "browser_pool": { "pool_size": pool_size, - "active_browsers": active_browser_count, "max_browsers": MAX_BROWSERS, - "browser_ttl_seconds": BROWSER_TTL, - "browser_ages_seconds": browser_ages, - "concurrent_operations_limit": MAX_CONCURRENT_OPERATIONS + "active_browsers": active_browser_count, + "browser_ttl_seconds": BROWSER_TTL }, - "timestamp": time.time() + "cache": { + "total_entries": cache_count, + "expiry_hours": CACHE_EXPIRY_HOURS + }, + "rate_limiting": { + "requests_per_minute": RATE_LIMIT_MINUTE, + "window_seconds": RATE_LIMIT_WINDOW + } } except Exception as e: raise HTTPException(status_code=500, detail=str(e)) @app.post("/emergency-cleanup") async def emergency_cleanup(x_api_key: Optional[str] = Header(None)): - """Force emergency cleanup of all browsers""" + """Emergency cleanup endpoint to force cleanup all browsers""" # Validate API key if not x_api_key or x_api_key != API_KEY: raise HTTPException(status_code=401, detail="Invalid API key") @@ -809,60 +807,14 @@ async def emergency_cleanup(x_api_key: Optional[str] = Header(None)): @asynccontextmanager async def get_browser(): - """Get a browser from the pool or create a new one if needed""" + """Context manager for getting a browser from the pool""" browser = None try: - # Try to get a browser from the pool with timeout - try: - browser = await asyncio.wait_for(browser_pool.get(), timeout=30.0) - except (asyncio.QueueEmpty, asyncio.TimeoutError): - # If pool is empty or timeout, create a new browser if under the limit - async with browser_lock: - current_browser_count = len(active_browsers) - if current_browser_count < MAX_BROWSERS: - browser = await create_browser() - else: - # If at limit, wait for a browser to become available with timeout - try: - browser = await asyncio.wait_for(browser_pool.get(), timeout=60.0) - except asyncio.TimeoutError: - # Emergency cleanup and create new browser - print("Emergency: Timeout waiting for browser, forcing cleanup") - await force_cleanup_all_browsers() - browser = await create_browser() - - yield browser - except Exception as e: - print(f"Error in get_browser: {e}") - # Emergency cleanup if we can't get a browser - await force_cleanup_all_browsers() - browser = await create_browser() + browser = await browser_pool.get() yield browser finally: - # Return browser to pool if it's still viable if browser: - try: - # Quick check if browser is still usable - await browser.pages() - # Check if browser is too old - if time.time() - browser_creation_times.get(browser, 0) > BROWSER_TTL: - print(f"Recycling old browser in get_browser (age: {time.time() - browser_creation_times.get(browser, 0):.0f}s)") - await cleanup_browser(browser) - browser = await create_browser() - - # Only put back if pool is not full - if not browser_pool.full(): - await browser_pool.put(browser) - else: - # Pool is full, cleanup this browser - await cleanup_browser(browser) - except Exception as e: - print(f"Browser health check failed in get_browser: {e}") - # If browser is not usable, close it and create a new one - await cleanup_browser(browser) - if not browser_pool.full(): - new_browser = await create_browser() - await browser_pool.put(new_browser) + await browser_pool.put(browser) if __name__ == "__main__": import uvicorn diff --git a/Dockers/puppeteer-api/requirements.txt b/Dockers/puppeteer-api/requirements.txt index df34eee..614b279 100644 --- a/Dockers/puppeteer-api/requirements.txt +++ b/Dockers/puppeteer-api/requirements.txt @@ -1,7 +1,7 @@ -fastapi==0.68.1 -uvicorn==0.15.0 -pyppeteer==1.0.2 +fastapi==0.104.1 +uvicorn==0.24.0 +playwright==1.40.0 psutil==6.0.0 -apscheduler +apscheduler==3.10.4 aiohttp==3.9.1 beautifulsoup4==4.12.2 \ No newline at end of file diff --git a/Dockers/puppeteer-api/test_playwright_migration.py b/Dockers/puppeteer-api/test_playwright_migration.py new file mode 100644 index 0000000..dc0cea1 --- /dev/null +++ b/Dockers/puppeteer-api/test_playwright_migration.py @@ -0,0 +1,120 @@ +#!/usr/bin/env python3 +""" +Test script to verify Playwright migration works correctly +""" + +import asyncio +import sys +import os + +# Add the current directory to Python path +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) + +from playwright.async_api import async_playwright + +async def test_playwright_installation(): + """Test that Playwright is properly installed and can launch a browser""" + print("Testing Playwright installation...") + + try: + async with async_playwright() as p: + browser = await p.chromium.launch(headless=True) + context = await browser.new_context() + page = await context.new_page() + + # Test basic navigation + response = await page.goto('https://example.com', wait_until='networkidle', timeout=10000) + title = await page.title() + + print(f"โœ… Successfully navigated to example.com") + print(f" Title: {title}") + print(f" Status: {response.status if response else 'No response'}") + + await context.close() + await browser.close() + + return True + + except Exception as e: + print(f"โŒ Playwright test failed: {e}") + return False + +async def test_browser_utils(): + """Test the updated browser_utils module""" + print("\nTesting browser_utils module...") + + try: + from app.utils.browser_utils import safe_browser_operation + + async def test_operation(page): + await page.goto('https://example.com', wait_until='networkidle', timeout=10000) + title = await page.title() + return {"status": "success", "title": title} + + result = await safe_browser_operation('https://example.com', test_operation) + + if result.get('status') == 'success': + print(f"โœ… browser_utils test passed") + print(f" Title: {result.get('title')}") + return True + else: + print(f"โŒ browser_utils test failed: {result}") + return False + + except Exception as e: + print(f"โŒ browser_utils test failed: {e}") + return False + +async def test_browser_service(): + """Test the updated browser service""" + print("\nTesting browser service...") + + try: + from app.services.browser import visit_url_service + + result = await visit_url_service('https://example.com') + + if result.get('status') == 'success': + print(f"โœ… browser service test passed") + print(f" Content length: {len(result.get('content', ''))}") + return True + else: + print(f"โŒ browser service test failed: {result}") + return False + + except Exception as e: + print(f"โŒ browser service test failed: {e}") + return False + +async def main(): + """Run all tests""" + print("๐Ÿงช Running Playwright migration tests...\n") + + tests = [ + test_playwright_installation, + test_browser_utils, + test_browser_service, + ] + + results = [] + for test in tests: + try: + result = await test() + results.append(result) + except Exception as e: + print(f"โŒ Test {test.__name__} failed with exception: {e}") + results.append(False) + + print(f"\n๐Ÿ“Š Test Results:") + print(f" Passed: {sum(results)}/{len(results)}") + + if all(results): + print("๐ŸŽ‰ All tests passed! Playwright migration is working correctly.") + return 0 + else: + print("โŒ Some tests failed. Please check the errors above.") + return 1 + +if __name__ == "__main__": + exit_code = asyncio.run(main()) + sys.exit(exit_code) \ No newline at end of file diff --git a/Dockers/puppeteer-api/version b/Dockers/puppeteer-api/version index b9bc2fd..56fea8a 100644 --- a/Dockers/puppeteer-api/version +++ b/Dockers/puppeteer-api/version @@ -1 +1 @@ -latest \ No newline at end of file +3.0.0 \ No newline at end of file