From de3919a4b516aa7081c345092620de077d487470 Mon Sep 17 00:00:00 2001 From: Bram Kelchtermans Date: Fri, 2 May 2025 21:15:30 +0200 Subject: [PATCH] split up application --- Dockers/puppeteer-api/Dockerfile | 2 +- Dockers/puppeteer-api/app/__init__.py | 1 + Dockers/puppeteer-api/app/config.py | 18 + Dockers/puppeteer-api/app/database.py | 144 +++++ Dockers/puppeteer-api/app/main.py | 39 ++ Dockers/puppeteer-api/app/routes/__init__.py | 1 + Dockers/puppeteer-api/app/routes/browser.py | 117 ++++ Dockers/puppeteer-api/app/routes/cache.py | 24 + Dockers/puppeteer-api/app/routes/health.py | 7 + .../puppeteer-api/app/services/__init__.py | 1 + Dockers/puppeteer-api/app/services/browser.py | 564 ++++++++++++++++++ Dockers/puppeteer-api/app/services/cache.py | 55 ++ Dockers/puppeteer-api/app/utils/__init__.py | 1 + .../puppeteer-api/app/utils/browser_utils.py | 43 ++ Dockers/puppeteer-api/run.py | 4 + 15 files changed, 1020 insertions(+), 1 deletion(-) create mode 100644 Dockers/puppeteer-api/app/__init__.py create mode 100644 Dockers/puppeteer-api/app/config.py create mode 100644 Dockers/puppeteer-api/app/database.py create mode 100644 Dockers/puppeteer-api/app/main.py create mode 100644 Dockers/puppeteer-api/app/routes/__init__.py create mode 100644 Dockers/puppeteer-api/app/routes/browser.py create mode 100644 Dockers/puppeteer-api/app/routes/cache.py create mode 100644 Dockers/puppeteer-api/app/routes/health.py create mode 100644 Dockers/puppeteer-api/app/services/__init__.py create mode 100644 Dockers/puppeteer-api/app/services/browser.py create mode 100644 Dockers/puppeteer-api/app/services/cache.py create mode 100644 Dockers/puppeteer-api/app/utils/__init__.py create mode 100644 Dockers/puppeteer-api/app/utils/browser_utils.py create mode 100644 Dockers/puppeteer-api/run.py diff --git a/Dockers/puppeteer-api/Dockerfile b/Dockers/puppeteer-api/Dockerfile index d849b37..18c43c0 100644 --- a/Dockers/puppeteer-api/Dockerfile +++ b/Dockers/puppeteer-api/Dockerfile @@ -50,4 +50,4 @@ USER puppeteer EXPOSE 8000 # Run the application -CMD ["python", "main.py"] +CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000"] diff --git a/Dockers/puppeteer-api/app/__init__.py b/Dockers/puppeteer-api/app/__init__.py new file mode 100644 index 0000000..2bcdea1 --- /dev/null +++ b/Dockers/puppeteer-api/app/__init__.py @@ -0,0 +1 @@ +# This file is intentionally left empty to make the directory a Python package \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/config.py b/Dockers/puppeteer-api/app/config.py new file mode 100644 index 0000000..03ba24d --- /dev/null +++ b/Dockers/puppeteer-api/app/config.py @@ -0,0 +1,18 @@ +import os + +# Get API key from environment variable +API_KEY = os.getenv('API_KEY') +if not API_KEY: + raise ValueError("API_KEY environment variable must be set") + +# Get cache expiry time from environment variable (default: 36 hours) +CACHE_EXPIRY_HOURS = int(os.getenv('CACHE_EXPIRY_HOURS', '36')) + +# Get cleanup cron schedule from environment variable (default: every day at 3 AM) +CLEANUP_CRON = os.getenv('CLEANUP_CRON', '0 3 * * *') + +# Define custom user agent +CUSTOM_USER_AGENT = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/69.0.3497.100 Safari/537.36' + +# Database path +DB_PATH = '/db/cache.db' \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/database.py b/Dockers/puppeteer-api/app/database.py new file mode 100644 index 0000000..6061d8f --- /dev/null +++ b/Dockers/puppeteer-api/app/database.py @@ -0,0 +1,144 @@ +import os +import sqlite3 +import time +import json +from datetime import datetime +from app.config import DB_PATH, CACHE_EXPIRY_HOURS + +# Initialize SQLite database +def init_db(): + global DB_PATH + + # Try to use the mounted volume first + db_path = '/db/cache.db' + db_dir = os.path.dirname(db_path) + + # Check if directory exists and is writable + dir_writable = False + if os.path.exists(db_dir): + try: + test_file = os.path.join(db_dir, '.write_test') + with open(test_file, 'w') as f: + f.write('test') + os.remove(test_file) + dir_writable = True + except (IOError, PermissionError): + print(f"Directory {db_dir} exists but is not writable") + dir_writable = False + + # If directory doesn't exist or isn't writable, try to create it + if not os.path.exists(db_dir) or not dir_writable: + try: + os.makedirs(db_dir, exist_ok=True) + # Test if we can write to the directory + test_file = os.path.join(db_dir, '.write_test') + with open(test_file, 'w') as f: + f.write('test') + os.remove(test_file) + print(f"Created directory: {db_dir}") + dir_writable = True + except Exception as e: + print(f"Warning: Could not create or write to directory {db_dir}: {e}") + # Fallback to using a local database file + db_path = 'cache.db' + print(f"Using local database file: {db_path}") + + try: + conn = sqlite3.connect(db_path) + cursor = conn.cursor() + cursor.execute(''' + CREATE TABLE IF NOT EXISTS cache ( + url TEXT, + route TEXT, + data TEXT, + timestamp INTEGER, + PRIMARY KEY (url, route) + ) + ''') + conn.commit() + conn.close() + print(f"Database initialized at {db_path}") + # Update the global DB_PATH + DB_PATH = db_path + except sqlite3.OperationalError as e: + print(f"Error initializing database at {db_path}: {e}") + # Fallback to using a local database file if the mounted volume has permission issues + db_path = 'cache.db' + print(f"Falling back to local database file: {db_path}") + try: + conn = sqlite3.connect(db_path) + cursor = conn.cursor() + cursor.execute(''' + CREATE TABLE IF NOT EXISTS cache ( + url TEXT, + route TEXT, + data TEXT, + timestamp INTEGER, + PRIMARY KEY (url, route) + ) + ''') + conn.commit() + conn.close() + print(f"Local database initialized at {db_path}") + # Update the global DB_PATH + DB_PATH = db_path + except sqlite3.OperationalError as e2: + print(f"Error initializing local database: {e2}") + raise + +# Get cached data if it exists and is not older than the expiry time +def get_cached_data(url, route): + conn = sqlite3.connect(DB_PATH) + cursor = conn.cursor() + cache_expiry = int(time.time()) - (CACHE_EXPIRY_HOURS * 60 * 60) # Convert hours to seconds + cursor.execute( + "SELECT data FROM cache WHERE url = ? AND route = ? AND timestamp > ?", + (url, route, cache_expiry) + ) + result = cursor.fetchone() + conn.close() + + if result: + print(f"Cache hit for {url} on route {route}") + return json.loads(result[0]) + return None + +# Save data to cache +def save_to_cache(url, route, data): + conn = sqlite3.connect(DB_PATH) + cursor = conn.cursor() + timestamp = int(time.time()) + + # Convert data to JSON string + data_json = json.dumps(data) + + cursor.execute( + "INSERT OR REPLACE INTO cache (url, route, data, timestamp) VALUES (?, ?, ?, ?)", + (url, route, data_json, timestamp) + ) + conn.commit() + conn.close() + print(f"Saved to cache: {url} on route {route}") + +# Function to clean up old cache entries +def cleanup_old_cache_entries(): + try: + print(f"Running scheduled cache cleanup (entries older than {CACHE_EXPIRY_HOURS} hours)") + conn = sqlite3.connect(DB_PATH) + cursor = conn.cursor() + + # Calculate the timestamp for entries older than the expiry time + expiry_timestamp = int(time.time()) - (CACHE_EXPIRY_HOURS * 60 * 60) + + # Get count of entries to be deleted + cursor.execute("SELECT COUNT(*) FROM cache WHERE timestamp < ?", (expiry_timestamp,)) + count = cursor.fetchone()[0] + + # Delete old entries + cursor.execute("DELETE FROM cache WHERE timestamp < ?", (expiry_timestamp,)) + conn.commit() + conn.close() + + print(f"Cache cleanup completed: {count} entries removed") + except Exception as e: + print(f"Error during cache cleanup: {e}") \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/main.py b/Dockers/puppeteer-api/app/main.py new file mode 100644 index 0000000..51827c7 --- /dev/null +++ b/Dockers/puppeteer-api/app/main.py @@ -0,0 +1,39 @@ +from fastapi import FastAPI +from apscheduler.schedulers.background import BackgroundScheduler +from apscheduler.triggers.cron import CronTrigger +from app.database import init_db, cleanup_old_cache_entries +from app.config import CLEANUP_CRON, CACHE_EXPIRY_HOURS +from app.routes import health, browser, cache + +# Create FastAPI app +app = FastAPI() + +# Include routers +app.include_router(health.router) +app.include_router(browser.router) +app.include_router(cache.router) + +# Initialize scheduler for periodic cache cleanup +scheduler = BackgroundScheduler() +scheduler.add_job( + cleanup_old_cache_entries, + CronTrigger.from_crontab(CLEANUP_CRON), + id='cache_cleanup_job', + replace_existing=True +) + +# Initialize database on startup +init_db() + +# Start the scheduler when the application starts +@app.on_event("startup") +def start_scheduler(): + scheduler.start() + print(f"Cache cleanup scheduler started with cron: {CLEANUP_CRON}") + print(f"Cache entries will expire after {CACHE_EXPIRY_HOURS} hours") + +# Shutdown the scheduler when the application stops +@app.on_event("shutdown") +def shutdown_scheduler(): + scheduler.shutdown(wait=False) + print("Cache cleanup scheduler stopped") \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/routes/__init__.py b/Dockers/puppeteer-api/app/routes/__init__.py new file mode 100644 index 0000000..2bcdea1 --- /dev/null +++ b/Dockers/puppeteer-api/app/routes/__init__.py @@ -0,0 +1 @@ +# This file is intentionally left empty to make the directory a Python package \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/routes/browser.py b/Dockers/puppeteer-api/app/routes/browser.py new file mode 100644 index 0000000..db35f04 --- /dev/null +++ b/Dockers/puppeteer-api/app/routes/browser.py @@ -0,0 +1,117 @@ +from fastapi import APIRouter, HTTPException, Header +from typing import Optional +from urllib.parse import unquote +from app.config import API_KEY +from app.database import get_cached_data, save_to_cache +from app.services.browser import ( + visit_url_service, + extract_seo_service, + extract_meta_tags_service, + detect_pagination_service +) + +router = APIRouter() + +@router.get("/") +async def visit_url(url: str, x_api_key: Optional[str] = Header(None)): + # Validate API key + if not x_api_key or x_api_key != API_KEY: + raise HTTPException(status_code=401, detail="Invalid API key") + + # Decode URL if it's encoded + decoded_url = unquote(url) + + # Check cache first + cached_result = get_cached_data(decoded_url, "visit") + if cached_result: + return cached_result + + try: + # Call the service function + result = await visit_url_service(decoded_url) + + # Save to cache + save_to_cache(decoded_url, "visit", result) + + return result + except Exception as e: + print(f"Error visiting URL {decoded_url}: {e}") + raise HTTPException(status_code=500, detail=str(e)) + +@router.get("/seo") +async def extract_seo(url: str, x_api_key: Optional[str] = Header(None)): + """Extract SEO information from a website""" + # Validate API key + if not x_api_key or x_api_key != API_KEY: + raise HTTPException(status_code=401, detail="Invalid API key") + + # Decode URL if it's encoded + decoded_url = unquote(url) + + # Check cache first + cached_result = get_cached_data(decoded_url, "seo") + if cached_result: + return cached_result + + try: + # Call the service function + result = await extract_seo_service(decoded_url) + + # Save to cache + save_to_cache(decoded_url, "seo", result) + + return result + except Exception as e: + raise HTTPException(status_code=500, detail=str(e)) + +@router.get("/meta") +async def extract_meta_tags(url: str, x_api_key: Optional[str] = Header(None)): + """Extract meta tags from a website""" + # Validate API key + if not x_api_key or x_api_key != API_KEY: + raise HTTPException(status_code=401, detail="Invalid API key") + + # Decode URL if it's encoded + decoded_url = unquote(url) + + # Check cache first + cached_result = get_cached_data(decoded_url, "meta") + if cached_result: + return cached_result + + try: + # Call the service function + result = await extract_meta_tags_service(decoded_url) + + # Save to cache + save_to_cache(decoded_url, "meta", result) + + return result + except Exception as e: + raise HTTPException(status_code=500, detail=str(e)) + +@router.get("/pagination") +async def detect_pagination(url: str, x_api_key: Optional[str] = Header(None)): + """Detect pagination on a website and determine the pagination pattern""" + # Validate API key + if not x_api_key or x_api_key != API_KEY: + raise HTTPException(status_code=401, detail="Invalid API key") + + # Decode URL if it's encoded + decoded_url = unquote(url) + + # Check cache first + cached_result = get_cached_data(decoded_url, "pagination") + if cached_result: + return cached_result + + try: + # Call the service function + result = await detect_pagination_service(decoded_url) + + # Save to cache + save_to_cache(decoded_url, "pagination", result) + + return result + except Exception as e: + raise HTTPException(status_code=500, detail=str(e)) \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/routes/cache.py b/Dockers/puppeteer-api/app/routes/cache.py new file mode 100644 index 0000000..e61fc77 --- /dev/null +++ b/Dockers/puppeteer-api/app/routes/cache.py @@ -0,0 +1,24 @@ +from fastapi import APIRouter, HTTPException, Header +from typing import Optional +from app.config import API_KEY +from app.services.cache import clear_cache, get_cache_stats + +router = APIRouter() + +@router.get("/cache/clear") +async def clear_cache_route(x_api_key: Optional[str] = Header(None)): + """Clear the entire cache database""" + # Validate API key + if not x_api_key or x_api_key != API_KEY: + raise HTTPException(status_code=401, detail="Invalid API key") + + return clear_cache() + +@router.get("/cache/stats") +async def cache_stats_route(x_api_key: Optional[str] = Header(None)): + """Get cache statistics""" + # Validate API key + if not x_api_key or x_api_key != API_KEY: + raise HTTPException(status_code=401, detail="Invalid API key") + + return get_cache_stats() \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/routes/health.py b/Dockers/puppeteer-api/app/routes/health.py new file mode 100644 index 0000000..6194522 --- /dev/null +++ b/Dockers/puppeteer-api/app/routes/health.py @@ -0,0 +1,7 @@ +from fastapi import APIRouter + +router = APIRouter() + +@router.head("/") +async def health_check(): + return {"status": "ok"} \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/services/__init__.py b/Dockers/puppeteer-api/app/services/__init__.py new file mode 100644 index 0000000..2bcdea1 --- /dev/null +++ b/Dockers/puppeteer-api/app/services/__init__.py @@ -0,0 +1 @@ +# This file is intentionally left empty to make the directory a Python package \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/services/browser.py b/Dockers/puppeteer-api/app/services/browser.py new file mode 100644 index 0000000..dd819a6 --- /dev/null +++ b/Dockers/puppeteer-api/app/services/browser.py @@ -0,0 +1,564 @@ +from app.utils.browser_utils import safe_browser_operation + +async def visit_url_service(decoded_url): + """Service function to visit a URL and get its content""" + print(f"Visiting URL: {decoded_url}") + + # Define the operation to perform with the browser + async def visit_operation(page): + try: + response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) + if not response: + print(f"Warning: No response object returned for {decoded_url}") + + # Get page content + content = await page.content() + return {"status": "success", "content": content} + except Exception as e: + print(f"Error during page navigation: {e}") + # Try to get content anyway + try: + content = await page.content() + return {"status": "partial", "content": content, "error": str(e)} + except: + raise Exception(f"Failed to get page content: {str(e)}") + + # Perform the operation + return await safe_browser_operation(decoded_url, visit_operation) + +async def extract_seo_service(decoded_url): + """Service function to extract SEO information from a website""" + print(f"Extracting SEO from: {decoded_url}") + + # Define the operation to perform with the browser + async def seo_operation(page): + try: + response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) + + # Extract SEO information + seo_data = await page.evaluate('''() => { + const data = { + title: document.title || '', + description: '', + canonical: '', + h1: [], + h2: [], + images: 0, + links: 0 + }; + + // Get meta description + const metaDescription = document.querySelector('meta[name="description"]'); + if (metaDescription) { + data.description = metaDescription.getAttribute('content') || ''; + } + + // Get canonical link + const canonicalLink = document.querySelector('link[rel="canonical"]'); + if (canonicalLink) { + data.canonical = canonicalLink.getAttribute('href') || ''; + } + + // Get h1 tags + document.querySelectorAll('h1').forEach(h1 => { + const text = h1.innerText.trim(); + if (text) data.h1.push(text); + }); + + // Get h2 tags + document.querySelectorAll('h2').forEach(h2 => { + const text = h2.innerText.trim(); + if (text) data.h2.push(text); + }); + + // Count images + data.images = document.querySelectorAll('img').length; + + // Count links + data.links = document.querySelectorAll('a').length; + + return data; + }''') + + result = { + "status": "success", + "url": decoded_url, + "seo": seo_data + } + + return result + + except Exception as e: + print(f"Error during SEO extraction: {e}") + return {"status": "error", "url": decoded_url, "error": str(e)} + + # Perform the operation + return await safe_browser_operation(decoded_url, seo_operation) + +async def extract_meta_tags_service(decoded_url): + """Service function to extract meta tags from a website""" + print(f"Extracting meta tags from: {decoded_url}") + + # Define the operation to perform with the browser + async def meta_operation(page): + try: + response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) + + # Extract all meta tags + meta_tags = await page.evaluate('''() => { + const metas = Array.from(document.querySelectorAll('meta')); + return metas.map(meta => { + const attributes = {}; + Array.from(meta.attributes).forEach(attr => { + attributes[attr.name] = attr.value; + }); + return attributes; + }); + }''') + + # Extract Open Graph tags + og_tags = await page.evaluate('''() => { + const ogTags = {}; + document.querySelectorAll('meta[property^="og:"]').forEach(tag => { + const property = tag.getAttribute('property'); + ogTags[property] = tag.getAttribute('content'); + }); + return ogTags; + }''') + + # Extract Twitter card tags + twitter_tags = await page.evaluate('''() => { + const twitterTags = {}; + document.querySelectorAll('meta[name^="twitter:"]').forEach(tag => { + const name = tag.getAttribute('name'); + twitterTags[name] = tag.getAttribute('content'); + }); + return twitterTags; + }''') + + result = { + "status": "success", + "url": decoded_url, + "meta_tags": meta_tags, + "open_graph": og_tags, + "twitter_card": twitter_tags, + "title": await page.title() + } + + return result + + except Exception as e: + print(f"Error during meta tag extraction: {e}") + return {"status": "error", "url": decoded_url, "error": str(e)} + + # Perform the operation + return await safe_browser_operation(decoded_url, meta_operation) + +async def detect_pagination_service(decoded_url): + """Service function to detect pagination on a website""" + print(f"Detecting pagination on: {decoded_url}") + + # Define the operation to perform with the browser + async def pagination_operation(page): + try: + # Navigate to the URL + await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) + original_url = page.url + + print(f"Successfully loaded page: {original_url}") + + # Analyze the page for pagination information + pagination_info = await page.evaluate('''() => { + // Find the last page number if available + const findLastPageNumber = () => { + // Get all links on the page + const links = Array.from(document.querySelectorAll('a')); + + // Strategy 1: Find numeric links (page numbers) + const numericLinks = links.filter(link => { + const text = link.innerText.trim(); + return /^[0-9]+$/.test(text) && link.href && link.href !== '#'; + }); + + if (numericLinks.length > 0) { + const numericValues = numericLinks.map(link => parseInt(link.innerText.trim())); + return Math.max(...numericValues); + } + + // Strategy 2: Look for "last page" link + const lastLinks = links.filter(link => { + const text = link.innerText.trim().toLowerCase(); + const classes = (link.className || '').toLowerCase(); + const ariaLabel = (link.getAttribute('aria-label') || '').toLowerCase(); + + return (text === 'last' || + classes.includes('last') || + ariaLabel.includes('last') || + link.getAttribute('rel') === 'last'); + }); + + if (lastLinks.length > 0) { + const lastLink = lastLinks[0]; + const href = lastLink.href; + + // Common patterns: page=X, /page/X, etc. + const pagePatterns = [ + /[?&]page=(\d+)/, + /[?&]p=(\d+)/, + /[?&]pg=(\d+)/, + /\/page\/(\d+)/, + /\/p\/(\d+)/, + /\/paged\/(\d+)/ + ]; + + for (const pattern of pagePatterns) { + const match = href.match(pattern); + if (match && match[1]) { + return parseInt(match[1]); + } + } + } + + // Strategy 3: Analyze all URLs for page numbers + const pageNumbersFromUrls = []; + links.forEach(link => { + if (!link.href || link.href === '#') return; + + // Check for common pagination URL patterns + const patterns = [ + /[?&]page=(\d+)/, + /[?&]p=(\d+)/, + /[?&]pg=(\d+)/, + /\/page\/(\d+)/, + /\/p\/(\d+)/, + /\/paged\/(\d+)/, + /\/pages\/(\d+)/ + ]; + + for (const pattern of patterns) { + const match = link.href.match(pattern); + if (match && match[1]) { + pageNumbersFromUrls.push(parseInt(match[1])); + } + } + }); + + if (pageNumbersFromUrls.length > 0) { + return Math.max(...pageNumbersFromUrls); + } + + return null; + }; + + // Find a pagination link to click + const findPaginationLink = () => { + const links = Array.from(document.querySelectorAll('a')); + + // Try to find a page "2" link first (most reliable) + const page2Link = links.find(link => { + const text = link.innerText.trim(); + return text === '2' && link.href && link.href !== '#'; + }); + + if (page2Link) { + return { element: page2Link, href: page2Link.href, type: 'numeric' }; + } + + // Try common "next page" selectors + const nextSelectors = [ + 'a.next', + 'a.page-next', + 'a[rel="next"]', + 'a[aria-label="Next page"]', + 'a[aria-label="next"]' + ]; + + for (const selector of nextSelectors) { + const element = document.querySelector(selector); + if (element && element.href && element.href !== '#') { + return { element, href: element.href, type: 'next' }; + } + } + + // Look for any link that might be pagination + const paginationLinks = links.filter(link => { + if (!link.href || link.href === '#') return false; + + const text = link.innerText.trim(); + const href = link.href; + + // Check for numeric text or next/prev indicators + const isNumeric = /^[0-9]+$/.test(text) && text !== '1'; + const isNextPrev = /next|prev|previous|older|newer/i.test(text) || + /[»«‹›<>]/.test(text); + + // Check for page parameter in URL + const hasPageParam = /[?&]page=|[?&]p=|[?&]pg=|\/page\/|\/p\//.test(href); + + return (isNumeric || isNextPrev || hasPageParam); + }); + + if (paginationLinks.length > 0) { + const link = paginationLinks[0]; + return { element: link, href: link.href, type: 'other' }; + } + + return null; + }; + + const lastPage = findLastPageNumber(); + const paginationLink = findPaginationLink(); + + if (paginationLink) { + // Click the link + paginationLink.element.click(); + return { + clicked: true, + href: paginationLink.href, + type: paginationLink.type, + lastPage + }; + } + + return { clicked: false, lastPage }; + }''') + + # If no pagination was found or clicked + if not pagination_info.get('clicked', False): + return { + "status": "success", + "url": decoded_url, + "hasPagination": False, + "urlTemplate": None, + "lastPage": pagination_info.get('lastPage') + } + + # Wait for navigation to complete after the click + try: + await page.waitForNavigation({'timeout': 10000, 'waitUntil': 'networkidle2'}) + except Exception as e: + print(f"Navigation timeout: {e}") + + # Get the new URL after clicking + next_page_url = page.url + + # If URL didn't change, pagination might be handled by AJAX + if next_page_url == original_url: + return { + "status": "success", + "url": decoded_url, + "hasPagination": True, + "urlTemplate": "AJAX pagination (URL doesn't change)", + "lastPage": pagination_info.get('lastPage') + } + + print(f"Navigation successful: {original_url} -> {next_page_url}") + + # Analyze the URL structure to determine pagination pattern + url_template = await page.evaluate('''(originalUrl, nextPageUrl) => { + // Helper function to parse URL query parameters + const parseQueryParams = (url) => { + const params = {}; + if (url.includes('?')) { + const queryString = url.split('?')[1].split('#')[0]; + queryString.split('&').forEach(param => { + if (param.includes('=')) { + const [key, value] = param.split('=', 2); + params[key] = value; + } + }); + } + return params; + }; + + // Check for query parameter based pagination + if (nextPageUrl.includes('?')) { + const originalParams = parseQueryParams(originalUrl); + const nextParams = parseQueryParams(nextPageUrl); + + // Find parameters that changed or were added + let paginationParam = null; + + // First check for common pagination parameter names + const commonPaginationParams = ['page', 'p', 'pg', 'paged', 'current_page', 'pagenum', 'pageNumber']; + + for (const key of commonPaginationParams) { + if (key in nextParams && + (!(key in originalParams) || originalParams[key] !== nextParams[key])) { + if (/^\d+$/.test(nextParams[key]) && parseInt(nextParams[key]) > 1) { + paginationParam = key; + break; + } + } + } + + // If no common parameter found, check all parameters + if (!paginationParam) { + for (const [key, value] of Object.entries(nextParams)) { + // Check if parameter is new or changed + if (!(key in originalParams) || originalParams[key] !== value) { + // Check if the value is numeric and could be a page number + if (/^\d+$/.test(value) && parseInt(value) > 1) { + paginationParam = key; + break; + } + } + } + } + + // If we found a pagination parameter + if (paginationParam) { + const baseUrl = nextPageUrl.split('?')[0]; + + // Reconstruct the URL template with all parameters + const queryParts = []; + for (const [key, value] of Object.entries(nextParams)) { + if (key === paginationParam) { + queryParts.push(`${key}={PAGE_NUMBER}`); + } else { + queryParts.push(`${key}=${value}`); + } + } + + return `${baseUrl}?${queryParts.join('&')}`; + } + } + + // Check for path-based pagination + const pathPatterns = ['/page/', '/p/', '/paged/', '/pages/']; + for (const pattern of pathPatterns) { + if (nextPageUrl.includes(pattern)) { + const parts = nextPageUrl.split(pattern); + let template = `${parts[0]}${pattern}{PAGE_NUMBER}`; + + // Add any suffix after the page number + if (parts.length > 1 && parts[1].includes('/')) { + const suffix = parts[1].split('/', 1)[1]; + if (suffix) { + template += `/${suffix}`; + } + } + + return template; + } + } + + // If we couldn't determine the pattern, try to make an educated guess + // For the specific case where a parameter like current_page=2 is added + const originalUrlObj = new URL(originalUrl); + const nextUrlObj = new URL(nextPageUrl); + + // Check if the paths are the same but query params differ + if (originalUrlObj.pathname === nextUrlObj.pathname) { + const originalParams = parseQueryParams(originalUrl); + const nextParams = parseQueryParams(nextPageUrl); + + // Find parameters that exist in next but not in original + const newParams = Object.keys(nextParams).filter(key => !(key in originalParams)); + + // If there's exactly one new parameter and it has a numeric value + if (newParams.length === 1 && /^\d+$/.test(nextParams[newParams[0]])) { + const paginationParam = newParams[0]; + const baseUrl = nextPageUrl.split('?')[0]; + + // Reconstruct the URL template + const queryParts = []; + for (const [key, value] of Object.entries(nextParams)) { + if (key === paginationParam) { + queryParts.push(`${key}={PAGE_NUMBER}`); + } else { + queryParts.push(`${key}=${value}`); + } + } + + return `${baseUrl}?${queryParts.join('&')}`; + } + } + + // If we still couldn't determine the pattern, return both URLs as examples + return `Pattern unclear. Example: ${originalUrl} → ${nextPageUrl}`; + }''', original_url, next_page_url) + + # Try to extract last page number from the next page if we didn't find it on the first page + if not pagination_info.get('lastPage'): + last_page_from_next = await page.evaluate('''() => { + // Get all links on the page + const links = Array.from(document.querySelectorAll('a')); + + // Strategy 1: Find numeric links (page numbers) + const numericLinks = links.filter(link => { + const text = link.innerText.trim(); + return /^[0-9]+$/.test(text) && link.href && link.href !== '#'; + }); + + if (numericLinks.length > 0) { + const numericValues = numericLinks.map(link => parseInt(link.innerText.trim())); + return Math.max(...numericValues); + } + + // Strategy 2: Analyze all URLs for page numbers + const pageNumbersFromUrls = []; + links.forEach(link => { + if (!link.href || link.href === '#') return; + + // Check for common pagination URL patterns + const patterns = [ + /[?&]page=(\d+)/, + /[?&]p=(\d+)/, + /[?&]pg=(\d+)/, + /\/page\/(\d+)/, + /\/p\/(\d+)/, + /\/paged\/(\d+)/, + /\/pages\/(\d+)/ + ]; + + for (const pattern of patterns) { + const match = link.href.match(pattern); + if (match && match[1]) { + pageNumbersFromUrls.push(parseInt(match[1])); + } + } + }); + + if (pageNumbersFromUrls.length > 0) { + return Math.max(...pageNumbersFromUrls); + } + + return null; + }''') + + if last_page_from_next: + pagination_info['lastPage'] = last_page_from_next + + # Check if the pattern is unclear + has_pagination = True + url_string_template = str(url_template) + if url_string_template and url_string_template.startswith("Pattern unclear"): + has_pagination = False + url_template = None + + result = { + "status": "success", + "url": decoded_url, + "hasPagination": has_pagination, + "urlTemplate": url_template, + "lastPage": pagination_info.get('lastPage'), + "originalUrl": original_url, + "nextPageUrl": next_page_url + } + + return result + + except Exception as e: + print(f"Error during pagination detection: {e}") + return { + "status": "error", + "url": decoded_url, + "error": str(e), + "hasPagination": False, + "urlTemplate": None, + "lastPage": None + } + + # Perform the operation + return await safe_browser_operation(decoded_url, pagination_operation) \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/services/cache.py b/Dockers/puppeteer-api/app/services/cache.py new file mode 100644 index 0000000..bfbe0ce --- /dev/null +++ b/Dockers/puppeteer-api/app/services/cache.py @@ -0,0 +1,55 @@ +import sqlite3 +import time +from datetime import datetime +from app.database import DB_PATH, CACHE_EXPIRY_HOURS + +def clear_cache(): + """Clear the entire cache database""" + conn = sqlite3.connect(DB_PATH) + cursor = conn.cursor() + cursor.execute("DELETE FROM cache") + conn.commit() + conn.close() + return {"status": "success", "message": "Cache cleared successfully"} + +def get_cache_stats(): + """Get cache statistics""" + conn = sqlite3.connect(DB_PATH) + cursor = conn.cursor() + + # Get total entries + cursor.execute("SELECT COUNT(*) FROM cache") + total_entries = cursor.fetchone()[0] + + # Get entries by route + cursor.execute("SELECT route, COUNT(*) FROM cache GROUP BY route") + routes = {route: count for route, count in cursor.fetchall()} + + # Get recent entries (last 24 hours) + recent_timestamp = int(time.time()) - (24 * 60 * 60) + cursor.execute("SELECT COUNT(*) FROM cache WHERE timestamp > ?", (recent_timestamp,)) + recent_entries = cursor.fetchone()[0] + + # Get oldest entry timestamp + cursor.execute("SELECT MIN(timestamp) FROM cache") + oldest_timestamp = cursor.fetchone()[0] + oldest_date = datetime.fromtimestamp(oldest_timestamp).isoformat() if oldest_timestamp else None + + # Get newest entry timestamp + cursor.execute("SELECT MAX(timestamp) FROM cache") + newest_timestamp = cursor.fetchone()[0] + newest_date = datetime.fromtimestamp(newest_timestamp).isoformat() if newest_timestamp else None + + conn.close() + + return { + "status": "success", + "stats": { + "total_entries": total_entries, + "entries_by_route": routes, + "recent_entries": recent_entries, + "oldest_entry": oldest_date, + "newest_entry": newest_date, + "cache_expiry_hours": CACHE_EXPIRY_HOURS + } + } \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/utils/__init__.py b/Dockers/puppeteer-api/app/utils/__init__.py new file mode 100644 index 0000000..2bcdea1 --- /dev/null +++ b/Dockers/puppeteer-api/app/utils/__init__.py @@ -0,0 +1 @@ +# This file is intentionally left empty to make the directory a Python package \ No newline at end of file diff --git a/Dockers/puppeteer-api/app/utils/browser_utils.py b/Dockers/puppeteer-api/app/utils/browser_utils.py new file mode 100644 index 0000000..6e7356d --- /dev/null +++ b/Dockers/puppeteer-api/app/utils/browser_utils.py @@ -0,0 +1,43 @@ +from pyppeteer import launch +from app.config import CUSTOM_USER_AGENT + +async def wait_for_network_idle(page): + """Wait until no network requests are in flight""" + await page.waitForNetworkIdle(idleTime=500, timeout=30000) + +async def safe_browser_operation(url, operation_func): + """Safely perform browser operations with proper cleanup""" + browser = None + try: + browser = await launch( + headless=True, + executablePath='/usr/bin/google-chrome', + args=['--no-sandbox', '--disable-setuid-sandbox'], + handleSIGINT=False, + handleSIGTERM=False, + handleSIGHUP=False + ) + + # Create new page with timeout + page = await browser.newPage() + page.setDefaultNavigationTimeout(30000) + + # Set custom user agent + await page.setUserAgent(CUSTOM_USER_AGENT) + + # Call the operation function that uses the page + result = await operation_func(page) + + # Explicitly close the page + await page.close() + + return result + + finally: + # Ensure browser is closed properly + if browser: + try: + await browser.close() + except Exception as e: + print(f"Error closing browser: {e}") + # We don't re-raise here to avoid masking the original error \ No newline at end of file diff --git a/Dockers/puppeteer-api/run.py b/Dockers/puppeteer-api/run.py new file mode 100644 index 0000000..46c8bbd --- /dev/null +++ b/Dockers/puppeteer-api/run.py @@ -0,0 +1,4 @@ +import uvicorn + +if __name__ == "__main__": + uvicorn.run("app.main:app", host="0.0.0.0", port=8000) \ No newline at end of file