Files
projects/Dockers/puppeteer-api/main.py
T
Bram e452511de0
Build and Push Docker Images / build-and-push (push) Successful in 21s
add pagination route
2025-05-02 13:08:12 +02:00

920 lines
36 KiB
Python

from fastapi import FastAPI, HTTPException, Header
from pyppeteer import launch
import os
import asyncio
import json
import re
import sqlite3
import time
from datetime import datetime, timedelta
from typing import Optional, Dict, List, Any
from urllib.parse import unquote
from apscheduler.schedulers.background import BackgroundScheduler
from apscheduler.triggers.cron import CronTrigger
app = FastAPI()
# Get API key from environment variable
API_KEY = os.getenv('API_KEY')
if not API_KEY:
raise ValueError("API_KEY environment variable must be set")
# Get cache expiry time from environment variable (default: 36 hours)
CACHE_EXPIRY_HOURS = int(os.getenv('CACHE_EXPIRY_HOURS', '36'))
# Get cleanup cron schedule from environment variable (default: every day at 3 AM)
CLEANUP_CRON = os.getenv('CLEANUP_CRON', '0 3 * * *')
# Define custom user agent
CUSTOM_USER_AGENT = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/69.0.3497.100 Safari/537.36'
# Initialize SQLite database
def init_db():
global DB_PATH
# Try to use the mounted volume first
db_path = '/db/cache.db'
db_dir = os.path.dirname(db_path)
# Check if directory exists and is writable
dir_writable = False
if os.path.exists(db_dir):
try:
test_file = os.path.join(db_dir, '.write_test')
with open(test_file, 'w') as f:
f.write('test')
os.remove(test_file)
dir_writable = True
except (IOError, PermissionError):
print(f"Directory {db_dir} exists but is not writable")
dir_writable = False
# If directory doesn't exist or isn't writable, try to create it
if not os.path.exists(db_dir) or not dir_writable:
try:
os.makedirs(db_dir, exist_ok=True)
# Test if we can write to the directory
test_file = os.path.join(db_dir, '.write_test')
with open(test_file, 'w') as f:
f.write('test')
os.remove(test_file)
print(f"Created directory: {db_dir}")
dir_writable = True
except Exception as e:
print(f"Warning: Could not create or write to directory {db_dir}: {e}")
# Fallback to using a local database file
db_path = 'cache.db'
print(f"Using local database file: {db_path}")
try:
conn = sqlite3.connect(db_path)
cursor = conn.cursor()
cursor.execute('''
CREATE TABLE IF NOT EXISTS cache (
url TEXT,
route TEXT,
data TEXT,
timestamp INTEGER,
PRIMARY KEY (url, route)
)
''')
conn.commit()
conn.close()
print(f"Database initialized at {db_path}")
# Update the global DB_PATH
DB_PATH = db_path
except sqlite3.OperationalError as e:
print(f"Error initializing database at {db_path}: {e}")
# Fallback to using a local database file if the mounted volume has permission issues
db_path = 'cache.db'
print(f"Falling back to local database file: {db_path}")
try:
conn = sqlite3.connect(db_path)
cursor = conn.cursor()
cursor.execute('''
CREATE TABLE IF NOT EXISTS cache (
url TEXT,
route TEXT,
data TEXT,
timestamp INTEGER,
PRIMARY KEY (url, route)
)
''')
conn.commit()
conn.close()
print(f"Local database initialized at {db_path}")
# Update the global DB_PATH
DB_PATH = db_path
except sqlite3.OperationalError as e2:
print(f"Error initializing local database: {e2}")
raise
# Define the database path
DB_PATH = '/db/cache.db'
# Get cached data if it exists and is not older than the expiry time
def get_cached_data(url, route):
conn = sqlite3.connect(DB_PATH)
cursor = conn.cursor()
cache_expiry = int(time.time()) - (CACHE_EXPIRY_HOURS * 60 * 60) # Convert hours to seconds
cursor.execute(
"SELECT data FROM cache WHERE url = ? AND route = ? AND timestamp > ?",
(url, route, cache_expiry)
)
result = cursor.fetchone()
conn.close()
if result:
print(f"Cache hit for {url} on route {route}")
return json.loads(result[0])
return None
# Save data to cache
def save_to_cache(url, route, data):
conn = sqlite3.connect(DB_PATH)
cursor = conn.cursor()
timestamp = int(time.time())
# Convert data to JSON string
data_json = json.dumps(data)
cursor.execute(
"INSERT OR REPLACE INTO cache (url, route, data, timestamp) VALUES (?, ?, ?, ?)",
(url, route, data_json, timestamp)
)
conn.commit()
conn.close()
print(f"Saved to cache: {url} on route {route}")
# Function to clean up old cache entries
def cleanup_old_cache_entries():
try:
print(f"Running scheduled cache cleanup (entries older than {CACHE_EXPIRY_HOURS} hours)")
conn = sqlite3.connect(DB_PATH)
cursor = conn.cursor()
# Calculate the timestamp for entries older than the expiry time
expiry_timestamp = int(time.time()) - (CACHE_EXPIRY_HOURS * 60 * 60)
# Get count of entries to be deleted
cursor.execute("SELECT COUNT(*) FROM cache WHERE timestamp < ?", (expiry_timestamp,))
count = cursor.fetchone()[0]
# Delete old entries
cursor.execute("DELETE FROM cache WHERE timestamp < ?", (expiry_timestamp,))
conn.commit()
conn.close()
print(f"Cache cleanup completed: {count} entries removed")
except Exception as e:
print(f"Error during cache cleanup: {e}")
# Initialize scheduler for periodic cache cleanup
scheduler = BackgroundScheduler()
scheduler.add_job(
cleanup_old_cache_entries,
CronTrigger.from_crontab(CLEANUP_CRON),
id='cache_cleanup_job',
replace_existing=True
)
# Initialize database on startup
init_db()
# Start the scheduler when the application starts
@app.on_event("startup")
def start_scheduler():
scheduler.start()
print(f"Cache cleanup scheduler started with cron: {CLEANUP_CRON}")
print(f"Cache entries will expire after {CACHE_EXPIRY_HOURS} hours")
# Shutdown the scheduler when the application stops
@app.on_event("shutdown")
def shutdown_scheduler():
scheduler.shutdown(wait=False)
print("Cache cleanup scheduler stopped")
async def wait_for_network_idle(page):
"""Wait until no network requests are in flight"""
await page.waitForNetworkIdle(idleTime=500, timeout=30000)
@app.head("/")
async def health_check():
return {"status": "ok"}
async def safe_browser_operation(url, operation_func):
"""Safely perform browser operations with proper cleanup"""
browser = None
try:
browser = await launch(
headless=True,
executablePath='/usr/bin/google-chrome',
args=['--no-sandbox', '--disable-setuid-sandbox'],
handleSIGINT=False,
handleSIGTERM=False,
handleSIGHUP=False
)
# Create new page with timeout
page = await browser.newPage()
page.setDefaultNavigationTimeout(30000)
# Set custom user agent
await page.setUserAgent(CUSTOM_USER_AGENT)
# Call the operation function that uses the page
result = await operation_func(page)
# Explicitly close the page
await page.close()
return result
finally:
# Ensure browser is closed properly
if browser:
try:
await browser.close()
except Exception as e:
print(f"Error closing browser: {e}")
# We don't re-raise here to avoid masking the original error
@app.get("/")
async def visit_url(url: str, x_api_key: Optional[str] = Header(None)):
# Validate API key
if not x_api_key or x_api_key != API_KEY:
raise HTTPException(status_code=401, detail="Invalid API key")
# Decode URL if it's encoded
decoded_url = unquote(url)
# Check cache first
cached_result = get_cached_data(decoded_url, "visit")
if cached_result:
return cached_result
try:
print(f"Visiting URL: {decoded_url}")
# Define the operation to perform with the browser
async def visit_operation(page):
try:
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
if not response:
print(f"Warning: No response object returned for {decoded_url}")
# Get page content
content = await page.content()
return {"status": "success", "content": content}
except Exception as e:
print(f"Error during page navigation: {e}")
# Try to get content anyway
try:
content = await page.content()
return {"status": "partial", "content": content, "error": str(e)}
except:
raise HTTPException(status_code=500, detail=f"Failed to get page content: {str(e)}")
# Perform the operation
result = await safe_browser_operation(decoded_url, visit_operation)
# Save to cache
save_to_cache(decoded_url, "visit", result)
return result
except Exception as e:
print(f"Error visiting URL {decoded_url}: {e}")
raise HTTPException(status_code=500, detail=str(e))
@app.get("/seo")
async def extract_seo(url: str, x_api_key: Optional[str] = Header(None)):
"""Extract SEO information from a website"""
# Validate API key
if not x_api_key or x_api_key != API_KEY:
raise HTTPException(status_code=401, detail="Invalid API key")
# Decode URL if it's encoded
decoded_url = unquote(url)
# Check cache first
cached_result = get_cached_data(decoded_url, "seo")
if cached_result:
return cached_result
try:
print(f"Extracting SEO from: {decoded_url}")
# Define the operation to perform with the browser
async def seo_operation(page):
try:
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
# Extract SEO information
seo_data = await page.evaluate('''() => {
const data = {
title: document.title || '',
description: '',
canonical: '',
h1: [],
h2: [],
images: 0,
links: 0
};
// Get meta description
const metaDescription = document.querySelector('meta[name="description"]');
if (metaDescription) {
data.description = metaDescription.getAttribute('content') || '';
}
// Get canonical link
const canonicalLink = document.querySelector('link[rel="canonical"]');
if (canonicalLink) {
data.canonical = canonicalLink.getAttribute('href') || '';
}
// Get h1 tags
document.querySelectorAll('h1').forEach(h1 => {
const text = h1.innerText.trim();
if (text) data.h1.push(text);
});
// Get h2 tags
document.querySelectorAll('h2').forEach(h2 => {
const text = h2.innerText.trim();
if (text) data.h2.push(text);
});
// Count images
data.images = document.querySelectorAll('img').length;
// Count links
data.links = document.querySelectorAll('a').length;
return data;
}''')
result = {
"status": "success",
"url": decoded_url,
"seo": seo_data
}
return result
except Exception as e:
print(f"Error during SEO extraction: {e}")
return {"status": "error", "url": decoded_url, "error": str(e)}
# Perform the operation
result = await safe_browser_operation(decoded_url, seo_operation)
# Save to cache
save_to_cache(decoded_url, "seo", result)
return result
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@app.get("/meta")
async def extract_meta_tags(url: str, x_api_key: Optional[str] = Header(None)):
"""Extract meta tags from a website"""
# Validate API key
if not x_api_key or x_api_key != API_KEY:
raise HTTPException(status_code=401, detail="Invalid API key")
# Decode URL if it's encoded
decoded_url = unquote(url)
# Check cache first
cached_result = get_cached_data(decoded_url, "meta")
if cached_result:
return cached_result
try:
print(f"Extracting meta tags from: {decoded_url}")
# Define the operation to perform with the browser
async def meta_operation(page):
try:
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
# Extract all meta tags
meta_tags = await page.evaluate('''() => {
const metas = Array.from(document.querySelectorAll('meta'));
return metas.map(meta => {
const attributes = {};
Array.from(meta.attributes).forEach(attr => {
attributes[attr.name] = attr.value;
});
return attributes;
});
}''')
# Extract Open Graph tags
og_tags = await page.evaluate('''() => {
const ogTags = {};
document.querySelectorAll('meta[property^="og:"]').forEach(tag => {
const property = tag.getAttribute('property');
ogTags[property] = tag.getAttribute('content');
});
return ogTags;
}''')
# Extract Twitter card tags
twitter_tags = await page.evaluate('''() => {
const twitterTags = {};
document.querySelectorAll('meta[name^="twitter:"]').forEach(tag => {
const name = tag.getAttribute('name');
twitterTags[name] = tag.getAttribute('content');
});
return twitterTags;
}''')
result = {
"status": "success",
"url": decoded_url,
"meta_tags": meta_tags,
"open_graph": og_tags,
"twitter_card": twitter_tags,
"title": await page.title()
}
return result
except Exception as e:
print(f"Error during meta tag extraction: {e}")
return {"status": "error", "url": decoded_url, "error": str(e)}
# Perform the operation
result = await safe_browser_operation(decoded_url, meta_operation)
# Save to cache
save_to_cache(decoded_url, "meta", result)
return result
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@app.get("/cache/clear")
async def clear_cache(x_api_key: Optional[str] = Header(None)):
"""Clear the entire cache database"""
# Validate API key
if not x_api_key or x_api_key != API_KEY:
raise HTTPException(status_code=401, detail="Invalid API key")
conn = sqlite3.connect(DB_PATH)
cursor = conn.cursor()
cursor.execute("DELETE FROM cache")
conn.commit()
conn.close()
return {"status": "success", "message": "Cache cleared successfully"}
@app.get("/cache/stats")
async def cache_stats(x_api_key: Optional[str] = Header(None)):
"""Get cache statistics"""
# Validate API key
if not x_api_key or x_api_key != API_KEY:
raise HTTPException(status_code=401, detail="Invalid API key")
conn = sqlite3.connect(DB_PATH)
cursor = conn.cursor()
# Get total entries
cursor.execute("SELECT COUNT(*) FROM cache")
total_entries = cursor.fetchone()[0]
# Get entries by route
cursor.execute("SELECT route, COUNT(*) FROM cache GROUP BY route")
routes = {route: count for route, count in cursor.fetchall()}
# Get recent entries (last 24 hours)
recent_timestamp = int(time.time()) - (24 * 60 * 60)
cursor.execute("SELECT COUNT(*) FROM cache WHERE timestamp > ?", (recent_timestamp,))
recent_entries = cursor.fetchone()[0]
# Get oldest entry timestamp
cursor.execute("SELECT MIN(timestamp) FROM cache")
oldest_timestamp = cursor.fetchone()[0]
oldest_date = datetime.fromtimestamp(oldest_timestamp).isoformat() if oldest_timestamp else None
# Get newest entry timestamp
cursor.execute("SELECT MAX(timestamp) FROM cache")
newest_timestamp = cursor.fetchone()[0]
newest_date = datetime.fromtimestamp(newest_timestamp).isoformat() if newest_timestamp else None
conn.close()
return {
"status": "success",
"stats": {
"total_entries": total_entries,
"entries_by_route": routes,
"recent_entries": recent_entries,
"oldest_entry": oldest_date,
"newest_entry": newest_date,
"cache_expiry_hours": CACHE_EXPIRY_HOURS,
"cleanup_schedule": CLEANUP_CRON
}
}
@app.get("/pagination")
async def detect_pagination(url: str, x_api_key: Optional[str] = Header(None)):
"""Detect pagination on a website and determine the pagination pattern"""
# Validate API key
if not x_api_key or x_api_key != API_KEY:
raise HTTPException(status_code=401, detail="Invalid API key")
# Decode URL if it's encoded
decoded_url = unquote(url)
# Check cache first
cached_result = get_cached_data(decoded_url, "pagination")
if cached_result:
return cached_result
try:
print(f"Detecting pagination on: {decoded_url}")
# Define the operation to perform with the browser
async def pagination_operation(page):
try:
# Navigate to the URL
await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
original_url = page.url
print(f"Successfully loaded page: {original_url}")
# Before clicking any pagination links, try to determine the last page
# by analyzing all pagination-related links on the current page
last_page = await page.evaluate('''() => {
console.log("Analyzing page for last page number before navigation");
// Get all links on the page
const links = Array.from(document.querySelectorAll('a'));
console.log(`Found ${links.length} links to analyze`);
let lastPage = null;
let lastPageSource = '';
// Strategy 1: Find numeric links (e.g., page numbers)
const numericLinks = links.filter(link => {
const text = link.innerText.trim();
return /^[0-9]+$/.test(text) && link.href && link.href !== '#';
});
if (numericLinks.length > 0) {
const numericValues = numericLinks.map(link => parseInt(link.innerText.trim()));
const maxNumeric = Math.max(...numericValues);
console.log(`Found highest numeric link: ${maxNumeric}`);
if (maxNumeric > 1) {
lastPage = maxNumeric;
lastPageSource = 'numeric-links';
}
}
// Strategy 2: Look for "last page" link
const lastLinks = links.filter(link => {
const text = link.innerText.trim().toLowerCase();
const classes = (link.className || '').toLowerCase();
const ariaLabel = (link.getAttribute('aria-label') || '').toLowerCase();
return (text === 'last' ||
classes.includes('last') ||
ariaLabel.includes('last') ||
link.getAttribute('rel') === 'last');
});
if (lastLinks.length > 0) {
console.log(`Found ${lastLinks.length} "last" links`);
// Try to extract page number from the URL
const lastLink = lastLinks[0];
const href = lastLink.href;
console.log(`Last link href: ${href}`);
// Common patterns: page=X, /page/X, etc.
const pagePatterns = [
/[?&]page=(\d+)/,
/[?&]p=(\d+)/,
/[?&]pg=(\d+)/,
/\/page\/(\d+)/,
/\/p\/(\d+)/,
/\/paged\/(\d+)/
];
for (const pattern of pagePatterns) {
const match = href.match(pattern);
if (match && match[1]) {
const pageNum = parseInt(match[1]);
console.log(`Found page number ${pageNum} in last link URL`);
if (pageNum > 1 && (lastPage === null || pageNum > lastPage)) {
lastPage = pageNum;
lastPageSource = 'last-link-url';
break;
}
}
}
}
// Strategy 3: Look for pagination text like "Page 1 of 42"
const paginationTexts = [];
document.querySelectorAll('.pagination, .pager, .paginator, .paging, .page-numbers, .pages, .page-navigation')
.forEach(el => paginationTexts.push(el.innerText));
// Also check for any element that might contain pagination info
document.querySelectorAll('[class*="pag"], [id*="pag"]')
.forEach(el => paginationTexts.push(el.innerText));
const pageOfPatterns = [
/page\s+\d+\s+of\s+(\d+)/i,
/page\s+\d+\s*\/\s*(\d+)/i,
/\d+\s*\/\s*(\d+)\s+pages/i,
/\d+\s*-\s*\d+\s+of\s+\d+\s+\(\s*(\d+)\s+pages\s*\)/i,
/showing\s+\d+\s*-\s*\d+\s+of\s+\d+\s+\(\s*(\d+)\s+pages\s*\)/i,
/\d+\s*-\s*\d+\s+of\s+\d+\s+items\s+\(\s*(\d+)\s+pages\s*\)/i
];
for (const text of paginationTexts) {
for (const pattern of pageOfPatterns) {
const match = text.match(pattern);
if (match && match[1]) {
const pageNum = parseInt(match[1]);
console.log(`Found page count ${pageNum} in text`);
if (pageNum > 1 && (lastPage === null || pageNum > lastPage)) {
lastPage = pageNum;
lastPageSource = 'pagination-text';
break;
}
}
}
}
// NEW STRATEGY: Analyze all link URLs for page numbers
console.log("Analyzing all link URLs for page numbers");
// Common URL patterns that indicate pagination
const urlPagePatterns = [
/[?&]page=(\d+)/,
/[?&]p=(\d+)/,
/[?&]pg=(\d+)/,
/\/page\/(\d+)/,
/\/p\/(\d+)/,
/\/paged\/(\d+)/,
/\/pages\/(\d+)/,
/[?&]offset=(\d+)/,
/[?&]start=(\d+)/,
/[?&]from=(\d+)/,
/[?&]paged=(\d+)/,
/[?&]pagenum=(\d+)/,
/[?&]pageNumber=(\d+)/,
/[?&]currentpage=(\d+)/
];
// Extract page numbers from all link URLs
const pageNumbersFromUrls = [];
links.forEach(link => {
if (!link.href || link.href === '#') return;
const href = link.href;
for (const pattern of urlPagePatterns) {
const match = href.match(pattern);
if (match && match[1]) {
const pageNum = parseInt(match[1]);
if (pageNum > 1) {
pageNumbersFromUrls.push(pageNum);
break;
}
}
}
});
if (pageNumbersFromUrls.length > 0) {
const maxPageFromUrls = Math.max(...pageNumbersFromUrls);
console.log(`Found highest page number in URLs: ${maxPageFromUrls} (from ${pageNumbersFromUrls.length} links)`);
if (maxPageFromUrls > 1 && (lastPage === null || maxPageFromUrls > lastPage)) {
lastPage = maxPageFromUrls;
lastPageSource = 'url-analysis';
}
}
console.log(`Pre-navigation last page detection: ${lastPage} (source: ${lastPageSource})`);
return { lastPage, lastPageSource };
}''');
print(f"Pre-navigation last page detection: {last_page.get('lastPage')} (source: {last_page.get('lastPageSource')})")
# Find and click on pagination elements
pagination_clicked = await page.evaluate('''async () => {
// Common pagination link selectors to try
const selectors = [
'a.next', 'a.page-next', 'a[rel="next"]',
'a[aria-label="Next page"]', 'a[aria-label="next"]',
'.pagination a:nth-child(2)', // Often the "2" link
'.pagination li:nth-child(2) a',
'a:contains("2")', 'a:contains("Next")', 'a:contains("»")'
];
// Try to find a page "2" link first
const links = Array.from(document.querySelectorAll('a'));
const page2Link = links.find(link => {
const text = link.innerText.trim();
return text === '2' && link.href && link.href !== '#';
});
if (page2Link) {
// Store the href before clicking
const href = page2Link.href;
// Click the link
page2Link.click();
return { clicked: true, href: href, type: 'numeric' };
}
// Try common selectors
for (const selector of selectors) {
try {
if (selector.includes(':contains')) {
// Handle jQuery-style :contains selector
const text = selector.match(/:contains\\("(.+)"\\)/)[1];
const link = links.find(l => l.innerText.includes(text) && l.href && l.href !== '#');
if (link) {
const href = link.href;
link.click();
return { clicked: true, href: href, type: 'selector' };
}
} else {
const element = document.querySelector(selector);
if (element && element.href && element.href !== '#') {
const href = element.href;
element.click();
return { clicked: true, href: href, type: 'selector' };
}
}
} catch (e) {
// Ignore errors for invalid selectors
}
}
// Look for any link that might be pagination
const paginationLinks = links.filter(link => {
if (!link.href || link.href === '#') return false;
const text = link.innerText.trim();
const href = link.href;
// Check for numeric text or next/prev indicators
const isNumeric = /^[0-9]+$/.test(text) && text !== '1';
const isNextPrev = /next|prev|previous|older|newer/i.test(text) ||
/[»«‹›<>]/.test(text);
// Check for page parameter in URL
const hasPageParam = /[?&]page=|[?&]p=|[?&]pg=|\/page\/|\/p\//.test(href);
return (isNumeric || isNextPrev || hasPageParam);
});
if (paginationLinks.length > 0) {
const link = paginationLinks[0];
const href = link.href;
link.click();
return { clicked: true, href: href, type: 'other' };
}
return { clicked: false };
}''')
# If no pagination was found or clicked
if not pagination_clicked.get('clicked', False):
return {
"status": "success",
"url": decoded_url,
"hasPagination": False,
"urlTemplate": None,
"lastPage": last_page.get('lastPage')
}
# Wait for navigation to complete after the click
try:
await page.waitForNavigation({'timeout': 10000, 'waitUntil': 'networkidle2'})
except Exception as e:
print(f"Navigation timeout: {e}")
# Get the new URL after clicking
next_page_url = page.url
# If URL didn't change, pagination might be handled by AJAX
if next_page_url == original_url:
return {
"status": "success",
"url": decoded_url,
"hasPagination": True,
"urlTemplate": "AJAX pagination (URL doesn't change)",
"lastPage": last_page.get('lastPage'),
"nextPageUrl": next_page_url,
"originalUrl": original_url,
"expectedHref": pagination_clicked.get('href')
}
print(f"Navigation successful: {original_url} -> {next_page_url}")
# Analyze the URL structure to determine pagination pattern
url_template = None
# Compare the original and new URLs to find the pagination pattern
if '?' in next_page_url:
# Extract query parameters from both URLs
original_params = {}
if '?' in original_url:
original_query = original_url.split('?')[1].split('#')[0]
for param in original_query.split('&'):
if '=' in param:
key, value = param.split('=', 1)
original_params[key] = value
next_query = next_page_url.split('?')[1].split('#')[0]
next_params = {}
for param in next_query.split('&'):
if '=' in param:
key, value = param.split('=', 1)
next_params[key] = value
# Find parameters that changed or were added
pagination_param = None
for key, value in next_params.items():
# Check if parameter is new or changed
if key not in original_params or original_params[key] != value:
# Check if the value is numeric and could be a page number
if value.isdigit() and int(value) > 1:
pagination_param = key
break
# If we found a pagination parameter
if pagination_param:
base_url = next_page_url.split('?')[0]
# Reconstruct the URL template with all parameters
query_parts = []
for key, value in next_params.items():
if key == pagination_param:
query_parts.append(f"{key}={{PAGE_NUMBER}}")
else:
query_parts.append(f"{key}={value}")
url_template = f"{base_url}?{'&'.join(query_parts)}"
# If we couldn't determine from query parameters, check for path-based pagination
if not url_template:
# Path-based pagination patterns
path_patterns = ['/page/', '/p/', '/paged/', '/pages/']
for pattern in path_patterns:
if pattern in next_page_url:
parts = next_page_url.split(pattern)
url_template = f"{parts[0]}{pattern}{{PAGE_NUMBER}}"
if len(parts) > 1 and '/' in parts[1]:
suffix = parts[1].split('/', 1)[1]
if suffix:
url_template += f"/{suffix}"
break
# If we still couldn't determine the pattern, use the original and next URLs as examples
if not url_template:
url_template = f"Pattern unclear. Example: {original_url}{next_page_url}"
result = {
"status": "success",
"hasPagination": True,
"urlTemplate": url_template,
"lastPage": last_page.get('lastPage')
}
return result
except Exception as e:
print(f"Error during pagination detection: {e}")
return {
"status": "error",
"url": decoded_url,
"error": str(e),
"hasPagination": False,
"urlTemplate": None,
"lastPage": None
}
# Perform the operation
result = await safe_browser_operation(decoded_url, pagination_operation)
# Save to cache
save_to_cache(decoded_url, "pagination", result)
return result
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
if __name__ == "__main__":
import uvicorn
uvicorn.run(app, host="0.0.0.0", port=8000)