920 lines
36 KiB
Python
920 lines
36 KiB
Python
from fastapi import FastAPI, HTTPException, Header
|
|
from pyppeteer import launch
|
|
import os
|
|
import asyncio
|
|
import json
|
|
import re
|
|
import sqlite3
|
|
import time
|
|
from datetime import datetime, timedelta
|
|
from typing import Optional, Dict, List, Any
|
|
from urllib.parse import unquote
|
|
from apscheduler.schedulers.background import BackgroundScheduler
|
|
from apscheduler.triggers.cron import CronTrigger
|
|
|
|
app = FastAPI()
|
|
|
|
# Get API key from environment variable
|
|
API_KEY = os.getenv('API_KEY')
|
|
if not API_KEY:
|
|
raise ValueError("API_KEY environment variable must be set")
|
|
|
|
# Get cache expiry time from environment variable (default: 36 hours)
|
|
CACHE_EXPIRY_HOURS = int(os.getenv('CACHE_EXPIRY_HOURS', '36'))
|
|
|
|
# Get cleanup cron schedule from environment variable (default: every day at 3 AM)
|
|
CLEANUP_CRON = os.getenv('CLEANUP_CRON', '0 3 * * *')
|
|
|
|
# Define custom user agent
|
|
CUSTOM_USER_AGENT = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/69.0.3497.100 Safari/537.36'
|
|
|
|
# Initialize SQLite database
|
|
def init_db():
|
|
global DB_PATH
|
|
|
|
# Try to use the mounted volume first
|
|
db_path = '/db/cache.db'
|
|
db_dir = os.path.dirname(db_path)
|
|
|
|
# Check if directory exists and is writable
|
|
dir_writable = False
|
|
if os.path.exists(db_dir):
|
|
try:
|
|
test_file = os.path.join(db_dir, '.write_test')
|
|
with open(test_file, 'w') as f:
|
|
f.write('test')
|
|
os.remove(test_file)
|
|
dir_writable = True
|
|
except (IOError, PermissionError):
|
|
print(f"Directory {db_dir} exists but is not writable")
|
|
dir_writable = False
|
|
|
|
# If directory doesn't exist or isn't writable, try to create it
|
|
if not os.path.exists(db_dir) or not dir_writable:
|
|
try:
|
|
os.makedirs(db_dir, exist_ok=True)
|
|
# Test if we can write to the directory
|
|
test_file = os.path.join(db_dir, '.write_test')
|
|
with open(test_file, 'w') as f:
|
|
f.write('test')
|
|
os.remove(test_file)
|
|
print(f"Created directory: {db_dir}")
|
|
dir_writable = True
|
|
except Exception as e:
|
|
print(f"Warning: Could not create or write to directory {db_dir}: {e}")
|
|
# Fallback to using a local database file
|
|
db_path = 'cache.db'
|
|
print(f"Using local database file: {db_path}")
|
|
|
|
try:
|
|
conn = sqlite3.connect(db_path)
|
|
cursor = conn.cursor()
|
|
cursor.execute('''
|
|
CREATE TABLE IF NOT EXISTS cache (
|
|
url TEXT,
|
|
route TEXT,
|
|
data TEXT,
|
|
timestamp INTEGER,
|
|
PRIMARY KEY (url, route)
|
|
)
|
|
''')
|
|
conn.commit()
|
|
conn.close()
|
|
print(f"Database initialized at {db_path}")
|
|
# Update the global DB_PATH
|
|
DB_PATH = db_path
|
|
except sqlite3.OperationalError as e:
|
|
print(f"Error initializing database at {db_path}: {e}")
|
|
# Fallback to using a local database file if the mounted volume has permission issues
|
|
db_path = 'cache.db'
|
|
print(f"Falling back to local database file: {db_path}")
|
|
try:
|
|
conn = sqlite3.connect(db_path)
|
|
cursor = conn.cursor()
|
|
cursor.execute('''
|
|
CREATE TABLE IF NOT EXISTS cache (
|
|
url TEXT,
|
|
route TEXT,
|
|
data TEXT,
|
|
timestamp INTEGER,
|
|
PRIMARY KEY (url, route)
|
|
)
|
|
''')
|
|
conn.commit()
|
|
conn.close()
|
|
print(f"Local database initialized at {db_path}")
|
|
# Update the global DB_PATH
|
|
DB_PATH = db_path
|
|
except sqlite3.OperationalError as e2:
|
|
print(f"Error initializing local database: {e2}")
|
|
raise
|
|
|
|
# Define the database path
|
|
DB_PATH = '/db/cache.db'
|
|
|
|
# Get cached data if it exists and is not older than the expiry time
|
|
def get_cached_data(url, route):
|
|
conn = sqlite3.connect(DB_PATH)
|
|
cursor = conn.cursor()
|
|
cache_expiry = int(time.time()) - (CACHE_EXPIRY_HOURS * 60 * 60) # Convert hours to seconds
|
|
cursor.execute(
|
|
"SELECT data FROM cache WHERE url = ? AND route = ? AND timestamp > ?",
|
|
(url, route, cache_expiry)
|
|
)
|
|
result = cursor.fetchone()
|
|
conn.close()
|
|
|
|
if result:
|
|
print(f"Cache hit for {url} on route {route}")
|
|
return json.loads(result[0])
|
|
return None
|
|
|
|
# Save data to cache
|
|
def save_to_cache(url, route, data):
|
|
conn = sqlite3.connect(DB_PATH)
|
|
cursor = conn.cursor()
|
|
timestamp = int(time.time())
|
|
|
|
# Convert data to JSON string
|
|
data_json = json.dumps(data)
|
|
|
|
cursor.execute(
|
|
"INSERT OR REPLACE INTO cache (url, route, data, timestamp) VALUES (?, ?, ?, ?)",
|
|
(url, route, data_json, timestamp)
|
|
)
|
|
conn.commit()
|
|
conn.close()
|
|
print(f"Saved to cache: {url} on route {route}")
|
|
|
|
# Function to clean up old cache entries
|
|
def cleanup_old_cache_entries():
|
|
try:
|
|
print(f"Running scheduled cache cleanup (entries older than {CACHE_EXPIRY_HOURS} hours)")
|
|
conn = sqlite3.connect(DB_PATH)
|
|
cursor = conn.cursor()
|
|
|
|
# Calculate the timestamp for entries older than the expiry time
|
|
expiry_timestamp = int(time.time()) - (CACHE_EXPIRY_HOURS * 60 * 60)
|
|
|
|
# Get count of entries to be deleted
|
|
cursor.execute("SELECT COUNT(*) FROM cache WHERE timestamp < ?", (expiry_timestamp,))
|
|
count = cursor.fetchone()[0]
|
|
|
|
# Delete old entries
|
|
cursor.execute("DELETE FROM cache WHERE timestamp < ?", (expiry_timestamp,))
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
print(f"Cache cleanup completed: {count} entries removed")
|
|
except Exception as e:
|
|
print(f"Error during cache cleanup: {e}")
|
|
|
|
# Initialize scheduler for periodic cache cleanup
|
|
scheduler = BackgroundScheduler()
|
|
scheduler.add_job(
|
|
cleanup_old_cache_entries,
|
|
CronTrigger.from_crontab(CLEANUP_CRON),
|
|
id='cache_cleanup_job',
|
|
replace_existing=True
|
|
)
|
|
|
|
# Initialize database on startup
|
|
init_db()
|
|
|
|
# Start the scheduler when the application starts
|
|
@app.on_event("startup")
|
|
def start_scheduler():
|
|
scheduler.start()
|
|
print(f"Cache cleanup scheduler started with cron: {CLEANUP_CRON}")
|
|
print(f"Cache entries will expire after {CACHE_EXPIRY_HOURS} hours")
|
|
|
|
# Shutdown the scheduler when the application stops
|
|
@app.on_event("shutdown")
|
|
def shutdown_scheduler():
|
|
scheduler.shutdown(wait=False)
|
|
print("Cache cleanup scheduler stopped")
|
|
|
|
async def wait_for_network_idle(page):
|
|
"""Wait until no network requests are in flight"""
|
|
await page.waitForNetworkIdle(idleTime=500, timeout=30000)
|
|
|
|
@app.head("/")
|
|
async def health_check():
|
|
return {"status": "ok"}
|
|
|
|
async def safe_browser_operation(url, operation_func):
|
|
"""Safely perform browser operations with proper cleanup"""
|
|
browser = None
|
|
try:
|
|
browser = await launch(
|
|
headless=True,
|
|
executablePath='/usr/bin/google-chrome',
|
|
args=['--no-sandbox', '--disable-setuid-sandbox'],
|
|
handleSIGINT=False,
|
|
handleSIGTERM=False,
|
|
handleSIGHUP=False
|
|
)
|
|
|
|
# Create new page with timeout
|
|
page = await browser.newPage()
|
|
page.setDefaultNavigationTimeout(30000)
|
|
|
|
# Set custom user agent
|
|
await page.setUserAgent(CUSTOM_USER_AGENT)
|
|
|
|
# Call the operation function that uses the page
|
|
result = await operation_func(page)
|
|
|
|
# Explicitly close the page
|
|
await page.close()
|
|
|
|
return result
|
|
|
|
finally:
|
|
# Ensure browser is closed properly
|
|
if browser:
|
|
try:
|
|
await browser.close()
|
|
except Exception as e:
|
|
print(f"Error closing browser: {e}")
|
|
# We don't re-raise here to avoid masking the original error
|
|
|
|
@app.get("/")
|
|
async def visit_url(url: str, x_api_key: Optional[str] = Header(None)):
|
|
# Validate API key
|
|
if not x_api_key or x_api_key != API_KEY:
|
|
raise HTTPException(status_code=401, detail="Invalid API key")
|
|
|
|
# Decode URL if it's encoded
|
|
decoded_url = unquote(url)
|
|
|
|
# Check cache first
|
|
cached_result = get_cached_data(decoded_url, "visit")
|
|
if cached_result:
|
|
return cached_result
|
|
|
|
try:
|
|
print(f"Visiting URL: {decoded_url}")
|
|
|
|
# Define the operation to perform with the browser
|
|
async def visit_operation(page):
|
|
try:
|
|
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
|
if not response:
|
|
print(f"Warning: No response object returned for {decoded_url}")
|
|
|
|
# Get page content
|
|
content = await page.content()
|
|
return {"status": "success", "content": content}
|
|
except Exception as e:
|
|
print(f"Error during page navigation: {e}")
|
|
# Try to get content anyway
|
|
try:
|
|
content = await page.content()
|
|
return {"status": "partial", "content": content, "error": str(e)}
|
|
except:
|
|
raise HTTPException(status_code=500, detail=f"Failed to get page content: {str(e)}")
|
|
|
|
# Perform the operation
|
|
result = await safe_browser_operation(decoded_url, visit_operation)
|
|
|
|
# Save to cache
|
|
save_to_cache(decoded_url, "visit", result)
|
|
|
|
return result
|
|
|
|
except Exception as e:
|
|
print(f"Error visiting URL {decoded_url}: {e}")
|
|
raise HTTPException(status_code=500, detail=str(e))
|
|
|
|
@app.get("/seo")
|
|
async def extract_seo(url: str, x_api_key: Optional[str] = Header(None)):
|
|
"""Extract SEO information from a website"""
|
|
# Validate API key
|
|
if not x_api_key or x_api_key != API_KEY:
|
|
raise HTTPException(status_code=401, detail="Invalid API key")
|
|
|
|
# Decode URL if it's encoded
|
|
decoded_url = unquote(url)
|
|
|
|
# Check cache first
|
|
cached_result = get_cached_data(decoded_url, "seo")
|
|
if cached_result:
|
|
return cached_result
|
|
|
|
try:
|
|
print(f"Extracting SEO from: {decoded_url}")
|
|
|
|
# Define the operation to perform with the browser
|
|
async def seo_operation(page):
|
|
try:
|
|
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
|
|
|
# Extract SEO information
|
|
seo_data = await page.evaluate('''() => {
|
|
const data = {
|
|
title: document.title || '',
|
|
description: '',
|
|
canonical: '',
|
|
h1: [],
|
|
h2: [],
|
|
images: 0,
|
|
links: 0
|
|
};
|
|
|
|
// Get meta description
|
|
const metaDescription = document.querySelector('meta[name="description"]');
|
|
if (metaDescription) {
|
|
data.description = metaDescription.getAttribute('content') || '';
|
|
}
|
|
|
|
// Get canonical link
|
|
const canonicalLink = document.querySelector('link[rel="canonical"]');
|
|
if (canonicalLink) {
|
|
data.canonical = canonicalLink.getAttribute('href') || '';
|
|
}
|
|
|
|
// Get h1 tags
|
|
document.querySelectorAll('h1').forEach(h1 => {
|
|
const text = h1.innerText.trim();
|
|
if (text) data.h1.push(text);
|
|
});
|
|
|
|
// Get h2 tags
|
|
document.querySelectorAll('h2').forEach(h2 => {
|
|
const text = h2.innerText.trim();
|
|
if (text) data.h2.push(text);
|
|
});
|
|
|
|
// Count images
|
|
data.images = document.querySelectorAll('img').length;
|
|
|
|
// Count links
|
|
data.links = document.querySelectorAll('a').length;
|
|
|
|
return data;
|
|
}''')
|
|
|
|
result = {
|
|
"status": "success",
|
|
"url": decoded_url,
|
|
"seo": seo_data
|
|
}
|
|
|
|
return result
|
|
|
|
except Exception as e:
|
|
print(f"Error during SEO extraction: {e}")
|
|
return {"status": "error", "url": decoded_url, "error": str(e)}
|
|
|
|
# Perform the operation
|
|
result = await safe_browser_operation(decoded_url, seo_operation)
|
|
|
|
# Save to cache
|
|
save_to_cache(decoded_url, "seo", result)
|
|
|
|
return result
|
|
|
|
except Exception as e:
|
|
raise HTTPException(status_code=500, detail=str(e))
|
|
|
|
@app.get("/meta")
|
|
async def extract_meta_tags(url: str, x_api_key: Optional[str] = Header(None)):
|
|
"""Extract meta tags from a website"""
|
|
# Validate API key
|
|
if not x_api_key or x_api_key != API_KEY:
|
|
raise HTTPException(status_code=401, detail="Invalid API key")
|
|
|
|
# Decode URL if it's encoded
|
|
decoded_url = unquote(url)
|
|
|
|
# Check cache first
|
|
cached_result = get_cached_data(decoded_url, "meta")
|
|
if cached_result:
|
|
return cached_result
|
|
|
|
try:
|
|
print(f"Extracting meta tags from: {decoded_url}")
|
|
|
|
# Define the operation to perform with the browser
|
|
async def meta_operation(page):
|
|
try:
|
|
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
|
|
|
# Extract all meta tags
|
|
meta_tags = await page.evaluate('''() => {
|
|
const metas = Array.from(document.querySelectorAll('meta'));
|
|
return metas.map(meta => {
|
|
const attributes = {};
|
|
Array.from(meta.attributes).forEach(attr => {
|
|
attributes[attr.name] = attr.value;
|
|
});
|
|
return attributes;
|
|
});
|
|
}''')
|
|
|
|
# Extract Open Graph tags
|
|
og_tags = await page.evaluate('''() => {
|
|
const ogTags = {};
|
|
document.querySelectorAll('meta[property^="og:"]').forEach(tag => {
|
|
const property = tag.getAttribute('property');
|
|
ogTags[property] = tag.getAttribute('content');
|
|
});
|
|
return ogTags;
|
|
}''')
|
|
|
|
# Extract Twitter card tags
|
|
twitter_tags = await page.evaluate('''() => {
|
|
const twitterTags = {};
|
|
document.querySelectorAll('meta[name^="twitter:"]').forEach(tag => {
|
|
const name = tag.getAttribute('name');
|
|
twitterTags[name] = tag.getAttribute('content');
|
|
});
|
|
return twitterTags;
|
|
}''')
|
|
|
|
result = {
|
|
"status": "success",
|
|
"url": decoded_url,
|
|
"meta_tags": meta_tags,
|
|
"open_graph": og_tags,
|
|
"twitter_card": twitter_tags,
|
|
"title": await page.title()
|
|
}
|
|
|
|
return result
|
|
|
|
except Exception as e:
|
|
print(f"Error during meta tag extraction: {e}")
|
|
return {"status": "error", "url": decoded_url, "error": str(e)}
|
|
|
|
# Perform the operation
|
|
result = await safe_browser_operation(decoded_url, meta_operation)
|
|
|
|
# Save to cache
|
|
save_to_cache(decoded_url, "meta", result)
|
|
|
|
return result
|
|
|
|
except Exception as e:
|
|
raise HTTPException(status_code=500, detail=str(e))
|
|
|
|
@app.get("/cache/clear")
|
|
async def clear_cache(x_api_key: Optional[str] = Header(None)):
|
|
"""Clear the entire cache database"""
|
|
# Validate API key
|
|
if not x_api_key or x_api_key != API_KEY:
|
|
raise HTTPException(status_code=401, detail="Invalid API key")
|
|
|
|
conn = sqlite3.connect(DB_PATH)
|
|
cursor = conn.cursor()
|
|
cursor.execute("DELETE FROM cache")
|
|
conn.commit()
|
|
conn.close()
|
|
|
|
return {"status": "success", "message": "Cache cleared successfully"}
|
|
|
|
@app.get("/cache/stats")
|
|
async def cache_stats(x_api_key: Optional[str] = Header(None)):
|
|
"""Get cache statistics"""
|
|
# Validate API key
|
|
if not x_api_key or x_api_key != API_KEY:
|
|
raise HTTPException(status_code=401, detail="Invalid API key")
|
|
|
|
conn = sqlite3.connect(DB_PATH)
|
|
cursor = conn.cursor()
|
|
|
|
# Get total entries
|
|
cursor.execute("SELECT COUNT(*) FROM cache")
|
|
total_entries = cursor.fetchone()[0]
|
|
|
|
# Get entries by route
|
|
cursor.execute("SELECT route, COUNT(*) FROM cache GROUP BY route")
|
|
routes = {route: count for route, count in cursor.fetchall()}
|
|
|
|
# Get recent entries (last 24 hours)
|
|
recent_timestamp = int(time.time()) - (24 * 60 * 60)
|
|
cursor.execute("SELECT COUNT(*) FROM cache WHERE timestamp > ?", (recent_timestamp,))
|
|
recent_entries = cursor.fetchone()[0]
|
|
|
|
# Get oldest entry timestamp
|
|
cursor.execute("SELECT MIN(timestamp) FROM cache")
|
|
oldest_timestamp = cursor.fetchone()[0]
|
|
oldest_date = datetime.fromtimestamp(oldest_timestamp).isoformat() if oldest_timestamp else None
|
|
|
|
# Get newest entry timestamp
|
|
cursor.execute("SELECT MAX(timestamp) FROM cache")
|
|
newest_timestamp = cursor.fetchone()[0]
|
|
newest_date = datetime.fromtimestamp(newest_timestamp).isoformat() if newest_timestamp else None
|
|
|
|
conn.close()
|
|
|
|
return {
|
|
"status": "success",
|
|
"stats": {
|
|
"total_entries": total_entries,
|
|
"entries_by_route": routes,
|
|
"recent_entries": recent_entries,
|
|
"oldest_entry": oldest_date,
|
|
"newest_entry": newest_date,
|
|
"cache_expiry_hours": CACHE_EXPIRY_HOURS,
|
|
"cleanup_schedule": CLEANUP_CRON
|
|
}
|
|
}
|
|
|
|
@app.get("/pagination")
|
|
async def detect_pagination(url: str, x_api_key: Optional[str] = Header(None)):
|
|
"""Detect pagination on a website and determine the pagination pattern"""
|
|
# Validate API key
|
|
if not x_api_key or x_api_key != API_KEY:
|
|
raise HTTPException(status_code=401, detail="Invalid API key")
|
|
|
|
# Decode URL if it's encoded
|
|
decoded_url = unquote(url)
|
|
|
|
# Check cache first
|
|
cached_result = get_cached_data(decoded_url, "pagination")
|
|
if cached_result:
|
|
return cached_result
|
|
|
|
try:
|
|
print(f"Detecting pagination on: {decoded_url}")
|
|
|
|
# Define the operation to perform with the browser
|
|
async def pagination_operation(page):
|
|
try:
|
|
# Navigate to the URL
|
|
await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
|
original_url = page.url
|
|
|
|
print(f"Successfully loaded page: {original_url}")
|
|
|
|
# Before clicking any pagination links, try to determine the last page
|
|
# by analyzing all pagination-related links on the current page
|
|
last_page = await page.evaluate('''() => {
|
|
console.log("Analyzing page for last page number before navigation");
|
|
|
|
// Get all links on the page
|
|
const links = Array.from(document.querySelectorAll('a'));
|
|
console.log(`Found ${links.length} links to analyze`);
|
|
|
|
let lastPage = null;
|
|
let lastPageSource = '';
|
|
|
|
// Strategy 1: Find numeric links (e.g., page numbers)
|
|
const numericLinks = links.filter(link => {
|
|
const text = link.innerText.trim();
|
|
return /^[0-9]+$/.test(text) && link.href && link.href !== '#';
|
|
});
|
|
|
|
if (numericLinks.length > 0) {
|
|
const numericValues = numericLinks.map(link => parseInt(link.innerText.trim()));
|
|
const maxNumeric = Math.max(...numericValues);
|
|
console.log(`Found highest numeric link: ${maxNumeric}`);
|
|
if (maxNumeric > 1) {
|
|
lastPage = maxNumeric;
|
|
lastPageSource = 'numeric-links';
|
|
}
|
|
}
|
|
|
|
// Strategy 2: Look for "last page" link
|
|
const lastLinks = links.filter(link => {
|
|
const text = link.innerText.trim().toLowerCase();
|
|
const classes = (link.className || '').toLowerCase();
|
|
const ariaLabel = (link.getAttribute('aria-label') || '').toLowerCase();
|
|
|
|
return (text === 'last' ||
|
|
classes.includes('last') ||
|
|
ariaLabel.includes('last') ||
|
|
link.getAttribute('rel') === 'last');
|
|
});
|
|
|
|
if (lastLinks.length > 0) {
|
|
console.log(`Found ${lastLinks.length} "last" links`);
|
|
// Try to extract page number from the URL
|
|
const lastLink = lastLinks[0];
|
|
const href = lastLink.href;
|
|
console.log(`Last link href: ${href}`);
|
|
|
|
// Common patterns: page=X, /page/X, etc.
|
|
const pagePatterns = [
|
|
/[?&]page=(\d+)/,
|
|
/[?&]p=(\d+)/,
|
|
/[?&]pg=(\d+)/,
|
|
/\/page\/(\d+)/,
|
|
/\/p\/(\d+)/,
|
|
/\/paged\/(\d+)/
|
|
];
|
|
|
|
for (const pattern of pagePatterns) {
|
|
const match = href.match(pattern);
|
|
if (match && match[1]) {
|
|
const pageNum = parseInt(match[1]);
|
|
console.log(`Found page number ${pageNum} in last link URL`);
|
|
if (pageNum > 1 && (lastPage === null || pageNum > lastPage)) {
|
|
lastPage = pageNum;
|
|
lastPageSource = 'last-link-url';
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// Strategy 3: Look for pagination text like "Page 1 of 42"
|
|
const paginationTexts = [];
|
|
document.querySelectorAll('.pagination, .pager, .paginator, .paging, .page-numbers, .pages, .page-navigation')
|
|
.forEach(el => paginationTexts.push(el.innerText));
|
|
|
|
// Also check for any element that might contain pagination info
|
|
document.querySelectorAll('[class*="pag"], [id*="pag"]')
|
|
.forEach(el => paginationTexts.push(el.innerText));
|
|
|
|
const pageOfPatterns = [
|
|
/page\s+\d+\s+of\s+(\d+)/i,
|
|
/page\s+\d+\s*\/\s*(\d+)/i,
|
|
/\d+\s*\/\s*(\d+)\s+pages/i,
|
|
/\d+\s*-\s*\d+\s+of\s+\d+\s+\(\s*(\d+)\s+pages\s*\)/i,
|
|
/showing\s+\d+\s*-\s*\d+\s+of\s+\d+\s+\(\s*(\d+)\s+pages\s*\)/i,
|
|
/\d+\s*-\s*\d+\s+of\s+\d+\s+items\s+\(\s*(\d+)\s+pages\s*\)/i
|
|
];
|
|
|
|
for (const text of paginationTexts) {
|
|
for (const pattern of pageOfPatterns) {
|
|
const match = text.match(pattern);
|
|
if (match && match[1]) {
|
|
const pageNum = parseInt(match[1]);
|
|
console.log(`Found page count ${pageNum} in text`);
|
|
if (pageNum > 1 && (lastPage === null || pageNum > lastPage)) {
|
|
lastPage = pageNum;
|
|
lastPageSource = 'pagination-text';
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// NEW STRATEGY: Analyze all link URLs for page numbers
|
|
console.log("Analyzing all link URLs for page numbers");
|
|
|
|
// Common URL patterns that indicate pagination
|
|
const urlPagePatterns = [
|
|
/[?&]page=(\d+)/,
|
|
/[?&]p=(\d+)/,
|
|
/[?&]pg=(\d+)/,
|
|
/\/page\/(\d+)/,
|
|
/\/p\/(\d+)/,
|
|
/\/paged\/(\d+)/,
|
|
/\/pages\/(\d+)/,
|
|
/[?&]offset=(\d+)/,
|
|
/[?&]start=(\d+)/,
|
|
/[?&]from=(\d+)/,
|
|
/[?&]paged=(\d+)/,
|
|
/[?&]pagenum=(\d+)/,
|
|
/[?&]pageNumber=(\d+)/,
|
|
/[?&]currentpage=(\d+)/
|
|
];
|
|
|
|
// Extract page numbers from all link URLs
|
|
const pageNumbersFromUrls = [];
|
|
links.forEach(link => {
|
|
if (!link.href || link.href === '#') return;
|
|
|
|
const href = link.href;
|
|
for (const pattern of urlPagePatterns) {
|
|
const match = href.match(pattern);
|
|
if (match && match[1]) {
|
|
const pageNum = parseInt(match[1]);
|
|
if (pageNum > 1) {
|
|
pageNumbersFromUrls.push(pageNum);
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
});
|
|
|
|
if (pageNumbersFromUrls.length > 0) {
|
|
const maxPageFromUrls = Math.max(...pageNumbersFromUrls);
|
|
console.log(`Found highest page number in URLs: ${maxPageFromUrls} (from ${pageNumbersFromUrls.length} links)`);
|
|
if (maxPageFromUrls > 1 && (lastPage === null || maxPageFromUrls > lastPage)) {
|
|
lastPage = maxPageFromUrls;
|
|
lastPageSource = 'url-analysis';
|
|
}
|
|
}
|
|
|
|
console.log(`Pre-navigation last page detection: ${lastPage} (source: ${lastPageSource})`);
|
|
return { lastPage, lastPageSource };
|
|
}''');
|
|
|
|
print(f"Pre-navigation last page detection: {last_page.get('lastPage')} (source: {last_page.get('lastPageSource')})")
|
|
|
|
# Find and click on pagination elements
|
|
pagination_clicked = await page.evaluate('''async () => {
|
|
// Common pagination link selectors to try
|
|
const selectors = [
|
|
'a.next', 'a.page-next', 'a[rel="next"]',
|
|
'a[aria-label="Next page"]', 'a[aria-label="next"]',
|
|
'.pagination a:nth-child(2)', // Often the "2" link
|
|
'.pagination li:nth-child(2) a',
|
|
'a:contains("2")', 'a:contains("Next")', 'a:contains("»")'
|
|
];
|
|
|
|
// Try to find a page "2" link first
|
|
const links = Array.from(document.querySelectorAll('a'));
|
|
const page2Link = links.find(link => {
|
|
const text = link.innerText.trim();
|
|
return text === '2' && link.href && link.href !== '#';
|
|
});
|
|
|
|
if (page2Link) {
|
|
// Store the href before clicking
|
|
const href = page2Link.href;
|
|
// Click the link
|
|
page2Link.click();
|
|
return { clicked: true, href: href, type: 'numeric' };
|
|
}
|
|
|
|
// Try common selectors
|
|
for (const selector of selectors) {
|
|
try {
|
|
if (selector.includes(':contains')) {
|
|
// Handle jQuery-style :contains selector
|
|
const text = selector.match(/:contains\\("(.+)"\\)/)[1];
|
|
const link = links.find(l => l.innerText.includes(text) && l.href && l.href !== '#');
|
|
if (link) {
|
|
const href = link.href;
|
|
link.click();
|
|
return { clicked: true, href: href, type: 'selector' };
|
|
}
|
|
} else {
|
|
const element = document.querySelector(selector);
|
|
if (element && element.href && element.href !== '#') {
|
|
const href = element.href;
|
|
element.click();
|
|
return { clicked: true, href: href, type: 'selector' };
|
|
}
|
|
}
|
|
} catch (e) {
|
|
// Ignore errors for invalid selectors
|
|
}
|
|
}
|
|
|
|
// Look for any link that might be pagination
|
|
const paginationLinks = links.filter(link => {
|
|
if (!link.href || link.href === '#') return false;
|
|
|
|
const text = link.innerText.trim();
|
|
const href = link.href;
|
|
|
|
// Check for numeric text or next/prev indicators
|
|
const isNumeric = /^[0-9]+$/.test(text) && text !== '1';
|
|
const isNextPrev = /next|prev|previous|older|newer/i.test(text) ||
|
|
/[»«‹›<>]/.test(text);
|
|
|
|
// Check for page parameter in URL
|
|
const hasPageParam = /[?&]page=|[?&]p=|[?&]pg=|\/page\/|\/p\//.test(href);
|
|
|
|
return (isNumeric || isNextPrev || hasPageParam);
|
|
});
|
|
|
|
if (paginationLinks.length > 0) {
|
|
const link = paginationLinks[0];
|
|
const href = link.href;
|
|
link.click();
|
|
return { clicked: true, href: href, type: 'other' };
|
|
}
|
|
|
|
return { clicked: false };
|
|
}''')
|
|
|
|
# If no pagination was found or clicked
|
|
if not pagination_clicked.get('clicked', False):
|
|
return {
|
|
"status": "success",
|
|
"url": decoded_url,
|
|
"hasPagination": False,
|
|
"urlTemplate": None,
|
|
"lastPage": last_page.get('lastPage')
|
|
}
|
|
|
|
# Wait for navigation to complete after the click
|
|
try:
|
|
await page.waitForNavigation({'timeout': 10000, 'waitUntil': 'networkidle2'})
|
|
except Exception as e:
|
|
print(f"Navigation timeout: {e}")
|
|
|
|
# Get the new URL after clicking
|
|
next_page_url = page.url
|
|
|
|
# If URL didn't change, pagination might be handled by AJAX
|
|
if next_page_url == original_url:
|
|
return {
|
|
"status": "success",
|
|
"url": decoded_url,
|
|
"hasPagination": True,
|
|
"urlTemplate": "AJAX pagination (URL doesn't change)",
|
|
"lastPage": last_page.get('lastPage'),
|
|
"nextPageUrl": next_page_url,
|
|
"originalUrl": original_url,
|
|
"expectedHref": pagination_clicked.get('href')
|
|
}
|
|
|
|
print(f"Navigation successful: {original_url} -> {next_page_url}")
|
|
|
|
# Analyze the URL structure to determine pagination pattern
|
|
url_template = None
|
|
|
|
# Compare the original and new URLs to find the pagination pattern
|
|
if '?' in next_page_url:
|
|
# Extract query parameters from both URLs
|
|
original_params = {}
|
|
if '?' in original_url:
|
|
original_query = original_url.split('?')[1].split('#')[0]
|
|
for param in original_query.split('&'):
|
|
if '=' in param:
|
|
key, value = param.split('=', 1)
|
|
original_params[key] = value
|
|
|
|
next_query = next_page_url.split('?')[1].split('#')[0]
|
|
next_params = {}
|
|
for param in next_query.split('&'):
|
|
if '=' in param:
|
|
key, value = param.split('=', 1)
|
|
next_params[key] = value
|
|
|
|
# Find parameters that changed or were added
|
|
pagination_param = None
|
|
for key, value in next_params.items():
|
|
# Check if parameter is new or changed
|
|
if key not in original_params or original_params[key] != value:
|
|
# Check if the value is numeric and could be a page number
|
|
if value.isdigit() and int(value) > 1:
|
|
pagination_param = key
|
|
break
|
|
|
|
# If we found a pagination parameter
|
|
if pagination_param:
|
|
base_url = next_page_url.split('?')[0]
|
|
|
|
# Reconstruct the URL template with all parameters
|
|
query_parts = []
|
|
for key, value in next_params.items():
|
|
if key == pagination_param:
|
|
query_parts.append(f"{key}={{PAGE_NUMBER}}")
|
|
else:
|
|
query_parts.append(f"{key}={value}")
|
|
|
|
url_template = f"{base_url}?{'&'.join(query_parts)}"
|
|
|
|
# If we couldn't determine from query parameters, check for path-based pagination
|
|
if not url_template:
|
|
# Path-based pagination patterns
|
|
path_patterns = ['/page/', '/p/', '/paged/', '/pages/']
|
|
for pattern in path_patterns:
|
|
if pattern in next_page_url:
|
|
parts = next_page_url.split(pattern)
|
|
url_template = f"{parts[0]}{pattern}{{PAGE_NUMBER}}"
|
|
if len(parts) > 1 and '/' in parts[1]:
|
|
suffix = parts[1].split('/', 1)[1]
|
|
if suffix:
|
|
url_template += f"/{suffix}"
|
|
break
|
|
|
|
# If we still couldn't determine the pattern, use the original and next URLs as examples
|
|
if not url_template:
|
|
url_template = f"Pattern unclear. Example: {original_url} → {next_page_url}"
|
|
|
|
result = {
|
|
"status": "success",
|
|
"hasPagination": True,
|
|
"urlTemplate": url_template,
|
|
"lastPage": last_page.get('lastPage')
|
|
}
|
|
|
|
return result
|
|
|
|
except Exception as e:
|
|
print(f"Error during pagination detection: {e}")
|
|
return {
|
|
"status": "error",
|
|
"url": decoded_url,
|
|
"error": str(e),
|
|
"hasPagination": False,
|
|
"urlTemplate": None,
|
|
"lastPage": None
|
|
}
|
|
|
|
# Perform the operation
|
|
result = await safe_browser_operation(decoded_url, pagination_operation)
|
|
|
|
# Save to cache
|
|
save_to_cache(decoded_url, "pagination", result)
|
|
|
|
return result
|
|
|
|
except Exception as e:
|
|
raise HTTPException(status_code=500, detail=str(e))
|
|
|
|
if __name__ == "__main__":
|
|
import uvicorn
|
|
uvicorn.run(app, host="0.0.0.0", port=8000)
|