why not
Build and Push Docker Images / build-and-push (push) Successful in 2m26s

This commit is contained in:
2025-06-21 20:31:38 +02:00
parent 6a4deb883f
commit 6051e41cfa
3 changed files with 449 additions and 761 deletions
@@ -153,10 +153,20 @@ async def detect_pagination_service(decoded_url):
async def pagination_operation(page):
try:
# Navigate to the URL
try:
await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
except Exception as e:
print(f"Error navigating to URL: {e}")
# Try to get the current URL even if navigation failed
try:
original_url = page.url
except:
original_url = decoded_url
else:
original_url = page.url
# Check for common pagination indicators
try:
pagination_data = await page.evaluate('''() => {
const data = {
hasPagination: false,
@@ -299,6 +309,16 @@ async def detect_pagination_service(decoded_url):
return data;
}''')
except Exception as e:
print(f"Error during JavaScript evaluation for pagination detection: {e}")
# Return a safe default if JavaScript evaluation fails
pagination_data = {
'hasPagination': False,
'paginationType': None,
'paginationElements': [],
'detectedParameter': None,
'lastPageNumber': None
}
# If pagination is detected, try to navigate to the next page by clicking
next_page_url = None
@@ -315,6 +335,7 @@ async def detect_pagination_service(decoded_url):
# First try to click on a numbered link (preferably "2" if we're on page 1)
try:
clicked = await page.evaluate('''() => {
try {
// First try to find and click on a "2" link or button
const page2Elements = Array.from(document.querySelectorAll('a[href], button, [role="button"]'))
.filter(el => {
@@ -395,6 +416,10 @@ async def detect_pagination_service(decoded_url):
}
return false;
} catch (error) {
console.error("Error during pagination click operation:", error);
return false;
}
}''')
if clicked:
@@ -407,12 +432,15 @@ async def detect_pagination_service(decoded_url):
except Exception as e:
print(f"Error clicking on pagination element: {e}")
# Continue with the process even if clicking fails
# If we successfully navigated to the next page, analyze the URL difference
if next_page_url:
print('Searching for pagination parameter')
# Parse both URLs
try:
original_parsed = await page.evaluate(f'''(originalUrl) => {{
try {{
const original = new URL(originalUrl);
const current = new URL(window.location.href);
@@ -468,7 +496,26 @@ async def detect_pagination_service(decoded_url):
originalUrl: originalUrl,
currentUrl: window.location.href
}};
}} catch (error) {{
console.error("Error during URL analysis:", error);
return {{
paramDiff: null,
pathDiff: null,
originalUrl: originalUrl,
currentUrl: window.location.href,
error: error.message
}};
}}
}}''', original_url)
except Exception as e:
print(f"Error during URL analysis: {e}")
original_parsed = {
'paramDiff': None,
'pathDiff': None,
'originalUrl': original_url,
'currentUrl': next_page_url,
'error': str(e)
}
# Determine the pagination parameter and create URL template
print(original_parsed)
@@ -492,13 +539,22 @@ async def detect_pagination_service(decoded_url):
step_size = None
# Create URL template for query parameter
try:
url_obj = await page.evaluate(f'''(url, paramName) => {{
try {{
const urlObj = new URL(url);
urlObj.searchParams.set(paramName, "{{PAGE_NUMBER}}");
return urlObj.toString();
}} catch (error) {{
console.error("Error creating URL template:", error);
return null;
}}
}}''', original_url, param_name)
url_template = url_obj
except Exception as e:
print(f"Error creating URL template for query parameter: {e}")
url_template = None
elif original_parsed['pathDiff']:
path_index = original_parsed['pathDiff']['index']
@@ -520,13 +576,22 @@ async def detect_pagination_service(decoded_url):
step_size = None
# Create URL template for path parameter
try:
url_template = await page.evaluate(f'''(url, pathIndex) => {{
try {{
const urlObj = new URL(url);
const pathSegments = urlObj.pathname.split('/').filter(s => s);
pathSegments[pathIndex] = "{{PAGE_NUMBER}}";
urlObj.pathname = '/' + pathSegments.join('/');
return urlObj.toString();
}} catch (error) {{
console.error("Error creating path URL template:", error);
return null;
}}
}}''', original_url, path_index)
except Exception as e:
print(f"Error creating URL template for path parameter: {e}")
url_template = None
if url_template:
# Decode URL-encoded characters in the template
@@ -536,11 +601,20 @@ async def detect_pagination_service(decoded_url):
# If we couldn't determine the URL template from navigation, try to infer it
if not url_template and pagination_data['detectedParameter']:
param_name = pagination_data['detectedParameter']['name']
try:
url_template = await page.evaluate(f'''(url, paramName) => {{
try {{
const urlObj = new URL(url);
urlObj.searchParams.set(paramName, "{{PAGE_NUMBER}}");
return urlObj.toString();
}} catch (error) {{
console.error("Error creating inferred URL template:", error);
return null;
}}
}}''', original_url, param_name)
except Exception as e:
print(f"Error creating inferred URL template: {e}")
url_template = None
# Return the pagination detection results with a simplified structure
result = {
+12 -457
View File
@@ -655,475 +655,30 @@ async def cache_stats(x_api_key: Optional[str] = Header(None)):
if not x_api_key or x_api_key != API_KEY:
raise HTTPException(status_code=401, detail="Invalid API key")
try:
conn = sqlite3.connect(DB_PATH)
cursor = conn.cursor()
# Get total entries
# Get total count
cursor.execute("SELECT COUNT(*) FROM cache")
total_entries = cursor.fetchone()[0]
total_count = cursor.fetchone()[0]
# Get entries by route
# Get count by route
cursor.execute("SELECT route, COUNT(*) FROM cache GROUP BY route")
routes = {route: count for route, count in cursor.fetchall()}
route_counts = dict(cursor.fetchall())
# Get recent entries (last 24 hours)
recent_timestamp = int(time.time()) - (24 * 60 * 60)
cursor.execute("SELECT COUNT(*) FROM cache WHERE timestamp > ?", (recent_timestamp,))
recent_entries = cursor.fetchone()[0]
# Get oldest entry timestamp
cursor.execute("SELECT MIN(timestamp) FROM cache")
oldest_timestamp = cursor.fetchone()[0]
oldest_date = datetime.fromtimestamp(oldest_timestamp).isoformat() if oldest_timestamp else None
# Get newest entry timestamp
cursor.execute("SELECT MAX(timestamp) FROM cache")
newest_timestamp = cursor.fetchone()[0]
newest_date = datetime.fromtimestamp(newest_timestamp).isoformat() if newest_timestamp else None
# Get oldest and newest entries
cursor.execute("SELECT MIN(timestamp), MAX(timestamp) FROM cache")
min_time, max_time = cursor.fetchone()
conn.close()
return {
"status": "success",
"stats": {
"total_entries": total_entries,
"entries_by_route": routes,
"recent_entries": recent_entries,
"oldest_entry": oldest_date,
"newest_entry": newest_date,
"cache_expiry_hours": CACHE_EXPIRY_HOURS,
"cleanup_schedule": CLEANUP_CRON
"total_entries": total_count,
"route_counts": route_counts,
"oldest_entry": min_time,
"newest_entry": max_time
}
}
@app.get("/pagination")
async def detect_pagination(url: str, x_api_key: Optional[str] = Header(None)):
"""Detect pagination on a website and determine the pagination pattern"""
# Validate API key
if not x_api_key or x_api_key != API_KEY:
raise HTTPException(status_code=401, detail="Invalid API key")
# Decode URL if it's encoded
decoded_url = unquote(url)
# Check cache first
cached_result = get_cached_data(decoded_url, "pagination")
if cached_result:
return cached_result
try:
print(f"Detecting pagination on: {decoded_url}")
# Define the operation to perform with the browser
async def pagination_operation(page):
try:
# Navigate to the URL
await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
original_url = page.url
print(f"Successfully loaded page: {original_url}")
# Analyze the page for pagination information
pagination_info = await page.evaluate('''() => {
// Find the last page number if available
const findLastPageNumber = () => {
// Get all links on the page
const links = Array.from(document.querySelectorAll('a'));
// Strategy 1: Find numeric links (page numbers)
const numericLinks = links.filter(link => {
const text = link.innerText.trim();
return /^[0-9]+$/.test(text) && link.href && link.href !== '#';
});
if (numericLinks.length > 0) {
const numericValues = numericLinks.map(link => parseInt(link.innerText.trim()));
return Math.max(...numericValues);
}
// Strategy 2: Look for "last page" link
const lastLinks = links.filter(link => {
const text = link.innerText.trim().toLowerCase();
const classes = (link.className || '').toLowerCase();
const ariaLabel = (link.getAttribute('aria-label') || '').toLowerCase();
return (text === 'last' ||
classes.includes('last') ||
ariaLabel.includes('last') ||
link.getAttribute('rel') === 'last');
});
if (lastLinks.length > 0) {
const lastLink = lastLinks[0];
const href = lastLink.href;
// Common patterns: page=X, /page/X, etc.
const pagePatterns = [
/[?&]page=(\d+)/,
/[?&]p=(\d+)/,
/[?&]pg=(\d+)/,
/\/page\/(\d+)/,
/\/p\/(\d+)/,
/\/paged\/(\d+)/
];
for (const pattern of pagePatterns) {
const match = href.match(pattern);
if (match && match[1]) {
return parseInt(match[1]);
}
}
}
// Strategy 3: Analyze all URLs for page numbers
const pageNumbersFromUrls = [];
links.forEach(link => {
if (!link.href || link.href === '#') return;
// Check for common pagination URL patterns
const patterns = [
/[?&]page=(\d+)/,
/[?&]p=(\d+)/,
/[?&]pg=(\d+)/,
/\/page\/(\d+)/,
/\/p\/(\d+)/,
/\/paged\/(\d+)/,
/\/pages\/(\d+)/
];
for (const pattern of patterns) {
const match = link.href.match(pattern);
if (match && match[1]) {
pageNumbersFromUrls.push(parseInt(match[1]));
}
}
});
if (pageNumbersFromUrls.length > 0) {
return Math.max(...pageNumbersFromUrls);
}
return null;
};
// Find a pagination link to click
const findPaginationLink = () => {
const links = Array.from(document.querySelectorAll('a'));
// Try to find a page "2" link first (most reliable)
const page2Link = links.find(link => {
const text = link.innerText.trim();
return text === '2' && link.href && link.href !== '#';
});
if (page2Link) {
return { element: page2Link, href: page2Link.href, type: 'numeric' };
}
// Try common "next page" selectors
const nextSelectors = [
'a.next',
'a.page-next',
'a[rel="next"]',
'a[aria-label="Next page"]',
'a[aria-label="next"]'
];
for (const selector of nextSelectors) {
const element = document.querySelector(selector);
if (element && element.href && element.href !== '#') {
return { element, href: element.href, type: 'next' };
}
}
// Look for any link that might be pagination
const paginationLinks = links.filter(link => {
if (!link.href || link.href === '#') return false;
const text = link.innerText.trim();
const href = link.href;
// Check for numeric text or next/prev indicators
const isNumeric = /^[0-9]+$/.test(text) && text !== '1';
const isNextPrev = /next|prev|previous|older|newer/i.test(text) ||
/[»«‹›<>]/.test(text);
// Check for page parameter in URL
const hasPageParam = /[?&]page=|[?&]p=|[?&]pg=|\/page\/|\/p\//.test(href);
return (isNumeric || isNextPrev || hasPageParam);
});
if (paginationLinks.length > 0) {
const link = paginationLinks[0];
return { element: link, href: link.href, type: 'other' };
}
return null;
};
const lastPage = findLastPageNumber();
const paginationLink = findPaginationLink();
if (paginationLink) {
// Click the link
paginationLink.element.click();
return {
clicked: true,
href: paginationLink.href,
type: paginationLink.type,
lastPage
};
}
return { clicked: false, lastPage };
}''')
# If no pagination was found or clicked
if not pagination_info.get('clicked', False):
return {
"status": "success",
"url": decoded_url,
"hasPagination": False,
"urlTemplate": None,
"lastPage": pagination_info.get('lastPage')
}
# Wait for navigation to complete after the click
try:
await page.waitForNavigation({'timeout': 10000, 'waitUntil': 'networkidle2'})
except Exception as e:
print(f"Navigation timeout: {e}")
# Get the new URL after clicking
next_page_url = page.url
# If URL didn't change, pagination might be handled by AJAX
if next_page_url == original_url:
return {
"status": "success",
"url": decoded_url,
"hasPagination": True,
"urlTemplate": "AJAX pagination (URL doesn't change)",
"lastPage": pagination_info.get('lastPage')
}
print(f"Navigation successful: {original_url} -> {next_page_url}")
# Analyze the URL structure to determine pagination pattern
url_template = await page.evaluate('''(originalUrl, nextPageUrl) => {
// Helper function to parse URL query parameters
const parseQueryParams = (url) => {
const params = {};
if (url.includes('?')) {
const queryString = url.split('?')[1].split('#')[0];
queryString.split('&').forEach(param => {
if (param.includes('=')) {
const [key, value] = param.split('=', 2);
params[key] = value;
}
});
}
return params;
};
// Check for query parameter based pagination
if (nextPageUrl.includes('?')) {
const originalParams = parseQueryParams(originalUrl);
const nextParams = parseQueryParams(nextPageUrl);
// Find parameters that changed or were added
let paginationParam = null;
// First check for common pagination parameter names
const commonPaginationParams = ['page', 'p', 'pg', 'paged', 'current_page', 'pagenum', 'pageNumber'];
for (const key of commonPaginationParams) {
if (key in nextParams &&
(!(key in originalParams) || originalParams[key] !== nextParams[key])) {
if (/^\d+$/.test(nextParams[key]) && parseInt(nextParams[key]) > 1) {
paginationParam = key;
break;
}
}
}
// If no common parameter found, check all parameters
if (!paginationParam) {
for (const [key, value] of Object.entries(nextParams)) {
// Check if parameter is new or changed
if (!(key in originalParams) || originalParams[key] !== value) {
// Check if the value is numeric and could be a page number
if (/^\d+$/.test(value) && parseInt(value) > 1) {
paginationParam = key;
break;
}
}
}
}
// If we found a pagination parameter
if (paginationParam) {
const baseUrl = nextPageUrl.split('?')[0];
// Reconstruct the URL template with all parameters
const queryParts = [];
for (const [key, value] of Object.entries(nextParams)) {
if (key === paginationParam) {
queryParts.push(`${key}={PAGE_NUMBER}`);
} else {
queryParts.push(`${key}=${value}`);
}
}
return `${baseUrl}?${queryParts.join('&')}`;
}
}
// Check for path-based pagination
const pathPatterns = ['/page/', '/p/', '/paged/', '/pages/'];
for (const pattern of pathPatterns) {
if (nextPageUrl.includes(pattern)) {
const parts = nextPageUrl.split(pattern);
let template = `${parts[0]}${pattern}{PAGE_NUMBER}`;
// Add any suffix after the page number
if (parts.length > 1 && parts[1].includes('/')) {
const suffix = parts[1].split('/', 1)[1];
if (suffix) {
template += `/${suffix}`;
}
}
return template;
}
}
// If we couldn't determine the pattern, try to make an educated guess
// For the specific case where a parameter like current_page=2 is added
const originalUrlObj = new URL(originalUrl);
const nextUrlObj = new URL(nextPageUrl);
// Check if the paths are the same but query params differ
if (originalUrlObj.pathname === nextUrlObj.pathname) {
const originalParams = parseQueryParams(originalUrl);
const nextParams = parseQueryParams(nextPageUrl);
// Find parameters that exist in next but not in original
const newParams = Object.keys(nextParams).filter(key => !(key in originalParams));
// If there's exactly one new parameter and it has a numeric value
if (newParams.length === 1 && /^\d+$/.test(nextParams[newParams[0]])) {
const paginationParam = newParams[0];
const baseUrl = nextPageUrl.split('?')[0];
// Reconstruct the URL template
const queryParts = [];
for (const [key, value] of Object.entries(nextParams)) {
if (key === paginationParam) {
queryParts.push(`${key}={PAGE_NUMBER}`);
} else {
queryParts.push(`${key}=${value}`);
}
}
return `${baseUrl}?${queryParts.join('&')}`;
}
}
// If we still couldn't determine the pattern, return both URLs as examples
return `Pattern unclear. Example: ${originalUrl} → ${nextPageUrl}`;
}''', original_url, next_page_url)
# Try to extract last page number from the next page if we didn't find it on the first page
if not pagination_info.get('lastPage'):
last_page_from_next = await page.evaluate('''() => {
// Get all links on the page
const links = Array.from(document.querySelectorAll('a'));
// Strategy 1: Find numeric links (page numbers)
const numericLinks = links.filter(link => {
const text = link.innerText.trim();
return /^[0-9]+$/.test(text) && link.href && link.href !== '#';
});
if (numericLinks.length > 0) {
const numericValues = numericLinks.map(link => parseInt(link.innerText.trim()));
return Math.max(...numericValues);
}
// Strategy 2: Analyze all URLs for page numbers
const pageNumbersFromUrls = [];
links.forEach(link => {
if (!link.href || link.href === '#') return;
// Check for common pagination URL patterns
const patterns = [
/[?&]page=(\d+)/,
/[?&]p=(\d+)/,
/[?&]pg=(\d+)/,
/\/page\/(\d+)/,
/\/p\/(\d+)/,
/\/paged\/(\d+)/,
/\/pages\/(\d+)/
];
for (const pattern of patterns) {
const match = link.href.match(pattern);
if (match && match[1]) {
pageNumbersFromUrls.push(parseInt(match[1]));
}
}
});
if (pageNumbersFromUrls.length > 0) {
return Math.max(...pageNumbersFromUrls);
}
return null;
}''')
if last_page_from_next:
pagination_info['lastPage'] = last_page_from_next
# Check if the pattern is unclear
has_pagination = True
url_string_template = str(url_template)
if url_string_template and url_string_template.startswith("Pattern unclear"):
has_pagination = False
url_template = None
result = {
"status": "success",
"url": decoded_url,
"hasPagination": has_pagination,
"urlTemplate": url_template,
"lastPage": pagination_info.get('lastPage'),
"originalUrl": original_url,
"nextPageUrl": next_page_url
}
return result
except Exception as e:
print(f"Error during pagination detection: {e}")
return {
"status": "error",
"url": decoded_url,
"error": str(e),
"hasPagination": False,
"urlTemplate": None,
"lastPage": None
}
# Perform the operation
result = await safe_browser_operation(decoded_url, pagination_operation)
# Save to cache
save_to_cache(decoded_url, "pagination", result)
return result
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
+59
View File
@@ -0,0 +1,59 @@
#!/usr/bin/env python3
"""
Test script for the pagination route to verify error handling and browser cleanup.
"""
import asyncio
import aiohttp
import json
import sys
async def test_pagination(url, api_key):
"""Test the pagination endpoint with a given URL"""
async with aiohttp.ClientSession() as session:
headers = {"X-API-Key": api_key}
# Test URL with pagination
test_url = f"http://localhost:8000/pagination?url={url}"
print(f"Testing pagination for: {url}")
try:
async with session.get(test_url, headers=headers) as response:
if response.status == 200:
result = await response.json()
print(f"✅ Success: {json.dumps(result, indent=2)}")
return True
else:
error_text = await response.text()
print(f"❌ Error {response.status}: {error_text}")
return False
except Exception as e:
print(f"❌ Exception: {e}")
return False
async def main():
"""Main test function"""
# Test URLs - some with pagination, some without
test_urls = [
"https://example.com", # No pagination
"https://httpbin.org/get", # No pagination
"https://news.ycombinator.com", # Has pagination
]
api_key = "test-key" # Replace with your actual API key
print("Testing pagination route...")
print("=" * 50)
for url in test_urls:
success = await test_pagination(url, api_key)
print("-" * 30)
# Small delay between tests
await asyncio.sleep(1)
print("Test completed!")
if __name__ == "__main__":
asyncio.run(main())