This commit is contained in:
@@ -52,9 +52,17 @@ This project has been successfully migrated from Puppeteer (pyppeteer) to Playwr
|
||||
- **System status endpoint**: `/status` - Monitor CPU, memory, file descriptors
|
||||
- **Browser pool monitoring**: Track active browsers and their ages
|
||||
- **Emergency cleanup endpoint**: `/emergency-cleanup` - Force cleanup when needed
|
||||
- **Timeout-based cleanup**: `/force-cleanup-old` - Force close old browser/page instances
|
||||
- **Monitoring script**: `monitor.py` - Real-time system monitoring
|
||||
|
||||
#### 4. Error Handling
|
||||
#### 4. Timeout Management
|
||||
|
||||
- **Configurable timeout**: `BROWSER_INSTANCE_TIMEOUT_MINUTES` (default: 10 minutes)
|
||||
- **Automatic cleanup**: Old browser and page instances are automatically force closed
|
||||
- **Page tracking**: All page instances are tracked with creation timestamps
|
||||
- **Health check integration**: Timeout cleanup runs every 2 minutes as part of health checks
|
||||
|
||||
#### 5. Error Handling
|
||||
|
||||
- **Timeout handling**: 30s timeout for getting browsers from pool
|
||||
- **Emergency recovery**: Auto-cleanup when timeouts occur
|
||||
@@ -70,6 +78,7 @@ API_KEY=your-api-key-here
|
||||
MAX_BROWSERS=3 # Maximum browser instances (default: 3)
|
||||
BROWSER_TTL=1800 # Browser time-to-live in seconds (default: 1800 = 30min)
|
||||
MAX_CONCURRENT_OPERATIONS=5 # Max concurrent operations (default: 5)
|
||||
BROWSER_INSTANCE_TIMEOUT_MINUTES=10 # Force close old browser/page instances after N minutes (default: 10)
|
||||
CACHE_EXPIRY_HOURS=36 # Cache expiry in hours (default: 36)
|
||||
CLEANUP_CRON=0 3 * * * # Cache cleanup schedule (default: daily at 3 AM)
|
||||
RATE_LIMIT_MINUTE=60 # Requests per minute (default: 60)
|
||||
@@ -82,12 +91,22 @@ RATE_LIMIT_MINUTE=60 # Requests per minute (default: 60)
|
||||
- `GET /` - Visit URL and get HTML content
|
||||
- `GET /seo` - Extract SEO information
|
||||
- `GET /meta` - Extract meta tags and Open Graph data
|
||||
- `GET /json` - Fetch JSON content from a website
|
||||
|
||||
The `/json` endpoint intelligently extracts JSON data from websites by:
|
||||
|
||||
- Directly parsing JSON responses (when Content-Type is application/json)
|
||||
- Extracting JSON from `<script type="application/json">` tags
|
||||
- Finding JSON in `data-json` attributes
|
||||
- Searching for JSON-like structures in page content
|
||||
- Parsing the entire page content as JSON if possible
|
||||
|
||||
### Management Endpoints
|
||||
|
||||
- `HEAD /` - Health check
|
||||
- `GET /status` - System status and browser pool information
|
||||
- `POST /emergency-cleanup` - Force cleanup all browsers
|
||||
- `POST /force-cleanup-old` - Force cleanup old browser/page instances (based on timeout)
|
||||
- `GET /cache/stats` - Cache statistics
|
||||
- `GET /cache/clear` - Clear all cache
|
||||
|
||||
@@ -102,8 +121,14 @@ curl -H "X-API-Key: your-api-key" "http://localhost:8000/?url=https://example.co
|
||||
# Extract SEO data
|
||||
curl -H "X-API-Key: your-api-key" "http://localhost:8000/seo?url=https://example.com"
|
||||
|
||||
# Fetch JSON content
|
||||
curl -H "X-API-Key: your-api-key" "http://localhost:8000/json?url=https://api.example.com/data"
|
||||
|
||||
# Get system status
|
||||
curl -H "X-API-Key: your-api-key" "http://localhost:8000/status"
|
||||
|
||||
# Force cleanup old browser/page instances
|
||||
curl -X POST -H "X-API-Key: your-api-key" "http://localhost:8000/force-cleanup-old"
|
||||
```
|
||||
|
||||
### Monitoring
|
||||
@@ -157,7 +182,9 @@ This will test:
|
||||
|
||||
1. Check system status: `GET /status`
|
||||
2. If browser pool is at capacity, trigger cleanup: `POST /emergency-cleanup`
|
||||
3. Consider reducing `MAX_BROWSERS` or `MAX_CONCURRENT_OPERATIONS`
|
||||
3. Force cleanup old instances: `POST /force-cleanup-old`
|
||||
4. Consider reducing `MAX_BROWSERS` or `MAX_CONCURRENT_OPERATIONS`
|
||||
5. Adjust `BROWSER_INSTANCE_TIMEOUT_MINUTES` to a lower value (e.g., 5 minutes)
|
||||
|
||||
### Resource Errors (Errno 11)
|
||||
|
||||
|
||||
@@ -11,6 +11,9 @@ CACHE_EXPIRY_HOURS = int(os.getenv('CACHE_EXPIRY_HOURS', '36'))
|
||||
# Get cleanup cron schedule from environment variable (default: every day at 3 AM)
|
||||
CLEANUP_CRON = os.getenv('CLEANUP_CRON', '0 3 * * *')
|
||||
|
||||
# Get browser instance timeout from environment variable (default: 10 minutes)
|
||||
BROWSER_INSTANCE_TIMEOUT_MINUTES = int(os.getenv('BROWSER_INSTANCE_TIMEOUT_MINUTES', '10'))
|
||||
|
||||
# Define custom user agent
|
||||
CUSTOM_USER_AGENT = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/69.0.3497.100 Safari/537.36'
|
||||
|
||||
|
||||
@@ -8,7 +8,8 @@ from app.services.browser import (
|
||||
extract_seo_service,
|
||||
extract_meta_tags_service,
|
||||
detect_pagination_service,
|
||||
capture_outgoing_calls_service
|
||||
capture_outgoing_calls_service,
|
||||
fetch_json_service
|
||||
)
|
||||
|
||||
router = APIRouter()
|
||||
@@ -153,3 +154,32 @@ async def capture_outgoing_calls(url: str, skipCache: bool = False, x_api_key: O
|
||||
except Exception as e:
|
||||
print(f"Error capturing outgoing calls for URL {decoded_url}: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@router.get("/json")
|
||||
async def fetch_json(url: str, skipCache: bool = False, x_api_key: Optional[str] = Header(None)):
|
||||
"""Fetch JSON content from a website"""
|
||||
# Validate API key
|
||||
if not x_api_key or x_api_key != API_KEY:
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
|
||||
# Decode URL if it's encoded
|
||||
decoded_url = unquote(url)
|
||||
|
||||
# Check cache first (unless skipCache is True)
|
||||
if not skipCache:
|
||||
cached_result = get_cached_data(decoded_url, "json")
|
||||
if cached_result:
|
||||
return cached_result
|
||||
|
||||
try:
|
||||
# Call the service function
|
||||
result = await fetch_json_service(decoded_url)
|
||||
|
||||
# Save to cache
|
||||
if(result["status"] == "success"):
|
||||
save_to_cache(decoded_url, "json", result)
|
||||
|
||||
return result
|
||||
except Exception as e:
|
||||
print(f"Error fetching JSON for URL {decoded_url}: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
@@ -894,3 +894,107 @@ async def capture_outgoing_calls_service(decoded_url):
|
||||
|
||||
# Perform the operation
|
||||
return await safe_browser_operation(decoded_url, outgoing_calls_operation)
|
||||
|
||||
async def fetch_json_service(decoded_url):
|
||||
"""Service function to fetch JSON content from a website"""
|
||||
print(f"Fetching JSON from: {decoded_url}")
|
||||
|
||||
# Define the operation to perform with the browser
|
||||
async def json_operation(page):
|
||||
try:
|
||||
# Navigate to the URL
|
||||
response = await page.goto(decoded_url, wait_until='networkidle', timeout=30000)
|
||||
|
||||
if not response:
|
||||
raise Exception("No response received from the URL")
|
||||
|
||||
# Check if the response is JSON
|
||||
content_type = response.headers.get('content-type', '').lower()
|
||||
if 'application/json' not in content_type and 'text/json' not in content_type:
|
||||
# If not JSON, try to find JSON content in the page
|
||||
json_content = await page.evaluate('''() => {
|
||||
// Look for JSON in script tags
|
||||
const scripts = document.querySelectorAll('script[type="application/json"], script[type="application/ld+json"]');
|
||||
if (scripts.length > 0) {
|
||||
return Array.from(scripts).map(script => {
|
||||
try {
|
||||
return JSON.parse(script.textContent);
|
||||
} catch (e) {
|
||||
return null;
|
||||
}
|
||||
}).filter(json => json !== null);
|
||||
}
|
||||
|
||||
// Look for JSON in data attributes
|
||||
const elementsWithData = document.querySelectorAll('[data-json]');
|
||||
if (elementsWithData.length > 0) {
|
||||
return Array.from(elementsWithData).map(el => {
|
||||
try {
|
||||
return JSON.parse(el.getAttribute('data-json'));
|
||||
} catch (e) {
|
||||
return null;
|
||||
}
|
||||
}).filter(json => json !== null);
|
||||
}
|
||||
|
||||
// Look for JSON in the page content (try to find JSON-like structures)
|
||||
const bodyText = document.body.innerText;
|
||||
const jsonMatches = bodyText.match(/\\{[^{}]*\\}/g);
|
||||
if (jsonMatches) {
|
||||
const validJsons = [];
|
||||
for (const match of jsonMatches) {
|
||||
try {
|
||||
const parsed = JSON.parse(match);
|
||||
validJsons.push(parsed);
|
||||
} catch (e) {
|
||||
// Skip invalid JSON
|
||||
}
|
||||
}
|
||||
if (validJsons.length > 0) {
|
||||
return validJsons;
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}''')
|
||||
|
||||
if json_content:
|
||||
return {
|
||||
"status": "success",
|
||||
"url": decoded_url,
|
||||
"content_type": "json_extracted",
|
||||
"json_data": json_content
|
||||
}
|
||||
else:
|
||||
# Try to get the page content and check if it's JSON
|
||||
content = await page.content()
|
||||
try:
|
||||
import json
|
||||
json_data = json.loads(content)
|
||||
return {
|
||||
"status": "success",
|
||||
"url": decoded_url,
|
||||
"content_type": "json_direct",
|
||||
"json_data": json_data
|
||||
}
|
||||
except json.JSONDecodeError:
|
||||
raise Exception("No JSON content found on the page")
|
||||
else:
|
||||
# Response is already JSON
|
||||
try:
|
||||
json_data = await response.json()
|
||||
return {
|
||||
"status": "success",
|
||||
"url": decoded_url,
|
||||
"content_type": "json_response",
|
||||
"json_data": json_data
|
||||
}
|
||||
except Exception as e:
|
||||
raise Exception(f"Failed to parse JSON response: {str(e)}")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error during JSON fetching: {e}")
|
||||
return {"status": "error", "url": decoded_url, "error": str(e)}
|
||||
|
||||
# Perform the operation
|
||||
return await safe_browser_operation(decoded_url, json_operation)
|
||||
|
||||
@@ -1,6 +1,11 @@
|
||||
from playwright.async_api import async_playwright
|
||||
from app.config import CUSTOM_USER_AGENT
|
||||
from app.config import CUSTOM_USER_AGENT, BROWSER_INSTANCE_TIMEOUT_MINUTES
|
||||
import asyncio
|
||||
import time
|
||||
|
||||
# Global tracking for browser instances created by this module
|
||||
page_creation_times = {}
|
||||
active_pages = set()
|
||||
|
||||
async def wait_for_network_idle(page):
|
||||
"""Wait until no network requests are in flight"""
|
||||
@@ -32,6 +37,10 @@ async def safe_browser_operation(url, operation_func):
|
||||
page = await context.new_page()
|
||||
page.set_default_timeout(30000)
|
||||
|
||||
# Track page creation time for force cleanup
|
||||
page_creation_times[page] = time.time()
|
||||
active_pages.add(page)
|
||||
|
||||
# Call the operation function that uses the page
|
||||
result = await operation_func(page)
|
||||
|
||||
@@ -46,9 +55,14 @@ async def safe_browser_operation(url, operation_func):
|
||||
"url": url
|
||||
}
|
||||
finally:
|
||||
# Ensure page is closed properly
|
||||
# Cleanup page tracking
|
||||
if page:
|
||||
try:
|
||||
# Remove from tracking
|
||||
if page in active_pages:
|
||||
active_pages.remove(page)
|
||||
if page in page_creation_times:
|
||||
del page_creation_times[page]
|
||||
await page.close()
|
||||
except Exception as e:
|
||||
print(f"Error closing page: {e}")
|
||||
@@ -73,3 +87,32 @@ async def safe_browser_operation(url, operation_func):
|
||||
await playwright.stop()
|
||||
except Exception as e:
|
||||
print(f"Error closing playwright: {e}")
|
||||
|
||||
async def force_cleanup_old_pages():
|
||||
"""Force cleanup old page instances created by this module"""
|
||||
print(f"Checking for old page instances in browser_utils (timeout: {BROWSER_INSTANCE_TIMEOUT_MINUTES} minutes)...")
|
||||
|
||||
current_time = time.time()
|
||||
timeout_seconds = BROWSER_INSTANCE_TIMEOUT_MINUTES * 60
|
||||
cleaned_pages = 0
|
||||
|
||||
# Clean up old pages
|
||||
pages_to_cleanup = []
|
||||
for page in list(active_pages):
|
||||
creation_time = page_creation_times.get(page, 0)
|
||||
if current_time - creation_time > timeout_seconds:
|
||||
pages_to_cleanup.append(page)
|
||||
print(f"Marking page for cleanup (age: {(current_time - creation_time)/60:.1f} minutes)")
|
||||
|
||||
for page in pages_to_cleanup:
|
||||
try:
|
||||
if page in active_pages:
|
||||
active_pages.remove(page)
|
||||
if page in page_creation_times:
|
||||
del page_creation_times[page]
|
||||
await page.close()
|
||||
cleaned_pages += 1
|
||||
except Exception as e:
|
||||
print(f"Error cleaning up old page: {e}")
|
||||
|
||||
print(f"Force cleanup completed: {cleaned_pages} pages cleaned")
|
||||
@@ -14,10 +14,12 @@ from urllib.parse import unquote
|
||||
from apscheduler.schedulers.background import BackgroundScheduler
|
||||
from apscheduler.triggers.cron import CronTrigger
|
||||
from fastapi.middleware.base import BaseHTTPMiddleware
|
||||
from app.config import BROWSER_INSTANCE_TIMEOUT_MINUTES
|
||||
|
||||
# Add imports for browser pool
|
||||
from asyncio import Queue, Lock, Semaphore
|
||||
from contextlib import asynccontextmanager
|
||||
from app.utils.browser_utils import force_cleanup_old_pages
|
||||
|
||||
app = FastAPI()
|
||||
|
||||
@@ -44,7 +46,9 @@ MAX_CONCURRENT_OPERATIONS = int(os.getenv('MAX_CONCURRENT_OPERATIONS', '5')) #
|
||||
browser_pool = Queue(maxsize=MAX_BROWSERS) # Add maxsize to prevent unbounded growth
|
||||
browser_lock = Lock()
|
||||
browser_creation_times = {}
|
||||
page_creation_times = {} # Track page creation times for force cleanup
|
||||
active_browsers = set() # Track active browsers
|
||||
active_pages = set() # Track active pages for force cleanup
|
||||
operation_semaphore = Semaphore(MAX_CONCURRENT_OPERATIONS) # Limit concurrent operations
|
||||
playwright_instance = None # Global playwright instance
|
||||
|
||||
@@ -152,10 +156,14 @@ async def cleanup_browser(browser):
|
||||
if browser in browser_creation_times:
|
||||
del browser_creation_times[browser]
|
||||
|
||||
# Close all pages first
|
||||
# Close all pages first and clean up page tracking
|
||||
pages = browser.contexts[0].pages if browser.contexts else []
|
||||
for page in pages:
|
||||
try:
|
||||
if page in active_pages:
|
||||
active_pages.remove(page)
|
||||
if page in page_creation_times:
|
||||
del page_creation_times[page]
|
||||
await page.close()
|
||||
except Exception as e:
|
||||
print(f"Error closing page: {e}")
|
||||
@@ -173,6 +181,9 @@ async def check_browser_health():
|
||||
# Sleep for 2 minutes between checks (reduced from 5 minutes)
|
||||
await asyncio.sleep(120)
|
||||
|
||||
# First, force cleanup old instances
|
||||
await force_cleanup_old_instances()
|
||||
|
||||
async with browser_lock:
|
||||
# Get all browsers from the pool
|
||||
browsers = []
|
||||
@@ -223,6 +234,58 @@ async def check_browser_health():
|
||||
print(f"Error in browser health check: {e}")
|
||||
await asyncio.sleep(60) # Wait before retrying
|
||||
|
||||
async def force_cleanup_old_instances():
|
||||
"""Force cleanup old browser and page instances based on timeout"""
|
||||
print(f"Checking for old browser/page instances (timeout: {BROWSER_INSTANCE_TIMEOUT_MINUTES} minutes)...")
|
||||
|
||||
current_time = time.time()
|
||||
timeout_seconds = BROWSER_INSTANCE_TIMEOUT_MINUTES * 60
|
||||
cleaned_browsers = 0
|
||||
cleaned_pages = 0
|
||||
|
||||
async with browser_lock:
|
||||
# Clean up old browsers
|
||||
browsers_to_cleanup = []
|
||||
for browser in list(active_browsers):
|
||||
creation_time = browser_creation_times.get(browser, 0)
|
||||
if current_time - creation_time > timeout_seconds:
|
||||
browsers_to_cleanup.append(browser)
|
||||
print(f"Marking browser for cleanup (age: {(current_time - creation_time)/60:.1f} minutes)")
|
||||
|
||||
for browser in browsers_to_cleanup:
|
||||
try:
|
||||
await cleanup_browser(browser)
|
||||
cleaned_browsers += 1
|
||||
except Exception as e:
|
||||
print(f"Error cleaning up old browser: {e}")
|
||||
|
||||
# Clean up old pages (this is a fallback for pages that might not be properly tracked)
|
||||
for browser in list(active_browsers):
|
||||
try:
|
||||
if browser.contexts:
|
||||
for context in browser.contexts:
|
||||
for page in context.pages:
|
||||
if page in page_creation_times:
|
||||
creation_time = page_creation_times[page]
|
||||
if current_time - creation_time > timeout_seconds:
|
||||
try:
|
||||
if page in active_pages:
|
||||
active_pages.remove(page)
|
||||
if page in page_creation_times:
|
||||
del page_creation_times[page]
|
||||
await page.close()
|
||||
cleaned_pages += 1
|
||||
print(f"Force closed old page (age: {(current_time - creation_time)/60:.1f} minutes)")
|
||||
except Exception as e:
|
||||
print(f"Error closing old page: {e}")
|
||||
except Exception as e:
|
||||
print(f"Error checking pages in browser: {e}")
|
||||
|
||||
# Also cleanup pages from browser_utils module
|
||||
await force_cleanup_old_pages()
|
||||
|
||||
print(f"Force cleanup completed: {cleaned_browsers} browsers, {cleaned_pages} pages cleaned")
|
||||
|
||||
async def force_cleanup_all_browsers():
|
||||
"""Force cleanup all browsers - useful for emergency situations"""
|
||||
print("Force cleaning up all browsers...")
|
||||
@@ -439,6 +502,10 @@ async def safe_browser_operation(url, operation_func):
|
||||
|
||||
page = await context.new_page()
|
||||
|
||||
# Track page creation time for force cleanup
|
||||
page_creation_times[page] = time.time()
|
||||
active_pages.add(page)
|
||||
|
||||
# Set up request interception for better performance
|
||||
await page.route("**/*", lambda route: route.abort()
|
||||
if route.request.resource_type in ['image', 'stylesheet', 'font', 'media']
|
||||
@@ -455,6 +522,11 @@ async def safe_browser_operation(url, operation_func):
|
||||
# Cleanup
|
||||
if page:
|
||||
try:
|
||||
# Remove from tracking
|
||||
if page in active_pages:
|
||||
active_pages.remove(page)
|
||||
if page in page_creation_times:
|
||||
del page_creation_times[page]
|
||||
await page.close()
|
||||
except:
|
||||
pass
|
||||
@@ -757,6 +829,7 @@ async def system_status(x_api_key: Optional[str] = Header(None)):
|
||||
# Get browser pool information
|
||||
pool_size = browser_pool.qsize()
|
||||
active_browser_count = len(active_browsers)
|
||||
active_page_count = len(active_pages)
|
||||
|
||||
# Get cache statistics
|
||||
conn = sqlite3.connect('/db/cache.db')
|
||||
@@ -778,7 +851,9 @@ async def system_status(x_api_key: Optional[str] = Header(None)):
|
||||
"pool_size": pool_size,
|
||||
"max_browsers": MAX_BROWSERS,
|
||||
"active_browsers": active_browser_count,
|
||||
"browser_ttl_seconds": BROWSER_TTL
|
||||
"active_pages": active_page_count,
|
||||
"browser_ttl_seconds": BROWSER_TTL,
|
||||
"instance_timeout_minutes": BROWSER_INSTANCE_TIMEOUT_MINUTES
|
||||
},
|
||||
"cache": {
|
||||
"total_entries": cache_count,
|
||||
@@ -805,6 +880,22 @@ async def emergency_cleanup(x_api_key: Optional[str] = Header(None)):
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@app.post("/force-cleanup-old")
|
||||
async def force_cleanup_old(x_api_key: Optional[str] = Header(None)):
|
||||
"""Force cleanup old browser and page instances based on timeout"""
|
||||
# Validate API key
|
||||
if not x_api_key or x_api_key != API_KEY:
|
||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
||||
|
||||
try:
|
||||
await force_cleanup_old_instances()
|
||||
return {
|
||||
"status": "success",
|
||||
"message": f"Force cleanup of old instances completed (timeout: {BROWSER_INSTANCE_TIMEOUT_MINUTES} minutes)"
|
||||
}
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@asynccontextmanager
|
||||
async def get_browser():
|
||||
"""Context manager for getting a browser from the pool"""
|
||||
|
||||
@@ -1,59 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Test script for the pagination route to verify error handling and browser cleanup.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import aiohttp
|
||||
import json
|
||||
import sys
|
||||
|
||||
async def test_pagination(url, api_key):
|
||||
"""Test the pagination endpoint with a given URL"""
|
||||
async with aiohttp.ClientSession() as session:
|
||||
headers = {"X-API-Key": api_key}
|
||||
|
||||
# Test URL with pagination
|
||||
test_url = f"http://localhost:8000/pagination?url={url}"
|
||||
|
||||
print(f"Testing pagination for: {url}")
|
||||
|
||||
try:
|
||||
async with session.get(test_url, headers=headers) as response:
|
||||
if response.status == 200:
|
||||
result = await response.json()
|
||||
print(f"✅ Success: {json.dumps(result, indent=2)}")
|
||||
return True
|
||||
else:
|
||||
error_text = await response.text()
|
||||
print(f"❌ Error {response.status}: {error_text}")
|
||||
return False
|
||||
except Exception as e:
|
||||
print(f"❌ Exception: {e}")
|
||||
return False
|
||||
|
||||
async def main():
|
||||
"""Main test function"""
|
||||
# Test URLs - some with pagination, some without
|
||||
test_urls = [
|
||||
"https://example.com", # No pagination
|
||||
"https://httpbin.org/get", # No pagination
|
||||
"https://news.ycombinator.com", # Has pagination
|
||||
]
|
||||
|
||||
api_key = "test-key" # Replace with your actual API key
|
||||
|
||||
print("Testing pagination route...")
|
||||
print("=" * 50)
|
||||
|
||||
for url in test_urls:
|
||||
success = await test_pagination(url, api_key)
|
||||
print("-" * 30)
|
||||
|
||||
# Small delay between tests
|
||||
await asyncio.sleep(1)
|
||||
|
||||
print("Test completed!")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
@@ -1,101 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
|
||||
# Mock the pagination detection logic to test the URL template generation
|
||||
async def test_url_template_generation():
|
||||
"""Test the URL template generation for the specific case"""
|
||||
|
||||
# Test case: /vacatures -> /vacatures/page/2
|
||||
original_url = "https://www.werkenbijabnamro.nl/vacatures"
|
||||
current_url = "https://www.werkenbijabnamro.nl/vacatures/page/2"
|
||||
|
||||
# Simulate the JavaScript logic for path difference detection
|
||||
def detect_path_diff(original_url, current_url):
|
||||
from urllib.parse import urlparse
|
||||
|
||||
original_parsed = urlparse(original_url)
|
||||
current_parsed = urlparse(current_url)
|
||||
|
||||
original_path = original_parsed.path
|
||||
current_path = current_parsed.path
|
||||
|
||||
if original_path != current_path:
|
||||
original_segments = [s for s in original_path.split('/') if s]
|
||||
current_segments = [s for s in current_path.split('/') if s]
|
||||
|
||||
# Case 2: Current path has more segments - check for added pagination segments
|
||||
if len(current_segments) > len(original_segments):
|
||||
# Look for patterns like /page/NUMBER or /p/NUMBER at the end
|
||||
import re
|
||||
page_pattern = re.compile(r'^(page|p)/(\d+)$', re.IGNORECASE)
|
||||
|
||||
# Check the last two segments of the current path
|
||||
if len(current_segments) >= 2:
|
||||
last_two_segments = '/'.join(current_segments[-2:])
|
||||
match = page_pattern.match(last_two_segments)
|
||||
|
||||
if match:
|
||||
return {
|
||||
'type': 'append',
|
||||
'pageSegment': match.group(1), # 'page' or 'p'
|
||||
'pageNumber': match.group(2), # the actual number
|
||||
'originalSegments': original_segments,
|
||||
'currentSegments': current_segments
|
||||
}
|
||||
|
||||
# If no pattern match, check if the last segment is numeric
|
||||
if len(current_segments) > 0:
|
||||
last_segment = current_segments[-1]
|
||||
if last_segment.isdigit():
|
||||
return {
|
||||
'type': 'append',
|
||||
'pageSegment': None,
|
||||
'pageNumber': last_segment,
|
||||
'originalSegments': original_segments,
|
||||
'currentSegments': current_segments
|
||||
}
|
||||
|
||||
return None
|
||||
|
||||
# Test the path difference detection
|
||||
path_diff = detect_path_diff(original_url, current_url)
|
||||
print("Path difference detection:")
|
||||
print(json.dumps(path_diff, indent=2))
|
||||
|
||||
# Test URL template generation
|
||||
def create_url_template(original_url, path_diff):
|
||||
from urllib.parse import urlparse, urlunparse
|
||||
|
||||
if path_diff and path_diff.get('type') == 'append':
|
||||
url_obj = urlparse(original_url)
|
||||
new_path = url_obj.path
|
||||
|
||||
# Remove trailing slash if present
|
||||
if new_path.endswith('/'):
|
||||
new_path = new_path[:-1]
|
||||
|
||||
# Append the pagination segment
|
||||
page_segment = path_diff.get('pageSegment')
|
||||
if page_segment:
|
||||
new_path += f'/{page_segment}/{{PAGE_NUMBER}}'
|
||||
else:
|
||||
new_path += '/{PAGE_NUMBER}'
|
||||
|
||||
url_obj = url_obj._replace(pathname=new_path)
|
||||
return urlunparse(url_obj)
|
||||
|
||||
return None
|
||||
|
||||
url_template = create_url_template(original_url, path_diff)
|
||||
print(f"\nGenerated URL template: {url_template}")
|
||||
|
||||
# Test with different page numbers
|
||||
if url_template:
|
||||
for page_num in [1, 2, 3, 10]:
|
||||
test_url = url_template.replace('{PAGE_NUMBER}', str(page_num))
|
||||
print(f"Page {page_num}: {test_url}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(test_url_template_generation())
|
||||
@@ -1,120 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Test script to verify Playwright migration works correctly
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import sys
|
||||
import os
|
||||
|
||||
# Add the current directory to Python path
|
||||
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
from playwright.async_api import async_playwright
|
||||
|
||||
async def test_playwright_installation():
|
||||
"""Test that Playwright is properly installed and can launch a browser"""
|
||||
print("Testing Playwright installation...")
|
||||
|
||||
try:
|
||||
async with async_playwright() as p:
|
||||
browser = await p.chromium.launch(headless=True)
|
||||
context = await browser.new_context()
|
||||
page = await context.new_page()
|
||||
|
||||
# Test basic navigation
|
||||
response = await page.goto('https://example.com', wait_until='networkidle', timeout=10000)
|
||||
title = await page.title()
|
||||
|
||||
print(f"✅ Successfully navigated to example.com")
|
||||
print(f" Title: {title}")
|
||||
print(f" Status: {response.status if response else 'No response'}")
|
||||
|
||||
await context.close()
|
||||
await browser.close()
|
||||
|
||||
return True
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ Playwright test failed: {e}")
|
||||
return False
|
||||
|
||||
async def test_browser_utils():
|
||||
"""Test the updated browser_utils module"""
|
||||
print("\nTesting browser_utils module...")
|
||||
|
||||
try:
|
||||
from app.utils.browser_utils import safe_browser_operation
|
||||
|
||||
async def test_operation(page):
|
||||
await page.goto('https://example.com', wait_until='networkidle', timeout=10000)
|
||||
title = await page.title()
|
||||
return {"status": "success", "title": title}
|
||||
|
||||
result = await safe_browser_operation('https://example.com', test_operation)
|
||||
|
||||
if result.get('status') == 'success':
|
||||
print(f"✅ browser_utils test passed")
|
||||
print(f" Title: {result.get('title')}")
|
||||
return True
|
||||
else:
|
||||
print(f"❌ browser_utils test failed: {result}")
|
||||
return False
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ browser_utils test failed: {e}")
|
||||
return False
|
||||
|
||||
async def test_browser_service():
|
||||
"""Test the updated browser service"""
|
||||
print("\nTesting browser service...")
|
||||
|
||||
try:
|
||||
from app.services.browser import visit_url_service
|
||||
|
||||
result = await visit_url_service('https://example.com')
|
||||
|
||||
if result.get('status') == 'success':
|
||||
print(f"✅ browser service test passed")
|
||||
print(f" Content length: {len(result.get('content', ''))}")
|
||||
return True
|
||||
else:
|
||||
print(f"❌ browser service test failed: {result}")
|
||||
return False
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ browser service test failed: {e}")
|
||||
return False
|
||||
|
||||
async def main():
|
||||
"""Run all tests"""
|
||||
print("🧪 Running Playwright migration tests...\n")
|
||||
|
||||
tests = [
|
||||
test_playwright_installation,
|
||||
test_browser_utils,
|
||||
test_browser_service,
|
||||
]
|
||||
|
||||
results = []
|
||||
for test in tests:
|
||||
try:
|
||||
result = await test()
|
||||
results.append(result)
|
||||
except Exception as e:
|
||||
print(f"❌ Test {test.__name__} failed with exception: {e}")
|
||||
results.append(False)
|
||||
|
||||
print(f"\n📊 Test Results:")
|
||||
print(f" Passed: {sum(results)}/{len(results)}")
|
||||
|
||||
if all(results):
|
||||
print("🎉 All tests passed! Playwright migration is working correctly.")
|
||||
return 0
|
||||
else:
|
||||
print("❌ Some tests failed. Please check the errors above.")
|
||||
return 1
|
||||
|
||||
if __name__ == "__main__":
|
||||
exit_code = asyncio.run(main())
|
||||
sys.exit(exit_code)
|
||||
Reference in New Issue
Block a user