This commit is contained in:
@@ -82,6 +82,15 @@ BROWSER_INSTANCE_TIMEOUT_MINUTES=10 # Force close old browser/page instances af
|
|||||||
CACHE_EXPIRY_HOURS=36 # Cache expiry in hours (default: 36)
|
CACHE_EXPIRY_HOURS=36 # Cache expiry in hours (default: 36)
|
||||||
CLEANUP_CRON=0 3 * * * # Cache cleanup schedule (default: daily at 3 AM)
|
CLEANUP_CRON=0 3 * * * # Cache cleanup schedule (default: daily at 3 AM)
|
||||||
RATE_LIMIT_MINUTE=60 # Requests per minute (default: 60)
|
RATE_LIMIT_MINUTE=60 # Requests per minute (default: 60)
|
||||||
|
|
||||||
|
# Database Configuration
|
||||||
|
# If all PostgreSQL credentials are provided, PostgreSQL will be used
|
||||||
|
# Otherwise, SQLite will be used as fallback
|
||||||
|
POSTGRES_HOST= # PostgreSQL host (optional)
|
||||||
|
POSTGRES_PORT=5432 # PostgreSQL port (default: 5432)
|
||||||
|
POSTGRES_DB= # PostgreSQL database name (optional)
|
||||||
|
POSTGRES_USER= # PostgreSQL username (optional)
|
||||||
|
POSTGRES_PASSWORD= # PostgreSQL password (optional)
|
||||||
```
|
```
|
||||||
|
|
||||||
## API Endpoints
|
## API Endpoints
|
||||||
@@ -138,11 +147,13 @@ python monitor.py your-api-key 30 http://localhost:8000 cleanup
|
|||||||
|
|
||||||
## Docker Usage
|
## Docker Usage
|
||||||
|
|
||||||
|
### Using SQLite (Default)
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Build the image
|
# Build the image
|
||||||
docker build -t playwright-api .
|
docker build -t playwright-api .
|
||||||
|
|
||||||
# Run with environment variables
|
# Run with SQLite (default)
|
||||||
docker run -d \
|
docker run -d \
|
||||||
--name playwright-api \
|
--name playwright-api \
|
||||||
-p 8000:8000 \
|
-p 8000:8000 \
|
||||||
@@ -153,6 +164,58 @@ docker run -d \
|
|||||||
playwright-api
|
playwright-api
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Using PostgreSQL
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Run with PostgreSQL
|
||||||
|
docker run -d \
|
||||||
|
--name playwright-api \
|
||||||
|
-p 8000:8000 \
|
||||||
|
-e API_KEY=your-api-key \
|
||||||
|
-e MAX_BROWSERS=3 \
|
||||||
|
-e BROWSER_TTL=1800 \
|
||||||
|
-e POSTGRES_HOST=your-postgres-host \
|
||||||
|
-e POSTGRES_PORT=5432 \
|
||||||
|
-e POSTGRES_DB=your-database-name \
|
||||||
|
-e POSTGRES_USER=your-username \
|
||||||
|
-e POSTGRES_PASSWORD=your-password \
|
||||||
|
playwright-api
|
||||||
|
```
|
||||||
|
|
||||||
|
### Using Docker Compose with PostgreSQL
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
version: "3.8"
|
||||||
|
services:
|
||||||
|
postgres:
|
||||||
|
image: postgres:15
|
||||||
|
environment:
|
||||||
|
POSTGRES_DB: playwright_cache
|
||||||
|
POSTGRES_USER: playwright_user
|
||||||
|
POSTGRES_PASSWORD: your_password
|
||||||
|
volumes:
|
||||||
|
- postgres_data:/var/lib/postgresql/data
|
||||||
|
ports:
|
||||||
|
- "5432:5432"
|
||||||
|
|
||||||
|
playwright-api:
|
||||||
|
build: .
|
||||||
|
ports:
|
||||||
|
- "8000:8000"
|
||||||
|
environment:
|
||||||
|
API_KEY: your-api-key
|
||||||
|
POSTGRES_HOST: postgres
|
||||||
|
POSTGRES_PORT: 5432
|
||||||
|
POSTGRES_DB: playwright_cache
|
||||||
|
POSTGRES_USER: playwright_user
|
||||||
|
POSTGRES_PASSWORD: your_password
|
||||||
|
depends_on:
|
||||||
|
- postgres
|
||||||
|
|
||||||
|
volumes:
|
||||||
|
postgres_data:
|
||||||
|
```
|
||||||
|
|
||||||
## Testing the Migration
|
## Testing the Migration
|
||||||
|
|
||||||
To verify that the Playwright migration works correctly:
|
To verify that the Playwright migration works correctly:
|
||||||
@@ -196,12 +259,37 @@ This will test:
|
|||||||
2. Check Docker build logs for browser installation
|
2. Check Docker build logs for browser installation
|
||||||
3. Verify system dependencies are installed
|
3. Verify system dependencies are installed
|
||||||
|
|
||||||
|
## Database Configuration
|
||||||
|
|
||||||
|
The API supports both SQLite and PostgreSQL for caching:
|
||||||
|
|
||||||
|
### SQLite (Default)
|
||||||
|
|
||||||
|
- **Automatic**: Used when no PostgreSQL credentials are provided
|
||||||
|
- **File-based**: Database stored in `/db/cache.db` (or `cache.db` as fallback)
|
||||||
|
- **Simple setup**: No additional services required
|
||||||
|
- **Suitable for**: Development, testing, and small deployments
|
||||||
|
|
||||||
|
### PostgreSQL
|
||||||
|
|
||||||
|
- **Configured via environment variables**: Set all PostgreSQL credentials to enable
|
||||||
|
- **Better performance**: For high-traffic applications
|
||||||
|
- **Scalable**: Can handle concurrent connections better
|
||||||
|
- **Suitable for**: Production deployments, high-traffic scenarios
|
||||||
|
|
||||||
|
### Migration
|
||||||
|
|
||||||
|
- **Automatic detection**: The system automatically chooses the database based on configuration
|
||||||
|
- **No data migration needed**: Each database type maintains its own cache
|
||||||
|
- **Backward compatible**: Existing SQLite setups continue to work unchanged
|
||||||
|
|
||||||
## Performance Tips
|
## Performance Tips
|
||||||
|
|
||||||
1. **Use caching**: The API caches results for 36 hours by default
|
1. **Use caching**: The API caches results for 36 hours by default
|
||||||
2. **Monitor resources**: Use the monitoring script to track usage
|
2. **Monitor resources**: Use the monitoring script to track usage
|
||||||
3. **Adjust limits**: Tune `MAX_BROWSERS` and `MAX_CONCURRENT_OPERATIONS` based on your server capacity
|
3. **Adjust limits**: Tune `MAX_BROWSERS` and `MAX_CONCURRENT_OPERATIONS` based on your server capacity
|
||||||
4. **Regular cleanup**: The system automatically recycles browsers every 30 minutes
|
4. **Regular cleanup**: The system automatically recycles browsers every 30 minutes
|
||||||
|
5. **Database choice**: Use PostgreSQL for high-traffic production deployments
|
||||||
|
|
||||||
## Logs
|
## Logs
|
||||||
|
|
||||||
|
|||||||
@@ -17,5 +17,16 @@ BROWSER_INSTANCE_TIMEOUT_MINUTES = int(os.getenv('BROWSER_INSTANCE_TIMEOUT_MINUT
|
|||||||
# Define custom user agent
|
# Define custom user agent
|
||||||
CUSTOM_USER_AGENT = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/69.0.3497.100 Safari/537.36'
|
CUSTOM_USER_AGENT = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/69.0.3497.100 Safari/537.36'
|
||||||
|
|
||||||
# Database path
|
# Database configuration
|
||||||
|
# Use PostgreSQL if credentials are provided, otherwise use SQLite
|
||||||
|
POSTGRES_HOST = os.getenv('POSTGRES_HOST')
|
||||||
|
POSTGRES_PORT = os.getenv('POSTGRES_PORT', '5432')
|
||||||
|
POSTGRES_DB = os.getenv('POSTGRES_DB')
|
||||||
|
POSTGRES_USER = os.getenv('POSTGRES_USER')
|
||||||
|
POSTGRES_PASSWORD = os.getenv('POSTGRES_PASSWORD')
|
||||||
|
|
||||||
|
# Determine database type
|
||||||
|
USE_POSTGRES = all([POSTGRES_HOST, POSTGRES_DB, POSTGRES_USER, POSTGRES_PASSWORD])
|
||||||
|
|
||||||
|
# Database path (for SQLite)
|
||||||
DB_PATH = '/db/cache.db'
|
DB_PATH = '/db/cache.db'
|
||||||
@@ -3,10 +3,34 @@ import sqlite3
|
|||||||
import time
|
import time
|
||||||
import json
|
import json
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from app.config import DB_PATH, CACHE_EXPIRY_HOURS
|
from app.config import (
|
||||||
|
DB_PATH, CACHE_EXPIRY_HOURS, USE_POSTGRES,
|
||||||
|
POSTGRES_HOST, POSTGRES_PORT, POSTGRES_DB, POSTGRES_USER, POSTGRES_PASSWORD
|
||||||
|
)
|
||||||
|
|
||||||
# Initialize SQLite database
|
# Import PostgreSQL dependencies only if needed
|
||||||
def init_db():
|
if USE_POSTGRES:
|
||||||
|
import psycopg2
|
||||||
|
from psycopg2.extras import RealDictCursor
|
||||||
|
|
||||||
|
class DatabaseManager:
|
||||||
|
def __init__(self):
|
||||||
|
self.db_type = "postgresql" if USE_POSTGRES else "sqlite"
|
||||||
|
self.connection_params = None
|
||||||
|
|
||||||
|
if USE_POSTGRES:
|
||||||
|
self.connection_params = {
|
||||||
|
'host': POSTGRES_HOST,
|
||||||
|
'port': POSTGRES_PORT,
|
||||||
|
'database': POSTGRES_DB,
|
||||||
|
'user': POSTGRES_USER,
|
||||||
|
'password': POSTGRES_PASSWORD
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
self._init_sqlite_path()
|
||||||
|
|
||||||
|
def _init_sqlite_path(self):
|
||||||
|
"""Initialize SQLite database path with fallback logic"""
|
||||||
global DB_PATH
|
global DB_PATH
|
||||||
|
|
||||||
# Try to use the mounted volume first
|
# Try to use the mounted volume first
|
||||||
@@ -43,8 +67,50 @@ def init_db():
|
|||||||
db_path = 'cache.db'
|
db_path = 'cache.db'
|
||||||
print(f"Using local database file: {db_path}")
|
print(f"Using local database file: {db_path}")
|
||||||
|
|
||||||
|
DB_PATH = db_path
|
||||||
|
|
||||||
|
def get_connection(self):
|
||||||
|
"""Get database connection based on configured database type"""
|
||||||
|
if USE_POSTGRES:
|
||||||
|
return psycopg2.connect(**self.connection_params)
|
||||||
|
else:
|
||||||
|
return sqlite3.connect(DB_PATH)
|
||||||
|
|
||||||
|
def init_db(self):
|
||||||
|
"""Initialize database and create tables"""
|
||||||
|
if USE_POSTGRES:
|
||||||
|
self._init_postgres_db()
|
||||||
|
else:
|
||||||
|
self._init_sqlite_db()
|
||||||
|
|
||||||
|
def _init_postgres_db(self):
|
||||||
|
"""Initialize PostgreSQL database"""
|
||||||
try:
|
try:
|
||||||
conn = sqlite3.connect(db_path)
|
conn = self.get_connection()
|
||||||
|
cursor = conn.cursor()
|
||||||
|
|
||||||
|
# Create cache table
|
||||||
|
cursor.execute('''
|
||||||
|
CREATE TABLE IF NOT EXISTS cache (
|
||||||
|
url TEXT,
|
||||||
|
route TEXT,
|
||||||
|
data TEXT,
|
||||||
|
timestamp BIGINT,
|
||||||
|
PRIMARY KEY (url, route)
|
||||||
|
)
|
||||||
|
''')
|
||||||
|
|
||||||
|
conn.commit()
|
||||||
|
conn.close()
|
||||||
|
print(f"PostgreSQL database initialized at {POSTGRES_HOST}:{POSTGRES_PORT}/{POSTGRES_DB}")
|
||||||
|
except Exception as e:
|
||||||
|
print(f"Error initializing PostgreSQL database: {e}")
|
||||||
|
raise
|
||||||
|
|
||||||
|
def _init_sqlite_db(self):
|
||||||
|
"""Initialize SQLite database"""
|
||||||
|
try:
|
||||||
|
conn = self.get_connection()
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
cursor.execute('''
|
cursor.execute('''
|
||||||
CREATE TABLE IF NOT EXISTS cache (
|
CREATE TABLE IF NOT EXISTS cache (
|
||||||
@@ -57,16 +123,15 @@ def init_db():
|
|||||||
''')
|
''')
|
||||||
conn.commit()
|
conn.commit()
|
||||||
conn.close()
|
conn.close()
|
||||||
print(f"Database initialized at {db_path}")
|
print(f"SQLite database initialized at {DB_PATH}")
|
||||||
# Update the global DB_PATH
|
|
||||||
DB_PATH = db_path
|
|
||||||
except sqlite3.OperationalError as e:
|
except sqlite3.OperationalError as e:
|
||||||
print(f"Error initializing database at {db_path}: {e}")
|
print(f"Error initializing database at {DB_PATH}: {e}")
|
||||||
# Fallback to using a local database file if the mounted volume has permission issues
|
# Fallback to using a local database file if the mounted volume has permission issues
|
||||||
db_path = 'cache.db'
|
global DB_PATH
|
||||||
print(f"Falling back to local database file: {db_path}")
|
DB_PATH = 'cache.db'
|
||||||
|
print(f"Falling back to local database file: {DB_PATH}")
|
||||||
try:
|
try:
|
||||||
conn = sqlite3.connect(db_path)
|
conn = sqlite3.connect(DB_PATH)
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
cursor.execute('''
|
cursor.execute('''
|
||||||
CREATE TABLE IF NOT EXISTS cache (
|
CREATE TABLE IF NOT EXISTS cache (
|
||||||
@@ -79,16 +144,14 @@ def init_db():
|
|||||||
''')
|
''')
|
||||||
conn.commit()
|
conn.commit()
|
||||||
conn.close()
|
conn.close()
|
||||||
print(f"Local database initialized at {db_path}")
|
print(f"Local database initialized at {DB_PATH}")
|
||||||
# Update the global DB_PATH
|
|
||||||
DB_PATH = db_path
|
|
||||||
except sqlite3.OperationalError as e2:
|
except sqlite3.OperationalError as e2:
|
||||||
print(f"Error initializing local database: {e2}")
|
print(f"Error initializing local database: {e2}")
|
||||||
raise
|
raise
|
||||||
|
|
||||||
# Get cached data if it exists and is not older than the expiry time
|
def get_cached_data(self, url, route):
|
||||||
def get_cached_data(url, route):
|
"""Get cached data if it exists and is not older than the expiry time"""
|
||||||
conn = sqlite3.connect(DB_PATH)
|
conn = self.get_connection()
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
|
|
||||||
# Calculate cache expiry time
|
# Calculate cache expiry time
|
||||||
@@ -99,7 +162,7 @@ def get_cached_data(url, route):
|
|||||||
cache_expiry = int(time.time()) - (CACHE_EXPIRY_HOURS * 60 * 60) # Convert hours to seconds
|
cache_expiry = int(time.time()) - (CACHE_EXPIRY_HOURS * 60 * 60) # Convert hours to seconds
|
||||||
|
|
||||||
cursor.execute(
|
cursor.execute(
|
||||||
"SELECT data FROM cache WHERE url = ? AND route = ? AND timestamp > ?",
|
"SELECT data FROM cache WHERE url = %s AND route = %s AND timestamp > %s",
|
||||||
(url, route, cache_expiry)
|
(url, route, cache_expiry)
|
||||||
)
|
)
|
||||||
result = cursor.fetchone()
|
result = cursor.fetchone()
|
||||||
@@ -110,28 +173,35 @@ def get_cached_data(url, route):
|
|||||||
return json.loads(result[0])
|
return json.loads(result[0])
|
||||||
return None
|
return None
|
||||||
|
|
||||||
# Save data to cache
|
def save_to_cache(self, url, route, data):
|
||||||
def save_to_cache(url, route, data):
|
"""Save data to cache"""
|
||||||
conn = sqlite3.connect(DB_PATH)
|
conn = self.get_connection()
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
timestamp = int(time.time())
|
timestamp = int(time.time())
|
||||||
|
|
||||||
# Convert data to JSON string
|
# Convert data to JSON string
|
||||||
data_json = json.dumps(data)
|
data_json = json.dumps(data)
|
||||||
|
|
||||||
|
if USE_POSTGRES:
|
||||||
|
cursor.execute(
|
||||||
|
"INSERT INTO cache (url, route, data, timestamp) VALUES (%s, %s, %s, %s) ON CONFLICT (url, route) DO UPDATE SET data = %s, timestamp = %s",
|
||||||
|
(url, route, data_json, timestamp, data_json, timestamp)
|
||||||
|
)
|
||||||
|
else:
|
||||||
cursor.execute(
|
cursor.execute(
|
||||||
"INSERT OR REPLACE INTO cache (url, route, data, timestamp) VALUES (?, ?, ?, ?)",
|
"INSERT OR REPLACE INTO cache (url, route, data, timestamp) VALUES (?, ?, ?, ?)",
|
||||||
(url, route, data_json, timestamp)
|
(url, route, data_json, timestamp)
|
||||||
)
|
)
|
||||||
|
|
||||||
conn.commit()
|
conn.commit()
|
||||||
conn.close()
|
conn.close()
|
||||||
print(f"Saved to cache: {url} on route {route}")
|
print(f"Saved to cache: {url} on route {route}")
|
||||||
|
|
||||||
# Function to clean up old cache entries
|
def cleanup_old_cache_entries(self):
|
||||||
def cleanup_old_cache_entries():
|
"""Clean up old cache entries"""
|
||||||
try:
|
try:
|
||||||
print(f"Running scheduled cache cleanup (entries older than {CACHE_EXPIRY_HOURS} hours, pagination: {CACHE_EXPIRY_HOURS * 31} hours)")
|
print(f"Running scheduled cache cleanup (entries older than {CACHE_EXPIRY_HOURS} hours, pagination: {CACHE_EXPIRY_HOURS * 31} hours)")
|
||||||
conn = sqlite3.connect(DB_PATH)
|
conn = self.get_connection()
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
|
|
||||||
# Calculate the timestamp for entries older than the expiry time
|
# Calculate the timestamp for entries older than the expiry time
|
||||||
@@ -139,18 +209,18 @@ def cleanup_old_cache_entries():
|
|||||||
pagination_expiry_timestamp = int(time.time()) - (CACHE_EXPIRY_HOURS * 31 * 60 * 60)
|
pagination_expiry_timestamp = int(time.time()) - (CACHE_EXPIRY_HOURS * 31 * 60 * 60)
|
||||||
|
|
||||||
# Get count of entries to be deleted (non-pagination)
|
# Get count of entries to be deleted (non-pagination)
|
||||||
cursor.execute("SELECT COUNT(*) FROM cache WHERE route != 'pagination' AND timestamp < ?", (expiry_timestamp,))
|
cursor.execute("SELECT COUNT(*) FROM cache WHERE route != %s AND timestamp < %s", ('pagination', expiry_timestamp))
|
||||||
count_non_pagination = cursor.fetchone()[0]
|
count_non_pagination = cursor.fetchone()[0]
|
||||||
|
|
||||||
# Get count of pagination entries to be deleted
|
# Get count of pagination entries to be deleted
|
||||||
cursor.execute("SELECT COUNT(*) FROM cache WHERE route = 'pagination' AND timestamp < ?", (pagination_expiry_timestamp,))
|
cursor.execute("SELECT COUNT(*) FROM cache WHERE route = %s AND timestamp < %s", ('pagination', pagination_expiry_timestamp))
|
||||||
count_pagination = cursor.fetchone()[0]
|
count_pagination = cursor.fetchone()[0]
|
||||||
|
|
||||||
# Delete old non-pagination entries
|
# Delete old non-pagination entries
|
||||||
cursor.execute("DELETE FROM cache WHERE route != 'pagination' AND timestamp < ?", (expiry_timestamp,))
|
cursor.execute("DELETE FROM cache WHERE route != %s AND timestamp < %s", ('pagination', expiry_timestamp))
|
||||||
|
|
||||||
# Delete old pagination entries
|
# Delete old pagination entries
|
||||||
cursor.execute("DELETE FROM cache WHERE route = 'pagination' AND timestamp < ?", (pagination_expiry_timestamp,))
|
cursor.execute("DELETE FROM cache WHERE route = %s AND timestamp < %s", ('pagination', pagination_expiry_timestamp))
|
||||||
|
|
||||||
conn.commit()
|
conn.commit()
|
||||||
conn.close()
|
conn.close()
|
||||||
@@ -158,3 +228,23 @@ def cleanup_old_cache_entries():
|
|||||||
print(f"Cache cleanup completed: {count_non_pagination} non-pagination entries and {count_pagination} pagination entries removed")
|
print(f"Cache cleanup completed: {count_non_pagination} non-pagination entries and {count_pagination} pagination entries removed")
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"Error during cache cleanup: {e}")
|
print(f"Error during cache cleanup: {e}")
|
||||||
|
|
||||||
|
# Create global database manager instance
|
||||||
|
db_manager = DatabaseManager()
|
||||||
|
|
||||||
|
# Backward compatibility functions
|
||||||
|
def init_db():
|
||||||
|
"""Initialize database (backward compatibility)"""
|
||||||
|
db_manager.init_db()
|
||||||
|
|
||||||
|
def get_cached_data(url, route):
|
||||||
|
"""Get cached data (backward compatibility)"""
|
||||||
|
return db_manager.get_cached_data(url, route)
|
||||||
|
|
||||||
|
def save_to_cache(url, route, data):
|
||||||
|
"""Save data to cache (backward compatibility)"""
|
||||||
|
db_manager.save_to_cache(url, route, data)
|
||||||
|
|
||||||
|
def cleanup_old_cache_entries():
|
||||||
|
"""Clean up old cache entries (backward compatibility)"""
|
||||||
|
db_manager.cleanup_old_cache_entries()
|
||||||
@@ -7,7 +7,6 @@ from app.services.browser import (
|
|||||||
visit_url_service,
|
visit_url_service,
|
||||||
extract_seo_service,
|
extract_seo_service,
|
||||||
extract_meta_tags_service,
|
extract_meta_tags_service,
|
||||||
detect_pagination_service,
|
|
||||||
capture_outgoing_calls_service,
|
capture_outgoing_calls_service,
|
||||||
get_resulting_url_service
|
get_resulting_url_service
|
||||||
)
|
)
|
||||||
@@ -98,33 +97,6 @@ async def extract_meta_tags(url: str, skipCache: bool = False, x_api_key: Option
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
raise HTTPException(status_code=500, detail=str(e))
|
raise HTTPException(status_code=500, detail=str(e))
|
||||||
|
|
||||||
@router.get("/pagination")
|
|
||||||
async def detect_pagination(url: str, skipCache: bool = False, x_api_key: Optional[str] = Header(None)):
|
|
||||||
"""Detect pagination on a website and determine the pagination pattern"""
|
|
||||||
# Validate API key
|
|
||||||
if not x_api_key or x_api_key != API_KEY:
|
|
||||||
raise HTTPException(status_code=401, detail="Invalid API key")
|
|
||||||
|
|
||||||
# Decode URL if it's encoded
|
|
||||||
decoded_url = unquote(url)
|
|
||||||
|
|
||||||
# Check cache first (unless skipCache is True)
|
|
||||||
if not skipCache:
|
|
||||||
cached_result = get_cached_data(decoded_url, "pagination")
|
|
||||||
if cached_result:
|
|
||||||
return cached_result
|
|
||||||
|
|
||||||
try:
|
|
||||||
# Call the service function
|
|
||||||
result = await detect_pagination_service(decoded_url)
|
|
||||||
|
|
||||||
# Save to cache
|
|
||||||
if(result["status"] == "success"):
|
|
||||||
save_to_cache(decoded_url, "pagination", result)
|
|
||||||
|
|
||||||
return result
|
|
||||||
except Exception as e:
|
|
||||||
raise HTTPException(status_code=500, detail=str(e))
|
|
||||||
|
|
||||||
@router.get("/outgoing-calls")
|
@router.get("/outgoing-calls")
|
||||||
async def capture_outgoing_calls(url: str, skipCache: bool = False, x_api_key: Optional[str] = Header(None)):
|
async def capture_outgoing_calls(url: str, skipCache: bool = False, x_api_key: Optional[str] = Header(None)):
|
||||||
|
|||||||
@@ -146,638 +146,6 @@ async def extract_meta_tags_service(decoded_url):
|
|||||||
print(f"Error during meta tag extraction: {e}")
|
print(f"Error during meta tag extraction: {e}")
|
||||||
return {"status": "error", "url": decoded_url, "error": str(e)}
|
return {"status": "error", "url": decoded_url, "error": str(e)}
|
||||||
|
|
||||||
async def detect_pagination_service(decoded_url):
|
|
||||||
"""Service function to detect pagination on a website"""
|
|
||||||
print(f"Detecting pagination on: {decoded_url}")
|
|
||||||
|
|
||||||
# Define the operation to perform with the browser
|
|
||||||
async def pagination_operation(page):
|
|
||||||
try:
|
|
||||||
# Navigate to the URL
|
|
||||||
try:
|
|
||||||
await page.goto(decoded_url, wait_until='networkidle', timeout=30000)
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error navigating to URL: {e}")
|
|
||||||
# Try to get the current URL even if navigation failed
|
|
||||||
try:
|
|
||||||
original_url = page.url
|
|
||||||
except:
|
|
||||||
original_url = decoded_url
|
|
||||||
else:
|
|
||||||
original_url = page.url
|
|
||||||
|
|
||||||
# Check for common pagination indicators
|
|
||||||
try:
|
|
||||||
pagination_data = await page.evaluate('''() => {
|
|
||||||
const data = {
|
|
||||||
hasPagination: false,
|
|
||||||
paginationType: null,
|
|
||||||
paginationElements: [],
|
|
||||||
detectedParameter: null,
|
|
||||||
lastPageNumber: null
|
|
||||||
};
|
|
||||||
|
|
||||||
// Look for numbered pagination links (1, 2, 3...)
|
|
||||||
const numberedLinks = Array.from(document.querySelectorAll('a, button, span'))
|
|
||||||
.filter(el => {
|
|
||||||
const text = el.innerText.trim();
|
|
||||||
|
|
||||||
// check for data-page attribute
|
|
||||||
const dataPage = el.getAttribute('data-page');
|
|
||||||
if (dataPage) {
|
|
||||||
return /^[0-9]+$/.test(dataPage);
|
|
||||||
}
|
|
||||||
|
|
||||||
return /^[0-9]+$/.test(text) &&
|
|
||||||
(el.tagName === 'A' || el.onclick ||
|
|
||||||
el.closest('button, [role="button"]'));
|
|
||||||
});
|
|
||||||
|
|
||||||
if(numberedLinks.length <= 1) {
|
|
||||||
return data;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Look for next/prev buttons
|
|
||||||
const nextButtons = Array.from(document.querySelectorAll('a, button, [role="button"]'))
|
|
||||||
.filter(el => {
|
|
||||||
const text = el.innerText.trim().toLowerCase();
|
|
||||||
const ariaLabel = el.getAttribute('aria-label')?.toLowerCase() || '';
|
|
||||||
const hasNextIcon = el.querySelector('i.fa-chevron-right, i.fa-arrow-right, svg[class*="arrow"], svg[class*="next"]');
|
|
||||||
|
|
||||||
return text.includes('next') ||
|
|
||||||
text.includes('›') ||
|
|
||||||
text.includes('»') ||
|
|
||||||
text.includes('→') ||
|
|
||||||
ariaLabel.includes('next') ||
|
|
||||||
hasNextIcon;
|
|
||||||
});
|
|
||||||
|
|
||||||
// Check for pagination containers
|
|
||||||
const paginationContainers = Array.from(document.querySelectorAll(
|
|
||||||
'.pagination, [class*="pagination"], [class*="pager"], nav[aria-label*="pagination"], [role="navigation"]'
|
|
||||||
));
|
|
||||||
|
|
||||||
// Collect all potential pagination elements
|
|
||||||
if (numberedLinks.length > 0) {
|
|
||||||
data.hasPagination = true;
|
|
||||||
data.paginationType = 'numbered';
|
|
||||||
|
|
||||||
// Get href attributes or other identifiers from numbered links
|
|
||||||
data.paginationElements = numberedLinks.slice(0, 5).map(el => {
|
|
||||||
return {
|
|
||||||
text: el.innerText.trim(),
|
|
||||||
dataPage: el.getAttribute('data-page'),
|
|
||||||
href: el.tagName === 'A' ? el.href : null,
|
|
||||||
classes: el.className,
|
|
||||||
id: el.id
|
|
||||||
};
|
|
||||||
});
|
|
||||||
|
|
||||||
// Try to find the last page number
|
|
||||||
const pageNumbers = numberedLinks
|
|
||||||
.map(el => parseInt(el.innerText.replace(/[^\d]/g, '')))
|
|
||||||
.filter(num => !isNaN(num));
|
|
||||||
console.log(pageNumbers);
|
|
||||||
if (pageNumbers.length > 0) {
|
|
||||||
data.lastPageNumber = Math.max(...pageNumbers);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Also look for a "last page" element that might have text like "Last" or "»"
|
|
||||||
const lastPageElement = Array.from(document.querySelectorAll('a, button'))
|
|
||||||
.find(el => {
|
|
||||||
const text = el.innerText.trim().toLowerCase();
|
|
||||||
const ariaLabel = el.getAttribute('aria-label')?.toLowerCase() || '';
|
|
||||||
return text.includes('last') ||
|
|
||||||
text === '»' ||
|
|
||||||
ariaLabel.includes('last page');
|
|
||||||
});
|
|
||||||
|
|
||||||
if (lastPageElement && lastPageElement.href) {
|
|
||||||
// Try to extract page number from the URL
|
|
||||||
try {
|
|
||||||
const url = new URL(lastPageElement.href);
|
|
||||||
// Check common pagination parameters
|
|
||||||
['page', 'p', 'pg'].forEach(param => {
|
|
||||||
if (url.searchParams.has(param)) {
|
|
||||||
const value = parseInt(url.searchParams.get(param));
|
|
||||||
if (!isNaN(value) && (data.lastPageNumber === null || value > data.lastPageNumber)) {
|
|
||||||
console.log("Last page number: " + value);
|
|
||||||
data.lastPageNumber = value;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
// Check for path-based pagination (like /page/10)
|
|
||||||
const pathMatch = url.pathname.match(/\/(?:page|p)\/(\d+)/i);
|
|
||||||
if (pathMatch && pathMatch[1]) {
|
|
||||||
const value = parseInt(pathMatch[1]);
|
|
||||||
if (!isNaN(value) && (data.lastPageNumber === null || value > data.lastPageNumber)) {
|
|
||||||
console.log("Last page number: " + value);
|
|
||||||
data.lastPageNumber = value;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} catch (e) {
|
|
||||||
console.error("Error parsing last page URL:", e);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} else if (nextButtons.length > 0) {
|
|
||||||
data.hasPagination = true;
|
|
||||||
data.paginationType = 'next-prev';
|
|
||||||
|
|
||||||
// Get information about next buttons
|
|
||||||
data.paginationElements = nextButtons.slice(0, 3).map(el => {
|
|
||||||
return {
|
|
||||||
text: el.innerText.trim(),
|
|
||||||
href: el.tagName === 'A' ? el.href : null,
|
|
||||||
classes: el.className,
|
|
||||||
id: el.id
|
|
||||||
};
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check for URL parameters that might indicate pagination
|
|
||||||
const currentUrl = window.location.href;
|
|
||||||
const urlParams = new URL(currentUrl).searchParams;
|
|
||||||
|
|
||||||
// Common pagination parameters
|
|
||||||
const paginationParams = ['page', 'p', 'pg', 'offset', 'o', 'from', 'start', 'limit'];
|
|
||||||
|
|
||||||
for (const param of paginationParams) {
|
|
||||||
if (urlParams.has(param)) {
|
|
||||||
data.detectedParameter = {
|
|
||||||
name: param,
|
|
||||||
value: urlParams.get(param)
|
|
||||||
};
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return data;
|
|
||||||
}''')
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error during JavaScript evaluation for pagination detection: {e}")
|
|
||||||
# Return a safe default if JavaScript evaluation fails
|
|
||||||
pagination_data = {
|
|
||||||
'hasPagination': False,
|
|
||||||
'paginationType': None,
|
|
||||||
'paginationElements': [],
|
|
||||||
'detectedParameter': None,
|
|
||||||
'lastPageNumber': None
|
|
||||||
}
|
|
||||||
|
|
||||||
print(f"Pagination data: {pagination_data}")
|
|
||||||
# If pagination is detected, try to navigate to the next page by clicking
|
|
||||||
next_page_url = None
|
|
||||||
pagination_parameter = None
|
|
||||||
url_template = None
|
|
||||||
step_size = None
|
|
||||||
original_parsed = None # Initialize the variable
|
|
||||||
|
|
||||||
if pagination_data['hasPagination']:
|
|
||||||
print("Pagination detected, attempting to click on a pagination element")
|
|
||||||
|
|
||||||
# Capture the original URL before any navigation
|
|
||||||
original_url_before_navigation = page.url
|
|
||||||
|
|
||||||
# Always try to click on a pagination element, regardless of type
|
|
||||||
clicked = False
|
|
||||||
|
|
||||||
# First try to click on a numbered link (preferably "2" if we're on page 1)
|
|
||||||
try:
|
|
||||||
clicked = await page.evaluate('''() => {
|
|
||||||
try {
|
|
||||||
// First try to find and click on a "2" link or button
|
|
||||||
const page2Elements = Array.from(document.querySelectorAll('a[href], button, [role="button"]'))
|
|
||||||
.filter(el => {
|
|
||||||
// Check for text content "2"
|
|
||||||
if (el.innerText.trim() === '2') {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check for href with page=2 or similar (for anchor elements)
|
|
||||||
if (el.tagName === 'A' && el.href) {
|
|
||||||
try {
|
|
||||||
const url = new URL(el.href, window.location.origin);
|
|
||||||
const paginationParams = ['page', 'p', 'pg', 'offset', 'o', 'from', 'start', 'limit',
|
|
||||||
'currentpage', 'pagenum', 'pageNumber', 'paged'];
|
|
||||||
|
|
||||||
for (const param of paginationParams) {
|
|
||||||
if (url.searchParams.has(param) && url.searchParams.get(param) === '2') {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check for path-based pagination like /page/2/ or /vacatures/page/2
|
|
||||||
const pathMatch = url.pathname.match(/\/(page|p)\/2\/?$/i);
|
|
||||||
if (pathMatch) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check for path-based pagination where /page/2 is appended to the current path
|
|
||||||
const currentPath = window.location.pathname;
|
|
||||||
const expectedPath = currentPath.replace(/\/$/, '') + '/page/2';
|
|
||||||
if (url.pathname === expectedPath) {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
} catch (e) {}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check for data attributes that might indicate pagination
|
|
||||||
if (el.getAttribute('data-page') === '2' ||
|
|
||||||
el.getAttribute('data-pagenumber') === '2' ||
|
|
||||||
el.getAttribute('data-page-number') === '2') {
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
return false;
|
|
||||||
});
|
|
||||||
|
|
||||||
if (page2Elements.length > 0) {
|
|
||||||
console.log("Clicking on page 2 element");
|
|
||||||
page2Elements[0].click();
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// If no "2" link found, try any numbered link or button
|
|
||||||
const numberedElements = Array.from(document.querySelectorAll('a[href], button, [role="button"]'))
|
|
||||||
.filter(el => /^\d+$/.test(el.innerText.trim()));
|
|
||||||
|
|
||||||
if (numberedElements.length > 0) {
|
|
||||||
// Sort by number and get the second one (likely page 2)
|
|
||||||
const sorted = numberedElements.sort((a, b) => {
|
|
||||||
return parseInt(a.innerText.trim()) - parseInt(b.innerText.trim());
|
|
||||||
});
|
|
||||||
|
|
||||||
// Get the second element if available (page 2), otherwise the first one
|
|
||||||
const elementToClick = sorted.length > 1 ? sorted[1] : sorted[0];
|
|
||||||
console.log("Clicking on numbered element: " + elementToClick.innerText);
|
|
||||||
elementToClick.click();
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
// If no numbered links, try next button
|
|
||||||
const nextTexts = ['next', '›', '»', '→'];
|
|
||||||
const nextElements = Array.from(document.querySelectorAll('a, button, [role="button"]'))
|
|
||||||
.filter(el => {
|
|
||||||
const text = el.textContent.trim().toLowerCase();
|
|
||||||
const ariaLabel = el.getAttribute('aria-label')?.toLowerCase() || '';
|
|
||||||
return nextTexts.some(t => text.includes(t)) ||
|
|
||||||
ariaLabel.includes('next') ||
|
|
||||||
el.querySelector('i.fa-chevron-right, i.fa-arrow-right, svg[class*="arrow"], svg[class*="next"]');
|
|
||||||
});
|
|
||||||
|
|
||||||
if (nextElements.length > 0) {
|
|
||||||
console.log("Clicking on next button");
|
|
||||||
nextElements[0].click();
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
return false;
|
|
||||||
} catch (error) {
|
|
||||||
console.error("Error during pagination click operation:", error);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
}''')
|
|
||||||
|
|
||||||
if clicked:
|
|
||||||
print("Successfully clicked on pagination element")
|
|
||||||
# Wait for navigation to complete
|
|
||||||
await page.wait_for_load_state('networkidle', timeout=10000)
|
|
||||||
await asyncio.sleep(2)
|
|
||||||
next_page_url = page.url
|
|
||||||
else:
|
|
||||||
print("No clickable pagination element found")
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error clicking on pagination element: {e}")
|
|
||||||
# Continue with the process even if clicking fails
|
|
||||||
|
|
||||||
print(f"Next page URL: {next_page_url}")
|
|
||||||
|
|
||||||
# If we successfully navigated to the next page, analyze the URL difference
|
|
||||||
if next_page_url and next_page_url != original_url_before_navigation:
|
|
||||||
print('Searching for pagination parameter')
|
|
||||||
# Parse both URLs using Python instead of JavaScript to avoid execution context issues
|
|
||||||
try:
|
|
||||||
from urllib.parse import urlparse, parse_qs
|
|
||||||
|
|
||||||
original_parsed_url = urlparse(original_url_before_navigation)
|
|
||||||
current_parsed_url = urlparse(next_page_url)
|
|
||||||
|
|
||||||
# Check for differences in query parameters
|
|
||||||
param_diff = None
|
|
||||||
original_params = parse_qs(original_parsed_url.query)
|
|
||||||
current_params = parse_qs(current_parsed_url.query)
|
|
||||||
|
|
||||||
# Common pagination parameters to check
|
|
||||||
pagination_params = ['page', 'p', 'pg', 'offset', 'o', 'from', 'start', 'limit', 'currentPage', 'current_page', 'currentpage', 'pagenum', 'pageNumber', 'paged']
|
|
||||||
|
|
||||||
for param in pagination_params:
|
|
||||||
original_value = original_params.get(param, [None])[0]
|
|
||||||
current_value = current_params.get(param, [None])[0]
|
|
||||||
|
|
||||||
if original_value != current_value and current_value is not None:
|
|
||||||
param_diff = {
|
|
||||||
'name': param,
|
|
||||||
'originalValue': original_value,
|
|
||||||
'currentValue': current_value
|
|
||||||
}
|
|
||||||
break
|
|
||||||
|
|
||||||
# Check for path differences
|
|
||||||
path_diff = None
|
|
||||||
original_path = original_parsed_url.path
|
|
||||||
current_path = current_parsed_url.path
|
|
||||||
|
|
||||||
if original_path != current_path:
|
|
||||||
original_segments = [s for s in original_path.split('/') if s]
|
|
||||||
current_segments = [s for s in current_path.split('/') if s]
|
|
||||||
|
|
||||||
# Case 1: Same number of segments - find the one that changed
|
|
||||||
if len(original_segments) == len(current_segments):
|
|
||||||
for i, (orig_seg, curr_seg) in enumerate(zip(original_segments, current_segments)):
|
|
||||||
if orig_seg != curr_seg:
|
|
||||||
# Check if the difference is numeric
|
|
||||||
if orig_seg.isdigit() and curr_seg.isdigit():
|
|
||||||
path_diff = {
|
|
||||||
'type': 'replace',
|
|
||||||
'index': i,
|
|
||||||
'originalValue': orig_seg,
|
|
||||||
'currentValue': curr_seg
|
|
||||||
}
|
|
||||||
break
|
|
||||||
|
|
||||||
# Case 2: Current path has more segments - check for added pagination segments
|
|
||||||
elif len(current_segments) > len(original_segments):
|
|
||||||
# Look for patterns like /page/NUMBER or /p/NUMBER at the end
|
|
||||||
import re
|
|
||||||
page_pattern = re.compile(r'^(page|p)/(\d+)$', re.IGNORECASE)
|
|
||||||
|
|
||||||
# Check the last two segments of the current path
|
|
||||||
if len(current_segments) >= 2:
|
|
||||||
last_two_segments = '/'.join(current_segments[-2:])
|
|
||||||
match = page_pattern.match(last_two_segments)
|
|
||||||
|
|
||||||
if match:
|
|
||||||
path_diff = {
|
|
||||||
'type': 'append',
|
|
||||||
'pageSegment': match.group(1), # 'page' or 'p'
|
|
||||||
'pageNumber': match.group(2), # the actual number
|
|
||||||
'originalSegments': original_segments,
|
|
||||||
'currentSegments': current_segments
|
|
||||||
}
|
|
||||||
|
|
||||||
# If no pattern match, check if the last segment is numeric
|
|
||||||
if not path_diff and current_segments:
|
|
||||||
last_segment = current_segments[-1]
|
|
||||||
if last_segment.isdigit():
|
|
||||||
path_diff = {
|
|
||||||
'type': 'append',
|
|
||||||
'pageSegment': None,
|
|
||||||
'pageNumber': last_segment,
|
|
||||||
'originalSegments': original_segments,
|
|
||||||
'currentSegments': current_segments
|
|
||||||
}
|
|
||||||
|
|
||||||
original_parsed = {
|
|
||||||
'paramDiff': param_diff,
|
|
||||||
'pathDiff': path_diff,
|
|
||||||
'originalUrl': original_url_before_navigation,
|
|
||||||
'currentUrl': next_page_url
|
|
||||||
}
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error during URL analysis: {e}")
|
|
||||||
original_parsed = {
|
|
||||||
'paramDiff': None,
|
|
||||||
'pathDiff': None,
|
|
||||||
'originalUrl': original_url_before_navigation,
|
|
||||||
'currentUrl': next_page_url,
|
|
||||||
'error': str(e)
|
|
||||||
}
|
|
||||||
|
|
||||||
# Determine the pagination parameter and create URL template
|
|
||||||
if original_parsed:
|
|
||||||
print(original_parsed)
|
|
||||||
if original_parsed.get('paramDiff'):
|
|
||||||
param_name = original_parsed['paramDiff']['name']
|
|
||||||
pagination_parameter = {
|
|
||||||
'type': 'query',
|
|
||||||
'name': param_name,
|
|
||||||
'value': original_parsed['paramDiff']['currentValue']
|
|
||||||
}
|
|
||||||
|
|
||||||
# --- STEP SIZE DETECTION FOR QUERY PARAM ---
|
|
||||||
orig_val = original_parsed['paramDiff']['originalValue'] if original_parsed['paramDiff']['originalValue'] is not None else 0
|
|
||||||
curr_val = original_parsed['paramDiff']['currentValue']
|
|
||||||
try:
|
|
||||||
if orig_val is not None and curr_val is not None:
|
|
||||||
orig_num = int(orig_val)
|
|
||||||
curr_num = int(curr_val)
|
|
||||||
step_size = abs(curr_num - orig_num)
|
|
||||||
except Exception:
|
|
||||||
step_size = None
|
|
||||||
|
|
||||||
# Create URL template for query parameter
|
|
||||||
try:
|
|
||||||
from urllib.parse import urlparse, urlencode, parse_qs
|
|
||||||
|
|
||||||
parsed_url = urlparse(original_url_before_navigation)
|
|
||||||
params = parse_qs(parsed_url.query)
|
|
||||||
params[param_name] = ['{PAGE_NUMBER}']
|
|
||||||
|
|
||||||
# Reconstruct the URL
|
|
||||||
new_query = urlencode(params, doseq=True)
|
|
||||||
url_template = f"{parsed_url.scheme}://{parsed_url.netloc}{parsed_url.path}"
|
|
||||||
if new_query:
|
|
||||||
url_template += f"?{new_query}"
|
|
||||||
if parsed_url.fragment:
|
|
||||||
url_template += f"#{parsed_url.fragment}"
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error creating URL template for query parameter: {e}")
|
|
||||||
url_template = None
|
|
||||||
|
|
||||||
elif original_parsed.get('pathDiff'):
|
|
||||||
path_diff = original_parsed['pathDiff']
|
|
||||||
|
|
||||||
if path_diff.get('type') == 'replace':
|
|
||||||
# Handle existing path segment replacement
|
|
||||||
path_index = path_diff['index']
|
|
||||||
pagination_parameter = {
|
|
||||||
'type': 'path',
|
|
||||||
'index': path_index,
|
|
||||||
'value': path_diff['currentValue']
|
|
||||||
}
|
|
||||||
|
|
||||||
# --- STEP SIZE DETECTION FOR PATH PARAM ---
|
|
||||||
orig_val = path_diff['originalValue']
|
|
||||||
curr_val = path_diff['currentValue']
|
|
||||||
try:
|
|
||||||
if orig_val is not None and curr_val is not None:
|
|
||||||
orig_num = int(orig_val)
|
|
||||||
curr_num = int(curr_val)
|
|
||||||
step_size = abs(curr_num - orig_num)
|
|
||||||
except Exception:
|
|
||||||
step_size = None
|
|
||||||
|
|
||||||
# Create URL template for path parameter replacement
|
|
||||||
try:
|
|
||||||
from urllib.parse import urlparse
|
|
||||||
|
|
||||||
parsed_url = urlparse(original_url_before_navigation)
|
|
||||||
path_segments = [s for s in parsed_url.path.split('/') if s]
|
|
||||||
path_segments[path_index] = '{PAGE_NUMBER}'
|
|
||||||
|
|
||||||
# Reconstruct the URL
|
|
||||||
new_path = '/' + '/'.join(path_segments)
|
|
||||||
url_template = f"{parsed_url.scheme}://{parsed_url.netloc}{new_path}"
|
|
||||||
if parsed_url.query:
|
|
||||||
url_template += f"?{parsed_url.query}"
|
|
||||||
if parsed_url.fragment:
|
|
||||||
url_template += f"#{parsed_url.fragment}"
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error creating URL template for path parameter: {e}")
|
|
||||||
url_template = None
|
|
||||||
|
|
||||||
elif path_diff.get('type') == 'append':
|
|
||||||
# Handle new pagination segments being appended
|
|
||||||
pagination_parameter = {
|
|
||||||
'type': 'path_append',
|
|
||||||
'pageSegment': path_diff.get('pageSegment'),
|
|
||||||
'pageNumber': path_diff['pageNumber']
|
|
||||||
}
|
|
||||||
|
|
||||||
# --- STEP SIZE DETECTION FOR APPENDED PATH PARAM ---
|
|
||||||
try:
|
|
||||||
curr_val = path_diff['pageNumber']
|
|
||||||
# For appended pagination, assume we started from page 1 (implicit)
|
|
||||||
orig_num = 1
|
|
||||||
curr_num = int(curr_val)
|
|
||||||
step_size = abs(curr_num - orig_num)
|
|
||||||
except Exception:
|
|
||||||
step_size = None
|
|
||||||
|
|
||||||
# Create URL template for appended path parameter
|
|
||||||
try:
|
|
||||||
from urllib.parse import urlparse
|
|
||||||
|
|
||||||
parsed_url = urlparse(original_url_before_navigation)
|
|
||||||
new_path = parsed_url.path
|
|
||||||
|
|
||||||
# Remove trailing slash if present
|
|
||||||
if new_path.endswith('/'):
|
|
||||||
new_path = new_path[:-1]
|
|
||||||
|
|
||||||
# Append the pagination segment
|
|
||||||
if path_diff.get('pageSegment'):
|
|
||||||
new_path += f"/{path_diff['pageSegment']}/{{PAGE_NUMBER}}"
|
|
||||||
else:
|
|
||||||
new_path += "/{PAGE_NUMBER}"
|
|
||||||
|
|
||||||
# Reconstruct the URL
|
|
||||||
url_template = f"{parsed_url.scheme}://{parsed_url.netloc}{new_path}"
|
|
||||||
if parsed_url.query:
|
|
||||||
url_template += f"?{parsed_url.query}"
|
|
||||||
if parsed_url.fragment:
|
|
||||||
url_template += f"#{parsed_url.fragment}"
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error creating URL template for appended path parameter: {e}")
|
|
||||||
url_template = None
|
|
||||||
|
|
||||||
if url_template:
|
|
||||||
# Decode URL-encoded characters in the template
|
|
||||||
import urllib.parse
|
|
||||||
url_template = urllib.parse.unquote(url_template)
|
|
||||||
|
|
||||||
# If we couldn't determine the URL template from navigation, try to infer it
|
|
||||||
if not url_template and pagination_data['detectedParameter']:
|
|
||||||
param_name = pagination_data['detectedParameter']['name']
|
|
||||||
try:
|
|
||||||
from urllib.parse import urlparse, urlencode, parse_qs
|
|
||||||
|
|
||||||
parsed_url = urlparse(original_url_before_navigation)
|
|
||||||
params = parse_qs(parsed_url.query)
|
|
||||||
params[param_name] = ['{PAGE_NUMBER}']
|
|
||||||
|
|
||||||
# Reconstruct the URL
|
|
||||||
new_query = urlencode(params, doseq=True)
|
|
||||||
url_template = f"{parsed_url.scheme}://{parsed_url.netloc}{parsed_url.path}"
|
|
||||||
if new_query:
|
|
||||||
url_template += f"?{new_query}"
|
|
||||||
if parsed_url.fragment:
|
|
||||||
url_template += f"#{parsed_url.fragment}"
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error creating inferred URL template: {e}")
|
|
||||||
url_template = None
|
|
||||||
|
|
||||||
# If still no template and we have pagination elements, try to infer from the current URL structure
|
|
||||||
if not url_template and pagination_data['hasPagination']:
|
|
||||||
try:
|
|
||||||
from urllib.parse import urlparse
|
|
||||||
import re
|
|
||||||
|
|
||||||
parsed_url = urlparse(original_url_before_navigation)
|
|
||||||
path = parsed_url.path
|
|
||||||
|
|
||||||
# Remove trailing slash if present
|
|
||||||
if path.endswith('/'):
|
|
||||||
path = path[:-1]
|
|
||||||
|
|
||||||
# Check if the current URL already has a pagination pattern
|
|
||||||
page_pattern = re.compile(r'/(page|p)/\d+$', re.IGNORECASE)
|
|
||||||
if page_pattern.search(path):
|
|
||||||
# Replace the existing page number with placeholder
|
|
||||||
path = page_pattern.sub(r'/\1/{PAGE_NUMBER}', path)
|
|
||||||
else:
|
|
||||||
# Add pagination pattern
|
|
||||||
path += '/page/{PAGE_NUMBER}'
|
|
||||||
|
|
||||||
# Reconstruct the URL
|
|
||||||
url_template = f"{parsed_url.scheme}://{parsed_url.netloc}{path}"
|
|
||||||
if parsed_url.query:
|
|
||||||
url_template += f"?{parsed_url.query}"
|
|
||||||
if parsed_url.fragment:
|
|
||||||
url_template += f"#{parsed_url.fragment}"
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error creating fallback URL template: {e}")
|
|
||||||
url_template = None
|
|
||||||
|
|
||||||
# Return the pagination detection results with a simplified structure
|
|
||||||
result = {
|
|
||||||
"status": "success",
|
|
||||||
"hasPagination": pagination_data['hasPagination'],
|
|
||||||
"urlTemplate": url_template,
|
|
||||||
"lastPage": pagination_data['lastPageNumber'],
|
|
||||||
"stepSize": step_size if step_size is not None and step_size >= 5 else 1
|
|
||||||
}
|
|
||||||
|
|
||||||
return result
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
print(f"Error during pagination detection: {e}")
|
|
||||||
return {
|
|
||||||
"status": "error",
|
|
||||||
"hasPagination": False,
|
|
||||||
"urlTemplate": None,
|
|
||||||
"lastPage": None,
|
|
||||||
"stepSize": None,
|
|
||||||
"error": str(e)
|
|
||||||
}
|
|
||||||
|
|
||||||
# Perform the operation
|
|
||||||
result = await safe_browser_operation(decoded_url, pagination_operation)
|
|
||||||
|
|
||||||
# Check if the result is an error from safe_browser_operation
|
|
||||||
if isinstance(result, dict) and result.get("status") == "error":
|
|
||||||
return result
|
|
||||||
|
|
||||||
return result
|
|
||||||
|
|
||||||
async def capture_outgoing_calls_service(decoded_url):
|
async def capture_outgoing_calls_service(decoded_url):
|
||||||
"""Service function to capture outgoing API calls from a website"""
|
"""Service function to capture outgoing API calls from a website"""
|
||||||
print(f"Capturing outgoing calls from: {decoded_url}")
|
print(f"Capturing outgoing calls from: {decoded_url}")
|
||||||
|
|||||||
@@ -1,11 +1,10 @@
|
|||||||
import sqlite3
|
|
||||||
import time
|
import time
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from app.database import DB_PATH, CACHE_EXPIRY_HOURS
|
from app.database import db_manager, CACHE_EXPIRY_HOURS
|
||||||
|
|
||||||
def clear_cache():
|
def clear_cache():
|
||||||
"""Clear the entire cache database"""
|
"""Clear the entire cache database"""
|
||||||
conn = sqlite3.connect(DB_PATH)
|
conn = db_manager.get_connection()
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
cursor.execute("DELETE FROM cache")
|
cursor.execute("DELETE FROM cache")
|
||||||
conn.commit()
|
conn.commit()
|
||||||
@@ -14,7 +13,7 @@ def clear_cache():
|
|||||||
|
|
||||||
def get_cache_stats():
|
def get_cache_stats():
|
||||||
"""Get cache statistics"""
|
"""Get cache statistics"""
|
||||||
conn = sqlite3.connect(DB_PATH)
|
conn = db_manager.get_connection()
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
|
|
||||||
# Get total entries
|
# Get total entries
|
||||||
@@ -27,7 +26,7 @@ def get_cache_stats():
|
|||||||
|
|
||||||
# Get recent entries (last 24 hours)
|
# Get recent entries (last 24 hours)
|
||||||
recent_timestamp = int(time.time()) - (24 * 60 * 60)
|
recent_timestamp = int(time.time()) - (24 * 60 * 60)
|
||||||
cursor.execute("SELECT COUNT(*) FROM cache WHERE timestamp > ?", (recent_timestamp,))
|
cursor.execute("SELECT COUNT(*) FROM cache WHERE timestamp > %s", (recent_timestamp,))
|
||||||
recent_entries = cursor.fetchone()[0]
|
recent_entries = cursor.fetchone()[0]
|
||||||
|
|
||||||
# Get oldest entry timestamp
|
# Get oldest entry timestamp
|
||||||
@@ -50,6 +49,7 @@ def get_cache_stats():
|
|||||||
"recent_entries": recent_entries,
|
"recent_entries": recent_entries,
|
||||||
"oldest_entry": oldest_date,
|
"oldest_entry": oldest_date,
|
||||||
"newest_entry": newest_date,
|
"newest_entry": newest_date,
|
||||||
"cache_expiry_hours": CACHE_EXPIRY_HOURS
|
"cache_expiry_hours": CACHE_EXPIRY_HOURS,
|
||||||
|
"database_type": db_manager.db_type
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -5,3 +5,4 @@ psutil==6.0.0
|
|||||||
apscheduler==3.10.4
|
apscheduler==3.10.4
|
||||||
aiohttp==3.9.1
|
aiohttp==3.9.1
|
||||||
beautifulsoup4==4.12.2
|
beautifulsoup4==4.12.2
|
||||||
|
psycopg2-binary==2.9.9
|
||||||
@@ -0,0 +1,152 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""
|
||||||
|
Test script to verify database configuration for both SQLite and PostgreSQL
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
import json
|
||||||
|
|
||||||
|
# Add the app directory to the Python path
|
||||||
|
sys.path.insert(0, os.path.join(os.path.dirname(__file__), 'app'))
|
||||||
|
|
||||||
|
def test_sqlite():
|
||||||
|
"""Test SQLite database functionality"""
|
||||||
|
print("=== Testing SQLite Database ===")
|
||||||
|
|
||||||
|
# Clear any existing PostgreSQL environment variables
|
||||||
|
for var in ['POSTGRES_HOST', 'POSTGRES_DB', 'POSTGRES_USER', 'POSTGRES_PASSWORD']:
|
||||||
|
if var in os.environ:
|
||||||
|
del os.environ[var]
|
||||||
|
|
||||||
|
try:
|
||||||
|
from app.database import db_manager, init_db, get_cached_data, save_to_cache, cleanup_old_cache_entries
|
||||||
|
|
||||||
|
print(f"Database type: {db_manager.db_type}")
|
||||||
|
|
||||||
|
# Initialize database
|
||||||
|
init_db()
|
||||||
|
print("✓ Database initialized successfully")
|
||||||
|
|
||||||
|
# Test cache operations
|
||||||
|
test_url = "https://example.com"
|
||||||
|
test_route = "test"
|
||||||
|
test_data = {"title": "Test Page", "content": "Test content"}
|
||||||
|
|
||||||
|
# Save to cache
|
||||||
|
save_to_cache(test_url, test_route, test_data)
|
||||||
|
print("✓ Data saved to cache")
|
||||||
|
|
||||||
|
# Retrieve from cache
|
||||||
|
cached_data = get_cached_data(test_url, test_route)
|
||||||
|
if cached_data and cached_data == test_data:
|
||||||
|
print("✓ Data retrieved from cache successfully")
|
||||||
|
else:
|
||||||
|
print("✗ Failed to retrieve data from cache")
|
||||||
|
return False
|
||||||
|
|
||||||
|
# Test cache stats
|
||||||
|
from app.services.cache import get_cache_stats
|
||||||
|
stats = get_cache_stats()
|
||||||
|
if stats['status'] == 'success':
|
||||||
|
print("✓ Cache stats retrieved successfully")
|
||||||
|
print(f" Database type: {stats['stats']['database_type']}")
|
||||||
|
print(f" Total entries: {stats['stats']['total_entries']}")
|
||||||
|
else:
|
||||||
|
print("✗ Failed to get cache stats")
|
||||||
|
return False
|
||||||
|
|
||||||
|
print("✓ SQLite database test passed!")
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"✗ SQLite test failed: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
def test_postgres():
|
||||||
|
"""Test PostgreSQL database functionality"""
|
||||||
|
print("\n=== Testing PostgreSQL Database ===")
|
||||||
|
|
||||||
|
# Check if PostgreSQL credentials are available
|
||||||
|
required_vars = ['POSTGRES_HOST', 'POSTGRES_DB', 'POSTGRES_USER', 'POSTGRES_PASSWORD']
|
||||||
|
missing_vars = [var for var in required_vars if not os.getenv(var)]
|
||||||
|
|
||||||
|
if missing_vars:
|
||||||
|
print(f"⚠ PostgreSQL test skipped - missing environment variables: {', '.join(missing_vars)}")
|
||||||
|
print("Set these variables to test PostgreSQL:")
|
||||||
|
print(" POSTGRES_HOST=your-host")
|
||||||
|
print(" POSTGRES_DB=your-database")
|
||||||
|
print(" POSTGRES_USER=your-username")
|
||||||
|
print(" POSTGRES_PASSWORD=your-password")
|
||||||
|
return True # Not a failure, just skipped
|
||||||
|
|
||||||
|
try:
|
||||||
|
from app.database import db_manager, init_db, get_cached_data, save_to_cache
|
||||||
|
|
||||||
|
print(f"Database type: {db_manager.db_type}")
|
||||||
|
|
||||||
|
# Initialize database
|
||||||
|
init_db()
|
||||||
|
print("✓ Database initialized successfully")
|
||||||
|
|
||||||
|
# Test cache operations
|
||||||
|
test_url = "https://example.com"
|
||||||
|
test_route = "test"
|
||||||
|
test_data = {"title": "Test Page", "content": "Test content"}
|
||||||
|
|
||||||
|
# Save to cache
|
||||||
|
save_to_cache(test_url, test_route, test_data)
|
||||||
|
print("✓ Data saved to cache")
|
||||||
|
|
||||||
|
# Retrieve from cache
|
||||||
|
cached_data = get_cached_data(test_url, test_route)
|
||||||
|
if cached_data and cached_data == test_data:
|
||||||
|
print("✓ Data retrieved from cache successfully")
|
||||||
|
else:
|
||||||
|
print("✗ Failed to retrieve data from cache")
|
||||||
|
return False
|
||||||
|
|
||||||
|
# Test cache stats
|
||||||
|
from app.services.cache import get_cache_stats
|
||||||
|
stats = get_cache_stats()
|
||||||
|
if stats['status'] == 'success':
|
||||||
|
print("✓ Cache stats retrieved successfully")
|
||||||
|
print(f" Database type: {stats['stats']['database_type']}")
|
||||||
|
print(f" Total entries: {stats['stats']['total_entries']}")
|
||||||
|
else:
|
||||||
|
print("✗ Failed to get cache stats")
|
||||||
|
return False
|
||||||
|
|
||||||
|
print("✓ PostgreSQL database test passed!")
|
||||||
|
return True
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"✗ PostgreSQL test failed: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
def main():
|
||||||
|
"""Run all database tests"""
|
||||||
|
print("Database Configuration Test")
|
||||||
|
print("=" * 50)
|
||||||
|
|
||||||
|
# Test SQLite
|
||||||
|
sqlite_success = test_sqlite()
|
||||||
|
|
||||||
|
# Test PostgreSQL
|
||||||
|
postgres_success = test_postgres()
|
||||||
|
|
||||||
|
print("\n" + "=" * 50)
|
||||||
|
print("Test Results:")
|
||||||
|
print(f"SQLite: {'✓ PASSED' if sqlite_success else '✗ FAILED'}")
|
||||||
|
print(f"PostgreSQL: {'✓ PASSED' if postgres_success else '✗ FAILED'}")
|
||||||
|
|
||||||
|
if sqlite_success and postgres_success:
|
||||||
|
print("\n🎉 All tests passed! Database configuration is working correctly.")
|
||||||
|
return 0
|
||||||
|
else:
|
||||||
|
print("\n❌ Some tests failed. Please check the configuration.")
|
||||||
|
return 1
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
Reference in New Issue
Block a user