split up application
Build and Push Docker Images / build-and-push (push) Successful in 33s

This commit is contained in:
2025-05-02 21:15:30 +02:00
parent 2b893ca47a
commit de3919a4b5
15 changed files with 1020 additions and 1 deletions
@@ -0,0 +1 @@
# This file is intentionally left empty to make the directory a Python package
@@ -0,0 +1,564 @@
from app.utils.browser_utils import safe_browser_operation
async def visit_url_service(decoded_url):
"""Service function to visit a URL and get its content"""
print(f"Visiting URL: {decoded_url}")
# Define the operation to perform with the browser
async def visit_operation(page):
try:
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
if not response:
print(f"Warning: No response object returned for {decoded_url}")
# Get page content
content = await page.content()
return {"status": "success", "content": content}
except Exception as e:
print(f"Error during page navigation: {e}")
# Try to get content anyway
try:
content = await page.content()
return {"status": "partial", "content": content, "error": str(e)}
except:
raise Exception(f"Failed to get page content: {str(e)}")
# Perform the operation
return await safe_browser_operation(decoded_url, visit_operation)
async def extract_seo_service(decoded_url):
"""Service function to extract SEO information from a website"""
print(f"Extracting SEO from: {decoded_url}")
# Define the operation to perform with the browser
async def seo_operation(page):
try:
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
# Extract SEO information
seo_data = await page.evaluate('''() => {
const data = {
title: document.title || '',
description: '',
canonical: '',
h1: [],
h2: [],
images: 0,
links: 0
};
// Get meta description
const metaDescription = document.querySelector('meta[name="description"]');
if (metaDescription) {
data.description = metaDescription.getAttribute('content') || '';
}
// Get canonical link
const canonicalLink = document.querySelector('link[rel="canonical"]');
if (canonicalLink) {
data.canonical = canonicalLink.getAttribute('href') || '';
}
// Get h1 tags
document.querySelectorAll('h1').forEach(h1 => {
const text = h1.innerText.trim();
if (text) data.h1.push(text);
});
// Get h2 tags
document.querySelectorAll('h2').forEach(h2 => {
const text = h2.innerText.trim();
if (text) data.h2.push(text);
});
// Count images
data.images = document.querySelectorAll('img').length;
// Count links
data.links = document.querySelectorAll('a').length;
return data;
}''')
result = {
"status": "success",
"url": decoded_url,
"seo": seo_data
}
return result
except Exception as e:
print(f"Error during SEO extraction: {e}")
return {"status": "error", "url": decoded_url, "error": str(e)}
# Perform the operation
return await safe_browser_operation(decoded_url, seo_operation)
async def extract_meta_tags_service(decoded_url):
"""Service function to extract meta tags from a website"""
print(f"Extracting meta tags from: {decoded_url}")
# Define the operation to perform with the browser
async def meta_operation(page):
try:
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
# Extract all meta tags
meta_tags = await page.evaluate('''() => {
const metas = Array.from(document.querySelectorAll('meta'));
return metas.map(meta => {
const attributes = {};
Array.from(meta.attributes).forEach(attr => {
attributes[attr.name] = attr.value;
});
return attributes;
});
}''')
# Extract Open Graph tags
og_tags = await page.evaluate('''() => {
const ogTags = {};
document.querySelectorAll('meta[property^="og:"]').forEach(tag => {
const property = tag.getAttribute('property');
ogTags[property] = tag.getAttribute('content');
});
return ogTags;
}''')
# Extract Twitter card tags
twitter_tags = await page.evaluate('''() => {
const twitterTags = {};
document.querySelectorAll('meta[name^="twitter:"]').forEach(tag => {
const name = tag.getAttribute('name');
twitterTags[name] = tag.getAttribute('content');
});
return twitterTags;
}''')
result = {
"status": "success",
"url": decoded_url,
"meta_tags": meta_tags,
"open_graph": og_tags,
"twitter_card": twitter_tags,
"title": await page.title()
}
return result
except Exception as e:
print(f"Error during meta tag extraction: {e}")
return {"status": "error", "url": decoded_url, "error": str(e)}
# Perform the operation
return await safe_browser_operation(decoded_url, meta_operation)
async def detect_pagination_service(decoded_url):
"""Service function to detect pagination on a website"""
print(f"Detecting pagination on: {decoded_url}")
# Define the operation to perform with the browser
async def pagination_operation(page):
try:
# Navigate to the URL
await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
original_url = page.url
print(f"Successfully loaded page: {original_url}")
# Analyze the page for pagination information
pagination_info = await page.evaluate('''() => {
// Find the last page number if available
const findLastPageNumber = () => {
// Get all links on the page
const links = Array.from(document.querySelectorAll('a'));
// Strategy 1: Find numeric links (page numbers)
const numericLinks = links.filter(link => {
const text = link.innerText.trim();
return /^[0-9]+$/.test(text) && link.href && link.href !== '#';
});
if (numericLinks.length > 0) {
const numericValues = numericLinks.map(link => parseInt(link.innerText.trim()));
return Math.max(...numericValues);
}
// Strategy 2: Look for "last page" link
const lastLinks = links.filter(link => {
const text = link.innerText.trim().toLowerCase();
const classes = (link.className || '').toLowerCase();
const ariaLabel = (link.getAttribute('aria-label') || '').toLowerCase();
return (text === 'last' ||
classes.includes('last') ||
ariaLabel.includes('last') ||
link.getAttribute('rel') === 'last');
});
if (lastLinks.length > 0) {
const lastLink = lastLinks[0];
const href = lastLink.href;
// Common patterns: page=X, /page/X, etc.
const pagePatterns = [
/[?&]page=(\d+)/,
/[?&]p=(\d+)/,
/[?&]pg=(\d+)/,
/\/page\/(\d+)/,
/\/p\/(\d+)/,
/\/paged\/(\d+)/
];
for (const pattern of pagePatterns) {
const match = href.match(pattern);
if (match && match[1]) {
return parseInt(match[1]);
}
}
}
// Strategy 3: Analyze all URLs for page numbers
const pageNumbersFromUrls = [];
links.forEach(link => {
if (!link.href || link.href === '#') return;
// Check for common pagination URL patterns
const patterns = [
/[?&]page=(\d+)/,
/[?&]p=(\d+)/,
/[?&]pg=(\d+)/,
/\/page\/(\d+)/,
/\/p\/(\d+)/,
/\/paged\/(\d+)/,
/\/pages\/(\d+)/
];
for (const pattern of patterns) {
const match = link.href.match(pattern);
if (match && match[1]) {
pageNumbersFromUrls.push(parseInt(match[1]));
}
}
});
if (pageNumbersFromUrls.length > 0) {
return Math.max(...pageNumbersFromUrls);
}
return null;
};
// Find a pagination link to click
const findPaginationLink = () => {
const links = Array.from(document.querySelectorAll('a'));
// Try to find a page "2" link first (most reliable)
const page2Link = links.find(link => {
const text = link.innerText.trim();
return text === '2' && link.href && link.href !== '#';
});
if (page2Link) {
return { element: page2Link, href: page2Link.href, type: 'numeric' };
}
// Try common "next page" selectors
const nextSelectors = [
'a.next',
'a.page-next',
'a[rel="next"]',
'a[aria-label="Next page"]',
'a[aria-label="next"]'
];
for (const selector of nextSelectors) {
const element = document.querySelector(selector);
if (element && element.href && element.href !== '#') {
return { element, href: element.href, type: 'next' };
}
}
// Look for any link that might be pagination
const paginationLinks = links.filter(link => {
if (!link.href || link.href === '#') return false;
const text = link.innerText.trim();
const href = link.href;
// Check for numeric text or next/prev indicators
const isNumeric = /^[0-9]+$/.test(text) && text !== '1';
const isNextPrev = /next|prev|previous|older|newer/i.test(text) ||
/[»«‹›<>]/.test(text);
// Check for page parameter in URL
const hasPageParam = /[?&]page=|[?&]p=|[?&]pg=|\/page\/|\/p\//.test(href);
return (isNumeric || isNextPrev || hasPageParam);
});
if (paginationLinks.length > 0) {
const link = paginationLinks[0];
return { element: link, href: link.href, type: 'other' };
}
return null;
};
const lastPage = findLastPageNumber();
const paginationLink = findPaginationLink();
if (paginationLink) {
// Click the link
paginationLink.element.click();
return {
clicked: true,
href: paginationLink.href,
type: paginationLink.type,
lastPage
};
}
return { clicked: false, lastPage };
}''')
# If no pagination was found or clicked
if not pagination_info.get('clicked', False):
return {
"status": "success",
"url": decoded_url,
"hasPagination": False,
"urlTemplate": None,
"lastPage": pagination_info.get('lastPage')
}
# Wait for navigation to complete after the click
try:
await page.waitForNavigation({'timeout': 10000, 'waitUntil': 'networkidle2'})
except Exception as e:
print(f"Navigation timeout: {e}")
# Get the new URL after clicking
next_page_url = page.url
# If URL didn't change, pagination might be handled by AJAX
if next_page_url == original_url:
return {
"status": "success",
"url": decoded_url,
"hasPagination": True,
"urlTemplate": "AJAX pagination (URL doesn't change)",
"lastPage": pagination_info.get('lastPage')
}
print(f"Navigation successful: {original_url} -> {next_page_url}")
# Analyze the URL structure to determine pagination pattern
url_template = await page.evaluate('''(originalUrl, nextPageUrl) => {
// Helper function to parse URL query parameters
const parseQueryParams = (url) => {
const params = {};
if (url.includes('?')) {
const queryString = url.split('?')[1].split('#')[0];
queryString.split('&').forEach(param => {
if (param.includes('=')) {
const [key, value] = param.split('=', 2);
params[key] = value;
}
});
}
return params;
};
// Check for query parameter based pagination
if (nextPageUrl.includes('?')) {
const originalParams = parseQueryParams(originalUrl);
const nextParams = parseQueryParams(nextPageUrl);
// Find parameters that changed or were added
let paginationParam = null;
// First check for common pagination parameter names
const commonPaginationParams = ['page', 'p', 'pg', 'paged', 'current_page', 'pagenum', 'pageNumber'];
for (const key of commonPaginationParams) {
if (key in nextParams &&
(!(key in originalParams) || originalParams[key] !== nextParams[key])) {
if (/^\d+$/.test(nextParams[key]) && parseInt(nextParams[key]) > 1) {
paginationParam = key;
break;
}
}
}
// If no common parameter found, check all parameters
if (!paginationParam) {
for (const [key, value] of Object.entries(nextParams)) {
// Check if parameter is new or changed
if (!(key in originalParams) || originalParams[key] !== value) {
// Check if the value is numeric and could be a page number
if (/^\d+$/.test(value) && parseInt(value) > 1) {
paginationParam = key;
break;
}
}
}
}
// If we found a pagination parameter
if (paginationParam) {
const baseUrl = nextPageUrl.split('?')[0];
// Reconstruct the URL template with all parameters
const queryParts = [];
for (const [key, value] of Object.entries(nextParams)) {
if (key === paginationParam) {
queryParts.push(`${key}={PAGE_NUMBER}`);
} else {
queryParts.push(`${key}=${value}`);
}
}
return `${baseUrl}?${queryParts.join('&')}`;
}
}
// Check for path-based pagination
const pathPatterns = ['/page/', '/p/', '/paged/', '/pages/'];
for (const pattern of pathPatterns) {
if (nextPageUrl.includes(pattern)) {
const parts = nextPageUrl.split(pattern);
let template = `${parts[0]}${pattern}{PAGE_NUMBER}`;
// Add any suffix after the page number
if (parts.length > 1 && parts[1].includes('/')) {
const suffix = parts[1].split('/', 1)[1];
if (suffix) {
template += `/${suffix}`;
}
}
return template;
}
}
// If we couldn't determine the pattern, try to make an educated guess
// For the specific case where a parameter like current_page=2 is added
const originalUrlObj = new URL(originalUrl);
const nextUrlObj = new URL(nextPageUrl);
// Check if the paths are the same but query params differ
if (originalUrlObj.pathname === nextUrlObj.pathname) {
const originalParams = parseQueryParams(originalUrl);
const nextParams = parseQueryParams(nextPageUrl);
// Find parameters that exist in next but not in original
const newParams = Object.keys(nextParams).filter(key => !(key in originalParams));
// If there's exactly one new parameter and it has a numeric value
if (newParams.length === 1 && /^\d+$/.test(nextParams[newParams[0]])) {
const paginationParam = newParams[0];
const baseUrl = nextPageUrl.split('?')[0];
// Reconstruct the URL template
const queryParts = [];
for (const [key, value] of Object.entries(nextParams)) {
if (key === paginationParam) {
queryParts.push(`${key}={PAGE_NUMBER}`);
} else {
queryParts.push(`${key}=${value}`);
}
}
return `${baseUrl}?${queryParts.join('&')}`;
}
}
// If we still couldn't determine the pattern, return both URLs as examples
return `Pattern unclear. Example: ${originalUrl} → ${nextPageUrl}`;
}''', original_url, next_page_url)
# Try to extract last page number from the next page if we didn't find it on the first page
if not pagination_info.get('lastPage'):
last_page_from_next = await page.evaluate('''() => {
// Get all links on the page
const links = Array.from(document.querySelectorAll('a'));
// Strategy 1: Find numeric links (page numbers)
const numericLinks = links.filter(link => {
const text = link.innerText.trim();
return /^[0-9]+$/.test(text) && link.href && link.href !== '#';
});
if (numericLinks.length > 0) {
const numericValues = numericLinks.map(link => parseInt(link.innerText.trim()));
return Math.max(...numericValues);
}
// Strategy 2: Analyze all URLs for page numbers
const pageNumbersFromUrls = [];
links.forEach(link => {
if (!link.href || link.href === '#') return;
// Check for common pagination URL patterns
const patterns = [
/[?&]page=(\d+)/,
/[?&]p=(\d+)/,
/[?&]pg=(\d+)/,
/\/page\/(\d+)/,
/\/p\/(\d+)/,
/\/paged\/(\d+)/,
/\/pages\/(\d+)/
];
for (const pattern of patterns) {
const match = link.href.match(pattern);
if (match && match[1]) {
pageNumbersFromUrls.push(parseInt(match[1]));
}
}
});
if (pageNumbersFromUrls.length > 0) {
return Math.max(...pageNumbersFromUrls);
}
return null;
}''')
if last_page_from_next:
pagination_info['lastPage'] = last_page_from_next
# Check if the pattern is unclear
has_pagination = True
url_string_template = str(url_template)
if url_string_template and url_string_template.startswith("Pattern unclear"):
has_pagination = False
url_template = None
result = {
"status": "success",
"url": decoded_url,
"hasPagination": has_pagination,
"urlTemplate": url_template,
"lastPage": pagination_info.get('lastPage'),
"originalUrl": original_url,
"nextPageUrl": next_page_url
}
return result
except Exception as e:
print(f"Error during pagination detection: {e}")
return {
"status": "error",
"url": decoded_url,
"error": str(e),
"hasPagination": False,
"urlTemplate": None,
"lastPage": None
}
# Perform the operation
return await safe_browser_operation(decoded_url, pagination_operation)
@@ -0,0 +1,55 @@
import sqlite3
import time
from datetime import datetime
from app.database import DB_PATH, CACHE_EXPIRY_HOURS
def clear_cache():
"""Clear the entire cache database"""
conn = sqlite3.connect(DB_PATH)
cursor = conn.cursor()
cursor.execute("DELETE FROM cache")
conn.commit()
conn.close()
return {"status": "success", "message": "Cache cleared successfully"}
def get_cache_stats():
"""Get cache statistics"""
conn = sqlite3.connect(DB_PATH)
cursor = conn.cursor()
# Get total entries
cursor.execute("SELECT COUNT(*) FROM cache")
total_entries = cursor.fetchone()[0]
# Get entries by route
cursor.execute("SELECT route, COUNT(*) FROM cache GROUP BY route")
routes = {route: count for route, count in cursor.fetchall()}
# Get recent entries (last 24 hours)
recent_timestamp = int(time.time()) - (24 * 60 * 60)
cursor.execute("SELECT COUNT(*) FROM cache WHERE timestamp > ?", (recent_timestamp,))
recent_entries = cursor.fetchone()[0]
# Get oldest entry timestamp
cursor.execute("SELECT MIN(timestamp) FROM cache")
oldest_timestamp = cursor.fetchone()[0]
oldest_date = datetime.fromtimestamp(oldest_timestamp).isoformat() if oldest_timestamp else None
# Get newest entry timestamp
cursor.execute("SELECT MAX(timestamp) FROM cache")
newest_timestamp = cursor.fetchone()[0]
newest_date = datetime.fromtimestamp(newest_timestamp).isoformat() if newest_timestamp else None
conn.close()
return {
"status": "success",
"stats": {
"total_entries": total_entries,
"entries_by_route": routes,
"recent_entries": recent_entries,
"oldest_entry": oldest_date,
"newest_entry": newest_date,
"cache_expiry_hours": CACHE_EXPIRY_HOURS
}
}