564 lines
23 KiB
Python
564 lines
23 KiB
Python
from app.utils.browser_utils import safe_browser_operation
|
|
|
|
async def visit_url_service(decoded_url):
|
|
"""Service function to visit a URL and get its content"""
|
|
print(f"Visiting URL: {decoded_url}")
|
|
|
|
# Define the operation to perform with the browser
|
|
async def visit_operation(page):
|
|
try:
|
|
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
|
if not response:
|
|
print(f"Warning: No response object returned for {decoded_url}")
|
|
|
|
# Get page content
|
|
content = await page.content()
|
|
return {"status": "success", "content": content}
|
|
except Exception as e:
|
|
print(f"Error during page navigation: {e}")
|
|
# Try to get content anyway
|
|
try:
|
|
content = await page.content()
|
|
return {"status": "partial", "content": content, "error": str(e)}
|
|
except:
|
|
raise Exception(f"Failed to get page content: {str(e)}")
|
|
|
|
# Perform the operation
|
|
return await safe_browser_operation(decoded_url, visit_operation)
|
|
|
|
async def extract_seo_service(decoded_url):
|
|
"""Service function to extract SEO information from a website"""
|
|
print(f"Extracting SEO from: {decoded_url}")
|
|
|
|
# Define the operation to perform with the browser
|
|
async def seo_operation(page):
|
|
try:
|
|
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
|
|
|
# Extract SEO information
|
|
seo_data = await page.evaluate('''() => {
|
|
const data = {
|
|
title: document.title || '',
|
|
description: '',
|
|
canonical: '',
|
|
h1: [],
|
|
h2: [],
|
|
images: 0,
|
|
links: 0
|
|
};
|
|
|
|
// Get meta description
|
|
const metaDescription = document.querySelector('meta[name="description"]');
|
|
if (metaDescription) {
|
|
data.description = metaDescription.getAttribute('content') || '';
|
|
}
|
|
|
|
// Get canonical link
|
|
const canonicalLink = document.querySelector('link[rel="canonical"]');
|
|
if (canonicalLink) {
|
|
data.canonical = canonicalLink.getAttribute('href') || '';
|
|
}
|
|
|
|
// Get h1 tags
|
|
document.querySelectorAll('h1').forEach(h1 => {
|
|
const text = h1.innerText.trim();
|
|
if (text) data.h1.push(text);
|
|
});
|
|
|
|
// Get h2 tags
|
|
document.querySelectorAll('h2').forEach(h2 => {
|
|
const text = h2.innerText.trim();
|
|
if (text) data.h2.push(text);
|
|
});
|
|
|
|
// Count images
|
|
data.images = document.querySelectorAll('img').length;
|
|
|
|
// Count links
|
|
data.links = document.querySelectorAll('a').length;
|
|
|
|
return data;
|
|
}''')
|
|
|
|
result = {
|
|
"status": "success",
|
|
"url": decoded_url,
|
|
"seo": seo_data
|
|
}
|
|
|
|
return result
|
|
|
|
except Exception as e:
|
|
print(f"Error during SEO extraction: {e}")
|
|
return {"status": "error", "url": decoded_url, "error": str(e)}
|
|
|
|
# Perform the operation
|
|
return await safe_browser_operation(decoded_url, seo_operation)
|
|
|
|
async def extract_meta_tags_service(decoded_url):
|
|
"""Service function to extract meta tags from a website"""
|
|
print(f"Extracting meta tags from: {decoded_url}")
|
|
|
|
# Define the operation to perform with the browser
|
|
async def meta_operation(page):
|
|
try:
|
|
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
|
|
|
# Extract all meta tags
|
|
meta_tags = await page.evaluate('''() => {
|
|
const metas = Array.from(document.querySelectorAll('meta'));
|
|
return metas.map(meta => {
|
|
const attributes = {};
|
|
Array.from(meta.attributes).forEach(attr => {
|
|
attributes[attr.name] = attr.value;
|
|
});
|
|
return attributes;
|
|
});
|
|
}''')
|
|
|
|
# Extract Open Graph tags
|
|
og_tags = await page.evaluate('''() => {
|
|
const ogTags = {};
|
|
document.querySelectorAll('meta[property^="og:"]').forEach(tag => {
|
|
const property = tag.getAttribute('property');
|
|
ogTags[property] = tag.getAttribute('content');
|
|
});
|
|
return ogTags;
|
|
}''')
|
|
|
|
# Extract Twitter card tags
|
|
twitter_tags = await page.evaluate('''() => {
|
|
const twitterTags = {};
|
|
document.querySelectorAll('meta[name^="twitter:"]').forEach(tag => {
|
|
const name = tag.getAttribute('name');
|
|
twitterTags[name] = tag.getAttribute('content');
|
|
});
|
|
return twitterTags;
|
|
}''')
|
|
|
|
result = {
|
|
"status": "success",
|
|
"url": decoded_url,
|
|
"meta_tags": meta_tags,
|
|
"open_graph": og_tags,
|
|
"twitter_card": twitter_tags,
|
|
"title": await page.title()
|
|
}
|
|
|
|
return result
|
|
|
|
except Exception as e:
|
|
print(f"Error during meta tag extraction: {e}")
|
|
return {"status": "error", "url": decoded_url, "error": str(e)}
|
|
|
|
# Perform the operation
|
|
return await safe_browser_operation(decoded_url, meta_operation)
|
|
|
|
async def detect_pagination_service(decoded_url):
|
|
"""Service function to detect pagination on a website"""
|
|
print(f"Detecting pagination on: {decoded_url}")
|
|
|
|
# Define the operation to perform with the browser
|
|
async def pagination_operation(page):
|
|
try:
|
|
# Navigate to the URL
|
|
await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
|
original_url = page.url
|
|
|
|
print(f"Successfully loaded page: {original_url}")
|
|
|
|
# Analyze the page for pagination information
|
|
pagination_info = await page.evaluate('''() => {
|
|
// Find the last page number if available
|
|
const findLastPageNumber = () => {
|
|
// Get all links on the page
|
|
const links = Array.from(document.querySelectorAll('a'));
|
|
|
|
// Strategy 1: Find numeric links (page numbers)
|
|
const numericLinks = links.filter(link => {
|
|
const text = link.innerText.trim();
|
|
return /^[0-9]+$/.test(text) && link.href && link.href !== '#';
|
|
});
|
|
|
|
if (numericLinks.length > 0) {
|
|
const numericValues = numericLinks.map(link => parseInt(link.innerText.trim()));
|
|
return Math.max(...numericValues);
|
|
}
|
|
|
|
// Strategy 2: Look for "last page" link
|
|
const lastLinks = links.filter(link => {
|
|
const text = link.innerText.trim().toLowerCase();
|
|
const classes = (link.className || '').toLowerCase();
|
|
const ariaLabel = (link.getAttribute('aria-label') || '').toLowerCase();
|
|
|
|
return (text === 'last' ||
|
|
classes.includes('last') ||
|
|
ariaLabel.includes('last') ||
|
|
link.getAttribute('rel') === 'last');
|
|
});
|
|
|
|
if (lastLinks.length > 0) {
|
|
const lastLink = lastLinks[0];
|
|
const href = lastLink.href;
|
|
|
|
// Common patterns: page=X, /page/X, etc.
|
|
const pagePatterns = [
|
|
/[?&]page=(\d+)/,
|
|
/[?&]p=(\d+)/,
|
|
/[?&]pg=(\d+)/,
|
|
/\/page\/(\d+)/,
|
|
/\/p\/(\d+)/,
|
|
/\/paged\/(\d+)/
|
|
];
|
|
|
|
for (const pattern of pagePatterns) {
|
|
const match = href.match(pattern);
|
|
if (match && match[1]) {
|
|
return parseInt(match[1]);
|
|
}
|
|
}
|
|
}
|
|
|
|
// Strategy 3: Analyze all URLs for page numbers
|
|
const pageNumbersFromUrls = [];
|
|
links.forEach(link => {
|
|
if (!link.href || link.href === '#') return;
|
|
|
|
// Check for common pagination URL patterns
|
|
const patterns = [
|
|
/[?&]page=(\d+)/,
|
|
/[?&]p=(\d+)/,
|
|
/[?&]pg=(\d+)/,
|
|
/\/page\/(\d+)/,
|
|
/\/p\/(\d+)/,
|
|
/\/paged\/(\d+)/,
|
|
/\/pages\/(\d+)/
|
|
];
|
|
|
|
for (const pattern of patterns) {
|
|
const match = link.href.match(pattern);
|
|
if (match && match[1]) {
|
|
pageNumbersFromUrls.push(parseInt(match[1]));
|
|
}
|
|
}
|
|
});
|
|
|
|
if (pageNumbersFromUrls.length > 0) {
|
|
return Math.max(...pageNumbersFromUrls);
|
|
}
|
|
|
|
return null;
|
|
};
|
|
|
|
// Find a pagination link to click
|
|
const findPaginationLink = () => {
|
|
const links = Array.from(document.querySelectorAll('a'));
|
|
|
|
// Try to find a page "2" link first (most reliable)
|
|
const page2Link = links.find(link => {
|
|
const text = link.innerText.trim();
|
|
return text === '2' && link.href && link.href !== '#';
|
|
});
|
|
|
|
if (page2Link) {
|
|
return { element: page2Link, href: page2Link.href, type: 'numeric' };
|
|
}
|
|
|
|
// Try common "next page" selectors
|
|
const nextSelectors = [
|
|
'a.next',
|
|
'a.page-next',
|
|
'a[rel="next"]',
|
|
'a[aria-label="Next page"]',
|
|
'a[aria-label="next"]'
|
|
];
|
|
|
|
for (const selector of nextSelectors) {
|
|
const element = document.querySelector(selector);
|
|
if (element && element.href && element.href !== '#') {
|
|
return { element, href: element.href, type: 'next' };
|
|
}
|
|
}
|
|
|
|
// Look for any link that might be pagination
|
|
const paginationLinks = links.filter(link => {
|
|
if (!link.href || link.href === '#') return false;
|
|
|
|
const text = link.innerText.trim();
|
|
const href = link.href;
|
|
|
|
// Check for numeric text or next/prev indicators
|
|
const isNumeric = /^[0-9]+$/.test(text) && text !== '1';
|
|
const isNextPrev = /next|prev|previous|older|newer/i.test(text) ||
|
|
/[»«‹›<>]/.test(text);
|
|
|
|
// Check for page parameter in URL
|
|
const hasPageParam = /[?&]page=|[?&]p=|[?&]pg=|\/page\/|\/p\//.test(href);
|
|
|
|
return (isNumeric || isNextPrev || hasPageParam);
|
|
});
|
|
|
|
if (paginationLinks.length > 0) {
|
|
const link = paginationLinks[0];
|
|
return { element: link, href: link.href, type: 'other' };
|
|
}
|
|
|
|
return null;
|
|
};
|
|
|
|
const lastPage = findLastPageNumber();
|
|
const paginationLink = findPaginationLink();
|
|
|
|
if (paginationLink) {
|
|
// Click the link
|
|
paginationLink.element.click();
|
|
return {
|
|
clicked: true,
|
|
href: paginationLink.href,
|
|
type: paginationLink.type,
|
|
lastPage
|
|
};
|
|
}
|
|
|
|
return { clicked: false, lastPage };
|
|
}''')
|
|
|
|
# If no pagination was found or clicked
|
|
if not pagination_info.get('clicked', False):
|
|
return {
|
|
"status": "success",
|
|
"url": decoded_url,
|
|
"hasPagination": False,
|
|
"urlTemplate": None,
|
|
"lastPage": pagination_info.get('lastPage')
|
|
}
|
|
|
|
# Wait for navigation to complete after the click
|
|
try:
|
|
await page.waitForNavigation({'timeout': 10000, 'waitUntil': 'networkidle2'})
|
|
except Exception as e:
|
|
print(f"Navigation timeout: {e}")
|
|
|
|
# Get the new URL after clicking
|
|
next_page_url = page.url
|
|
|
|
# If URL didn't change, pagination might be handled by AJAX
|
|
if next_page_url == original_url:
|
|
return {
|
|
"status": "success",
|
|
"url": decoded_url,
|
|
"hasPagination": True,
|
|
"urlTemplate": "AJAX pagination (URL doesn't change)",
|
|
"lastPage": pagination_info.get('lastPage')
|
|
}
|
|
|
|
print(f"Navigation successful: {original_url} -> {next_page_url}")
|
|
|
|
# Analyze the URL structure to determine pagination pattern
|
|
url_template = await page.evaluate('''(originalUrl, nextPageUrl) => {
|
|
// Helper function to parse URL query parameters
|
|
const parseQueryParams = (url) => {
|
|
const params = {};
|
|
if (url.includes('?')) {
|
|
const queryString = url.split('?')[1].split('#')[0];
|
|
queryString.split('&').forEach(param => {
|
|
if (param.includes('=')) {
|
|
const [key, value] = param.split('=', 2);
|
|
params[key] = value;
|
|
}
|
|
});
|
|
}
|
|
return params;
|
|
};
|
|
|
|
// Check for query parameter based pagination
|
|
if (nextPageUrl.includes('?')) {
|
|
const originalParams = parseQueryParams(originalUrl);
|
|
const nextParams = parseQueryParams(nextPageUrl);
|
|
|
|
// Find parameters that changed or were added
|
|
let paginationParam = null;
|
|
|
|
// First check for common pagination parameter names
|
|
const commonPaginationParams = ['page', 'p', 'pg', 'paged', 'current_page', 'pagenum', 'pageNumber'];
|
|
|
|
for (const key of commonPaginationParams) {
|
|
if (key in nextParams &&
|
|
(!(key in originalParams) || originalParams[key] !== nextParams[key])) {
|
|
if (/^\d+$/.test(nextParams[key]) && parseInt(nextParams[key]) > 1) {
|
|
paginationParam = key;
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
|
|
// If no common parameter found, check all parameters
|
|
if (!paginationParam) {
|
|
for (const [key, value] of Object.entries(nextParams)) {
|
|
// Check if parameter is new or changed
|
|
if (!(key in originalParams) || originalParams[key] !== value) {
|
|
// Check if the value is numeric and could be a page number
|
|
if (/^\d+$/.test(value) && parseInt(value) > 1) {
|
|
paginationParam = key;
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// If we found a pagination parameter
|
|
if (paginationParam) {
|
|
const baseUrl = nextPageUrl.split('?')[0];
|
|
|
|
// Reconstruct the URL template with all parameters
|
|
const queryParts = [];
|
|
for (const [key, value] of Object.entries(nextParams)) {
|
|
if (key === paginationParam) {
|
|
queryParts.push(`${key}={PAGE_NUMBER}`);
|
|
} else {
|
|
queryParts.push(`${key}=${value}`);
|
|
}
|
|
}
|
|
|
|
return `${baseUrl}?${queryParts.join('&')}`;
|
|
}
|
|
}
|
|
|
|
// Check for path-based pagination
|
|
const pathPatterns = ['/page/', '/p/', '/paged/', '/pages/'];
|
|
for (const pattern of pathPatterns) {
|
|
if (nextPageUrl.includes(pattern)) {
|
|
const parts = nextPageUrl.split(pattern);
|
|
let template = `${parts[0]}${pattern}{PAGE_NUMBER}`;
|
|
|
|
// Add any suffix after the page number
|
|
if (parts.length > 1 && parts[1].includes('/')) {
|
|
const suffix = parts[1].split('/', 1)[1];
|
|
if (suffix) {
|
|
template += `/${suffix}`;
|
|
}
|
|
}
|
|
|
|
return template;
|
|
}
|
|
}
|
|
|
|
// If we couldn't determine the pattern, try to make an educated guess
|
|
// For the specific case where a parameter like current_page=2 is added
|
|
const originalUrlObj = new URL(originalUrl);
|
|
const nextUrlObj = new URL(nextPageUrl);
|
|
|
|
// Check if the paths are the same but query params differ
|
|
if (originalUrlObj.pathname === nextUrlObj.pathname) {
|
|
const originalParams = parseQueryParams(originalUrl);
|
|
const nextParams = parseQueryParams(nextPageUrl);
|
|
|
|
// Find parameters that exist in next but not in original
|
|
const newParams = Object.keys(nextParams).filter(key => !(key in originalParams));
|
|
|
|
// If there's exactly one new parameter and it has a numeric value
|
|
if (newParams.length === 1 && /^\d+$/.test(nextParams[newParams[0]])) {
|
|
const paginationParam = newParams[0];
|
|
const baseUrl = nextPageUrl.split('?')[0];
|
|
|
|
// Reconstruct the URL template
|
|
const queryParts = [];
|
|
for (const [key, value] of Object.entries(nextParams)) {
|
|
if (key === paginationParam) {
|
|
queryParts.push(`${key}={PAGE_NUMBER}`);
|
|
} else {
|
|
queryParts.push(`${key}=${value}`);
|
|
}
|
|
}
|
|
|
|
return `${baseUrl}?${queryParts.join('&')}`;
|
|
}
|
|
}
|
|
|
|
// If we still couldn't determine the pattern, return both URLs as examples
|
|
return `Pattern unclear. Example: ${originalUrl} → ${nextPageUrl}`;
|
|
}''', original_url, next_page_url)
|
|
|
|
# Try to extract last page number from the next page if we didn't find it on the first page
|
|
if not pagination_info.get('lastPage'):
|
|
last_page_from_next = await page.evaluate('''() => {
|
|
// Get all links on the page
|
|
const links = Array.from(document.querySelectorAll('a'));
|
|
|
|
// Strategy 1: Find numeric links (page numbers)
|
|
const numericLinks = links.filter(link => {
|
|
const text = link.innerText.trim();
|
|
return /^[0-9]+$/.test(text) && link.href && link.href !== '#';
|
|
});
|
|
|
|
if (numericLinks.length > 0) {
|
|
const numericValues = numericLinks.map(link => parseInt(link.innerText.trim()));
|
|
return Math.max(...numericValues);
|
|
}
|
|
|
|
// Strategy 2: Analyze all URLs for page numbers
|
|
const pageNumbersFromUrls = [];
|
|
links.forEach(link => {
|
|
if (!link.href || link.href === '#') return;
|
|
|
|
// Check for common pagination URL patterns
|
|
const patterns = [
|
|
/[?&]page=(\d+)/,
|
|
/[?&]p=(\d+)/,
|
|
/[?&]pg=(\d+)/,
|
|
/\/page\/(\d+)/,
|
|
/\/p\/(\d+)/,
|
|
/\/paged\/(\d+)/,
|
|
/\/pages\/(\d+)/
|
|
];
|
|
|
|
for (const pattern of patterns) {
|
|
const match = link.href.match(pattern);
|
|
if (match && match[1]) {
|
|
pageNumbersFromUrls.push(parseInt(match[1]));
|
|
}
|
|
}
|
|
});
|
|
|
|
if (pageNumbersFromUrls.length > 0) {
|
|
return Math.max(...pageNumbersFromUrls);
|
|
}
|
|
|
|
return null;
|
|
}''')
|
|
|
|
if last_page_from_next:
|
|
pagination_info['lastPage'] = last_page_from_next
|
|
|
|
# Check if the pattern is unclear
|
|
has_pagination = True
|
|
url_string_template = str(url_template)
|
|
if url_string_template and url_string_template.startswith("Pattern unclear"):
|
|
has_pagination = False
|
|
url_template = None
|
|
|
|
result = {
|
|
"status": "success",
|
|
"url": decoded_url,
|
|
"hasPagination": has_pagination,
|
|
"urlTemplate": url_template,
|
|
"lastPage": pagination_info.get('lastPage'),
|
|
"originalUrl": original_url,
|
|
"nextPageUrl": next_page_url
|
|
}
|
|
|
|
return result
|
|
|
|
except Exception as e:
|
|
print(f"Error during pagination detection: {e}")
|
|
return {
|
|
"status": "error",
|
|
"url": decoded_url,
|
|
"error": str(e),
|
|
"hasPagination": False,
|
|
"urlTemplate": None,
|
|
"lastPage": None
|
|
}
|
|
|
|
# Perform the operation
|
|
return await safe_browser_operation(decoded_url, pagination_operation) |