This commit is contained in:
@@ -0,0 +1,564 @@
|
||||
from app.utils.browser_utils import safe_browser_operation
|
||||
|
||||
async def visit_url_service(decoded_url):
|
||||
"""Service function to visit a URL and get its content"""
|
||||
print(f"Visiting URL: {decoded_url}")
|
||||
|
||||
# Define the operation to perform with the browser
|
||||
async def visit_operation(page):
|
||||
try:
|
||||
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
||||
if not response:
|
||||
print(f"Warning: No response object returned for {decoded_url}")
|
||||
|
||||
# Get page content
|
||||
content = await page.content()
|
||||
return {"status": "success", "content": content}
|
||||
except Exception as e:
|
||||
print(f"Error during page navigation: {e}")
|
||||
# Try to get content anyway
|
||||
try:
|
||||
content = await page.content()
|
||||
return {"status": "partial", "content": content, "error": str(e)}
|
||||
except:
|
||||
raise Exception(f"Failed to get page content: {str(e)}")
|
||||
|
||||
# Perform the operation
|
||||
return await safe_browser_operation(decoded_url, visit_operation)
|
||||
|
||||
async def extract_seo_service(decoded_url):
|
||||
"""Service function to extract SEO information from a website"""
|
||||
print(f"Extracting SEO from: {decoded_url}")
|
||||
|
||||
# Define the operation to perform with the browser
|
||||
async def seo_operation(page):
|
||||
try:
|
||||
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
||||
|
||||
# Extract SEO information
|
||||
seo_data = await page.evaluate('''() => {
|
||||
const data = {
|
||||
title: document.title || '',
|
||||
description: '',
|
||||
canonical: '',
|
||||
h1: [],
|
||||
h2: [],
|
||||
images: 0,
|
||||
links: 0
|
||||
};
|
||||
|
||||
// Get meta description
|
||||
const metaDescription = document.querySelector('meta[name="description"]');
|
||||
if (metaDescription) {
|
||||
data.description = metaDescription.getAttribute('content') || '';
|
||||
}
|
||||
|
||||
// Get canonical link
|
||||
const canonicalLink = document.querySelector('link[rel="canonical"]');
|
||||
if (canonicalLink) {
|
||||
data.canonical = canonicalLink.getAttribute('href') || '';
|
||||
}
|
||||
|
||||
// Get h1 tags
|
||||
document.querySelectorAll('h1').forEach(h1 => {
|
||||
const text = h1.innerText.trim();
|
||||
if (text) data.h1.push(text);
|
||||
});
|
||||
|
||||
// Get h2 tags
|
||||
document.querySelectorAll('h2').forEach(h2 => {
|
||||
const text = h2.innerText.trim();
|
||||
if (text) data.h2.push(text);
|
||||
});
|
||||
|
||||
// Count images
|
||||
data.images = document.querySelectorAll('img').length;
|
||||
|
||||
// Count links
|
||||
data.links = document.querySelectorAll('a').length;
|
||||
|
||||
return data;
|
||||
}''')
|
||||
|
||||
result = {
|
||||
"status": "success",
|
||||
"url": decoded_url,
|
||||
"seo": seo_data
|
||||
}
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error during SEO extraction: {e}")
|
||||
return {"status": "error", "url": decoded_url, "error": str(e)}
|
||||
|
||||
# Perform the operation
|
||||
return await safe_browser_operation(decoded_url, seo_operation)
|
||||
|
||||
async def extract_meta_tags_service(decoded_url):
|
||||
"""Service function to extract meta tags from a website"""
|
||||
print(f"Extracting meta tags from: {decoded_url}")
|
||||
|
||||
# Define the operation to perform with the browser
|
||||
async def meta_operation(page):
|
||||
try:
|
||||
response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
||||
|
||||
# Extract all meta tags
|
||||
meta_tags = await page.evaluate('''() => {
|
||||
const metas = Array.from(document.querySelectorAll('meta'));
|
||||
return metas.map(meta => {
|
||||
const attributes = {};
|
||||
Array.from(meta.attributes).forEach(attr => {
|
||||
attributes[attr.name] = attr.value;
|
||||
});
|
||||
return attributes;
|
||||
});
|
||||
}''')
|
||||
|
||||
# Extract Open Graph tags
|
||||
og_tags = await page.evaluate('''() => {
|
||||
const ogTags = {};
|
||||
document.querySelectorAll('meta[property^="og:"]').forEach(tag => {
|
||||
const property = tag.getAttribute('property');
|
||||
ogTags[property] = tag.getAttribute('content');
|
||||
});
|
||||
return ogTags;
|
||||
}''')
|
||||
|
||||
# Extract Twitter card tags
|
||||
twitter_tags = await page.evaluate('''() => {
|
||||
const twitterTags = {};
|
||||
document.querySelectorAll('meta[name^="twitter:"]').forEach(tag => {
|
||||
const name = tag.getAttribute('name');
|
||||
twitterTags[name] = tag.getAttribute('content');
|
||||
});
|
||||
return twitterTags;
|
||||
}''')
|
||||
|
||||
result = {
|
||||
"status": "success",
|
||||
"url": decoded_url,
|
||||
"meta_tags": meta_tags,
|
||||
"open_graph": og_tags,
|
||||
"twitter_card": twitter_tags,
|
||||
"title": await page.title()
|
||||
}
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error during meta tag extraction: {e}")
|
||||
return {"status": "error", "url": decoded_url, "error": str(e)}
|
||||
|
||||
# Perform the operation
|
||||
return await safe_browser_operation(decoded_url, meta_operation)
|
||||
|
||||
async def detect_pagination_service(decoded_url):
|
||||
"""Service function to detect pagination on a website"""
|
||||
print(f"Detecting pagination on: {decoded_url}")
|
||||
|
||||
# Define the operation to perform with the browser
|
||||
async def pagination_operation(page):
|
||||
try:
|
||||
# Navigate to the URL
|
||||
await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000)
|
||||
original_url = page.url
|
||||
|
||||
print(f"Successfully loaded page: {original_url}")
|
||||
|
||||
# Analyze the page for pagination information
|
||||
pagination_info = await page.evaluate('''() => {
|
||||
// Find the last page number if available
|
||||
const findLastPageNumber = () => {
|
||||
// Get all links on the page
|
||||
const links = Array.from(document.querySelectorAll('a'));
|
||||
|
||||
// Strategy 1: Find numeric links (page numbers)
|
||||
const numericLinks = links.filter(link => {
|
||||
const text = link.innerText.trim();
|
||||
return /^[0-9]+$/.test(text) && link.href && link.href !== '#';
|
||||
});
|
||||
|
||||
if (numericLinks.length > 0) {
|
||||
const numericValues = numericLinks.map(link => parseInt(link.innerText.trim()));
|
||||
return Math.max(...numericValues);
|
||||
}
|
||||
|
||||
// Strategy 2: Look for "last page" link
|
||||
const lastLinks = links.filter(link => {
|
||||
const text = link.innerText.trim().toLowerCase();
|
||||
const classes = (link.className || '').toLowerCase();
|
||||
const ariaLabel = (link.getAttribute('aria-label') || '').toLowerCase();
|
||||
|
||||
return (text === 'last' ||
|
||||
classes.includes('last') ||
|
||||
ariaLabel.includes('last') ||
|
||||
link.getAttribute('rel') === 'last');
|
||||
});
|
||||
|
||||
if (lastLinks.length > 0) {
|
||||
const lastLink = lastLinks[0];
|
||||
const href = lastLink.href;
|
||||
|
||||
// Common patterns: page=X, /page/X, etc.
|
||||
const pagePatterns = [
|
||||
/[?&]page=(\d+)/,
|
||||
/[?&]p=(\d+)/,
|
||||
/[?&]pg=(\d+)/,
|
||||
/\/page\/(\d+)/,
|
||||
/\/p\/(\d+)/,
|
||||
/\/paged\/(\d+)/
|
||||
];
|
||||
|
||||
for (const pattern of pagePatterns) {
|
||||
const match = href.match(pattern);
|
||||
if (match && match[1]) {
|
||||
return parseInt(match[1]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Strategy 3: Analyze all URLs for page numbers
|
||||
const pageNumbersFromUrls = [];
|
||||
links.forEach(link => {
|
||||
if (!link.href || link.href === '#') return;
|
||||
|
||||
// Check for common pagination URL patterns
|
||||
const patterns = [
|
||||
/[?&]page=(\d+)/,
|
||||
/[?&]p=(\d+)/,
|
||||
/[?&]pg=(\d+)/,
|
||||
/\/page\/(\d+)/,
|
||||
/\/p\/(\d+)/,
|
||||
/\/paged\/(\d+)/,
|
||||
/\/pages\/(\d+)/
|
||||
];
|
||||
|
||||
for (const pattern of patterns) {
|
||||
const match = link.href.match(pattern);
|
||||
if (match && match[1]) {
|
||||
pageNumbersFromUrls.push(parseInt(match[1]));
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
if (pageNumbersFromUrls.length > 0) {
|
||||
return Math.max(...pageNumbersFromUrls);
|
||||
}
|
||||
|
||||
return null;
|
||||
};
|
||||
|
||||
// Find a pagination link to click
|
||||
const findPaginationLink = () => {
|
||||
const links = Array.from(document.querySelectorAll('a'));
|
||||
|
||||
// Try to find a page "2" link first (most reliable)
|
||||
const page2Link = links.find(link => {
|
||||
const text = link.innerText.trim();
|
||||
return text === '2' && link.href && link.href !== '#';
|
||||
});
|
||||
|
||||
if (page2Link) {
|
||||
return { element: page2Link, href: page2Link.href, type: 'numeric' };
|
||||
}
|
||||
|
||||
// Try common "next page" selectors
|
||||
const nextSelectors = [
|
||||
'a.next',
|
||||
'a.page-next',
|
||||
'a[rel="next"]',
|
||||
'a[aria-label="Next page"]',
|
||||
'a[aria-label="next"]'
|
||||
];
|
||||
|
||||
for (const selector of nextSelectors) {
|
||||
const element = document.querySelector(selector);
|
||||
if (element && element.href && element.href !== '#') {
|
||||
return { element, href: element.href, type: 'next' };
|
||||
}
|
||||
}
|
||||
|
||||
// Look for any link that might be pagination
|
||||
const paginationLinks = links.filter(link => {
|
||||
if (!link.href || link.href === '#') return false;
|
||||
|
||||
const text = link.innerText.trim();
|
||||
const href = link.href;
|
||||
|
||||
// Check for numeric text or next/prev indicators
|
||||
const isNumeric = /^[0-9]+$/.test(text) && text !== '1';
|
||||
const isNextPrev = /next|prev|previous|older|newer/i.test(text) ||
|
||||
/[»«‹›<>]/.test(text);
|
||||
|
||||
// Check for page parameter in URL
|
||||
const hasPageParam = /[?&]page=|[?&]p=|[?&]pg=|\/page\/|\/p\//.test(href);
|
||||
|
||||
return (isNumeric || isNextPrev || hasPageParam);
|
||||
});
|
||||
|
||||
if (paginationLinks.length > 0) {
|
||||
const link = paginationLinks[0];
|
||||
return { element: link, href: link.href, type: 'other' };
|
||||
}
|
||||
|
||||
return null;
|
||||
};
|
||||
|
||||
const lastPage = findLastPageNumber();
|
||||
const paginationLink = findPaginationLink();
|
||||
|
||||
if (paginationLink) {
|
||||
// Click the link
|
||||
paginationLink.element.click();
|
||||
return {
|
||||
clicked: true,
|
||||
href: paginationLink.href,
|
||||
type: paginationLink.type,
|
||||
lastPage
|
||||
};
|
||||
}
|
||||
|
||||
return { clicked: false, lastPage };
|
||||
}''')
|
||||
|
||||
# If no pagination was found or clicked
|
||||
if not pagination_info.get('clicked', False):
|
||||
return {
|
||||
"status": "success",
|
||||
"url": decoded_url,
|
||||
"hasPagination": False,
|
||||
"urlTemplate": None,
|
||||
"lastPage": pagination_info.get('lastPage')
|
||||
}
|
||||
|
||||
# Wait for navigation to complete after the click
|
||||
try:
|
||||
await page.waitForNavigation({'timeout': 10000, 'waitUntil': 'networkidle2'})
|
||||
except Exception as e:
|
||||
print(f"Navigation timeout: {e}")
|
||||
|
||||
# Get the new URL after clicking
|
||||
next_page_url = page.url
|
||||
|
||||
# If URL didn't change, pagination might be handled by AJAX
|
||||
if next_page_url == original_url:
|
||||
return {
|
||||
"status": "success",
|
||||
"url": decoded_url,
|
||||
"hasPagination": True,
|
||||
"urlTemplate": "AJAX pagination (URL doesn't change)",
|
||||
"lastPage": pagination_info.get('lastPage')
|
||||
}
|
||||
|
||||
print(f"Navigation successful: {original_url} -> {next_page_url}")
|
||||
|
||||
# Analyze the URL structure to determine pagination pattern
|
||||
url_template = await page.evaluate('''(originalUrl, nextPageUrl) => {
|
||||
// Helper function to parse URL query parameters
|
||||
const parseQueryParams = (url) => {
|
||||
const params = {};
|
||||
if (url.includes('?')) {
|
||||
const queryString = url.split('?')[1].split('#')[0];
|
||||
queryString.split('&').forEach(param => {
|
||||
if (param.includes('=')) {
|
||||
const [key, value] = param.split('=', 2);
|
||||
params[key] = value;
|
||||
}
|
||||
});
|
||||
}
|
||||
return params;
|
||||
};
|
||||
|
||||
// Check for query parameter based pagination
|
||||
if (nextPageUrl.includes('?')) {
|
||||
const originalParams = parseQueryParams(originalUrl);
|
||||
const nextParams = parseQueryParams(nextPageUrl);
|
||||
|
||||
// Find parameters that changed or were added
|
||||
let paginationParam = null;
|
||||
|
||||
// First check for common pagination parameter names
|
||||
const commonPaginationParams = ['page', 'p', 'pg', 'paged', 'current_page', 'pagenum', 'pageNumber'];
|
||||
|
||||
for (const key of commonPaginationParams) {
|
||||
if (key in nextParams &&
|
||||
(!(key in originalParams) || originalParams[key] !== nextParams[key])) {
|
||||
if (/^\d+$/.test(nextParams[key]) && parseInt(nextParams[key]) > 1) {
|
||||
paginationParam = key;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// If no common parameter found, check all parameters
|
||||
if (!paginationParam) {
|
||||
for (const [key, value] of Object.entries(nextParams)) {
|
||||
// Check if parameter is new or changed
|
||||
if (!(key in originalParams) || originalParams[key] !== value) {
|
||||
// Check if the value is numeric and could be a page number
|
||||
if (/^\d+$/.test(value) && parseInt(value) > 1) {
|
||||
paginationParam = key;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// If we found a pagination parameter
|
||||
if (paginationParam) {
|
||||
const baseUrl = nextPageUrl.split('?')[0];
|
||||
|
||||
// Reconstruct the URL template with all parameters
|
||||
const queryParts = [];
|
||||
for (const [key, value] of Object.entries(nextParams)) {
|
||||
if (key === paginationParam) {
|
||||
queryParts.push(`${key}={PAGE_NUMBER}`);
|
||||
} else {
|
||||
queryParts.push(`${key}=${value}`);
|
||||
}
|
||||
}
|
||||
|
||||
return `${baseUrl}?${queryParts.join('&')}`;
|
||||
}
|
||||
}
|
||||
|
||||
// Check for path-based pagination
|
||||
const pathPatterns = ['/page/', '/p/', '/paged/', '/pages/'];
|
||||
for (const pattern of pathPatterns) {
|
||||
if (nextPageUrl.includes(pattern)) {
|
||||
const parts = nextPageUrl.split(pattern);
|
||||
let template = `${parts[0]}${pattern}{PAGE_NUMBER}`;
|
||||
|
||||
// Add any suffix after the page number
|
||||
if (parts.length > 1 && parts[1].includes('/')) {
|
||||
const suffix = parts[1].split('/', 1)[1];
|
||||
if (suffix) {
|
||||
template += `/${suffix}`;
|
||||
}
|
||||
}
|
||||
|
||||
return template;
|
||||
}
|
||||
}
|
||||
|
||||
// If we couldn't determine the pattern, try to make an educated guess
|
||||
// For the specific case where a parameter like current_page=2 is added
|
||||
const originalUrlObj = new URL(originalUrl);
|
||||
const nextUrlObj = new URL(nextPageUrl);
|
||||
|
||||
// Check if the paths are the same but query params differ
|
||||
if (originalUrlObj.pathname === nextUrlObj.pathname) {
|
||||
const originalParams = parseQueryParams(originalUrl);
|
||||
const nextParams = parseQueryParams(nextPageUrl);
|
||||
|
||||
// Find parameters that exist in next but not in original
|
||||
const newParams = Object.keys(nextParams).filter(key => !(key in originalParams));
|
||||
|
||||
// If there's exactly one new parameter and it has a numeric value
|
||||
if (newParams.length === 1 && /^\d+$/.test(nextParams[newParams[0]])) {
|
||||
const paginationParam = newParams[0];
|
||||
const baseUrl = nextPageUrl.split('?')[0];
|
||||
|
||||
// Reconstruct the URL template
|
||||
const queryParts = [];
|
||||
for (const [key, value] of Object.entries(nextParams)) {
|
||||
if (key === paginationParam) {
|
||||
queryParts.push(`${key}={PAGE_NUMBER}`);
|
||||
} else {
|
||||
queryParts.push(`${key}=${value}`);
|
||||
}
|
||||
}
|
||||
|
||||
return `${baseUrl}?${queryParts.join('&')}`;
|
||||
}
|
||||
}
|
||||
|
||||
// If we still couldn't determine the pattern, return both URLs as examples
|
||||
return `Pattern unclear. Example: ${originalUrl} → ${nextPageUrl}`;
|
||||
}''', original_url, next_page_url)
|
||||
|
||||
# Try to extract last page number from the next page if we didn't find it on the first page
|
||||
if not pagination_info.get('lastPage'):
|
||||
last_page_from_next = await page.evaluate('''() => {
|
||||
// Get all links on the page
|
||||
const links = Array.from(document.querySelectorAll('a'));
|
||||
|
||||
// Strategy 1: Find numeric links (page numbers)
|
||||
const numericLinks = links.filter(link => {
|
||||
const text = link.innerText.trim();
|
||||
return /^[0-9]+$/.test(text) && link.href && link.href !== '#';
|
||||
});
|
||||
|
||||
if (numericLinks.length > 0) {
|
||||
const numericValues = numericLinks.map(link => parseInt(link.innerText.trim()));
|
||||
return Math.max(...numericValues);
|
||||
}
|
||||
|
||||
// Strategy 2: Analyze all URLs for page numbers
|
||||
const pageNumbersFromUrls = [];
|
||||
links.forEach(link => {
|
||||
if (!link.href || link.href === '#') return;
|
||||
|
||||
// Check for common pagination URL patterns
|
||||
const patterns = [
|
||||
/[?&]page=(\d+)/,
|
||||
/[?&]p=(\d+)/,
|
||||
/[?&]pg=(\d+)/,
|
||||
/\/page\/(\d+)/,
|
||||
/\/p\/(\d+)/,
|
||||
/\/paged\/(\d+)/,
|
||||
/\/pages\/(\d+)/
|
||||
];
|
||||
|
||||
for (const pattern of patterns) {
|
||||
const match = link.href.match(pattern);
|
||||
if (match && match[1]) {
|
||||
pageNumbersFromUrls.push(parseInt(match[1]));
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
if (pageNumbersFromUrls.length > 0) {
|
||||
return Math.max(...pageNumbersFromUrls);
|
||||
}
|
||||
|
||||
return null;
|
||||
}''')
|
||||
|
||||
if last_page_from_next:
|
||||
pagination_info['lastPage'] = last_page_from_next
|
||||
|
||||
# Check if the pattern is unclear
|
||||
has_pagination = True
|
||||
url_string_template = str(url_template)
|
||||
if url_string_template and url_string_template.startswith("Pattern unclear"):
|
||||
has_pagination = False
|
||||
url_template = None
|
||||
|
||||
result = {
|
||||
"status": "success",
|
||||
"url": decoded_url,
|
||||
"hasPagination": has_pagination,
|
||||
"urlTemplate": url_template,
|
||||
"lastPage": pagination_info.get('lastPage'),
|
||||
"originalUrl": original_url,
|
||||
"nextPageUrl": next_page_url
|
||||
}
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error during pagination detection: {e}")
|
||||
return {
|
||||
"status": "error",
|
||||
"url": decoded_url,
|
||||
"error": str(e),
|
||||
"hasPagination": False,
|
||||
"urlTemplate": None,
|
||||
"lastPage": None
|
||||
}
|
||||
|
||||
# Perform the operation
|
||||
return await safe_browser_operation(decoded_url, pagination_operation)
|
||||
Reference in New Issue
Block a user