from app.utils.browser_utils import safe_browser_operation async def visit_url_service(decoded_url): """Service function to visit a URL and get its content""" print(f"Visiting URL: {decoded_url}") # Define the operation to perform with the browser async def visit_operation(page): try: response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) if not response: print(f"Warning: No response object returned for {decoded_url}") # Get page content content = await page.content() return {"status": "success", "content": content} except Exception as e: print(f"Error during page navigation: {e}") # Try to get content anyway try: content = await page.content() return {"status": "partial", "content": content, "error": str(e)} except: raise Exception(f"Failed to get page content: {str(e)}") # Perform the operation return await safe_browser_operation(decoded_url, visit_operation) async def extract_seo_service(decoded_url): """Service function to extract SEO information from a website""" print(f"Extracting SEO from: {decoded_url}") # Define the operation to perform with the browser async def seo_operation(page): try: response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) # Extract SEO information seo_data = await page.evaluate('''() => { const data = { title: document.title || '', description: '', canonical: '', h1: [], h2: [], images: 0, links: 0 }; // Get meta description const metaDescription = document.querySelector('meta[name="description"]'); if (metaDescription) { data.description = metaDescription.getAttribute('content') || ''; } // Get canonical link const canonicalLink = document.querySelector('link[rel="canonical"]'); if (canonicalLink) { data.canonical = canonicalLink.getAttribute('href') || ''; } // Get h1 tags document.querySelectorAll('h1').forEach(h1 => { const text = h1.innerText.trim(); if (text) data.h1.push(text); }); // Get h2 tags document.querySelectorAll('h2').forEach(h2 => { const text = h2.innerText.trim(); if (text) data.h2.push(text); }); // Count images data.images = document.querySelectorAll('img').length; // Count links data.links = document.querySelectorAll('a').length; return data; }''') result = { "status": "success", "url": decoded_url, "seo": seo_data } return result except Exception as e: print(f"Error during SEO extraction: {e}") return {"status": "error", "url": decoded_url, "error": str(e)} # Perform the operation return await safe_browser_operation(decoded_url, seo_operation) async def extract_meta_tags_service(decoded_url): """Service function to extract meta tags from a website""" print(f"Extracting meta tags from: {decoded_url}") # Define the operation to perform with the browser async def meta_operation(page): try: response = await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) # Extract all meta tags meta_tags = await page.evaluate('''() => { const metas = Array.from(document.querySelectorAll('meta')); return metas.map(meta => { const attributes = {}; Array.from(meta.attributes).forEach(attr => { attributes[attr.name] = attr.value; }); return attributes; }); }''') # Extract Open Graph tags og_tags = await page.evaluate('''() => { const ogTags = {}; document.querySelectorAll('meta[property^="og:"]').forEach(tag => { const property = tag.getAttribute('property'); ogTags[property] = tag.getAttribute('content'); }); return ogTags; }''') # Extract Twitter card tags twitter_tags = await page.evaluate('''() => { const twitterTags = {}; document.querySelectorAll('meta[name^="twitter:"]').forEach(tag => { const name = tag.getAttribute('name'); twitterTags[name] = tag.getAttribute('content'); }); return twitterTags; }''') result = { "status": "success", "url": decoded_url, "meta_tags": meta_tags, "open_graph": og_tags, "twitter_card": twitter_tags, "title": await page.title() } return result except Exception as e: print(f"Error during meta tag extraction: {e}") return {"status": "error", "url": decoded_url, "error": str(e)} # Perform the operation return await safe_browser_operation(decoded_url, meta_operation) async def detect_pagination_service(decoded_url): """Service function to detect pagination on a website""" print(f"Detecting pagination on: {decoded_url}") # Define the operation to perform with the browser async def pagination_operation(page): try: # Navigate to the URL await page.goto(decoded_url, waitUntil='networkidle2', timeout=30000) original_url = page.url print(f"Successfully loaded page: {original_url}") # Analyze the page for pagination information pagination_info = await page.evaluate('''() => { // Find the last page number if available const findLastPageNumber = () => { // Get all links on the page const links = Array.from(document.querySelectorAll('a')); // Strategy 1: Find numeric links (page numbers) const numericLinks = links.filter(link => { const text = link.innerText.trim(); return /^[0-9]+$/.test(text) && link.href && link.href !== '#'; }); if (numericLinks.length > 0) { const numericValues = numericLinks.map(link => parseInt(link.innerText.trim())); return Math.max(...numericValues); } // Strategy 2: Look for "last page" link const lastLinks = links.filter(link => { const text = link.innerText.trim().toLowerCase(); const classes = (link.className || '').toLowerCase(); const ariaLabel = (link.getAttribute('aria-label') || '').toLowerCase(); return (text === 'last' || classes.includes('last') || ariaLabel.includes('last') || link.getAttribute('rel') === 'last'); }); if (lastLinks.length > 0) { const lastLink = lastLinks[0]; const href = lastLink.href; // Common patterns: page=X, /page/X, etc. const pagePatterns = [ /[?&]page=(\d+)/, /[?&]p=(\d+)/, /[?&]pg=(\d+)/, /\/page\/(\d+)/, /\/p\/(\d+)/, /\/paged\/(\d+)/ ]; for (const pattern of pagePatterns) { const match = href.match(pattern); if (match && match[1]) { return parseInt(match[1]); } } } // Strategy 3: Analyze all URLs for page numbers const pageNumbersFromUrls = []; links.forEach(link => { if (!link.href || link.href === '#') return; // Check for common pagination URL patterns const patterns = [ /[?&]page=(\d+)/, /[?&]p=(\d+)/, /[?&]pg=(\d+)/, /\/page\/(\d+)/, /\/p\/(\d+)/, /\/paged\/(\d+)/, /\/pages\/(\d+)/ ]; for (const pattern of patterns) { const match = link.href.match(pattern); if (match && match[1]) { pageNumbersFromUrls.push(parseInt(match[1])); } } }); if (pageNumbersFromUrls.length > 0) { return Math.max(...pageNumbersFromUrls); } return null; }; // Find a pagination link to click const findPaginationLink = () => { const links = Array.from(document.querySelectorAll('a')); // Try to find a page "2" link first (most reliable) const page2Link = links.find(link => { const text = link.innerText.trim(); return text === '2' && link.href && link.href !== '#'; }); if (page2Link) { return { element: page2Link, href: page2Link.href, type: 'numeric' }; } // Try common "next page" selectors const nextSelectors = [ 'a.next', 'a.page-next', 'a[rel="next"]', 'a[aria-label="Next page"]', 'a[aria-label="next"]' ]; for (const selector of nextSelectors) { const element = document.querySelector(selector); if (element && element.href && element.href !== '#') { return { element, href: element.href, type: 'next' }; } } // Look for any link that might be pagination const paginationLinks = links.filter(link => { if (!link.href || link.href === '#') return false; const text = link.innerText.trim(); const href = link.href; // Check for numeric text or next/prev indicators const isNumeric = /^[0-9]+$/.test(text) && text !== '1'; const isNextPrev = /next|prev|previous|older|newer/i.test(text) || /[»«‹›<>]/.test(text); // Check for page parameter in URL const hasPageParam = /[?&]page=|[?&]p=|[?&]pg=|\/page\/|\/p\//.test(href); return (isNumeric || isNextPrev || hasPageParam); }); if (paginationLinks.length > 0) { const link = paginationLinks[0]; return { element: link, href: link.href, type: 'other' }; } return null; }; const lastPage = findLastPageNumber(); const paginationLink = findPaginationLink(); if (paginationLink) { // Click the link paginationLink.element.click(); return { clicked: true, href: paginationLink.href, type: paginationLink.type, lastPage }; } return { clicked: false, lastPage }; }''') # If no pagination was found or clicked if not pagination_info.get('clicked', False): return { "status": "success", "url": decoded_url, "hasPagination": False, "urlTemplate": None, "lastPage": pagination_info.get('lastPage') } # Wait for navigation to complete after the click try: await page.waitForNavigation({'timeout': 10000, 'waitUntil': 'networkidle2'}) except Exception as e: print(f"Navigation timeout: {e}") # Get the new URL after clicking next_page_url = page.url # If URL didn't change, pagination might be handled by AJAX if next_page_url == original_url: return { "status": "success", "url": decoded_url, "hasPagination": True, "urlTemplate": "AJAX pagination (URL doesn't change)", "lastPage": pagination_info.get('lastPage') } print(f"Navigation successful: {original_url} -> {next_page_url}") # Analyze the URL structure to determine pagination pattern url_template = await page.evaluate('''(originalUrl, nextPageUrl) => { // Helper function to parse URL query parameters const parseQueryParams = (url) => { const params = {}; if (url.includes('?')) { const queryString = url.split('?')[1].split('#')[0]; queryString.split('&').forEach(param => { if (param.includes('=')) { const [key, value] = param.split('=', 2); params[key] = value; } }); } return params; }; // Check for query parameter based pagination if (nextPageUrl.includes('?')) { const originalParams = parseQueryParams(originalUrl); const nextParams = parseQueryParams(nextPageUrl); // Find parameters that changed or were added let paginationParam = null; // First check for common pagination parameter names const commonPaginationParams = ['page', 'p', 'pg', 'paged', 'current_page', 'pagenum', 'pageNumber']; for (const key of commonPaginationParams) { if (key in nextParams && (!(key in originalParams) || originalParams[key] !== nextParams[key])) { if (/^\d+$/.test(nextParams[key]) && parseInt(nextParams[key]) > 1) { paginationParam = key; break; } } } // If no common parameter found, check all parameters if (!paginationParam) { for (const [key, value] of Object.entries(nextParams)) { // Check if parameter is new or changed if (!(key in originalParams) || originalParams[key] !== value) { // Check if the value is numeric and could be a page number if (/^\d+$/.test(value) && parseInt(value) > 1) { paginationParam = key; break; } } } } // If we found a pagination parameter if (paginationParam) { const baseUrl = nextPageUrl.split('?')[0]; // Reconstruct the URL template with all parameters const queryParts = []; for (const [key, value] of Object.entries(nextParams)) { if (key === paginationParam) { queryParts.push(`${key}={PAGE_NUMBER}`); } else { queryParts.push(`${key}=${value}`); } } return `${baseUrl}?${queryParts.join('&')}`; } } // Check for path-based pagination const pathPatterns = ['/page/', '/p/', '/paged/', '/pages/']; for (const pattern of pathPatterns) { if (nextPageUrl.includes(pattern)) { const parts = nextPageUrl.split(pattern); let template = `${parts[0]}${pattern}{PAGE_NUMBER}`; // Add any suffix after the page number if (parts.length > 1 && parts[1].includes('/')) { const suffix = parts[1].split('/', 1)[1]; if (suffix) { template += `/${suffix}`; } } return template; } } // If we couldn't determine the pattern, try to make an educated guess // For the specific case where a parameter like current_page=2 is added const originalUrlObj = new URL(originalUrl); const nextUrlObj = new URL(nextPageUrl); // Check if the paths are the same but query params differ if (originalUrlObj.pathname === nextUrlObj.pathname) { const originalParams = parseQueryParams(originalUrl); const nextParams = parseQueryParams(nextPageUrl); // Find parameters that exist in next but not in original const newParams = Object.keys(nextParams).filter(key => !(key in originalParams)); // If there's exactly one new parameter and it has a numeric value if (newParams.length === 1 && /^\d+$/.test(nextParams[newParams[0]])) { const paginationParam = newParams[0]; const baseUrl = nextPageUrl.split('?')[0]; // Reconstruct the URL template const queryParts = []; for (const [key, value] of Object.entries(nextParams)) { if (key === paginationParam) { queryParts.push(`${key}={PAGE_NUMBER}`); } else { queryParts.push(`${key}=${value}`); } } return `${baseUrl}?${queryParts.join('&')}`; } } // If we still couldn't determine the pattern, return both URLs as examples return `Pattern unclear. Example: ${originalUrl} → ${nextPageUrl}`; }''', original_url, next_page_url) # Try to extract last page number from the next page if we didn't find it on the first page if not pagination_info.get('lastPage'): last_page_from_next = await page.evaluate('''() => { // Get all links on the page const links = Array.from(document.querySelectorAll('a')); // Strategy 1: Find numeric links (page numbers) const numericLinks = links.filter(link => { const text = link.innerText.trim(); return /^[0-9]+$/.test(text) && link.href && link.href !== '#'; }); if (numericLinks.length > 0) { const numericValues = numericLinks.map(link => parseInt(link.innerText.trim())); return Math.max(...numericValues); } // Strategy 2: Analyze all URLs for page numbers const pageNumbersFromUrls = []; links.forEach(link => { if (!link.href || link.href === '#') return; // Check for common pagination URL patterns const patterns = [ /[?&]page=(\d+)/, /[?&]p=(\d+)/, /[?&]pg=(\d+)/, /\/page\/(\d+)/, /\/p\/(\d+)/, /\/paged\/(\d+)/, /\/pages\/(\d+)/ ]; for (const pattern of patterns) { const match = link.href.match(pattern); if (match && match[1]) { pageNumbersFromUrls.push(parseInt(match[1])); } } }); if (pageNumbersFromUrls.length > 0) { return Math.max(...pageNumbersFromUrls); } return null; }''') if last_page_from_next: pagination_info['lastPage'] = last_page_from_next # Check if the pattern is unclear has_pagination = True url_string_template = str(url_template) if url_string_template and url_string_template.startswith("Pattern unclear"): has_pagination = False url_template = None result = { "status": "success", "url": decoded_url, "hasPagination": has_pagination, "urlTemplate": url_template, "lastPage": pagination_info.get('lastPage'), "originalUrl": original_url, "nextPageUrl": next_page_url } return result except Exception as e: print(f"Error during pagination detection: {e}") return { "status": "error", "url": decoded_url, "error": str(e), "hasPagination": False, "urlTemplate": None, "lastPage": None } # Perform the operation return await safe_browser_operation(decoded_url, pagination_operation)