This commit is contained in:
@@ -146,638 +146,6 @@ async def extract_meta_tags_service(decoded_url):
|
||||
print(f"Error during meta tag extraction: {e}")
|
||||
return {"status": "error", "url": decoded_url, "error": str(e)}
|
||||
|
||||
async def detect_pagination_service(decoded_url):
|
||||
"""Service function to detect pagination on a website"""
|
||||
print(f"Detecting pagination on: {decoded_url}")
|
||||
|
||||
# Define the operation to perform with the browser
|
||||
async def pagination_operation(page):
|
||||
try:
|
||||
# Navigate to the URL
|
||||
try:
|
||||
await page.goto(decoded_url, wait_until='networkidle', timeout=30000)
|
||||
except Exception as e:
|
||||
print(f"Error navigating to URL: {e}")
|
||||
# Try to get the current URL even if navigation failed
|
||||
try:
|
||||
original_url = page.url
|
||||
except:
|
||||
original_url = decoded_url
|
||||
else:
|
||||
original_url = page.url
|
||||
|
||||
# Check for common pagination indicators
|
||||
try:
|
||||
pagination_data = await page.evaluate('''() => {
|
||||
const data = {
|
||||
hasPagination: false,
|
||||
paginationType: null,
|
||||
paginationElements: [],
|
||||
detectedParameter: null,
|
||||
lastPageNumber: null
|
||||
};
|
||||
|
||||
// Look for numbered pagination links (1, 2, 3...)
|
||||
const numberedLinks = Array.from(document.querySelectorAll('a, button, span'))
|
||||
.filter(el => {
|
||||
const text = el.innerText.trim();
|
||||
|
||||
// check for data-page attribute
|
||||
const dataPage = el.getAttribute('data-page');
|
||||
if (dataPage) {
|
||||
return /^[0-9]+$/.test(dataPage);
|
||||
}
|
||||
|
||||
return /^[0-9]+$/.test(text) &&
|
||||
(el.tagName === 'A' || el.onclick ||
|
||||
el.closest('button, [role="button"]'));
|
||||
});
|
||||
|
||||
if(numberedLinks.length <= 1) {
|
||||
return data;
|
||||
}
|
||||
|
||||
// Look for next/prev buttons
|
||||
const nextButtons = Array.from(document.querySelectorAll('a, button, [role="button"]'))
|
||||
.filter(el => {
|
||||
const text = el.innerText.trim().toLowerCase();
|
||||
const ariaLabel = el.getAttribute('aria-label')?.toLowerCase() || '';
|
||||
const hasNextIcon = el.querySelector('i.fa-chevron-right, i.fa-arrow-right, svg[class*="arrow"], svg[class*="next"]');
|
||||
|
||||
return text.includes('next') ||
|
||||
text.includes('›') ||
|
||||
text.includes('»') ||
|
||||
text.includes('→') ||
|
||||
ariaLabel.includes('next') ||
|
||||
hasNextIcon;
|
||||
});
|
||||
|
||||
// Check for pagination containers
|
||||
const paginationContainers = Array.from(document.querySelectorAll(
|
||||
'.pagination, [class*="pagination"], [class*="pager"], nav[aria-label*="pagination"], [role="navigation"]'
|
||||
));
|
||||
|
||||
// Collect all potential pagination elements
|
||||
if (numberedLinks.length > 0) {
|
||||
data.hasPagination = true;
|
||||
data.paginationType = 'numbered';
|
||||
|
||||
// Get href attributes or other identifiers from numbered links
|
||||
data.paginationElements = numberedLinks.slice(0, 5).map(el => {
|
||||
return {
|
||||
text: el.innerText.trim(),
|
||||
dataPage: el.getAttribute('data-page'),
|
||||
href: el.tagName === 'A' ? el.href : null,
|
||||
classes: el.className,
|
||||
id: el.id
|
||||
};
|
||||
});
|
||||
|
||||
// Try to find the last page number
|
||||
const pageNumbers = numberedLinks
|
||||
.map(el => parseInt(el.innerText.replace(/[^\d]/g, '')))
|
||||
.filter(num => !isNaN(num));
|
||||
console.log(pageNumbers);
|
||||
if (pageNumbers.length > 0) {
|
||||
data.lastPageNumber = Math.max(...pageNumbers);
|
||||
}
|
||||
|
||||
// Also look for a "last page" element that might have text like "Last" or "»"
|
||||
const lastPageElement = Array.from(document.querySelectorAll('a, button'))
|
||||
.find(el => {
|
||||
const text = el.innerText.trim().toLowerCase();
|
||||
const ariaLabel = el.getAttribute('aria-label')?.toLowerCase() || '';
|
||||
return text.includes('last') ||
|
||||
text === '»' ||
|
||||
ariaLabel.includes('last page');
|
||||
});
|
||||
|
||||
if (lastPageElement && lastPageElement.href) {
|
||||
// Try to extract page number from the URL
|
||||
try {
|
||||
const url = new URL(lastPageElement.href);
|
||||
// Check common pagination parameters
|
||||
['page', 'p', 'pg'].forEach(param => {
|
||||
if (url.searchParams.has(param)) {
|
||||
const value = parseInt(url.searchParams.get(param));
|
||||
if (!isNaN(value) && (data.lastPageNumber === null || value > data.lastPageNumber)) {
|
||||
console.log("Last page number: " + value);
|
||||
data.lastPageNumber = value;
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
// Check for path-based pagination (like /page/10)
|
||||
const pathMatch = url.pathname.match(/\/(?:page|p)\/(\d+)/i);
|
||||
if (pathMatch && pathMatch[1]) {
|
||||
const value = parseInt(pathMatch[1]);
|
||||
if (!isNaN(value) && (data.lastPageNumber === null || value > data.lastPageNumber)) {
|
||||
console.log("Last page number: " + value);
|
||||
data.lastPageNumber = value;
|
||||
}
|
||||
}
|
||||
} catch (e) {
|
||||
console.error("Error parsing last page URL:", e);
|
||||
}
|
||||
}
|
||||
} else if (nextButtons.length > 0) {
|
||||
data.hasPagination = true;
|
||||
data.paginationType = 'next-prev';
|
||||
|
||||
// Get information about next buttons
|
||||
data.paginationElements = nextButtons.slice(0, 3).map(el => {
|
||||
return {
|
||||
text: el.innerText.trim(),
|
||||
href: el.tagName === 'A' ? el.href : null,
|
||||
classes: el.className,
|
||||
id: el.id
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
// Check for URL parameters that might indicate pagination
|
||||
const currentUrl = window.location.href;
|
||||
const urlParams = new URL(currentUrl).searchParams;
|
||||
|
||||
// Common pagination parameters
|
||||
const paginationParams = ['page', 'p', 'pg', 'offset', 'o', 'from', 'start', 'limit'];
|
||||
|
||||
for (const param of paginationParams) {
|
||||
if (urlParams.has(param)) {
|
||||
data.detectedParameter = {
|
||||
name: param,
|
||||
value: urlParams.get(param)
|
||||
};
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return data;
|
||||
}''')
|
||||
except Exception as e:
|
||||
print(f"Error during JavaScript evaluation for pagination detection: {e}")
|
||||
# Return a safe default if JavaScript evaluation fails
|
||||
pagination_data = {
|
||||
'hasPagination': False,
|
||||
'paginationType': None,
|
||||
'paginationElements': [],
|
||||
'detectedParameter': None,
|
||||
'lastPageNumber': None
|
||||
}
|
||||
|
||||
print(f"Pagination data: {pagination_data}")
|
||||
# If pagination is detected, try to navigate to the next page by clicking
|
||||
next_page_url = None
|
||||
pagination_parameter = None
|
||||
url_template = None
|
||||
step_size = None
|
||||
original_parsed = None # Initialize the variable
|
||||
|
||||
if pagination_data['hasPagination']:
|
||||
print("Pagination detected, attempting to click on a pagination element")
|
||||
|
||||
# Capture the original URL before any navigation
|
||||
original_url_before_navigation = page.url
|
||||
|
||||
# Always try to click on a pagination element, regardless of type
|
||||
clicked = False
|
||||
|
||||
# First try to click on a numbered link (preferably "2" if we're on page 1)
|
||||
try:
|
||||
clicked = await page.evaluate('''() => {
|
||||
try {
|
||||
// First try to find and click on a "2" link or button
|
||||
const page2Elements = Array.from(document.querySelectorAll('a[href], button, [role="button"]'))
|
||||
.filter(el => {
|
||||
// Check for text content "2"
|
||||
if (el.innerText.trim() === '2') {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Check for href with page=2 or similar (for anchor elements)
|
||||
if (el.tagName === 'A' && el.href) {
|
||||
try {
|
||||
const url = new URL(el.href, window.location.origin);
|
||||
const paginationParams = ['page', 'p', 'pg', 'offset', 'o', 'from', 'start', 'limit',
|
||||
'currentpage', 'pagenum', 'pageNumber', 'paged'];
|
||||
|
||||
for (const param of paginationParams) {
|
||||
if (url.searchParams.has(param) && url.searchParams.get(param) === '2') {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check for path-based pagination like /page/2/ or /vacatures/page/2
|
||||
const pathMatch = url.pathname.match(/\/(page|p)\/2\/?$/i);
|
||||
if (pathMatch) {
|
||||
return true;
|
||||
}
|
||||
|
||||
// Check for path-based pagination where /page/2 is appended to the current path
|
||||
const currentPath = window.location.pathname;
|
||||
const expectedPath = currentPath.replace(/\/$/, '') + '/page/2';
|
||||
if (url.pathname === expectedPath) {
|
||||
return true;
|
||||
}
|
||||
} catch (e) {}
|
||||
}
|
||||
|
||||
// Check for data attributes that might indicate pagination
|
||||
if (el.getAttribute('data-page') === '2' ||
|
||||
el.getAttribute('data-pagenumber') === '2' ||
|
||||
el.getAttribute('data-page-number') === '2') {
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
});
|
||||
|
||||
if (page2Elements.length > 0) {
|
||||
console.log("Clicking on page 2 element");
|
||||
page2Elements[0].click();
|
||||
return true;
|
||||
}
|
||||
|
||||
// If no "2" link found, try any numbered link or button
|
||||
const numberedElements = Array.from(document.querySelectorAll('a[href], button, [role="button"]'))
|
||||
.filter(el => /^\d+$/.test(el.innerText.trim()));
|
||||
|
||||
if (numberedElements.length > 0) {
|
||||
// Sort by number and get the second one (likely page 2)
|
||||
const sorted = numberedElements.sort((a, b) => {
|
||||
return parseInt(a.innerText.trim()) - parseInt(b.innerText.trim());
|
||||
});
|
||||
|
||||
// Get the second element if available (page 2), otherwise the first one
|
||||
const elementToClick = sorted.length > 1 ? sorted[1] : sorted[0];
|
||||
console.log("Clicking on numbered element: " + elementToClick.innerText);
|
||||
elementToClick.click();
|
||||
return true;
|
||||
}
|
||||
|
||||
// If no numbered links, try next button
|
||||
const nextTexts = ['next', '›', '»', '→'];
|
||||
const nextElements = Array.from(document.querySelectorAll('a, button, [role="button"]'))
|
||||
.filter(el => {
|
||||
const text = el.textContent.trim().toLowerCase();
|
||||
const ariaLabel = el.getAttribute('aria-label')?.toLowerCase() || '';
|
||||
return nextTexts.some(t => text.includes(t)) ||
|
||||
ariaLabel.includes('next') ||
|
||||
el.querySelector('i.fa-chevron-right, i.fa-arrow-right, svg[class*="arrow"], svg[class*="next"]');
|
||||
});
|
||||
|
||||
if (nextElements.length > 0) {
|
||||
console.log("Clicking on next button");
|
||||
nextElements[0].click();
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
} catch (error) {
|
||||
console.error("Error during pagination click operation:", error);
|
||||
return false;
|
||||
}
|
||||
}''')
|
||||
|
||||
if clicked:
|
||||
print("Successfully clicked on pagination element")
|
||||
# Wait for navigation to complete
|
||||
await page.wait_for_load_state('networkidle', timeout=10000)
|
||||
await asyncio.sleep(2)
|
||||
next_page_url = page.url
|
||||
else:
|
||||
print("No clickable pagination element found")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error clicking on pagination element: {e}")
|
||||
# Continue with the process even if clicking fails
|
||||
|
||||
print(f"Next page URL: {next_page_url}")
|
||||
|
||||
# If we successfully navigated to the next page, analyze the URL difference
|
||||
if next_page_url and next_page_url != original_url_before_navigation:
|
||||
print('Searching for pagination parameter')
|
||||
# Parse both URLs using Python instead of JavaScript to avoid execution context issues
|
||||
try:
|
||||
from urllib.parse import urlparse, parse_qs
|
||||
|
||||
original_parsed_url = urlparse(original_url_before_navigation)
|
||||
current_parsed_url = urlparse(next_page_url)
|
||||
|
||||
# Check for differences in query parameters
|
||||
param_diff = None
|
||||
original_params = parse_qs(original_parsed_url.query)
|
||||
current_params = parse_qs(current_parsed_url.query)
|
||||
|
||||
# Common pagination parameters to check
|
||||
pagination_params = ['page', 'p', 'pg', 'offset', 'o', 'from', 'start', 'limit', 'currentPage', 'current_page', 'currentpage', 'pagenum', 'pageNumber', 'paged']
|
||||
|
||||
for param in pagination_params:
|
||||
original_value = original_params.get(param, [None])[0]
|
||||
current_value = current_params.get(param, [None])[0]
|
||||
|
||||
if original_value != current_value and current_value is not None:
|
||||
param_diff = {
|
||||
'name': param,
|
||||
'originalValue': original_value,
|
||||
'currentValue': current_value
|
||||
}
|
||||
break
|
||||
|
||||
# Check for path differences
|
||||
path_diff = None
|
||||
original_path = original_parsed_url.path
|
||||
current_path = current_parsed_url.path
|
||||
|
||||
if original_path != current_path:
|
||||
original_segments = [s for s in original_path.split('/') if s]
|
||||
current_segments = [s for s in current_path.split('/') if s]
|
||||
|
||||
# Case 1: Same number of segments - find the one that changed
|
||||
if len(original_segments) == len(current_segments):
|
||||
for i, (orig_seg, curr_seg) in enumerate(zip(original_segments, current_segments)):
|
||||
if orig_seg != curr_seg:
|
||||
# Check if the difference is numeric
|
||||
if orig_seg.isdigit() and curr_seg.isdigit():
|
||||
path_diff = {
|
||||
'type': 'replace',
|
||||
'index': i,
|
||||
'originalValue': orig_seg,
|
||||
'currentValue': curr_seg
|
||||
}
|
||||
break
|
||||
|
||||
# Case 2: Current path has more segments - check for added pagination segments
|
||||
elif len(current_segments) > len(original_segments):
|
||||
# Look for patterns like /page/NUMBER or /p/NUMBER at the end
|
||||
import re
|
||||
page_pattern = re.compile(r'^(page|p)/(\d+)$', re.IGNORECASE)
|
||||
|
||||
# Check the last two segments of the current path
|
||||
if len(current_segments) >= 2:
|
||||
last_two_segments = '/'.join(current_segments[-2:])
|
||||
match = page_pattern.match(last_two_segments)
|
||||
|
||||
if match:
|
||||
path_diff = {
|
||||
'type': 'append',
|
||||
'pageSegment': match.group(1), # 'page' or 'p'
|
||||
'pageNumber': match.group(2), # the actual number
|
||||
'originalSegments': original_segments,
|
||||
'currentSegments': current_segments
|
||||
}
|
||||
|
||||
# If no pattern match, check if the last segment is numeric
|
||||
if not path_diff and current_segments:
|
||||
last_segment = current_segments[-1]
|
||||
if last_segment.isdigit():
|
||||
path_diff = {
|
||||
'type': 'append',
|
||||
'pageSegment': None,
|
||||
'pageNumber': last_segment,
|
||||
'originalSegments': original_segments,
|
||||
'currentSegments': current_segments
|
||||
}
|
||||
|
||||
original_parsed = {
|
||||
'paramDiff': param_diff,
|
||||
'pathDiff': path_diff,
|
||||
'originalUrl': original_url_before_navigation,
|
||||
'currentUrl': next_page_url
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error during URL analysis: {e}")
|
||||
original_parsed = {
|
||||
'paramDiff': None,
|
||||
'pathDiff': None,
|
||||
'originalUrl': original_url_before_navigation,
|
||||
'currentUrl': next_page_url,
|
||||
'error': str(e)
|
||||
}
|
||||
|
||||
# Determine the pagination parameter and create URL template
|
||||
if original_parsed:
|
||||
print(original_parsed)
|
||||
if original_parsed.get('paramDiff'):
|
||||
param_name = original_parsed['paramDiff']['name']
|
||||
pagination_parameter = {
|
||||
'type': 'query',
|
||||
'name': param_name,
|
||||
'value': original_parsed['paramDiff']['currentValue']
|
||||
}
|
||||
|
||||
# --- STEP SIZE DETECTION FOR QUERY PARAM ---
|
||||
orig_val = original_parsed['paramDiff']['originalValue'] if original_parsed['paramDiff']['originalValue'] is not None else 0
|
||||
curr_val = original_parsed['paramDiff']['currentValue']
|
||||
try:
|
||||
if orig_val is not None and curr_val is not None:
|
||||
orig_num = int(orig_val)
|
||||
curr_num = int(curr_val)
|
||||
step_size = abs(curr_num - orig_num)
|
||||
except Exception:
|
||||
step_size = None
|
||||
|
||||
# Create URL template for query parameter
|
||||
try:
|
||||
from urllib.parse import urlparse, urlencode, parse_qs
|
||||
|
||||
parsed_url = urlparse(original_url_before_navigation)
|
||||
params = parse_qs(parsed_url.query)
|
||||
params[param_name] = ['{PAGE_NUMBER}']
|
||||
|
||||
# Reconstruct the URL
|
||||
new_query = urlencode(params, doseq=True)
|
||||
url_template = f"{parsed_url.scheme}://{parsed_url.netloc}{parsed_url.path}"
|
||||
if new_query:
|
||||
url_template += f"?{new_query}"
|
||||
if parsed_url.fragment:
|
||||
url_template += f"#{parsed_url.fragment}"
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error creating URL template for query parameter: {e}")
|
||||
url_template = None
|
||||
|
||||
elif original_parsed.get('pathDiff'):
|
||||
path_diff = original_parsed['pathDiff']
|
||||
|
||||
if path_diff.get('type') == 'replace':
|
||||
# Handle existing path segment replacement
|
||||
path_index = path_diff['index']
|
||||
pagination_parameter = {
|
||||
'type': 'path',
|
||||
'index': path_index,
|
||||
'value': path_diff['currentValue']
|
||||
}
|
||||
|
||||
# --- STEP SIZE DETECTION FOR PATH PARAM ---
|
||||
orig_val = path_diff['originalValue']
|
||||
curr_val = path_diff['currentValue']
|
||||
try:
|
||||
if orig_val is not None and curr_val is not None:
|
||||
orig_num = int(orig_val)
|
||||
curr_num = int(curr_val)
|
||||
step_size = abs(curr_num - orig_num)
|
||||
except Exception:
|
||||
step_size = None
|
||||
|
||||
# Create URL template for path parameter replacement
|
||||
try:
|
||||
from urllib.parse import urlparse
|
||||
|
||||
parsed_url = urlparse(original_url_before_navigation)
|
||||
path_segments = [s for s in parsed_url.path.split('/') if s]
|
||||
path_segments[path_index] = '{PAGE_NUMBER}'
|
||||
|
||||
# Reconstruct the URL
|
||||
new_path = '/' + '/'.join(path_segments)
|
||||
url_template = f"{parsed_url.scheme}://{parsed_url.netloc}{new_path}"
|
||||
if parsed_url.query:
|
||||
url_template += f"?{parsed_url.query}"
|
||||
if parsed_url.fragment:
|
||||
url_template += f"#{parsed_url.fragment}"
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error creating URL template for path parameter: {e}")
|
||||
url_template = None
|
||||
|
||||
elif path_diff.get('type') == 'append':
|
||||
# Handle new pagination segments being appended
|
||||
pagination_parameter = {
|
||||
'type': 'path_append',
|
||||
'pageSegment': path_diff.get('pageSegment'),
|
||||
'pageNumber': path_diff['pageNumber']
|
||||
}
|
||||
|
||||
# --- STEP SIZE DETECTION FOR APPENDED PATH PARAM ---
|
||||
try:
|
||||
curr_val = path_diff['pageNumber']
|
||||
# For appended pagination, assume we started from page 1 (implicit)
|
||||
orig_num = 1
|
||||
curr_num = int(curr_val)
|
||||
step_size = abs(curr_num - orig_num)
|
||||
except Exception:
|
||||
step_size = None
|
||||
|
||||
# Create URL template for appended path parameter
|
||||
try:
|
||||
from urllib.parse import urlparse
|
||||
|
||||
parsed_url = urlparse(original_url_before_navigation)
|
||||
new_path = parsed_url.path
|
||||
|
||||
# Remove trailing slash if present
|
||||
if new_path.endswith('/'):
|
||||
new_path = new_path[:-1]
|
||||
|
||||
# Append the pagination segment
|
||||
if path_diff.get('pageSegment'):
|
||||
new_path += f"/{path_diff['pageSegment']}/{{PAGE_NUMBER}}"
|
||||
else:
|
||||
new_path += "/{PAGE_NUMBER}"
|
||||
|
||||
# Reconstruct the URL
|
||||
url_template = f"{parsed_url.scheme}://{parsed_url.netloc}{new_path}"
|
||||
if parsed_url.query:
|
||||
url_template += f"?{parsed_url.query}"
|
||||
if parsed_url.fragment:
|
||||
url_template += f"#{parsed_url.fragment}"
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error creating URL template for appended path parameter: {e}")
|
||||
url_template = None
|
||||
|
||||
if url_template:
|
||||
# Decode URL-encoded characters in the template
|
||||
import urllib.parse
|
||||
url_template = urllib.parse.unquote(url_template)
|
||||
|
||||
# If we couldn't determine the URL template from navigation, try to infer it
|
||||
if not url_template and pagination_data['detectedParameter']:
|
||||
param_name = pagination_data['detectedParameter']['name']
|
||||
try:
|
||||
from urllib.parse import urlparse, urlencode, parse_qs
|
||||
|
||||
parsed_url = urlparse(original_url_before_navigation)
|
||||
params = parse_qs(parsed_url.query)
|
||||
params[param_name] = ['{PAGE_NUMBER}']
|
||||
|
||||
# Reconstruct the URL
|
||||
new_query = urlencode(params, doseq=True)
|
||||
url_template = f"{parsed_url.scheme}://{parsed_url.netloc}{parsed_url.path}"
|
||||
if new_query:
|
||||
url_template += f"?{new_query}"
|
||||
if parsed_url.fragment:
|
||||
url_template += f"#{parsed_url.fragment}"
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error creating inferred URL template: {e}")
|
||||
url_template = None
|
||||
|
||||
# If still no template and we have pagination elements, try to infer from the current URL structure
|
||||
if not url_template and pagination_data['hasPagination']:
|
||||
try:
|
||||
from urllib.parse import urlparse
|
||||
import re
|
||||
|
||||
parsed_url = urlparse(original_url_before_navigation)
|
||||
path = parsed_url.path
|
||||
|
||||
# Remove trailing slash if present
|
||||
if path.endswith('/'):
|
||||
path = path[:-1]
|
||||
|
||||
# Check if the current URL already has a pagination pattern
|
||||
page_pattern = re.compile(r'/(page|p)/\d+$', re.IGNORECASE)
|
||||
if page_pattern.search(path):
|
||||
# Replace the existing page number with placeholder
|
||||
path = page_pattern.sub(r'/\1/{PAGE_NUMBER}', path)
|
||||
else:
|
||||
# Add pagination pattern
|
||||
path += '/page/{PAGE_NUMBER}'
|
||||
|
||||
# Reconstruct the URL
|
||||
url_template = f"{parsed_url.scheme}://{parsed_url.netloc}{path}"
|
||||
if parsed_url.query:
|
||||
url_template += f"?{parsed_url.query}"
|
||||
if parsed_url.fragment:
|
||||
url_template += f"#{parsed_url.fragment}"
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error creating fallback URL template: {e}")
|
||||
url_template = None
|
||||
|
||||
# Return the pagination detection results with a simplified structure
|
||||
result = {
|
||||
"status": "success",
|
||||
"hasPagination": pagination_data['hasPagination'],
|
||||
"urlTemplate": url_template,
|
||||
"lastPage": pagination_data['lastPageNumber'],
|
||||
"stepSize": step_size if step_size is not None and step_size >= 5 else 1
|
||||
}
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error during pagination detection: {e}")
|
||||
return {
|
||||
"status": "error",
|
||||
"hasPagination": False,
|
||||
"urlTemplate": None,
|
||||
"lastPage": None,
|
||||
"stepSize": None,
|
||||
"error": str(e)
|
||||
}
|
||||
|
||||
# Perform the operation
|
||||
result = await safe_browser_operation(decoded_url, pagination_operation)
|
||||
|
||||
# Check if the result is an error from safe_browser_operation
|
||||
if isinstance(result, dict) and result.get("status") == "error":
|
||||
return result
|
||||
|
||||
return result
|
||||
|
||||
async def capture_outgoing_calls_service(decoded_url):
|
||||
"""Service function to capture outgoing API calls from a website"""
|
||||
print(f"Capturing outgoing calls from: {decoded_url}")
|
||||
|
||||
Reference in New Issue
Block a user