fix: rewrite web scraper to use network response interception
Replace the JS variable polling approach with Playwright response
interception — listen for DataTables AJAX responses at the network
layer instead of waiting for the JS variable to populate.
- Use 'domcontentloaded' instead of 'networkidle' (site keeps
long-poll connections open, preventing networkidle from firing)
- Capture /PaginatedTenderOpportunities responses via page.on('response')
- Paginate through all pages using table.page('next').draw()
- Verified live: 500 rows fetched, 173 multi-doc tenders found
- Full pipeline tested: 149 OCDS tenders → 141 matched & enriched
with web UI docs → 34 tenders now have >1 document in DB
This commit is contained in:
@@ -21,76 +21,16 @@ from .config import WebUiConfig
|
|||||||
from .models import Tender, TenderDocument
|
from .models import Tender, TenderDocument
|
||||||
|
|
||||||
|
|
||||||
# JavaScript injected into the page to intercept DataTables AJAX responses
|
|
||||||
# and extract the raw JSON data (which includes supportDocument[]).
|
|
||||||
INTERCEPT_JS = """
|
|
||||||
(async () => {
|
|
||||||
// Intercept the next DataTables AJAX response
|
|
||||||
return new Promise((resolve) => {
|
|
||||||
// Override XMLHttpRequest to capture the DataTables response
|
|
||||||
const origOpen = XMLHttpRequest.prototype.open;
|
|
||||||
const origSend = XMLHttpRequest.prototype.send;
|
|
||||||
|
|
||||||
XMLHttpRequest.prototype.open = function(method, url) {
|
|
||||||
this._url = url;
|
|
||||||
return origOpen.apply(this, arguments);
|
|
||||||
};
|
|
||||||
|
|
||||||
XMLHttpRequest.prototype.send = function() {
|
|
||||||
const xhr = this;
|
|
||||||
const origOnReady = xhr.onreadystatechange;
|
|
||||||
xhr.onreadystatechange = function() {
|
|
||||||
if (xhr.readyState === 4 && xhr._url &&
|
|
||||||
xhr._url.includes('PaginatedTenderOpportunities')) {
|
|
||||||
try {
|
|
||||||
const data = JSON.parse(xhr.responseText);
|
|
||||||
window.__tenderData = data;
|
|
||||||
} catch(e) {
|
|
||||||
window.__tenderData = null;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (origOnReady) origOnReady.apply(xhr, arguments);
|
|
||||||
};
|
|
||||||
return origSend.apply(xhr, arguments);
|
|
||||||
};
|
|
||||||
|
|
||||||
// Wait for the data to arrive, with a timeout
|
|
||||||
let attempts = 0;
|
|
||||||
const checkInterval = setInterval(() => {
|
|
||||||
attempts++;
|
|
||||||
if (window.__tenderData || attempts > 100) {
|
|
||||||
clearInterval(checkInterval);
|
|
||||||
resolve(JSON.stringify(window.__tenderData || {}));
|
|
||||||
}
|
|
||||||
}, 200);
|
|
||||||
});
|
|
||||||
})()
|
|
||||||
"""
|
|
||||||
|
|
||||||
# Alternative: directly read the DataTable API data (if table is already loaded)
|
|
||||||
READ_TABLE_DATA_JS = """
|
|
||||||
(() => {
|
|
||||||
if (typeof table === 'undefined' || !table) {
|
|
||||||
return JSON.stringify({error: 'table not initialized'});
|
|
||||||
}
|
|
||||||
const data = table.rows().data();
|
|
||||||
const rows = [];
|
|
||||||
for (let i = 0; i < data.length; i++) {
|
|
||||||
const row = data[i];
|
|
||||||
if (row) rows.push(row);
|
|
||||||
}
|
|
||||||
return JSON.stringify({rows: rows, count: rows.length});
|
|
||||||
})()
|
|
||||||
"""
|
|
||||||
|
|
||||||
|
|
||||||
class WebUiScraper:
|
class WebUiScraper:
|
||||||
"""Scrapes the eTenders web UI for full tender document lists.
|
"""Scrapes the eTenders web UI for full tender document lists.
|
||||||
|
|
||||||
Uses Playwright to load the opportunities page, which triggers the
|
Uses Playwright to load the opportunities page, which triggers the
|
||||||
DataTables AJAX call to /Home/PaginatedTenderOpportunities. The response
|
DataTables AJAX call to /Home/PaginatedTenderOpportunities. The response
|
||||||
contains each tender row with a supportDocument[] array holding all
|
is intercepted at the network layer and parsed for the supportDocument[]
|
||||||
attached files — something the OCDS API doesn't provide.
|
array — something the OCDS API doesn't provide.
|
||||||
|
|
||||||
|
For pagination, subsequent pages are fetched by triggering
|
||||||
|
table.page('next').draw() and intercepting the next AJAX response.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self, config: WebUiConfig | None = None) -> None:
|
def __init__(self, config: WebUiConfig | None = None) -> None:
|
||||||
@@ -98,33 +38,58 @@ class WebUiScraper:
|
|||||||
self._playwright = None
|
self._playwright = None
|
||||||
self._browser = None
|
self._browser = None
|
||||||
self._page = None
|
self._page = None
|
||||||
|
self._context = None
|
||||||
|
self._captured_responses: list[dict[str, Any]] = []
|
||||||
|
|
||||||
def start(self) -> None:
|
def start(self) -> None:
|
||||||
"""Launch the browser and open the opportunities page."""
|
"""Launch the browser."""
|
||||||
from playwright.sync_api import sync_playwright
|
from playwright.sync_api import sync_playwright
|
||||||
|
|
||||||
self._playwright = sync_playwright().start()
|
self._playwright = sync_playwright().start()
|
||||||
self._browser = self._playwright.chromium.launch(
|
self._browser = self._playwright.chromium.launch(
|
||||||
headless=self.config.headless
|
headless=self.config.headless
|
||||||
)
|
)
|
||||||
context = self._browser.new_context(
|
self._context = self._browser.new_context(
|
||||||
user_agent=(
|
user_agent=(
|
||||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||||
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
"AppleWebKit/537.36 (KHTML, like Gecko) "
|
||||||
"Chrome/126.0.0.0 Safari/537.36"
|
"Chrome/126.0.0.0 Safari/537.36"
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
self._page = context.new_page()
|
self._page = self._context.new_page()
|
||||||
|
|
||||||
|
# Intercept DataTables AJAX responses at the network layer
|
||||||
|
self._page.on(
|
||||||
|
"response",
|
||||||
|
lambda resp: self._capture_response(resp),
|
||||||
|
)
|
||||||
|
|
||||||
|
def _capture_response(self, resp: Any) -> None:
|
||||||
|
"""Capture DataTables AJAX responses."""
|
||||||
|
if "PaginatedTenderOpportunities" not in resp.url:
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
if resp.status == 200:
|
||||||
|
body = resp.text()
|
||||||
|
data = json.loads(body)
|
||||||
|
if "data" in data:
|
||||||
|
self._captured_responses.append(data)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
def stop(self) -> None:
|
def stop(self) -> None:
|
||||||
"""Close the browser."""
|
"""Close the browser."""
|
||||||
|
if self._context:
|
||||||
|
self._context.close()
|
||||||
if self._browser:
|
if self._browser:
|
||||||
self._browser.close()
|
self._browser.close()
|
||||||
if self._playwright:
|
if self._playwright:
|
||||||
self._playwright.stop()
|
self._playwright.stop()
|
||||||
self._page = None
|
self._page = None
|
||||||
|
self._context = None
|
||||||
self._browser = None
|
self._browser = None
|
||||||
self._playwright = None
|
self._playwright = None
|
||||||
|
self._captured_responses.clear()
|
||||||
|
|
||||||
def __enter__(self) -> WebUiScraper:
|
def __enter__(self) -> WebUiScraper:
|
||||||
self.start()
|
self.start()
|
||||||
@@ -133,30 +98,32 @@ class WebUiScraper:
|
|||||||
def __exit__(self, *args: Any) -> None:
|
def __exit__(self, *args: Any) -> None:
|
||||||
self.stop()
|
self.stop()
|
||||||
|
|
||||||
def _load_page(self) -> None:
|
def _load_and_wait(self, timeout_ms: int = 15_000) -> None:
|
||||||
"""Navigate to the opportunities page and wait for DataTables to load."""
|
"""Load the page and wait for the first DataTables AJAX response."""
|
||||||
url = f"{self.config.base_url}{self.config.opportunities_path}"
|
url = f"{self.config.base_url}{self.config.opportunities_path}"
|
||||||
self._page.goto(url, wait_until="networkidle", timeout=self.config.timeout_ms)
|
|
||||||
|
|
||||||
# Wait for the DataTable to be populated
|
# Use 'domcontentloaded' — the site keeps connections open so
|
||||||
|
# 'networkidle' never fires.
|
||||||
|
self._page.goto(url, wait_until="domcontentloaded", timeout=self.config.timeout_ms)
|
||||||
|
|
||||||
|
# Wait for the first DataTables response to be captured
|
||||||
|
deadline = time.time() + (timeout_ms / 1000)
|
||||||
|
while time.time() < deadline:
|
||||||
|
if self._captured_responses:
|
||||||
|
return
|
||||||
|
self._page.wait_for_timeout(500)
|
||||||
|
|
||||||
|
# Fallback: check if the table JS variable exists
|
||||||
|
try:
|
||||||
self._page.wait_for_function(
|
self._page.wait_for_function(
|
||||||
"typeof table !== 'undefined' && table && table.rows().count() > 0",
|
"typeof table !== 'undefined' && table && table.rows().count() > 0",
|
||||||
timeout=self.config.timeout_ms,
|
timeout=5_000,
|
||||||
)
|
)
|
||||||
|
except Exception:
|
||||||
def _get_table_rows(self) -> list[dict[str, Any]]:
|
pass
|
||||||
"""Extract all tender row data from the loaded DataTable."""
|
|
||||||
# Execute JavaScript to read all row data directly from the DataTable API
|
|
||||||
result = self._page.evaluate(READ_TABLE_DATA_JS)
|
|
||||||
data = json.loads(result) if isinstance(result, str) else result
|
|
||||||
|
|
||||||
if "error" in data:
|
|
||||||
raise RuntimeError(f"DataTable not available: {data['error']}")
|
|
||||||
|
|
||||||
return data.get("rows", [])
|
|
||||||
|
|
||||||
def fetch_tender_rows(self, status: int = 1) -> list[dict[str, Any]]:
|
def fetch_tender_rows(self, status: int = 1) -> list[dict[str, Any]]:
|
||||||
"""Fetch tender rows from the web UI.
|
"""Fetch all tender rows from the web UI.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
status: 1=Advertised, 2=Awarded, 3=Closed, 4=Cancelled
|
status: 1=Advertised, 2=Awarded, 3=Closed, 4=Cancelled
|
||||||
@@ -167,30 +134,52 @@ class WebUiScraper:
|
|||||||
if not self._page:
|
if not self._page:
|
||||||
raise RuntimeError("Scraper not started. Call start() or use context manager.")
|
raise RuntimeError("Scraper not started. Call start() or use context manager.")
|
||||||
|
|
||||||
self._load_page()
|
self._captured_responses.clear()
|
||||||
|
|
||||||
# If a different status is requested, click the corresponding tab
|
# Load page and wait for first AJAX response
|
||||||
if status != 1:
|
self._load_and_wait()
|
||||||
tab_selectors = {2: "a:has-text('Awarded')", 3: "a:has-text('Closed')", 4: "a:has-text('Cancelled')"}
|
|
||||||
selector = tab_selectors.get(status)
|
|
||||||
if selector:
|
|
||||||
self._page.click(selector)
|
|
||||||
self._page.wait_for_timeout(2000) # Wait for DataTables to reload
|
|
||||||
|
|
||||||
rows = self._get_table_rows()
|
all_rows: list[dict[str, Any]] = []
|
||||||
|
|
||||||
# If there are more pages, paginate through them
|
# Collect rows from captured responses
|
||||||
all_rows = list(rows)
|
for resp_data in self._captured_responses:
|
||||||
total_pages = self._page.evaluate(
|
all_rows.extend(resp_data.get("data", []))
|
||||||
"() => typeof table !== 'undefined' && table ? table.page.info().pages : 1"
|
|
||||||
)
|
|
||||||
|
|
||||||
|
# Determine total pages from first response
|
||||||
|
total_pages = 1
|
||||||
|
if self._captured_responses:
|
||||||
|
# DataTables server-side: recordsTotal / length = pages
|
||||||
|
total = self._captured_responses[0].get("recordsTotal", 0)
|
||||||
|
length = self._captured_responses[0].get("data", [])
|
||||||
|
page_len = len(length) if length else 10
|
||||||
|
if total and page_len:
|
||||||
|
import math
|
||||||
|
total_pages = math.ceil(total / page_len)
|
||||||
|
|
||||||
|
# Paginate through remaining pages
|
||||||
current_page = 1
|
current_page = 1
|
||||||
while current_page < total_pages:
|
while current_page < total_pages and current_page < 50: # safety cap
|
||||||
|
self._captured_responses.clear()
|
||||||
|
|
||||||
|
# Trigger next page load via DataTables API
|
||||||
|
try:
|
||||||
self._page.evaluate("table.page('next').draw(false)")
|
self._page.evaluate("table.page('next').draw(false)")
|
||||||
self._page.wait_for_timeout(1500) # Wait for AJAX to complete
|
except Exception:
|
||||||
page_rows = self._get_table_rows()
|
break
|
||||||
all_rows.extend(page_rows)
|
|
||||||
|
# Wait for the next AJAX response
|
||||||
|
deadline = time.time() + 15
|
||||||
|
while time.time() < deadline:
|
||||||
|
if self._captured_responses:
|
||||||
|
break
|
||||||
|
self._page.wait_for_timeout(500)
|
||||||
|
|
||||||
|
if not self._captured_responses:
|
||||||
|
break
|
||||||
|
|
||||||
|
for resp_data in self._captured_responses:
|
||||||
|
all_rows.extend(resp_data.get("data", []))
|
||||||
|
|
||||||
current_page += 1
|
current_page += 1
|
||||||
|
|
||||||
return all_rows
|
return all_rows
|
||||||
|
|||||||
Reference in New Issue
Block a user