fix: harden price scraping (findings 1, 2, 6)
Finding 1: parsePriceValue mishandled thousands separators — "1.299,00"
and "1,299.00" both parsed to 1.3. Both the meta-tag path and the
text-regex path now funnel through a single normalizeAmount() helper
that picks the last separator as the decimal one, and the text regex
also matches thousands-grouped amounts.
Finding 2: the text fallback read <script>/<style> contents via
$('body').text(); those elements are now stripped from a clone first.
Finding 6: the default fetch now uses a 15s AbortSignal.timeout and a
browser-like User-Agent, so one unresponsive host cannot stall the
whole sequential run. The fetchImpl injection point is unchanged.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_013gwc3aCiyQxmCWQke8RvEz
This commit is contained in:
+44
-6
@@ -4,10 +4,32 @@ function roundMoney(n) {
|
||||
return Math.round(n * 100) / 100;
|
||||
}
|
||||
|
||||
function parsePriceValue(raw) {
|
||||
// Turns a human-written amount into a plain JS-parsable number string.
|
||||
// When both separators occur, the last one is the decimal separator
|
||||
// ("1.299,00" and "1,299.00" both mean 1299.00); a lone comma is decimal
|
||||
// only when exactly two digits follow it, otherwise it groups thousands.
|
||||
function normalizeAmount(raw) {
|
||||
if (raw == null) return null;
|
||||
const cleaned = String(raw).trim().replace(',', '.');
|
||||
const value = parseFloat(cleaned);
|
||||
const match = String(raw).trim().match(/-?\d[\d.,]*/);
|
||||
if (!match) return null;
|
||||
const str = match[0].replace(/[.,]+$/, '');
|
||||
const lastDot = str.lastIndexOf('.');
|
||||
const lastComma = str.lastIndexOf(',');
|
||||
if (lastDot !== -1 && lastComma !== -1) {
|
||||
return lastComma > lastDot
|
||||
? str.replace(/\./g, '').replace(',', '.')
|
||||
: str.replace(/,/g, '');
|
||||
}
|
||||
if (lastComma !== -1) {
|
||||
return /,\d{2}$/.test(str) ? str.replace(',', '.') : str.replace(/,/g, '');
|
||||
}
|
||||
return str;
|
||||
}
|
||||
|
||||
function parsePriceValue(raw) {
|
||||
const normalized = normalizeAmount(raw);
|
||||
if (normalized == null) return null;
|
||||
const value = parseFloat(normalized);
|
||||
return Number.isFinite(value) ? roundMoney(value) : null;
|
||||
}
|
||||
|
||||
@@ -59,9 +81,14 @@ function findPriceInMeta($) {
|
||||
return null;
|
||||
}
|
||||
|
||||
// Thousands-grouped amounts first, then a plain amount, so "€1.299,00"
|
||||
// is not truncated to "€1.29".
|
||||
const TEXT_PRICE_RE = /€\s?(\d{1,3}(?:[.,]\d{3})+[.,]\d{2}|\d+[.,]\d{2})/;
|
||||
|
||||
function findPriceInText($) {
|
||||
const text = $('body').text();
|
||||
const match = text.match(/€\s?(\d+[.,]\d{2})/);
|
||||
const text = $('body').clone().find('script, style, noscript').remove().end()
|
||||
.text();
|
||||
const match = text.match(TEXT_PRICE_RE);
|
||||
return match ? parsePriceValue(match[1]) : null;
|
||||
}
|
||||
|
||||
@@ -76,7 +103,18 @@ function extractPrice(html) {
|
||||
return { price: null, method: null };
|
||||
}
|
||||
|
||||
async function fetchAndExtractPrice(url, { fetchImpl = fetch } = {}) {
|
||||
const FETCH_TIMEOUT_MS = 15000;
|
||||
const USER_AGENT = 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
+ '(KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36';
|
||||
|
||||
function defaultFetch(url) {
|
||||
return fetch(url, {
|
||||
signal: AbortSignal.timeout(FETCH_TIMEOUT_MS),
|
||||
headers: { 'User-Agent': USER_AGENT },
|
||||
});
|
||||
}
|
||||
|
||||
async function fetchAndExtractPrice(url, { fetchImpl = defaultFetch } = {}) {
|
||||
let response;
|
||||
try {
|
||||
response = await fetchImpl(url);
|
||||
|
||||
Reference in New Issue
Block a user