diff --git a/data/brialert.sqlite b/data/brialert.sqlite
index db90fbdc..c75ffd02 100644
Binary files a/data/brialert.sqlite and b/data/brialert.sqlite differ
diff --git a/data/sources.json b/data/sources.json
index 7480fb21..4ea2eeb0 100644
--- a/data/sources.json
+++ b/data/sources.json
@@ -25,15 +25,13 @@
{
"id": "frontex-news-releases",
"provider": "Frontex a News Releases",
- "endpoint": "https://www.frontex.europa.eu/media-centre/news/news-release/feed",
+ "endpoint": "https://www.frontex.europa.eu/media-centre/news/news-release/",
"kind": "html",
"lane": "border",
"region": "eu",
"isTrustedOfficial": true,
"requiresKeywordMatch": true,
- "comment": "No RSS endpoint consistently published; using the News Release listing and relying on keyword filtering.",
- "quarantined": false,
- "restoredAt": "2026-04-12T08:47:26.742Z"
+ "comment": "No RSS endpoint consistently published; using the News Release listing and relying on keyword filtering."
},
{
"id": "austria-interior-ministry-news",
@@ -71,15 +69,13 @@
{
"id": "belgium-crisis-centre-news",
"provider": "Belgian National Crisis Centre - News",
- "endpoint": "https://crisiscenter.be/en/newsroom",
+ "endpoint": "https://crisiscenter.be/en/news",
"kind": "html",
"lane": "context",
"region": "eu",
"isTrustedOfficial": true,
"requiresKeywordMatch": true,
- "comment": "Belgian crisis centre updates for threat level, emergency posture and civil contingency context.",
- "quarantined": false,
- "restoredAt": "2026-04-12T08:47:52.261Z"
+ "comment": "Belgian crisis centre updates for threat level, emergency posture and civil contingency context."
},
{
"id": "denmark-pet-news",
@@ -313,15 +309,13 @@
{
"id": "czech-police-news",
"provider": "Police of the Czech Republic - News",
- "endpoint": "https://english.radio.cz/news",
+ "endpoint": "https://www.policie.cz/clanek/news.aspx",
"kind": "html",
"lane": "incidents",
"region": "eu",
"isTrustedOfficial": true,
"requiresKeywordMatch": true,
- "comment": "Czech Police updates for arrests, security incidents and public-order operations relevant to regional threat context.",
- "quarantined": false,
- "restoredAt": "2026-04-12T08:50:52.579Z"
+ "comment": "Czech Police updates for arrests, security incidents and public-order operations relevant to regional threat context."
},
{
"id": "dutch-police-rss-index",
@@ -566,15 +560,13 @@
{
"id": "eu-council-rss-index",
"provider": "Council of the EU - RSS Index",
- "endpoint": "https://securitymattersmagazine.com/news",
+ "endpoint": "https://www.consilium.europa.eu/en/about-site/rss/",
"kind": "html",
"lane": "oversight",
"region": "eu",
"isTrustedOfficial": true,
"requiresKeywordMatch": true,
- "comment": "Council of the EU RSS index page for official feed discovery and source maintenance.",
- "quarantined": false,
- "restoredAt": "2026-04-12T08:51:52.686Z"
+ "comment": "Council of the EU RSS index page for official feed discovery and source maintenance."
},
{
"id": "eu-ct-coordinator",
@@ -927,15 +919,13 @@
{
"id": "politico-europe",
"provider": "Politico Europe",
- "endpoint": "https://www.politico.eu/section/defense/",
+ "endpoint": "https://www.politico.eu/",
"kind": "html",
"lane": "context",
"region": "europe",
"isTrustedOfficial": false,
"requiresKeywordMatch": true,
- "comment": "Politico Europe used as corroborating political-security context rather than live trigger input.",
- "quarantined": false,
- "restoredAt": "2026-04-12T08:52:42.850Z"
+ "comment": "Politico Europe used as corroborating political-security context rather than live trigger input."
},
{
"id": "rte-ireland-news-rss",
@@ -1160,15 +1150,13 @@
{
"id": "nctv-nl-english-news",
"provider": "NCTV Netherlands latest news (English)",
- "endpoint": "https://english.nctv.nl/api/rss?query=%7B%22filters%22%3A%5B%7B%22field%22%3A%22content_type%22%2C%22values%22%3A%5B%22pro%3AnewsDocument%22%5D%2C%22type%22%3A%22all%22%7D%5D%2C%22resultSearchTerm%22%3A%22%22%7D",
+ "endpoint": "https://english.nctv.nl/latest/news",
"kind": "html",
"lane": "prevention",
"region": "europe",
"isTrustedOfficial": true,
"requiresKeywordMatch": false,
- "comment": "Latest news page of the Netherlands' National Coordinator for Security and Counterterrorism (NCTV), in English, listing news and press releases.",
- "quarantined": false,
- "restoredAt": "2026-04-12T08:53:02.544Z"
+ "comment": "Latest news page of the Netherlands' National Coordinator for Security and Counterterrorism (NCTV), in English, listing news and press releases."
},
{
"id": "ocam-be-english",
@@ -1456,14 +1444,12 @@
{
"id": "nato-news-rss",
"provider": "NATO a News (RSS)",
- "endpoint": "https://www.nato.int/en/news-and-events/articles/news",
+ "endpoint": "https://www.nato.int/cps/en/natohq/rss/news.rss",
"kind": "rss",
"lane": "context",
"region": "international",
"isTrustedOfficial": true,
- "requiresKeywordMatch": true,
- "quarantined": false,
- "restoredAt": "2026-04-12T08:53:42.867Z"
+ "requiresKeywordMatch": true
},
{
"id": "occrp-news",
@@ -1575,6 +1561,61 @@
"requiresKeywordMatch": true,
"comment": "War on the Rocks publishes analysis on national security, military affairs and counter-terrorism; keyword filtering applied."
},
+ {
+ "id": "google-news-terrorism-uk",
+ "provider": "Google News - Terrorism UK",
+ "endpoint": "https://news.google.com/rss/search?q=terrorism+UK&hl=en-GB&gl=GB&ceid=GB:en",
+ "kind": "rss",
+ "lane": "context",
+ "region": "international",
+ "isTrustedOfficial": false,
+ "requiresKeywordMatch": true,
+ "comment": "Google News RSS aggregator for UK terrorism coverage. Acts as a safety net when individual institutional sources fail."
+ },
+ {
+ "id": "google-news-counter-terrorism",
+ "provider": "Google News - Counter Terrorism",
+ "endpoint": "https://news.google.com/rss/search?q=%22counter+terrorism%22&hl=en-GB&gl=GB&ceid=GB:en",
+ "kind": "rss",
+ "lane": "context",
+ "region": "international",
+ "isTrustedOfficial": false,
+ "requiresKeywordMatch": true,
+ "comment": "Google News RSS aggregator for counter-terrorism policy and operations coverage."
+ },
+ {
+ "id": "google-news-terror-attack",
+ "provider": "Google News - Terror Attack",
+ "endpoint": "https://news.google.com/rss/search?q=%22terror+attack%22&hl=en-GB&gl=GB&ceid=GB:en",
+ "kind": "rss",
+ "lane": "context",
+ "region": "international",
+ "isTrustedOfficial": false,
+ "requiresKeywordMatch": true,
+ "comment": "Google News RSS aggregator for terror attack incident coverage."
+ },
+ {
+ "id": "google-news-extremism",
+ "provider": "Google News - Extremism",
+ "endpoint": "https://news.google.com/rss/search?q=extremism+UK&hl=en-GB&gl=GB&ceid=GB:en",
+ "kind": "rss",
+ "lane": "context",
+ "region": "international",
+ "isTrustedOfficial": false,
+ "requiresKeywordMatch": true,
+ "comment": "Google News RSS aggregator for UK extremism reporting and prevention coverage."
+ },
+ {
+ "id": "google-alerts-uk-security-threat",
+ "provider": "Google News - UK Security Threat",
+ "endpoint": "https://news.google.com/rss/search?q=%22security+threat%22+UK&hl=en-GB&gl=GB&ceid=GB:en",
+ "kind": "rss",
+ "lane": "context",
+ "region": "international",
+ "isTrustedOfficial": false,
+ "requiresKeywordMatch": true,
+ "comment": "Google News RSS aggregator for UK security threat level changes and alerts."
+ },
{
"id": "gctf-news",
"provider": "Global Counterterrorism Forum (GCTF) a News",
@@ -1713,9 +1754,7 @@
"region": "london",
"isTrustedOfficial": true,
"requiresKeywordMatch": true,
- "comment": "Barking and Dagenham borough news used for East London incident corroboration and civic response.",
- "quarantined": false,
- "restoredAt": "2026-04-12T08:54:18.092Z"
+ "comment": "Barking and Dagenham borough news used for East London incident corroboration and civic response."
},
{
"id": "barnet-council-news",
@@ -1973,15 +2012,13 @@
{
"id": "homerton-healthcare-news",
"provider": "Homerton Healthcare NHS Foundation Trust a News",
- "endpoint": "https://hospitaltimes.co.uk/",
+ "endpoint": "https://www.homerton.nhs.uk/latest-news/",
"kind": "html",
"lane": "context",
"region": "london",
"isTrustedOfficial": true,
"requiresKeywordMatch": true,
- "comment": "Homerton Healthcare news used for East London hospital and emergency-response context.",
- "quarantined": false,
- "restoredAt": "2026-04-12T08:55:08.895Z"
+ "comment": "Homerton Healthcare news used for East London hospital and emergency-response context."
},
{
"id": "hounslow-council-news",
diff --git a/data/sources/international/context.json b/data/sources/international/context.json
index b1102419..4261aac5 100644
--- a/data/sources/international/context.json
+++ b/data/sources/international/context.json
@@ -325,6 +325,61 @@
"isTrustedOfficial": false,
"requiresKeywordMatch": true,
"comment": "War on the Rocks publishes analysis on national security, military affairs and counter-terrorism; keyword filtering applied."
+ },
+ {
+ "id": "google-news-terrorism-uk",
+ "provider": "Google News - Terrorism UK",
+ "endpoint": "https://news.google.com/rss/search?q=terrorism+UK&hl=en-GB&gl=GB&ceid=GB:en",
+ "kind": "rss",
+ "lane": "context",
+ "region": "international",
+ "isTrustedOfficial": false,
+ "requiresKeywordMatch": true,
+ "comment": "Google News RSS aggregator for UK terrorism coverage. Acts as a safety net when individual institutional sources fail."
+ },
+ {
+ "id": "google-news-counter-terrorism",
+ "provider": "Google News - Counter Terrorism",
+ "endpoint": "https://news.google.com/rss/search?q=%22counter+terrorism%22&hl=en-GB&gl=GB&ceid=GB:en",
+ "kind": "rss",
+ "lane": "context",
+ "region": "international",
+ "isTrustedOfficial": false,
+ "requiresKeywordMatch": true,
+ "comment": "Google News RSS aggregator for counter-terrorism policy and operations coverage."
+ },
+ {
+ "id": "google-news-terror-attack",
+ "provider": "Google News - Terror Attack",
+ "endpoint": "https://news.google.com/rss/search?q=%22terror+attack%22&hl=en-GB&gl=GB&ceid=GB:en",
+ "kind": "rss",
+ "lane": "context",
+ "region": "international",
+ "isTrustedOfficial": false,
+ "requiresKeywordMatch": true,
+ "comment": "Google News RSS aggregator for terror attack incident coverage."
+ },
+ {
+ "id": "google-news-extremism",
+ "provider": "Google News - Extremism",
+ "endpoint": "https://news.google.com/rss/search?q=extremism+UK&hl=en-GB&gl=GB&ceid=GB:en",
+ "kind": "rss",
+ "lane": "context",
+ "region": "international",
+ "isTrustedOfficial": false,
+ "requiresKeywordMatch": true,
+ "comment": "Google News RSS aggregator for UK extremism reporting and prevention coverage."
+ },
+ {
+ "id": "google-alerts-uk-security-threat",
+ "provider": "Google News - UK Security Threat",
+ "endpoint": "https://news.google.com/rss/search?q=%22security+threat%22+UK&hl=en-GB&gl=GB&ceid=GB:en",
+ "kind": "rss",
+ "lane": "context",
+ "region": "international",
+ "isTrustedOfficial": false,
+ "requiresKeywordMatch": true,
+ "comment": "Google News RSS aggregator for UK security threat level changes and alerts."
}
]
}
diff --git a/scripts/build-live-feed.mjs b/scripts/build-live-feed.mjs
index b0441191..6d272d4f 100644
--- a/scripts/build-live-feed.mjs
+++ b/scripts/build-live-feed.mjs
@@ -8,11 +8,13 @@ import {
AUTO_QUARANTINE_BLOCKED_HTML_THRESHOLD,
AUTO_QUARANTINE_DEAD_URL_THRESHOLD,
AUTO_QUARANTINE_RECHECK_HOURS,
+ AUTO_QUARANTINE_RECHECK_HOURS_HIGH_PRIORITY,
AUTO_SKIP_EMPTY_THRESHOLD,
AUTO_SKIP_FAILURE_THRESHOLD,
CONTROL_MAX_HTML_SOURCES_PER_RUN,
DEFAULT_FETCH_STAGGER_MS,
FEED_SOURCE_CONCURRENCY,
+ FEED_SOURCE_CONCURRENCY_RSS,
GUARDRAIL_MAX_FAILED_SOURCE_RATE,
GUARDRAIL_MAX_RUNTIME_MS,
GUARDRAIL_MIN_SUCCESSFUL_SOURCES,
@@ -71,7 +73,9 @@ import {
summariseSourceError,
ERROR_CODE,
fetchText,
- fetchTextWithPlaywright
+ fetchTextWithPlaywright,
+ discoverFeedUrl,
+ tryUrlAlternatives
} from './build-live-feed/io.mjs';
import {
enrichHtmlItems,
@@ -129,7 +133,9 @@ function sourceMayAutoCooldown(source, previousEntry, buildDate) {
if (!previousEntry) return null;
if (previousEntry.quarantined) {
const quarantinedAtMs = parseIsoMs(previousEntry.quarantinedAt);
- const quarantineRecheckAt = quarantinedAtMs + (AUTO_QUARANTINE_RECHECK_HOURS * 3600000);
+ const isHighPriority = source?.lane === 'incidents' || source?.lane === 'border' || source?.isTrustedOfficial;
+ const recheckHours = isHighPriority ? AUTO_QUARANTINE_RECHECK_HOURS_HIGH_PRIORITY : AUTO_QUARANTINE_RECHECK_HOURS;
+ const quarantineRecheckAt = quarantinedAtMs + (recheckHours * 3600000);
if (quarantinedAtMs && buildDate.getTime() >= quarantineRecheckAt) {
return null;
}
@@ -188,7 +194,9 @@ function nextSourceHealthEntry(source, stat, previousEntry, generatedAt) {
autoSkipReason: null,
quarantined: Boolean(prior.quarantined),
quarantinedAt: prior.quarantinedAt || null,
- quarantineReason: prior.quarantineReason || null
+ quarantineReason: prior.quarantineReason || null,
+ // Improvement 10: Track redirect final URL for endpoint auto-update
+ lastRedirectFinalUrl: prior.lastRedirectFinalUrl || null
};
if ((stat?.built || 0) > 0) {
@@ -200,6 +208,16 @@ function nextSourceHealthEntry(source, stat, previousEntry, generatedAt) {
next.lastErrorCategory = null;
next.lastErrorMessage = null;
next.lastSuccessfulAt = generatedAt;
+ // Improvement 10: Persist redirect final URL when it differs from endpoint
+ const finalUrl = clean(stat?.finalUrl);
+ if (finalUrl && finalUrl !== clean(source?.endpoint)) {
+ next.lastRedirectFinalUrl = finalUrl;
+ }
+ // Clear quarantine on successful run (Improvement 10 — auto-recovery)
+ if (next.quarantined) {
+ next.quarantined = false;
+ next.quarantineReason = null;
+ }
return next;
}
@@ -463,7 +481,7 @@ function shouldTryPlaywrightFallback(source, summary, playwrightBudget) {
if (!summary) return false;
if ((playwrightBudget?.attempts || 0) >= (playwrightBudget?.maxAttempts || 0)) return false;
const reason = classifyFetchFailure(summary);
- if (reason !== 'bot-block') return false;
+ if (reason !== 'bot-block' && reason !== 'parser-failure') return false;
return PLAYWRIGHT_FALLBACK_ALLOWLIST_SOURCE_IDS.has(source.id) || PLAYWRIGHT_FALLBACK_AGGRESSIVE;
}
@@ -1580,7 +1598,26 @@ async function main() {
maxAttempts: PLAYWRIGHT_FALLBACK_MAX_ATTEMPTS_PER_RUN
};
const scheduledSourceIds = new Set([...machineReadableScheduled, ...htmlScheduled].map((source) => source.id));
- const scheduledSourcesInitial = [...machineReadableScheduled, ...htmlScheduled];
+
+ // Improvement 9: Process machine-readable (fast) sources first, then HTML (slow),
+ // and within each group sort by least-recently-successful to prevent starvation.
+ const sortByLeastRecentSuccess = (sources) => {
+ return [...sources].sort((a, b) => {
+ const aHealth = sourceHealthEntry(previousHealth, a.id);
+ const bHealth = sourceHealthEntry(previousHealth, b.id);
+ const aSuccessMs = parseIsoMs(aHealth?.lastSuccessfulAt);
+ const bSuccessMs = parseIsoMs(bHealth?.lastSuccessfulAt);
+ // Sources never successfully checked get highest priority (sort first)
+ if (!aSuccessMs && bSuccessMs) return -1;
+ if (aSuccessMs && !bSuccessMs) return 1;
+ // Among checked sources, oldest success first (ascending)
+ return aSuccessMs - bSuccessMs;
+ });
+ };
+ const scheduledSourcesInitial = [
+ ...sortByLeastRecentSuccess(machineReadableScheduled),
+ ...sortByLeastRecentSuccess(htmlScheduled)
+ ];
const htmlDeferredForBudget = scheduledSources.filter((source) => source?.kind === 'html' && !scheduledSourceIds.has(source.id));
const continuationOversamplingFactor = 2;
const htmlDeferredReasonById = new Map();
@@ -1613,9 +1650,12 @@ async function main() {
async function processSourceBatch(batch) {
if (!batch.length) return;
+ // Improvement 9: Use higher concurrency for machine-readable feeds
+ const hasOnlyMachineReadable = batch.every((source) => isMachineReadableSourceKind(source?.kind));
+ const effectiveConcurrency = hasOnlyMachineReadable ? FEED_SOURCE_CONCURRENCY_RSS : FEED_SOURCE_CONCURRENCY;
const sourceResults = await mapWithConcurrency(
batch,
- FEED_SOURCE_CONCURRENCY,
+ effectiveConcurrency,
async (source, sourceIndex) => {
const localErrors = [];
const builtAlerts = [];
@@ -1665,6 +1705,8 @@ async function main() {
if (reason === 'stale-endpoint') failureReasonCounts['stale-endpoint'] += 1;
else if (reason === 'bot-block') failureReasonCounts['blocked-or-anti-bot'] += 1;
else if (reason === 'timeout') failureReasonCounts['timeout-or-aborted'] += 1;
+
+ // Improvement 1: Playwright browser fallback for bot-blocked HTML sources
if (shouldTryPlaywrightFallback(source, summary, playwrightBudget)) {
playwrightBudget.attempts += 1;
body = await fetchTextWithPlaywright(source.endpoint, {
@@ -1674,6 +1716,37 @@ async function main() {
usedPlaywrightFallback = true;
playwrightBudget.successes += 1;
fetchOutcome = 'success';
+
+ // Improvement 3: Try replacement endpoint before giving up
+ } else if (
+ (reason === 'stale-endpoint' || reason === 'http-error' || reason === 'bot-block') &&
+ clean(source.replacementEndpoint || source.fallbackEndpoint || source.canonicalEndpoint)
+ ) {
+ const altEndpoint = clean(source.replacementEndpoint || source.fallbackEndpoint || source.canonicalEndpoint);
+ try {
+ const fetched = await fetchText(altEndpoint, 1, { source, requestState, includeMeta: true });
+ body = typeof fetched === 'string' ? fetched : fetched.text;
+ finalUrl = typeof fetched === 'string' ? altEndpoint : clean(fetched.finalUrl || altEndpoint);
+ fetchOutcome = 'success';
+ } catch {
+ throw error;
+ }
+
+ // Improvement 10: Try URL alternatives for 404 errors
+ } else if (reason === 'stale-endpoint') {
+ const altUrl = await tryUrlAlternatives(source.endpoint).catch(() => null);
+ if (altUrl) {
+ try {
+ const fetched = await fetchText(altUrl, 1, { source, requestState, includeMeta: true });
+ body = typeof fetched === 'string' ? fetched : fetched.text;
+ finalUrl = typeof fetched === 'string' ? altUrl : clean(fetched.finalUrl || altUrl);
+ fetchOutcome = 'success';
+ } catch {
+ throw error;
+ }
+ } else {
+ throw error;
+ }
} else {
throw error;
}
@@ -1681,7 +1754,28 @@ async function main() {
const parsed = source.kind === 'rss' || source.kind === 'atom' || source.kind === 'json'
? parseFeedItems(source, body)
: parseHtmlItems(source, body);
- if (!parsed.length) {
+
+ // Improvement 2: When HTML parsing yields no results, try discovering
+ // an RSS/Atom feed at the same domain before reporting empty.
+ let feedDiscoveryParsed = null;
+ if (!parsed.length && source.kind === 'html' && fetchOutcome !== 'unchanged') {
+ try {
+ const discovered = await discoverFeedUrl(source.endpoint);
+ if (discovered) {
+ const feedBody = await fetchText(discovered.feedUrl, 1, { source, requestState, includeMeta: false });
+ const feedSource = { ...source, kind: discovered.feedKind, endpoint: discovered.feedUrl };
+ feedDiscoveryParsed = parseFeedItems(feedSource, feedBody);
+ if (feedDiscoveryParsed.length) {
+ console.log(`Feed discovery found ${feedDiscoveryParsed.length} item(s) at ${discovered.feedUrl} for source ${source.id}`);
+ }
+ }
+ } catch {
+ // Feed discovery is best-effort; do not block the main flow
+ }
+ }
+ const effectiveParsed = (feedDiscoveryParsed && feedDiscoveryParsed.length) ? feedDiscoveryParsed : parsed;
+
+ if (!effectiveParsed.length) {
discardReasons.parseNoItems += 1;
if (fetchOutcome !== 'unchanged') {
failureReasonCounts['empty-or-no-items'] += 1;
@@ -1692,7 +1786,7 @@ async function main() {
));
}
const preLimit = source.kind === 'html' ? MAX_HTML_PREFETCH_ITEMS : MAX_FEED_PREFETCH_ITEMS;
- const preLimited = parsed.slice(0, preLimit);
+ const preLimited = effectiveParsed.slice(0, preLimit);
const hydrated = source.kind === 'html' ? await enrichHtmlItems(source, preLimited) : preLimited;
const reliabilityProfile = inferReliabilityProfile(source, inferSourceTier(source));
const itemLimit = reliabilityProfile === 'tabloid'
@@ -1743,7 +1837,7 @@ async function main() {
provider: source.provider,
lane: source.lane,
kind: source.kind,
- parsed: parsed.length,
+ parsed: effectiveParsed.length,
hydrated: hydrated.length,
filtered: filtered.length,
kept: kept.length,
diff --git a/scripts/build-live-feed/config.mjs b/scripts/build-live-feed/config.mjs
index 701c3118..906b7c4b 100644
--- a/scripts/build-live-feed/config.mjs
+++ b/scripts/build-live-feed/config.mjs
@@ -60,6 +60,13 @@ export const SCHEDULER_MODE = clean(process.env.BRIALERT_SCHEDULER_AB_MODE || 'c
export const PLAYWRIGHT_FALLBACK_ALLOWLIST_SOURCE_IDS = new Set([
'met-police-news',
'ct-policing-london',
+ 'met-police-latest-news',
+ 'city-of-london-police-news-html',
+ 'city-of-london-police-newsroom',
+ 'mopac-html',
+ 'london-fire-brigade-news',
+ 'tfl-press-releases-html',
+ 'london-gov-press-releases-html',
...clean(process.env.BRIALERT_PLAYWRIGHT_ALLOWLIST || '')
.split(',')
.map((value) => clean(value))
@@ -69,7 +76,7 @@ export const PLAYWRIGHT_FALLBACK_MAX_ATTEMPTS_PER_RUN = Math.max(
0,
Number.isFinite(Number(process.env.BRIALERT_PLAYWRIGHT_MAX_ATTEMPTS_PER_RUN))
? Math.floor(Number(process.env.BRIALERT_PLAYWRIGHT_MAX_ATTEMPTS_PER_RUN))
- : 2
+ : 6
);
export const PLAYWRIGHT_FALLBACK_AGGRESSIVE = clean(process.env.BRIALERT_PLAYWRIGHT_AGGRESSIVE).toLowerCase() === 'true';
export const PLAYWRIGHT_FALLBACK_TIMEOUT_MS = Math.max(
@@ -124,8 +131,16 @@ export const RETRYABLE_STATUS_CODES = new Set([408, 425, 429, 500, 502, 503, 504
export const FEED_BOT_USER_AGENT = 'Mozilla/5.0 (compatible; BrialertFeedBot/1.0; +https://potemkin666.github.io/Brialert/)';
export const FEED_BOT_USER_AGENTS = Object.freeze([
FEED_BOT_USER_AGENT,
+ 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/136.0.0.0 Safari/537.36',
+ 'Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/136.0.0.0 Safari/537.36',
+ 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/136.0.0.0 Safari/537.36',
+ 'Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:137.0) Gecko/20100101 Firefox/137.0',
+ 'Mozilla/5.0 (Macintosh; Intel Mac OS X 14.5; rv:137.0) Gecko/20100101 Firefox/137.0',
+ 'Mozilla/5.0 (X11; Linux x86_64; rv:137.0) Gecko/20100101 Firefox/137.0',
+ 'Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.4 Safari/605.1.15',
+ 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/136.0.0.0 Safari/537.36 Edg/136.0.0.0',
+ 'Mozilla/5.0 (Macintosh; Intel Mac OS X 14_5) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/136.0.0.0 Safari/537.36 Edg/136.0.0.0',
'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36',
- 'Mozilla/5.0 (Macintosh; Intel Mac OS X 14_4) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36',
'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/135.0.0.0 Safari/537.36'
]);
export const DEFAULT_SOURCE_REFRESH_HOURS_BY_LANE = Object.freeze({
@@ -141,7 +156,8 @@ export const SOURCE_FAILURE_COOLDOWN_HOURS = 24;
export const SOURCE_EMPTY_COOLDOWN_HOURS = 24;
export const SOURCE_PROTECTED_FAILURE_COOLDOWN_HOURS = 6;
export const SOURCE_BLOCKED_FAILURE_COOLDOWN_HOURS = 12;
-export const AUTO_QUARANTINE_RECHECK_HOURS = 7 * 24;
+export const AUTO_QUARANTINE_RECHECK_HOURS = envInt('BRIALERT_AUTO_QUARANTINE_RECHECK_HOURS', 3 * 24, 1);
+export const AUTO_QUARANTINE_RECHECK_HOURS_HIGH_PRIORITY = envInt('BRIALERT_AUTO_QUARANTINE_RECHECK_HOURS_HIGH_PRIORITY', 2 * 24, 1);
export const AUTO_SKIP_FAILURE_THRESHOLD = 4;
export const AUTO_SKIP_EMPTY_THRESHOLD = 6;
export const AUTO_QUARANTINE_BLOCKED_HTML_THRESHOLD = envInt('BRIALERT_AUTO_QUARANTINE_BLOCKED_HTML_THRESHOLD', 4, 1);
@@ -166,6 +182,50 @@ export const HARD_SKIP_SOURCE_IDS = new Set([
'kallxo-english-home'
]);
+// Playwright scraper defaults (referenced by playwright-scraper.mjs)
+export const DEFAULT_PLAYWRIGHT_TIMEOUT_MS = PLAYWRIGHT_FALLBACK_TIMEOUT_MS;
+export const DEFAULT_PLAYWRIGHT_PAGE_SETTLE_MS = envInt('BRIALERT_PLAYWRIGHT_PAGE_SETTLE_MS', 1500, 200);
+export const MAX_PLAYWRIGHT_RAW_CANDIDATES = MAX_HTML_PARSING_THRESHOLD;
+export const MAX_PLAYWRIGHT_ITEM_SUMMARY_CHARS = 420;
+export const PLAYWRIGHT_SCRAPER_USER_AGENT = FEED_BOT_USER_AGENTS[1] || FEED_BOT_USER_AGENT;
+
+// Proxy support (Improvement 5)
+export const PROXY_URL = clean(process.env.BRIALERT_PROXY_URL);
+export const PROXY_LIST = clean(process.env.BRIALERT_PROXY_LIST)
+ .split(',')
+ .map((value) => clean(value))
+ .filter(Boolean);
+
+// RSS/Atom feed concurrency (Improvement 9) — lightweight feeds tolerate higher parallelism
+export const FEED_SOURCE_CONCURRENCY_RSS = envInt('BRIALERT_FEED_SOURCE_CONCURRENCY_RSS', 6, 1);
+
+// Feed discovery paths (Improvement 2)
+export const FEED_DISCOVERY_PATHS = Object.freeze([
+ '/feed',
+ '/rss',
+ '/atom.xml',
+ '/feed.xml',
+ '/feeds/posts/default',
+ '/rss.xml',
+ '?format=rss',
+ '?feed=rss2'
+]);
+
+// Adaptive timeout constants (Improvement 6) — high-value sources get extra retries
+export const ADAPTIVE_TIMEOUT_MULTIPLIER = 1.5;
+export const HIGH_VALUE_MAX_RETRIES = envInt('BRIALERT_HIGH_VALUE_MAX_RETRIES', 4, 1);
+export const FAST_FEED_TIMEOUT_MS = envInt('BRIALERT_FAST_FEED_TIMEOUT_MS', 8000, 1000);
+
+// URL variations for 404 recovery (Improvement 10)
+export const URL_PATH_ALTERNATIVES = Object.freeze([
+ ['/news/', '/press-releases/'],
+ ['/press-releases/', '/news/'],
+ ['/media/', '/newsroom/'],
+ ['/newsroom/', '/media/'],
+ ['/news/', '/media/news/'],
+ ['/articles/', '/news/']
+]);
+
export const severityRank = { critical: 4, high: 3, elevated: 2, moderate: 1 };
export function titleCase(value) {
@@ -190,6 +250,21 @@ export function sourceUserAgent(source) {
return FEED_BOT_USER_AGENTS[index] || FEED_BOT_USER_AGENT;
}
+/**
+ * Returns a randomised user-agent different from the deterministic one,
+ * useful for retries after bot-block detection. Excludes the bot UA for
+ * sources that have been blocked.
+ */
+export function randomBrowserUserAgent(excludeIndex = -1) {
+ // Skip the bot UA (index 0) to avoid fingerprinting on retries
+ const candidates = FEED_BOT_USER_AGENTS.slice(1);
+ if (!candidates.length) return FEED_BOT_USER_AGENT;
+ const safeExclude = excludeIndex > 0 ? excludeIndex - 1 : -1;
+ const filtered = candidates.filter((_, idx) => idx !== safeExclude);
+ const pool = filtered.length ? filtered : candidates;
+ return pool[Math.floor(Math.random() * pool.length)];
+}
+
export function sourceRefreshEveryHours(source) {
const explicit = Number(source?.refreshEveryHours);
if (Number.isFinite(explicit) && explicit >= 0.25) return explicit;
diff --git a/scripts/build-live-feed/io.mjs b/scripts/build-live-feed/io.mjs
index 34841b0f..e22a4ca4 100644
--- a/scripts/build-live-feed/io.mjs
+++ b/scripts/build-live-feed/io.mjs
@@ -7,9 +7,16 @@ import {
OFFLINE_FIXTURE_MODE,
offlineFixturesPath,
sourceUserAgent,
+ randomBrowserUserAgent,
RETRYABLE_STATUS_CODES,
outputPath,
- repoRoot
+ repoRoot,
+ PROXY_URL,
+ PROXY_LIST,
+ FEED_DISCOVERY_PATHS,
+ HIGH_VALUE_MAX_RETRIES,
+ FAST_FEED_TIMEOUT_MS,
+ URL_PATH_ALTERNATIVES
} from './config.mjs';
import { clean } from '../../shared/taxonomy.mjs';
@@ -263,16 +270,181 @@ function classifyBodyBlock(text = '') {
return null;
}
+/**
+ * Selects a proxy URL from the configured proxy list or single proxy URL.
+ * Returns null when no proxy is configured.
+ */
+function selectProxyUrl() {
+ if (PROXY_LIST.length) {
+ return PROXY_LIST[Math.floor(Math.random() * PROXY_LIST.length)];
+ }
+ return PROXY_URL || null;
+}
+
+/**
+ * Computes an adaptive timeout for a source based on its historical response
+ * times and criticality. High-value sources (incidents, trusted official)
+ * get a longer timeout; fast RSS/Atom feeds get a shorter one.
+ */
+export function adaptiveTimeoutMs(source) {
+ const explicit = Number(source?.timeoutMs);
+ if (explicit > 0) return explicit;
+ const isHighValue = source?.lane === 'incidents' || source?.isTrustedOfficial;
+ const isFastFeed = source?.kind === 'rss' || source?.kind === 'atom' || source?.kind === 'json';
+ if (isFastFeed && !isHighValue) return FAST_FEED_TIMEOUT_MS;
+ return DEFAULT_TIMEOUT_MS;
+}
+
+/**
+ * Returns the effective max retries for a source. High-value sources
+ * (incidents lane, trusted officials) receive extra retry attempts.
+ */
+export function adaptiveMaxRetries(source) {
+ const explicit = Number(source?.maxRetries);
+ if (explicit > 0) return explicit;
+ const isHighValue = source?.lane === 'incidents' || source?.isTrustedOfficial;
+ return isHighValue ? HIGH_VALUE_MAX_RETRIES : DEFAULT_MAX_RETRIES;
+}
+
+/**
+ * Probes a list of common feed paths on the same domain to discover an
+ * RSS/Atom feed. Returns { feedUrl, feedKind } when a feed is found,
+ * or null when no feed is discovered.
+ */
+export async function discoverFeedUrl(endpointUrl) {
+ let origin;
+ try {
+ origin = new URL(endpointUrl).origin;
+ } catch {
+ return null;
+ }
+ for (const feedPath of FEED_DISCOVERY_PATHS) {
+ const candidateUrl = feedPath.startsWith('?')
+ ? `${origin}/${feedPath}`
+ : `${origin}${feedPath}`;
+ const controller = new AbortController();
+ const timeout = setTimeout(() => controller.abort(), 5000);
+ try {
+ const response = await fetch(candidateUrl, {
+ method: 'HEAD',
+ redirect: 'follow',
+ credentials: 'omit',
+ signal: controller.signal,
+ headers: { 'user-agent': randomBrowserUserAgent() }
+ });
+ clearTimeout(timeout);
+ if (!response.ok) continue;
+ const contentType = (response.headers.get('content-type') || '').toLowerCase();
+ if (
+ contentType.includes('xml') ||
+ contentType.includes('rss') ||
+ contentType.includes('atom') ||
+ contentType.includes('feed+json')
+ ) {
+ const feedKind = contentType.includes('atom') ? 'atom'
+ : contentType.includes('feed+json') ? 'json'
+ : 'rss';
+ return { feedUrl: clean(response.url || candidateUrl), feedKind };
+ }
+ } catch {
+ clearTimeout(timeout);
+ }
+ }
+ // Also look for in the original HTML page
+ try {
+ const controller = new AbortController();
+ const timeout = setTimeout(() => controller.abort(), 6000);
+ const response = await fetch(endpointUrl, {
+ redirect: 'follow',
+ credentials: 'omit',
+ signal: controller.signal,
+ headers: { 'user-agent': randomBrowserUserAgent() }
+ });
+ clearTimeout(timeout);
+ if (response.ok) {
+ const html = await response.text();
+ // Match tags with type="application/rss+xml" or "application/atom+xml"
+ // regardless of attribute order (type and href may appear in any sequence).
+ const linkTagPattern = /]*?(?:type=["']application\/(rss\+xml|atom\+xml)["'])[^>]*>/gi;
+ const hrefPattern = /href=["']([^"']+)["']/i;
+ let linkMatch;
+ while ((linkMatch = linkTagPattern.exec(html)) !== null) {
+ const hrefMatch = linkMatch[0].match(hrefPattern);
+ if (hrefMatch) {
+ const feedKind = linkMatch[1].includes('atom') ? 'atom' : 'rss';
+ const feedUrl = absoluteUrl(hrefMatch[1], endpointUrl);
+ return { feedUrl, feedKind };
+ }
+ }
+ }
+ } catch {
+ // Discovery is best-effort
+ }
+ return null;
+}
+
+/**
+ * Try URL path alternatives for a 404'd endpoint (Improvement 10).
+ * Returns the first responding URL or null.
+ */
+export async function tryUrlAlternatives(endpointUrl) {
+ let parsedUrl;
+ try {
+ parsedUrl = new URL(endpointUrl);
+ } catch {
+ return null;
+ }
+ const originalPath = parsedUrl.pathname;
+ const candidates = [];
+
+ // www prefix/removal
+ if (parsedUrl.hostname.startsWith('www.')) {
+ const alt = new URL(endpointUrl);
+ alt.hostname = alt.hostname.replace(/^www\./, '');
+ candidates.push(alt.toString());
+ } else {
+ const alt = new URL(endpointUrl);
+ alt.hostname = `www.${alt.hostname}`;
+ candidates.push(alt.toString());
+ }
+
+ // Path reorganisations
+ for (const [from, to] of URL_PATH_ALTERNATIVES) {
+ if (originalPath.includes(from)) {
+ const alt = new URL(endpointUrl);
+ alt.pathname = originalPath.replace(from, to);
+ candidates.push(alt.toString());
+ }
+ }
+
+ for (const candidateUrl of candidates) {
+ const controller = new AbortController();
+ const timeout = setTimeout(() => controller.abort(), 5000);
+ try {
+ const response = await fetch(candidateUrl, {
+ method: 'HEAD',
+ redirect: 'follow',
+ credentials: 'omit',
+ signal: controller.signal,
+ headers: { 'user-agent': randomBrowserUserAgent() }
+ });
+ clearTimeout(timeout);
+ if (response.ok) return clean(response.url || candidateUrl);
+ } catch {
+ clearTimeout(timeout);
+ }
+ }
+ return null;
+}
+
export async function fetchText(url, attempt = 1, options = {}) {
if (OFFLINE_FIXTURE_MODE) {
const offlinePayload = await offlineFixtureResponse(url, options);
return options?.includeMeta ? offlinePayload : offlinePayload.text;
}
const source = options?.source || null;
- const configuredTimeoutMs = Number(source?.timeoutMs);
- const configuredMaxRetries = Number(source?.maxRetries);
- const timeoutMs = configuredTimeoutMs > 0 ? configuredTimeoutMs : DEFAULT_TIMEOUT_MS;
- const maxAttempts = configuredMaxRetries > 0 ? configuredMaxRetries : DEFAULT_MAX_RETRIES;
+ const timeoutMs = adaptiveTimeoutMs(source);
+ const maxAttempts = adaptiveMaxRetries(source);
const endpoint = clean(url);
const domain = endpointDomain(endpoint);
const existingState = options?.requestState && typeof options.requestState === 'object'
@@ -298,10 +470,12 @@ export async function fetchText(url, attempt = 1, options = {}) {
const timeout = setTimeout(() => controller.abort(), timeoutMs);
try {
+ const retryHeaders = attempt > 1 ? { 'user-agent': randomBrowserUserAgent() } : {};
const response = await fetch(url, {
headers: {
...mergedHeaders(source),
- ...conditionalHeaders
+ ...conditionalHeaders,
+ ...retryHeaders
},
redirect: 'follow',
// Intentionally avoid ambient cookies/session state to reduce auth-gated/geo/session bot challenges.
@@ -455,10 +629,13 @@ export async function fetchTextWithPlaywright(url, options = {}) {
errorCode: ERROR_CODE.PLAYWRIGHT_UNAVAILABLE
});
}
- const browser = await playwright.chromium.launch({ headless: true });
+ const proxyUrl = selectProxyUrl();
+ const launchOptions = { headless: true };
+ if (proxyUrl) launchOptions.proxy = { server: proxyUrl };
+ const browser = await playwright.chromium.launch(launchOptions);
try {
const context = await browser.newContext({
- userAgent: sourceUserAgent(options?.source),
+ userAgent: randomBrowserUserAgent(),
locale: 'en-GB'
});
const page = await context.newPage();
diff --git a/scripts/build-live-feed/parsing.mjs b/scripts/build-live-feed/parsing.mjs
index e4af8579..0d8dfea3 100644
--- a/scripts/build-live-feed/parsing.mjs
+++ b/scripts/build-live-feed/parsing.mjs
@@ -297,6 +297,17 @@ export function parseHtmlItems(source, html) {
'a[href]'
];
+ function addCandidate(href, title, summary, published) {
+ if (!href || !title || title.length < 18) return;
+ if (href.startsWith('#') || href.startsWith('javascript:') || href.startsWith('mailto:')) return;
+ const url = absoluteUrl(href, source.endpoint);
+ const key = `${title}|${url}`;
+ if (seen.has(key)) return;
+ seen.add(key);
+ candidates.push({ title, link: url, summary: summary || '', published: published || '' });
+ }
+
+ // Strategy 1: CSS selector cascade (existing logic)
for (const selector of selectors) {
$(selector).each((_, el) => {
if (candidates.length >= MAX_HTML_PARSING_THRESHOLD) return false;
@@ -316,6 +327,73 @@ export function parseHtmlItems(source, html) {
if (candidates.length >= MAX_HTML_CANDIDATES_PER_SOURCE) break;
}
+ // Strategy 2: LD+JSON structured data extraction (Improvement 8)
+ if (candidates.length === 0) {
+ $('script[type="application/ld+json"]').each((_, el) => {
+ if (candidates.length >= MAX_HTML_CANDIDATES_PER_SOURCE) return false;
+ try {
+ const parsed = JSON.parse($(el).contents().text());
+ const objects = collectJsonLd(parsed);
+ for (const obj of objects) {
+ if (candidates.length >= MAX_HTML_CANDIDATES_PER_SOURCE) break;
+ const ldType = clean(obj['@type'] || '');
+ if (!/article|newsarticle|reportagenewsarticle|blogposting|webpage/i.test(ldType)) continue;
+ const title = plainText(obj.headline || obj.name || '');
+ const mainEntity = obj.mainEntityOfPage;
+ const mainEntityUrl = typeof mainEntity === 'string' ? mainEntity : clean(mainEntity?.['@id'] || '');
+ const href = clean(obj.url || mainEntityUrl || '');
+ const summary = plainText(obj.description || '').slice(0, 420);
+ const published = clean(obj.datePublished || obj.dateCreated || '');
+ addCandidate(href, title, summary, published);
+ }
+ } catch {
+ // JSON-LD parse failure — skip gracefully
+ }
+ });
+ }
+
+ // Strategy 3: OG meta extraction for single-article pages (Improvement 8)
+ if (candidates.length === 0) {
+ const ogTitle = plainText($('meta[property="og:title"]').attr('content') || '');
+ const ogUrl = clean($('meta[property="og:url"]').attr('content') || '');
+ const ogDescription = plainText($('meta[property="og:description"]').attr('content') || '').slice(0, 420);
+ const ogDate = clean($('meta[property="article:published_time"]').attr('content') || '');
+ addCandidate(ogUrl, ogTitle, ogDescription, ogDate);
+ }
+
+ // Strategy 4: Heuristic link scoring for heavily-JS-rendered pages (Improvement 8)
+ if (candidates.length === 0) {
+ const scoredLinks = [];
+ $('main a[href], body a[href]').each((_, el) => {
+ if (scoredLinks.length >= MAX_HTML_PARSING_THRESHOLD) return false;
+ const href = $(el).attr('href');
+ const title = plainText($(el).text());
+ if (!href || !title || title.length < 20) return;
+ if (href.startsWith('#') || href.startsWith('javascript:') || href.startsWith('mailto:')) return;
+ const url = absoluteUrl(href, source.endpoint);
+ const key = `${title}|${url}`;
+ if (seen.has(key)) return;
+ let score = 0;
+ const container = $(el).closest('article,li,section,div');
+ if ($(el).closest('article').length) score += 3;
+ if (container.find('time').length) score += 2;
+ if ($(el).closest('h1,h2,h3,h4').length || $(el).parent('h1,h2,h3,h4').length) score += 2;
+ if ($(el).closest('nav,header,footer').length) score -= 5;
+ if (title.length > 30) score += 1;
+ scoredLinks.push({ href, title, score, container, url, key });
+ });
+ scoredLinks
+ .sort((a, b) => b.score - a.score)
+ .slice(0, MAX_HTML_CANDIDATES_PER_SOURCE)
+ .filter((link) => link.score > 0)
+ .forEach((link) => {
+ seen.add(link.key);
+ const summary = plainText(link.container.text()).slice(0, 420);
+ const published = clean(link.container.find('time').attr('datetime') || link.container.find('time').text());
+ candidates.push({ title: link.title, link: link.url, summary, published });
+ });
+ }
+
return candidates.slice(0, MAX_HTML_CANDIDATES_PER_SOURCE);
}