feat(news): add scraper adapters, article deduplication, blocklist service, and remove tracked publish artifacts

This commit is contained in:
2026-08-24 21:35:33 +02:00
parent 44b161d509
commit 8112598602
400 changed files with 1812 additions and 113367 deletions
@@ -1,62 +1,65 @@
using System;
using System.Collections.Concurrent;
using System.Collections.Generic;
using System.IO;
using System.Linq;
using System.Text.Json;
using System.Text.RegularExpressions;
using System.Threading;
using System.Threading.Tasks;
using FinlyticAssets.Models;
using FinlyticAssets.Util;
using FinlyticCore.Dtos.News;
using FinlyticCore.Services;
using FinlyticCore.Util;
using FinlyticNews.Entities;
using FinlyticNews.Util;
using Microsoft.Extensions.Configuration;
using Microsoft.Extensions.DependencyInjection;
using Microsoft.Extensions.Hosting;
namespace FinlyticNews.Services;
/// <summary>
/// A background worker service that orchestrates link discovery, Playwright scraping,
/// pre-filtering, n8n AI enrichment, database persistence, and MQTT broadcasts.
/// Background worker service orchestrating link discovery, Playwright scraping,
/// non-AI deduplication, in-memory asset matching, plausibility validation, and MQTT broadcasts.
/// </summary>
public class NewsScraperBackgroundService : BackgroundService
{
private record CompiledAssetMatcher(
AssetIndex Asset,
string CoreName,
Regex? WordRegex,
Regex? CoreWordRegex
);
private readonly IServiceScopeFactory _scopeFactory;
private readonly IFinlyticLogger<NewsScraperBackgroundService> _finlyticLogger;
private readonly NewsMqttClient _mqttClient;
private readonly string _indexPath;
private List<CompiledAssetMatcher>? _cachedAssetMatchers;
private DateTime _lastIndexLoadTime = DateTime.MinValue;
private readonly INewsBlocklistService _blocklistService;
private readonly IArticleDeduplicationService _deduplicationService;
private readonly IAssetMatcherService _assetMatcherService;
private readonly IAssetValidationService _assetValidationService;
public NewsScraperBackgroundService(
IServiceScopeFactory scopeFactory,
IFinlyticLogger<NewsScraperBackgroundService> finlyticLogger,
NewsMqttClient mqttClient,
IConfiguration configuration)
INewsBlocklistService blocklistService,
IArticleDeduplicationService deduplicationService,
IAssetMatcherService assetMatcherService,
IAssetValidationService assetValidationService)
{
_scopeFactory = scopeFactory;
_finlyticLogger = finlyticLogger;
_mqttClient = mqttClient;
_indexPath = Path.Combine(Volumes.IndexRelativePath, "index.json");
_blocklistService = blocklistService;
_deduplicationService = deduplicationService;
_assetMatcherService = assetMatcherService;
_assetValidationService = assetValidationService;
}
protected override async Task ExecuteAsync(CancellationToken stoppingToken)
{
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] NewsScraperBackgroundService started.");
// Initialize in-memory blocklist and deduplication cache on startup
try
{
await _blocklistService.InitializeAsync(stoppingToken);
await _deduplicationService.InitializeAsync(stoppingToken);
await _assetMatcherService.ReloadIndexAsync(stoppingToken);
}
catch (Exception ex)
{
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] Error during startup service initialization.");
}
while (!stoppingToken.IsCancellationRequested)
{
try
@@ -76,7 +79,7 @@ public class NewsScraperBackgroundService : BackgroundService
}
catch (Exception ex) when (ex is not OperationCanceledException)
{
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] An unhandled exception occurred during news scraping cycle.");
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] Unhandled exception in news scraping cycle.");
}
int intervalMinutes = 15;
@@ -89,7 +92,7 @@ public class NewsScraperBackgroundService : BackgroundService
catch { }
var jitterSeconds = Random.Shared.Next(0, 60);
var nextRunDelay = TimeSpan.FromMinutes(intervalMinutes) + TimeSpan.FromSeconds(jitterSeconds);
var nextRunDelay = TimeSpan.FromMinutes(Math.Max(1, intervalMinutes)) + TimeSpan.FromSeconds(jitterSeconds);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Scraping cycle completed. Next cycle in {Delay} (interval: {Minutes}m).", nextRunDelay, intervalMinutes);
try
@@ -111,29 +114,26 @@ public class NewsScraperBackgroundService : BackgroundService
var dbService = scope.ServiceProvider.GetRequiredService<INewsDbService>();
var discoveryService = scope.ServiceProvider.GetRequiredService<IArticleDiscoveryService>();
var scraperService = scope.ServiceProvider.GetRequiredService<IPlaywrightScraperService>();
var n8nService = scope.ServiceProvider.GetRequiredService<IN8nService>();
var settings = scope.ServiceProvider.GetRequiredService<ISettingsService>();
var maxArticlesPerFeed = await settings.GetSettingAsync(SettingKeys.MaxArticlesPerFeed, stoppingToken);
var assetMatchers = await GetOrLoadAssetMatchersAsync();
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Loaded {Count} asset index items for text pre-filtering.", assetMatchers.Count);
var failedArticles = await dbService.GetArticlesByStatusAsync("Scraping");
if (failedArticles.Count > 0)
// Process any previously interrupted articles
var pendingScraping = await dbService.GetArticlesByStatusAsync("Scraping");
if (pendingScraping.Count > 0)
{
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Found {Count} articles in status 'Scraping' that failed to scrape previously. Retrying...", failedArticles.Count);
foreach (var article in failedArticles)
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Retrying {Count} pending scraping articles...", pendingScraping.Count);
foreach (var article in pendingScraping)
{
if (stoppingToken.IsCancellationRequested) return;
await ProcessSingleArticleAsync(article, scraperService, n8nService, dbService, assetMatchers, stoppingToken);
await ProcessSingleArticleAsync(article, scraperService, dbService, stoppingToken);
}
}
var sources = await dbService.GetSourcesAsync();
if (sources.Count == 0)
{
await _finlyticLogger.LogWarningAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] No article sources configured in database. Skipping cycle.");
await _finlyticLogger.LogWarningAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] No article sources configured in database.");
return;
}
@@ -141,30 +141,35 @@ public class NewsScraperBackgroundService : BackgroundService
{
if (stoppingToken.IsCancellationRequested) break;
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Starting article link discovery for source: {SourceName} ({Url})", source.Name, source.Source);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Discovering articles from source: {Name} ({Url})", source.Name, source.Source);
var discoveredArticles = await discoveryService.DiscoverLinksAsync(source.Source, source.Type, stoppingToken);
if (discoveredArticles == null || discoveredArticles.Count == 0)
{
await _finlyticLogger.LogDebugAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] No links discovered from source: {SourceName}", source.Name);
continue;
}
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Discovered {Count} potential article links from {SourceName}.", discoveredArticles.Count, source.Name);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Discovered {Count} potential article links from {Name}.", discoveredArticles.Count, source.Name);
var toProcess = discoveredArticles.Take(maxArticlesPerFeed > 0 ? maxArticlesPerFeed : 20);
foreach (var discovered in toProcess)
{
if (stoppingToken.IsCancellationRequested) break;
// 1. Fast Blocklist & Duplicate check before creating DB entry
if (_blocklistService.IsBlocked(discovered.Url))
{
await _finlyticLogger.LogDebugAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Skipping blocked article URL: {Url}", discovered.Url);
continue;
}
if (await dbService.IsUrlDuplicateAsync(discovered.Url))
{
await _finlyticLogger.LogDebugAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Skipping duplicate article URL: {Url}", discovered.Url);
continue;
}
NewsArticleEntity article;
NewsArticleEntity? article;
try
{
article = await dbService.CreatePendingArticleAsync(
@@ -178,17 +183,13 @@ public class NewsScraperBackgroundService : BackgroundService
}
catch (Exception ex)
{
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] Failed to register initial pending state for URL: {Url}. Skipping.", discovered.Url);
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] Failed to create pending article for {Url}", discovered.Url);
continue;
}
if (article.Id == Guid.Empty)
{
await _finlyticLogger.LogWarningAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Created pending article has invalid/empty ID for URL: {Url}. Skipping.", discovered.Url);
continue;
}
if (article == null || article.Id == Guid.Empty) continue;
await ProcessSingleArticleAsync(article, scraperService, n8nService, dbService, assetMatchers, stoppingToken);
await ProcessSingleArticleAsync(article, scraperService, dbService, stoppingToken);
}
}
}
@@ -196,25 +197,24 @@ public class NewsScraperBackgroundService : BackgroundService
private async Task ProcessSingleArticleAsync(
NewsArticleEntity article,
IPlaywrightScraperService scraperService,
IN8nService n8nService,
INewsDbService dbService,
List<CompiledAssetMatcher> assetMatchers,
CancellationToken stoppingToken)
{
try
{
await dbService.UpdateArticleStatusAsync(article.Id, "Processing");
var (resolvedUrl, rawContent) = await scraperService.ScrapeArticleAsync(article.SourceUrl);
// 1. Playwright Headless Scrape & Redirect Resolution
var (resolvedUrl, scrapeResult) = await scraperService.ScrapeArticleAsync(article.SourceUrl);
if (!string.IsNullOrWhiteSpace(resolvedUrl) &&
!resolvedUrl.Equals(article.SourceUrl, StringComparison.OrdinalIgnoreCase))
// 2. Redirect Handling
if (!string.IsNullOrWhiteSpace(resolvedUrl) && !resolvedUrl.Equals(article.SourceUrl, StringComparison.OrdinalIgnoreCase))
{
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Redirect detected. Initial: {OldUrl} -> Resolved: {NewUrl}", article.SourceUrl, resolvedUrl);
if (await dbService.IsUrlDuplicateAsync(resolvedUrl))
if (_blocklistService.IsBlocked(resolvedUrl) || await dbService.IsUrlDuplicateAsync(resolvedUrl))
{
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Redirected URL {ResolvedUrl} is a duplicate. Terminating processing.", resolvedUrl);
await dbService.UpdateArticleStatusAsync(article.Id, "Failed");
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Redirected URL {Url} is already blocked/duplicate.", resolvedUrl);
await _blocklistService.BlockUrlAsync(article.SourceUrl, "RedirectToDuplicate", stoppingToken);
await dbService.DeleteArticleAsync(article.Id);
return;
}
@@ -222,66 +222,84 @@ public class NewsScraperBackgroundService : BackgroundService
article.SourceUrl = resolvedUrl;
}
if (string.IsNullOrWhiteSpace(rawContent) || rawContent.Length < 60)
// 3. Content Validity Check
var title = !string.IsNullOrWhiteSpace(scrapeResult.Title) ? scrapeResult.Title : article.Title;
var content = scrapeResult.TextContent;
if (string.IsNullOrWhiteSpace(content) || content.Length < 80 || string.IsNullOrWhiteSpace(title))
{
await dbService.UpdateArticleStatusAsync(article.Id, "Failed");
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Article {Id} has empty or insufficient content ({Length} chars). Adding to blocklist.", article.Id, content?.Length ?? 0);
await _blocklistService.BlockUrlAsync(article.SourceUrl, "EmptyOrInvalidContent", stoppingToken);
await dbService.DeleteArticleAsync(article.Id);
return;
}
var discoveredIsins = article.MatchedAssets.Select(m => m.Isin).Where(i => !string.IsNullOrEmpty(i)).ToList();
var preFilteredAssets = PreFilterAssets(rawContent, article.Title, assetMatchers, discoveredIsins);
var publishedAt = scrapeResult.PublishedAt ?? article.PublishedAt;
if (preFilteredAssets.Count == 0)
// 4. Non-AI Content & Title Deduplication Check (SimHash + N-Gram)
var dupCheck = _deduplicationService.CheckDuplicate(title, content, publishedAt);
if (dupCheck.IsDuplicate)
{
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Pre-filtering: Article {Id} does not reference any known assets. Terminating pipeline.", article.Id);
await dbService.UpdateArticleStatusAsync(article.Id, "Failed");
await _finlyticLogger.LogInfoAsync(SettingKeys.DeduplicationChannel, "[NewsScraperBackgroundService] Article {Id} identified as duplicate of {OriginalId} (Reason: {Reason}). Blocking URL.", article.Id, dupCheck.DuplicateOfArticleId?.ToString() ?? "Unknown", dupCheck.Reason ?? "Duplicate content");
await _blocklistService.BlockUrlAsync(article.SourceUrl, $"Duplicate:{dupCheck.DuplicateOfArticleId}", stoppingToken);
await dbService.DeleteArticleAsync(article.Id);
return;
}
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Pre-filtering matched {Count} assets for article {Id}.", preFilteredAssets.Count, article.Id);
var n8nResponse = await n8nService.AnalyzeArticleAsync(rawContent, preFilteredAssets, stoppingToken);
if (n8nResponse == null)
// 5. In-Memory Asset Matching (ISIN Regex, Name Tokenizer, Suffix Trimming)
var candidateMatches = _assetMatcherService.MatchAssets(title, content, article.SourceUrl);
if (candidateMatches.Count == 0)
{
await dbService.UpdateArticleStatusAsync(article.Id, "Failed");
await _finlyticLogger.LogInfoAsync(SettingKeys.MatcherChannel, "[NewsScraperBackgroundService] No candidate assets matched for article {Id} ('{Title}'). Blocking URL.", article.Id, title);
await _blocklistService.BlockUrlAsync(article.SourceUrl, "NoMatchedAssets", stoppingToken);
await dbService.DeleteArticleAsync(article.Id);
return;
}
var matchedEntities = new List<MatchedAssetEntity>();
if (n8nResponse.MatchedAssets != null && n8nResponse.MatchedAssets.Count > 0)
// 6. Multi-Criteria Asset Validation (Sector Clustering, Title Weighting, Existence)
var validatedMatches = await _assetValidationService.ValidateCandidateAssetsAsync(candidateMatches, title, content, stoppingToken);
if (validatedMatches.Count == 0)
{
foreach (var asset in n8nResponse.MatchedAssets)
{
var preMatch = preFilteredAssets.FirstOrDefault(p => p.Name.Equals(asset.Name, StringComparison.OrdinalIgnoreCase));
var isin = preMatch?.Isin ?? asset.Ticker ?? "";
if (string.IsNullOrWhiteSpace(isin)) continue;
matchedEntities.Add(new MatchedAssetEntity
{
Id = Guid.NewGuid(),
NewsArticleId = article.Id,
Isin = isin.Trim().ToUpperInvariant(),
Name = !string.IsNullOrWhiteSpace(asset.Name) ? asset.Name.Trim() : isin.Trim().ToUpperInvariant()
});
}
await _finlyticLogger.LogInfoAsync(SettingKeys.MatcherChannel, "[NewsScraperBackgroundService] All candidate matches pruned during validation for article {Id}. Blocking URL.", article.Id);
await _blocklistService.BlockUrlAsync(article.SourceUrl, "NoValidatedAssets", stoppingToken);
await dbService.DeleteArticleAsync(article.Id);
return;
}
if (matchedEntities.Count == 0)
// 7. Prepare Matched Asset Entities
var matchedEntities = validatedMatches.Select(m => new MatchedAssetEntity
{
foreach (var preMatch in preFilteredAssets)
{
matchedEntities.Add(new MatchedAssetEntity
{
Id = Guid.NewGuid(),
NewsArticleId = article.Id,
Isin = preMatch.Isin,
Name = preMatch.Name
});
}
Id = Guid.NewGuid(),
NewsArticleId = article.Id,
Isin = m.Isin,
Name = m.Name
}).ToList();
// 8. Generate Clean Summary & Excerpt
var summary = scrapeResult.Excerpt;
if (string.IsNullOrWhiteSpace(summary))
{
summary = content.Length > 280 ? content[..280] + "..." : content;
}
var updatedArticle = await dbService.SaveArticleClassificationAsync(article.Id, n8nResponse, matchedEntities);
// 9. Persist Completed Article
var updatedArticle = await dbService.SaveProcessedArticleAsync(
id: article.Id,
title: title,
author: scrapeResult.Author,
summary: summary,
contentRaw: content,
language: scrapeResult.Language ?? article.Language ?? "de",
publishedAt: publishedAt,
titleHash: dupCheck.TitleHash,
simHash: dupCheck.SimHash,
matchedAssets: matchedEntities
);
// 10. Register in deduplication memory cache
_deduplicationService.RegisterArticle(article.Id, dupCheck.TitleHash, dupCheck.SimHash, title, publishedAt);
// 11. Broadcast via MQTT
if (updatedArticle != null)
{
var dto = new NewsArticleDto
@@ -304,124 +322,13 @@ public class NewsScraperBackgroundService : BackgroundService
};
await _mqttClient.BroadcastArticleAsync(dto);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Successfully processed and broadcasted article {Id} ('{Title}') with {Count} matched assets.", article.Id, title, matchedEntities.Count);
}
}
catch (Exception ex)
{
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] Failed to complete processing pipeline for article: {Url}. Transitioning to 'Scraping' for next cycle retry.", article.SourceUrl);
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] Exception processing article {Url}. Flagging for next cycle retry.", article.SourceUrl);
await dbService.UpdateArticleStatusAsync(article.Id, "Scraping");
}
}
private List<FilteredAssetPayload> PreFilterAssets(
string content,
string? title,
List<CompiledAssetMatcher> assetMatchers,
List<string>? priorityIsins = null)
{
if (assetMatchers.Count == 0) return new List<FilteredAssetPayload>();
var fullText = (title != null ? title + " " + content : content);
var matched = new Dictionary<string, FilteredAssetPayload>(StringComparer.OrdinalIgnoreCase);
if (priorityIsins != null && priorityIsins.Count > 0)
{
foreach (var isin in priorityIsins)
{
var match = assetMatchers.FirstOrDefault(m => string.Equals(m.Asset.Isin, isin, StringComparison.OrdinalIgnoreCase));
if (match != null && !matched.ContainsKey(match.Asset.Isin))
{
matched[match.Asset.Isin] = new FilteredAssetPayload(match.Asset.Name, match.Asset.Isin);
}
}
}
foreach (var m in assetMatchers)
{
if (matched.ContainsKey(m.Asset.Isin)) continue;
if (fullText.Contains(m.Asset.Isin, StringComparison.OrdinalIgnoreCase))
{
matched[m.Asset.Isin] = new FilteredAssetPayload(m.Asset.Name, m.Asset.Isin);
continue;
}
if (m.WordRegex != null && m.WordRegex.IsMatch(fullText))
{
matched[m.Asset.Isin] = new FilteredAssetPayload(m.Asset.Name, m.Asset.Isin);
continue;
}
if (m.CoreWordRegex != null && m.CoreWordRegex.IsMatch(fullText))
{
matched[m.Asset.Isin] = new FilteredAssetPayload(m.Asset.Name, m.Asset.Isin);
}
}
return matched.Values.ToList();
}
private async Task<List<CompiledAssetMatcher>> GetOrLoadAssetMatchersAsync()
{
if (_cachedAssetMatchers != null && (DateTime.UtcNow - _lastIndexLoadTime).TotalMinutes < 60)
{
return _cachedAssetMatchers;
}
if (!File.Exists(_indexPath))
{
await _finlyticLogger.LogWarningAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Asset index file not found at: {Path}. Pre-filtering will match 0 assets.", _indexPath);
return new List<CompiledAssetMatcher>();
}
try
{
using var stream = File.OpenRead(_indexPath);
var indexList = await JsonSerializer.DeserializeAsync<List<AssetIndex>>(stream);
if (indexList == null || indexList.Count == 0)
{
return new List<CompiledAssetMatcher>();
}
var compiled = new List<CompiledAssetMatcher>(indexList.Count);
foreach (var asset in indexList)
{
if (string.IsNullOrWhiteSpace(asset.Name) || string.IsNullOrWhiteSpace(asset.Isin))
continue;
var rawName = asset.Name.Trim();
var coreName = ExtractCoreName(rawName);
Regex? wordRegex = null;
if (rawName.Length >= 4)
{
wordRegex = new Regex($@"\b{Regex.Escape(rawName)}\b", RegexOptions.IgnoreCase | RegexOptions.CultureInvariant | RegexOptions.Compiled);
}
Regex? coreWordRegex = null;
if (!string.IsNullOrWhiteSpace(coreName) && coreName.Length >= 4 && !coreName.Equals(rawName, StringComparison.OrdinalIgnoreCase))
{
coreWordRegex = new Regex($@"\b{Regex.Escape(coreName)}\b", RegexOptions.IgnoreCase | RegexOptions.CultureInvariant | RegexOptions.Compiled);
}
compiled.Add(new CompiledAssetMatcher(asset, coreName, wordRegex, coreWordRegex));
}
_cachedAssetMatchers = compiled;
_lastIndexLoadTime = DateTime.UtcNow;
return _cachedAssetMatchers;
}
catch (Exception ex)
{
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] Failed to read or parse asset index file from {Path}.", _indexPath);
return _cachedAssetMatchers ?? new List<CompiledAssetMatcher>();
}
}
private static string ExtractCoreName(string rawName)
{
var cleaned = Regex.Replace(rawName, @"\b(AG|SE|SA|NV|PLC|INC|CORP|LLC|GMBH|CO|KG|HOLDING|GROUP|CLASS\s+[A-Z])\b", "", RegexOptions.IgnoreCase);
return cleaned.Trim(' ', '.', ',', '-');
}
}