feat(news): dynamic settings, IFinlyticLogger, live log streaming, and EF migration

This commit is contained in:
2026-08-15 21:30:01 +02:00
parent a1f2b888f6
commit a94c36a878
12 changed files with 800 additions and 555 deletions
@@ -1,18 +1,22 @@
using System;
using System.Collections.Concurrent;
using System.Collections.Generic;
using System.IO;
using System.Linq;
using System.Text.Json;
using System.Text.Json.Serialization;
using System.Text.RegularExpressions;
using System.Threading;
using System.Threading.Tasks;
using FinlyticAssets.Models;
using FinlyticAssets.Util;
using FinlyticCore.Dtos.News;
using FinlyticCore.Services;
using FinlyticCore.Util;
using FinlyticNews.Entities;
using FinlyticNews.Util;
using Microsoft.Extensions.Configuration;
using Microsoft.Extensions.DependencyInjection;
using Microsoft.Extensions.Hosting;
using Microsoft.Extensions.Logging;
namespace FinlyticNews.Services;
@@ -22,9 +26,6 @@ namespace FinlyticNews.Services;
/// </summary>
public class NewsScraperBackgroundService : BackgroundService
{
/// <summary>
/// Internal wrapper to associate compiled regex patterns with the unmodified AssetIndex record.
/// </summary>
private record CompiledAssetMatcher(
AssetIndex Asset,
string CoreName,
@@ -33,63 +34,63 @@ public class NewsScraperBackgroundService : BackgroundService
);
private readonly IServiceScopeFactory _scopeFactory;
private readonly ILogger<NewsScraperBackgroundService> _logger;
private readonly IFinlyticLogger<NewsScraperBackgroundService> _finlyticLogger;
private readonly NewsMqttClient _mqttClient;
private readonly int _intervalMinutes;
private readonly string _indexPath;
// In-Memory Cache for compiled asset matchers to prevent re-reading & re-compiling Regex
private List<CompiledAssetMatcher>? _cachedAssetMatchers;
private DateTime _lastIndexLoadTime = DateTime.MinValue;
public NewsScraperBackgroundService(
IServiceScopeFactory scopeFactory,
ILogger<NewsScraperBackgroundService> logger,
IFinlyticLogger<NewsScraperBackgroundService> finlyticLogger,
NewsMqttClient mqttClient,
IConfiguration configuration)
{
_scopeFactory = scopeFactory;
_logger = logger;
_finlyticLogger = finlyticLogger;
_mqttClient = mqttClient;
_intervalMinutes = configuration.GetValue<int>("ScrapingSettings:IntervalMinutes", 15);
_indexPath = Path.Combine(Volumes.IndexRelativePath, "index.json");
}
protected override async Task ExecuteAsync(CancellationToken stoppingToken)
{
_logger.LogInformation("[{Channel}] NewsScraperBackgroundService started. Interval: {Minutes} minutes.", "NewsChannel", _intervalMinutes);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] NewsScraperBackgroundService started.");
while (!stoppingToken.IsCancellationRequested)
{
try
{
await RunScrapingCycleAsync(stoppingToken);
using var scope = _scopeFactory.CreateScope();
var settings = scope.ServiceProvider.GetRequiredService<ISettingsService>();
bool enabled = await settings.GetSettingAsync(SettingKeys.EnableAutoScraping, stoppingToken);
if (enabled)
{
await RunScrapingCycleAsync(stoppingToken);
}
else
{
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Auto-scraping is disabled via settings.");
}
}
catch (Exception ex) when (ex is not OperationCanceledException)
{
_logger.LogError(ex, "[{Channel}] An unhandled exception occurred during news scraping cycle.", "NewsChannel");
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] An unhandled exception occurred during news scraping cycle.");
}
int intervalMinutes = _intervalMinutes;
int intervalMinutes = 15;
try
{
using var scope = _scopeFactory.CreateScope();
var settingsDb = scope.ServiceProvider.GetService<ISettingsDbService>();
if (settingsDb != null)
{
var settings = await settingsDb.GetSettingsAsync();
if (settings?.ScrapingIntervalMinutes > 0)
{
intervalMinutes = settings.ScrapingIntervalMinutes;
}
}
var settings = scope.ServiceProvider.GetRequiredService<ISettingsService>();
intervalMinutes = await settings.GetSettingAsync(SettingKeys.ScrapeIntervalMinutes, stoppingToken);
}
catch { /* Ignore settings DB lookup failures */ }
catch { }
var jitterSeconds = Random.Shared.Next(0, 300);
var jitterSeconds = Random.Shared.Next(0, 60);
var nextRunDelay = TimeSpan.FromMinutes(intervalMinutes) + TimeSpan.FromSeconds(jitterSeconds);
_logger.LogInformation("[{Channel}] Scraping cycle completed. Next cycle in {Delay} (interval: {Minutes}m).", "NewsChannel", nextRunDelay, intervalMinutes);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Scraping cycle completed. Next cycle in {Delay} (interval: {Minutes}m).", nextRunDelay, intervalMinutes);
try
{
@@ -101,7 +102,7 @@ public class NewsScraperBackgroundService : BackgroundService
}
}
_logger.LogInformation("[{Channel}] NewsScraperBackgroundService stopping.", "NewsChannel");
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] NewsScraperBackgroundService stopping.");
}
private async Task RunScrapingCycleAsync(CancellationToken stoppingToken)
@@ -111,28 +112,28 @@ public class NewsScraperBackgroundService : BackgroundService
var discoveryService = scope.ServiceProvider.GetRequiredService<IArticleDiscoveryService>();
var scraperService = scope.ServiceProvider.GetRequiredService<IPlaywrightScraperService>();
var n8nService = scope.ServiceProvider.GetRequiredService<IN8nService>();
var settings = scope.ServiceProvider.GetRequiredService<ISettingsService>();
var maxArticlesPerFeed = await settings.GetSettingAsync(SettingKeys.MaxArticlesPerFeed, stoppingToken);
// Load pre-compiled asset index matchers for zero-latency pre-filtering
var assetMatchers = await GetOrLoadAssetMatchersAsync();
_logger.LogInformation("[{Channel}] Loaded {Count} asset index items for text pre-filtering.", "NewsChannel", assetMatchers.Count);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Loaded {Count} asset index items for text pre-filtering.", assetMatchers.Count);
// 1. Scraping Retry Phase: query articles in status "Scraping" (failed Playwright runs)
var failedArticles = await dbService.GetArticlesByStatusAsync("Scraping");
if (failedArticles.Count > 0)
{
_logger.LogInformation("[{Channel}] Found {Count} articles in status 'Scraping' that failed to scrape previously. Retrying...", "NewsChannel", failedArticles.Count);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Found {Count} articles in status 'Scraping' that failed to scrape previously. Retrying...", failedArticles.Count);
foreach (var article in failedArticles)
{
if (stoppingToken.IsCancellationRequested) break;
await ProcessSingleArticleAsync(article, dbService, scraperService, n8nService, assetMatchers, stoppingToken);
if (stoppingToken.IsCancellationRequested) return;
await ProcessSingleArticleAsync(article, scraperService, n8nService, dbService, assetMatchers, stoppingToken);
}
}
// 2. Link Discovery Phase: query RSS feeds and listing pages
var sources = await dbService.GetSourcesAsync();
if (sources.Count == 0)
{
_logger.LogWarning("[{Channel}] No article sources configured in database. Skipping cycle.", "NewsChannel");
await _finlyticLogger.LogWarningAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] No article sources configured in database. Skipping cycle.");
return;
}
@@ -140,86 +141,80 @@ public class NewsScraperBackgroundService : BackgroundService
{
if (stoppingToken.IsCancellationRequested) break;
_logger.LogInformation("[{Channel}] Starting article link discovery for source: {SourceName} ({Url})", "NewsChannel", source.Name, source.Source);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Starting article link discovery for source: {SourceName} ({Url})", source.Name, source.Source);
var discoveredArticles = await discoveryService.DiscoverLinksAsync(source.Source, source.Type, stoppingToken);
if (discoveredArticles == null || discoveredArticles.Count == 0)
{
_logger.LogDebug("No links discovered from source: {SourceName}", source.Name);
await _finlyticLogger.LogDebugAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] No links discovered from source: {SourceName}", source.Name);
continue;
}
_logger.LogInformation("[{Channel}] Discovered {Count} potential article links from {SourceName}.", "NewsChannel", discoveredArticles.Count, source.Name);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Discovered {Count} potential article links from {SourceName}.", discoveredArticles.Count, source.Name);
foreach (var discovered in discoveredArticles)
var toProcess = discoveredArticles.Take(maxArticlesPerFeed > 0 ? maxArticlesPerFeed : 20);
foreach (var discovered in toProcess)
{
if (stoppingToken.IsCancellationRequested) break;
// Idempotency Check & Deduplication
var isDuplicate = await dbService.IsUrlDuplicateAsync(discovered.Url);
if (isDuplicate)
if (await dbService.IsUrlDuplicateAsync(discovered.Url))
{
_logger.LogDebug("Skipping duplicate article URL: {Url}", discovered.Url);
await _finlyticLogger.LogDebugAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Skipping duplicate article URL: {Url}", discovered.Url);
continue;
}
// Register Initial Lock State in the database ("Pending")
NewsArticleEntity? article;
NewsArticleEntity article;
try
{
article = await dbService.CreatePendingArticleAsync(
discovered.Url,
discovered.Isins,
discovered.Title,
discovered.Summary,
discovered.PublishedAt,
discovered.Language);
discovered.Url,
discovered.Isins,
discovered.Title,
discovered.Summary,
discovered.PublishedAt,
discovered.Language
);
}
catch (Exception ex)
{
_logger.LogError(ex, "[{Channel}] Failed to register initial pending state for URL: {Url}. Skipping.", "NewsChannel", discovered.Url);
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] Failed to register initial pending state for URL: {Url}. Skipping.", discovered.Url);
continue;
}
if (article == null || article.Id == Guid.Empty)
if (article.Id == Guid.Empty)
{
_logger.LogWarning("[{Channel}] Created pending article has invalid/empty ID for URL: {Url}. Skipping.", "NewsChannel", discovered.Url);
await _finlyticLogger.LogWarningAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Created pending article has invalid/empty ID for URL: {Url}. Skipping.", discovered.Url);
continue;
}
// Process single article pipeline
await ProcessSingleArticleAsync(article, dbService, scraperService, n8nService, assetMatchers, stoppingToken);
await ProcessSingleArticleAsync(article, scraperService, n8nService, dbService, assetMatchers, stoppingToken);
}
}
}
private async Task ProcessSingleArticleAsync(
NewsArticleEntity article,
INewsDbService dbService,
IPlaywrightScraperService scraperService,
IN8nService n8nService,
INewsDbService dbService,
List<CompiledAssetMatcher> assetMatchers,
CancellationToken stoppingToken)
{
try
{
// 3. Extraction with Headless Browser (Transitions to "Processing")
await dbService.UpdateArticleStatusAsync(article.Id, "Processing");
var (resolvedUrl, rawText) = await scraperService.ScrapeArticleAsync(article.SourceUrl);
if (string.IsNullOrWhiteSpace(rawText))
{
throw new InvalidOperationException("Scraping returned empty text body content.");
}
var (resolvedUrl, rawContent) = await scraperService.ScrapeArticleAsync(article.SourceUrl);
// Update resolved URL if redirect occurred
if (!string.Equals(resolvedUrl, article.SourceUrl, StringComparison.OrdinalIgnoreCase))
if (!string.IsNullOrWhiteSpace(resolvedUrl) &&
!resolvedUrl.Equals(article.SourceUrl, StringComparison.OrdinalIgnoreCase))
{
_logger.LogInformation("[{Channel}] Redirect detected. Initial: {OldUrl} -> Resolved: {NewUrl}", "NewsChannel", article.SourceUrl, resolvedUrl);
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Redirect detected. Initial: {OldUrl} -> Resolved: {NewUrl}", article.SourceUrl, resolvedUrl);
if (await dbService.IsUrlDuplicateAsync(resolvedUrl))
{
_logger.LogInformation("[{Channel}] Redirected URL {ResolvedUrl} is a duplicate. Terminating processing.", "NewsChannel", resolvedUrl);
await dbService.UpdateArticleStatusAsync(article.Id, "Duplicate");
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Redirected URL {ResolvedUrl} is a duplicate. Terminating processing.", resolvedUrl);
await dbService.UpdateArticleStatusAsync(article.Id, "Failed");
return;
}
@@ -227,108 +222,85 @@ public class NewsScraperBackgroundService : BackgroundService
article.SourceUrl = resolvedUrl;
}
// 4. Pre-filtering Assets (Optimized with Pre-Compiled Regex Patterns)
var title = article.Title ?? string.Empty;
var summary = article.Summary ?? string.Empty;
var preFilteredAssets = assetMatchers.Where(matcher =>
if (string.IsNullOrWhiteSpace(rawContent) || rawContent.Length < 60)
{
var asset = matcher.Asset;
// ISIN direct match
if (rawText.Contains(asset.Isin, StringComparison.OrdinalIgnoreCase) ||
title.Contains(asset.Isin, StringComparison.OrdinalIgnoreCase) ||
summary.Contains(asset.Isin, StringComparison.OrdinalIgnoreCase))
{
return true;
}
// Fast regex word boundary check on Full Name
if (matcher.WordRegex != null && (matcher.WordRegex.IsMatch(rawText) || matcher.WordRegex.IsMatch(title)))
{
return true;
}
// Fast regex word boundary check on Core Name
if (matcher.CoreName.Length >= 3 && matcher.CoreWordRegex != null &&
(matcher.CoreWordRegex.IsMatch(rawText) || matcher.CoreWordRegex.IsMatch(title)))
{
return true;
}
return false;
})
.Select(m => new FilteredAssetPayload(m.Asset.Name, m.Asset.Isin))
.ToList();
if (preFilteredAssets.Count == 0)
{
_logger.LogInformation("[{Channel}] Pre-filtering: Article {Id} does not reference any known assets. Terminating pipeline.", "NewsChannel", article.Id);
await dbService.UpdateArticleStatusAsync(article.Id, "Failed");
return;
}
_logger.LogInformation("[{Channel}] Pre-filtering matched {Count} assets for article {Id}.", "NewsChannel", preFilteredAssets.Count, article.Id);
var discoveredIsins = article.MatchedAssets.Select(m => m.Isin).Where(i => !string.IsNullOrEmpty(i)).ToList();
var preFilteredAssets = PreFilterAssets(rawContent, article.Title, assetMatchers, discoveredIsins);
// 5. Send to n8n Webhook Pipeline
var n8nResponse = await n8nService.AnalyzeArticleAsync(rawText, preFilteredAssets, stoppingToken);
if (n8nResponse == null)
if (preFilteredAssets.Count == 0)
{
throw new InvalidOperationException("n8n AI webhook execution returned null or failed.");
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Pre-filtering: Article {Id} does not reference any known assets. Terminating pipeline.", article.Id);
await dbService.UpdateArticleStatusAsync(article.Id, "Failed");
return;
}
// 6. Map and Save Completed Classification
var matchedEntities = new List<MatchedAssetEntity>();
foreach (var n8nAsset in n8nResponse.MatchedAssets)
await _finlyticLogger.LogInfoAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Pre-filtering matched {Count} assets for article {Id}.", preFilteredAssets.Count, article.Id);
var n8nResponse = await n8nService.AnalyzeArticleAsync(rawContent, preFilteredAssets, stoppingToken);
if (n8nResponse == null)
{
var n8nCoreName = ExtractCoreAssetName(n8nAsset.Name);
await dbService.UpdateArticleStatusAsync(article.Id, "Failed");
return;
}
var matchedIsin = preFilteredAssets.FirstOrDefault(fa =>
fa.Name.Equals(n8nAsset.Name, StringComparison.OrdinalIgnoreCase) ||
n8nAsset.Name.Contains(fa.Name, StringComparison.OrdinalIgnoreCase) ||
(n8nCoreName.Length >= 3 && ExtractCoreAssetName(fa.Name).Equals(n8nCoreName, StringComparison.OrdinalIgnoreCase)))?.Isin;
if (string.IsNullOrWhiteSpace(matchedIsin))
var matchedEntities = new List<MatchedAssetEntity>();
if (n8nResponse.MatchedAssets != null && n8nResponse.MatchedAssets.Count > 0)
{
foreach (var asset in n8nResponse.MatchedAssets)
{
matchedIsin = assetMatchers.FirstOrDefault(m =>
m.Asset.Name.Equals(n8nAsset.Name, StringComparison.OrdinalIgnoreCase) ||
n8nAsset.Name.Contains(m.Asset.Name, StringComparison.OrdinalIgnoreCase) ||
(n8nCoreName.Length >= 3 && m.CoreName.Equals(n8nCoreName, StringComparison.OrdinalIgnoreCase)))?.Asset.Isin;
}
var preMatch = preFilteredAssets.FirstOrDefault(p => p.Name.Equals(asset.Name, StringComparison.OrdinalIgnoreCase));
var isin = preMatch?.Isin ?? asset.Ticker ?? "";
if (string.IsNullOrWhiteSpace(isin)) continue;
if (!string.IsNullOrWhiteSpace(matchedIsin))
{
matchedEntities.Add(new MatchedAssetEntity
{
Id = Guid.NewGuid(),
NewsArticleId = article.Id,
Name = n8nAsset.Name,
Isin = matchedIsin
Isin = isin.Trim().ToUpperInvariant(),
Name = !string.IsNullOrWhiteSpace(asset.Name) ? asset.Name.Trim() : isin.Trim().ToUpperInvariant()
});
}
}
var completedArticle = await dbService.SaveArticleClassificationAsync(article.Id, n8nResponse, matchedEntities);
if (completedArticle != null)
if (matchedEntities.Count == 0)
{
foreach (var preMatch in preFilteredAssets)
{
matchedEntities.Add(new MatchedAssetEntity
{
Id = Guid.NewGuid(),
NewsArticleId = article.Id,
Isin = preMatch.Isin,
Name = preMatch.Name
});
}
}
var updatedArticle = await dbService.SaveArticleClassificationAsync(article.Id, n8nResponse, matchedEntities);
if (updatedArticle != null)
{
// 7. MQTT Broadcast (Sends completed article to downstream services)
var dto = new NewsArticleDto
{
Id = completedArticle.Id,
Title = completedArticle.Title,
Author = completedArticle.Author,
Summary = completedArticle.Summary,
ContentRaw = completedArticle.ContentRaw,
Language = completedArticle.Language,
SourceUrl = completedArticle.SourceUrl,
ScrapedAt = completedArticle.ScrapedAt,
PublishedAt = completedArticle.PublishedAt,
MatchedAssets = completedArticle.MatchedAssets.Select(m => new MatchedAssetDto
Id = updatedArticle.Id,
Title = updatedArticle.Title,
Author = updatedArticle.Author,
Summary = updatedArticle.Summary,
ContentRaw = updatedArticle.ContentRaw,
Language = updatedArticle.Language,
SourceUrl = updatedArticle.SourceUrl,
ScrapedAt = updatedArticle.ScrapedAt,
PublishedAt = updatedArticle.PublishedAt,
Status = updatedArticle.Status,
MatchedAssets = updatedArticle.MatchedAssets.Select(m => new MatchedAssetDto
{
Name = m.Name,
Isin = m.Isin
}).ToList(),
Status = completedArticle.Status
}).ToList()
};
await _mqttClient.BroadcastArticleAsync(dto);
@@ -336,103 +308,120 @@ public class NewsScraperBackgroundService : BackgroundService
}
catch (Exception ex)
{
_logger.LogError(ex, "[{Channel}] Failed to complete processing pipeline for article: {Url}. Transitioning to 'Scraping' for next cycle retry.", "NewsChannel", article.SourceUrl);
try
{
await dbService.UpdateArticleStatusAsync(article.Id, "Scraping");
}
catch { /* Suppress database secondary errors */ }
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] Failed to complete processing pipeline for article: {Url}. Transitioning to 'Scraping' for next cycle retry.", article.SourceUrl);
await dbService.UpdateArticleStatusAsync(article.Id, "Scraping");
}
}
/// <summary>
/// Returns cached compiled asset matchers or parses the index file from disk if stale/missing.
/// </summary>
private List<FilteredAssetPayload> PreFilterAssets(
string content,
string? title,
List<CompiledAssetMatcher> assetMatchers,
List<string>? priorityIsins = null)
{
if (assetMatchers.Count == 0) return new List<FilteredAssetPayload>();
var fullText = (title != null ? title + " " + content : content);
var matched = new Dictionary<string, FilteredAssetPayload>(StringComparer.OrdinalIgnoreCase);
if (priorityIsins != null && priorityIsins.Count > 0)
{
foreach (var isin in priorityIsins)
{
var match = assetMatchers.FirstOrDefault(m => string.Equals(m.Asset.Isin, isin, StringComparison.OrdinalIgnoreCase));
if (match != null && !matched.ContainsKey(match.Asset.Isin))
{
matched[match.Asset.Isin] = new FilteredAssetPayload(match.Asset.Name, match.Asset.Isin);
}
}
}
foreach (var m in assetMatchers)
{
if (matched.ContainsKey(m.Asset.Isin)) continue;
if (fullText.Contains(m.Asset.Isin, StringComparison.OrdinalIgnoreCase))
{
matched[m.Asset.Isin] = new FilteredAssetPayload(m.Asset.Name, m.Asset.Isin);
continue;
}
if (m.WordRegex != null && m.WordRegex.IsMatch(fullText))
{
matched[m.Asset.Isin] = new FilteredAssetPayload(m.Asset.Name, m.Asset.Isin);
continue;
}
if (m.CoreWordRegex != null && m.CoreWordRegex.IsMatch(fullText))
{
matched[m.Asset.Isin] = new FilteredAssetPayload(m.Asset.Name, m.Asset.Isin);
}
}
return matched.Values.ToList();
}
private async Task<List<CompiledAssetMatcher>> GetOrLoadAssetMatchersAsync()
{
if (_cachedAssetMatchers != null && (DateTime.UtcNow - _lastIndexLoadTime).TotalMinutes < 30)
if (_cachedAssetMatchers != null && (DateTime.UtcNow - _lastIndexLoadTime).TotalMinutes < 60)
{
return _cachedAssetMatchers;
}
if (!File.Exists(_indexPath))
{
_logger.LogWarning("[{Channel}] Asset index file not found at: {Path}. Pre-filtering will match 0 assets.", "NewsChannel", _indexPath);
return [];
await _finlyticLogger.LogWarningAsync(SettingKeys.NewsChannel, "[NewsScraperBackgroundService] Asset index file not found at: {Path}. Pre-filtering will match 0 assets.", _indexPath);
return new List<CompiledAssetMatcher>();
}
try
{
await using var stream = File.OpenRead(_indexPath);
// Standard Deserialization for AssetIndex list
var rawList = await JsonSerializer.DeserializeAsync<List<AssetIndex>>(stream);
using var stream = File.OpenRead(_indexPath);
var indexList = await JsonSerializer.DeserializeAsync<List<AssetIndex>>(stream);
if (rawList != null)
if (indexList == null || indexList.Count == 0)
{
_cachedAssetMatchers = rawList.Select(asset =>
{
var coreName = ExtractCoreAssetName(asset.Name);
return new CompiledAssetMatcher(
Asset: asset,
CoreName: coreName,
WordRegex: BuildWordRegex(asset.Name),
CoreWordRegex: BuildWordRegex(coreName)
);
}).ToList();
_lastIndexLoadTime = DateTime.UtcNow;
return _cachedAssetMatchers;
return new List<CompiledAssetMatcher>();
}
var compiled = new List<CompiledAssetMatcher>(indexList.Count);
foreach (var asset in indexList)
{
if (string.IsNullOrWhiteSpace(asset.Name) || string.IsNullOrWhiteSpace(asset.Isin))
continue;
var rawName = asset.Name.Trim();
var coreName = ExtractCoreName(rawName);
Regex? wordRegex = null;
if (rawName.Length >= 4)
{
wordRegex = new Regex($@"\b{Regex.Escape(rawName)}\b", RegexOptions.IgnoreCase | RegexOptions.CultureInvariant | RegexOptions.Compiled);
}
Regex? coreWordRegex = null;
if (!string.IsNullOrWhiteSpace(coreName) && coreName.Length >= 4 && !coreName.Equals(rawName, StringComparison.OrdinalIgnoreCase))
{
coreWordRegex = new Regex($@"\b{Regex.Escape(coreName)}\b", RegexOptions.IgnoreCase | RegexOptions.CultureInvariant | RegexOptions.Compiled);
}
compiled.Add(new CompiledAssetMatcher(asset, coreName, wordRegex, coreWordRegex));
}
_cachedAssetMatchers = compiled;
_lastIndexLoadTime = DateTime.UtcNow;
return _cachedAssetMatchers;
}
catch (Exception ex)
{
_logger.LogError(ex, "[{Channel}] Failed to read or parse asset index file from {Path}.", "NewsChannel", _indexPath);
}
return _cachedAssetMatchers ?? [];
}
/// <summary>
/// Helper to pre-compile Word Boundary Regex for an asset name.
/// </summary>
private static Regex? BuildWordRegex(string name)
{
if (string.IsNullOrWhiteSpace(name)) return null;
try
{
return new Regex($@"\b{Regex.Escape(name)}\b", RegexOptions.IgnoreCase | RegexOptions.CultureInvariant | RegexOptions.Compiled);
}
catch
{
return null;
await _finlyticLogger.LogErrorAsync(SettingKeys.NewsChannel, ex, "[NewsScraperBackgroundService] Failed to read or parse asset index file from {Path}.", _indexPath);
return _cachedAssetMatchers ?? new List<CompiledAssetMatcher>();
}
}
/// <summary>
/// Extracts the core name of an asset by removing parenthetical metadata and corporate suffixes.
/// </summary>
private static string ExtractCoreAssetName(string name)
private static string ExtractCoreName(string rawName)
{
if (string.IsNullOrWhiteSpace(name)) return string.Empty;
int parenIndex = name.IndexOf('(');
if (parenIndex >= 0)
{
name = name[..parenIndex];
}
name = name.Trim();
var suffixes = new[] { "Inc.", "Inc", "AG", "SE", "Co.", "Co", "Corp.", "Corp", "Ltd.", "Ltd", "plc", "GmbH", "SA", "NV", "Group" };
foreach (var suffix in suffixes)
{
if (name.EndsWith(" " + suffix, StringComparison.OrdinalIgnoreCase))
{
name = name[..^suffix.Length].Trim();
}
}
return name;
var cleaned = Regex.Replace(rawName, @"\b(AG|SE|SA|NV|PLC|INC|CORP|LLC|GMBH|CO|KG|HOLDING|GROUP|CLASS\s+[A-Z])\b", "", RegexOptions.IgnoreCase);
return cleaned.Trim(' ', '.', ',', '-');
}
}