using System.Text.RegularExpressions; using System.Web; using CvMatcher.Models.Settings; using Microsoft.Extensions.Logging; namespace CvSearchJob.Services; /// /// Config-driven HTML scraper that fetches a provider's job listing page and extracts matching job URLs. /// Uses a two-stage anchor filter: href must contain the provider's link pattern, and anchor text must /// contain at least one CV keyword. /// public sealed class HtmlJobSearcher { private readonly HttpClient _http; private readonly ILogger _logger; public HtmlJobSearcher(HttpClient http, ILogger logger) { _http = http; _logger = logger; _http.Timeout = TimeSpan.FromSeconds(20); _http.DefaultRequestHeaders.UserAgent.ParseAdd("Mozilla/5.0 (compatible; MyAi.ro CV-Search/1.0)"); } /// /// Fetches the provider's search result page for the combined initial + CV keywords, parses all anchor /// tags, applies the two-stage filter, and returns up to absolute URLs. /// Returns an empty list when the HTTP request fails rather than throwing. /// /// Provider configuration including search URL template, link filter, and result cap. /// Keywords extracted from the user's CV to inject into the search query. /// Cancellation token. /// Deduplicated list of absolute job page URLs (query string stripped). public async Task> SearchJobUrlsAsync( JobProviderConfig provider, IReadOnlyList cvKeywords, CancellationToken ct) { var allKeywords = provider.InitialKeywords .Concat(cvKeywords) .Where(k => !string.IsNullOrWhiteSpace(k)) .Distinct(StringComparer.OrdinalIgnoreCase) .ToList(); if (allKeywords.Count == 0) return []; var keywordsEncoded = HttpUtility.UrlEncode(string.Join(" ", allKeywords)); var searchUrl = provider.SearchUrlTemplate.Replace("{keywords}", keywordsEncoded); string html; try { html = await _http.GetStringAsync(searchUrl, ct); } catch (Exception ex) { _logger.LogWarning(ex, "Failed to fetch search results from {Provider} at {Url}", provider.Name, searchUrl); return []; } var baseUri = new Uri(searchUrl); var results = new List(); var seen = new HashSet(StringComparer.OrdinalIgnoreCase); // Match all anchor tags capturing href and inner text var anchorPattern = new Regex(@"]+href=[""']([^""']+)[""'][^>]*>(.*?)", RegexOptions.IgnoreCase | RegexOptions.Singleline); foreach (Match match in anchorPattern.Matches(html)) { if (results.Count >= provider.MaxResults) break; var href = match.Groups[1].Value.Trim(); var anchorText = Regex.Replace(match.Groups[2].Value, "<[^>]+>", " ").Trim(); if (!href.Contains(provider.JobLinkContains, StringComparison.OrdinalIgnoreCase)) continue; // Stage 2: anchor text must contain at least one CV keyword if (!cvKeywords.Any(k => anchorText.Contains(k, StringComparison.OrdinalIgnoreCase))) continue; // Make absolute URL if (!Uri.TryCreate(href, UriKind.Absolute, out var absoluteUri)) { if (!Uri.TryCreate(baseUri, href, out absoluteUri)) continue; } // Strip query string and fragment so different tracking variants of the same URL collapse to one. var url = absoluteUri.GetLeftPart(UriPartial.Path); if (seen.Add(url)) results.Add(url); } _logger.LogInformation("Provider {Provider}: found {Count} job URLs", provider.Name, results.Count); return results; } }