16bb195cb5
- Update CLAUDE.md: replace incorrect 'no XML doc on internal code' rule with the correct convention (XML doc on all public methods and non-trivial private/protected helpers) - Restore /// <summary> on FileDownloadController private helpers (HandleRangeRequest, StreamRangeAsync) - Add full XML doc to all service contracts: ICaptchaVerifier, IEmailSender, ICvMatcherService, IJobTextExtractor, IJobTokenService, IDocumentClassifier, IRagService, ITextChunker, ITextExtractor, IEmailTemplateService, ITemplateService - Add /// <summary> and /// <inheritdoc /> to all concrete service classes and their methods: RecaptchaVerifier, EmailApiEmailSender, SmtpEmailDispatcher, CvMatcherService, JobTextExtractor, JobTokenService, RagService, DocumentClassifier, TextChunker, TextExtractor, HtmlJobSearcher, CvSearchEmailSender, CvSearchJobTask, EmailTemplateService, DbTemplateService Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
102 lines
4.0 KiB
C#
102 lines
4.0 KiB
C#
using System.Text.RegularExpressions;
|
|
using System.Web;
|
|
using CvMatcher.Models.Settings;
|
|
using Microsoft.Extensions.Logging;
|
|
|
|
namespace CvSearchJob.Services;
|
|
|
|
/// <summary>
|
|
/// Config-driven HTML scraper that fetches a provider's job listing page and extracts matching job URLs.
|
|
/// Uses a two-stage anchor filter: href must contain the provider's link pattern, and anchor text must
|
|
/// contain at least one CV keyword.
|
|
/// </summary>
|
|
public sealed class HtmlJobSearcher
|
|
{
|
|
private readonly HttpClient _http;
|
|
private readonly ILogger<HtmlJobSearcher> _logger;
|
|
|
|
public HtmlJobSearcher(HttpClient http, ILogger<HtmlJobSearcher> logger)
|
|
{
|
|
_http = http;
|
|
_logger = logger;
|
|
_http.Timeout = TimeSpan.FromSeconds(20);
|
|
_http.DefaultRequestHeaders.UserAgent.ParseAdd("Mozilla/5.0 (compatible; MyAi.ro CV-Search/1.0)");
|
|
}
|
|
|
|
/// <summary>
|
|
/// Fetches the provider's search result page for the combined initial + CV keywords, parses all anchor
|
|
/// tags, applies the two-stage filter, and returns up to <see cref="JobProviderConfig.MaxResults"/> absolute URLs.
|
|
/// Returns an empty list when the HTTP request fails rather than throwing.
|
|
/// </summary>
|
|
/// <param name="provider">Provider configuration including search URL template, link filter, and result cap.</param>
|
|
/// <param name="cvKeywords">Keywords extracted from the user's CV to inject into the search query.</param>
|
|
/// <param name="ct">Cancellation token.</param>
|
|
/// <returns>Deduplicated list of absolute job page URLs (query string stripped).</returns>
|
|
public async Task<IReadOnlyList<string>> SearchJobUrlsAsync(
|
|
JobProviderConfig provider,
|
|
IReadOnlyList<string> cvKeywords,
|
|
CancellationToken ct)
|
|
{
|
|
var allKeywords = provider.InitialKeywords
|
|
.Concat(cvKeywords)
|
|
.Where(k => !string.IsNullOrWhiteSpace(k))
|
|
.Distinct(StringComparer.OrdinalIgnoreCase)
|
|
.ToList();
|
|
|
|
if (allKeywords.Count == 0)
|
|
return [];
|
|
|
|
var keywordsEncoded = HttpUtility.UrlEncode(string.Join(" ", allKeywords));
|
|
var searchUrl = provider.SearchUrlTemplate.Replace("{keywords}", keywordsEncoded);
|
|
|
|
string html;
|
|
try
|
|
{
|
|
html = await _http.GetStringAsync(searchUrl, ct);
|
|
}
|
|
catch (Exception ex)
|
|
{
|
|
_logger.LogWarning(ex, "Failed to fetch search results from {Provider} at {Url}", provider.Name, searchUrl);
|
|
return [];
|
|
}
|
|
|
|
var baseUri = new Uri(searchUrl);
|
|
var results = new List<string>();
|
|
var seen = new HashSet<string>(StringComparer.OrdinalIgnoreCase);
|
|
|
|
// Match all anchor tags capturing href and inner text
|
|
var anchorPattern = new Regex(@"<a[^>]+href=[""']([^""']+)[""'][^>]*>(.*?)</a>",
|
|
RegexOptions.IgnoreCase | RegexOptions.Singleline);
|
|
|
|
foreach (Match match in anchorPattern.Matches(html))
|
|
{
|
|
if (results.Count >= provider.MaxResults) break;
|
|
|
|
var href = match.Groups[1].Value.Trim();
|
|
var anchorText = Regex.Replace(match.Groups[2].Value, "<[^>]+>", " ").Trim();
|
|
|
|
if (!href.Contains(provider.JobLinkContains, StringComparison.OrdinalIgnoreCase))
|
|
continue;
|
|
|
|
// Stage 2: anchor text must contain at least one CV keyword
|
|
if (!cvKeywords.Any(k => anchorText.Contains(k, StringComparison.OrdinalIgnoreCase)))
|
|
continue;
|
|
|
|
// Make absolute URL
|
|
if (!Uri.TryCreate(href, UriKind.Absolute, out var absoluteUri))
|
|
{
|
|
if (!Uri.TryCreate(baseUri, href, out absoluteUri))
|
|
continue;
|
|
}
|
|
|
|
// Strip query string and fragment so different tracking variants of the same URL collapse to one.
|
|
var url = absoluteUri.GetLeftPart(UriPartial.Path);
|
|
if (seen.Add(url))
|
|
results.Add(url);
|
|
}
|
|
|
|
_logger.LogInformation("Provider {Provider}: found {Count} job URLs", provider.Name, results.Count);
|
|
return results;
|
|
}
|
|
}
|