e95ed36647
Phases 1-10 of the planned refactoring:
Phase 1: rename shared-models -> common
- namespace Shared.Models -> Common throughout
- remove stale AspNetCore.Http.Features 5.0 reference
Phase 2: create shared-data with abstract BaseEntity
- BaseEntity: required string Id { get; init; } + DateTime CreatedAt { get; init; }
Phase 3: rename myai-models -> myai-data
- namespace MyAi.Models -> MyAi.Data
- MigrationsAssembly("myai-data")
Phase 4: rename cv-search-models -> cv-search-data
- namespace CvSearch.Models -> CvSearch.Data
- move JobSearchSettings to cv-matcher-api-models
- JobSearch*Entity now inherits BaseEntity
Phase 5: extract rag-data from rag-api
- new project: Apis/rag-data with RagDbContext + entities + migrations
- RagDocumentEntity inherits BaseEntity; cache entities use CacheKey PK
- fix duplicate AddHttpClient<RagAiClient>/AddScoped registrations in rag-api
- MigrationsAssembly("rag-data")
Phase 6: extract cv-matcher-data from cv-matcher-api
- new project: Apis/cv-matcher-data with CvMatcherDbContext + entities + migrations
- CvMatchResultEntity inherits BaseEntity; CvMatcherChatCacheEntity uses CacheKey PK
- MigrationsAssembly("cv-matcher-data")
Phase 7: create empty cv-cleanup-job-models and cv-search-job-models
Phase 8: update all 5 Dockerfiles for renamed/new projects
Phase 9: reorganise .sln virtual folders (Apis/Jobs/Models/Data/Helpers)
- update root CLAUDE.md with new project taxonomy and migration commands
- update cv-matcher-api/CLAUDE.md and cv-search-job/CLAUDE.md
Phase 10: add Directory.Packages.props for centralised NuGet versions
- remove Version= from all PackageReference elements in active .csproj files
No database changes. No runtime behaviour changes.
All MigrationId strings in __EFMigrationsHistory are unaffected.
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
174 lines
9.5 KiB
C#
174 lines
9.5 KiB
C#
using Microsoft.AspNetCore.Mvc;
|
|
using Api.Services.Contracts;
|
|
using Rag.Models.Requests;
|
|
using Rag.Models.Responses;
|
|
using Swashbuckle.AspNetCore.Annotations;
|
|
using Common.Responses;
|
|
|
|
namespace Api.Controllers;
|
|
|
|
/// <summary>
|
|
/// Internal endpoints for indexing documents into the vector store and performing semantic search.
|
|
/// Routes are prefixed with <c>api/rag</c>. Protected by the internal API key middleware — not reachable from the public internet.
|
|
/// </summary>
|
|
[ApiController]
|
|
[Route("api/rag")]
|
|
public sealed class RagController : ControllerBase
|
|
{
|
|
private readonly IRagService _ragService;
|
|
private readonly ILogger<RagController> _logger;
|
|
|
|
public RagController(IRagService ragService, ILogger<RagController> logger)
|
|
{
|
|
_ragService = ragService;
|
|
_logger = logger;
|
|
}
|
|
|
|
/// <summary>
|
|
/// Indexes a PDF file or plain-text document into the vector store via multipart/form-data.
|
|
/// Chunks the content, generates embeddings, and stores them for semantic retrieval.
|
|
/// Returns immediately from cache if an identical document was previously indexed.
|
|
/// </summary>
|
|
/// <param name="request">The indexing request: either a PDF file or raw text, plus optional title, source URL, and document type.</param>
|
|
/// <param name="ct">Cancellation token.</param>
|
|
/// <returns>
|
|
/// 200 OK with an <see cref="IndexDocumentResponse"/> containing the document ID, chunk count, and cache status;
|
|
/// 400 Bad Request if neither a file nor text is provided, or the request is otherwise invalid.
|
|
/// </returns>
|
|
[HttpPost("documents")]
|
|
[RequestSizeLimit(10 * 1024 * 1024)]
|
|
[SwaggerOperation(Summary = "Index document (multipart)", Description = "Indexes a PDF or plain-text document via multipart/form-data. Returns from cache if the same content was previously indexed.")]
|
|
[SwaggerResponse(StatusCodes.Status200OK, "Document indexed successfully", typeof(IndexDocumentResponse))]
|
|
[SwaggerResponse(StatusCodes.Status400BadRequest, "Neither file nor text provided, or request is invalid", typeof(ErrorResponse))]
|
|
[ProducesResponseType(StatusCodes.Status200OK)]
|
|
[ProducesResponseType(typeof(ErrorResponse), StatusCodes.Status400BadRequest)]
|
|
public async Task<ActionResult<IndexDocumentResponse>> IndexDocument(
|
|
[FromForm] IndexDocumentUploadRequest request,
|
|
CancellationToken ct)
|
|
{
|
|
try
|
|
{
|
|
_logger.LogInformation("Index document request received. HasFile={HasFile}, DocumentType={DocumentType}, Title={Title}, SourceUrl={SourceUrl}",
|
|
request.File is not null, request.DocumentType, request.Title, request.SourceUrl);
|
|
|
|
if (request.File is not null)
|
|
{
|
|
var result = await _ragService.IndexPdfAsync(request.File, request.DocumentType, request.Title, request.SourceUrl, ct);
|
|
_logger.LogInformation("Indexed PDF document. DocumentId={DocumentId}, DocumentType={DocumentType}, Chunks={Chunks}, Cached={Cached}",
|
|
result.DocumentId, result.DocumentType, result.Chunks, result.Cached);
|
|
return Ok(result);
|
|
}
|
|
|
|
var textResult = await _ragService.IndexTextAsync(new IndexDocumentRequest
|
|
{
|
|
Text = request.Text,
|
|
DocumentType = request.DocumentType,
|
|
Title = request.Title,
|
|
SourceUrl = request.SourceUrl
|
|
}, ct);
|
|
_logger.LogInformation("Indexed text document. DocumentId={DocumentId}, DocumentType={DocumentType}, Chunks={Chunks}, Cached={Cached}",
|
|
textResult.DocumentId, textResult.DocumentType, textResult.Chunks, textResult.Cached);
|
|
return Ok(textResult);
|
|
}
|
|
catch (InvalidOperationException ex)
|
|
{
|
|
_logger.LogWarning(ex, "Invalid document indexing request.");
|
|
return BadRequest(new ErrorResponse { Error = ex.Message, Code = "invalid_request" });
|
|
}
|
|
}
|
|
|
|
/// <summary>
|
|
/// Indexes a plain-text document sent as JSON into the vector store.
|
|
/// Returns immediately from cache if an identical document was previously indexed.
|
|
/// </summary>
|
|
/// <param name="request">The indexing request containing the raw text and optional title, source URL, and document type.</param>
|
|
/// <param name="ct">Cancellation token.</param>
|
|
/// <returns>
|
|
/// 200 OK with an <see cref="IndexDocumentResponse"/> containing the document ID, chunk count, and cache status;
|
|
/// 400 Bad Request if the text is empty or the request is otherwise invalid.
|
|
/// </returns>
|
|
[HttpPost("documents/json")]
|
|
[SwaggerOperation(Summary = "Index document (JSON)", Description = "Indexes a plain-text document sent as JSON. Returns from cache if the same content was previously indexed.")]
|
|
[SwaggerResponse(StatusCodes.Status200OK, "Document indexed successfully", typeof(IndexDocumentResponse))]
|
|
[SwaggerResponse(StatusCodes.Status400BadRequest, "Text missing or request invalid", typeof(ErrorResponse))]
|
|
[ProducesResponseType(StatusCodes.Status200OK)]
|
|
[ProducesResponseType(typeof(ErrorResponse), StatusCodes.Status400BadRequest)]
|
|
public async Task<ActionResult<IndexDocumentResponse>> IndexJsonDocument([FromBody] IndexDocumentRequest request, CancellationToken ct)
|
|
{
|
|
try
|
|
{
|
|
_logger.LogInformation("JSON document indexing request received. DocumentType={DocumentType}, Title={Title}, SourceUrl={SourceUrl}",
|
|
request.DocumentType, request.Title, request.SourceUrl);
|
|
var result = await _ragService.IndexTextAsync(request, ct);
|
|
_logger.LogInformation("Indexed JSON document. DocumentId={DocumentId}, DocumentType={DocumentType}, Chunks={Chunks}, Cached={Cached}",
|
|
result.DocumentId, result.DocumentType, result.Chunks, result.Cached);
|
|
return Ok(result);
|
|
}
|
|
catch (InvalidOperationException ex)
|
|
{
|
|
_logger.LogWarning(ex, "Invalid JSON document indexing request.");
|
|
return BadRequest(new ErrorResponse { Error = ex.Message, Code = "invalid_request" });
|
|
}
|
|
}
|
|
|
|
/// <summary>
|
|
/// Performs semantic (vector) search over indexed documents.
|
|
/// Embeds the query, retrieves the closest chunks by cosine similarity, and returns the ranked results.
|
|
/// </summary>
|
|
/// <param name="request">The search request: query text, optional document type filter, and maximum result count.</param>
|
|
/// <param name="ct">Cancellation token.</param>
|
|
/// <returns>
|
|
/// 200 OK with a <see cref="SearchResponse"/> containing the ranked matching chunks with scores and metadata;
|
|
/// 400 Bad Request if the query is empty or the request is otherwise invalid.
|
|
/// </returns>
|
|
[HttpPost("search")]
|
|
[SwaggerOperation(Summary = "Semantic search", Description = "Embeds the query and retrieves the closest document chunks by vector similarity.")]
|
|
[SwaggerResponse(StatusCodes.Status200OK, "Search results returned", typeof(SearchResponse))]
|
|
[SwaggerResponse(StatusCodes.Status400BadRequest, "Query missing or request invalid", typeof(ErrorResponse))]
|
|
[ProducesResponseType(StatusCodes.Status200OK)]
|
|
[ProducesResponseType(typeof(ErrorResponse), StatusCodes.Status400BadRequest)]
|
|
public async Task<ActionResult<SearchResponse>> Search([FromBody] SearchRequest request, CancellationToken ct)
|
|
{
|
|
try
|
|
{
|
|
_logger.LogInformation("Semantic search request received. TargetTypes={TargetTypes}, TopK={TopK}",
|
|
string.Join(',', request.TargetDocumentTypes ?? System.Array.Empty<string>()), request.TopK);
|
|
var result = await _ragService.SearchAsync(request, ct);
|
|
_logger.LogInformation("Semantic search completed. ResultCount={ResultCount}", result.Results.Count);
|
|
return Ok(result);
|
|
}
|
|
catch (InvalidOperationException ex)
|
|
{
|
|
_logger.LogWarning(ex, "Invalid semantic search request.");
|
|
return BadRequest(new ErrorResponse { Error = ex.Message, Code = "invalid_request" });
|
|
}
|
|
}
|
|
|
|
/// <summary>
|
|
/// Returns the stored details for a previously indexed document, including its extracted text and metadata.
|
|
/// </summary>
|
|
/// <param name="id">The document ID returned when the document was indexed.</param>
|
|
/// <param name="ct">Cancellation token.</param>
|
|
/// <returns>
|
|
/// 200 OK with a <see cref="RagDocumentDetailsResponse"/> containing the document text and metadata;
|
|
/// 404 Not Found if no document with the given ID exists in the store.
|
|
/// </returns>
|
|
[HttpGet("documents/{id}")]
|
|
[SwaggerOperation(Summary = "Get document details", Description = "Returns the stored text and metadata for a previously indexed document.")]
|
|
[SwaggerResponse(StatusCodes.Status200OK, "Document details returned", typeof(RagDocumentDetailsResponse))]
|
|
[SwaggerResponse(StatusCodes.Status404NotFound, "Document not found", typeof(ErrorResponse))]
|
|
[ProducesResponseType(StatusCodes.Status200OK)]
|
|
[ProducesResponseType(typeof(ErrorResponse), StatusCodes.Status404NotFound)]
|
|
public async Task<ActionResult<RagDocumentDetailsResponse>> GetDocument(string id, CancellationToken ct)
|
|
{
|
|
_logger.LogInformation("Get document request received. DocumentId={DocumentId}", id);
|
|
var document = await _ragService.GetDocumentAsync(id, ct);
|
|
if (document is null)
|
|
{
|
|
_logger.LogWarning("Document not found. DocumentId={DocumentId}", id);
|
|
return NotFound(new ErrorResponse { Error = "Document not found.", Code = "document_not_found" });
|
|
}
|
|
return Ok(document);
|
|
}
|
|
}
|