mirror of
https://github.com/Stirling-Tools/Stirling-PDF.git
synced 2026-09-03 05:10:16 +03:00
Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0d70133e64 | ||
|
|
07de12746c | ||
|
|
469e7c499c | ||
|
|
40ccbc15cc | ||
|
|
62944d7423 | ||
|
|
7ce98d29f2 | ||
|
|
65c01e8078 | ||
|
|
4700542c75 |
@@ -13,6 +13,11 @@ on:
|
||||
required: false
|
||||
type: boolean
|
||||
default: true
|
||||
build_engine:
|
||||
description: "Build & push the stirling-pdf-engine image (plus the -docparse addon variant)."
|
||||
required: false
|
||||
type: boolean
|
||||
default: false
|
||||
force_unoserver_rebuild:
|
||||
description: "Rebuild stirling-unoserver even if its source hash is unchanged."
|
||||
required: false
|
||||
@@ -51,6 +56,8 @@ jobs:
|
||||
env:
|
||||
RUN_MAIN_APP: ${{ github.event_name != 'workflow_dispatch' || inputs.build_main_app }}
|
||||
RUN_UNOSERVER: ${{ github.event_name != 'workflow_dispatch' || inputs.build_unoserver }}
|
||||
# Engine images are dispatch-only for now; flip the default once the addon stabilises.
|
||||
RUN_ENGINE: ${{ github.event_name == 'workflow_dispatch' && inputs.build_engine }}
|
||||
steps:
|
||||
- name: Harden Runner
|
||||
uses: step-security/harden-runner@ab7a9404c0f3da075243ca237b5fac12c98deaa5 # v2.19.3
|
||||
@@ -219,6 +226,62 @@ jobs:
|
||||
cosign sign --key env://COSIGN_PRIVATE_KEY --yes "${tag}@${DIGEST}"
|
||||
done
|
||||
|
||||
- name: Generate tags for engine
|
||||
id: meta-engine
|
||||
uses: docker/metadata-action@80c7e94dd9b9319bd5eb7a0e0fe9291e23a2a2e9 # v6.1.0
|
||||
if: env.RUN_ENGINE == 'true'
|
||||
with:
|
||||
images: |
|
||||
ghcr.io/${{ steps.repoowner.outputs.lowercase }}/stirling-pdf-engine
|
||||
${{ secrets.DOCKER_HUB_ORG_USERNAME }}/stirling-pdf-engine
|
||||
tags: |
|
||||
type=raw,value=${{ steps.versionNumber.outputs.versionNumber }}
|
||||
type=raw,value=latest
|
||||
|
||||
- name: Build and push engine image
|
||||
id: build-push-engine
|
||||
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0
|
||||
if: env.RUN_ENGINE == 'true' && steps.meta-engine.outputs.tags != ''
|
||||
with:
|
||||
builder: ${{ steps.buildx.outputs.name }}
|
||||
context: ./engine
|
||||
push: true
|
||||
cache-from: type=gha,scope=stirling-pdf-engine
|
||||
cache-to: type=gha,mode=max,scope=stirling-pdf-engine
|
||||
tags: ${{ steps.meta-engine.outputs.tags }}
|
||||
labels: ${{ steps.meta-engine.outputs.labels }}
|
||||
platforms: linux/amd64,linux/arm64/v8
|
||||
provenance: true
|
||||
sbom: true
|
||||
|
||||
- name: Generate tags for engine docparse addon
|
||||
id: meta-engine-docparse
|
||||
uses: docker/metadata-action@80c7e94dd9b9319bd5eb7a0e0fe9291e23a2a2e9 # v6.1.0
|
||||
if: env.RUN_ENGINE == 'true'
|
||||
with:
|
||||
images: |
|
||||
ghcr.io/${{ steps.repoowner.outputs.lowercase }}/stirling-pdf-engine
|
||||
${{ secrets.DOCKER_HUB_ORG_USERNAME }}/stirling-pdf-engine
|
||||
tags: |
|
||||
type=raw,value=${{ steps.versionNumber.outputs.versionNumber }}-docparse
|
||||
type=raw,value=latest-docparse
|
||||
|
||||
- name: Build and push engine docparse addon image
|
||||
uses: docker/build-push-action@53b7df96c91f9c12dcc8a07bcb9ccacbed38856a # v7.3.0
|
||||
if: env.RUN_ENGINE == 'true' && steps.meta-engine-docparse.outputs.tags != ''
|
||||
with:
|
||||
builder: ${{ steps.buildx.outputs.name }}
|
||||
context: ./engine
|
||||
push: true
|
||||
cache-from: type=gha,scope=stirling-pdf-engine-docparse
|
||||
cache-to: type=gha,mode=max,scope=stirling-pdf-engine-docparse
|
||||
tags: ${{ steps.meta-engine-docparse.outputs.tags }}
|
||||
labels: ${{ steps.meta-engine-docparse.outputs.labels }}
|
||||
build-args: DOCPARSE=true
|
||||
platforms: linux/amd64
|
||||
provenance: true
|
||||
sbom: true
|
||||
|
||||
- name: Generate tags for ultra-lite
|
||||
id: meta-lite
|
||||
uses: docker/metadata-action@80c7e94dd9b9319bd5eb7a0e0fe9291e23a2a2e9 # v6.1.0
|
||||
|
||||
@@ -433,6 +433,20 @@ public class EndpointConfiguration {
|
||||
addEndpointToGroup("Automation", "automate"); // Alias for handleData (user-friendly name)
|
||||
addEndpointToGroup("Automation", "pipeline");
|
||||
|
||||
// Adding endpoints to "DocParse" group (parsing, splitting, chunking, extraction,
|
||||
// templating)
|
||||
addEndpointToGroup("DocParse", "parse-document");
|
||||
addEndpointToGroup("DocParse", "extract-fields");
|
||||
addEndpointToGroup("DocParse", "smart-split");
|
||||
addEndpointToGroup("DocParse", "chunk-document");
|
||||
addEndpointToGroup("DocParse", "rag-ingest");
|
||||
addEndpointToGroup("DocParse", "rag-documents");
|
||||
addEndpointToGroup("DocParse", "rag-search");
|
||||
addEndpointToGroup("DocParse", "rag-ask");
|
||||
addEndpointToGroup("DocParse", "extract-tables");
|
||||
addEndpointToGroup("DocParse", "suggest-schema");
|
||||
addEndpointToGroup("DocParse", "fill-template");
|
||||
|
||||
// Adding endpoints to "DeveloperTools" group
|
||||
addEndpointToGroup("DeveloperTools", "show-javascript");
|
||||
|
||||
|
||||
@@ -77,6 +77,7 @@ public class ApplicationProperties {
|
||||
private ProcessExecutor processExecutor = new ProcessExecutor();
|
||||
private PdfEditor pdfEditor = new PdfEditor();
|
||||
private AiEngine aiEngine = new AiEngine();
|
||||
private Docparse docparse = new Docparse();
|
||||
private Mcp mcp = new Mcp();
|
||||
private InternalApi internalApi = new InternalApi();
|
||||
private Cluster cluster = new Cluster();
|
||||
@@ -425,6 +426,24 @@ public class ApplicationProperties {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* DocParse settings (top-level {@code docparse.*}): document understanding for ingestion
|
||||
* pipelines. The basic tier (text layer) always works; the advanced tier lives in the engine's
|
||||
* docparse addon.
|
||||
*/
|
||||
@Data
|
||||
public static class Docparse {
|
||||
|
||||
/** Master switch; hides the DocParse endpoints when false. */
|
||||
private boolean enabled = true;
|
||||
|
||||
/** Requested tier: 'auto', 'basic', or 'advanced'. 'auto' resolves per document. */
|
||||
private String mode = "auto";
|
||||
|
||||
/** Mirrors DOCPARSE_AUTO_INSTALL for the engine's boot-time addon install script. */
|
||||
private boolean autoInstall = false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Model Context Protocol (MCP) server configuration. All keys live under the top-level {@code
|
||||
* mcp.*} prefix. {@link #enabled} defaults to {@code false}: when off, no MCP beans are wired,
|
||||
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
package stirling.software.common.service;
|
||||
|
||||
/**
|
||||
* View of the engine's DocParse capability for modules that cannot see the proprietary
|
||||
* implementation (e.g. ConfigController in core). Implemented by the proprietary
|
||||
* DocparseCapabilityService; absent when the proprietary module is not loaded.
|
||||
*/
|
||||
public interface DocparseCapabilityServiceInterface {
|
||||
|
||||
/**
|
||||
* Whether the engine reports the docparse addon (advanced tier) as installed. Must be cheap and
|
||||
* non-blocking: returns the cached probe result, {@code false} when the engine is disabled,
|
||||
* unreachable, or not yet probed.
|
||||
*/
|
||||
boolean isAdvancedInstalled();
|
||||
}
|
||||
@@ -53,7 +53,7 @@ public class InternalApiClient {
|
||||
// ApiConnectionResolver.
|
||||
private static final Pattern ALLOWED_ENDPOINT_PATH =
|
||||
Pattern.compile(
|
||||
"^/api/v1/(general|misc|security|convert|filter|integration)(/[A-Za-z0-9_-]+)+$"
|
||||
"^/api/v1/(general|misc|security|convert|filter|integration|docparse)(/[A-Za-z0-9_-]+)+$"
|
||||
+ "|^/api/v1/ai/tools(/[A-Za-z0-9_-]+)+$");
|
||||
|
||||
/**
|
||||
|
||||
+16
-1
@@ -24,6 +24,7 @@ import stirling.software.common.annotations.api.ConfigApi;
|
||||
import stirling.software.common.configuration.AppConfig;
|
||||
import stirling.software.common.configuration.interfaces.ShowAdminInterface;
|
||||
import stirling.software.common.model.ApplicationProperties;
|
||||
import stirling.software.common.service.DocparseCapabilityServiceInterface;
|
||||
import stirling.software.common.service.ServerCertificateServiceInterface;
|
||||
import stirling.software.common.service.UserServiceInterface;
|
||||
import stirling.software.common.util.GeneralUtils;
|
||||
@@ -41,6 +42,7 @@ public class ConfigController {
|
||||
private final ShowAdminInterface showAdmin;
|
||||
private final stirling.software.common.service.LicenseServiceInterface licenseService;
|
||||
private final stirling.software.SPDF.config.ExternalAppDepConfig externalAppDepConfig;
|
||||
private final DocparseCapabilityServiceInterface docparseCapabilityService;
|
||||
|
||||
public ConfigController(
|
||||
ApplicationProperties applicationProperties,
|
||||
@@ -54,7 +56,9 @@ public class ConfigController {
|
||||
ShowAdminInterface showAdmin,
|
||||
@org.springframework.beans.factory.annotation.Autowired(required = false)
|
||||
stirling.software.common.service.LicenseServiceInterface licenseService,
|
||||
stirling.software.SPDF.config.ExternalAppDepConfig externalAppDepConfig) {
|
||||
stirling.software.SPDF.config.ExternalAppDepConfig externalAppDepConfig,
|
||||
@org.springframework.beans.factory.annotation.Autowired(required = false)
|
||||
DocparseCapabilityServiceInterface docparseCapabilityService) {
|
||||
this.applicationProperties = applicationProperties;
|
||||
this.applicationContext = applicationContext;
|
||||
this.endpointConfiguration = endpointConfiguration;
|
||||
@@ -63,6 +67,7 @@ public class ConfigController {
|
||||
this.showAdmin = showAdmin;
|
||||
this.licenseService = licenseService;
|
||||
this.externalAppDepConfig = externalAppDepConfig;
|
||||
this.docparseCapabilityService = docparseCapabilityService;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -350,6 +355,16 @@ public class ConfigController {
|
||||
Map.entry("pdfComment", aiFeatures.isPdfComment()),
|
||||
Map.entry("classify", aiFeatures.isClassify())));
|
||||
|
||||
// DocParse settings; "advanced" reflects the cached engine capability probe and is
|
||||
// false when the engine is disabled, unreachable, or the proprietary module is absent.
|
||||
boolean docparseEnabled = applicationProperties.getDocparse().isEnabled();
|
||||
configData.put("docparseEnabled", docparseEnabled);
|
||||
configData.put(
|
||||
"docparseAdvanced",
|
||||
docparseEnabled
|
||||
&& docparseCapabilityService != null
|
||||
&& docparseCapabilityService.isAdvancedInstalled());
|
||||
|
||||
// Timestamp TSA settings — single source of truth for presets + admin URLs
|
||||
ApplicationProperties.Security.Timestamp tsConfig =
|
||||
applicationProperties.getSecurity().getTimestamp();
|
||||
|
||||
@@ -396,6 +396,14 @@ aiEngine:
|
||||
pdfComment: true # AI-authored PDF comments/annotations
|
||||
classify: true # Automatic document classification/labelling
|
||||
|
||||
# DocParse: document understanding for ingestion pipelines (chunking + knowledge-base
|
||||
# indexing). The basic tier (text layer) always works; the advanced tier (layout parsing)
|
||||
# requires the engine's docparse addon. Env overrides: DOCPARSE_ENABLED, DOCPARSE_MODE.
|
||||
docparse:
|
||||
enabled: true # Master switch; hides the DocParse endpoints when false
|
||||
mode: auto # Tier selection: 'auto' (best available), 'basic', or 'advanced'
|
||||
autoInstall: false # Mirrors DOCPARSE_AUTO_INSTALL for the engine's boot-time addon install script
|
||||
|
||||
policies:
|
||||
# Folder automations can read from and write to the directories you allow here, so treat this as a
|
||||
# security boundary. Leave allowedFolderRoots empty (default) to disable folder sources/outputs,
|
||||
|
||||
+2
-1
@@ -74,7 +74,8 @@ class ConfigControllerMoreTest {
|
||||
userService,
|
||||
showAdmin,
|
||||
licenseService,
|
||||
externalAppDepConfig);
|
||||
externalAppDepConfig,
|
||||
null);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
|
||||
+2
-1
@@ -52,7 +52,8 @@ class ConfigControllerTest {
|
||||
userService,
|
||||
showAdmin,
|
||||
licenseService,
|
||||
mock(stirling.software.SPDF.config.ExternalAppDepConfig.class));
|
||||
mock(stirling.software.SPDF.config.ExternalAppDepConfig.class),
|
||||
null);
|
||||
}
|
||||
|
||||
@Test
|
||||
|
||||
+532
@@ -0,0 +1,532 @@
|
||||
package stirling.software.proprietary.controller.api;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.StringWriter;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.util.Base64;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.zip.ZipEntry;
|
||||
import java.util.zip.ZipOutputStream;
|
||||
|
||||
import org.apache.commons.csv.CSVFormat;
|
||||
import org.apache.commons.csv.CSVPrinter;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.springframework.core.io.ByteArrayResource;
|
||||
import org.springframework.core.io.Resource;
|
||||
import org.springframework.http.HttpHeaders;
|
||||
import org.springframework.http.HttpStatus;
|
||||
import org.springframework.http.MediaType;
|
||||
import org.springframework.http.ResponseEntity;
|
||||
import org.springframework.web.bind.annotation.GetMapping;
|
||||
import org.springframework.web.bind.annotation.ModelAttribute;
|
||||
import org.springframework.web.bind.annotation.PostMapping;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
import org.springframework.web.bind.annotation.RequestMapping;
|
||||
import org.springframework.web.bind.annotation.RequestParam;
|
||||
import org.springframework.web.bind.annotation.RestController;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
import org.springframework.web.server.ResponseStatusException;
|
||||
|
||||
import io.swagger.v3.oas.annotations.Operation;
|
||||
import io.swagger.v3.oas.annotations.tags.Tag;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.common.annotations.AutoJobPostMapping;
|
||||
import stirling.software.common.enumeration.ResourceWeight;
|
||||
import stirling.software.common.service.CustomPDFDocumentFactory;
|
||||
import stirling.software.common.util.FormUtils;
|
||||
import stirling.software.common.util.GeneralUtils;
|
||||
import stirling.software.common.util.TempFile;
|
||||
import stirling.software.common.util.TempFileManager;
|
||||
import stirling.software.common.util.WebResponseUtils;
|
||||
import stirling.software.proprietary.model.api.docparse.ChunkDocumentApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.ExtractFieldsApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.ExtractTablesApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.ParseDocumentApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.RagAskApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.RagIngestApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.RagSearchApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.SmartSplitApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.SuggestSchemaApiRequest;
|
||||
import stirling.software.proprietary.model.docparse.ChunkDocumentResponse;
|
||||
import stirling.software.proprietary.model.docparse.DocChunk;
|
||||
import stirling.software.proprietary.model.docparse.DocTable;
|
||||
import stirling.software.proprietary.model.docparse.DocparseCapabilitiesView;
|
||||
import stirling.software.proprietary.model.docparse.DocparseMode;
|
||||
import stirling.software.proprietary.model.docparse.ExtractFieldsResponse;
|
||||
import stirling.software.proprietary.model.docparse.ExtractTablesResponse;
|
||||
import stirling.software.proprietary.model.docparse.FillDocxResponse;
|
||||
import stirling.software.proprietary.model.docparse.ParseDocumentResponse;
|
||||
import stirling.software.proprietary.model.docparse.RagIngestResponse;
|
||||
import stirling.software.proprietary.model.docparse.RagStatsView;
|
||||
import stirling.software.proprietary.model.docparse.SmartSplitResponse;
|
||||
import stirling.software.proprietary.model.docparse.SplitPart;
|
||||
import stirling.software.proprietary.model.docparse.SuggestSchemaResponse;
|
||||
import stirling.software.proprietary.service.AiToolResponseHeaders;
|
||||
import stirling.software.proprietary.service.DocParseService;
|
||||
|
||||
import tools.jackson.databind.ObjectMapper;
|
||||
import tools.jackson.databind.node.ObjectNode;
|
||||
|
||||
/**
|
||||
* Public DocParse ingestion API. Thin HTTP layer over {@link DocParseService}, which owns the
|
||||
* engine wire contract; this class owns the pipeline step shape (report header, export ZIP).
|
||||
*/
|
||||
@Slf4j
|
||||
@RestController
|
||||
@RequestMapping("/api/v1/docparse")
|
||||
@RequiredArgsConstructor
|
||||
@Tag(
|
||||
name = "DocParse",
|
||||
description =
|
||||
"Document ingestion: chunk, embed, and index documents into the searchable"
|
||||
+ " knowledge base, or export the parsed content (markdown, chunks JSONL)"
|
||||
+ " for external systems.")
|
||||
public class DocParseController {
|
||||
|
||||
private static final MediaType CSV = MediaType.parseMediaType("text/csv");
|
||||
|
||||
private static final MediaType MARKDOWN = MediaType.parseMediaType("text/markdown");
|
||||
private static final MediaType DOCX =
|
||||
MediaType.parseMediaType(
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document");
|
||||
|
||||
private final DocParseService docParseService;
|
||||
private final CustomPDFDocumentFactory pdfDocumentFactory;
|
||||
private final TempFileManager tempFileManager;
|
||||
private final ObjectMapper objectMapper;
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/rag-ingest",
|
||||
resourceWeight = ResourceWeight.LARGE_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Chunk, embed, and index a document into the RAG store (pipeline shape)",
|
||||
description =
|
||||
"Ingests the document into the engine's RAG store under a stable documentId"
|
||||
+ " (default: content hash). Returns the ORIGINAL PDF unchanged as the"
|
||||
+ " body, with the ingest summary JSON in the X-Stirling-Tool-Report"
|
||||
+ " header so policy pipelines pick it up as the step report. With"
|
||||
+ " exportMarkdown/exportChunksJsonl the body becomes a ZIP holding the"
|
||||
+ " original plus the corpus files, ready for delivery to external"
|
||||
+ " systems. Input:PDF Output:PDF/ZIP Type:SISO")
|
||||
public ResponseEntity<Resource> ragIngest(@ModelAttribute RagIngestApiRequest request)
|
||||
throws IOException {
|
||||
MultipartFile file = request.getFileInput();
|
||||
boolean export = request.isExportMarkdown() || request.isExportChunksJsonl();
|
||||
RagIngestResponse result =
|
||||
docParseService.ragIngest(
|
||||
file,
|
||||
request.getDocumentId(),
|
||||
request.getChunkSize(),
|
||||
request.getOverlap(),
|
||||
DocparseMode.fromWire(request.getMode()),
|
||||
request.isIndex(),
|
||||
request.isExportMarkdown(),
|
||||
request.isExportChunksJsonl());
|
||||
|
||||
// The report header must stay small: summary fields only, never the echoed content.
|
||||
ObjectNode report = objectMapper.createObjectNode();
|
||||
report.put("mode", result.mode().wire());
|
||||
report.put("documentId", result.documentId());
|
||||
report.put("chunksIndexed", result.chunksIndexed());
|
||||
report.put("pages", result.pages());
|
||||
report.put("indexed", request.isIndex());
|
||||
|
||||
String fileName = DocParseService.fileName(file);
|
||||
byte[] original = file.getBytes();
|
||||
HttpHeaders headers = new HttpHeaders();
|
||||
headers.set(AiToolResponseHeaders.TOOL_REPORT, objectMapper.writeValueAsString(report));
|
||||
|
||||
if (!export) {
|
||||
headers.setContentType(MediaType.APPLICATION_PDF);
|
||||
headers.setContentDispositionFormData("attachment", fileName);
|
||||
headers.setContentLength(original.length);
|
||||
return ResponseEntity.ok().headers(headers).body(new ByteArrayResource(original));
|
||||
}
|
||||
|
||||
byte[] zip = exportZip(fileName, original, result, request);
|
||||
headers.setContentType(MediaType.parseMediaType("application/zip"));
|
||||
headers.setContentDispositionFormData("attachment", baseName(fileName) + "-ingested.zip");
|
||||
headers.setContentLength(zip.length);
|
||||
return ResponseEntity.ok().headers(headers).body(new ByteArrayResource(zip));
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/extract-fields",
|
||||
resourceWeight = ResourceWeight.LARGE_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Extract typed fields from a document (pipeline shape)",
|
||||
description =
|
||||
"Extracts the fields described by the JSON Schema and returns the ORIGINAL PDF"
|
||||
+ " unchanged as the body, with the extraction JSON in the"
|
||||
+ " X-Stirling-Tool-Report header so policy pipelines pick it up as the"
|
||||
+ " step report. Use /extract-fields/json for the raw JSON."
|
||||
+ " Input:PDF Output:PDF Type:SISO")
|
||||
public ResponseEntity<Resource> extractFields(@ModelAttribute ExtractFieldsApiRequest request)
|
||||
throws IOException {
|
||||
MultipartFile file = request.getFileInput();
|
||||
ExtractFieldsResponse result =
|
||||
docParseService.extractFields(
|
||||
file,
|
||||
request.getFieldsSchema(),
|
||||
DocparseMode.fromWire(request.getMode()),
|
||||
request.getInstructions());
|
||||
byte[] original = file.getBytes();
|
||||
HttpHeaders headers = new HttpHeaders();
|
||||
headers.setContentType(MediaType.APPLICATION_PDF);
|
||||
headers.setContentDispositionFormData("attachment", DocParseService.fileName(file));
|
||||
headers.setContentLength(original.length);
|
||||
headers.set(AiToolResponseHeaders.TOOL_REPORT, objectMapper.writeValueAsString(result));
|
||||
return ResponseEntity.ok().headers(headers).body(new ByteArrayResource(original));
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/extract-fields/json",
|
||||
resourceWeight = ResourceWeight.LARGE_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Extract typed fields from a document (JSON)",
|
||||
description =
|
||||
"Extracts the fields described by the JSON Schema and returns the extraction"
|
||||
+ " result (fields, confidence, citations) as JSON."
|
||||
+ " Input:PDF Output:JSON Type:SISO")
|
||||
public ResponseEntity<ExtractFieldsResponse> extractFieldsJson(
|
||||
@ModelAttribute ExtractFieldsApiRequest request) throws IOException {
|
||||
return ResponseEntity.ok(
|
||||
docParseService.extractFields(
|
||||
request.getFileInput(),
|
||||
request.getFieldsSchema(),
|
||||
DocparseMode.fromWire(request.getMode()),
|
||||
request.getInstructions()));
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/suggest-schema",
|
||||
resourceWeight = ResourceWeight.LARGE_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Suggest an extraction schema for a document",
|
||||
description =
|
||||
"Reads the document and proposes the fields worth extracting (name, type,"
|
||||
+ " description), ready to feed into /extract-fields as a JSON Schema."
|
||||
+ " Input:PDF Output:JSON Type:SISO")
|
||||
public ResponseEntity<SuggestSchemaResponse> suggestSchema(
|
||||
@ModelAttribute SuggestSchemaApiRequest request) throws IOException {
|
||||
return ResponseEntity.ok(
|
||||
docParseService.suggestSchema(request.getFileInput(), request.getMaxFields()));
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/parse-document",
|
||||
resourceWeight = ResourceWeight.XLARGE_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Parse a document into structured blocks, tables, and markdown",
|
||||
description =
|
||||
"Parses the PDF into layout blocks, tables, and a markdown rendering. The"
|
||||
+ " basic tier reads the text layer; the advanced tier (docparse addon)"
|
||||
+ " adds OCR, real table structure, and bounding boxes."
|
||||
+ " Input:PDF Output:JSON Type:SISO")
|
||||
public ResponseEntity<?> parseDocument(@ModelAttribute ParseDocumentApiRequest request)
|
||||
throws IOException {
|
||||
ParseDocumentResponse result =
|
||||
docParseService.parse(
|
||||
request.getFileInput(),
|
||||
DocparseMode.fromWire(request.getMode()),
|
||||
request.isWithOcr());
|
||||
if ("markdown".equalsIgnoreCase(request.getOutputFormat())) {
|
||||
return WebResponseUtils.bytesToWebResponse(
|
||||
result.markdown().getBytes(StandardCharsets.UTF_8),
|
||||
outputName(request.getFileInput(), "_parsed.md"),
|
||||
MARKDOWN);
|
||||
}
|
||||
return ResponseEntity.ok(result);
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/smart-split",
|
||||
resourceWeight = ResourceWeight.LARGE_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Split a document at content-derived boundaries",
|
||||
description =
|
||||
"Asks the engine where sub-documents start (per the natural-language rule) and"
|
||||
+ " returns a ZIP with one PDF per part, named from the part labels."
|
||||
+ " Input:PDF Output:ZIP-PDF Type:SIMO")
|
||||
public ResponseEntity<Resource> smartSplit(@ModelAttribute SmartSplitApiRequest request)
|
||||
throws IOException {
|
||||
MultipartFile file = request.getFileInput();
|
||||
SmartSplitResponse split =
|
||||
docParseService.split(file, request.getRule(), request.getMaxParts());
|
||||
if (split.parts().isEmpty()) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.UNPROCESSABLE_ENTITY,
|
||||
"The split rule produced no parts for this document");
|
||||
}
|
||||
TempFile zipTempFile = tempFileManager.createManagedTempFile(".zip");
|
||||
try {
|
||||
try (TempFile sourceTempFile = new TempFile(tempFileManager, ".pdf")) {
|
||||
Files.copy(
|
||||
file.getInputStream(),
|
||||
sourceTempFile.getPath(),
|
||||
StandardCopyOption.REPLACE_EXISTING);
|
||||
try (ZipOutputStream zipOut =
|
||||
new ZipOutputStream(Files.newOutputStream(zipTempFile.getPath()))) {
|
||||
writeParts(sourceTempFile, split.parts(), zipOut);
|
||||
}
|
||||
}
|
||||
return WebResponseUtils.zipFileToWebResponse(
|
||||
zipTempFile,
|
||||
GeneralUtils.generateFilename(file.getOriginalFilename(), "_split.zip"));
|
||||
} catch (Exception e) {
|
||||
zipTempFile.close();
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/chunk-document",
|
||||
resourceWeight = ResourceWeight.MEDIUM_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Chunk a document for RAG",
|
||||
description =
|
||||
"Splits the document text into overlapping chunks with page spans and (advanced"
|
||||
+ " tier) heading breadcrumbs. Input:PDF Output:JSON Type:SISO")
|
||||
public ResponseEntity<ChunkDocumentResponse> chunkDocument(
|
||||
@ModelAttribute ChunkDocumentApiRequest request) throws IOException {
|
||||
return ResponseEntity.ok(
|
||||
docParseService.chunk(
|
||||
request.getFileInput(),
|
||||
request.getChunkSize(),
|
||||
request.getOverlap(),
|
||||
DocparseMode.fromWire(request.getMode())));
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/fill-template",
|
||||
resourceWeight = ResourceWeight.SMALL_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Fill a DOCX template with JSON data",
|
||||
description =
|
||||
"Replaces the template's placeholders with values from the JSON object and"
|
||||
+ " returns the filled DOCX. Replacement counts and missing keys ride"
|
||||
+ " the X-Stirling-Tool-Report header."
|
||||
+ " Input:DOCX Output:DOCX Type:SISO")
|
||||
public ResponseEntity<Resource> fillTemplate(
|
||||
@RequestParam("templateFile") MultipartFile templateFile,
|
||||
@RequestParam("data") String data)
|
||||
throws IOException {
|
||||
FillDocxResponse result = docParseService.fillDocx(templateFile, data);
|
||||
byte[] filled = Base64.getDecoder().decode(result.docxBase64());
|
||||
HttpHeaders headers = new HttpHeaders();
|
||||
headers.setContentType(DOCX);
|
||||
headers.setContentDispositionFormData(
|
||||
"attachment",
|
||||
GeneralUtils.generateFilename(templateFile.getOriginalFilename(), "_filled.docx"));
|
||||
headers.setContentLength(filled.length);
|
||||
headers.set(
|
||||
AiToolResponseHeaders.TOOL_REPORT,
|
||||
objectMapper.writeValueAsString(
|
||||
new FillDocxResponse("", result.replaced(), result.missing())));
|
||||
return ResponseEntity.ok().headers(headers).body(new ByteArrayResource(filled));
|
||||
}
|
||||
|
||||
@GetMapping("/capabilities")
|
||||
@Operation(
|
||||
summary = "DocParse capability summary",
|
||||
description =
|
||||
"Merged view of the Java settings and the engine's capability probe, so"
|
||||
+ " clients can gate advanced-tier UI.")
|
||||
public ResponseEntity<DocparseCapabilitiesView> capabilities() {
|
||||
return ResponseEntity.ok(docParseService.capabilitiesView());
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/extract-tables",
|
||||
resourceWeight = ResourceWeight.LARGE_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Extract tables from a document",
|
||||
description =
|
||||
"Extracts table structure and returns CSV (all tables concatenated, blank line"
|
||||
+ " between them) or the structured JSON table list."
|
||||
+ " Input:PDF Output:CSV/JSON Type:SISO")
|
||||
public ResponseEntity<?> extractTables(@ModelAttribute ExtractTablesApiRequest request)
|
||||
throws IOException {
|
||||
ExtractTablesResponse result = docParseService.tables(request.getFileInput());
|
||||
if ("json".equalsIgnoreCase(request.getOutputFormat())) {
|
||||
return ResponseEntity.ok(result);
|
||||
}
|
||||
return WebResponseUtils.bytesToWebResponse(
|
||||
tablesToCsv(result.tables()).getBytes(StandardCharsets.UTF_8),
|
||||
outputName(request.getFileInput(), "_tables.csv"),
|
||||
CSV);
|
||||
}
|
||||
|
||||
@GetMapping("/rag-stats")
|
||||
@Operation(
|
||||
summary = "RAG store statistics",
|
||||
description =
|
||||
"The engine's document-store totals (backend, documents, chunks, embedding"
|
||||
+ " model) merged with the DocParse capability fields. Answers with"
|
||||
+ " zeros and engineReachable=false when the engine is down.")
|
||||
public ResponseEntity<RagStatsView> ragStats() {
|
||||
return ResponseEntity.ok(docParseService.ragStats());
|
||||
}
|
||||
|
||||
@GetMapping("/rag-documents")
|
||||
@Operation(
|
||||
summary = "List documents in the RAG store",
|
||||
description =
|
||||
"Engine passthrough of the caller-visible indexed documents (documentId,"
|
||||
+ " source, chunk count).")
|
||||
public ResponseEntity<String> ragDocuments() throws IOException {
|
||||
return jsonPassthrough(docParseService.ragDocuments());
|
||||
}
|
||||
|
||||
@PostMapping(value = "/rag-search", consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
@Operation(
|
||||
summary = "Semantic search over the RAG store",
|
||||
description =
|
||||
"Searches the caller-visible indexed documents and returns the top passages"
|
||||
+ " with scores, page spans, and heading breadcrumbs.")
|
||||
public ResponseEntity<String> ragSearch(@RequestBody RagSearchApiRequest request)
|
||||
throws IOException {
|
||||
return jsonPassthrough(docParseService.ragSearch(request.getQuery(), request.getTopK()));
|
||||
}
|
||||
|
||||
@PostMapping(value = "/rag-ask", consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
@Operation(
|
||||
summary = "Ask a question over the RAG store",
|
||||
description =
|
||||
"Answers the question from the caller-visible indexed documents and returns"
|
||||
+ " the answer with its supporting passages.")
|
||||
public ResponseEntity<String> ragAsk(@RequestBody RagAskApiRequest request) throws IOException {
|
||||
return jsonPassthrough(docParseService.ragAsk(request.getQuestion(), request.getTopK()));
|
||||
}
|
||||
|
||||
private static ResponseEntity<String> jsonPassthrough(String engineJson) {
|
||||
return ResponseEntity.ok().contentType(MediaType.APPLICATION_JSON).body(engineJson);
|
||||
}
|
||||
|
||||
/** Original + requested corpus files in one ZIP, so destinations receive them together. */
|
||||
private byte[] exportZip(
|
||||
String fileName, byte[] original, RagIngestResponse result, RagIngestApiRequest request)
|
||||
throws IOException {
|
||||
String base = baseName(fileName);
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
try (ZipOutputStream zip = new ZipOutputStream(out)) {
|
||||
zip.putNextEntry(new ZipEntry(fileName));
|
||||
zip.write(original);
|
||||
zip.closeEntry();
|
||||
if (request.isExportMarkdown()) {
|
||||
zip.putNextEntry(new ZipEntry(base + ".md"));
|
||||
zip.write(
|
||||
(result.markdown() == null ? "" : result.markdown())
|
||||
.getBytes(StandardCharsets.UTF_8));
|
||||
zip.closeEntry();
|
||||
}
|
||||
if (request.isExportChunksJsonl()) {
|
||||
zip.putNextEntry(new ZipEntry(base + ".chunks.jsonl"));
|
||||
zip.write(chunksJsonl(result).getBytes(StandardCharsets.UTF_8));
|
||||
zip.closeEntry();
|
||||
}
|
||||
}
|
||||
return out.toByteArray();
|
||||
}
|
||||
|
||||
/** One chunk per line, each self-describing (documentId + source travel on every line). */
|
||||
private String chunksJsonl(RagIngestResponse result) {
|
||||
if (result.chunks() == null) {
|
||||
return "";
|
||||
}
|
||||
StringBuilder lines = new StringBuilder();
|
||||
for (DocChunk chunk : result.chunks()) {
|
||||
ObjectNode line = objectMapper.createObjectNode();
|
||||
line.put("documentId", result.documentId());
|
||||
line.put("index", chunk.index());
|
||||
line.put("text", chunk.text());
|
||||
if (chunk.pageStart() != null) {
|
||||
line.put("pageStart", chunk.pageStart());
|
||||
}
|
||||
if (chunk.pageEnd() != null) {
|
||||
line.put("pageEnd", chunk.pageEnd());
|
||||
}
|
||||
var headings = line.putArray("headingPath");
|
||||
chunk.headingPath().forEach(headings::add);
|
||||
lines.append(objectMapper.writeValueAsString(line)).append('\n');
|
||||
}
|
||||
return lines.toString();
|
||||
}
|
||||
|
||||
private static String baseName(String fileName) {
|
||||
int dot = fileName.lastIndexOf('.');
|
||||
return dot > 0 ? fileName.substring(0, dot) : fileName;
|
||||
}
|
||||
|
||||
private void writeParts(TempFile sourceTempFile, List<SplitPart> parts, ZipOutputStream zipOut)
|
||||
throws IOException {
|
||||
for (int i = 0; i < parts.size(); i++) {
|
||||
SplitPart part = parts.get(i);
|
||||
// Load per part and remove pages outside the range: avoids the PDFBox cross-document
|
||||
// addPage pitfalls while keeping shared resources intact.
|
||||
try (PDDocument partDoc = pdfDocumentFactory.load(sourceTempFile.getFile())) {
|
||||
int pageCount = partDoc.getNumberOfPages();
|
||||
int start = Math.clamp(part.startPage(), 1, pageCount);
|
||||
int end = Math.clamp(part.endPage(), start, pageCount);
|
||||
for (int p = pageCount - 1; p >= 0; p--) {
|
||||
int pageNumber = p + 1;
|
||||
if (pageNumber < start || pageNumber > end) {
|
||||
partDoc.removePage(p);
|
||||
}
|
||||
}
|
||||
FormUtils.pruneOrphanedFormFields(partDoc);
|
||||
zipOut.putNextEntry(new ZipEntry(partEntryName(i, part)));
|
||||
partDoc.save(zipOut);
|
||||
zipOut.closeEntry();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private static String partEntryName(int index, SplitPart part) {
|
||||
String label = part.label() == null ? "" : part.label().trim();
|
||||
String sanitized = label.replaceAll("[^A-Za-z0-9 ._-]", "_").replaceAll("\\s+", "_");
|
||||
if (sanitized.isBlank() || sanitized.chars().allMatch(c -> c == '_' || c == '.')) {
|
||||
sanitized = "part";
|
||||
}
|
||||
// Index prefix keeps entries unique even when labels repeat.
|
||||
return String.format(Locale.ROOT, "%02d_%s.pdf", index + 1, sanitized);
|
||||
}
|
||||
|
||||
private static String tablesToCsv(List<DocTable> tables) throws IOException {
|
||||
CSVFormat format = CSVFormat.EXCEL.builder().setEscape('"').build();
|
||||
StringWriter writer = new StringWriter();
|
||||
try (CSVPrinter printer = format.print(writer)) {
|
||||
boolean first = true;
|
||||
for (DocTable table : tables) {
|
||||
if (!first) {
|
||||
printer.println();
|
||||
}
|
||||
first = false;
|
||||
for (List<String> row : table.cells()) {
|
||||
printer.printRecord(row);
|
||||
}
|
||||
}
|
||||
}
|
||||
return writer.toString();
|
||||
}
|
||||
|
||||
private static String outputName(MultipartFile file, String suffix) {
|
||||
return GeneralUtils.removeExtension(DocParseService.fileName(file)) + suffix;
|
||||
}
|
||||
}
|
||||
+27
@@ -0,0 +1,27 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class ChunkDocumentApiRequest extends PDFFile {
|
||||
|
||||
@Schema(description = "Target chunk size in characters (64-32768)", defaultValue = "512")
|
||||
private int chunkSize = 512;
|
||||
|
||||
@Schema(
|
||||
description = "Overlap between adjacent chunks in characters (0-4096)",
|
||||
defaultValue = "64")
|
||||
private int overlap = 64;
|
||||
|
||||
@Schema(
|
||||
description = "Tier to use: 'auto' picks per document, or force 'basic'/'advanced'",
|
||||
allowableValues = {"auto", "basic", "advanced"},
|
||||
defaultValue = "auto")
|
||||
private String mode = "auto";
|
||||
}
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class ExtractFieldsApiRequest extends PDFFile {
|
||||
|
||||
@Schema(
|
||||
description = "JSON Schema object describing the fields to extract, as a JSON string",
|
||||
requiredMode = Schema.RequiredMode.REQUIRED,
|
||||
example =
|
||||
"{\"type\":\"object\",\"properties\":{\"invoiceNumber\":{\"type\":\"string\"}}}")
|
||||
private String fieldsSchema;
|
||||
|
||||
@Schema(
|
||||
description = "Tier to use: 'auto' picks per document, or force 'basic'/'advanced'",
|
||||
allowableValues = {"auto", "basic", "advanced"},
|
||||
defaultValue = "auto")
|
||||
private String mode = "auto";
|
||||
|
||||
@Schema(description = "Optional natural-language guidance for the extraction")
|
||||
private String instructions;
|
||||
}
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class ExtractTablesApiRequest extends PDFFile {
|
||||
|
||||
@Schema(
|
||||
description = "Response format: CSV text or the structured JSON table list",
|
||||
allowableValues = {"csv", "json"},
|
||||
defaultValue = "csv")
|
||||
private String outputFormat = "csv";
|
||||
}
|
||||
+30
@@ -0,0 +1,30 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class ParseDocumentApiRequest extends PDFFile {
|
||||
|
||||
@Schema(
|
||||
description = "Tier to use: 'auto' picks per document, or force 'basic'/'advanced'",
|
||||
allowableValues = {"auto", "basic", "advanced"},
|
||||
defaultValue = "auto")
|
||||
private String mode = "auto";
|
||||
|
||||
@Schema(
|
||||
description = "Apply OCR when parsing scanned pages (advanced tier only)",
|
||||
defaultValue = "true")
|
||||
private boolean withOcr = true;
|
||||
|
||||
@Schema(
|
||||
description = "Response format: full JSON result or the markdown rendering only",
|
||||
allowableValues = {"json", "markdown"},
|
||||
defaultValue = "json")
|
||||
private String outputFormat = "json";
|
||||
}
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
|
||||
@Data
|
||||
public class RagAskApiRequest {
|
||||
|
||||
@Schema(
|
||||
description = "Question to answer from the indexed documents",
|
||||
requiredMode = Schema.RequiredMode.REQUIRED)
|
||||
private String question;
|
||||
|
||||
@Schema(description = "Number of passages to ground the answer on (1-20)", defaultValue = "5")
|
||||
private int topK = 5;
|
||||
}
|
||||
+52
@@ -0,0 +1,52 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class RagIngestApiRequest extends PDFFile {
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"Stable identifier for the ingested document; re-ingesting the same id replaces"
|
||||
+ " its chunks. Defaults to a content hash of the uploaded bytes.")
|
||||
private String documentId;
|
||||
|
||||
@Schema(description = "Target chunk size in characters (64-32768)", defaultValue = "512")
|
||||
private int chunkSize = 512;
|
||||
|
||||
@Schema(
|
||||
description = "Overlap between adjacent chunks in characters (0-4096)",
|
||||
defaultValue = "64")
|
||||
private int overlap = 64;
|
||||
|
||||
@Schema(
|
||||
description = "Tier to use: 'auto' picks per document, or force 'basic'/'advanced'",
|
||||
allowableValues = {"auto", "basic", "advanced"},
|
||||
defaultValue = "auto")
|
||||
private String mode = "auto";
|
||||
|
||||
@Schema(
|
||||
description = "Index the document into the built-in knowledge base",
|
||||
defaultValue = "true")
|
||||
private boolean index = true;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"Also return the parsed document as a markdown file, for delivery to external"
|
||||
+ " systems (vector DBs, training corpora)",
|
||||
defaultValue = "false")
|
||||
private boolean exportMarkdown = false;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"Also return the chunks as a JSONL file (one chunk per line with page span and"
|
||||
+ " heading breadcrumb), ready for external embedding or indexing",
|
||||
defaultValue = "false")
|
||||
private boolean exportChunksJsonl = false;
|
||||
}
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
|
||||
@Data
|
||||
public class RagSearchApiRequest {
|
||||
|
||||
@Schema(
|
||||
description = "Natural-language search query",
|
||||
requiredMode = Schema.RequiredMode.REQUIRED)
|
||||
private String query;
|
||||
|
||||
@Schema(description = "Number of passages to return (1-50)", defaultValue = "10")
|
||||
private int topK = 10;
|
||||
}
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class SmartSplitApiRequest extends PDFFile {
|
||||
|
||||
@Schema(
|
||||
description = "Natural-language boundary rule, e.g. 'split where a new invoice starts'",
|
||||
requiredMode = Schema.RequiredMode.REQUIRED)
|
||||
private String rule;
|
||||
|
||||
@Schema(description = "Maximum number of parts to produce (1-500)", defaultValue = "50")
|
||||
private int maxParts = 50;
|
||||
}
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class SuggestSchemaApiRequest extends PDFFile {
|
||||
|
||||
@Schema(description = "Maximum number of fields to suggest (1-20)", defaultValue = "10")
|
||||
private int maxFields = 10;
|
||||
}
|
||||
+14
@@ -0,0 +1,14 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import stirling.software.proprietary.model.api.ai.AiPageText;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/docparse/chunk}. */
|
||||
public record ChunkDocumentRequest(
|
||||
String fileName,
|
||||
List<AiPageText> pages,
|
||||
String contentBase64,
|
||||
int chunkSize,
|
||||
int overlap,
|
||||
DocparseMode mode) {}
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Engine response for {@code POST /api/v1/docparse/chunk}. */
|
||||
public record ChunkDocumentResponse(DocparseTier mode, List<DocChunk> chunks) {
|
||||
|
||||
public ChunkDocumentResponse {
|
||||
chunks = chunks == null ? List.of() : chunks;
|
||||
}
|
||||
}
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* One layout block. {@code bbox} is [x0, y0, x1, y1] normalized to 0..1 with a top-left origin;
|
||||
* {@code null} in basic tier (no layout model ran). Mirrors {@code docparse.py DocBlock}.
|
||||
*/
|
||||
public record DocBlock(String type, String text, int page, List<Double> bbox, Double confidence) {}
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** One RAG chunk with page span and heading breadcrumb. Mirrors {@code docparse.py DocChunk}. */
|
||||
public record DocChunk(
|
||||
int index, String text, Integer pageStart, Integer pageEnd, List<String> headingPath) {
|
||||
|
||||
public DocChunk {
|
||||
headingPath = headingPath == null ? List.of() : headingPath;
|
||||
}
|
||||
}
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** One extracted table. Mirrors {@code docparse.py DocTable}. */
|
||||
public record DocTable(
|
||||
int page, List<Double> bbox, List<List<String>> cells, String markdown, Double confidence) {
|
||||
|
||||
public DocTable {
|
||||
cells = cells == null ? List.of() : cells;
|
||||
}
|
||||
}
|
||||
+25
@@ -0,0 +1,25 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* What the engine can actually do right now; Java caches and republishes this. Mirrors {@code
|
||||
* docparse.py DocparseCapabilities}.
|
||||
*/
|
||||
public record DocparseCapabilities(
|
||||
boolean advancedInstalled,
|
||||
String doclingVersion,
|
||||
String torchVersion,
|
||||
boolean modelsAvailable,
|
||||
String modelsPath,
|
||||
List<String> errors) {
|
||||
|
||||
public DocparseCapabilities {
|
||||
errors = errors == null ? List.of() : errors;
|
||||
}
|
||||
|
||||
/** The addon-absent view used when the engine is disabled, unreachable, or probing failed. */
|
||||
public static DocparseCapabilities absent(String reason) {
|
||||
return new DocparseCapabilities(false, null, null, false, null, List.of(reason));
|
||||
}
|
||||
}
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** Merged capability view served by {@code GET /api/v1/docparse/capabilities} (Java side). */
|
||||
public record DocparseCapabilitiesView(
|
||||
boolean enabled,
|
||||
String mode,
|
||||
boolean advancedInstalled,
|
||||
boolean engineReachable,
|
||||
String doclingVersion) {}
|
||||
+35
@@ -0,0 +1,35 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.Locale;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonCreator;
|
||||
import com.fasterxml.jackson.annotation.JsonValue;
|
||||
|
||||
/**
|
||||
* What the caller asked for; {@code AUTO} resolves per request. Wire values are lowercase to match
|
||||
* {@code engine/src/stirling/contracts/docparse.py DocparseMode}.
|
||||
*/
|
||||
public enum DocparseMode {
|
||||
AUTO("auto"),
|
||||
BASIC("basic"),
|
||||
ADVANCED("advanced");
|
||||
|
||||
private final String wire;
|
||||
|
||||
DocparseMode(String wire) {
|
||||
this.wire = wire;
|
||||
}
|
||||
|
||||
@JsonValue
|
||||
public String wire() {
|
||||
return wire;
|
||||
}
|
||||
|
||||
@JsonCreator
|
||||
public static DocparseMode fromWire(String value) {
|
||||
if (value == null || value.isBlank()) {
|
||||
return AUTO;
|
||||
}
|
||||
return valueOf(value.trim().toUpperCase(Locale.ROOT));
|
||||
}
|
||||
}
|
||||
+31
@@ -0,0 +1,31 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.Locale;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonCreator;
|
||||
import com.fasterxml.jackson.annotation.JsonValue;
|
||||
|
||||
/**
|
||||
* Which implementation actually served a request. Wire values are lowercase to match {@code
|
||||
* engine/src/stirling/contracts/docparse.py DocparseTier}.
|
||||
*/
|
||||
public enum DocparseTier {
|
||||
BASIC("basic"),
|
||||
ADVANCED("advanced");
|
||||
|
||||
private final String wire;
|
||||
|
||||
DocparseTier(String wire) {
|
||||
this.wire = wire;
|
||||
}
|
||||
|
||||
@JsonValue
|
||||
public String wire() {
|
||||
return wire;
|
||||
}
|
||||
|
||||
@JsonCreator
|
||||
public static DocparseTier fromWire(String value) {
|
||||
return valueOf(value.trim().toUpperCase(Locale.ROOT));
|
||||
}
|
||||
}
|
||||
+5
@@ -0,0 +1,5 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** Engine response for {@code GET /api/v1/documents/stats}: the RAG document store totals. */
|
||||
public record DocumentStoreStats(
|
||||
String backend, long documents, long chunks, String embeddingModel) {}
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import stirling.software.proprietary.model.api.ai.AiPageText;
|
||||
|
||||
import tools.jackson.databind.JsonNode;
|
||||
|
||||
/**
|
||||
* Engine request for {@code POST /api/v1/docparse/extract}. {@code pages} drives the basic tier
|
||||
* (Java-extracted text); {@code contentBase64} lets the advanced tier parse the raw file itself.
|
||||
*/
|
||||
public record ExtractFieldsRequest(
|
||||
String fileName,
|
||||
JsonNode fieldsSchema,
|
||||
List<AiPageText> pages,
|
||||
String contentBase64,
|
||||
DocparseMode mode,
|
||||
String instructions) {}
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Engine response for {@code POST /api/v1/docparse/extract}. */
|
||||
public record ExtractFieldsResponse(
|
||||
DocparseTier mode, List<ExtractedField> fields, double overallConfidence) {
|
||||
|
||||
public ExtractFieldsResponse {
|
||||
fields = fields == null ? List.of() : fields;
|
||||
}
|
||||
}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/docparse/tables}. */
|
||||
public record ExtractTablesRequest(String fileName, String contentBase64) {}
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Engine response for {@code POST /api/v1/docparse/tables}. */
|
||||
public record ExtractTablesResponse(DocparseTier mode, List<DocTable> tables) {
|
||||
|
||||
public ExtractTablesResponse {
|
||||
tables = tables == null ? List.of() : tables;
|
||||
}
|
||||
}
|
||||
+16
@@ -0,0 +1,16 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import tools.jackson.databind.JsonNode;
|
||||
|
||||
/**
|
||||
* One extracted field with confidence and citations. Mirrors {@code docparse.py ExtractedField}.
|
||||
*/
|
||||
public record ExtractedField(
|
||||
String name, JsonNode value, double confidence, List<FieldCitation> citations) {
|
||||
|
||||
public ExtractedField {
|
||||
citations = citations == null ? List.of() : citations;
|
||||
}
|
||||
}
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Where a value came from. {@code quote} is always set; {@code bbox} only when a layout parse ran
|
||||
* (advanced tier); offsets index into the cited page's text. Mirrors {@code docparse.py
|
||||
* FieldCitation}.
|
||||
*/
|
||||
public record FieldCitation(
|
||||
Integer page, List<Double> bbox, String quote, Integer startOffset, Integer endOffset) {}
|
||||
+6
@@ -0,0 +1,6 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import tools.jackson.databind.JsonNode;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/docparse/fill-docx}. */
|
||||
public record FillDocxRequest(String templateBase64, JsonNode data) {}
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Engine response for {@code POST /api/v1/docparse/fill-docx}. */
|
||||
public record FillDocxResponse(String docxBase64, int replaced, List<String> missing) {
|
||||
|
||||
public FillDocxResponse {
|
||||
missing = missing == null ? List.of() : missing;
|
||||
}
|
||||
}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/docparse/parse}. */
|
||||
public record ParseDocumentRequest(String fileName, String contentBase64, boolean withOcr) {}
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Engine response for {@code POST /api/v1/docparse/parse}; also produced by the Java basic tier.
|
||||
*/
|
||||
public record ParseDocumentResponse(
|
||||
DocparseTier mode,
|
||||
int pages,
|
||||
List<DocBlock> blocks,
|
||||
List<DocTable> tables,
|
||||
String markdown,
|
||||
boolean ocrApplied) {
|
||||
|
||||
public ParseDocumentResponse {
|
||||
blocks = blocks == null ? List.of() : blocks;
|
||||
tables = tables == null ? List.of() : tables;
|
||||
markdown = markdown == null ? "" : markdown;
|
||||
}
|
||||
}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/documents/ask}: grounded Q&A over the RAG store. */
|
||||
public record RagAskRequest(String question, int topK) {}
|
||||
+29
@@ -0,0 +1,29 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.time.Instant;
|
||||
import java.util.List;
|
||||
|
||||
import stirling.software.proprietary.model.api.ai.AiPageText;
|
||||
|
||||
/**
|
||||
* Engine request for {@code POST /api/v1/docparse/rag-ingest}. Owner semantics mirror {@code POST
|
||||
* /api/v1/documents}: {@code ownerId} is the tenant, {@code readPrincipals} the explicit readers,
|
||||
* and a null {@code expiresAt} keeps the ingested content until an explicit delete. {@code index}
|
||||
* false skips the store (export-only); {@code includeMarkdown}/{@code includeChunks} echo the
|
||||
* parsed content back so the caller can emit corpus files.
|
||||
*/
|
||||
public record RagIngestRequest(
|
||||
String fileName,
|
||||
String documentId,
|
||||
String source,
|
||||
String ownerId,
|
||||
List<String> readPrincipals,
|
||||
Instant expiresAt,
|
||||
List<AiPageText> pages,
|
||||
String contentBase64,
|
||||
int chunkSize,
|
||||
int overlap,
|
||||
DocparseMode mode,
|
||||
boolean index,
|
||||
boolean includeMarkdown,
|
||||
boolean includeChunks) {}
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Engine response for {@code POST /api/v1/docparse/rag-ingest}. {@code markdown} and {@code chunks}
|
||||
* are only present when the request asked for them via includeMarkdown/includeChunks.
|
||||
*/
|
||||
public record RagIngestResponse(
|
||||
DocparseTier mode,
|
||||
String documentId,
|
||||
int chunksIndexed,
|
||||
int pages,
|
||||
String markdown,
|
||||
List<DocChunk> chunks) {}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/documents/search}: semantic search over the RAG store. */
|
||||
public record RagSearchRequest(String query, int topK) {}
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/**
|
||||
* Merged RAG store view served by {@code GET /api/v1/docparse/rag-stats} (Java side): the engine's
|
||||
* document-store totals plus the cached DocParse capability fields. When the engine is unreachable
|
||||
* the totals are zero and {@code engineReachable} is false.
|
||||
*/
|
||||
public record RagStatsView(
|
||||
String backend,
|
||||
long documents,
|
||||
long chunks,
|
||||
String embeddingModel,
|
||||
boolean advancedInstalled,
|
||||
String doclingVersion,
|
||||
boolean engineReachable) {}
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import stirling.software.proprietary.model.api.ai.AiPageText;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/docparse/split}. */
|
||||
public record SmartSplitRequest(
|
||||
String fileName, String rule, List<AiPageText> pages, int maxParts) {}
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Engine response for {@code POST /api/v1/docparse/split}. */
|
||||
public record SmartSplitResponse(List<SplitPart> parts) {
|
||||
|
||||
public SmartSplitResponse {
|
||||
parts = parts == null ? List.of() : parts;
|
||||
}
|
||||
}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** One sub-document page range (1-based, inclusive). Mirrors {@code docparse.py SplitPart}. */
|
||||
public record SplitPart(int startPage, int endPage, String label, double confidence) {}
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import stirling.software.proprietary.model.api.ai.AiPageText;
|
||||
|
||||
/**
|
||||
* Engine request for {@code POST /api/v1/docparse/suggest-schema}. {@code pages} drives the basic
|
||||
* tier (Java-extracted text); {@code contentBase64} lets the advanced tier parse the raw file.
|
||||
*/
|
||||
public record SuggestSchemaRequest(
|
||||
String fileName, List<AiPageText> pages, String contentBase64, int maxFields) {}
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Engine response for {@code POST /api/v1/docparse/suggest-schema}. */
|
||||
public record SuggestSchemaResponse(DocparseTier mode, List<SuggestedField> fields) {
|
||||
|
||||
public SuggestSchemaResponse {
|
||||
fields = fields == null ? List.of() : fields;
|
||||
}
|
||||
}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** One field the engine proposes for an extraction schema. */
|
||||
public record SuggestedField(String name, String type, String description) {}
|
||||
+51
@@ -0,0 +1,51 @@
|
||||
package stirling.software.proprietary.security.filter;
|
||||
|
||||
import java.io.IOException;
|
||||
|
||||
import org.slf4j.MDC;
|
||||
import org.springframework.core.Ordered;
|
||||
import org.springframework.core.annotation.Order;
|
||||
import org.springframework.security.core.Authentication;
|
||||
import org.springframework.security.core.context.SecurityContextHolder;
|
||||
import org.springframework.stereotype.Component;
|
||||
import org.springframework.web.filter.OncePerRequestFilter;
|
||||
|
||||
import jakarta.servlet.FilterChain;
|
||||
import jakarta.servlet.ServletException;
|
||||
import jakarta.servlet.http.HttpServletRequest;
|
||||
import jakarta.servlet.http.HttpServletResponse;
|
||||
|
||||
/**
|
||||
* Stamps the authenticated principal into MDC for every request. Async job workers resolve the
|
||||
* caller via UserService's MDC fallback; without this the fallback only worked when the audit
|
||||
* aspect (a pro feature) happened to populate it.
|
||||
*/
|
||||
@Component
|
||||
@Order(Ordered.LOWEST_PRECEDENCE)
|
||||
public class PrincipalMdcFilter extends OncePerRequestFilter {
|
||||
|
||||
static final String MDC_KEY = "auditPrincipal";
|
||||
|
||||
@Override
|
||||
protected void doFilterInternal(
|
||||
HttpServletRequest request, HttpServletResponse response, FilterChain filterChain)
|
||||
throws ServletException, IOException {
|
||||
String previous = MDC.get(MDC_KEY);
|
||||
Authentication authentication = SecurityContextHolder.getContext().getAuthentication();
|
||||
boolean stamped = false;
|
||||
if (previous == null
|
||||
&& authentication != null
|
||||
&& authentication.isAuthenticated()
|
||||
&& !"anonymousUser".equals(authentication.getPrincipal())) {
|
||||
MDC.put(MDC_KEY, authentication.getName());
|
||||
stamped = true;
|
||||
}
|
||||
try {
|
||||
filterChain.doFilter(request, response);
|
||||
} finally {
|
||||
if (stamped) {
|
||||
MDC.remove(MDC_KEY);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+7
@@ -254,6 +254,13 @@ public class AiEngineClient {
|
||||
|
||||
private void checkResponseStatus(HttpResponse<String> response) {
|
||||
int status = response.statusCode();
|
||||
// 501 = capability not implemented (e.g. docparse addon missing); keep the status and
|
||||
// body so callers can surface the machine-readable addonRequired detail.
|
||||
if (status == 501) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.NOT_IMPLEMENTED,
|
||||
"AI engine capability not implemented: " + response.body());
|
||||
}
|
||||
if (status >= 500) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.BAD_GATEWAY, "AI engine returned error: " + status);
|
||||
|
||||
+554
@@ -0,0 +1,554 @@
|
||||
package stirling.software.proprietary.service;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Base64;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.http.HttpStatus;
|
||||
import org.springframework.stereotype.Service;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
import org.springframework.web.server.ResponseStatusException;
|
||||
|
||||
import io.github.pixee.security.Filenames;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.common.model.ApplicationProperties;
|
||||
import stirling.software.common.service.CustomPDFDocumentFactory;
|
||||
import stirling.software.common.service.UserServiceInterface;
|
||||
import stirling.software.proprietary.model.api.ai.AiPageText;
|
||||
import stirling.software.proprietary.model.docparse.ChunkDocumentRequest;
|
||||
import stirling.software.proprietary.model.docparse.ChunkDocumentResponse;
|
||||
import stirling.software.proprietary.model.docparse.DocBlock;
|
||||
import stirling.software.proprietary.model.docparse.DocparseCapabilities;
|
||||
import stirling.software.proprietary.model.docparse.DocparseCapabilitiesView;
|
||||
import stirling.software.proprietary.model.docparse.DocparseMode;
|
||||
import stirling.software.proprietary.model.docparse.DocparseTier;
|
||||
import stirling.software.proprietary.model.docparse.DocumentStoreStats;
|
||||
import stirling.software.proprietary.model.docparse.ExtractFieldsRequest;
|
||||
import stirling.software.proprietary.model.docparse.ExtractFieldsResponse;
|
||||
import stirling.software.proprietary.model.docparse.ExtractTablesRequest;
|
||||
import stirling.software.proprietary.model.docparse.ExtractTablesResponse;
|
||||
import stirling.software.proprietary.model.docparse.FillDocxRequest;
|
||||
import stirling.software.proprietary.model.docparse.FillDocxResponse;
|
||||
import stirling.software.proprietary.model.docparse.ParseDocumentRequest;
|
||||
import stirling.software.proprietary.model.docparse.ParseDocumentResponse;
|
||||
import stirling.software.proprietary.model.docparse.RagAskRequest;
|
||||
import stirling.software.proprietary.model.docparse.RagIngestRequest;
|
||||
import stirling.software.proprietary.model.docparse.RagIngestResponse;
|
||||
import stirling.software.proprietary.model.docparse.RagSearchRequest;
|
||||
import stirling.software.proprietary.model.docparse.RagStatsView;
|
||||
import stirling.software.proprietary.model.docparse.SmartSplitRequest;
|
||||
import stirling.software.proprietary.model.docparse.SmartSplitResponse;
|
||||
import stirling.software.proprietary.model.docparse.SuggestSchemaRequest;
|
||||
import stirling.software.proprietary.model.docparse.SuggestSchemaResponse;
|
||||
|
||||
import tools.jackson.databind.JsonNode;
|
||||
import tools.jackson.databind.ObjectMapper;
|
||||
|
||||
/**
|
||||
* DocParse ingestion: per-page text extraction (reusing the same {@link PdfContentExtractor} the AI
|
||||
* chat path uses) plus engine dispatch for chunk + embed + index. The engine owns chunking,
|
||||
* embedding, and the document store; Java owns identity, limits, and the wire contract from {@code
|
||||
* engine/src/stirling/contracts/docparse.py}.
|
||||
*/
|
||||
@Slf4j
|
||||
@Service
|
||||
public class DocParseService {
|
||||
|
||||
private static final String PARSE_ENDPOINT = "/api/v1/docparse/parse";
|
||||
private static final String EXTRACT_ENDPOINT = "/api/v1/docparse/extract";
|
||||
private static final String SPLIT_ENDPOINT = "/api/v1/docparse/split";
|
||||
private static final String CHUNK_ENDPOINT = "/api/v1/docparse/chunk";
|
||||
private static final String TABLES_ENDPOINT = "/api/v1/docparse/tables";
|
||||
private static final String FILL_DOCX_ENDPOINT = "/api/v1/docparse/fill-docx";
|
||||
private static final String SUGGEST_SCHEMA_ENDPOINT = "/api/v1/docparse/suggest-schema";
|
||||
private static final String RAG_INGEST_ENDPOINT = "/api/v1/docparse/rag-ingest";
|
||||
private static final String DOCUMENT_STATS_ENDPOINT = "/api/v1/documents/stats";
|
||||
private static final String DOCUMENT_LIST_ENDPOINT = "/api/v1/documents/list";
|
||||
private static final String DOCUMENT_SEARCH_ENDPOINT = "/api/v1/documents/search";
|
||||
private static final String DOCUMENT_ASK_ENDPOINT = "/api/v1/documents/ask";
|
||||
|
||||
/** Below this average of extractable chars per page the document is treated as scanned. */
|
||||
static final int SCANNED_AVG_CHARS_PER_PAGE = 100;
|
||||
|
||||
/** Pages sampled for the scanned heuristic; keeps the probe cheap on huge documents. */
|
||||
private static final int SCANNED_SAMPLE_PAGES = 20;
|
||||
|
||||
private final AiEngineClient aiEngineClient;
|
||||
private final DocparseCapabilityService capabilityService;
|
||||
private final CustomPDFDocumentFactory pdfDocumentFactory;
|
||||
private final PdfContentExtractor pdfContentExtractor;
|
||||
private final ApplicationProperties applicationProperties;
|
||||
private final ObjectMapper objectMapper;
|
||||
private final FileIdStrategy fileIdStrategy;
|
||||
private final UserServiceInterface userService;
|
||||
|
||||
public DocParseService(
|
||||
AiEngineClient aiEngineClient,
|
||||
DocparseCapabilityService capabilityService,
|
||||
CustomPDFDocumentFactory pdfDocumentFactory,
|
||||
PdfContentExtractor pdfContentExtractor,
|
||||
ApplicationProperties applicationProperties,
|
||||
ObjectMapper objectMapper,
|
||||
FileIdStrategy fileIdStrategy,
|
||||
@Autowired(required = false) UserServiceInterface userService) {
|
||||
this.aiEngineClient = aiEngineClient;
|
||||
this.capabilityService = capabilityService;
|
||||
this.pdfDocumentFactory = pdfDocumentFactory;
|
||||
this.pdfContentExtractor = pdfContentExtractor;
|
||||
this.applicationProperties = applicationProperties;
|
||||
this.objectMapper = objectMapper;
|
||||
this.fileIdStrategy = fileIdStrategy;
|
||||
this.userService = userService;
|
||||
}
|
||||
|
||||
/** Throws 503 when the docparse.enabled master switch is off. */
|
||||
public void requireEnabled() {
|
||||
if (!applicationProperties.getDocparse().isEnabled()) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.SERVICE_UNAVAILABLE, "DocParse is disabled");
|
||||
}
|
||||
}
|
||||
|
||||
public DocparseCapabilitiesView capabilitiesView() {
|
||||
ApplicationProperties.Docparse config = applicationProperties.getDocparse();
|
||||
DocparseCapabilities capabilities = capabilityService.capabilities();
|
||||
return new DocparseCapabilitiesView(
|
||||
config.isEnabled(),
|
||||
config.getMode(),
|
||||
capabilities.advancedInstalled(),
|
||||
capabilityService.isEngineReachable(),
|
||||
capabilities.doclingVersion());
|
||||
}
|
||||
|
||||
public ParseDocumentResponse parse(
|
||||
MultipartFile file, DocparseMode requestedMode, boolean withOcr) throws IOException {
|
||||
requireEnabled();
|
||||
DocparseTier tier;
|
||||
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
|
||||
tier =
|
||||
resolveTier(
|
||||
requestedMode,
|
||||
capabilityService.capabilities(),
|
||||
false,
|
||||
looksScanned(document));
|
||||
if (tier == DocparseTier.BASIC) {
|
||||
return basicParse(document);
|
||||
}
|
||||
}
|
||||
ParseDocumentRequest request =
|
||||
new ParseDocumentRequest(fileName(file), encodeBase64(file), withOcr);
|
||||
String responseJson =
|
||||
aiEngineClient.postLongRunning(
|
||||
PARSE_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
|
||||
return objectMapper.readValue(responseJson, ParseDocumentResponse.class);
|
||||
}
|
||||
|
||||
public SmartSplitResponse split(MultipartFile file, String rule, int maxParts)
|
||||
throws IOException {
|
||||
requireEnabled();
|
||||
if (rule == null || rule.isBlank()) {
|
||||
throw new ResponseStatusException(HttpStatus.BAD_REQUEST, "A split rule is required");
|
||||
}
|
||||
List<AiPageText> pages;
|
||||
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
|
||||
pages = extractPages(document);
|
||||
}
|
||||
SmartSplitRequest request =
|
||||
new SmartSplitRequest(fileName(file), rule, pages, Math.clamp(maxParts, 1, 500));
|
||||
String responseJson =
|
||||
aiEngineClient.post(
|
||||
SPLIT_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
|
||||
return objectMapper.readValue(responseJson, SmartSplitResponse.class);
|
||||
}
|
||||
|
||||
public ChunkDocumentResponse chunk(
|
||||
MultipartFile file, int chunkSize, int overlap, DocparseMode mode) throws IOException {
|
||||
requireEnabled();
|
||||
List<AiPageText> pages;
|
||||
DocparseTier tier;
|
||||
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
|
||||
pages = extractPages(document);
|
||||
tier =
|
||||
resolveTier(
|
||||
mode, capabilityService.capabilities(), false, looksScanned(document));
|
||||
}
|
||||
ChunkDocumentRequest request =
|
||||
new ChunkDocumentRequest(
|
||||
fileName(file),
|
||||
pages,
|
||||
tier == DocparseTier.ADVANCED ? encodeBase64(file) : null,
|
||||
Math.clamp(chunkSize, 64, 32_768),
|
||||
Math.clamp(overlap, 0, 4_096),
|
||||
toMode(tier));
|
||||
String responseJson =
|
||||
aiEngineClient.postLongRunning(
|
||||
CHUNK_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
|
||||
return objectMapper.readValue(responseJson, ChunkDocumentResponse.class);
|
||||
}
|
||||
|
||||
/**
|
||||
* Chunk, embed, and index the document into the engine's RAG store, and/or echo the parsed
|
||||
* content back for corpus export. Text extraction and tier routing happen here; the engine
|
||||
* handles both basic (text-only) and advanced (layout) tiers.
|
||||
*/
|
||||
public RagIngestResponse ragIngest(
|
||||
MultipartFile file,
|
||||
String documentId,
|
||||
int chunkSize,
|
||||
int overlap,
|
||||
DocparseMode mode,
|
||||
boolean index,
|
||||
boolean includeMarkdown,
|
||||
boolean includeChunks)
|
||||
throws IOException {
|
||||
requireEnabled();
|
||||
if (!index && !includeMarkdown && !includeChunks) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.BAD_REQUEST,
|
||||
"Nothing to do: enable index, exportMarkdown, or exportChunksJsonl");
|
||||
}
|
||||
// Content hash default: re-ingesting identical bytes dedupes to the same document.
|
||||
String docId =
|
||||
(documentId == null || documentId.isBlank())
|
||||
? fileIdStrategy.idFor(file)
|
||||
: documentId.trim();
|
||||
List<AiPageText> pages;
|
||||
DocparseTier tier;
|
||||
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
|
||||
tier =
|
||||
resolveTier(
|
||||
mode, capabilityService.capabilities(), false, looksScanned(document));
|
||||
// The advanced tier parses the raw file engine-side; extracting pages
|
||||
// too would double both the PDFBox work and the payload.
|
||||
pages = tier == DocparseTier.BASIC ? extractPages(document) : null;
|
||||
}
|
||||
String callerId = currentUserId();
|
||||
// Null expiresAt = persistent until explicit delete; ingest here is a deliberate
|
||||
// knowledge-base action, unlike the TTL'd auto-ingest in AiWorkflowService.
|
||||
RagIngestRequest request =
|
||||
new RagIngestRequest(
|
||||
fileName(file),
|
||||
docId,
|
||||
fileName(file),
|
||||
callerId,
|
||||
// Engine forbids an empty list here (min_length=1); null means
|
||||
// "default to the owner" on the engine side.
|
||||
callerId == null ? null : List.of(callerId),
|
||||
null,
|
||||
pages,
|
||||
tier == DocparseTier.ADVANCED ? encodeBase64(file) : null,
|
||||
Math.clamp(chunkSize, 64, 32_768),
|
||||
Math.clamp(overlap, 0, 4_096),
|
||||
toMode(tier),
|
||||
index,
|
||||
includeMarkdown,
|
||||
includeChunks);
|
||||
String responseJson =
|
||||
aiEngineClient.postLongRunning(
|
||||
RAG_INGEST_ENDPOINT, objectMapper.writeValueAsString(request), callerId);
|
||||
return objectMapper.readValue(responseJson, RagIngestResponse.class);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the tier that will serve a request. The settings mode wins when stricter: a settings
|
||||
* {@code basic} always forces basic, a settings {@code advanced} upgrades everything except an
|
||||
* explicit basic request. {@code auto} picks advanced only when the addon is installed and the
|
||||
* document actually needs it (scanned, or the operation needs layout).
|
||||
*/
|
||||
DocparseTier resolveTier(
|
||||
DocparseMode requested,
|
||||
DocparseCapabilities capability,
|
||||
boolean needsLayout,
|
||||
boolean looksScanned) {
|
||||
DocparseMode effective = effectiveMode(settingsMode(), requested);
|
||||
return switch (effective) {
|
||||
case BASIC -> DocparseTier.BASIC;
|
||||
case ADVANCED -> requireAdvanced(capability);
|
||||
case AUTO ->
|
||||
capability.advancedInstalled() && (looksScanned || needsLayout)
|
||||
? DocparseTier.ADVANCED
|
||||
: DocparseTier.BASIC;
|
||||
};
|
||||
}
|
||||
|
||||
/** Extract the fields described by a JSON Schema, with confidence and citations. */
|
||||
public ExtractFieldsResponse extractFields(
|
||||
MultipartFile file, String fieldsSchemaJson, DocparseMode mode, String instructions)
|
||||
throws IOException {
|
||||
requireEnabled();
|
||||
JsonNode schema = parseJsonObject(fieldsSchemaJson, "fieldsSchema");
|
||||
List<AiPageText> pages;
|
||||
DocparseTier tier;
|
||||
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
|
||||
pages = extractPages(document);
|
||||
tier =
|
||||
resolveTier(
|
||||
mode, capabilityService.capabilities(), false, looksScanned(document));
|
||||
}
|
||||
ExtractFieldsRequest request =
|
||||
new ExtractFieldsRequest(
|
||||
fileName(file),
|
||||
schema,
|
||||
pages,
|
||||
tier == DocparseTier.ADVANCED ? encodeBase64(file) : null,
|
||||
toMode(tier),
|
||||
instructions);
|
||||
String responseJson =
|
||||
aiEngineClient.postLongRunning(
|
||||
EXTRACT_ENDPOINT,
|
||||
objectMapper.writeValueAsString(request),
|
||||
currentUserId());
|
||||
return objectMapper.readValue(responseJson, ExtractFieldsResponse.class);
|
||||
}
|
||||
|
||||
/** Propose an extraction schema from the document's first pages. */
|
||||
public SuggestSchemaResponse suggestSchema(MultipartFile file, int maxFields)
|
||||
throws IOException {
|
||||
requireEnabled();
|
||||
List<AiPageText> pages;
|
||||
DocparseTier tier;
|
||||
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
|
||||
pages = extractPages(document);
|
||||
tier =
|
||||
resolveTier(
|
||||
DocparseMode.AUTO,
|
||||
capabilityService.capabilities(),
|
||||
false,
|
||||
looksScanned(document));
|
||||
}
|
||||
SuggestSchemaRequest request =
|
||||
new SuggestSchemaRequest(
|
||||
fileName(file),
|
||||
pages,
|
||||
tier == DocparseTier.ADVANCED ? encodeBase64(file) : null,
|
||||
Math.clamp(maxFields, 1, 20));
|
||||
String responseJson =
|
||||
aiEngineClient.postLongRunning(
|
||||
SUGGEST_SCHEMA_ENDPOINT,
|
||||
objectMapper.writeValueAsString(request),
|
||||
currentUserId());
|
||||
return objectMapper.readValue(responseJson, SuggestSchemaResponse.class);
|
||||
}
|
||||
|
||||
private JsonNode parseJsonObject(String json, String fieldName) {
|
||||
if (json == null || json.isBlank()) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.BAD_REQUEST, "'" + fieldName + "' is required");
|
||||
}
|
||||
JsonNode node;
|
||||
try {
|
||||
node = objectMapper.readTree(json);
|
||||
} catch (Exception e) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.BAD_REQUEST, "'" + fieldName + "' is not valid JSON");
|
||||
}
|
||||
if (!node.isObject()) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.BAD_REQUEST, "'" + fieldName + "' must be a JSON object");
|
||||
}
|
||||
return node;
|
||||
}
|
||||
|
||||
private static void requireNonBlank(String value, String fieldName) {
|
||||
if (value == null || value.isBlank()) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.BAD_REQUEST, "'" + fieldName + "' is required");
|
||||
}
|
||||
}
|
||||
|
||||
/** Engine RAG store totals merged with the cached capability fields; graceful when down. */
|
||||
public RagStatsView ragStats() {
|
||||
DocparseCapabilities capabilities = capabilityService.capabilities();
|
||||
try {
|
||||
// The engine's documents routes are user-gated; an id-less probe 401s
|
||||
// and would read as "engine offline" in the UI.
|
||||
String json = aiEngineClient.get(DOCUMENT_STATS_ENDPOINT, currentUserId());
|
||||
DocumentStoreStats stats = objectMapper.readValue(json, DocumentStoreStats.class);
|
||||
return new RagStatsView(
|
||||
stats.backend(),
|
||||
stats.documents(),
|
||||
stats.chunks(),
|
||||
stats.embeddingModel(),
|
||||
capabilities.advancedInstalled(),
|
||||
capabilities.doclingVersion(),
|
||||
true);
|
||||
} catch (Exception e) {
|
||||
log.debug("RAG stats probe failed: {}", e.getMessage());
|
||||
return new RagStatsView(
|
||||
null,
|
||||
0,
|
||||
0,
|
||||
null,
|
||||
capabilities.advancedInstalled(),
|
||||
capabilities.doclingVersion(),
|
||||
false);
|
||||
}
|
||||
}
|
||||
|
||||
/** Engine document-list passthrough; X-User-Id scopes it to the caller's ACLs. */
|
||||
public String ragDocuments() throws IOException {
|
||||
requireEnabled();
|
||||
return aiEngineClient.get(DOCUMENT_LIST_ENDPOINT, currentUserId());
|
||||
}
|
||||
|
||||
/** Semantic-search passthrough over the caller-visible RAG documents. */
|
||||
public String ragSearch(String query, int topK) throws IOException {
|
||||
requireEnabled();
|
||||
requireNonBlank(query, "query");
|
||||
RagSearchRequest request = new RagSearchRequest(query, Math.clamp(topK, 1, 50));
|
||||
return aiEngineClient.post(
|
||||
DOCUMENT_SEARCH_ENDPOINT,
|
||||
objectMapper.writeValueAsString(request),
|
||||
currentUserId());
|
||||
}
|
||||
|
||||
/** Grounded-answer passthrough; long-running because local models answer slowly. */
|
||||
public String ragAsk(String question, int topK) throws IOException {
|
||||
requireEnabled();
|
||||
requireNonBlank(question, "question");
|
||||
RagAskRequest request = new RagAskRequest(question, Math.clamp(topK, 1, 20));
|
||||
return aiEngineClient.postLongRunning(
|
||||
DOCUMENT_ASK_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
|
||||
}
|
||||
|
||||
/**
|
||||
* The settings mode wins when stricter: a settings {@code basic} always forces basic, a
|
||||
* settings {@code advanced} upgrades everything except an explicit basic request.
|
||||
*/
|
||||
static DocparseMode effectiveMode(DocparseMode settings, DocparseMode requested) {
|
||||
DocparseMode request = requested == null ? DocparseMode.AUTO : requested;
|
||||
if (settings == DocparseMode.BASIC) {
|
||||
return DocparseMode.BASIC;
|
||||
}
|
||||
if (settings == DocparseMode.ADVANCED) {
|
||||
return request == DocparseMode.BASIC ? DocparseMode.BASIC : DocparseMode.ADVANCED;
|
||||
}
|
||||
return request;
|
||||
}
|
||||
|
||||
/** True when the sampled average of extractable chars per page falls below the threshold. */
|
||||
boolean looksScanned(PDDocument document) throws IOException {
|
||||
int pageCount = document.getNumberOfPages();
|
||||
if (pageCount == 0) {
|
||||
return false;
|
||||
}
|
||||
int sampled = Math.min(pageCount, SCANNED_SAMPLE_PAGES);
|
||||
long totalChars = 0;
|
||||
for (int page = 1; page <= sampled; page++) {
|
||||
String text = pdfContentExtractor.extractPageTextRaw(document, page);
|
||||
totalChars += text == null ? 0 : text.length();
|
||||
}
|
||||
return (totalChars / sampled) < SCANNED_AVG_CHARS_PER_PAGE;
|
||||
}
|
||||
|
||||
private DocparseTier requireAdvanced(DocparseCapabilities capability) {
|
||||
if (!capability.advancedInstalled()) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.NOT_IMPLEMENTED,
|
||||
"The advanced DocParse tier requires the docparse addon"
|
||||
+ " (addonRequired=docparse); install it or use mode=basic");
|
||||
}
|
||||
return DocparseTier.ADVANCED;
|
||||
}
|
||||
|
||||
private static DocparseMode toMode(DocparseTier tier) {
|
||||
return tier == DocparseTier.ADVANCED ? DocparseMode.ADVANCED : DocparseMode.BASIC;
|
||||
}
|
||||
|
||||
private static String encodeBase64(MultipartFile file) throws IOException {
|
||||
return Base64.getEncoder().encodeToString(file.getBytes());
|
||||
}
|
||||
|
||||
public ExtractTablesResponse tables(MultipartFile file) throws IOException {
|
||||
requireEnabled();
|
||||
// Tables need layout, so a forced-advanced setting without the addon must 501 here
|
||||
// rather than let the engine silently fall back; the engine picks the tier otherwise.
|
||||
resolveTier(DocparseMode.AUTO, capabilityService.capabilities(), true, false);
|
||||
ExtractTablesRequest request = new ExtractTablesRequest(fileName(file), encodeBase64(file));
|
||||
String responseJson =
|
||||
aiEngineClient.postLongRunning(
|
||||
TABLES_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
|
||||
return objectMapper.readValue(responseJson, ExtractTablesResponse.class);
|
||||
}
|
||||
|
||||
public FillDocxResponse fillDocx(MultipartFile templateFile, String dataJson)
|
||||
throws IOException {
|
||||
requireEnabled();
|
||||
JsonNode data = parseJsonObject(dataJson, "data");
|
||||
FillDocxRequest request =
|
||||
new FillDocxRequest(
|
||||
Base64.getEncoder().encodeToString(templateFile.getBytes()), data);
|
||||
String responseJson =
|
||||
aiEngineClient.post(
|
||||
FILL_DOCX_ENDPOINT,
|
||||
objectMapper.writeValueAsString(request),
|
||||
currentUserId());
|
||||
return objectMapper.readValue(responseJson, FillDocxResponse.class);
|
||||
}
|
||||
|
||||
/** Basic tier parse: PDFBox text layer only, one paragraph block per non-blank page. */
|
||||
ParseDocumentResponse basicParse(PDDocument document) throws IOException {
|
||||
int pageCount = document.getNumberOfPages();
|
||||
List<DocBlock> blocks = new ArrayList<>();
|
||||
StringBuilder markdown = new StringBuilder();
|
||||
for (int page = 1; page <= pageCount; page++) {
|
||||
String text = pdfContentExtractor.extractPageTextRaw(document, page);
|
||||
if (text == null || text.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
blocks.add(new DocBlock("paragraph", text, page, null, null));
|
||||
if (!markdown.isEmpty()) {
|
||||
markdown.append("\n\n");
|
||||
}
|
||||
markdown.append(text);
|
||||
}
|
||||
return new ParseDocumentResponse(
|
||||
DocparseTier.BASIC, pageCount, blocks, List.of(), markdown.toString(), false);
|
||||
}
|
||||
|
||||
/** Extract per-page text for the engine, capped by the shared aiEngine limits. */
|
||||
List<AiPageText> extractPages(PDDocument document) throws IOException {
|
||||
ApplicationProperties.AiEngine.Limits limits =
|
||||
applicationProperties.getAiEngine().getLimits();
|
||||
int maxPages = Math.min(document.getNumberOfPages(), limits.getMaxPages());
|
||||
int remainingCharacters = limits.getMaxCharacters();
|
||||
List<AiPageText> pages = new ArrayList<>();
|
||||
for (int page = 1; page <= maxPages && remainingCharacters > 0; page++) {
|
||||
String text = pdfContentExtractor.extractPageTextRaw(document, page);
|
||||
if (text == null || text.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
if (text.length() > remainingCharacters) {
|
||||
text = text.substring(0, remainingCharacters);
|
||||
}
|
||||
pages.add(new AiPageText(page, text));
|
||||
remainingCharacters -= text.length();
|
||||
}
|
||||
return pages;
|
||||
}
|
||||
|
||||
private DocparseMode settingsMode() {
|
||||
try {
|
||||
return DocparseMode.fromWire(applicationProperties.getDocparse().getMode());
|
||||
} catch (IllegalArgumentException e) {
|
||||
log.warn(
|
||||
"Unknown docparse.mode '{}'; falling back to auto",
|
||||
applicationProperties.getDocparse().getMode());
|
||||
return DocparseMode.AUTO;
|
||||
}
|
||||
}
|
||||
|
||||
public static String fileName(MultipartFile file) {
|
||||
String name = Filenames.toSimpleFileName(file.getOriginalFilename());
|
||||
return (name == null || name.isBlank()) ? "document.pdf" : name;
|
||||
}
|
||||
|
||||
private String currentUserId() {
|
||||
return userService != null ? userService.getCurrentUsername() : null;
|
||||
}
|
||||
}
|
||||
+123
@@ -0,0 +1,123 @@
|
||||
package stirling.software.proprietary.service;
|
||||
|
||||
import java.time.Duration;
|
||||
import java.time.Instant;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.common.model.ApplicationProperties;
|
||||
import stirling.software.common.service.DocparseCapabilityServiceInterface;
|
||||
import stirling.software.proprietary.model.docparse.DocparseCapabilities;
|
||||
|
||||
import tools.jackson.databind.ObjectMapper;
|
||||
|
||||
/**
|
||||
* Probes the engine's {@code GET /api/v1/docparse/capabilities} and caches the answer for 5
|
||||
* minutes. Reports "addon absent" when the AI engine is disabled or the probe fails, so callers can
|
||||
* always route to the basic tier without special-casing errors.
|
||||
*/
|
||||
@Slf4j
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class DocparseCapabilityService implements DocparseCapabilityServiceInterface {
|
||||
|
||||
private static final String CAPABILITIES_ENDPOINT = "/api/v1/docparse/capabilities";
|
||||
private static final Duration CACHE_TTL = Duration.ofMinutes(5);
|
||||
|
||||
private final AiEngineClient aiEngineClient;
|
||||
private final ApplicationProperties applicationProperties;
|
||||
private final ObjectMapper objectMapper;
|
||||
|
||||
private record Snapshot(
|
||||
DocparseCapabilities capabilities, boolean engineReachable, Instant fetchedAt) {}
|
||||
|
||||
private volatile Snapshot snapshot;
|
||||
private final AtomicBoolean refreshing = new AtomicBoolean();
|
||||
|
||||
/** The cached capabilities, refreshed synchronously when stale. */
|
||||
public DocparseCapabilities capabilities() {
|
||||
return freshSnapshot().capabilities();
|
||||
}
|
||||
|
||||
/** Whether the last capability probe reached the engine. */
|
||||
public boolean isEngineReachable() {
|
||||
Snapshot current = snapshot;
|
||||
return current != null && current.engineReachable();
|
||||
}
|
||||
|
||||
/** {@code refresh(true)} bypasses the cache and re-probes the engine now. */
|
||||
public DocparseCapabilities refresh(boolean force) {
|
||||
if (!force) {
|
||||
return capabilities();
|
||||
}
|
||||
Snapshot fresh = fetch();
|
||||
snapshot = fresh;
|
||||
return fresh.capabilities();
|
||||
}
|
||||
|
||||
/**
|
||||
* Non-blocking read for app-config: returns the last known value and kicks off a background
|
||||
* refresh when stale, so a slow/unreachable engine never delays page load.
|
||||
*/
|
||||
@Override
|
||||
public boolean isAdvancedInstalled() {
|
||||
Snapshot current = snapshot;
|
||||
if (current == null || isStale(current)) {
|
||||
triggerAsyncRefresh();
|
||||
}
|
||||
return current != null && current.capabilities().advancedInstalled();
|
||||
}
|
||||
|
||||
private Snapshot freshSnapshot() {
|
||||
Snapshot current = snapshot;
|
||||
if (current != null && !isStale(current)) {
|
||||
return current;
|
||||
}
|
||||
Snapshot fresh = fetch();
|
||||
snapshot = fresh;
|
||||
return fresh;
|
||||
}
|
||||
|
||||
private void triggerAsyncRefresh() {
|
||||
if (!refreshing.compareAndSet(false, true)) {
|
||||
return;
|
||||
}
|
||||
Thread.ofVirtual()
|
||||
.name("docparse-capability-refresh")
|
||||
.start(
|
||||
() -> {
|
||||
try {
|
||||
snapshot = fetch();
|
||||
} finally {
|
||||
refreshing.set(false);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
private Snapshot fetch() {
|
||||
if (!applicationProperties.getAiEngine().isEnabled()) {
|
||||
return new Snapshot(
|
||||
DocparseCapabilities.absent("AI engine is disabled"), false, Instant.now());
|
||||
}
|
||||
try {
|
||||
String json = aiEngineClient.get(CAPABILITIES_ENDPOINT, null);
|
||||
DocparseCapabilities capabilities =
|
||||
objectMapper.readValue(json, DocparseCapabilities.class);
|
||||
return new Snapshot(capabilities, true, Instant.now());
|
||||
} catch (Exception e) {
|
||||
log.debug("DocParse capability probe failed: {}", e.getMessage());
|
||||
return new Snapshot(
|
||||
DocparseCapabilities.absent("Capability probe failed: " + e.getMessage()),
|
||||
false,
|
||||
Instant.now());
|
||||
}
|
||||
}
|
||||
|
||||
private static boolean isStale(Snapshot current) {
|
||||
return current.fetchedAt().plus(CACHE_TTL).isBefore(Instant.now());
|
||||
}
|
||||
}
|
||||
+93
@@ -0,0 +1,93 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import stirling.software.proprietary.model.api.ai.AiPageText;
|
||||
|
||||
import tools.jackson.databind.JsonNode;
|
||||
import tools.jackson.databind.json.JsonMapper;
|
||||
|
||||
/**
|
||||
* The Java DTOs must serialize to exactly the camelCase wire shapes defined in {@code
|
||||
* engine/src/stirling/contracts/docparse.py}; a drift here breaks ingestion silently.
|
||||
*/
|
||||
class DocparseWireContractTest {
|
||||
|
||||
private final JsonMapper mapper = JsonMapper.builder().build();
|
||||
|
||||
@Test
|
||||
void ragIngestRequestSerializesTheEngineContract() {
|
||||
RagIngestRequest request =
|
||||
new RagIngestRequest(
|
||||
"report.pdf",
|
||||
"doc-1",
|
||||
"report.pdf",
|
||||
"user:alice",
|
||||
List.of("user:alice"),
|
||||
null,
|
||||
List.of(new AiPageText(1, "hello")),
|
||||
null,
|
||||
512,
|
||||
64,
|
||||
DocparseMode.AUTO,
|
||||
true,
|
||||
false,
|
||||
true);
|
||||
JsonNode json = mapper.readTree(mapper.writeValueAsString(request));
|
||||
assertEquals("report.pdf", json.get("fileName").asText());
|
||||
assertEquals("doc-1", json.get("documentId").asText());
|
||||
assertEquals("user:alice", json.get("ownerId").asText());
|
||||
assertEquals("user:alice", json.get("readPrincipals").get(0).asText());
|
||||
assertEquals(1, json.get("pages").get(0).get("pageNumber").asInt());
|
||||
assertEquals("hello", json.get("pages").get(0).get("text").asText());
|
||||
assertEquals(512, json.get("chunkSize").asInt());
|
||||
assertEquals(64, json.get("overlap").asInt());
|
||||
assertEquals("auto", json.get("mode").asText());
|
||||
assertTrue(json.get("index").asBoolean());
|
||||
assertFalse(json.get("includeMarkdown").asBoolean());
|
||||
assertTrue(json.get("includeChunks").asBoolean());
|
||||
}
|
||||
|
||||
@Test
|
||||
void ragIngestResponseReadsTheEngineShapeIncludingEchoedContent() {
|
||||
String engineJson =
|
||||
"{\"mode\":\"basic\",\"documentId\":\"doc-1\",\"chunksIndexed\":2,\"pages\":3,"
|
||||
+ "\"markdown\":\"# Title\",\"chunks\":[{\"index\":0,\"text\":\"t\","
|
||||
+ "\"pageStart\":1,\"pageEnd\":2,\"headingPath\":[\"Intro\"]}]}";
|
||||
RagIngestResponse response = mapper.readValue(engineJson, RagIngestResponse.class);
|
||||
assertEquals(DocparseTier.BASIC, response.mode());
|
||||
assertEquals("doc-1", response.documentId());
|
||||
assertEquals(2, response.chunksIndexed());
|
||||
assertEquals(3, response.pages());
|
||||
assertEquals("# Title", response.markdown());
|
||||
assertEquals(1, response.chunks().size());
|
||||
assertEquals(List.of("Intro"), response.chunks().get(0).headingPath());
|
||||
}
|
||||
|
||||
@Test
|
||||
void ragIngestResponseToleratesAbsentEchoFields() {
|
||||
String engineJson =
|
||||
"{\"mode\":\"basic\",\"documentId\":\"d\",\"chunksIndexed\":0,\"pages\":1}";
|
||||
RagIngestResponse response = mapper.readValue(engineJson, RagIngestResponse.class);
|
||||
assertNull(response.markdown());
|
||||
assertNull(response.chunks());
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesReadTheEngineProbeShape() {
|
||||
String engineJson =
|
||||
"{\"advancedInstalled\":false,\"doclingVersion\":null,\"torchVersion\":null,"
|
||||
+ "\"modelsAvailable\":false,\"modelsPath\":null,\"errors\":[\"missing\"]}";
|
||||
DocparseCapabilities capabilities =
|
||||
mapper.readValue(engineJson, DocparseCapabilities.class);
|
||||
assertFalse(capabilities.advancedInstalled());
|
||||
assertEquals(List.of("missing"), capabilities.errors());
|
||||
}
|
||||
}
|
||||
+78
@@ -0,0 +1,78 @@
|
||||
package stirling.software.proprietary.security.filter;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.MDC;
|
||||
import org.springframework.mock.web.MockHttpServletRequest;
|
||||
import org.springframework.mock.web.MockHttpServletResponse;
|
||||
import org.springframework.security.authentication.AnonymousAuthenticationToken;
|
||||
import org.springframework.security.authentication.UsernamePasswordAuthenticationToken;
|
||||
import org.springframework.security.core.authority.AuthorityUtils;
|
||||
import org.springframework.security.core.context.SecurityContextHolder;
|
||||
|
||||
class PrincipalMdcFilterTest {
|
||||
|
||||
private final PrincipalMdcFilter filter = new PrincipalMdcFilter();
|
||||
|
||||
@AfterEach
|
||||
void cleanUp() {
|
||||
SecurityContextHolder.clearContext();
|
||||
MDC.remove(PrincipalMdcFilter.MDC_KEY);
|
||||
}
|
||||
|
||||
@Test
|
||||
void stampsAuthenticatedPrincipalDuringTheChainAndClearsAfter() throws Exception {
|
||||
SecurityContextHolder.getContext()
|
||||
.setAuthentication(
|
||||
new UsernamePasswordAuthenticationToken(
|
||||
"alice", "n/a", AuthorityUtils.createAuthorityList("ROLE_USER")));
|
||||
String[] seen = new String[1];
|
||||
|
||||
filter.doFilter(
|
||||
new MockHttpServletRequest(),
|
||||
new MockHttpServletResponse(),
|
||||
(req, res) -> seen[0] = MDC.get(PrincipalMdcFilter.MDC_KEY));
|
||||
|
||||
assertEquals("alice", seen[0]);
|
||||
assertNull(MDC.get(PrincipalMdcFilter.MDC_KEY));
|
||||
}
|
||||
|
||||
@Test
|
||||
void anonymousRequestsAreNotStamped() throws Exception {
|
||||
SecurityContextHolder.getContext()
|
||||
.setAuthentication(
|
||||
new AnonymousAuthenticationToken(
|
||||
"key",
|
||||
"anonymousUser",
|
||||
AuthorityUtils.createAuthorityList("ROLE_ANONYMOUS")));
|
||||
String[] seen = new String[] {"sentinel"};
|
||||
|
||||
filter.doFilter(
|
||||
new MockHttpServletRequest(),
|
||||
new MockHttpServletResponse(),
|
||||
(req, res) -> seen[0] = MDC.get(PrincipalMdcFilter.MDC_KEY));
|
||||
|
||||
assertNull(seen[0]);
|
||||
}
|
||||
|
||||
@Test
|
||||
void existingMdcPrincipalIsLeftUntouched() throws Exception {
|
||||
MDC.put(PrincipalMdcFilter.MDC_KEY, "policy-run-owner");
|
||||
SecurityContextHolder.getContext()
|
||||
.setAuthentication(
|
||||
new UsernamePasswordAuthenticationToken(
|
||||
"alice", "n/a", AuthorityUtils.createAuthorityList("ROLE_USER")));
|
||||
String[] seen = new String[1];
|
||||
|
||||
filter.doFilter(
|
||||
new MockHttpServletRequest(),
|
||||
new MockHttpServletResponse(),
|
||||
(req, res) -> seen[0] = MDC.get(PrincipalMdcFilter.MDC_KEY));
|
||||
|
||||
assertEquals("policy-run-owner", seen[0]);
|
||||
assertEquals("policy-run-owner", MDC.get(PrincipalMdcFilter.MDC_KEY));
|
||||
}
|
||||
}
|
||||
+377
@@ -0,0 +1,377 @@
|
||||
package stirling.software.proprietary.service;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.mockito.ArgumentMatchers.any;
|
||||
import static org.mockito.ArgumentMatchers.anyBoolean;
|
||||
import static org.mockito.ArgumentMatchers.anyInt;
|
||||
import static org.mockito.ArgumentMatchers.eq;
|
||||
import static org.mockito.ArgumentMatchers.isNull;
|
||||
import static org.mockito.Mockito.verifyNoInteractions;
|
||||
import static org.mockito.Mockito.when;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.extension.ExtendWith;
|
||||
import org.mockito.ArgumentCaptor;
|
||||
import org.mockito.Mock;
|
||||
import org.mockito.junit.jupiter.MockitoExtension;
|
||||
import org.springframework.http.HttpStatus;
|
||||
import org.springframework.mock.web.MockMultipartFile;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
import org.springframework.web.server.ResponseStatusException;
|
||||
|
||||
import stirling.software.common.model.ApplicationProperties;
|
||||
import stirling.software.common.service.CustomPDFDocumentFactory;
|
||||
import stirling.software.proprietary.model.docparse.DocparseCapabilities;
|
||||
import stirling.software.proprietary.model.docparse.DocparseMode;
|
||||
import stirling.software.proprietary.model.docparse.DocparseTier;
|
||||
import stirling.software.proprietary.model.docparse.RagIngestResponse;
|
||||
|
||||
import tools.jackson.databind.JsonNode;
|
||||
import tools.jackson.databind.json.JsonMapper;
|
||||
|
||||
/** Wire-building and mode capping for the ingestion path. */
|
||||
@ExtendWith(MockitoExtension.class)
|
||||
class DocParseServiceTest {
|
||||
|
||||
private static final String ENGINE_RESPONSE =
|
||||
"{\"mode\":\"basic\",\"documentId\":\"doc-1\",\"chunksIndexed\":3,\"pages\":2,"
|
||||
+ "\"markdown\":null,\"chunks\":null}";
|
||||
|
||||
@Mock private AiEngineClient aiEngineClient;
|
||||
@Mock private DocparseCapabilityService capabilityService;
|
||||
@Mock private CustomPDFDocumentFactory pdfDocumentFactory;
|
||||
@Mock private PdfContentExtractor pdfContentExtractor;
|
||||
@Mock private FileIdStrategy fileIdStrategy;
|
||||
|
||||
private ApplicationProperties properties;
|
||||
private DocParseService service;
|
||||
private final JsonMapper jsonMapper = JsonMapper.builder().build();
|
||||
|
||||
@BeforeEach
|
||||
void setUp() {
|
||||
properties = new ApplicationProperties();
|
||||
service =
|
||||
new DocParseService(
|
||||
aiEngineClient,
|
||||
capabilityService,
|
||||
pdfDocumentFactory,
|
||||
pdfContentExtractor,
|
||||
properties,
|
||||
jsonMapper,
|
||||
fileIdStrategy,
|
||||
null);
|
||||
}
|
||||
|
||||
// --- effectiveMode: the settings mode wins when stricter ---
|
||||
|
||||
@Test
|
||||
void settingsBasicForcesBasic() {
|
||||
assertEquals(
|
||||
DocparseMode.BASIC,
|
||||
DocParseService.effectiveMode(DocparseMode.BASIC, DocparseMode.ADVANCED));
|
||||
assertEquals(
|
||||
DocparseMode.BASIC,
|
||||
DocParseService.effectiveMode(DocparseMode.BASIC, DocparseMode.AUTO));
|
||||
}
|
||||
|
||||
@Test
|
||||
void settingsAdvancedUpgradesAutoButRespectsExplicitBasic() {
|
||||
assertEquals(
|
||||
DocparseMode.ADVANCED,
|
||||
DocParseService.effectiveMode(DocparseMode.ADVANCED, DocparseMode.AUTO));
|
||||
assertEquals(
|
||||
DocparseMode.BASIC,
|
||||
DocParseService.effectiveMode(DocparseMode.ADVANCED, DocparseMode.BASIC));
|
||||
}
|
||||
|
||||
@Test
|
||||
void settingsAutoPassesRequestThroughAndNullMeansAuto() {
|
||||
assertEquals(
|
||||
DocparseMode.ADVANCED,
|
||||
DocParseService.effectiveMode(DocparseMode.AUTO, DocparseMode.ADVANCED));
|
||||
assertEquals(DocparseMode.AUTO, DocParseService.effectiveMode(DocparseMode.AUTO, null));
|
||||
}
|
||||
|
||||
// --- ragIngest wire building ---
|
||||
|
||||
private MultipartFile pdfFile() {
|
||||
return new MockMultipartFile("fileInput", "invoice.pdf", "application/pdf", new byte[] {1});
|
||||
}
|
||||
|
||||
private JsonNode ingestAndCaptureRequest(
|
||||
String documentId, boolean index, boolean markdown, boolean chunks) throws IOException {
|
||||
try (PDDocument document = new PDDocument()) {
|
||||
document.addPage(new PDPage());
|
||||
document.addPage(new PDPage());
|
||||
when(pdfDocumentFactory.load(any(MultipartFile.class), anyBoolean()))
|
||||
.thenReturn(document);
|
||||
when(pdfContentExtractor.extractPageTextRaw(any(), eq(1))).thenReturn("page one");
|
||||
when(pdfContentExtractor.extractPageTextRaw(any(), eq(2))).thenReturn("page two");
|
||||
when(capabilityService.capabilities()).thenReturn(absent());
|
||||
ArgumentCaptor<String> body = ArgumentCaptor.forClass(String.class);
|
||||
when(aiEngineClient.postLongRunning(
|
||||
eq("/api/v1/docparse/rag-ingest"), body.capture(), isNull()))
|
||||
.thenReturn(ENGINE_RESPONSE);
|
||||
|
||||
RagIngestResponse response =
|
||||
service.ragIngest(
|
||||
pdfFile(),
|
||||
documentId,
|
||||
512,
|
||||
64,
|
||||
DocparseMode.AUTO,
|
||||
index,
|
||||
markdown,
|
||||
chunks);
|
||||
assertEquals(3, response.chunksIndexed());
|
||||
return jsonMapper.readTree(body.getValue());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void ragIngestExtractsPagesAndSendsTheWireContract() throws IOException {
|
||||
JsonNode request = ingestAndCaptureRequest("doc-1", true, false, false);
|
||||
assertEquals("invoice.pdf", request.get("fileName").asText());
|
||||
assertEquals("doc-1", request.get("documentId").asText());
|
||||
// AUTO resolves to a concrete tier before the wire; without the addon that is basic.
|
||||
assertEquals("basic", request.get("mode").asText());
|
||||
assertEquals(2, request.get("pages").size());
|
||||
assertEquals("page one", request.get("pages").get(0).get("text").asText());
|
||||
assertEquals(512, request.get("chunkSize").asInt());
|
||||
assertEquals(64, request.get("overlap").asInt());
|
||||
assertTrue(request.get("index").asBoolean());
|
||||
assertFalse(request.get("includeMarkdown").asBoolean());
|
||||
assertFalse(request.get("includeChunks").asBoolean());
|
||||
}
|
||||
|
||||
@Test
|
||||
void ragIngestDefaultsDocumentIdToContentHash() throws IOException {
|
||||
when(fileIdStrategy.idFor(any(MultipartFile.class))).thenReturn("sha-abc");
|
||||
JsonNode request = ingestAndCaptureRequest(" ", true, false, false);
|
||||
assertEquals("sha-abc", request.get("documentId").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void ragIngestForwardsExportFlags() throws IOException {
|
||||
JsonNode request = ingestAndCaptureRequest("doc-1", false, true, true);
|
||||
assertFalse(request.get("index").asBoolean());
|
||||
assertTrue(request.get("includeMarkdown").asBoolean());
|
||||
assertTrue(request.get("includeChunks").asBoolean());
|
||||
}
|
||||
|
||||
@Test
|
||||
void ragIngestSettingsBasicCapsTheWireMode() throws IOException {
|
||||
properties.getDocparse().setMode("basic");
|
||||
JsonNode request = ingestAndCaptureRequest("doc-1", true, false, false);
|
||||
assertEquals("basic", request.get("mode").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void ragIngestWithNothingToDoIs400() {
|
||||
ResponseStatusException error =
|
||||
assertThrows(
|
||||
ResponseStatusException.class,
|
||||
() ->
|
||||
service.ragIngest(
|
||||
pdfFile(),
|
||||
"doc",
|
||||
512,
|
||||
64,
|
||||
DocparseMode.AUTO,
|
||||
false,
|
||||
false,
|
||||
false));
|
||||
assertEquals(HttpStatus.BAD_REQUEST, error.getStatusCode());
|
||||
verifyNoInteractions(aiEngineClient);
|
||||
}
|
||||
|
||||
@Test
|
||||
void ragIngestWhenDisabledIs503() {
|
||||
properties.getDocparse().setEnabled(false);
|
||||
ResponseStatusException error =
|
||||
assertThrows(
|
||||
ResponseStatusException.class,
|
||||
() ->
|
||||
service.ragIngest(
|
||||
pdfFile(),
|
||||
"doc",
|
||||
512,
|
||||
64,
|
||||
DocparseMode.AUTO,
|
||||
true,
|
||||
false,
|
||||
false));
|
||||
assertEquals(HttpStatus.SERVICE_UNAVAILABLE, error.getStatusCode());
|
||||
verifyNoInteractions(aiEngineClient);
|
||||
}
|
||||
|
||||
// --- tier resolution: auto and explicit requests ---
|
||||
|
||||
private DocparseCapabilities installed() {
|
||||
return new DocparseCapabilities(true, "2.55.0", "2.6.0", true, "/models", List.of());
|
||||
}
|
||||
|
||||
private DocparseCapabilities absent() {
|
||||
return new DocparseCapabilities(false, null, null, false, null, List.of());
|
||||
}
|
||||
|
||||
private void settingsMode(String mode) {
|
||||
properties.getDocparse().setMode(mode);
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoPicksAdvancedWhenScannedAndInstalled() {
|
||||
settingsMode("auto");
|
||||
assertEquals(
|
||||
DocparseTier.ADVANCED,
|
||||
service.resolveTier(DocparseMode.AUTO, installed(), false, true));
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoPicksAdvancedWhenLayoutNeededAndInstalled() {
|
||||
settingsMode("auto");
|
||||
assertEquals(
|
||||
DocparseTier.ADVANCED,
|
||||
service.resolveTier(DocparseMode.AUTO, installed(), true, false));
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoPicksBasicForBornDigitalDocument() {
|
||||
settingsMode("auto");
|
||||
assertEquals(
|
||||
DocparseTier.BASIC,
|
||||
service.resolveTier(DocparseMode.AUTO, installed(), false, false));
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoPicksBasicWhenAddonMissingEvenIfScanned() {
|
||||
settingsMode("auto");
|
||||
assertEquals(
|
||||
DocparseTier.BASIC, service.resolveTier(DocparseMode.AUTO, absent(), false, true));
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitBasicRequestAlwaysBasic() {
|
||||
settingsMode("auto");
|
||||
assertEquals(
|
||||
DocparseTier.BASIC,
|
||||
service.resolveTier(DocparseMode.BASIC, installed(), true, true));
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitAdvancedRequestUsesAdvancedWhenInstalled() {
|
||||
settingsMode("auto");
|
||||
assertEquals(
|
||||
DocparseTier.ADVANCED,
|
||||
service.resolveTier(DocparseMode.ADVANCED, installed(), false, false));
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitAdvancedRequestWithoutAddonReturns501() {
|
||||
settingsMode("auto");
|
||||
ResponseStatusException e =
|
||||
assertThrows(
|
||||
ResponseStatusException.class,
|
||||
() -> service.resolveTier(DocparseMode.ADVANCED, absent(), false, false));
|
||||
assertEquals(HttpStatus.NOT_IMPLEMENTED, e.getStatusCode());
|
||||
}
|
||||
|
||||
@Test
|
||||
void settingsBasicOverridesAdvancedRequest() {
|
||||
settingsMode("basic");
|
||||
assertEquals(
|
||||
DocparseTier.BASIC,
|
||||
service.resolveTier(DocparseMode.ADVANCED, installed(), true, true));
|
||||
}
|
||||
|
||||
@Test
|
||||
void settingsAdvancedUpgradesAutoRequest() {
|
||||
settingsMode("advanced");
|
||||
assertEquals(
|
||||
DocparseTier.ADVANCED,
|
||||
service.resolveTier(DocparseMode.AUTO, installed(), false, false));
|
||||
}
|
||||
|
||||
@Test
|
||||
void settingsAdvancedHonoursStricterBasicRequest() {
|
||||
settingsMode("advanced");
|
||||
assertEquals(
|
||||
DocparseTier.BASIC,
|
||||
service.resolveTier(DocparseMode.BASIC, installed(), false, false));
|
||||
}
|
||||
|
||||
@Test
|
||||
void settingsAdvancedWithoutAddonReturns501() {
|
||||
settingsMode("advanced");
|
||||
ResponseStatusException e =
|
||||
assertThrows(
|
||||
ResponseStatusException.class,
|
||||
() -> service.resolveTier(DocparseMode.AUTO, absent(), false, false));
|
||||
assertEquals(HttpStatus.NOT_IMPLEMENTED, e.getStatusCode());
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullRequestBehavesAsAuto() {
|
||||
settingsMode("auto");
|
||||
assertEquals(DocparseTier.BASIC, service.resolveTier(null, installed(), false, false));
|
||||
assertEquals(DocparseTier.ADVANCED, service.resolveTier(null, installed(), false, true));
|
||||
}
|
||||
|
||||
// --- scanned heuristic ---
|
||||
|
||||
@Test
|
||||
void looksScannedWhenAveragePageTextBelowThreshold() throws IOException {
|
||||
try (PDDocument document = new PDDocument()) {
|
||||
document.addPage(new PDPage());
|
||||
document.addPage(new PDPage());
|
||||
when(pdfContentExtractor.extractPageTextRaw(eq(document), anyInt()))
|
||||
.thenReturn("short");
|
||||
assertTrue(service.looksScanned(document));
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void doesNotLookScannedWithRealTextLayer() throws IOException {
|
||||
try (PDDocument document = new PDDocument()) {
|
||||
document.addPage(new PDPage());
|
||||
document.addPage(new PDPage());
|
||||
when(pdfContentExtractor.extractPageTextRaw(eq(document), anyInt()))
|
||||
.thenReturn("x".repeat(500));
|
||||
assertFalse(service.looksScanned(document));
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void fileNameFallsBackWhenMissing() {
|
||||
MultipartFile nameless =
|
||||
new MockMultipartFile("fileInput", "", "application/pdf", new byte[] {1});
|
||||
assertEquals("document.pdf", DocParseService.fileName(nameless));
|
||||
assertEquals("invoice.pdf", DocParseService.fileName(pdfFile()));
|
||||
}
|
||||
|
||||
@Test
|
||||
void extractPagesSkipsBlankPagesAndCapsCharacters() throws IOException {
|
||||
properties.getAiEngine().getLimits().setMaxCharacters(12);
|
||||
try (PDDocument document = new PDDocument()) {
|
||||
document.addPage(new PDPage());
|
||||
document.addPage(new PDPage());
|
||||
document.addPage(new PDPage());
|
||||
when(pdfContentExtractor.extractPageTextRaw(any(), eq(1))).thenReturn("0123456789");
|
||||
when(pdfContentExtractor.extractPageTextRaw(any(), eq(2))).thenReturn(" ");
|
||||
when(pdfContentExtractor.extractPageTextRaw(any(), eq(3))).thenReturn("abcdef");
|
||||
var pages = service.extractPages(document);
|
||||
assertEquals(2, pages.size());
|
||||
// Page 3 is truncated to the remaining budget (12 - 10 = 2 chars).
|
||||
assertEquals("ab", pages.get(1).getText());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
# Stirling-PDF + AI engine with the DocParse addon.
|
||||
# Two ways to get the addon; pick ONE per deployment:
|
||||
# 1. Dynamic install (default here): standard engine image downloads the addon
|
||||
# (~1.6 GB, one-time) into the docparse-data volume at first boot.
|
||||
# 2. Baked image (air-gapped): build the engine with `--build-arg DOCPARSE=true`
|
||||
# (uncomment DOCPARSE below) and drop DOCPARSE_AUTO_INSTALL.
|
||||
services:
|
||||
stirling-pdf:
|
||||
build:
|
||||
context: ../..
|
||||
dockerfile: docker/embedded/Dockerfile.fat
|
||||
container_name: stirling-pdf-docparse
|
||||
restart: unless-stopped
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -f http://localhost:8080$${SYSTEM_ROOTURIPATH:-''}/api/v1/info/status | grep -q 'UP'"]
|
||||
interval: 5s
|
||||
timeout: 10s
|
||||
retries: 16
|
||||
ports:
|
||||
- "8080:8080"
|
||||
volumes:
|
||||
- ../../stirling/latest/data:/usr/share/tessdata:rw
|
||||
- ../../stirling/latest/config:/configs:rw
|
||||
- ../../stirling/latest/logs:/logs:rw
|
||||
environment:
|
||||
SECURITY_ENABLELOGIN: "false"
|
||||
SYSTEM_DEFAULTLOCALE: en-US
|
||||
AIENGINE_ENABLED: "true"
|
||||
AIENGINE_URL: http://stirling-pdf-engine:5001
|
||||
DOCPARSE_ENABLED: "true"
|
||||
DOCPARSE_MODE: auto
|
||||
STIRLING_ENGINE_SHARED_SECRET: change-me
|
||||
depends_on:
|
||||
- stirling-pdf-engine
|
||||
networks:
|
||||
- stirling-network
|
||||
|
||||
stirling-pdf-engine:
|
||||
build:
|
||||
context: ../../engine
|
||||
# args:
|
||||
# DOCPARSE: "true" # bake the addon + models into the image instead
|
||||
container_name: stirling-pdf-engine
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
- docparse-data:/configs/docparse:rw
|
||||
environment:
|
||||
DOCPARSE_AUTO_INSTALL: "true"
|
||||
STIRLING_ENGINE_REQUIRE_AUTH: "true"
|
||||
STIRLING_ENGINE_SHARED_SECRET: change-me
|
||||
networks:
|
||||
- stirling-network
|
||||
|
||||
networks:
|
||||
stirling-network:
|
||||
driver: bridge
|
||||
|
||||
volumes:
|
||||
docparse-data:
|
||||
+11
-1
@@ -15,11 +15,19 @@ WORKDIR /app/engine
|
||||
|
||||
COPY pyproject.toml uv.lock .env ./
|
||||
COPY scripts/ ./scripts/
|
||||
# DOCPARSE=true bakes the docparse addon (Docling + CPU torch, ~1.6 GB) and its
|
||||
# model weights into the image for air-gapped deployments; default stays lean.
|
||||
ARG DOCPARSE=false
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen --no-dev
|
||||
if [ "$DOCPARSE" = "true" ]; then uv sync --frozen --no-dev --extra docparse; else uv sync --frozen --no-dev; fi
|
||||
|
||||
COPY src/ ./src/
|
||||
|
||||
RUN if [ "$DOCPARSE" = "true" ]; then \
|
||||
mkdir -p /opt/docparse/models \
|
||||
&& .venv/bin/python scripts/prefetch_docparse_models.py --output /opt/docparse/models; \
|
||||
fi
|
||||
|
||||
WORKDIR /app
|
||||
COPY Taskfile.yml ./
|
||||
COPY .taskfiles/ ./.taskfiles/
|
||||
@@ -34,4 +42,6 @@ ENV STIRLING_ENGINE_PORT=5001
|
||||
|
||||
EXPOSE 5001
|
||||
|
||||
RUN chmod +x /app/engine/scripts/docker-entrypoint.sh /app/engine/scripts/init_docparse.sh
|
||||
ENTRYPOINT ["/app/engine/scripts/docker-entrypoint.sh"]
|
||||
CMD ["task", "engine:run"]
|
||||
|
||||
@@ -14,12 +14,24 @@ dependencies = [
|
||||
"pydantic-ai-slim[voyageai]>=1.99.0,<2.0.0",
|
||||
"pydantic-settings>=2.0.0",
|
||||
"python-dotenv>=1.2.1",
|
||||
# Small (MIT) and always installed: DOCX template filling needs no addon.
|
||||
"python-docx>=1.1.2",
|
||||
"sqlite-vec>=0.1.6",
|
||||
"uvicorn>=0.35.0",
|
||||
"opentelemetry-sdk>=1.39.0",
|
||||
"posthog>=3.0.0",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
# DocParse advanced tier (layout parsing, tables, OCR, bbox citations).
|
||||
# ~1.6 GB installed with CPU torch; never in the default image. Delivered via
|
||||
# the addon engine image or the runtime dynamic install.
|
||||
docparse = [
|
||||
"docling>=2.55.0",
|
||||
"torch>=2.6.0",
|
||||
"torchvision>=0.21.0",
|
||||
]
|
||||
|
||||
[dependency-groups]
|
||||
dev = [
|
||||
"anyio>=4.0.0",
|
||||
@@ -34,6 +46,17 @@ dev = [
|
||||
requires = ["hatchling"]
|
||||
build-backend = "hatchling.build"
|
||||
|
||||
# Linux installs of the docparse extra must never pull CUDA wheels (~7 GB);
|
||||
# pin torch to the CPU index there. Windows/macOS PyPI wheels are already CPU.
|
||||
[[tool.uv.index]]
|
||||
name = "pytorch-cpu"
|
||||
url = "https://download.pytorch.org/whl/cpu"
|
||||
explicit = true
|
||||
|
||||
[tool.uv.sources]
|
||||
torch = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
|
||||
torchvision = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
|
||||
|
||||
[tool.hatch.build.targets.wheel]
|
||||
packages = ["src"]
|
||||
exclude = [
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
#!/bin/sh
|
||||
# Engine entrypoint: resolve where the docparse addon lives (baked vs dynamic)
|
||||
# before handing off to the server command.
|
||||
set -e
|
||||
|
||||
# Baked addon image (--build-arg DOCPARSE=true) prefetches models here.
|
||||
if [ -z "${STIRLING_DOCPARSE_HOME:-}" ] && [ -d /opt/docparse/models ]; then
|
||||
export STIRLING_DOCPARSE_HOME=/opt/docparse
|
||||
fi
|
||||
|
||||
if [ "${DOCPARSE_AUTO_INSTALL:-false}" = "true" ]; then
|
||||
export STIRLING_DOCPARSE_HOME="${STIRLING_DOCPARSE_HOME:-/configs/docparse}"
|
||||
/app/engine/scripts/init_docparse.sh || echo "[docparse] dynamic install failed; continuing with basic tier" >&2
|
||||
fi
|
||||
|
||||
exec "$@"
|
||||
@@ -0,0 +1,43 @@
|
||||
#!/bin/sh
|
||||
# Dynamic docparse install: put the addon's locked package delta into
|
||||
# $STIRLING_DOCPARSE_HOME/site (a volume, so it survives image upgrades) and
|
||||
# prefetch model weights into $STIRLING_DOCPARSE_HOME/models.
|
||||
# Idempotent: keyed on a marker derived from uv.lock, re-runs after upgrades.
|
||||
set -e
|
||||
|
||||
ENGINE_DIR=/app/engine
|
||||
HOME_DIR="${STIRLING_DOCPARSE_HOME:-/configs/docparse}"
|
||||
SITE_DIR="$HOME_DIR/site"
|
||||
MODELS_DIR="$HOME_DIR/models"
|
||||
PYTHON="$ENGINE_DIR/.venv/bin/python"
|
||||
|
||||
mkdir -p "$SITE_DIR" "$MODELS_DIR"
|
||||
|
||||
LOCK_HASH=$(sha256sum "$ENGINE_DIR/uv.lock" | cut -c1-16)
|
||||
MARKER="$HOME_DIR/.installed-$LOCK_HASH"
|
||||
|
||||
if [ ! -f "$MARKER" ]; then
|
||||
echo "[docparse] installing addon packages into $SITE_DIR (one-time, ~1.6 GB)"
|
||||
cd "$ENGINE_DIR"
|
||||
# Delta = locked docparse resolution minus what the base venv already has.
|
||||
uv export --frozen --no-dev --no-emit-project --no-hashes -o /tmp/docparse-base.req
|
||||
uv export --frozen --extra docparse --no-dev --no-emit-project --no-hashes -o /tmp/docparse-full.req
|
||||
grep -vxFf /tmp/docparse-base.req /tmp/docparse-full.req > /tmp/docparse-delta.req || true
|
||||
uv pip install \
|
||||
--python "$PYTHON" \
|
||||
--target "$SITE_DIR" \
|
||||
--no-deps \
|
||||
--extra-index-url https://download.pytorch.org/whl/cpu \
|
||||
--index-strategy unsafe-best-match \
|
||||
-r /tmp/docparse-delta.req
|
||||
rm -f "$HOME_DIR"/.installed-* /tmp/docparse-*.req
|
||||
touch "$MARKER"
|
||||
fi
|
||||
|
||||
if [ ! -d "$MODELS_DIR/docling" ] && [ -z "$(ls -A "$MODELS_DIR" 2>/dev/null)" ]; then
|
||||
echo "[docparse] prefetching model weights into $MODELS_DIR"
|
||||
PYTHONPATH="$SITE_DIR" "$PYTHON" "$ENGINE_DIR/scripts/prefetch_docparse_models.py" --output "$MODELS_DIR" \
|
||||
|| echo "[docparse] model prefetch failed; docling will fetch into its cache on first use" >&2
|
||||
fi
|
||||
|
||||
echo "[docparse] ready (site=$SITE_DIR, models=$MODELS_DIR)"
|
||||
@@ -0,0 +1,39 @@
|
||||
"""Prefetch Docling model weights for offline/air-gapped docparse.
|
||||
|
||||
Usage:
|
||||
uv run --extra docparse python scripts/prefetch_docparse_models.py --output /configs/docparse/models
|
||||
|
||||
The output directory is what STIRLING_DOCPARSE_HOME/models points at; the
|
||||
parser passes it to Docling as ``artifacts_path`` so no network is touched at
|
||||
request time.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import importlib
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--output", required=True, help="Directory to download model weights into")
|
||||
args = parser.parse_args()
|
||||
|
||||
try:
|
||||
# importlib keeps this file typecheckable without the docparse extra installed
|
||||
downloader = importlib.import_module("docling.utils.model_downloader")
|
||||
except ImportError:
|
||||
print("docling is not installed; run with: uv run --extra docparse ...", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
output = Path(args.output)
|
||||
output.mkdir(parents=True, exist_ok=True)
|
||||
path = downloader.download_models(output_dir=output, progress=True)
|
||||
print(f"docparse models ready at {path}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from .document_classifier import DocumentClassifierAgent
|
||||
from .execution import ExecutionPlanningAgent
|
||||
from .knowledge_ask import KnowledgeAskAgent
|
||||
from .orchestrator import OrchestratorAgent
|
||||
from .pdf_create import PdfCreateAgent
|
||||
from .pdf_edit import PdfEditAgent, PdfEditParameterSelector, PdfEditPlanSelection
|
||||
@@ -12,6 +13,7 @@ from .user_spec import UserSpecAgent
|
||||
__all__ = [
|
||||
"DocumentClassifierAgent",
|
||||
"ExecutionPlanningAgent",
|
||||
"KnowledgeAskAgent",
|
||||
"OrchestratorAgent",
|
||||
"PdfCreateAgent",
|
||||
"PdfEditAgent",
|
||||
|
||||
@@ -0,0 +1,134 @@
|
||||
"""Grounded Q&A over the caller's stored documents.
|
||||
|
||||
Retrieval runs the same cross-collection search as ``POST /documents/search``;
|
||||
one smart-model pass then answers only from the retrieved passages, citing
|
||||
document and page inline. No retrieval hit means a plain "not found" answer.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from pydantic import Field
|
||||
from pydantic_ai import Agent
|
||||
|
||||
from stirling.agents.output_mode import output_retries, structured_output
|
||||
from stirling.contracts import AskDocumentsRequest, AskDocumentsResponse, DocumentPassage
|
||||
from stirling.documents import CollectionSearchHit
|
||||
from stirling.documents.service import PAGE_NUMBER_METADATA_KEY
|
||||
from stirling.models import ApiModel, PrincipalId
|
||||
from stirling.services import AppRuntime
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Metadata keys written by docparse rag-ingest (_chunk_metadata) for structure-aware chunks.
|
||||
_PAGE_START_KEY = "page_start"
|
||||
_PAGE_END_KEY = "page_end"
|
||||
_HEADING_PATH_KEY = "heading_path"
|
||||
_HEADING_PATH_SEPARATOR = " > "
|
||||
|
||||
_SYSTEM_PROMPT = (
|
||||
"You answer questions using ONLY the numbered passages you are given.\n"
|
||||
"\n"
|
||||
"Rules:\n"
|
||||
"- Every statement must come from the passages. Never use outside knowledge, never guess.\n"
|
||||
"- Cite the document and page inline right after each fact, "
|
||||
'e.g. "(invoice.pdf p.2)" or "(report.pdf p.4-6)", using the names and pages '
|
||||
"shown in each passage header.\n"
|
||||
"- If the passages do not answer the question, say plainly that the stored "
|
||||
"documents do not cover it. Do not attempt a partial guess.\n"
|
||||
"- Answer in the same language as the question."
|
||||
)
|
||||
|
||||
_NO_PASSAGES_ANSWER = "I couldn't find anything relevant to that question in your stored documents."
|
||||
|
||||
|
||||
class _AskOutput(ApiModel):
|
||||
"""Raw model answer for the single ask pass."""
|
||||
|
||||
answer: str = Field(description="The answer grounded in the passages, with inline citations.")
|
||||
|
||||
|
||||
def _meta_int(value: str | None) -> int | None:
|
||||
if value is None:
|
||||
return None
|
||||
try:
|
||||
return int(value)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def passage_from_hit(hit: CollectionSearchHit) -> DocumentPassage:
|
||||
"""Map a store search hit onto the wire passage shape.
|
||||
|
||||
Docparse chunks carry page bounds and a heading path; plain page-text
|
||||
chunks only carry ``page_number``, which maps to both bounds.
|
||||
"""
|
||||
meta = hit.result.document.metadata
|
||||
page_start = _meta_int(meta.get(_PAGE_START_KEY))
|
||||
page_end = _meta_int(meta.get(_PAGE_END_KEY))
|
||||
if page_start is None and page_end is None:
|
||||
page_start = page_end = _meta_int(meta.get(PAGE_NUMBER_METADATA_KEY))
|
||||
heading = meta.get(_HEADING_PATH_KEY)
|
||||
source = meta.get("source")
|
||||
if source and ":page:" in source:
|
||||
# Page-text chunk sources look like "report.pdf:page:3"; show the file name.
|
||||
source = source.rsplit(":page:", 1)[0]
|
||||
return DocumentPassage(
|
||||
document_id=hit.collection,
|
||||
text=hit.result.document.text,
|
||||
score=hit.result.score,
|
||||
page_start=page_start,
|
||||
page_end=page_end,
|
||||
heading_path=heading.split(_HEADING_PATH_SEPARATOR) if heading else [],
|
||||
source=source or None,
|
||||
)
|
||||
|
||||
|
||||
def format_passages(passages: list[DocumentPassage]) -> str:
|
||||
"""Render passages for the prompt with the citation handle in each header."""
|
||||
return "\n\n".join(_format_passage(i, passage) for i, passage in enumerate(passages, 1))
|
||||
|
||||
|
||||
def _format_passage(index: int, passage: DocumentPassage) -> str:
|
||||
name = passage.source or passage.document_id
|
||||
if passage.page_start is None:
|
||||
pages = ""
|
||||
elif passage.page_end is not None and passage.page_end != passage.page_start:
|
||||
pages = f" p.{passage.page_start}-{passage.page_end}"
|
||||
else:
|
||||
pages = f" p.{passage.page_start}"
|
||||
return f"[Passage {index} | {name}{pages}]\n{passage.text}"
|
||||
|
||||
|
||||
class KnowledgeAskAgent:
|
||||
"""Answers a question from the caller's stored documents.
|
||||
|
||||
Retrieves the top passages the caller can read (same path as the search
|
||||
endpoint), then runs one smart-model pass over just those passages.
|
||||
"""
|
||||
|
||||
def __init__(self, runtime: AppRuntime) -> None:
|
||||
self.runtime = runtime
|
||||
# Ollama/custom block tool-calling under native json-schema output; see agents.output_mode.
|
||||
provider = runtime.settings.chat_provider
|
||||
self._agent: Agent[None, _AskOutput] = Agent(
|
||||
model=runtime.smart_model,
|
||||
output_type=structured_output([_AskOutput], chat_provider=provider),
|
||||
system_prompt=_SYSTEM_PROMPT,
|
||||
model_settings=runtime.smart_model_settings,
|
||||
retries=output_retries(provider),
|
||||
)
|
||||
|
||||
async def ask(self, request: AskDocumentsRequest, principals: list[PrincipalId]) -> AskDocumentsResponse:
|
||||
hits = await self.runtime.documents.search_with_collections(
|
||||
request.question, principals=principals, top_k=request.top_k
|
||||
)
|
||||
passages = [passage_from_hit(hit) for hit in hits]
|
||||
if not passages:
|
||||
logger.info("[knowledge-ask] question=%r -> 0 passages", request.question)
|
||||
return AskDocumentsResponse(answer=_NO_PASSAGES_ANSWER, passages=[])
|
||||
prompt = f"Question: {request.question}\n\nPassages:\n{format_passages(passages)}"
|
||||
logger.debug("[knowledge-ask] prompt:\n%s", prompt)
|
||||
result = await self._agent.run(prompt)
|
||||
return AskDocumentsResponse(answer=result.output.answer, passages=passages)
|
||||
@@ -17,6 +17,7 @@ from stirling.api.routes import (
|
||||
agent_capabilities_router,
|
||||
agent_draft_router,
|
||||
config_router,
|
||||
docparse_router,
|
||||
document_classifier_router,
|
||||
document_router,
|
||||
execution_router,
|
||||
@@ -30,6 +31,7 @@ from stirling.api.routes.config import CONFIG_APPLY_ERRORS, apply_to_app, resolv
|
||||
from stirling.config import AppSettings, load_settings
|
||||
from stirling.config.config_cache import cache_stamp, load_config
|
||||
from stirling.contracts import HealthResponse
|
||||
from stirling.docparse import activate_site
|
||||
from stirling.documents import DocumentService, EmbeddingService
|
||||
from stirling.services import setup_posthog_tracking
|
||||
|
||||
@@ -145,6 +147,8 @@ def _restore_cached_config(
|
||||
async def lifespan(fast_api: FastAPI):
|
||||
# Load env vars on startup so we can immediately crash if required env vars aren't set
|
||||
settings = _load_startup_settings(fast_api)
|
||||
# Initialize docparse addon if available (engine boot-time setup).
|
||||
activate_site(settings.docparse_home)
|
||||
# Precedence: env < persisted cache < live push. Stamp first so a push landing mid-boot
|
||||
# is re-adopted by the watcher rather than mistaken for the config we just restored.
|
||||
fast_api.state.config_cache_stamp = cache_stamp()
|
||||
@@ -217,6 +221,7 @@ app.include_router(ledger_router, dependencies=_user_gate)
|
||||
app.include_router(pdf_comments_router, dependencies=_user_gate)
|
||||
app.include_router(agent_capabilities_router, dependencies=_user_gate)
|
||||
app.include_router(document_classifier_router, dependencies=_user_gate)
|
||||
app.include_router(docparse_router, dependencies=_user_gate)
|
||||
# Config push is a system sync with no X-User-Id, so it is guarded by the shared secret
|
||||
# and allow_config_push flag only, deliberately NOT the per-user identity gate.
|
||||
app.include_router(config_router)
|
||||
|
||||
@@ -10,6 +10,7 @@ from pydantic_ai.models import Model
|
||||
from stirling.agents import (
|
||||
DocumentClassifierAgent,
|
||||
ExecutionPlanningAgent,
|
||||
KnowledgeAskAgent,
|
||||
OrchestratorAgent,
|
||||
PdfEditAgent,
|
||||
PdfQuestionAgent,
|
||||
@@ -18,6 +19,7 @@ from stirling.agents import (
|
||||
from stirling.agents.ledger import MathAuditorAgent
|
||||
from stirling.agents.pdf_comment import PdfCommentAgent
|
||||
from stirling.config import AppSettings
|
||||
from stirling.docparse import ExtractFieldsAgent, SmartSplitAgent, SuggestSchemaAgent
|
||||
from stirling.documents import DocumentService, EmbeddingService
|
||||
from stirling.services import AppRuntime, build_runtime
|
||||
|
||||
@@ -35,6 +37,10 @@ class AppState:
|
||||
math_auditor_agent: MathAuditorAgent
|
||||
pdf_comment_agent: PdfCommentAgent
|
||||
document_classifier_agent: DocumentClassifierAgent
|
||||
knowledge_ask_agent: KnowledgeAskAgent
|
||||
extract_fields_agent: ExtractFieldsAgent
|
||||
smart_split_agent: SmartSplitAgent
|
||||
suggest_schema_agent: SuggestSchemaAgent
|
||||
|
||||
|
||||
def build_app_state(
|
||||
@@ -63,6 +69,10 @@ def build_app_state(
|
||||
math_auditor_agent=MathAuditorAgent(runtime),
|
||||
pdf_comment_agent=PdfCommentAgent(runtime),
|
||||
document_classifier_agent=DocumentClassifierAgent(runtime),
|
||||
knowledge_ask_agent=KnowledgeAskAgent(runtime),
|
||||
extract_fields_agent=ExtractFieldsAgent(runtime),
|
||||
smart_split_agent=SmartSplitAgent(runtime),
|
||||
suggest_schema_agent=SuggestSchemaAgent(runtime),
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -7,6 +7,7 @@ from fastapi import Depends, HTTPException, Request, status
|
||||
from stirling.agents import (
|
||||
DocumentClassifierAgent,
|
||||
ExecutionPlanningAgent,
|
||||
KnowledgeAskAgent,
|
||||
OrchestratorAgent,
|
||||
PdfEditAgent,
|
||||
PdfQuestionAgent,
|
||||
@@ -15,6 +16,7 @@ from stirling.agents import (
|
||||
from stirling.agents.ledger import MathAuditorAgent
|
||||
from stirling.agents.pdf_comment import PdfCommentAgent
|
||||
from stirling.config import AppSettings, load_settings
|
||||
from stirling.docparse import ExtractFieldsAgent, SmartSplitAgent, SuggestSchemaAgent
|
||||
from stirling.documents import DocumentService
|
||||
from stirling.models import UserId
|
||||
from stirling.services import AppRuntime, current_user_id
|
||||
@@ -60,6 +62,22 @@ def get_document_classifier_agent(request: Request) -> DocumentClassifierAgent:
|
||||
return request.app.state.document_classifier_agent
|
||||
|
||||
|
||||
def get_knowledge_ask_agent(request: Request) -> KnowledgeAskAgent:
|
||||
return request.app.state.knowledge_ask_agent
|
||||
|
||||
|
||||
def get_extract_fields_agent(request: Request) -> ExtractFieldsAgent:
|
||||
return request.app.state.extract_fields_agent
|
||||
|
||||
|
||||
def get_smart_split_agent(request: Request) -> SmartSplitAgent:
|
||||
return request.app.state.smart_split_agent
|
||||
|
||||
|
||||
def get_suggest_schema_agent(request: Request) -> SuggestSchemaAgent:
|
||||
return request.app.state.suggest_schema_agent
|
||||
|
||||
|
||||
def require_user_id() -> UserId:
|
||||
"""FastAPI dependency for routes that touch per-user storage.
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
from .agent_capabilities import router as agent_capabilities_router
|
||||
from .agent_drafts import router as agent_draft_router
|
||||
from .config import router as config_router
|
||||
from .docparse import router as docparse_router
|
||||
from .document_classifier import router as document_classifier_router
|
||||
from .documents import router as document_router
|
||||
from .execution import router as execution_router
|
||||
@@ -14,6 +15,7 @@ __all__ = [
|
||||
"agent_capabilities_router",
|
||||
"agent_draft_router",
|
||||
"config_router",
|
||||
"docparse_router",
|
||||
"document_classifier_router",
|
||||
"document_router",
|
||||
"execution_router",
|
||||
|
||||
@@ -0,0 +1,314 @@
|
||||
"""DocParse routes: parse, extract, split, chunk, tables, fill, capabilities.
|
||||
|
||||
Tier routing happens here: requests carrying raw file bytes can use the
|
||||
advanced (Docling) path when the addon is installed; text-only requests run
|
||||
the basic path. Forcing ``advanced`` without the addon returns 501 with a
|
||||
machine-readable ``addonRequired`` detail that Java maps onto its own error.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import binascii
|
||||
import logging
|
||||
from typing import Annotated
|
||||
|
||||
import anyio.to_thread
|
||||
from fastapi import APIRouter, Depends, HTTPException, status
|
||||
|
||||
from stirling.api.dependencies import (
|
||||
get_document_service,
|
||||
get_extract_fields_agent,
|
||||
get_smart_split_agent,
|
||||
get_suggest_schema_agent,
|
||||
require_user_id,
|
||||
)
|
||||
from stirling.config import AppSettings, load_settings
|
||||
from stirling.contracts.docparse import (
|
||||
ChunkDocumentRequest,
|
||||
ChunkDocumentResponse,
|
||||
DocChunk,
|
||||
DocparseCapabilities,
|
||||
DocparseMode,
|
||||
DocparseTier,
|
||||
ExtractFieldsRequest,
|
||||
ExtractFieldsResponse,
|
||||
ExtractTablesRequest,
|
||||
ExtractTablesResponse,
|
||||
FillDocxRequest,
|
||||
FillDocxResponse,
|
||||
ParseDocumentRequest,
|
||||
ParseDocumentResponse,
|
||||
RagIngestRequest,
|
||||
RagIngestResponse,
|
||||
SmartSplitRequest,
|
||||
SmartSplitResponse,
|
||||
SuggestSchemaRequest,
|
||||
SuggestSchemaResponse,
|
||||
)
|
||||
from stirling.docparse import basic_chunks, fill_docx, probe_capabilities
|
||||
from stirling.docparse.capability import models_dir
|
||||
from stirling.docparse.chunking import advanced_chunks
|
||||
from stirling.docparse.extractor import ExtractFieldsAgent, SchemaError, pages_from_parse
|
||||
from stirling.docparse.splitter import SmartSplitAgent
|
||||
from stirling.docparse.suggest_schema import SuggestSchemaAgent
|
||||
from stirling.documents import DocumentService
|
||||
from stirling.documents.service import CONTENT_TYPE_METADATA_KEY, DOCPARSE_CHUNK_CONTENT_TYPE
|
||||
from stirling.models import OwnerId, PrincipalId, UserId
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter(prefix="/api/v1/docparse", tags=["docparse"])
|
||||
|
||||
_ADDON_REQUIRED_DETAIL = {
|
||||
"addonRequired": "docparse",
|
||||
"message": "The docparse addon (Docling) is not installed on the engine. "
|
||||
"Install the engine's 'docparse' extra or enable DOCPARSE_AUTO_INSTALL.",
|
||||
}
|
||||
|
||||
|
||||
def _settings() -> AppSettings:
|
||||
return load_settings()
|
||||
|
||||
|
||||
def _capabilities(settings: AppSettings, *, refresh: bool = False) -> DocparseCapabilities:
|
||||
return probe_capabilities(settings.docparse_home, refresh=refresh)
|
||||
|
||||
|
||||
def _require_advanced(settings: AppSettings) -> str | None:
|
||||
caps = _capabilities(settings)
|
||||
if not caps.advanced_installed:
|
||||
raise HTTPException(status_code=status.HTTP_501_NOT_IMPLEMENTED, detail=_ADDON_REQUIRED_DETAIL)
|
||||
directory = models_dir(settings.docparse_home)
|
||||
return str(directory) if caps.models_available and directory is not None else None
|
||||
|
||||
|
||||
def _decode_content(content_base64: str) -> bytes:
|
||||
try:
|
||||
return base64.b64decode(content_base64, validate=True)
|
||||
except (binascii.Error, ValueError) as error:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY, detail="contentBase64 is not valid base64"
|
||||
) from error
|
||||
|
||||
|
||||
async def _parse_advanced(content_base64: str, file_name: str, *, with_ocr: bool, artifacts: str | None):
|
||||
"""Run the Docling parse off-thread; unparsable documents are a caller error."""
|
||||
from stirling.docparse.parser import parse_pdf_bytes # deferred: touches docling
|
||||
|
||||
data = _decode_content(content_base64)
|
||||
try:
|
||||
return await anyio.to_thread.run_sync(
|
||||
lambda: parse_pdf_bytes(data, file_name, with_ocr=with_ocr, artifacts_path=artifacts)
|
||||
)
|
||||
except ValueError as error:
|
||||
raise HTTPException(status_code=status.HTTP_422_UNPROCESSABLE_ENTITY, detail=str(error)) from error
|
||||
|
||||
|
||||
@router.get("/capabilities", response_model=DocparseCapabilities)
|
||||
async def capabilities(refresh: bool = False) -> DocparseCapabilities:
|
||||
return _capabilities(_settings(), refresh=refresh)
|
||||
|
||||
|
||||
@router.post("/parse", response_model=ParseDocumentResponse)
|
||||
async def parse_document(request: ParseDocumentRequest) -> ParseDocumentResponse:
|
||||
"""Advanced-tier layout parse. The basic tier lives Java-side (PDFBox) and
|
||||
never reaches the engine, so this endpoint requires the addon outright."""
|
||||
settings = _settings()
|
||||
artifacts = _require_advanced(settings)
|
||||
return await _parse_advanced(
|
||||
request.content_base64, request.file_name, with_ocr=request.with_ocr, artifacts=artifacts
|
||||
)
|
||||
|
||||
|
||||
@router.post("/extract", response_model=ExtractFieldsResponse)
|
||||
async def extract_fields(
|
||||
request: ExtractFieldsRequest,
|
||||
agent: Annotated[ExtractFieldsAgent, Depends(get_extract_fields_agent)],
|
||||
) -> ExtractFieldsResponse:
|
||||
settings = _settings()
|
||||
caps = _capabilities(settings)
|
||||
|
||||
use_advanced = request.mode is DocparseMode.ADVANCED or (
|
||||
request.mode is DocparseMode.AUTO and caps.advanced_installed and request.content_base64 is not None
|
||||
)
|
||||
parse = None
|
||||
if use_advanced:
|
||||
artifacts = _require_advanced(settings)
|
||||
if request.content_base64 is None:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
||||
detail="advanced extraction needs contentBase64 (the raw file)",
|
||||
)
|
||||
parse = await _parse_advanced(request.content_base64, request.file_name, with_ocr=True, artifacts=artifacts)
|
||||
|
||||
pages = request.pages or (pages_from_parse(parse) if parse is not None else None)
|
||||
if not pages:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
||||
detail="send pages (extracted text) or contentBase64 with the addon installed",
|
||||
)
|
||||
try:
|
||||
return await agent.extract(request, pages, parse)
|
||||
except SchemaError as error:
|
||||
raise HTTPException(status_code=status.HTTP_422_UNPROCESSABLE_ENTITY, detail=str(error)) from error
|
||||
|
||||
|
||||
@router.post("/suggest-schema", response_model=SuggestSchemaResponse)
|
||||
async def suggest_schema(
|
||||
request: SuggestSchemaRequest,
|
||||
agent: Annotated[SuggestSchemaAgent, Depends(get_suggest_schema_agent)],
|
||||
) -> SuggestSchemaResponse:
|
||||
"""Propose an extraction schema from the document's first pages.
|
||||
Tier routing: pages -> basic; contentBase64 + addon -> advanced parse."""
|
||||
settings = _settings()
|
||||
caps = _capabilities(settings)
|
||||
pages = request.pages
|
||||
tier = DocparseTier.BASIC
|
||||
if not pages and request.content_base64 is not None and caps.advanced_installed:
|
||||
artifacts = _require_advanced(settings)
|
||||
parse = await _parse_advanced(request.content_base64, request.file_name, with_ocr=True, artifacts=artifacts)
|
||||
pages = pages_from_parse(parse)
|
||||
tier = DocparseTier.ADVANCED
|
||||
if not pages:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
||||
detail="send pages (extracted text) or contentBase64 with the addon installed",
|
||||
)
|
||||
return await agent.suggest(request, pages, tier)
|
||||
|
||||
|
||||
@router.post("/split", response_model=SmartSplitResponse)
|
||||
async def smart_split(
|
||||
request: SmartSplitRequest,
|
||||
agent: Annotated[SmartSplitAgent, Depends(get_smart_split_agent)],
|
||||
) -> SmartSplitResponse:
|
||||
return await agent.split(request)
|
||||
|
||||
|
||||
@router.post("/chunk", response_model=ChunkDocumentResponse)
|
||||
async def chunk_document(request: ChunkDocumentRequest) -> ChunkDocumentResponse:
|
||||
settings = _settings()
|
||||
caps = _capabilities(settings)
|
||||
use_advanced = request.mode is DocparseMode.ADVANCED or (
|
||||
request.mode is DocparseMode.AUTO and caps.advanced_installed and request.content_base64 is not None
|
||||
)
|
||||
if use_advanced:
|
||||
artifacts = _require_advanced(settings)
|
||||
if request.content_base64 is None:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
||||
detail="advanced chunking needs contentBase64 (the raw file)",
|
||||
)
|
||||
parse = await _parse_advanced(request.content_base64, request.file_name, with_ocr=True, artifacts=artifacts)
|
||||
return advanced_chunks(parse, request.chunk_size, request.overlap)
|
||||
|
||||
if not request.pages:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
||||
detail="send pages (extracted text) or contentBase64 with the addon installed",
|
||||
)
|
||||
return basic_chunks(request.pages, request.chunk_size, request.overlap)
|
||||
|
||||
|
||||
def _chunk_metadata(chunk: DocChunk) -> dict[str, str]:
|
||||
meta = {CONTENT_TYPE_METADATA_KEY: DOCPARSE_CHUNK_CONTENT_TYPE}
|
||||
if chunk.page_start is not None:
|
||||
meta["page_start"] = str(chunk.page_start)
|
||||
if chunk.page_end is not None:
|
||||
meta["page_end"] = str(chunk.page_end)
|
||||
if chunk.heading_path:
|
||||
meta["heading_path"] = " > ".join(chunk.heading_path)
|
||||
return meta
|
||||
|
||||
|
||||
@router.post("/rag-ingest", response_model=RagIngestResponse)
|
||||
async def rag_ingest(
|
||||
request: RagIngestRequest,
|
||||
documents: Annotated[DocumentService, Depends(get_document_service)],
|
||||
user_id: Annotated[UserId, Depends(require_user_id)],
|
||||
) -> RagIngestResponse:
|
||||
"""Chunk the document, then embed and index into the document store.
|
||||
Re-ingesting a documentId replaces its stored content (never duplicates).
|
||||
``index=False`` skips the store; ``includeMarkdown``/``includeChunks``
|
||||
echo the content back for corpus export."""
|
||||
settings = _settings()
|
||||
caps = _capabilities(settings)
|
||||
chunk_size = request.chunk_size if request.chunk_size is not None else settings.rag_chunk_size
|
||||
overlap = request.overlap if request.overlap is not None else settings.rag_chunk_overlap
|
||||
|
||||
if not request.index and not request.include_markdown and not request.include_chunks:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
||||
detail="nothing to do: enable index, includeMarkdown, or includeChunks",
|
||||
)
|
||||
|
||||
use_advanced = request.mode is DocparseMode.ADVANCED or (
|
||||
request.mode is DocparseMode.AUTO and caps.advanced_installed and request.content_base64 is not None
|
||||
)
|
||||
if use_advanced:
|
||||
artifacts = _require_advanced(settings)
|
||||
if request.content_base64 is None:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
||||
detail="advanced chunking needs contentBase64 (the raw file)",
|
||||
)
|
||||
parse = await _parse_advanced(request.content_base64, request.file_name, with_ocr=True, artifacts=artifacts)
|
||||
chunked = advanced_chunks(parse, chunk_size, overlap)
|
||||
page_count = parse.pages
|
||||
markdown = parse.markdown if request.include_markdown else None
|
||||
else:
|
||||
if not request.pages:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
||||
detail="send pages (extracted text) or contentBase64 with the addon installed",
|
||||
)
|
||||
chunked = basic_chunks(request.pages, chunk_size, overlap)
|
||||
page_count = max(p.page_number for p in request.pages)
|
||||
markdown = None
|
||||
if request.include_markdown:
|
||||
markdown = "\n\n".join(p.text for p in request.pages if p.text.strip())
|
||||
|
||||
chunks_indexed = 0
|
||||
if request.index:
|
||||
# Owner/ACL semantics mirror IngestDocumentRequest; omitted values default
|
||||
# to the authenticated caller (personal-doc behaviour).
|
||||
owner_id = request.owner_id if request.owner_id is not None else OwnerId(user_id)
|
||||
read_principals = request.read_principals or [PrincipalId(owner_id)]
|
||||
chunks_indexed = await documents.ingest_prepared(
|
||||
collection=request.document_id,
|
||||
chunks=[(chunk.text, _chunk_metadata(chunk)) for chunk in chunked.chunks],
|
||||
source=request.source,
|
||||
owner_id=owner_id,
|
||||
read_principals=read_principals,
|
||||
expires_at=request.expires_at,
|
||||
)
|
||||
logger.info(
|
||||
"docparse: rag-ingested %s: %d chunks indexed, %d pages", request.document_id, chunks_indexed, page_count
|
||||
)
|
||||
return RagIngestResponse(
|
||||
mode=chunked.mode,
|
||||
document_id=request.document_id,
|
||||
chunks_indexed=chunks_indexed,
|
||||
pages=page_count,
|
||||
markdown=markdown,
|
||||
chunks=chunked.chunks if request.include_chunks else None,
|
||||
)
|
||||
|
||||
|
||||
@router.post("/tables", response_model=ExtractTablesResponse)
|
||||
async def extract_tables(request: ExtractTablesRequest) -> ExtractTablesResponse:
|
||||
settings = _settings()
|
||||
artifacts = _require_advanced(settings)
|
||||
parse = await _parse_advanced(request.content_base64, request.file_name, with_ocr=True, artifacts=artifacts)
|
||||
return ExtractTablesResponse(mode=parse.mode, tables=parse.tables)
|
||||
|
||||
|
||||
@router.post("/fill-docx", response_model=FillDocxResponse)
|
||||
async def fill_docx_template(request: FillDocxRequest) -> FillDocxResponse:
|
||||
try:
|
||||
return await anyio.to_thread.run_sync(lambda: fill_docx(request))
|
||||
except (KeyError, ValueError) as error:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY, detail=f"invalid docx template: {error}"
|
||||
) from error
|
||||
@@ -5,13 +5,24 @@ from typing import Annotated
|
||||
|
||||
from fastapi import APIRouter, Depends
|
||||
|
||||
from stirling.api.dependencies import get_document_service, require_user_id
|
||||
from stirling.agents.knowledge_ask import KnowledgeAskAgent, passage_from_hit
|
||||
from stirling.api.dependencies import get_document_service, get_knowledge_ask_agent, require_user_id
|
||||
from stirling.config import load_settings
|
||||
from stirling.contracts import (
|
||||
DeleteDocumentResponse,
|
||||
IngestDocumentRequest,
|
||||
IngestDocumentResponse,
|
||||
)
|
||||
from stirling.contracts.documents import PurgeOwnerResponse
|
||||
from stirling.contracts.documents import (
|
||||
AskDocumentsRequest,
|
||||
AskDocumentsResponse,
|
||||
DocumentStatsResponse,
|
||||
DocumentSummary,
|
||||
ListDocumentsResponse,
|
||||
PurgeOwnerResponse,
|
||||
SearchDocumentsRequest,
|
||||
SearchDocumentsResponse,
|
||||
)
|
||||
from stirling.documents import DocumentService
|
||||
from stirling.models import FileId, OwnerId, PrincipalId, UserId
|
||||
|
||||
@@ -45,6 +56,70 @@ async def ingest_document(
|
||||
return IngestDocumentResponse(document_id=request.document_id, chunks_indexed=chunks_indexed)
|
||||
|
||||
|
||||
@router.get("/stats", response_model=DocumentStatsResponse)
|
||||
async def document_stats(
|
||||
documents: Annotated[DocumentService, Depends(get_document_service)],
|
||||
_user_id: Annotated[UserId, Depends(require_user_id)],
|
||||
) -> DocumentStatsResponse:
|
||||
"""Deployment-wide store counts for the admin dashboard.
|
||||
|
||||
Not tenant-filtered: counts cover every owner's content, so this reports
|
||||
what the whole store holds, not what the caller can read.
|
||||
"""
|
||||
settings = load_settings()
|
||||
counts = await documents.stats()
|
||||
return DocumentStatsResponse(
|
||||
backend=settings.documents_backend.value,
|
||||
documents=counts.documents,
|
||||
chunks=counts.chunks,
|
||||
embedding_model=settings.rag_embedding_model,
|
||||
)
|
||||
|
||||
|
||||
@router.get("/list", response_model=ListDocumentsResponse)
|
||||
async def list_documents(
|
||||
documents: Annotated[DocumentService, Depends(get_document_service)],
|
||||
user_id: Annotated[UserId, Depends(require_user_id)],
|
||||
) -> ListDocumentsResponse:
|
||||
"""Per-document rollup of what the caller can read: distinct document ids
|
||||
with their stored source label and chunk count. Never shows another
|
||||
principal's documents.
|
||||
"""
|
||||
summaries = await documents.list_documents([PrincipalId(user_id)])
|
||||
return ListDocumentsResponse(
|
||||
documents=[
|
||||
DocumentSummary(document_id=FileId(s.collection), source=s.source, chunks=s.chunks) for s in summaries
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
@router.post("/search", response_model=SearchDocumentsResponse)
|
||||
async def search_documents(
|
||||
request: SearchDocumentsRequest,
|
||||
documents: Annotated[DocumentService, Depends(get_document_service)],
|
||||
user_id: Annotated[UserId, Depends(require_user_id)],
|
||||
) -> SearchDocumentsResponse:
|
||||
"""Semantic search across every document the caller can read.
|
||||
|
||||
Same retrieval path as the RAG toolset: embed the query, search the
|
||||
caller's readable collections, merge by score.
|
||||
"""
|
||||
hits = await documents.search_with_collections(
|
||||
request.query, principals=[PrincipalId(user_id)], top_k=request.top_k
|
||||
)
|
||||
return SearchDocumentsResponse(passages=[passage_from_hit(hit) for hit in hits])
|
||||
|
||||
|
||||
@router.post("/ask", response_model=AskDocumentsResponse)
|
||||
async def ask_documents(
|
||||
request: AskDocumentsRequest,
|
||||
agent: Annotated[KnowledgeAskAgent, Depends(get_knowledge_ask_agent)],
|
||||
user_id: Annotated[UserId, Depends(require_user_id)],
|
||||
) -> AskDocumentsResponse:
|
||||
"""Answer a question from the caller's stored documents with inline citations."""
|
||||
return await agent.ask(request, principals=[PrincipalId(user_id)])
|
||||
|
||||
|
||||
@router.delete("/by-id/{document_id}", response_model=DeleteDocumentResponse)
|
||||
async def delete_document(
|
||||
document_id: FileId,
|
||||
|
||||
@@ -101,6 +101,11 @@ class AppSettings(BaseSettings):
|
||||
max_pages: int = Field(validation_alias="STIRLING_MAX_PAGES")
|
||||
max_characters: int = Field(validation_alias="STIRLING_MAX_CHARACTERS")
|
||||
|
||||
# DocParse addon home: <home>/site holds a dynamically installed docling
|
||||
# stack (prepended to sys.path at startup), <home>/models holds prefetched
|
||||
# model weights. Empty = addon only usable when baked into the image.
|
||||
docparse_home: str = Field(default="", validation_alias="STIRLING_DOCPARSE_HOME")
|
||||
|
||||
# When true, API routes reject requests that lack an X-User-Id header at
|
||||
# the boundary. Self-hosted deployments with security disabled have no
|
||||
# user identity and leave this off; multi-tenant deployments turn it on so
|
||||
|
||||
@@ -42,6 +42,36 @@ from .contradiction import (
|
||||
ContradictionReport,
|
||||
ContradictionSeverity,
|
||||
)
|
||||
from .docparse import (
|
||||
BlockType,
|
||||
ChunkDocumentRequest,
|
||||
ChunkDocumentResponse,
|
||||
DocBlock,
|
||||
DocChunk,
|
||||
DocparseCapabilities,
|
||||
DocparseMode,
|
||||
DocparseTier,
|
||||
DocTable,
|
||||
ExtractedField,
|
||||
ExtractFieldsRequest,
|
||||
ExtractFieldsResponse,
|
||||
ExtractTablesRequest,
|
||||
ExtractTablesResponse,
|
||||
FieldCitation,
|
||||
FillDocxRequest,
|
||||
FillDocxResponse,
|
||||
ParseDocumentRequest,
|
||||
ParseDocumentResponse,
|
||||
RagIngestRequest,
|
||||
RagIngestResponse,
|
||||
SmartSplitRequest,
|
||||
SmartSplitResponse,
|
||||
SplitPart,
|
||||
SuggestedField,
|
||||
SuggestedFieldType,
|
||||
SuggestSchemaRequest,
|
||||
SuggestSchemaResponse,
|
||||
)
|
||||
from .document_classifier import (
|
||||
ClassifyDocumentRequest,
|
||||
ClassifyDocumentResponse,
|
||||
@@ -49,13 +79,21 @@ from .document_classifier import (
|
||||
LabelOption,
|
||||
)
|
||||
from .documents import (
|
||||
AskDocumentsRequest,
|
||||
AskDocumentsResponse,
|
||||
DeleteDocumentResponse,
|
||||
DocumentPassage,
|
||||
DocumentStatsResponse,
|
||||
DocumentSummary,
|
||||
IngestDocumentRequest,
|
||||
IngestDocumentResponse,
|
||||
ListDocumentsResponse,
|
||||
Page,
|
||||
PageRange,
|
||||
PageText,
|
||||
PurgeOwnerResponse,
|
||||
SearchDocumentsRequest,
|
||||
SearchDocumentsResponse,
|
||||
)
|
||||
from .execution import (
|
||||
AgentExecutionRequest,
|
||||
@@ -140,10 +178,15 @@ __all__ = [
|
||||
"AiFile",
|
||||
"AiToolAgentStep",
|
||||
"ArtifactKind",
|
||||
"AskDocumentsRequest",
|
||||
"AskDocumentsResponse",
|
||||
"BlockType",
|
||||
"CannotContinueExecutionAction",
|
||||
"ChunkDocumentRequest",
|
||||
"ChunkDocumentResponse",
|
||||
"Claim",
|
||||
"ClassifyDocumentRequest",
|
||||
"ClassifyDocumentResponse",
|
||||
"Claim",
|
||||
"CommentSpec",
|
||||
"CompletedExecutionAction",
|
||||
"ConfigApplyResponse",
|
||||
@@ -156,30 +199,45 @@ __all__ = [
|
||||
"ContradictionSeverity",
|
||||
"ConversationMessage",
|
||||
"DeleteDocumentResponse",
|
||||
"PurgeOwnerResponse",
|
||||
"Discrepancy",
|
||||
"DocumentClassificationResponse",
|
||||
"LabelOption",
|
||||
"DocumentMeta",
|
||||
"DocumentSections",
|
||||
"DiscrepancyKind",
|
||||
"DocBlock",
|
||||
"DocChunk",
|
||||
"DocTable",
|
||||
"DocparseCapabilities",
|
||||
"DocparseMode",
|
||||
"DocparseTier",
|
||||
"DocumentClassificationResponse",
|
||||
"DocumentMeta",
|
||||
"DocumentPassage",
|
||||
"DocumentSections",
|
||||
"DocumentStatsResponse",
|
||||
"DocumentSummary",
|
||||
"EditCannotDoResponse",
|
||||
"EditClarificationRequest",
|
||||
"EditPlanResponse",
|
||||
"Evidence",
|
||||
"ExecutionContext",
|
||||
"ExecutionStepResult",
|
||||
"ExtractFieldsRequest",
|
||||
"ExtractFieldsResponse",
|
||||
"ExtractTablesRequest",
|
||||
"ExtractTablesResponse",
|
||||
"ExtractedField",
|
||||
"ExtractedFileText",
|
||||
"ExtractedTextArtifact",
|
||||
"FieldCitation",
|
||||
"FillDocxRequest",
|
||||
"FillDocxResponse",
|
||||
"Folio",
|
||||
"FolioManifest",
|
||||
"FolioType",
|
||||
"format_conversation_history",
|
||||
"format_file_names",
|
||||
"GenerateFileResponse",
|
||||
"HealthResponse",
|
||||
"IngestDocumentRequest",
|
||||
"IngestDocumentResponse",
|
||||
"LabelOption",
|
||||
"ListDocumentsResponse",
|
||||
"MathAuditorToolReportArtifact",
|
||||
"NeedContentFileRequest",
|
||||
"NeedContentResponse",
|
||||
@@ -190,6 +248,8 @@ __all__ = [
|
||||
"Page",
|
||||
"PageRange",
|
||||
"PageText",
|
||||
"ParseDocumentRequest",
|
||||
"ParseDocumentResponse",
|
||||
"PdfCommentInstruction",
|
||||
"PdfCommentReport",
|
||||
"PdfCommentRequest",
|
||||
@@ -212,9 +272,21 @@ __all__ = [
|
||||
"PdfReviewOrchestrateResponse",
|
||||
"PdfTextSelection",
|
||||
"ProgressEvent",
|
||||
"PurgeOwnerResponse",
|
||||
"RagIngestRequest",
|
||||
"RagIngestResponse",
|
||||
"Requisition",
|
||||
"SearchDocumentsRequest",
|
||||
"SearchDocumentsResponse",
|
||||
"Severity",
|
||||
"SmartSplitRequest",
|
||||
"SmartSplitResponse",
|
||||
"SplitPart",
|
||||
"StepKind",
|
||||
"SuggestSchemaRequest",
|
||||
"SuggestSchemaResponse",
|
||||
"SuggestedField",
|
||||
"SuggestedFieldType",
|
||||
"SupportedCapability",
|
||||
"TextChunk",
|
||||
"ToolCallExecutionAction",
|
||||
@@ -228,4 +300,6 @@ __all__ = [
|
||||
"WholeDocSliceDone",
|
||||
"WorkflowArtifact",
|
||||
"WorkflowOutcome",
|
||||
"format_conversation_history",
|
||||
"format_file_names",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,257 @@
|
||||
"""Wire contracts for the DocParse ingestion capability (chunking + rag-ingest).
|
||||
|
||||
Java counterpart DTOs live under ``stirling.software.proprietary.model.docparse``
|
||||
and must stay in sync. Tier model: ``basic`` (text layer) runs everywhere;
|
||||
``advanced`` (layout parsing) arrives with the docparse addon.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
from enum import StrEnum
|
||||
|
||||
from pydantic import Field, JsonValue
|
||||
|
||||
from stirling.contracts.documents import PageText
|
||||
from stirling.models import ApiModel, FileId, OwnerId, PrincipalId
|
||||
|
||||
|
||||
class DocparseTier(StrEnum):
|
||||
"""Which implementation actually served a request."""
|
||||
|
||||
BASIC = "basic"
|
||||
ADVANCED = "advanced"
|
||||
|
||||
|
||||
class DocparseMode(StrEnum):
|
||||
"""What the caller asked for; AUTO resolves per-request."""
|
||||
|
||||
AUTO = "auto"
|
||||
BASIC = "basic"
|
||||
ADVANCED = "advanced"
|
||||
|
||||
|
||||
class BlockType(StrEnum):
|
||||
"""Normalized layout block labels; Docling labels map onto these."""
|
||||
|
||||
HEADING = "heading"
|
||||
PARAGRAPH = "paragraph"
|
||||
LIST_ITEM = "list_item"
|
||||
TABLE = "table"
|
||||
FIGURE = "figure"
|
||||
CAPTION = "caption"
|
||||
CODE = "code"
|
||||
FORMULA = "formula"
|
||||
PAGE_HEADER = "page_header"
|
||||
PAGE_FOOTER = "page_footer"
|
||||
FOOTNOTE = "footnote"
|
||||
OTHER = "other"
|
||||
|
||||
|
||||
class DocBlock(ApiModel):
|
||||
"""One layout block. ``bbox`` is [x0, y0, x1, y1] normalized to 0..1 with a
|
||||
top-left origin; ``None`` in basic tier (no layout model ran)."""
|
||||
|
||||
type: BlockType
|
||||
text: str
|
||||
page: int = Field(ge=1)
|
||||
bbox: list[float] | None = None
|
||||
confidence: float | None = Field(default=None, ge=0.0, le=1.0)
|
||||
|
||||
|
||||
class DocTable(ApiModel):
|
||||
page: int = Field(ge=1)
|
||||
bbox: list[float] | None = None
|
||||
cells: list[list[str]]
|
||||
markdown: str
|
||||
confidence: float | None = Field(default=None, ge=0.0, le=1.0)
|
||||
|
||||
|
||||
class ParseDocumentRequest(ApiModel):
|
||||
file_name: str = Field(min_length=1)
|
||||
content_base64: str = Field(min_length=1)
|
||||
with_ocr: bool = True
|
||||
|
||||
|
||||
class ParseDocumentResponse(ApiModel):
|
||||
mode: DocparseTier
|
||||
pages: int = Field(ge=0)
|
||||
blocks: list[DocBlock] = Field(default_factory=list)
|
||||
tables: list[DocTable] = Field(default_factory=list)
|
||||
markdown: str = ""
|
||||
ocr_applied: bool = False
|
||||
|
||||
|
||||
class ExtractTablesRequest(ApiModel):
|
||||
file_name: str = Field(min_length=1)
|
||||
content_base64: str = Field(min_length=1)
|
||||
|
||||
|
||||
class ExtractTablesResponse(ApiModel):
|
||||
mode: DocparseTier
|
||||
tables: list[DocTable] = Field(default_factory=list)
|
||||
|
||||
|
||||
class DocChunk(ApiModel):
|
||||
index: int = Field(ge=0)
|
||||
text: str
|
||||
page_start: int | None = Field(default=None, ge=1)
|
||||
page_end: int | None = Field(default=None, ge=1)
|
||||
heading_path: list[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
class ChunkDocumentResponse(ApiModel):
|
||||
mode: DocparseTier
|
||||
chunks: list[DocChunk] = Field(default_factory=list)
|
||||
|
||||
|
||||
class RagIngestRequest(ApiModel):
|
||||
"""Chunk the document, then optionally embed and replace-index it.
|
||||
|
||||
``owner_id``/``read_principals`` default to the calling user (personal-doc
|
||||
semantics); ``chunk_size``/``overlap`` default to the engine's RAG settings.
|
||||
``index=False`` skips the store entirely (export-only ingestion);
|
||||
``include_markdown``/``include_chunks`` echo the parsed content back so the
|
||||
caller can emit corpus files (markdown, chunks JSONL).
|
||||
"""
|
||||
|
||||
file_name: str = Field(min_length=1)
|
||||
document_id: FileId = Field(min_length=1)
|
||||
source: str = Field(default="docparse", min_length=1)
|
||||
owner_id: OwnerId | None = None
|
||||
read_principals: list[PrincipalId] | None = Field(default=None, min_length=1)
|
||||
expires_at: datetime | None = None
|
||||
pages: list[PageText] | None = None
|
||||
content_base64: str | None = None
|
||||
chunk_size: int | None = Field(default=None, ge=64, le=32_768)
|
||||
overlap: int | None = Field(default=None, ge=0, le=4_096)
|
||||
mode: DocparseMode = DocparseMode.AUTO
|
||||
index: bool = True
|
||||
include_markdown: bool = False
|
||||
include_chunks: bool = False
|
||||
|
||||
|
||||
class RagIngestResponse(ApiModel):
|
||||
mode: DocparseTier
|
||||
document_id: FileId
|
||||
chunks_indexed: int = Field(ge=0)
|
||||
pages: int = Field(ge=0)
|
||||
markdown: str | None = None
|
||||
chunks: list[DocChunk] | None = None
|
||||
|
||||
|
||||
class FieldCitation(ApiModel):
|
||||
"""Where a value came from. ``quote`` is always set; ``bbox`` only when a
|
||||
layout parse ran (advanced tier); offsets index into the cited page's text."""
|
||||
|
||||
page: int | None = Field(default=None, ge=1)
|
||||
bbox: list[float] | None = None
|
||||
quote: str
|
||||
start_offset: int | None = Field(default=None, ge=0)
|
||||
end_offset: int | None = Field(default=None, ge=0)
|
||||
|
||||
|
||||
class ExtractedField(ApiModel):
|
||||
name: str
|
||||
value: JsonValue = None
|
||||
confidence: float = Field(ge=0.0, le=1.0)
|
||||
citations: list[FieldCitation] = Field(default_factory=list)
|
||||
|
||||
|
||||
class ExtractFieldsRequest(ApiModel):
|
||||
"""``pages`` drives the basic tier (caller-extracted text); ``content_base64``
|
||||
lets the advanced tier parse the raw file itself. Send either or both."""
|
||||
|
||||
file_name: str = Field(min_length=1)
|
||||
fields_schema: dict[str, JsonValue]
|
||||
pages: list[PageText] | None = None
|
||||
content_base64: str | None = None
|
||||
mode: DocparseMode = DocparseMode.AUTO
|
||||
instructions: str | None = None
|
||||
|
||||
|
||||
class ExtractFieldsResponse(ApiModel):
|
||||
mode: DocparseTier
|
||||
fields: list[ExtractedField] = Field(default_factory=list)
|
||||
overall_confidence: float = Field(ge=0.0, le=1.0)
|
||||
|
||||
|
||||
class SuggestedFieldType(StrEnum):
|
||||
"""Scalar types the schema suggester may propose; the extractor's leaf subset."""
|
||||
|
||||
STRING = "string"
|
||||
NUMBER = "number"
|
||||
INTEGER = "integer"
|
||||
BOOLEAN = "boolean"
|
||||
|
||||
|
||||
class SuggestedField(ApiModel):
|
||||
name: str = Field(description="snake_case field identifier, e.g. 'invoice_number'.")
|
||||
type: SuggestedFieldType
|
||||
description: str = ""
|
||||
|
||||
|
||||
class SuggestSchemaRequest(ApiModel):
|
||||
"""``pages`` drives the basic tier (caller-extracted text); ``content_base64``
|
||||
lets the advanced tier parse the raw file itself. Send either."""
|
||||
|
||||
file_name: str = Field(min_length=1)
|
||||
pages: list[PageText] | None = None
|
||||
content_base64: str | None = None
|
||||
max_fields: int = Field(default=8, ge=1, le=20)
|
||||
|
||||
|
||||
class SuggestSchemaResponse(ApiModel):
|
||||
mode: DocparseTier
|
||||
fields: list[SuggestedField] = Field(default_factory=list)
|
||||
|
||||
|
||||
class SplitPart(ApiModel):
|
||||
start_page: int = Field(ge=1)
|
||||
end_page: int = Field(ge=1)
|
||||
label: str
|
||||
confidence: float = Field(ge=0.0, le=1.0)
|
||||
|
||||
|
||||
class SmartSplitRequest(ApiModel):
|
||||
file_name: str = Field(min_length=1)
|
||||
rule: str = Field(
|
||||
min_length=1, description="Natural-language boundary rule, e.g. 'split where a new invoice starts'."
|
||||
)
|
||||
pages: list[PageText]
|
||||
max_parts: int = Field(default=50, ge=1, le=500)
|
||||
|
||||
|
||||
class SmartSplitResponse(ApiModel):
|
||||
parts: list[SplitPart] = Field(default_factory=list)
|
||||
|
||||
|
||||
class ChunkDocumentRequest(ApiModel):
|
||||
file_name: str = Field(min_length=1)
|
||||
pages: list[PageText] | None = None
|
||||
content_base64: str | None = None
|
||||
chunk_size: int = Field(default=512, ge=64, le=32_768)
|
||||
overlap: int = Field(default=64, ge=0, le=4_096)
|
||||
mode: DocparseMode = DocparseMode.AUTO
|
||||
|
||||
|
||||
class FillDocxRequest(ApiModel):
|
||||
template_base64: str = Field(min_length=1)
|
||||
data: dict[str, JsonValue]
|
||||
|
||||
|
||||
class FillDocxResponse(ApiModel):
|
||||
docx_base64: str
|
||||
replaced: int = Field(ge=0)
|
||||
missing: list[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
class DocparseCapabilities(ApiModel):
|
||||
"""What the engine can actually do right now; Java caches and republishes this."""
|
||||
|
||||
advanced_installed: bool
|
||||
docling_version: str | None = None
|
||||
torch_version: str | None = None
|
||||
models_available: bool = False
|
||||
models_path: str | None = None
|
||||
errors: list[str] = Field(default_factory=list)
|
||||
@@ -75,3 +75,65 @@ class PurgeOwnerResponse(ApiModel):
|
||||
|
||||
owner_id: OwnerId
|
||||
deleted: int = Field(ge=0)
|
||||
|
||||
|
||||
class DocumentStatsResponse(ApiModel):
|
||||
"""Returned by ``GET /api/v1/documents/stats``. Deployment-wide counts
|
||||
(every owner's content) powering the admin dashboard."""
|
||||
|
||||
backend: str
|
||||
documents: int = Field(ge=0)
|
||||
chunks: int = Field(ge=0)
|
||||
embedding_model: str
|
||||
|
||||
|
||||
class DocumentSummary(ApiModel):
|
||||
"""One stored document the caller can read: its id, source label, chunk count."""
|
||||
|
||||
document_id: FileId
|
||||
source: str
|
||||
chunks: int = Field(ge=0)
|
||||
|
||||
|
||||
class ListDocumentsResponse(ApiModel):
|
||||
"""Returned by ``GET /api/v1/documents/list``. Caller-scoped rollup."""
|
||||
|
||||
documents: list[DocumentSummary]
|
||||
|
||||
|
||||
class SearchDocumentsRequest(ApiModel):
|
||||
"""Semantic search over every document the caller can read."""
|
||||
|
||||
query: str = Field(min_length=1)
|
||||
top_k: int = Field(default=8, ge=1, le=50)
|
||||
|
||||
|
||||
class DocumentPassage(ApiModel):
|
||||
"""A retrieved chunk on the wire. Page bounds and heading path come from
|
||||
chunk metadata when present (docparse chunks carry them); nulls otherwise."""
|
||||
|
||||
document_id: FileId
|
||||
text: str
|
||||
score: float
|
||||
page_start: int | None = None
|
||||
page_end: int | None = None
|
||||
heading_path: list[str] = Field(default_factory=list)
|
||||
source: str | None = None
|
||||
|
||||
|
||||
class SearchDocumentsResponse(ApiModel):
|
||||
passages: list[DocumentPassage]
|
||||
|
||||
|
||||
class AskDocumentsRequest(ApiModel):
|
||||
"""Question answered only from the caller's stored documents."""
|
||||
|
||||
question: str = Field(min_length=1)
|
||||
top_k: int = Field(default=8, ge=1, le=20)
|
||||
|
||||
|
||||
class AskDocumentsResponse(ApiModel):
|
||||
"""Grounded answer with inline citations plus the passages it drew from."""
|
||||
|
||||
answer: str
|
||||
passages: list[DocumentPassage]
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
"""DocParse: document understanding for ingestion pipelines.
|
||||
|
||||
This package holds the basic (text-layer) tier; the advanced tier (Docling
|
||||
layout parsing) is delivered as an optional addon and probed at runtime.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from stirling.docparse.capability import activate_site, probe_capabilities
|
||||
from stirling.docparse.chunking import advanced_chunks, basic_chunks
|
||||
from stirling.docparse.docxfill import fill_docx
|
||||
from stirling.docparse.extractor import ExtractFieldsAgent
|
||||
from stirling.docparse.splitter import SmartSplitAgent
|
||||
from stirling.docparse.suggest_schema import SuggestSchemaAgent
|
||||
|
||||
__all__ = [
|
||||
"ExtractFieldsAgent",
|
||||
"SmartSplitAgent",
|
||||
"SuggestSchemaAgent",
|
||||
"activate_site",
|
||||
"advanced_chunks",
|
||||
"basic_chunks",
|
||||
"fill_docx",
|
||||
"probe_capabilities",
|
||||
]
|
||||
@@ -0,0 +1,104 @@
|
||||
"""Probe whether the docparse addon (Docling) is importable and its models present.
|
||||
|
||||
The addon arrives either baked into the engine image (uv extra) or dynamically
|
||||
installed into ``$STIRLING_DOCPARSE_HOME/site`` on a mounted volume; in the
|
||||
latter case :func:`activate_site` prepends that directory to ``sys.path`` at
|
||||
startup so the probe and the parser see it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import importlib.metadata
|
||||
import importlib.util
|
||||
import logging
|
||||
import sys
|
||||
import threading
|
||||
from pathlib import Path
|
||||
|
||||
from stirling.contracts.docparse import DocparseCapabilities
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_lock = threading.Lock()
|
||||
_cached: DocparseCapabilities | None = None
|
||||
|
||||
|
||||
def site_dir(docparse_home: str) -> Path | None:
|
||||
return Path(docparse_home) / "site" if docparse_home else None
|
||||
|
||||
|
||||
def models_dir(docparse_home: str) -> Path | None:
|
||||
return Path(docparse_home) / "models" if docparse_home else None
|
||||
|
||||
|
||||
def activate_site(docparse_home: str) -> bool:
|
||||
"""Prepend the dynamic-install site dir to ``sys.path`` if it exists.
|
||||
|
||||
Idempotent; returns True when the path is active. Called once from the app
|
||||
lifespan before the first capability probe.
|
||||
"""
|
||||
site = site_dir(docparse_home)
|
||||
if site is None or not site.is_dir():
|
||||
return False
|
||||
site_str = str(site)
|
||||
if site_str not in sys.path:
|
||||
sys.path.insert(0, site_str)
|
||||
logger.info("docparse: activated dynamic site dir %s", site_str)
|
||||
return True
|
||||
|
||||
|
||||
def _version_of(distribution: str) -> str | None:
|
||||
try:
|
||||
return importlib.metadata.version(distribution)
|
||||
except importlib.metadata.PackageNotFoundError:
|
||||
return None
|
||||
|
||||
|
||||
def _models_available(docparse_home: str) -> tuple[bool, str | None]:
|
||||
"""Models count as available when the prefetch dir has content, or when no
|
||||
home is configured at all (Docling then downloads into its default cache)."""
|
||||
directory = models_dir(docparse_home)
|
||||
if directory is None:
|
||||
return True, None
|
||||
if directory.is_dir() and any(directory.iterdir()):
|
||||
return True, str(directory)
|
||||
return False, str(directory)
|
||||
|
||||
|
||||
def probe_capabilities(docparse_home: str, *, refresh: bool = False) -> DocparseCapabilities:
|
||||
"""Report what the docparse layer can do right now. Cached after first call
|
||||
(imports are expensive); ``refresh=True`` re-probes, e.g. after a dynamic install."""
|
||||
global _cached
|
||||
with _lock:
|
||||
if _cached is not None and not refresh:
|
||||
return _cached
|
||||
|
||||
errors: list[str] = []
|
||||
advanced = importlib.util.find_spec("docling") is not None
|
||||
docling_version: str | None = None
|
||||
torch_version: str | None = None
|
||||
if advanced:
|
||||
docling_version = _version_of("docling")
|
||||
torch_version = _version_of("torch")
|
||||
if torch_version is None:
|
||||
advanced = False
|
||||
errors.append("docling present but torch missing; install is incomplete")
|
||||
models_ok, models_path = _models_available(docparse_home)
|
||||
|
||||
_cached = DocparseCapabilities(
|
||||
advanced_installed=advanced,
|
||||
docling_version=docling_version,
|
||||
torch_version=torch_version,
|
||||
models_available=models_ok,
|
||||
models_path=models_path,
|
||||
errors=errors,
|
||||
)
|
||||
logger.info(
|
||||
"docparse capabilities: advanced=%s docling=%s torch=%s models=%s",
|
||||
advanced,
|
||||
docling_version,
|
||||
torch_version,
|
||||
models_ok,
|
||||
)
|
||||
return _cached
|
||||
@@ -0,0 +1,96 @@
|
||||
"""RAG chunking, both tiers.
|
||||
|
||||
Basic: the existing character chunker per page. Advanced: structure-aware
|
||||
packing over parse blocks - heading-bounded, heading breadcrumbs attached,
|
||||
page ranges tracked. No tokenizer dependency; sizes are characters, matching
|
||||
the rest of the engine."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from stirling.contracts.docparse import BlockType, ChunkDocumentResponse, DocChunk, DocparseTier, ParseDocumentResponse
|
||||
from stirling.contracts.documents import PageText
|
||||
from stirling.documents.chunker import chunk_text
|
||||
|
||||
# Blocks that are noise for retrieval purposes.
|
||||
_SKIP_TYPES = {BlockType.PAGE_HEADER, BlockType.PAGE_FOOTER}
|
||||
|
||||
|
||||
def basic_chunks(pages: list[PageText], chunk_size: int, overlap: int) -> ChunkDocumentResponse:
|
||||
chunks: list[DocChunk] = []
|
||||
for page in pages:
|
||||
for piece in chunk_text(page.text, chunk_size=chunk_size, overlap=overlap):
|
||||
chunks.append(
|
||||
DocChunk(
|
||||
index=len(chunks),
|
||||
text=piece,
|
||||
page_start=page.page_number,
|
||||
page_end=page.page_number,
|
||||
)
|
||||
)
|
||||
return ChunkDocumentResponse(mode=DocparseTier.BASIC, chunks=chunks)
|
||||
|
||||
|
||||
def advanced_chunks(parse: ParseDocumentResponse, chunk_size: int, overlap: int) -> ChunkDocumentResponse:
|
||||
"""Pack layout blocks into chunks that never straddle a heading boundary."""
|
||||
chunks: list[DocChunk] = []
|
||||
heading_path: list[str] = []
|
||||
buffer: list[str] = []
|
||||
buffer_len = 0
|
||||
page_start: int | None = None
|
||||
page_end: int | None = None
|
||||
buffer_headings: list[str] = []
|
||||
|
||||
def flush() -> None:
|
||||
nonlocal buffer, buffer_len, page_start, page_end, buffer_headings
|
||||
text = "\n\n".join(buffer).strip()
|
||||
if text:
|
||||
chunks.append(
|
||||
DocChunk(
|
||||
index=len(chunks),
|
||||
text=text,
|
||||
page_start=page_start,
|
||||
page_end=page_end,
|
||||
heading_path=list(buffer_headings),
|
||||
)
|
||||
)
|
||||
buffer = []
|
||||
buffer_len = 0
|
||||
page_start = None
|
||||
page_end = None
|
||||
buffer_headings = list(heading_path)
|
||||
|
||||
buffer_headings = []
|
||||
for block in parse.blocks:
|
||||
if block.type in _SKIP_TYPES:
|
||||
continue
|
||||
if block.type is BlockType.HEADING:
|
||||
flush()
|
||||
heading_path = [*heading_path[-2:], block.text.strip()] if block.text.strip() else heading_path
|
||||
buffer_headings = list(heading_path)
|
||||
continue
|
||||
|
||||
text = block.text
|
||||
if not text.strip():
|
||||
continue
|
||||
if buffer_len + len(text) > chunk_size and buffer:
|
||||
flush()
|
||||
# A single oversized block falls back to the character chunker.
|
||||
if len(text) > chunk_size:
|
||||
for piece in chunk_text(text, chunk_size=chunk_size, overlap=overlap):
|
||||
chunks.append(
|
||||
DocChunk(
|
||||
index=len(chunks),
|
||||
text=piece,
|
||||
page_start=block.page,
|
||||
page_end=block.page,
|
||||
heading_path=list(buffer_headings),
|
||||
)
|
||||
)
|
||||
continue
|
||||
buffer.append(text)
|
||||
buffer_len += len(text)
|
||||
page_start = block.page if page_start is None else min(page_start, block.page)
|
||||
page_end = block.page if page_end is None else max(page_end, block.page)
|
||||
|
||||
flush()
|
||||
return ChunkDocumentResponse(mode=DocparseTier.ADVANCED, chunks=chunks)
|
||||
@@ -0,0 +1,178 @@
|
||||
"""Fill DOCX templates from JSON data: ``{{ dotted.path }}`` placeholders.
|
||||
|
||||
Scalar placeholders are replaced everywhere (body, tables, headers, footers).
|
||||
A table row whose text contains ``{{#items.field}}`` markers is treated as a
|
||||
row template: it is cloned once per element of the ``items`` array. Unresolved
|
||||
placeholders are left in place and reported back so the caller can surface them.
|
||||
|
||||
Formatting caveat: a placeholder split across styled runs collapses that
|
||||
paragraph's text into its first run's style.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import copy
|
||||
import io
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
from pydantic import JsonValue
|
||||
|
||||
from stirling.contracts.docparse import FillDocxRequest, FillDocxResponse
|
||||
|
||||
_PLACEHOLDER = re.compile(r"\{\{\s*(#?[\w.]+)\s*\}\}")
|
||||
|
||||
|
||||
def _resolve(path: str, data: dict[str, Any]) -> Any | None:
|
||||
node: Any = data
|
||||
for part in path.split("."):
|
||||
if isinstance(node, dict) and part in node:
|
||||
node = node[part]
|
||||
else:
|
||||
return None
|
||||
return node
|
||||
|
||||
|
||||
def _render_scalar(value: Any) -> str:
|
||||
if value is None:
|
||||
return ""
|
||||
if isinstance(value, bool):
|
||||
return "true" if value else "false"
|
||||
if isinstance(value, list):
|
||||
return ", ".join(_render_scalar(v) for v in value)
|
||||
return str(value)
|
||||
|
||||
|
||||
class _Stats:
|
||||
def __init__(self) -> None:
|
||||
self.replaced = 0
|
||||
self.missing: set[str] = set()
|
||||
|
||||
|
||||
def _fill_paragraph(paragraph: Any, data: dict[str, Any], stats: _Stats) -> None:
|
||||
text = paragraph.text
|
||||
if "{{" not in text:
|
||||
return
|
||||
|
||||
def substitute(match: re.Match[str]) -> str:
|
||||
path = match.group(1)
|
||||
if path.startswith("#"):
|
||||
return match.group(0) # row-template marker, handled at table level
|
||||
value = _resolve(path, data)
|
||||
if value is None:
|
||||
stats.missing.add(path)
|
||||
return match.group(0)
|
||||
stats.replaced += 1
|
||||
return _render_scalar(value)
|
||||
|
||||
rendered = _PLACEHOLDER.sub(substitute, text)
|
||||
if rendered == text:
|
||||
return
|
||||
# Collapse into the first run to survive placeholders split across runs.
|
||||
if paragraph.runs:
|
||||
paragraph.runs[0].text = rendered
|
||||
for run in paragraph.runs[1:]:
|
||||
run.text = ""
|
||||
else:
|
||||
paragraph.add_run(rendered)
|
||||
|
||||
|
||||
def _row_template_array(row: Any) -> str | None:
|
||||
"""Return the array name when the row carries ``{{#name.field}}`` markers."""
|
||||
names = {
|
||||
match.group(1)[1:].split(".")[0]
|
||||
for cell in row.cells
|
||||
for match in _PLACEHOLDER.finditer(cell.text)
|
||||
if match.group(1).startswith("#")
|
||||
}
|
||||
return names.pop() if len(names) == 1 else None
|
||||
|
||||
|
||||
def _fill_table(table: Any, data: dict[str, Any], stats: _Stats) -> None:
|
||||
for row in list(table.rows):
|
||||
array_name = _row_template_array(row)
|
||||
if array_name is None:
|
||||
continue
|
||||
items = _resolve(array_name, data)
|
||||
if not isinstance(items, list):
|
||||
stats.missing.add(array_name)
|
||||
continue
|
||||
for _ in items:
|
||||
new_row = copy.deepcopy(row._tr)
|
||||
row._tr.addprevious(new_row)
|
||||
# Clones sit before the template; rewrite their markers, then drop the template.
|
||||
_rewrite_cloned_rows(table, row, array_name, items, data, stats)
|
||||
row._tr.getparent().remove(row._tr)
|
||||
|
||||
|
||||
def _rewrite_cloned_rows(
|
||||
table: Any, template_row: Any, array_name: str, items: list[Any], data: dict[str, Any], stats: _Stats
|
||||
) -> None:
|
||||
marker_prefix = f"#{array_name}"
|
||||
clones = [
|
||||
r for r in table.rows if r._tr is not template_row._tr and marker_prefix in "".join(c.text for c in r.cells)
|
||||
]
|
||||
for row, item in zip(clones, items, strict=False):
|
||||
scoped = dict(data)
|
||||
scoped[array_name] = item if isinstance(item, dict) else {"value": item}
|
||||
for cell in row.cells:
|
||||
for paragraph in cell.paragraphs:
|
||||
text = paragraph.text
|
||||
|
||||
def substitute(match: re.Match[str]) -> str:
|
||||
path = match.group(1)
|
||||
if not path.startswith(marker_prefix):
|
||||
return match.group(0)
|
||||
item_path = path[1:] # "#items.field" -> "items.field"
|
||||
value = _resolve(item_path, scoped)
|
||||
if value is None and "." not in item_path:
|
||||
value = scoped[array_name].get("value") if isinstance(scoped[array_name], dict) else None
|
||||
if value is None:
|
||||
stats.missing.add(item_path)
|
||||
return match.group(0)
|
||||
stats.replaced += 1
|
||||
return _render_scalar(value)
|
||||
|
||||
rendered = _PLACEHOLDER.sub(substitute, text)
|
||||
if rendered != text:
|
||||
if paragraph.runs:
|
||||
paragraph.runs[0].text = rendered
|
||||
for run in paragraph.runs[1:]:
|
||||
run.text = ""
|
||||
else:
|
||||
paragraph.add_run(rendered)
|
||||
|
||||
|
||||
def _walk_paragraphs(document: Any) -> list[tuple[Any, Any]]:
|
||||
"""Yield (paragraph, containing table or None) across body, tables, headers, footers."""
|
||||
found: list[tuple[Any, Any]] = [(p, None) for p in document.paragraphs]
|
||||
for table in document.tables:
|
||||
for row in table.rows:
|
||||
for cell in row.cells:
|
||||
found.extend((p, table) for p in cell.paragraphs)
|
||||
for section in document.sections:
|
||||
for part in (section.header, section.footer):
|
||||
found.extend((p, None) for p in part.paragraphs)
|
||||
return found
|
||||
|
||||
|
||||
def fill_docx(request: FillDocxRequest) -> FillDocxResponse:
|
||||
import docx # local import: python-docx is small but only needed here
|
||||
|
||||
data: dict[str, JsonValue] = dict(request.data)
|
||||
document = docx.Document(io.BytesIO(base64.b64decode(request.template_base64)))
|
||||
stats = _Stats()
|
||||
|
||||
for table in document.tables:
|
||||
_fill_table(table, data, stats)
|
||||
for paragraph, _table in _walk_paragraphs(document):
|
||||
_fill_paragraph(paragraph, data, stats)
|
||||
|
||||
out = io.BytesIO()
|
||||
document.save(out)
|
||||
return FillDocxResponse(
|
||||
docx_base64=base64.b64encode(out.getvalue()).decode("ascii"),
|
||||
replaced=stats.replaced,
|
||||
missing=sorted(stats.missing),
|
||||
)
|
||||
@@ -0,0 +1,146 @@
|
||||
"""Extraction accuracy harness: score a gold-labelled case set against a live engine.
|
||||
|
||||
Case layout (one directory per document):
|
||||
|
||||
cases/
|
||||
invoice-001/
|
||||
input.pdf # the document
|
||||
expected.json # {"fieldsSchema": {...}, "fields": {"invoice_number": "INV-1", ...},
|
||||
# "pages": [{"pageNumber": 1, "text": "..."}]? (optional, for basic tier)}
|
||||
|
||||
Run:
|
||||
uv run python -m stirling.docparse.evals cases/ --engine http://localhost:5001 --output report.json
|
||||
|
||||
Scoring per field: exact match, then normalized match (case/whitespace/currency
|
||||
punctuation collapsed, numeric tolerance 1e-6). The report aggregates per-case
|
||||
and overall accuracy so pipeline changes can be regression-tracked. Stdlib
|
||||
HTTP only - the harness must run anywhere the engine runs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
_NORMALIZE_STRIP = re.compile(r"[\s,€$£%]+")
|
||||
|
||||
|
||||
def normalized_equal(expected: Any, actual: Any) -> bool:
|
||||
if expected is None or actual is None:
|
||||
return expected is actual
|
||||
try:
|
||||
return abs(float(expected) - float(actual)) < 1e-6
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if isinstance(expected, list) and isinstance(actual, list):
|
||||
return len(expected) == len(actual) and all(normalized_equal(e, a) for e, a in zip(expected, actual))
|
||||
return _NORMALIZE_STRIP.sub("", str(expected)).casefold() == _NORMALIZE_STRIP.sub("", str(actual)).casefold()
|
||||
|
||||
|
||||
def _call_extract(engine: str, secret: str | None, payload: dict[str, Any]) -> dict[str, Any]:
|
||||
request = urllib.request.Request(
|
||||
f"{engine.rstrip('/')}/api/v1/docparse/extract",
|
||||
data=json.dumps(payload).encode("utf-8"),
|
||||
headers={"Content-Type": "application/json", **({"X-Engine-Auth": secret} if secret else {})},
|
||||
method="POST",
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=600) as response:
|
||||
return json.loads(response.read().decode("utf-8"))
|
||||
|
||||
|
||||
def score_case(engine: str, secret: str | None, case_dir: Path) -> dict[str, Any]:
|
||||
expected_spec = json.loads((case_dir / "expected.json").read_text(encoding="utf-8"))
|
||||
payload: dict[str, Any] = {
|
||||
"fileName": "input.pdf",
|
||||
"fieldsSchema": expected_spec["fieldsSchema"],
|
||||
}
|
||||
input_pdf = case_dir / "input.pdf"
|
||||
if input_pdf.exists():
|
||||
payload["contentBase64"] = base64.b64encode(input_pdf.read_bytes()).decode("ascii")
|
||||
if expected_spec.get("pages"):
|
||||
payload["pages"] = expected_spec["pages"]
|
||||
|
||||
response = _call_extract(engine, secret, payload)
|
||||
actual = {field["name"]: field for field in response.get("fields", [])}
|
||||
|
||||
fields: list[dict[str, Any]] = []
|
||||
exact = 0
|
||||
normalized = 0
|
||||
for name, expected_value in expected_spec.get("fields", {}).items():
|
||||
actual_field = actual.get(name, {})
|
||||
actual_value = actual_field.get("value")
|
||||
is_exact = expected_value == actual_value
|
||||
is_normalized = is_exact or normalized_equal(expected_value, actual_value)
|
||||
exact += is_exact
|
||||
normalized += is_normalized
|
||||
fields.append(
|
||||
{
|
||||
"name": name,
|
||||
"expected": expected_value,
|
||||
"actual": actual_value,
|
||||
"exact": is_exact,
|
||||
"normalized": is_normalized,
|
||||
"confidence": actual_field.get("confidence"),
|
||||
"cited": bool(actual_field.get("citations")),
|
||||
}
|
||||
)
|
||||
|
||||
total = len(fields) or 1
|
||||
return {
|
||||
"case": case_dir.name,
|
||||
"mode": response.get("mode"),
|
||||
"fields": fields,
|
||||
"exactAccuracy": round(exact / total, 4),
|
||||
"normalizedAccuracy": round(normalized / total, 4),
|
||||
}
|
||||
|
||||
|
||||
def run(cases_root: Path, engine: str, secret: str | None) -> dict[str, Any]:
|
||||
case_dirs = sorted(d for d in cases_root.iterdir() if d.is_dir() and (d / "expected.json").exists())
|
||||
results: list[dict[str, Any]] = []
|
||||
failures: list[dict[str, str]] = []
|
||||
for case_dir in case_dirs:
|
||||
try:
|
||||
results.append(score_case(engine, secret, case_dir))
|
||||
except (urllib.error.URLError, OSError, KeyError, ValueError) as error:
|
||||
failures.append({"case": case_dir.name, "error": str(error)})
|
||||
|
||||
scored = [r for r in results if r["fields"]]
|
||||
overall = {
|
||||
"cases": len(case_dirs),
|
||||
"scored": len(scored),
|
||||
"errors": failures,
|
||||
"exactAccuracy": round(sum(r["exactAccuracy"] for r in scored) / len(scored), 4) if scored else 0.0,
|
||||
"normalizedAccuracy": round(sum(r["normalizedAccuracy"] for r in scored) / len(scored), 4) if scored else 0.0,
|
||||
"results": results,
|
||||
}
|
||||
return overall
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Score docparse extraction against a gold case set.")
|
||||
parser.add_argument("cases", help="Directory of case subdirectories")
|
||||
parser.add_argument("--engine", default="http://localhost:5001", help="Engine base URL")
|
||||
parser.add_argument("--secret", default=None, help="X-Engine-Auth shared secret, if the engine requires it")
|
||||
parser.add_argument("--output", default=None, help="Write the JSON report here (default: stdout)")
|
||||
args = parser.parse_args()
|
||||
|
||||
report = run(Path(args.cases), args.engine, args.secret)
|
||||
rendered = json.dumps(report, indent=2)
|
||||
if args.output:
|
||||
Path(args.output).write_text(rendered, encoding="utf-8")
|
||||
print(f"exact={report['exactAccuracy']} normalized={report['normalizedAccuracy']} -> {args.output}")
|
||||
else:
|
||||
print(rendered)
|
||||
return 0 if not report["errors"] else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,320 @@
|
||||
"""Schema-driven field extraction with grounded citations and confidence.
|
||||
|
||||
The caller supplies a JSON Schema (subset: scalar types, enums, arrays of
|
||||
scalars, nested objects). We build a dynamic pydantic output model where every
|
||||
leaf answers ``{value, quote, confidence}``, run one smart-model pass, then
|
||||
ground each quote against the page text in code. Model confidence is damped
|
||||
when a quote can't be found - the model asserts, the grounding decides how
|
||||
much to believe it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, Field, JsonValue, create_model
|
||||
from pydantic_ai import Agent
|
||||
|
||||
from stirling.agents.output_mode import output_retries, structured_output
|
||||
from stirling.contracts.docparse import (
|
||||
DocparseTier,
|
||||
ExtractedField,
|
||||
ExtractFieldsRequest,
|
||||
ExtractFieldsResponse,
|
||||
FieldCitation,
|
||||
ParseDocumentResponse,
|
||||
)
|
||||
from stirling.contracts.documents import PageText
|
||||
from stirling.models import ApiModel
|
||||
from stirling.services import AppRuntime
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Confidence multiplier when the supporting quote can't be found in the document.
|
||||
UNGROUNDED_PENALTY = 0.6
|
||||
# Floor when the model omitted quote/confidence but the value itself is found
|
||||
# verbatim in the document - the grounding is real even if the model was terse.
|
||||
VALUE_GROUNDED_FLOOR = 0.5
|
||||
MAX_SCHEMA_DEPTH = 3
|
||||
MAX_FIELDS = 100
|
||||
|
||||
_SYSTEM_PROMPT = (
|
||||
"You extract structured fields from a document.\n"
|
||||
"\n"
|
||||
"Rules:\n"
|
||||
"- For every field, return the value exactly as the schema types it, a short VERBATIM quote "
|
||||
"from the document that supports it, and your confidence from 0.0 to 1.0.\n"
|
||||
"- The quote must be copied character-for-character from the document text, at most 200 characters.\n"
|
||||
"- If the document does not contain the field, return value null, quote null, confidence 0.0. "
|
||||
"Never guess or fabricate.\n"
|
||||
"- Dates: return them formatted as the schema/description asks; quote the original text.\n"
|
||||
"- The document may be in any language."
|
||||
)
|
||||
|
||||
|
||||
class SchemaError(ValueError):
|
||||
"""The supplied JSON Schema is outside the supported subset."""
|
||||
|
||||
|
||||
def _scalar_type(spec: dict[str, Any]) -> Any:
|
||||
# Enums stay str-typed; the allowed values travel in the field description
|
||||
# (dynamic Literal types don't typecheck and local models handle them badly).
|
||||
match spec.get("type"):
|
||||
case "string":
|
||||
return str
|
||||
case "integer":
|
||||
return int
|
||||
case "number":
|
||||
return float
|
||||
case "boolean":
|
||||
return bool
|
||||
case _:
|
||||
raise SchemaError(f"Unsupported schema type: {spec.get('type')!r}")
|
||||
|
||||
|
||||
def _describe(spec: dict[str, Any]) -> str | None:
|
||||
description = spec.get("description") if isinstance(spec.get("description"), str) else None
|
||||
enum = spec.get("enum")
|
||||
if isinstance(enum, list) and enum:
|
||||
allowed = ", ".join(str(v) for v in enum)
|
||||
description = f"{description + ' ' if description else ''}Allowed values: {allowed}."
|
||||
return description
|
||||
|
||||
|
||||
def _leaf_answer_model(name: str, value_type: Any, description: str | None) -> type[BaseModel]:
|
||||
return create_model(
|
||||
f"Answer_{re.sub(r'[^A-Za-z0-9]', '_', name)}",
|
||||
__base__=ApiModel,
|
||||
value=(value_type | None, Field(default=None, description=description or None)),
|
||||
quote=(str | None, Field(default=None, max_length=400)),
|
||||
confidence=(float, Field(default=0.0, ge=0.0, le=1.0)),
|
||||
)
|
||||
|
||||
|
||||
def build_output_model(
|
||||
fields_schema: dict[str, Any], *, _depth: int = 0, _name: str = "ExtractionOutput"
|
||||
) -> type[BaseModel]:
|
||||
"""Turn the caller's JSON Schema into a pydantic model of leaf answers."""
|
||||
if _depth > MAX_SCHEMA_DEPTH:
|
||||
raise SchemaError(f"Schema nesting deeper than {MAX_SCHEMA_DEPTH} is not supported")
|
||||
properties = fields_schema.get("properties")
|
||||
if not isinstance(properties, dict) or not properties:
|
||||
raise SchemaError("Schema must be an object with a non-empty 'properties' map")
|
||||
if len(properties) > MAX_FIELDS:
|
||||
raise SchemaError(f"Schema has more than {MAX_FIELDS} fields")
|
||||
|
||||
model_fields: dict[str, Any] = {}
|
||||
for raw_name, spec in properties.items():
|
||||
name = str(raw_name)
|
||||
if not re.fullmatch(r"[A-Za-z_][A-Za-z0-9_]*", name):
|
||||
raise SchemaError(f"Field name {name!r} must be a valid identifier")
|
||||
if not isinstance(spec, dict):
|
||||
raise SchemaError(f"Field {name!r} must map to a schema object")
|
||||
description = _describe(spec)
|
||||
|
||||
if spec.get("type") == "object":
|
||||
nested = build_output_model(spec, _depth=_depth + 1, _name=f"{_name}_{name}")
|
||||
model_fields[name] = (nested, Field(...))
|
||||
elif spec.get("type") == "array":
|
||||
items = spec.get("items")
|
||||
if not isinstance(items, dict):
|
||||
raise SchemaError(f"Array field {name!r} needs an 'items' schema")
|
||||
if items.get("type") == "object":
|
||||
raise SchemaError(f"Array field {name!r}: arrays of objects are not supported yet")
|
||||
item_type: Any = _scalar_type(items)
|
||||
answer = _leaf_answer_model(name, list[item_type], description)
|
||||
model_fields[name] = (answer, Field(...))
|
||||
else:
|
||||
answer = _leaf_answer_model(name, _scalar_type(spec), description)
|
||||
model_fields[name] = (answer, Field(...))
|
||||
|
||||
return create_model(_name, __base__=ApiModel, **model_fields)
|
||||
|
||||
|
||||
_WHITESPACE = re.compile(r"\s+")
|
||||
|
||||
|
||||
def _normalize(text: str) -> str:
|
||||
return _WHITESPACE.sub(" ", text).strip().casefold()
|
||||
|
||||
|
||||
def find_quote(quote: str, pages: list[PageText]) -> tuple[int, int, int] | None:
|
||||
"""Locate ``quote`` in the page texts, whitespace-insensitively.
|
||||
|
||||
Returns (page_number, start_offset, end_offset) into the page's raw text,
|
||||
or None. Offsets are approximate under whitespace collapsing: we search the
|
||||
normalized page, then map back by counting non-space characters.
|
||||
"""
|
||||
needle = _normalize(quote)
|
||||
if not needle:
|
||||
return None
|
||||
for page in pages:
|
||||
haystack = _normalize(page.text)
|
||||
idx = haystack.find(needle)
|
||||
if idx < 0:
|
||||
continue
|
||||
start = _denormalize_offset(page.text, idx)
|
||||
end = _denormalize_offset(page.text, idx + len(needle))
|
||||
return page.page_number, start, min(end, len(page.text))
|
||||
return None
|
||||
|
||||
|
||||
def _denormalize_offset(raw: str, normalized_offset: int) -> int:
|
||||
"""Map an offset in the normalized string back into the raw string."""
|
||||
count = 0
|
||||
in_space = True # leading whitespace is stripped by _normalize
|
||||
for i, ch in enumerate(raw):
|
||||
if ch.isspace():
|
||||
if in_space:
|
||||
continue
|
||||
in_space = True
|
||||
else:
|
||||
in_space = False
|
||||
if count >= normalized_offset:
|
||||
return i
|
||||
count += 1
|
||||
return len(raw)
|
||||
|
||||
|
||||
def _bbox_for_quote(quote: str, parse: ParseDocumentResponse | None) -> list[float] | None:
|
||||
if parse is None:
|
||||
return None
|
||||
needle = _normalize(quote)
|
||||
if not needle:
|
||||
return None
|
||||
for block in parse.blocks:
|
||||
if block.bbox is not None and needle in _normalize(block.text):
|
||||
return block.bbox
|
||||
return None
|
||||
|
||||
|
||||
def flatten_answers(output: BaseModel, prefix: str = "") -> list[tuple[str, JsonValue, str | None, float]]:
|
||||
"""Walk the dynamic output model into (dotted_name, value, quote, confidence) leaves."""
|
||||
leaves: list[tuple[str, JsonValue, str | None, float]] = []
|
||||
for name in type(output).model_fields:
|
||||
node = getattr(output, name)
|
||||
dotted = f"{prefix}{name}"
|
||||
if isinstance(node, BaseModel) and "confidence" in type(node).model_fields:
|
||||
value = getattr(node, "value", None)
|
||||
quote = getattr(node, "quote", None)
|
||||
confidence = float(getattr(node, "confidence", 0.0) or 0.0)
|
||||
leaves.append((dotted, value, quote, confidence))
|
||||
elif isinstance(node, BaseModel):
|
||||
leaves.extend(flatten_answers(node, prefix=f"{dotted}."))
|
||||
return leaves
|
||||
|
||||
|
||||
def _format_pages(pages: list[PageText], max_characters: int) -> str:
|
||||
parts: list[str] = []
|
||||
used = 0
|
||||
for page in pages:
|
||||
snippet = page.text[: max(0, max_characters - used)]
|
||||
parts.append(f"[Page {page.page_number}]\n{snippet}")
|
||||
used += len(snippet)
|
||||
if used >= max_characters:
|
||||
break
|
||||
return "\n\n".join(parts) if parts else "(no extractable text)"
|
||||
|
||||
|
||||
def pages_from_parse(parse: ParseDocumentResponse) -> list[PageText]:
|
||||
"""Rebuild per-page text from parse blocks (advanced path with no caller text)."""
|
||||
by_page: dict[int, list[str]] = {}
|
||||
for block in parse.blocks:
|
||||
by_page.setdefault(block.page, []).append(block.text)
|
||||
return [PageText(page_number=n, text="\n".join(t)) for n, t in sorted(by_page.items())]
|
||||
|
||||
|
||||
class ExtractFieldsAgent:
|
||||
"""One smart-model pass over the document, then code-side grounding."""
|
||||
|
||||
def __init__(self, runtime: AppRuntime) -> None:
|
||||
self.runtime = runtime
|
||||
|
||||
async def extract(
|
||||
self,
|
||||
request: ExtractFieldsRequest,
|
||||
pages: list[PageText],
|
||||
parse: ParseDocumentResponse | None,
|
||||
) -> ExtractFieldsResponse:
|
||||
output_model = build_output_model(dict(request.fields_schema))
|
||||
provider = self.runtime.settings.chat_provider
|
||||
agent: Agent[None, BaseModel] = Agent(
|
||||
model=self.runtime.smart_model,
|
||||
output_type=structured_output([output_model], chat_provider=provider),
|
||||
system_prompt=_SYSTEM_PROMPT,
|
||||
model_settings=self.runtime.smart_model_settings,
|
||||
retries=output_retries(provider),
|
||||
)
|
||||
prompt = self._build_prompt(request, pages)
|
||||
result = await agent.run(prompt)
|
||||
fields = self._ground(result.output, pages, parse)
|
||||
overall = round(min((f.confidence for f in fields), default=0.0), 4)
|
||||
tier = DocparseTier.ADVANCED if parse is not None else DocparseTier.BASIC
|
||||
return ExtractFieldsResponse(mode=tier, fields=fields, overall_confidence=overall)
|
||||
|
||||
def _build_prompt(self, request: ExtractFieldsRequest, pages: list[PageText]) -> str:
|
||||
instructions = f"Additional instructions: {request.instructions}\n\n" if request.instructions else ""
|
||||
return (
|
||||
f"{instructions}"
|
||||
f"Document file name: {request.file_name}\n"
|
||||
f"Document content:\n{_format_pages(pages, self.runtime.settings.max_characters)}"
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _ground(
|
||||
output: BaseModel,
|
||||
pages: list[PageText],
|
||||
parse: ParseDocumentResponse | None,
|
||||
) -> list[ExtractedField]:
|
||||
fields: list[ExtractedField] = []
|
||||
for name, value, quote, model_confidence in flatten_answers(output):
|
||||
citations: list[FieldCitation] = []
|
||||
confidence = max(0.0, min(1.0, model_confidence))
|
||||
if value is None:
|
||||
confidence = 0.0
|
||||
elif quote:
|
||||
located = find_quote(quote, pages)
|
||||
if located is not None:
|
||||
page_number, start, end = located
|
||||
citations.append(
|
||||
FieldCitation(
|
||||
page=page_number,
|
||||
bbox=_bbox_for_quote(quote, parse),
|
||||
quote=quote,
|
||||
start_offset=start,
|
||||
end_offset=end,
|
||||
)
|
||||
)
|
||||
else:
|
||||
citations.append(FieldCitation(page=None, bbox=None, quote=quote))
|
||||
confidence *= UNGROUNDED_PENALTY
|
||||
else:
|
||||
# Terse models (local Ollama especially) often skip the quote;
|
||||
# grounding the value itself keeps citations and a usable score.
|
||||
value_text = _render_value(value)
|
||||
located = find_quote(value_text, pages) if value_text else None
|
||||
if located is not None:
|
||||
page_number, start, end = located
|
||||
citations.append(
|
||||
FieldCitation(
|
||||
page=page_number,
|
||||
bbox=_bbox_for_quote(value_text, parse),
|
||||
quote=value_text,
|
||||
start_offset=start,
|
||||
end_offset=end,
|
||||
)
|
||||
)
|
||||
confidence = max(confidence, VALUE_GROUNDED_FLOOR)
|
||||
else:
|
||||
confidence *= UNGROUNDED_PENALTY
|
||||
fields.append(ExtractedField(name=name, value=value, confidence=round(confidence, 4), citations=citations))
|
||||
return fields
|
||||
|
||||
|
||||
def _render_value(value: JsonValue) -> str:
|
||||
"""A searchable text form of a leaf value; empty when nothing sensible exists."""
|
||||
if value is None or isinstance(value, (dict, list)) or isinstance(value, bool):
|
||||
return ""
|
||||
return str(value)
|
||||
@@ -0,0 +1,188 @@
|
||||
"""Advanced-tier document parsing via Docling.
|
||||
|
||||
Everything Docling is imported lazily through :mod:`importlib` so this module
|
||||
imports cleanly (and pyright passes) when the addon isn't installed. Callers
|
||||
must check :func:`stirling.docparse.capability.probe_capabilities` first; the
|
||||
route returns 501 otherwise.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import io
|
||||
import logging
|
||||
import threading
|
||||
from typing import Any
|
||||
|
||||
from stirling.contracts.docparse import (
|
||||
BlockType,
|
||||
DocBlock,
|
||||
DocparseTier,
|
||||
DocTable,
|
||||
ParseDocumentResponse,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_converter_lock = threading.Lock()
|
||||
_converter: Any | None = None
|
||||
_converter_key: tuple[str, bool] | None = None
|
||||
|
||||
# Docling DocItemLabel values → our normalized block vocabulary.
|
||||
_LABEL_MAP: dict[str, BlockType] = {
|
||||
"title": BlockType.HEADING,
|
||||
"section_header": BlockType.HEADING,
|
||||
"paragraph": BlockType.PARAGRAPH,
|
||||
"text": BlockType.PARAGRAPH,
|
||||
"list_item": BlockType.LIST_ITEM,
|
||||
"table": BlockType.TABLE,
|
||||
"picture": BlockType.FIGURE,
|
||||
"chart": BlockType.FIGURE,
|
||||
"caption": BlockType.CAPTION,
|
||||
"code": BlockType.CODE,
|
||||
"formula": BlockType.FORMULA,
|
||||
"page_header": BlockType.PAGE_HEADER,
|
||||
"page_footer": BlockType.PAGE_FOOTER,
|
||||
"footnote": BlockType.FOOTNOTE,
|
||||
}
|
||||
|
||||
|
||||
def _get_converter(artifacts_path: str | None, with_ocr: bool) -> Any:
|
||||
"""Build (once) and cache the Docling converter; model load costs seconds."""
|
||||
global _converter, _converter_key
|
||||
key = (artifacts_path or "", with_ocr)
|
||||
with _converter_lock:
|
||||
if _converter is not None and _converter_key == key:
|
||||
return _converter
|
||||
|
||||
pdf_options_mod = importlib.import_module("docling.datamodel.pipeline_options")
|
||||
converter_mod = importlib.import_module("docling.document_converter")
|
||||
base_models = importlib.import_module("docling.datamodel.base_models")
|
||||
|
||||
pipeline_options = pdf_options_mod.PdfPipelineOptions()
|
||||
pipeline_options.do_ocr = with_ocr
|
||||
pipeline_options.do_table_structure = True
|
||||
if artifacts_path:
|
||||
pipeline_options.artifacts_path = artifacts_path
|
||||
# Never reach the network when a prefetched model dir is configured.
|
||||
pipeline_options.enable_remote_services = False
|
||||
|
||||
input_format = base_models.InputFormat.PDF
|
||||
pdf_format_option = converter_mod.PdfFormatOption(pipeline_options=pipeline_options)
|
||||
_converter = converter_mod.DocumentConverter(format_options={input_format: pdf_format_option})
|
||||
_converter_key = key
|
||||
return _converter
|
||||
|
||||
|
||||
def _normalize_bbox(prov: Any, page_sizes: dict[int, tuple[float, float]]) -> tuple[int, list[float] | None]:
|
||||
"""Docling prov → (page, [x0, y0, x1, y1] normalized, top-left origin)."""
|
||||
page_no = int(getattr(prov, "page_no", 1) or 1)
|
||||
bbox = getattr(prov, "bbox", None)
|
||||
size = page_sizes.get(page_no)
|
||||
if bbox is None or size is None or size[0] <= 0 or size[1] <= 0:
|
||||
return page_no, None
|
||||
width, height = size
|
||||
try:
|
||||
# Docling boxes are bottom-left origin; flip to top-left before normalizing.
|
||||
top_left = bbox.to_top_left_origin(page_height=height) if hasattr(bbox, "to_top_left_origin") else bbox
|
||||
x0 = float(top_left.l) / width
|
||||
y0 = float(top_left.t) / height
|
||||
x1 = float(top_left.r) / width
|
||||
y1 = float(top_left.b) / height
|
||||
except (AttributeError, TypeError, ValueError):
|
||||
return page_no, None
|
||||
clamp = lambda v: max(0.0, min(1.0, v)) # noqa: E731
|
||||
x0, x1 = sorted((clamp(x0), clamp(x1)))
|
||||
y0, y1 = sorted((clamp(y0), clamp(y1)))
|
||||
return page_no, [round(x0, 5), round(y0, 5), round(x1, 5), round(y1, 5)]
|
||||
|
||||
|
||||
def _document_confidence(result: Any) -> float | None:
|
||||
"""Pull a single 0..1 confidence out of Docling's confidence report, if any."""
|
||||
report = getattr(result, "confidence", None)
|
||||
if report is None:
|
||||
return None
|
||||
for attr in ("mean_score", "mean_grade_score", "score"):
|
||||
value = getattr(report, attr, None)
|
||||
if isinstance(value, (int, float)) and 0.0 <= float(value) <= 1.0:
|
||||
return round(float(value), 4)
|
||||
return None
|
||||
|
||||
|
||||
def _table_cells(item: Any) -> list[list[str]]:
|
||||
data = getattr(item, "data", None)
|
||||
grid = getattr(data, "grid", None) or []
|
||||
cells: list[list[str]] = []
|
||||
for row in grid:
|
||||
cells.append([str(getattr(cell, "text", "") or "") for cell in row])
|
||||
return cells
|
||||
|
||||
|
||||
def _cells_to_markdown(cells: list[list[str]]) -> str:
|
||||
if not cells:
|
||||
return ""
|
||||
esc = lambda s: s.replace("|", "\\|").replace("\n", " ") # noqa: E731
|
||||
lines = ["| " + " | ".join(esc(c) for c in cells[0]) + " |"]
|
||||
lines.append("|" + "---|" * len(cells[0]))
|
||||
for row in cells[1:]:
|
||||
lines.append("| " + " | ".join(esc(c) for c in row) + " |")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def parse_pdf_bytes(
|
||||
data: bytes,
|
||||
file_name: str,
|
||||
*,
|
||||
with_ocr: bool = True,
|
||||
artifacts_path: str | None = None,
|
||||
) -> ParseDocumentResponse:
|
||||
"""Synchronous, CPU-heavy; call from a worker thread (routes use ``anyio.to_thread``)."""
|
||||
io_mod = importlib.import_module("docling_core.types.io")
|
||||
converter = _get_converter(artifacts_path, with_ocr)
|
||||
|
||||
stream = io_mod.DocumentStream(name=file_name or "document.pdf", stream=io.BytesIO(data))
|
||||
try:
|
||||
result = converter.convert(stream)
|
||||
except Exception as error:
|
||||
raise ValueError(f"document could not be parsed: {error}") from error
|
||||
doc = result.document
|
||||
|
||||
page_sizes: dict[int, tuple[float, float]] = {}
|
||||
for page_no, page in (getattr(doc, "pages", None) or {}).items():
|
||||
size = getattr(page, "size", None)
|
||||
if size is not None:
|
||||
page_sizes[int(page_no)] = (float(size.width), float(size.height))
|
||||
|
||||
doc_confidence = _document_confidence(result)
|
||||
blocks: list[DocBlock] = []
|
||||
tables: list[DocTable] = []
|
||||
for item, _level in doc.iterate_items():
|
||||
provs = getattr(item, "prov", None) or []
|
||||
page, bbox = _normalize_bbox(provs[0], page_sizes) if provs else (1, None)
|
||||
label = str(getattr(item, "label", "") or "").lower()
|
||||
block_type = _LABEL_MAP.get(label, BlockType.OTHER)
|
||||
|
||||
if block_type is BlockType.TABLE:
|
||||
cells = _table_cells(item)
|
||||
tables.append(
|
||||
DocTable(
|
||||
page=page, bbox=bbox, cells=cells, markdown=_cells_to_markdown(cells), confidence=doc_confidence
|
||||
)
|
||||
)
|
||||
|
||||
text = str(getattr(item, "text", "") or "")
|
||||
if not text and block_type not in (BlockType.TABLE, BlockType.FIGURE):
|
||||
continue
|
||||
blocks.append(DocBlock(type=block_type, text=text, page=page, bbox=bbox, confidence=doc_confidence))
|
||||
|
||||
markdown = doc.export_to_markdown()
|
||||
pages = len(page_sizes) or len(getattr(doc, "pages", None) or {})
|
||||
logger.info("docparse: parsed %s: %d pages, %d blocks, %d tables", file_name, pages, len(blocks), len(tables))
|
||||
return ParseDocumentResponse(
|
||||
mode=DocparseTier.ADVANCED,
|
||||
pages=pages,
|
||||
blocks=blocks,
|
||||
tables=tables,
|
||||
markdown=markdown,
|
||||
ocr_applied=with_ocr,
|
||||
)
|
||||
@@ -0,0 +1,105 @@
|
||||
"""Content-based document splitting: an LLM finds sub-document boundaries.
|
||||
|
||||
Works entirely from caller-supplied page text (basic tier friendly); the fast
|
||||
model sees a bounded per-page preview and answers with boundary start pages,
|
||||
which are then validated in code (monotonic, in range, capped)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from pydantic import Field
|
||||
from pydantic_ai import Agent
|
||||
|
||||
from stirling.agents.output_mode import output_retries, structured_output
|
||||
from stirling.contracts.docparse import SmartSplitRequest, SmartSplitResponse, SplitPart
|
||||
from stirling.contracts.documents import PageText
|
||||
from stirling.models import ApiModel
|
||||
from stirling.services import AppRuntime
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Per-page preview budget; boundaries are recognisable from page openings.
|
||||
PAGE_PREVIEW_CHARS = 600
|
||||
|
||||
_SYSTEM_PROMPT = (
|
||||
"You split a multi-document file into its component documents.\n"
|
||||
"\n"
|
||||
"You are shown the beginning of every page. Apply the user's splitting rule and "
|
||||
"answer with every page where a NEW component document starts.\n"
|
||||
"Rules:\n"
|
||||
"- Page 1 always starts the first component.\n"
|
||||
"- Give each component a short descriptive label (e.g. 'Invoice #4821', 'Cover letter').\n"
|
||||
"- Give your confidence 0.0-1.0 per boundary.\n"
|
||||
"- If the rule doesn't match anything, return just the page-1 component spanning the whole file."
|
||||
)
|
||||
|
||||
|
||||
class _Boundary(ApiModel):
|
||||
start_page: int = Field(ge=1, description="First page of this component document.")
|
||||
label: str = Field(description="Short human label for the component.")
|
||||
confidence: float = Field(ge=0.0, le=1.0)
|
||||
|
||||
|
||||
class _SplitOutput(ApiModel):
|
||||
boundaries: list[_Boundary] = Field(default_factory=list)
|
||||
|
||||
|
||||
def _format_pages(pages: list[PageText], max_pages: int) -> str:
|
||||
shown = pages[:max_pages]
|
||||
parts = [f"[Page {p.page_number}] {p.text[:PAGE_PREVIEW_CHARS]}" for p in shown]
|
||||
if len(pages) > max_pages:
|
||||
parts.append(f"({len(pages) - max_pages} further pages omitted)")
|
||||
return "\n\n".join(parts) if parts else "(no extractable text)"
|
||||
|
||||
|
||||
def validate_boundaries(output: _SplitOutput, page_count: int, max_parts: int) -> list[SplitPart]:
|
||||
"""Coerce the model's boundaries into a clean, complete partition of 1..page_count."""
|
||||
starts: dict[int, _Boundary] = {}
|
||||
for boundary in output.boundaries:
|
||||
if 1 <= boundary.start_page <= page_count and boundary.start_page not in starts:
|
||||
starts[boundary.start_page] = boundary
|
||||
if 1 not in starts:
|
||||
starts[1] = _Boundary(start_page=1, label="Document", confidence=1.0)
|
||||
|
||||
ordered = [starts[k] for k in sorted(starts)][:max_parts]
|
||||
parts: list[SplitPart] = []
|
||||
for i, boundary in enumerate(ordered):
|
||||
end_page = ordered[i + 1].start_page - 1 if i + 1 < len(ordered) else page_count
|
||||
parts.append(
|
||||
SplitPart(
|
||||
start_page=boundary.start_page,
|
||||
end_page=end_page,
|
||||
label=boundary.label.strip() or f"Part {i + 1}",
|
||||
confidence=round(boundary.confidence, 4),
|
||||
)
|
||||
)
|
||||
return parts
|
||||
|
||||
|
||||
class SmartSplitAgent:
|
||||
def __init__(self, runtime: AppRuntime) -> None:
|
||||
self.runtime = runtime
|
||||
provider = runtime.settings.chat_provider
|
||||
self._agent: Agent[None, _SplitOutput] = Agent(
|
||||
model=runtime.fast_model,
|
||||
output_type=structured_output([_SplitOutput], chat_provider=provider),
|
||||
system_prompt=_SYSTEM_PROMPT,
|
||||
model_settings=runtime.fast_model_settings,
|
||||
retries=output_retries(provider),
|
||||
)
|
||||
|
||||
async def split(self, request: SmartSplitRequest) -> SmartSplitResponse:
|
||||
pages = request.pages
|
||||
if not pages:
|
||||
return SmartSplitResponse(parts=[])
|
||||
page_count = max(p.page_number for p in pages)
|
||||
prompt = (
|
||||
f"Splitting rule: {request.rule}\n\n"
|
||||
f"Document file name: {request.file_name}\n"
|
||||
f"Pages:\n{_format_pages(pages, self.runtime.settings.max_pages)}"
|
||||
)
|
||||
result = await self._agent.run(prompt)
|
||||
parts = validate_boundaries(result.output, page_count, request.max_parts)
|
||||
logger.info("docparse: split %s into %d parts", request.file_name, len(parts))
|
||||
return SmartSplitResponse(parts=parts)
|
||||
@@ -0,0 +1,120 @@
|
||||
"""Schema suggestion: the fast model proposes extractable fields for a document.
|
||||
|
||||
Reads a bounded window of the first pages (the fields worth extracting from a
|
||||
document type are evident from its opening) and answers with candidate fields,
|
||||
which are then validated in code: names coerced to snake_case, duplicates and
|
||||
unsupported types dropped, capped at the caller's maxFields."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
|
||||
from pydantic import Field
|
||||
from pydantic_ai import Agent
|
||||
|
||||
from stirling.agents.output_mode import output_retries, structured_output
|
||||
from stirling.contracts.docparse import (
|
||||
DocparseTier,
|
||||
SuggestedField,
|
||||
SuggestedFieldType,
|
||||
SuggestSchemaRequest,
|
||||
SuggestSchemaResponse,
|
||||
)
|
||||
from stirling.contracts.documents import PageText
|
||||
from stirling.models import ApiModel
|
||||
from stirling.services import AppRuntime
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# First pages read; a fixed window keeps cost flat regardless of length.
|
||||
WINDOW_PAGES = 3
|
||||
# Per-page preview budget; field candidates show up near page openings.
|
||||
PAGE_PREVIEW_CHARS = 2_000
|
||||
|
||||
_SYSTEM_PROMPT = (
|
||||
"You design an extraction schema for a document type.\n"
|
||||
"\n"
|
||||
"You are shown the first pages of a document. Propose the most useful fields "
|
||||
"a user would want extracted from documents of this type.\n"
|
||||
"Rules:\n"
|
||||
"- Name each field as a snake_case identifier (e.g. 'invoice_number').\n"
|
||||
"- Type each field as one of: string, number, integer, boolean.\n"
|
||||
"- Give each field a one-sentence description of what it holds.\n"
|
||||
"- Propose fields for the document TYPE, not only values visible on these pages.\n"
|
||||
"- Order fields from most to least useful."
|
||||
)
|
||||
|
||||
|
||||
class _SuggestedField(ApiModel):
|
||||
# Loosely typed on purpose: bad names/types are dropped in code, not retried.
|
||||
name: str = Field(description="snake_case identifier for the field.")
|
||||
type: str = Field(description="One of: string, number, integer, boolean.")
|
||||
description: str = ""
|
||||
|
||||
|
||||
class _SuggestOutput(ApiModel):
|
||||
fields: list[_SuggestedField] = Field(default_factory=list)
|
||||
|
||||
|
||||
_IDENTIFIER = re.compile(r"[a-z][a-z0-9_]*")
|
||||
_CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])")
|
||||
_NON_ALNUM = re.compile(r"[^A-Za-z0-9]+")
|
||||
_VALID_TYPES = {t.value for t in SuggestedFieldType}
|
||||
|
||||
|
||||
def to_snake_case(name: str) -> str:
|
||||
"""Coerce a model-proposed name into snake_case ('Invoice No.' -> 'invoice_no')."""
|
||||
return _NON_ALNUM.sub("_", _CAMEL_BOUNDARY.sub("_", name.strip())).strip("_").lower()
|
||||
|
||||
|
||||
def validate_fields(output: _SuggestOutput, max_fields: int) -> list[SuggestedField]:
|
||||
"""Keep unique snake_case names with supported types, capped at ``max_fields``."""
|
||||
kept: list[SuggestedField] = []
|
||||
seen: set[str] = set()
|
||||
for field in output.fields:
|
||||
name = to_snake_case(field.name)
|
||||
type_name = field.type.strip().lower()
|
||||
if not _IDENTIFIER.fullmatch(name) or name in seen or type_name not in _VALID_TYPES:
|
||||
continue
|
||||
seen.add(name)
|
||||
kept.append(
|
||||
SuggestedField(name=name, type=SuggestedFieldType(type_name), description=field.description.strip())
|
||||
)
|
||||
if len(kept) == max_fields:
|
||||
break
|
||||
return kept
|
||||
|
||||
|
||||
def _format_pages(pages: list[PageText]) -> str:
|
||||
shown = pages[:WINDOW_PAGES]
|
||||
parts = [f"[Page {p.page_number}] {p.text[:PAGE_PREVIEW_CHARS]}" for p in shown]
|
||||
if len(pages) > WINDOW_PAGES:
|
||||
parts.append(f"({len(pages) - WINDOW_PAGES} further pages omitted)")
|
||||
return "\n\n".join(parts) if parts else "(no extractable text)"
|
||||
|
||||
|
||||
class SuggestSchemaAgent:
|
||||
def __init__(self, runtime: AppRuntime) -> None:
|
||||
self.runtime = runtime
|
||||
provider = runtime.settings.chat_provider
|
||||
self._agent: Agent[None, _SuggestOutput] = Agent(
|
||||
model=runtime.fast_model,
|
||||
output_type=structured_output([_SuggestOutput], chat_provider=provider),
|
||||
system_prompt=_SYSTEM_PROMPT,
|
||||
model_settings=runtime.fast_model_settings,
|
||||
retries=output_retries(provider),
|
||||
)
|
||||
|
||||
async def suggest(
|
||||
self, request: SuggestSchemaRequest, pages: list[PageText], tier: DocparseTier
|
||||
) -> SuggestSchemaResponse:
|
||||
prompt = (
|
||||
f"Propose up to {request.max_fields} fields.\n\n"
|
||||
f"Document file name: {request.file_name}\n"
|
||||
f"Document content (first pages):\n{_format_pages(pages)}"
|
||||
)
|
||||
result = await self._agent.run(prompt)
|
||||
fields = validate_fields(result.output, request.max_fields)
|
||||
logger.info("docparse: suggested %d fields for %s", len(fields), request.file_name)
|
||||
return SuggestSchemaResponse(mode=tier, fields=fields)
|
||||
@@ -3,11 +3,20 @@ from __future__ import annotations
|
||||
from stirling.documents.embedder import EmbeddingService
|
||||
from stirling.documents.pgvector_store import PgVectorStore
|
||||
from stirling.documents.rag_capability import RagCapability
|
||||
from stirling.documents.service import DocumentService
|
||||
from stirling.documents.service import CollectionSearchHit, DocumentService
|
||||
from stirling.documents.sqlite_vec_store import SqliteVecStore
|
||||
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
|
||||
from stirling.documents.store import (
|
||||
CollectionSummary,
|
||||
Document,
|
||||
DocumentStore,
|
||||
SearchResult,
|
||||
StoredPage,
|
||||
StoreStats,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"CollectionSearchHit",
|
||||
"CollectionSummary",
|
||||
"Document",
|
||||
"DocumentService",
|
||||
"DocumentStore",
|
||||
@@ -16,5 +25,6 @@ __all__ = [
|
||||
"RagCapability",
|
||||
"SearchResult",
|
||||
"SqliteVecStore",
|
||||
"StoreStats",
|
||||
"StoredPage",
|
||||
]
|
||||
|
||||
@@ -10,7 +10,14 @@ from pgvector.psycopg import register_vector_async
|
||||
from psycopg_pool import AsyncConnectionPool
|
||||
|
||||
from stirling.contracts.documents import Page, PageRange
|
||||
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
|
||||
from stirling.documents.store import (
|
||||
CollectionSummary,
|
||||
Document,
|
||||
DocumentStore,
|
||||
SearchResult,
|
||||
StoredPage,
|
||||
StoreStats,
|
||||
)
|
||||
from stirling.models import OwnerId, PrincipalId
|
||||
|
||||
_READ_PERMISSION = "read"
|
||||
@@ -411,5 +418,44 @@ class PgVectorStore(DocumentStore):
|
||||
rows = await cur.fetchall()
|
||||
return [r[0] for r in rows]
|
||||
|
||||
async def list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
|
||||
if not principals:
|
||||
return []
|
||||
await self._ensure_ready()
|
||||
async with self._pool.connection() as conn:
|
||||
async with conn.cursor() as cur:
|
||||
# MIN(owner_id) mirrors _readable_owner_for's ORDER BY owner_id LIMIT 1.
|
||||
await cur.execute(
|
||||
"""
|
||||
SELECT r.collection, m.source, COUNT(d.id)
|
||||
FROM (
|
||||
SELECT collection, MIN(owner_id) AS owner_id
|
||||
FROM document_acl
|
||||
WHERE permission = %s AND principal_id = ANY(%s)
|
||||
GROUP BY collection
|
||||
) r
|
||||
JOIN documents_meta m ON m.collection = r.collection AND m.owner_id = r.owner_id
|
||||
LEFT JOIN rag_documents d ON d.collection = r.collection AND d.owner_id = r.owner_id
|
||||
GROUP BY r.collection, m.source
|
||||
ORDER BY r.collection
|
||||
""",
|
||||
(_READ_PERMISSION, list(principals)),
|
||||
)
|
||||
rows = await cur.fetchall()
|
||||
return [CollectionSummary(collection=r[0], source=r[1], chunks=int(r[2])) for r in rows]
|
||||
|
||||
async def stats(self) -> StoreStats:
|
||||
await self._ensure_ready()
|
||||
async with self._pool.connection() as conn:
|
||||
async with conn.cursor() as cur:
|
||||
await cur.execute("SELECT COUNT(DISTINCT collection) FROM documents_meta")
|
||||
doc_row = await cur.fetchone()
|
||||
await cur.execute("SELECT COUNT(*) FROM rag_documents")
|
||||
chunk_row = await cur.fetchone()
|
||||
return StoreStats(
|
||||
documents=int(doc_row[0]) if doc_row else 0,
|
||||
chunks=int(chunk_row[0]) if chunk_row else 0,
|
||||
)
|
||||
|
||||
async def close(self) -> None:
|
||||
await self._pool.close()
|
||||
|
||||
@@ -1,18 +1,36 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
|
||||
from stirling.contracts.documents import Page, PageRange, PageText
|
||||
from stirling.documents.embedder import EmbeddingService
|
||||
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
|
||||
from stirling.documents.store import (
|
||||
CollectionSummary,
|
||||
Document,
|
||||
DocumentStore,
|
||||
SearchResult,
|
||||
StoredPage,
|
||||
StoreStats,
|
||||
)
|
||||
from stirling.models import FileId, OwnerId, PrincipalId
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CollectionSearchHit:
|
||||
"""A search result tagged with the collection it came from."""
|
||||
|
||||
collection: FileId
|
||||
result: SearchResult
|
||||
|
||||
|
||||
PAGE_NUMBER_METADATA_KEY = "page_number"
|
||||
CONTENT_TYPE_METADATA_KEY = "content_type"
|
||||
PAGE_TEXT_CONTENT_TYPE = "page_text"
|
||||
DOCPARSE_CHUNK_CONTENT_TYPE = "docparse_chunk"
|
||||
|
||||
|
||||
class DocumentService:
|
||||
@@ -109,6 +127,42 @@ class DocumentService:
|
||||
await self._store.add_documents(collection, chunks, embeddings, owner_id)
|
||||
return len(chunks)
|
||||
|
||||
async def ingest_prepared(
|
||||
self,
|
||||
collection: FileId,
|
||||
chunks: list[tuple[str, dict[str, str]]],
|
||||
source: str,
|
||||
owner_id: OwnerId,
|
||||
read_principals: list[PrincipalId],
|
||||
expires_at: datetime | None,
|
||||
) -> int:
|
||||
"""Replace-ingest pre-chunked content (e.g. docparse structure-aware chunks).
|
||||
|
||||
Same lifecycle as :meth:`ingest` - wipe the ``(collection, owner_id)``
|
||||
pair, recreate it, grant reads, embed in batches, upsert - but chunk
|
||||
``(text, metadata)`` pairs arrive pre-built and no page representation
|
||||
is written. Returns the number of vector chunks indexed.
|
||||
"""
|
||||
if not read_principals:
|
||||
raise ValueError("read_principals must not be empty - every doc needs at least one reader")
|
||||
|
||||
await self._store.delete_collection(collection, owner_id)
|
||||
await self._store.ensure_collection(collection, source, owner_id, expires_at)
|
||||
await self._store.grant_read(collection, owner_id, read_principals)
|
||||
|
||||
documents: list[Document] = []
|
||||
for i, (text, metadata) in enumerate(chunks):
|
||||
if not text.strip():
|
||||
continue
|
||||
meta = {**metadata, "source": source, "chunk_index": str(i)}
|
||||
documents.append(Document(id=f"{source}:docparse:{i}", text=text, metadata=meta))
|
||||
|
||||
if not documents:
|
||||
return 0
|
||||
embeddings = await self._embedder.embed_documents([doc.text for doc in documents])
|
||||
await self._store.add_documents(collection, documents, embeddings, owner_id)
|
||||
return len(documents)
|
||||
|
||||
async def search(
|
||||
self,
|
||||
query: str,
|
||||
@@ -150,6 +204,32 @@ class DocumentService:
|
||||
all_results.sort(key=lambda r: r.score, reverse=True)
|
||||
return all_results[:k]
|
||||
|
||||
async def search_with_collections(
|
||||
self,
|
||||
query: str,
|
||||
principals: list[PrincipalId],
|
||||
top_k: int | None = None,
|
||||
) -> list[CollectionSearchHit]:
|
||||
"""Cross-collection search like :meth:`search`, but every result keeps
|
||||
the collection it came from. Restricted to what ``principals`` can read.
|
||||
"""
|
||||
k = top_k if top_k is not None else self._default_top_k
|
||||
query_embedding = await self._embedder.embed_query(query)
|
||||
hits: list[CollectionSearchHit] = []
|
||||
for col_name in await self._store.list_collections(principals):
|
||||
try:
|
||||
results = await self._store.search(col_name, query_embedding, k, principals)
|
||||
except Exception: # noqa: BLE001 - any backend error on one collection should not stop the others
|
||||
logger.warning(
|
||||
"Skipping collection %s during cross-collection search",
|
||||
col_name,
|
||||
exc_info=True,
|
||||
)
|
||||
continue
|
||||
hits.extend(CollectionSearchHit(collection=FileId(col_name), result=r) for r in results)
|
||||
hits.sort(key=lambda hit: hit.result.score, reverse=True)
|
||||
return hits[:k]
|
||||
|
||||
async def read_pages(
|
||||
self,
|
||||
collection: FileId,
|
||||
@@ -186,6 +266,10 @@ class DocumentService:
|
||||
"""List collections readable by at least one of ``principals``."""
|
||||
return [FileId(name) for name in await self._store.list_collections(principals)]
|
||||
|
||||
async def list_documents(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
|
||||
"""Per-document rollup (source, chunk count) readable by ``principals``."""
|
||||
return await self._store.list_collection_summaries(principals)
|
||||
|
||||
async def grant_read(
|
||||
self,
|
||||
collection: FileId,
|
||||
@@ -204,6 +288,10 @@ class DocumentService:
|
||||
"""Revoke a principal's access on an existing doc."""
|
||||
await self._store.revoke(collection, owner_id, principal)
|
||||
|
||||
async def stats(self) -> StoreStats:
|
||||
"""Deployment-wide document/chunk counts from the backing store."""
|
||||
return await self._store.stats()
|
||||
|
||||
async def close(self) -> None:
|
||||
"""Release the underlying store's resources."""
|
||||
await self._store.close()
|
||||
|
||||
@@ -11,7 +11,14 @@ from pathlib import Path
|
||||
import sqlite_vec
|
||||
|
||||
from stirling.contracts.documents import Page, PageRange
|
||||
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
|
||||
from stirling.documents.store import (
|
||||
CollectionSummary,
|
||||
Document,
|
||||
DocumentStore,
|
||||
SearchResult,
|
||||
StoredPage,
|
||||
StoreStats,
|
||||
)
|
||||
from stirling.models import OwnerId, PrincipalId
|
||||
|
||||
_READ_PERMISSION = "read"
|
||||
@@ -538,6 +545,42 @@ class SqliteVecStore(DocumentStore):
|
||||
).fetchall()
|
||||
return [r[0] for r in rows]
|
||||
|
||||
async def list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
|
||||
async with self._lock:
|
||||
return await asyncio.to_thread(self._sync_list_collection_summaries, principals)
|
||||
|
||||
def _sync_list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
|
||||
if not principals:
|
||||
return []
|
||||
placeholders = ",".join("?" * len(principals))
|
||||
# MIN(owner_id) mirrors _readable_owner_for's ORDER BY owner_id LIMIT 1.
|
||||
rows = self._conn.execute(
|
||||
f"""
|
||||
SELECT r.collection, m.source, COUNT(d.id)
|
||||
FROM (
|
||||
SELECT collection, MIN(owner_id) AS owner_id
|
||||
FROM document_acl
|
||||
WHERE permission = ? AND principal_id IN ({placeholders})
|
||||
GROUP BY collection
|
||||
) r
|
||||
JOIN documents_meta m ON m.collection = r.collection AND m.owner_id = r.owner_id
|
||||
LEFT JOIN documents d ON d.collection = r.collection AND d.owner_id = r.owner_id
|
||||
GROUP BY r.collection, m.source
|
||||
ORDER BY r.collection
|
||||
""",
|
||||
(_READ_PERMISSION, *principals),
|
||||
).fetchall()
|
||||
return [CollectionSummary(collection=r[0], source=r[1], chunks=int(r[2])) for r in rows]
|
||||
|
||||
async def stats(self) -> StoreStats:
|
||||
async with self._lock:
|
||||
return await asyncio.to_thread(self._sync_stats)
|
||||
|
||||
def _sync_stats(self) -> StoreStats:
|
||||
documents = self._conn.execute("SELECT COUNT(DISTINCT collection) FROM documents_meta").fetchone()[0]
|
||||
chunks = self._conn.execute("SELECT COUNT(*) FROM documents").fetchone()[0]
|
||||
return StoreStats(documents=int(documents), chunks=int(chunks))
|
||||
|
||||
async def close(self) -> None:
|
||||
async with self._lock:
|
||||
await asyncio.to_thread(self._sync_close)
|
||||
|
||||
@@ -34,6 +34,23 @@ class StoredPage:
|
||||
char_count: int
|
||||
|
||||
|
||||
@dataclass
|
||||
class StoreStats:
|
||||
"""Deployment-wide counts: distinct document ids and total vector-chunk rows."""
|
||||
|
||||
documents: int
|
||||
chunks: int
|
||||
|
||||
|
||||
@dataclass
|
||||
class CollectionSummary:
|
||||
"""Rollup row for one readable collection: stored source label + chunk count."""
|
||||
|
||||
collection: str
|
||||
source: str
|
||||
chunks: int
|
||||
|
||||
|
||||
class DocumentStore(ABC):
|
||||
"""Abstract interface for document storage backends.
|
||||
|
||||
@@ -148,6 +165,20 @@ class DocumentStore(ABC):
|
||||
async def list_collections(self, principals: list[PrincipalId]) -> list[str]:
|
||||
"""Return collection names readable by at least one of ``principals``."""
|
||||
|
||||
@abstractmethod
|
||||
async def list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
|
||||
"""Per-collection rollup (source + chunk count) readable by ``principals``.
|
||||
|
||||
Counts cover the same owner's copy a read would resolve to, so the
|
||||
rollup never leaks another tenant's content.
|
||||
"""
|
||||
|
||||
# ── deployment-wide stats (not tenant-scoped) ──────────────────────────
|
||||
|
||||
@abstractmethod
|
||||
async def stats(self) -> StoreStats:
|
||||
"""Count every owner's content: distinct document ids + total chunk rows."""
|
||||
|
||||
# ── lifecycle ──────────────────────────────────────────────────────────
|
||||
|
||||
@abstractmethod
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from stirling.contracts import PageText
|
||||
from stirling.contracts.docparse import BlockType, DocBlock, DocparseTier, ParseDocumentResponse
|
||||
from stirling.docparse.chunking import advanced_chunks, basic_chunks
|
||||
|
||||
|
||||
def test_basic_chunks_carry_page_numbers() -> None:
|
||||
pages = [
|
||||
PageText(page_number=1, text="alpha " * 200),
|
||||
PageText(page_number=2, text="beta " * 200),
|
||||
]
|
||||
result = basic_chunks(pages, chunk_size=256, overlap=32)
|
||||
assert result.mode is DocparseTier.BASIC
|
||||
assert len(result.chunks) >= 4
|
||||
assert {c.page_start for c in result.chunks} == {1, 2}
|
||||
assert [c.index for c in result.chunks] == list(range(len(result.chunks)))
|
||||
|
||||
|
||||
def _parse_fixture() -> ParseDocumentResponse:
|
||||
blocks = [
|
||||
DocBlock(type=BlockType.PAGE_HEADER, text="CONFIDENTIAL", page=1),
|
||||
DocBlock(type=BlockType.HEADING, text="1. Introduction", page=1),
|
||||
DocBlock(type=BlockType.PARAGRAPH, text="Short intro paragraph.", page=1),
|
||||
DocBlock(type=BlockType.PARAGRAPH, text="Second paragraph on same topic.", page=1),
|
||||
DocBlock(type=BlockType.HEADING, text="2. Terms", page=2),
|
||||
DocBlock(type=BlockType.PARAGRAPH, text="terms " * 300, page=2),
|
||||
]
|
||||
return ParseDocumentResponse(mode=DocparseTier.ADVANCED, pages=2, blocks=blocks, tables=[], markdown="")
|
||||
|
||||
|
||||
def test_advanced_chunks_respect_headings_and_skip_furniture() -> None:
|
||||
result = advanced_chunks(_parse_fixture(), chunk_size=512, overlap=64)
|
||||
assert result.mode is DocparseTier.ADVANCED
|
||||
texts = [c.text for c in result.chunks]
|
||||
assert all("CONFIDENTIAL" not in t for t in texts)
|
||||
intro = next(c for c in result.chunks if "Short intro" in c.text)
|
||||
assert intro.heading_path[-1] == "1. Introduction"
|
||||
assert intro.page_start == 1
|
||||
# Intro chunk must not bleed into the Terms section.
|
||||
assert "terms" not in intro.text
|
||||
|
||||
|
||||
def test_advanced_chunks_split_oversized_blocks() -> None:
|
||||
result = advanced_chunks(_parse_fixture(), chunk_size=512, overlap=64)
|
||||
terms_chunks = [c for c in result.chunks if c.heading_path and c.heading_path[-1] == "2. Terms"]
|
||||
assert len(terms_chunks) > 1
|
||||
assert all(c.page_start == 2 for c in terms_chunks)
|
||||
@@ -0,0 +1,70 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import io
|
||||
from typing import Any
|
||||
|
||||
import docx
|
||||
|
||||
from stirling.contracts.docparse import FillDocxRequest
|
||||
from stirling.docparse.docxfill import fill_docx
|
||||
|
||||
|
||||
def _template_base64() -> str:
|
||||
document = docx.Document()
|
||||
document.add_paragraph("Dear {{ customer.name }},")
|
||||
document.add_paragraph("Your total is {{ total }}.")
|
||||
document.add_paragraph("Unknown: {{ nowhere.field }}")
|
||||
table = document.add_table(rows=2, cols=2)
|
||||
table.rows[0].cells[0].text = "Item"
|
||||
table.rows[0].cells[1].text = "Price"
|
||||
table.rows[1].cells[0].text = "{{#items.name}}"
|
||||
table.rows[1].cells[1].text = "{{#items.price}}"
|
||||
buffer = io.BytesIO()
|
||||
document.save(buffer)
|
||||
return base64.b64encode(buffer.getvalue()).decode("ascii")
|
||||
|
||||
|
||||
def _load(response_base64: str) -> Any:
|
||||
return docx.Document(io.BytesIO(base64.b64decode(response_base64)))
|
||||
|
||||
|
||||
def test_fills_scalars_tables_and_reports_missing() -> None:
|
||||
request = FillDocxRequest(
|
||||
template_base64=_template_base64(),
|
||||
data={
|
||||
"customer": {"name": "ACME GmbH"},
|
||||
"total": 12.5,
|
||||
"items": [
|
||||
{"name": "Widget", "price": "2.00"},
|
||||
{"name": "Gadget", "price": "10.50"},
|
||||
],
|
||||
},
|
||||
)
|
||||
response = fill_docx(request)
|
||||
filled = _load(response.docx_base64)
|
||||
|
||||
paragraphs = [p.text for p in filled.paragraphs]
|
||||
assert "Dear ACME GmbH," in paragraphs
|
||||
assert "Your total is 12.5." in paragraphs
|
||||
# Unresolved placeholders stay put and are reported.
|
||||
assert any("{{ nowhere.field }}" in p for p in paragraphs)
|
||||
assert response.missing == ["nowhere.field"]
|
||||
|
||||
table = filled.tables[0]
|
||||
rendered_rows = [[cell.text for cell in row.cells] for row in table.rows]
|
||||
assert ["Widget", "2.00"] in rendered_rows
|
||||
assert ["Gadget", "10.50"] in rendered_rows
|
||||
# The template row is gone.
|
||||
assert all("{{#" not in cell for row in rendered_rows for cell in row)
|
||||
assert response.replaced >= 6
|
||||
|
||||
|
||||
def test_empty_items_removes_template_row() -> None:
|
||||
request = FillDocxRequest(
|
||||
template_base64=_template_base64(),
|
||||
data={"customer": {"name": "X"}, "total": 1, "items": []},
|
||||
)
|
||||
response = fill_docx(request)
|
||||
filled = _load(response.docx_base64)
|
||||
assert len(filled.tables[0].rows) == 1 # only the header remains
|
||||
@@ -0,0 +1,92 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
||||
from stirling.contracts import PageText
|
||||
from stirling.docparse.extractor import (
|
||||
VALUE_GROUNDED_FLOOR,
|
||||
ExtractFieldsAgent,
|
||||
build_output_model,
|
||||
find_quote,
|
||||
)
|
||||
from stirling.docparse.splitter import _Boundary, _SplitOutput, validate_boundaries
|
||||
|
||||
|
||||
def _pages() -> list[PageText]:
|
||||
return [
|
||||
PageText(page_number=1, text="Invoice INV-123\nTotal due: 1,240.00 EUR"),
|
||||
PageText(page_number=2, text="Payment terms\nNet 30 days from receipt."),
|
||||
]
|
||||
|
||||
|
||||
def test_find_quote_exact() -> None:
|
||||
located = find_quote("Invoice INV-123", _pages())
|
||||
assert located is not None
|
||||
page, start, end = located
|
||||
assert page == 1
|
||||
assert start == 0
|
||||
|
||||
|
||||
def test_find_quote_is_whitespace_insensitive() -> None:
|
||||
located = find_quote("Total due: 1,240.00 EUR", _pages())
|
||||
assert located is not None
|
||||
assert located[0] == 1
|
||||
|
||||
|
||||
def test_find_quote_is_case_insensitive_and_crosses_pages() -> None:
|
||||
located = find_quote("net 30 DAYS", _pages())
|
||||
assert located is not None
|
||||
assert located[0] == 2
|
||||
|
||||
|
||||
def test_find_quote_missing_returns_none() -> None:
|
||||
assert find_quote("does not appear", _pages()) is None
|
||||
assert find_quote(" ", _pages()) is None
|
||||
|
||||
|
||||
def test_validate_boundaries_partitions_cleanly() -> None:
|
||||
output = _SplitOutput(
|
||||
boundaries=[
|
||||
_Boundary(start_page=4, label="Invoice B", confidence=0.8),
|
||||
_Boundary(start_page=1, label="Invoice A", confidence=0.9),
|
||||
_Boundary(start_page=4, label="dup", confidence=0.1),
|
||||
_Boundary(start_page=99, label="out of range", confidence=0.5),
|
||||
]
|
||||
)
|
||||
parts = validate_boundaries(output, page_count=6, max_parts=10)
|
||||
assert [(p.start_page, p.end_page) for p in parts] == [(1, 3), (4, 6)]
|
||||
assert parts[0].label == "Invoice A"
|
||||
|
||||
|
||||
def test_validate_boundaries_inserts_page_one() -> None:
|
||||
output = _SplitOutput(boundaries=[_Boundary(start_page=3, label="Part", confidence=0.7)])
|
||||
parts = validate_boundaries(output, page_count=5, max_parts=10)
|
||||
assert parts[0].start_page == 1
|
||||
assert parts[1].start_page == 3
|
||||
assert parts[-1].end_page == 5
|
||||
|
||||
|
||||
def test_validate_boundaries_empty_output_spans_whole_file() -> None:
|
||||
parts = validate_boundaries(_SplitOutput(), page_count=7, max_parts=10)
|
||||
assert [(p.start_page, p.end_page) for p in parts] == [(1, 7)]
|
||||
|
||||
|
||||
def _answers(quote: str | None, confidence: float) -> BaseModel:
|
||||
model = build_output_model({"type": "object", "properties": {"invoice_number": {"type": "string"}}})
|
||||
return model.model_validate({"invoiceNumber": {"value": "INV-123", "quote": quote, "confidence": confidence}})
|
||||
|
||||
|
||||
def test_ground_falls_back_to_value_when_quote_missing() -> None:
|
||||
# Terse local models return the value but no quote; the value itself grounds.
|
||||
pages = [PageText(page_number=1, text="Invoice INV-123 issued today.")]
|
||||
fields = ExtractFieldsAgent._ground(_answers(quote=None, confidence=0.0), pages, None)
|
||||
assert fields[0].citations and fields[0].citations[0].page == 1
|
||||
assert fields[0].citations[0].quote == "INV-123"
|
||||
assert fields[0].confidence >= VALUE_GROUNDED_FLOOR
|
||||
|
||||
|
||||
def test_ground_penalises_when_nothing_grounds() -> None:
|
||||
pages = [PageText(page_number=1, text="completely unrelated text")]
|
||||
fields = ExtractFieldsAgent._ground(_answers(quote=None, confidence=0.9), pages, None)
|
||||
assert not fields[0].citations
|
||||
assert fields[0].confidence < 0.9
|
||||
@@ -0,0 +1,247 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Iterator
|
||||
from datetime import datetime
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from stirling.api import app
|
||||
from stirling.api.dependencies import get_document_service
|
||||
from stirling.documents import DocumentService, SqliteVecStore
|
||||
from stirling.models import FileId, OwnerId, PrincipalId
|
||||
|
||||
HEADERS = {"X-User-Id": "test-user"}
|
||||
|
||||
|
||||
class StubDocumentService:
|
||||
"""Records ingest_prepared calls so the route's passthrough can be asserted."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.calls: list[dict[str, Any]] = []
|
||||
|
||||
async def ingest_prepared(
|
||||
self,
|
||||
collection: FileId,
|
||||
chunks: list[tuple[str, dict[str, str]]],
|
||||
source: str,
|
||||
owner_id: OwnerId,
|
||||
read_principals: list[PrincipalId],
|
||||
expires_at: datetime | None,
|
||||
) -> int:
|
||||
self.calls.append(
|
||||
{
|
||||
"collection": collection,
|
||||
"chunks": chunks,
|
||||
"source": source,
|
||||
"owner_id": owner_id,
|
||||
"read_principals": read_principals,
|
||||
"expires_at": expires_at,
|
||||
}
|
||||
)
|
||||
return len(chunks)
|
||||
|
||||
|
||||
class StubEmbedder:
|
||||
"""Deterministic embeddings: no network, no provider needed."""
|
||||
|
||||
def __init__(self, dim: int = 8) -> None:
|
||||
self._dim = dim
|
||||
|
||||
async def embed_query(self, text: str) -> list[float]:
|
||||
h = hash(text) % 1000
|
||||
return [(h + i) / 1000.0 for i in range(self._dim)]
|
||||
|
||||
async def embed_documents(self, texts: list[str]) -> list[list[float]]:
|
||||
return [await self.embed_query(t) for t in texts]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def stub_service() -> StubDocumentService:
|
||||
return StubDocumentService()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def client(stub_service: StubDocumentService) -> Iterator[TestClient]:
|
||||
app.dependency_overrides[get_document_service] = lambda: stub_service
|
||||
try:
|
||||
yield TestClient(app)
|
||||
finally:
|
||||
app.dependency_overrides.pop(get_document_service, None)
|
||||
|
||||
|
||||
def test_rag_ingest_basic_tier_indexes_chunks_with_metadata(
|
||||
client: TestClient, stub_service: StubDocumentService
|
||||
) -> None:
|
||||
response = client.post(
|
||||
"/api/v1/docparse/rag-ingest",
|
||||
json={
|
||||
"fileName": "report.pdf",
|
||||
"documentId": "doc-1",
|
||||
"pages": [{"pageNumber": 1, "text": "para one"}, {"pageNumber": 2, "text": "para two"}],
|
||||
"chunkSize": 64,
|
||||
},
|
||||
headers=HEADERS,
|
||||
)
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["mode"] == "basic"
|
||||
assert body["documentId"] == "doc-1"
|
||||
assert body["chunksIndexed"] == 2
|
||||
assert body["pages"] == 2
|
||||
assert body["markdown"] is None
|
||||
assert body["chunks"] is None
|
||||
|
||||
call = stub_service.calls[0]
|
||||
assert call["collection"] == "doc-1"
|
||||
assert call["source"] == "docparse"
|
||||
text, metadata = call["chunks"][0]
|
||||
assert text == "para one"
|
||||
assert metadata["content_type"] == "docparse_chunk"
|
||||
assert metadata["page_start"] == "1"
|
||||
assert metadata["page_end"] == "1"
|
||||
|
||||
|
||||
def test_rag_ingest_defaults_owner_and_readers_to_caller(client: TestClient, stub_service: StubDocumentService) -> None:
|
||||
client.post(
|
||||
"/api/v1/docparse/rag-ingest",
|
||||
json={"fileName": "a.pdf", "documentId": "d", "pages": [{"pageNumber": 1, "text": "t"}]},
|
||||
headers=HEADERS,
|
||||
)
|
||||
call = stub_service.calls[0]
|
||||
assert call["owner_id"] == "test-user"
|
||||
assert call["read_principals"] == ["test-user"]
|
||||
assert call["expires_at"] is None
|
||||
|
||||
|
||||
def test_rag_ingest_passes_explicit_owner_acl_and_expiry_through(
|
||||
client: TestClient, stub_service: StubDocumentService
|
||||
) -> None:
|
||||
client.post(
|
||||
"/api/v1/docparse/rag-ingest",
|
||||
json={
|
||||
"fileName": "a.pdf",
|
||||
"documentId": "d",
|
||||
"source": "handbook.pdf",
|
||||
"ownerId": "org:acme",
|
||||
"readPrincipals": ["group:eng", "user:bob"],
|
||||
"expiresAt": "2030-01-01T00:00:00Z",
|
||||
"pages": [{"pageNumber": 1, "text": "t"}],
|
||||
},
|
||||
headers=HEADERS,
|
||||
)
|
||||
call = stub_service.calls[0]
|
||||
assert call["owner_id"] == "org:acme"
|
||||
assert call["read_principals"] == ["group:eng", "user:bob"]
|
||||
assert call["source"] == "handbook.pdf"
|
||||
assert call["expires_at"] is not None
|
||||
|
||||
|
||||
def test_rag_ingest_export_only_skips_the_store(client: TestClient, stub_service: StubDocumentService) -> None:
|
||||
response = client.post(
|
||||
"/api/v1/docparse/rag-ingest",
|
||||
json={
|
||||
"fileName": "a.pdf",
|
||||
"documentId": "d",
|
||||
"pages": [{"pageNumber": 1, "text": "alpha"}, {"pageNumber": 2, "text": "beta"}],
|
||||
"index": False,
|
||||
"includeMarkdown": True,
|
||||
"includeChunks": True,
|
||||
},
|
||||
headers=HEADERS,
|
||||
)
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert stub_service.calls == []
|
||||
assert body["chunksIndexed"] == 0
|
||||
assert body["markdown"] == "alpha\n\nbeta"
|
||||
assert [c["text"] for c in body["chunks"]] == ["alpha", "beta"]
|
||||
assert body["chunks"][0]["pageStart"] == 1
|
||||
|
||||
|
||||
def test_rag_ingest_index_off_with_no_export_is_422(client: TestClient) -> None:
|
||||
response = client.post(
|
||||
"/api/v1/docparse/rag-ingest",
|
||||
json={
|
||||
"fileName": "a.pdf",
|
||||
"documentId": "d",
|
||||
"pages": [{"pageNumber": 1, "text": "t"}],
|
||||
"index": False,
|
||||
},
|
||||
headers=HEADERS,
|
||||
)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_rag_ingest_advanced_mode_needs_the_addon(client: TestClient) -> None:
|
||||
response = client.post(
|
||||
"/api/v1/docparse/rag-ingest",
|
||||
json={
|
||||
"fileName": "x.pdf",
|
||||
"documentId": "d",
|
||||
"mode": "advanced",
|
||||
"pages": [{"pageNumber": 1, "text": "t"}],
|
||||
},
|
||||
headers=HEADERS,
|
||||
)
|
||||
assert response.status_code == 501
|
||||
assert response.json()["detail"]["addonRequired"] == "docparse"
|
||||
|
||||
|
||||
def test_rag_ingest_without_pages_is_422(client: TestClient) -> None:
|
||||
response = client.post(
|
||||
"/api/v1/docparse/rag-ingest", json={"fileName": "x.pdf", "documentId": "d"}, headers=HEADERS
|
||||
)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_rag_ingest_rejects_missing_user_header(client: TestClient) -> None:
|
||||
response = client.post(
|
||||
"/api/v1/docparse/rag-ingest",
|
||||
json={"fileName": "x.pdf", "documentId": "d", "pages": [{"pageNumber": 1, "text": "t"}]},
|
||||
)
|
||||
assert response.status_code == 401
|
||||
|
||||
|
||||
def test_rag_ingest_rejects_empty_document_id(client: TestClient) -> None:
|
||||
response = client.post(
|
||||
"/api/v1/docparse/rag-ingest",
|
||||
json={"fileName": "x.pdf", "documentId": "", "pages": [{"pageNumber": 1, "text": "t"}]},
|
||||
headers=HEADERS,
|
||||
)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_capabilities_reports_probe_shape() -> None:
|
||||
# Not asserting a value: a dev venv may genuinely have docling installed.
|
||||
client = TestClient(app)
|
||||
response = client.get("/api/v1/docparse/capabilities", headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
assert isinstance(response.json()["advancedInstalled"], bool)
|
||||
|
||||
|
||||
# ── real service: replacement semantics ─────────────────────────────────
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_rag_ingest_reingest_replaces_instead_of_duplicating() -> None:
|
||||
service = DocumentService(embedder=StubEmbedder(), store=SqliteVecStore.ephemeral(), default_top_k=3) # type: ignore[arg-type]
|
||||
app.dependency_overrides[get_document_service] = lambda: service
|
||||
try:
|
||||
client = TestClient(app)
|
||||
payload = {
|
||||
"fileName": "report.pdf",
|
||||
"documentId": "doc-replace",
|
||||
"pages": [{"pageNumber": 1, "text": "first version"}],
|
||||
}
|
||||
assert client.post("/api/v1/docparse/rag-ingest", json=payload, headers=HEADERS).status_code == 200
|
||||
payload["pages"] = [{"pageNumber": 1, "text": "second version"}]
|
||||
assert client.post("/api/v1/docparse/rag-ingest", json=payload, headers=HEADERS).status_code == 200
|
||||
finally:
|
||||
app.dependency_overrides.pop(get_document_service, None)
|
||||
|
||||
results = await service.search("version", principals=[PrincipalId("test-user")], collection=FileId("doc-replace"))
|
||||
assert [r.document.text for r in results] == ["second version"]
|
||||
assert results[0].document.metadata["content_type"] == "docparse_chunk"
|
||||
assert results[0].document.metadata["source"] == "docparse"
|
||||
@@ -0,0 +1,79 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
|
||||
from stirling.docparse.extractor import MAX_SCHEMA_DEPTH, SchemaError, build_output_model, flatten_answers
|
||||
|
||||
|
||||
def _schema(properties: dict[str, Any]) -> dict[str, Any]:
|
||||
return {"type": "object", "properties": properties}
|
||||
|
||||
|
||||
def test_builds_scalar_fields() -> None:
|
||||
model = build_output_model(
|
||||
_schema(
|
||||
{
|
||||
"invoice_number": {"type": "string", "description": "The invoice id"},
|
||||
"total": {"type": "number"},
|
||||
"line_count": {"type": "integer"},
|
||||
"paid": {"type": "boolean"},
|
||||
}
|
||||
)
|
||||
)
|
||||
instance = model.model_validate(
|
||||
{
|
||||
"invoiceNumber": {"value": "INV-1", "quote": "Invoice INV-1", "confidence": 0.9},
|
||||
"total": {"value": 12.5, "quote": None, "confidence": 0.8},
|
||||
"lineCount": {"value": 3, "quote": None, "confidence": 0.7},
|
||||
"paid": {"value": True, "quote": None, "confidence": 0.6},
|
||||
}
|
||||
)
|
||||
leaves = dict((name, value) for name, value, _quote, _conf in flatten_answers(instance))
|
||||
assert leaves == {"invoice_number": "INV-1", "total": 12.5, "line_count": 3, "paid": True}
|
||||
|
||||
|
||||
def test_nested_objects_flatten_to_dotted_names() -> None:
|
||||
model = build_output_model(_schema({"vendor": {"type": "object", "properties": {"name": {"type": "string"}}}}))
|
||||
instance = model.model_validate({"vendor": {"name": {"value": "ACME", "quote": None, "confidence": 0.5}}})
|
||||
names = [name for name, _v, _q, _c in flatten_answers(instance)]
|
||||
assert names == ["vendor.name"]
|
||||
|
||||
|
||||
def test_arrays_of_scalars() -> None:
|
||||
model = build_output_model(_schema({"tags": {"type": "array", "items": {"type": "string"}}}))
|
||||
instance = model.model_validate({"tags": {"value": ["a", "b"], "quote": None, "confidence": 1.0}})
|
||||
leaves = flatten_answers(instance)
|
||||
assert leaves[0][1] == ["a", "b"]
|
||||
|
||||
|
||||
def test_enum_lands_in_description_not_type() -> None:
|
||||
model = build_output_model(_schema({"currency": {"type": "string", "enum": ["EUR", "USD"]}}))
|
||||
answer_model = model.model_fields["currency"].annotation
|
||||
description = answer_model.model_fields["value"].description # type: ignore[union-attr]
|
||||
assert "EUR" in description and "USD" in description
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"schema",
|
||||
[
|
||||
{"type": "object", "properties": {}},
|
||||
{"type": "object"},
|
||||
{"type": "object", "properties": {"bad-name": {"type": "string"}}},
|
||||
{"type": "object", "properties": {"x": {"type": "date"}}},
|
||||
{"type": "object", "properties": {"x": {"type": "array"}}},
|
||||
{"type": "object", "properties": {"x": {"type": "array", "items": {"type": "object"}}}},
|
||||
],
|
||||
)
|
||||
def test_rejects_unsupported_schemas(schema: dict[str, Any]) -> None:
|
||||
with pytest.raises(SchemaError):
|
||||
build_output_model(schema)
|
||||
|
||||
|
||||
def test_rejects_over_deep_nesting() -> None:
|
||||
schema: dict = {"type": "string"}
|
||||
for _ in range(MAX_SCHEMA_DEPTH + 2):
|
||||
schema = {"type": "object", "properties": {"child": schema}}
|
||||
with pytest.raises(SchemaError):
|
||||
build_output_model(schema)
|
||||
@@ -0,0 +1,129 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
from collections.abc import Iterator
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from stirling.api import app
|
||||
from stirling.api.dependencies import get_suggest_schema_agent
|
||||
from stirling.api.routes import docparse as docparse_routes
|
||||
from stirling.contracts.docparse import (
|
||||
DocparseCapabilities,
|
||||
DocparseTier,
|
||||
SuggestedField,
|
||||
SuggestedFieldType,
|
||||
SuggestSchemaRequest,
|
||||
SuggestSchemaResponse,
|
||||
)
|
||||
from stirling.contracts.documents import PageText
|
||||
from stirling.docparse.suggest_schema import _SuggestedField, _SuggestOutput, to_snake_case, validate_fields
|
||||
|
||||
|
||||
def _force_addon(monkeypatch: pytest.MonkeyPatch, installed: bool) -> None:
|
||||
caps = DocparseCapabilities(advanced_installed=installed, models_available=installed)
|
||||
monkeypatch.setattr(docparse_routes, "probe_capabilities", lambda _home, refresh=False: caps)
|
||||
|
||||
|
||||
class StubSuggestAgent:
|
||||
def __init__(self) -> None:
|
||||
self.seen_pages: list[PageText] | None = None
|
||||
|
||||
async def suggest(
|
||||
self, _request: SuggestSchemaRequest, pages: list[PageText], tier: DocparseTier
|
||||
) -> SuggestSchemaResponse:
|
||||
self.seen_pages = pages
|
||||
return SuggestSchemaResponse(
|
||||
mode=tier,
|
||||
fields=[SuggestedField(name="invoice_number", type=SuggestedFieldType.STRING, description="The number.")],
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def stub_agent() -> StubSuggestAgent:
|
||||
return StubSuggestAgent()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def client(stub_agent: StubSuggestAgent) -> Iterator[TestClient]:
|
||||
app.dependency_overrides[get_suggest_schema_agent] = lambda: stub_agent
|
||||
try:
|
||||
yield TestClient(app)
|
||||
finally:
|
||||
app.dependency_overrides.pop(get_suggest_schema_agent, None)
|
||||
|
||||
|
||||
# ── validation (model proposes, code decides) ───────────────────────────
|
||||
|
||||
|
||||
def _output(*fields: tuple[str, str]) -> _SuggestOutput:
|
||||
return _SuggestOutput(fields=[_SuggestedField(name=n, type=t, description="d") for n, t in fields])
|
||||
|
||||
|
||||
def test_names_are_coerced_to_snake_case() -> None:
|
||||
fields = validate_fields(_output(("Invoice Number", "string"), ("dueDate", "string")), max_fields=8)
|
||||
assert [f.name for f in fields] == ["invoice_number", "due_date"]
|
||||
|
||||
|
||||
def test_invalid_types_are_dropped() -> None:
|
||||
fields = validate_fields(_output(("total", "money"), ("count", "integer"), ("date", "datetime")), max_fields=8)
|
||||
assert [f.name for f in fields] == ["count"]
|
||||
assert fields[0].type is SuggestedFieldType.INTEGER
|
||||
|
||||
|
||||
def test_duplicate_names_collapse_to_first() -> None:
|
||||
fields = validate_fields(_output(("total", "number"), ("Total", "string"), ("total", "integer")), max_fields=8)
|
||||
assert len(fields) == 1
|
||||
assert fields[0].type is SuggestedFieldType.NUMBER
|
||||
|
||||
|
||||
def test_result_is_capped_at_max_fields() -> None:
|
||||
fields = validate_fields(_output(*[(f"field_{i}", "string") for i in range(10)]), max_fields=3)
|
||||
assert [f.name for f in fields] == ["field_0", "field_1", "field_2"]
|
||||
|
||||
|
||||
def test_names_that_cannot_become_identifiers_are_dropped() -> None:
|
||||
fields = validate_fields(_output(("123abc", "string"), ("!!!", "string"), ("ok_name", "string")), max_fields=8)
|
||||
assert [f.name for f in fields] == ["ok_name"]
|
||||
|
||||
|
||||
def test_snake_case_coercion_examples() -> None:
|
||||
assert to_snake_case("Invoice No.") == "invoice_no"
|
||||
assert to_snake_case("invoiceNumber") == "invoice_number"
|
||||
assert to_snake_case("TotalUSD") == "total_usd"
|
||||
assert to_snake_case(" already_snake ") == "already_snake"
|
||||
|
||||
|
||||
# ── route ───────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_suggest_schema_basic_tier_from_pages(client: TestClient) -> None:
|
||||
response = client.post(
|
||||
"/api/v1/docparse/suggest-schema",
|
||||
json={"fileName": "invoice.pdf", "pages": [{"pageNumber": 1, "text": "Invoice INV-1"}]},
|
||||
)
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["mode"] == "basic"
|
||||
assert body["fields"][0] == {"name": "invoice_number", "type": "string", "description": "The number."}
|
||||
|
||||
|
||||
def test_suggest_schema_without_pages_or_content_is_422(client: TestClient) -> None:
|
||||
response = client.post("/api/v1/docparse/suggest-schema", json={"fileName": "x.pdf"})
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_suggest_schema_content_without_addon_is_422(client: TestClient, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
_force_addon(monkeypatch, installed=False)
|
||||
payload = {"fileName": "scan.pdf", "contentBase64": base64.b64encode(b"%PDF-1.4").decode()}
|
||||
response = client.post("/api/v1/docparse/suggest-schema", json=payload)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_suggest_schema_rejects_max_fields_above_cap(client: TestClient) -> None:
|
||||
response = client.post(
|
||||
"/api/v1/docparse/suggest-schema",
|
||||
json={"fileName": "x.pdf", "pages": [{"pageNumber": 1, "text": "t"}], "maxFields": 21},
|
||||
)
|
||||
assert response.status_code == 422
|
||||
@@ -7,7 +7,7 @@ from stirling.documents.chunker import chunk_text
|
||||
from stirling.documents.rag_capability import RagCapability
|
||||
from stirling.documents.service import DocumentService
|
||||
from stirling.documents.sqlite_vec_store import SqliteVecStore
|
||||
from stirling.documents.store import Document, SearchResult
|
||||
from stirling.documents.store import CollectionSummary, Document, SearchResult
|
||||
from stirling.models import FileId, OwnerId, PrincipalId
|
||||
|
||||
# Personal-doc tests reuse the same opaque string in all three roles — keeps the
|
||||
@@ -178,6 +178,27 @@ class TestSqliteVecStore:
|
||||
assert await store.list_collections(OWNER_PRINCIPALS) == []
|
||||
assert await store.list_collections(OTHER_OWNER_PRINCIPALS) == ["c.pdf"]
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_stats_count_distinct_document_ids_and_chunk_rows(self) -> None:
|
||||
"""Stats span every owner; a document id shared by two owners counts once."""
|
||||
store = SqliteVecStore.ephemeral()
|
||||
empty = await store.stats()
|
||||
assert (empty.documents, empty.chunks) == (0, 0)
|
||||
|
||||
for owner, principals, name, texts in (
|
||||
(OWNER, OWNER_PRINCIPALS, "doc-a", ["one", "two"]),
|
||||
(OWNER, OWNER_PRINCIPALS, "doc-b", ["three"]),
|
||||
(OTHER_OWNER, OTHER_OWNER_PRINCIPALS, "doc-b", ["four"]),
|
||||
):
|
||||
await store.ensure_collection(name, f"{name}.pdf", owner, None)
|
||||
await store.grant_read(name, owner, principals)
|
||||
docs = [Document(id=str(i), text=t, metadata={}) for i, t in enumerate(texts)]
|
||||
await store.add_documents(name, docs, [[1.0, 0.0]] * len(docs), owner)
|
||||
|
||||
stats = await store.stats()
|
||||
assert stats.documents == 2
|
||||
assert stats.chunks == 4
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_reap_expired_drops_collections_past_expires_at(self) -> None:
|
||||
"""TTL backstop: rows with ``expires_at`` in the past go away on reap."""
|
||||
@@ -239,6 +260,47 @@ class TestSqliteVecStore:
|
||||
# Owner still can.
|
||||
assert await store.has_collection("doc", OWNER_PRINCIPALS) is True
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_list_collection_summaries_rolls_up_readable_collections(self) -> None:
|
||||
"""Rollup: one row per readable collection with its source and chunk count."""
|
||||
store = SqliteVecStore.ephemeral()
|
||||
await store.ensure_collection("doc-a", "a.pdf", OWNER, None)
|
||||
await store.grant_read("doc-a", OWNER, OWNER_PRINCIPALS)
|
||||
docs = [Document(id="1", text="one", metadata={}), Document(id="2", text="two", metadata={})]
|
||||
await store.add_documents("doc-a", docs, [[1.0, 0.0], [0.0, 1.0]], OWNER)
|
||||
# Collection with no vector chunks yet: still listed, zero count.
|
||||
await store.ensure_collection("doc-b", "b.pdf", OWNER, None)
|
||||
await store.grant_read("doc-b", OWNER, OWNER_PRINCIPALS)
|
||||
|
||||
summaries = await store.list_collection_summaries(OWNER_PRINCIPALS)
|
||||
assert summaries == [
|
||||
CollectionSummary(collection="doc-a", source="a.pdf", chunks=2),
|
||||
CollectionSummary(collection="doc-b", source="b.pdf", chunks=0),
|
||||
]
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_list_collection_summaries_scoped_to_principals(self) -> None:
|
||||
"""One principal's rollup never lists, or counts, another owner's copy."""
|
||||
store = SqliteVecStore.ephemeral()
|
||||
await store.ensure_collection("shared-id", "alice.pdf", OWNER, None)
|
||||
await store.grant_read("shared-id", OWNER, OWNER_PRINCIPALS)
|
||||
await store.add_documents("shared-id", [Document(id="1", text="alice", metadata={})], [[1.0, 0.0]], OWNER)
|
||||
await store.ensure_collection("shared-id", "bob.pdf", OTHER_OWNER, None)
|
||||
await store.grant_read("shared-id", OTHER_OWNER, OTHER_OWNER_PRINCIPALS)
|
||||
bob_docs = [Document(id="1", text="bob", metadata={}), Document(id="2", text="bob2", metadata={})]
|
||||
await store.add_documents("shared-id", bob_docs, [[1.0, 0.0], [0.0, 1.0]], OTHER_OWNER)
|
||||
await store.ensure_collection("bob-only", "bob-only.pdf", OTHER_OWNER, None)
|
||||
await store.grant_read("bob-only", OTHER_OWNER, OTHER_OWNER_PRINCIPALS)
|
||||
|
||||
assert await store.list_collection_summaries(OWNER_PRINCIPALS) == [
|
||||
CollectionSummary(collection="shared-id", source="alice.pdf", chunks=1)
|
||||
]
|
||||
assert await store.list_collection_summaries(OTHER_OWNER_PRINCIPALS) == [
|
||||
CollectionSummary(collection="bob-only", source="bob-only.pdf", chunks=0),
|
||||
CollectionSummary(collection="shared-id", source="bob.pdf", chunks=2),
|
||||
]
|
||||
assert await store.list_collection_summaries([]) == []
|
||||
|
||||
|
||||
# DocumentService (with stub embedder)
|
||||
|
||||
@@ -398,6 +460,31 @@ class TestDocumentService:
|
||||
multi_results = await documents.search("deploy", principals=[hr_group, eng_group])
|
||||
assert len(multi_results) > 0
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_search_with_collections_tags_results_and_respects_acl(self, documents: DocumentService) -> None:
|
||||
"""Collection-tagged search only reaches collections the caller can read."""
|
||||
await documents.ingest(
|
||||
FileId("col-a"),
|
||||
_pages("Alpha content."),
|
||||
source="a.pdf",
|
||||
owner_id=OWNER,
|
||||
read_principals=OWNER_PRINCIPALS,
|
||||
expires_at=None,
|
||||
)
|
||||
await documents.ingest(
|
||||
FileId("col-b"),
|
||||
_pages("Beta content."),
|
||||
source="b.pdf",
|
||||
owner_id=OTHER_OWNER,
|
||||
read_principals=OTHER_OWNER_PRINCIPALS,
|
||||
expires_at=None,
|
||||
)
|
||||
|
||||
hits = await documents.search_with_collections("content", principals=OWNER_PRINCIPALS)
|
||||
assert hits
|
||||
assert {hit.collection for hit in hits} == {"col-a"}
|
||||
assert all(hit.result.document.text for hit in hits)
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_delete_collection(self, documents: DocumentService) -> None:
|
||||
await documents.ingest(
|
||||
|
||||
@@ -6,9 +6,10 @@ import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from stirling.api import app
|
||||
from stirling.api.dependencies import get_document_service
|
||||
from stirling.api.dependencies import get_document_service, get_knowledge_ask_agent
|
||||
from stirling.contracts import AskDocumentsRequest, AskDocumentsResponse, DocumentPassage
|
||||
from stirling.documents import Document, DocumentService, SqliteVecStore
|
||||
from stirling.models import FileId, PrincipalId, UserId
|
||||
from stirling.models import FileId, OwnerId, PrincipalId, UserId
|
||||
|
||||
USER = UserId("test-user")
|
||||
USER_PRINCIPALS = [PrincipalId("test-user")]
|
||||
@@ -348,6 +349,274 @@ def test_purge_by_owner_rejects_missing_user_header(client: TestClient) -> None:
|
||||
assert response.status_code == 401
|
||||
|
||||
|
||||
# ── GET /documents/list ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _ingest(client: TestClient, document_id: str, source: str, texts: list[str], owner: str) -> None:
|
||||
client.post(
|
||||
"/api/v1/documents",
|
||||
json={
|
||||
"documentId": document_id,
|
||||
"source": source,
|
||||
"pageText": [{"pageNumber": i, "text": t} for i, t in enumerate(texts, 1)],
|
||||
"ownerId": owner,
|
||||
"readPrincipals": [owner],
|
||||
"expiresAt": None,
|
||||
},
|
||||
headers={"X-User-Id": owner},
|
||||
)
|
||||
|
||||
|
||||
def test_list_documents_returns_caller_rollup(client: TestClient) -> None:
|
||||
_ingest(client, "list-a", "a.pdf", ["Page one text.", "Page two text."], USER)
|
||||
_ingest(client, "list-b", "b.pdf", ["Only page."], USER)
|
||||
|
||||
response = client.get("/api/v1/documents/list", headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
documents = response.json()["documents"]
|
||||
assert [d["documentId"] for d in documents] == ["list-a", "list-b"]
|
||||
by_id = {d["documentId"]: d for d in documents}
|
||||
assert by_id["list-a"]["source"] == "a.pdf"
|
||||
assert by_id["list-a"]["chunks"] >= 2
|
||||
assert by_id["list-b"]["source"] == "b.pdf"
|
||||
assert by_id["list-b"]["chunks"] >= 1
|
||||
|
||||
|
||||
def test_list_documents_empty_for_new_user(client: TestClient) -> None:
|
||||
response = client.get("/api/v1/documents/list", headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
assert response.json() == {"documents": []}
|
||||
|
||||
|
||||
def test_list_documents_hides_other_users_documents(client: TestClient) -> None:
|
||||
"""User A must never see user B's documents in the rollup."""
|
||||
_ingest(client, "alice-doc", "alice.pdf", ["alice content"], "alice")
|
||||
_ingest(client, "bob-doc", "bob.pdf", ["bob content"], "bob")
|
||||
|
||||
alice_docs = client.get("/api/v1/documents/list", headers={"X-User-Id": "alice"}).json()["documents"]
|
||||
bob_docs = client.get("/api/v1/documents/list", headers={"X-User-Id": "bob"}).json()["documents"]
|
||||
assert [d["documentId"] for d in alice_docs] == ["alice-doc"]
|
||||
assert [d["documentId"] for d in bob_docs] == ["bob-doc"]
|
||||
|
||||
|
||||
def test_list_documents_rejects_missing_user_header(client: TestClient) -> None:
|
||||
assert client.get("/api/v1/documents/list").status_code == 401
|
||||
|
||||
|
||||
# ── POST /documents/search ──────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_search_documents_maps_page_text_chunks(client: TestClient) -> None:
|
||||
"""Plain ingested chunks only carry page_number: both bounds map to it and
|
||||
the ":page:N" suffix is stripped off the source."""
|
||||
client.post(
|
||||
"/api/v1/documents",
|
||||
json={
|
||||
"documentId": "report",
|
||||
"source": "report.pdf",
|
||||
"pageText": [{"pageNumber": 3, "text": "The launch is planned for October."}],
|
||||
"ownerId": USER,
|
||||
"readPrincipals": [USER],
|
||||
"expiresAt": None,
|
||||
},
|
||||
headers=HEADERS,
|
||||
)
|
||||
|
||||
response = client.post("/api/v1/documents/search", json={"query": "launch", "topK": 5}, headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
passages = response.json()["passages"]
|
||||
assert len(passages) >= 1
|
||||
passage = passages[0]
|
||||
assert passage["documentId"] == "report"
|
||||
assert passage["pageStart"] == 3
|
||||
assert passage["pageEnd"] == 3
|
||||
assert passage["headingPath"] == []
|
||||
assert passage["source"] == "report.pdf"
|
||||
assert "launch" in passage["text"]
|
||||
assert isinstance(passage["score"], float)
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_search_documents_maps_docparse_chunk_metadata(client: TestClient, service: DocumentService) -> None:
|
||||
"""Docparse chunks carry page bounds + heading path; they map straight onto the wire."""
|
||||
await service.ingest_prepared(
|
||||
collection=FileId("dp-doc"),
|
||||
chunks=[
|
||||
(
|
||||
"Revenue grew 12% in Q2.",
|
||||
{
|
||||
"content_type": "docparse_chunk",
|
||||
"page_start": "2",
|
||||
"page_end": "3",
|
||||
"heading_path": "Report > Finance",
|
||||
},
|
||||
)
|
||||
],
|
||||
source="q2.pdf",
|
||||
owner_id=OwnerId(USER),
|
||||
read_principals=USER_PRINCIPALS,
|
||||
expires_at=None,
|
||||
)
|
||||
|
||||
response = client.post("/api/v1/documents/search", json={"query": "revenue"}, headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
passage = response.json()["passages"][0]
|
||||
assert passage["documentId"] == "dp-doc"
|
||||
assert passage["pageStart"] == 2
|
||||
assert passage["pageEnd"] == 3
|
||||
assert passage["headingPath"] == ["Report", "Finance"]
|
||||
assert passage["source"] == "q2.pdf"
|
||||
|
||||
|
||||
def test_search_documents_cannot_see_other_users_documents(client: TestClient) -> None:
|
||||
"""User B searching for user A's content must get nothing back."""
|
||||
_ingest(client, "alice-doc", "alice.pdf", ["The secret launch code is October."], "alice")
|
||||
|
||||
bob = client.post("/api/v1/documents/search", json={"query": "secret launch"}, headers={"X-User-Id": "bob"})
|
||||
assert bob.status_code == 200
|
||||
assert bob.json()["passages"] == []
|
||||
|
||||
alice = client.post("/api/v1/documents/search", json={"query": "secret launch"}, headers={"X-User-Id": "alice"})
|
||||
assert alice.json()["passages"] != []
|
||||
|
||||
|
||||
def test_search_documents_rejects_empty_query(client: TestClient) -> None:
|
||||
response = client.post("/api/v1/documents/search", json={"query": ""}, headers=HEADERS)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_search_documents_rejects_top_k_above_cap(client: TestClient) -> None:
|
||||
response = client.post("/api/v1/documents/search", json={"query": "x", "topK": 51}, headers=HEADERS)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_search_documents_rejects_missing_user_header(client: TestClient) -> None:
|
||||
assert client.post("/api/v1/documents/search", json={"query": "x"}).status_code == 401
|
||||
|
||||
|
||||
# ── POST /documents/ask ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
class StubKnowledgeAskAgent:
|
||||
"""Stands in for KnowledgeAskAgent so route tests don't call a model."""
|
||||
|
||||
def __init__(self, response: AskDocumentsResponse) -> None:
|
||||
self._response = response
|
||||
self.calls: list[tuple[AskDocumentsRequest, list[PrincipalId]]] = []
|
||||
|
||||
async def ask(self, request: AskDocumentsRequest, principals: list[PrincipalId]) -> AskDocumentsResponse:
|
||||
self.calls.append((request, principals))
|
||||
return self._response
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ask_agent() -> StubKnowledgeAskAgent:
|
||||
return StubKnowledgeAskAgent(
|
||||
AskDocumentsResponse(
|
||||
answer="Revenue grew 12% (q2.pdf p.2).",
|
||||
passages=[
|
||||
DocumentPassage(
|
||||
document_id=FileId("dp-doc"),
|
||||
text="Revenue grew 12% in Q2.",
|
||||
score=0.91,
|
||||
page_start=2,
|
||||
page_end=3,
|
||||
heading_path=["Report", "Finance"],
|
||||
source="q2.pdf",
|
||||
)
|
||||
],
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ask_client(client: TestClient, ask_agent: StubKnowledgeAskAgent) -> Iterator[TestClient]:
|
||||
app.dependency_overrides[get_knowledge_ask_agent] = lambda: ask_agent
|
||||
try:
|
||||
yield client
|
||||
finally:
|
||||
app.dependency_overrides.pop(get_knowledge_ask_agent, None)
|
||||
|
||||
|
||||
def test_ask_documents_returns_answer_and_passages(ask_client: TestClient) -> None:
|
||||
response = ask_client.post("/api/v1/documents/ask", json={"question": "How did revenue do?"}, headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["answer"] == "Revenue grew 12% (q2.pdf p.2)."
|
||||
assert body["passages"] == [
|
||||
{
|
||||
"documentId": "dp-doc",
|
||||
"text": "Revenue grew 12% in Q2.",
|
||||
"score": 0.91,
|
||||
"pageStart": 2,
|
||||
"pageEnd": 3,
|
||||
"headingPath": ["Report", "Finance"],
|
||||
"source": "q2.pdf",
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
def test_ask_documents_scopes_to_calling_user(ask_client: TestClient, ask_agent: StubKnowledgeAskAgent) -> None:
|
||||
"""The route hands the agent exactly the caller's principal set."""
|
||||
ask_client.post("/api/v1/documents/ask", json={"question": "anything"}, headers=HEADERS)
|
||||
request, principals = ask_agent.calls[0]
|
||||
assert principals == [PrincipalId(USER)]
|
||||
assert request.top_k == 8
|
||||
|
||||
|
||||
def test_ask_documents_rejects_empty_question(ask_client: TestClient) -> None:
|
||||
response = ask_client.post("/api/v1/documents/ask", json={"question": ""}, headers=HEADERS)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_ask_documents_rejects_top_k_above_cap(ask_client: TestClient) -> None:
|
||||
response = ask_client.post("/api/v1/documents/ask", json={"question": "x", "topK": 21}, headers=HEADERS)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_ask_documents_rejects_missing_user_header(ask_client: TestClient) -> None:
|
||||
assert ask_client.post("/api/v1/documents/ask", json={"question": "x"}).status_code == 401
|
||||
|
||||
|
||||
# ── GET /documents/stats ────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_stats_on_empty_store_reports_zero(client: TestClient) -> None:
|
||||
response = client.get("/api/v1/documents/stats", headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["documents"] == 0
|
||||
assert body["chunks"] == 0
|
||||
assert body["backend"] in ("sqlite", "pgvector")
|
||||
assert body["embeddingModel"]
|
||||
|
||||
|
||||
def test_stats_counts_seeded_documents_across_owners(client: TestClient) -> None:
|
||||
"""Stats are deployment-wide: both owners' content is counted."""
|
||||
for owner, doc in (("alice", "doc-a"), ("bob", "doc-b")):
|
||||
client.post(
|
||||
"/api/v1/documents",
|
||||
json={
|
||||
"documentId": doc,
|
||||
"source": f"{doc}.pdf",
|
||||
"pageText": [{"pageNumber": 1, "text": "Some content for the stats endpoint."}],
|
||||
"ownerId": owner,
|
||||
"readPrincipals": [owner],
|
||||
"expiresAt": None,
|
||||
},
|
||||
headers={"X-User-Id": owner},
|
||||
)
|
||||
response = client.get("/api/v1/documents/stats", headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["documents"] == 2
|
||||
assert body["chunks"] >= 2
|
||||
|
||||
|
||||
def test_stats_rejects_missing_user_header(client: TestClient) -> None:
|
||||
assert client.get("/api/v1/documents/stats").status_code == 401
|
||||
|
||||
|
||||
def test_delete_document_only_affects_calling_user(client: TestClient) -> None:
|
||||
"""Two users with the same document id: one user's delete must not remove the other's."""
|
||||
alice_body = {
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from stirling.agents.knowledge_ask import KnowledgeAskAgent, format_passages, passage_from_hit
|
||||
from stirling.config import AppSettings
|
||||
from stirling.contracts import AskDocumentsRequest, DocumentPassage
|
||||
from stirling.documents import CollectionSearchHit, Document, DocumentService, SearchResult, SqliteVecStore
|
||||
from stirling.models import FileId, PrincipalId
|
||||
from stirling.services import build_runtime
|
||||
|
||||
PRINCIPALS = [PrincipalId("test-user")]
|
||||
|
||||
|
||||
def _hit(metadata: dict[str, str], text: str = "chunk text", score: float = 0.8) -> CollectionSearchHit:
|
||||
return CollectionSearchHit(
|
||||
collection=FileId("doc-1"),
|
||||
result=SearchResult(document=Document(id="c1", text=text, metadata=metadata), score=score),
|
||||
)
|
||||
|
||||
|
||||
# ── passage_from_hit ────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_passage_from_hit_maps_docparse_metadata() -> None:
|
||||
passage = passage_from_hit(
|
||||
_hit({"source": "q2.pdf", "page_start": "2", "page_end": "3", "heading_path": "Report > Finance"})
|
||||
)
|
||||
assert passage.document_id == "doc-1"
|
||||
assert passage.page_start == 2
|
||||
assert passage.page_end == 3
|
||||
assert passage.heading_path == ["Report", "Finance"]
|
||||
assert passage.source == "q2.pdf"
|
||||
|
||||
|
||||
def test_passage_from_hit_falls_back_to_page_number() -> None:
|
||||
"""Plain page-text chunks: page_number fills both bounds, source drops the page suffix."""
|
||||
passage = passage_from_hit(_hit({"source": "report.pdf:page:4", "page_number": "4"}))
|
||||
assert passage.page_start == 4
|
||||
assert passage.page_end == 4
|
||||
assert passage.heading_path == []
|
||||
assert passage.source == "report.pdf"
|
||||
|
||||
|
||||
def test_passage_from_hit_tolerates_missing_and_bad_metadata() -> None:
|
||||
passage = passage_from_hit(_hit({"page_start": "not-a-number"}))
|
||||
assert passage.page_start is None
|
||||
assert passage.page_end is None
|
||||
assert passage.heading_path == []
|
||||
assert passage.source is None
|
||||
|
||||
|
||||
# ── format_passages ─────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_format_passages_includes_citation_handles() -> None:
|
||||
passages = [
|
||||
DocumentPassage(document_id=FileId("d1"), text="Alpha.", score=0.9, page_start=2, page_end=3, source="a.pdf"),
|
||||
DocumentPassage(document_id=FileId("d2"), text="Beta.", score=0.5, page_start=7, page_end=7, source="b.pdf"),
|
||||
DocumentPassage(document_id=FileId("d3"), text="Gamma.", score=0.4),
|
||||
]
|
||||
rendered = format_passages(passages)
|
||||
assert "[Passage 1 | a.pdf p.2-3]\nAlpha." in rendered
|
||||
assert "[Passage 2 | b.pdf p.7]\nBeta." in rendered
|
||||
# No source or pages: fall back to the document id alone.
|
||||
assert "[Passage 3 | d3]\nGamma." in rendered
|
||||
|
||||
|
||||
# ── KnowledgeAskAgent ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
class _StubEmbedder:
|
||||
"""Deterministic embeddings so the agent test needs no provider."""
|
||||
|
||||
async def embed_query(self, text: str) -> list[float]:
|
||||
return [1.0, 0.0]
|
||||
|
||||
async def embed_documents(self, texts: list[str]) -> list[list[float]]:
|
||||
return [[1.0, 0.0] for _ in texts]
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_ask_answers_plainly_when_nothing_retrieved(app_settings: AppSettings) -> None:
|
||||
"""Empty retrieval short-circuits: no model call, honest not-found answer."""
|
||||
documents = DocumentService(embedder=_StubEmbedder(), store=SqliteVecStore.ephemeral(), default_top_k=3) # type: ignore[arg-type]
|
||||
runtime = build_runtime(app_settings, documents=documents)
|
||||
agent = KnowledgeAskAgent(runtime)
|
||||
|
||||
response = await agent.ask(AskDocumentsRequest(question="What is the launch date?"), principals=PRINCIPALS)
|
||||
assert response.passages == []
|
||||
assert "couldn't find" in response.answer
|
||||
Generated
+1136
-13
File diff suppressed because it is too large
Load Diff
@@ -2986,6 +2986,38 @@ summary_one = "Ran 1 tool"
|
||||
summary_other = "Ran {{count}} tools"
|
||||
unknownTool = "Unknown tool"
|
||||
|
||||
[chunkDocument]
|
||||
intro = "Turns a document into retrieval-ready chunks in three layers, so answers cite the right section instead of a random page."
|
||||
processorCallout = "To index automatically, add the 'Index into knowledge base' step to an ingestion policy in the"
|
||||
processorLink = "Processor"
|
||||
submit = "Prepare chunks"
|
||||
|
||||
[chunkDocument.chunkSize]
|
||||
label = "Chunk size (characters)"
|
||||
|
||||
[chunkDocument.error]
|
||||
failed = "Failed to chunk document"
|
||||
|
||||
[chunkDocument.layers]
|
||||
chunk = "Structure-aware chunks: each carries its heading breadcrumb and page range"
|
||||
embed = "Ready to embed: exported as JSONL for any vector store"
|
||||
parse = "Layout-aware parse: headings, paragraphs, and tables are recognized as structure"
|
||||
|
||||
[chunkDocument.mode]
|
||||
advanced = "Advanced"
|
||||
auto = "Auto"
|
||||
basic = "Basic"
|
||||
label = "Mode"
|
||||
|
||||
[chunkDocument.overlap]
|
||||
label = "Overlap (characters)"
|
||||
|
||||
[chunkDocument.results]
|
||||
title = "Chunks (JSONL)"
|
||||
|
||||
[chunkDocument.settings]
|
||||
title = "Chunking settings"
|
||||
|
||||
[cloudBadge]
|
||||
tooltip = "This operation will use your cloud credits"
|
||||
|
||||
@@ -3573,6 +3605,13 @@ tags = "automation,folder,scanning,watch folder,hot folder,automatic processing,
|
||||
[devSsoGuide]
|
||||
tags = "SSO,single sign-on,authentication,SAML,OAuth,OIDC,login,enterprise,identity provider,IdP"
|
||||
|
||||
[docparse.intro]
|
||||
advancedOff = "Advanced parsing: off"
|
||||
advancedOn = "Advanced parsing: on"
|
||||
aiLayoutModel = "AI layout model"
|
||||
basicFallback = "Scanned documents fall back to basic text extraction - install the DocParse addon for layout AI"
|
||||
usesAi = "Uses AI"
|
||||
|
||||
[dropdownList]
|
||||
searchPlaceholder = "Search..."
|
||||
|
||||
@@ -3700,6 +3739,56 @@ _value = "Error"
|
||||
dismissAllErrors = "Dismiss All Errors"
|
||||
generic = "An error occurred"
|
||||
|
||||
[extractFields]
|
||||
intro = "Describe the fields you need and AI reads the document and returns each value with a confidence score and a citation you can verify."
|
||||
submit = "Extract fields"
|
||||
|
||||
[extractFields.error]
|
||||
failed = "Failed to extract fields"
|
||||
|
||||
[extractFields.fields]
|
||||
add = "Add field"
|
||||
description = "Description"
|
||||
descriptionPlaceholder = "What to look for"
|
||||
label = "Fields to extract"
|
||||
name = "Name"
|
||||
namePlaceholder = "invoice_number"
|
||||
remove = "Remove field"
|
||||
type = "Type"
|
||||
|
||||
[extractFields.instructions]
|
||||
label = "Instructions"
|
||||
placeholder = "e.g. Amounts are in EUR unless stated otherwise"
|
||||
|
||||
[extractFields.mode]
|
||||
advanced = "Advanced"
|
||||
auto = "Auto"
|
||||
basic = "Basic"
|
||||
label = "Mode"
|
||||
|
||||
[extractFields.presets]
|
||||
contract = "Contract"
|
||||
custom = "Custom"
|
||||
invoice = "Invoice"
|
||||
label = "Preset template"
|
||||
purchaseOrder = "Purchase order"
|
||||
receipt = "Receipt"
|
||||
|
||||
[extractFields.results]
|
||||
title = "Extraction report"
|
||||
|
||||
[extractFields.resultsPanel]
|
||||
title = "Extracted fields"
|
||||
|
||||
[extractFields.settings]
|
||||
title = "Extraction schema"
|
||||
|
||||
[extractFields.suggest]
|
||||
button = "Suggest fields (AI)"
|
||||
failed = "Could not suggest fields"
|
||||
failedBody = "The document could not be analyzed. Add fields manually or try again."
|
||||
needsFile = "Select a file first to suggest fields"
|
||||
|
||||
[extractImages]
|
||||
allowDuplicates = "Save duplicate images"
|
||||
selectText = "Select image format to convert extracted images to"
|
||||
@@ -4183,6 +4272,24 @@ upload = "Upload"
|
||||
uploadFile = "Upload File"
|
||||
uploadFiles = "Upload Files"
|
||||
|
||||
[fillTemplate]
|
||||
hint = "The input file must be a .docx template (not a PDF). Each placeholder in the template is replaced with the matching JSON value."
|
||||
intro = "Replaces the placeholders in a Word (.docx) template with your JSON data and returns the filled document - deterministic, no AI involved."
|
||||
submit = "Fill template"
|
||||
|
||||
[fillTemplate.data]
|
||||
invalid = "Enter a valid JSON object"
|
||||
label = "Data (JSON)"
|
||||
|
||||
[fillTemplate.error]
|
||||
failed = "Failed to fill template"
|
||||
|
||||
[fillTemplate.results]
|
||||
title = "Filled document"
|
||||
|
||||
[fillTemplate.settings]
|
||||
title = "Template data"
|
||||
|
||||
[firstLogin]
|
||||
allFieldsRequired = "All fields are required"
|
||||
changePassword = "Change Password"
|
||||
@@ -4522,6 +4629,11 @@ desc = "Change document restrictions and permissions"
|
||||
tags = "permissions,restrictions,rights,access control,allow,deny,printing,copying,editing,modify permissions,security settings,user rights"
|
||||
title = "Change Permissions"
|
||||
|
||||
[home.chunkDocument]
|
||||
desc = "Layout-aware parse to structure-aware chunks with heading breadcrumbs and page ranges, ready to embed"
|
||||
tags = "chunk,RAG,prepare,split text,segments,embedding,vector,ingest,LLM,retrieval,JSONL,overlap,knowledge base,index"
|
||||
title = "Prepare for RAG"
|
||||
|
||||
[home.compare]
|
||||
desc = "Compares and shows the differences between 2 PDF Documents"
|
||||
tags = "difference,compare,diff,compare PDFs,compare documents,find differences,show differences,changes,what changed,track changes,revisions,version compare,side by side,contrast,delta"
|
||||
@@ -4567,6 +4679,11 @@ desc = "Add or edit bookmarks and table of contents in PDF documents"
|
||||
tags = "bookmarks,contents,edit,table of contents,TOC,outline,navigation,chapters,sections,add bookmarks,edit bookmarks,PDF outline"
|
||||
title = "Edit Table of Contents"
|
||||
|
||||
[home.extractFields]
|
||||
desc = "Pull typed fields out of a document with confidence scores and citations"
|
||||
tags = "extract,fields,schema,structured data,invoice,form data,key value,confidence,citations,capture,parse"
|
||||
title = "Extract Fields"
|
||||
|
||||
[home.extractImages]
|
||||
desc = "Extracts all images from a PDF and saves them to zip"
|
||||
tags = "pull,save,export,extract images,get images,save images,export images,extract photos,extract pictures,pull images,download images,rip images,extract graphics,save photos"
|
||||
@@ -4577,6 +4694,11 @@ desc = "Extract specific pages from a PDF document"
|
||||
tags = "pull,select,copy,extract,extract pages,get pages,pull out,save pages,export pages,copy pages,select pages,specific pages"
|
||||
title = "Extract Pages"
|
||||
|
||||
[home.fillTemplate]
|
||||
desc = "Fill a DOCX template's placeholders from JSON data"
|
||||
tags = "template,DOCX,fill,merge fields,mail merge,generate document,placeholders,letters,contracts,Word"
|
||||
title = "Fill Template"
|
||||
|
||||
[home.flatten]
|
||||
desc = "Remove all interactive elements and forms from a PDF"
|
||||
tags = "simplify,remove,interactive,flatten,flatten form,remove form fields,make static,finalize form,lock form,disable editing,convert to image,non-editable"
|
||||
@@ -4630,6 +4752,11 @@ desc = "Merge multiple pages of a PDF document into a single page"
|
||||
tags = "layout,arrange,combine,N-up,2-up,4-up,multiple per page,pages per sheet,layout pages,tile,grid layout,multi-page layout,combine on page,handout"
|
||||
title = "Multi-Page Layout"
|
||||
|
||||
[home.parseDocument]
|
||||
desc = "Layout-aware parsing to structured JSON or Markdown, with optional OCR"
|
||||
tags = "parse,layout,structure,blocks,markdown,JSON,docling,OCR,scan,document understanding,convert"
|
||||
title = "Parse Document"
|
||||
|
||||
[home.pdfCommentAgent]
|
||||
desc = "Ask AI to annotate a PDF with sticky-note comments based on your prompt"
|
||||
tags = "AI,agent,comment,annotate,sticky note,review,feedback,notes"
|
||||
@@ -4743,6 +4870,11 @@ desc = "Adds signature to PDF by drawing, text or image"
|
||||
tags = "signature,autograph,e-sign,electronic signature,digital signature,sign document,approval,signoff,authorize,endorse,ink signature,handwriting"
|
||||
title = "Sign"
|
||||
|
||||
[home.smartSplit]
|
||||
desc = "Split a PDF into sub-documents using a natural-language boundary rule"
|
||||
tags = "split,smart,boundaries,separate,invoices,batches,content split,divide,rules,auto split"
|
||||
title = "Smart Split"
|
||||
|
||||
[home.split]
|
||||
desc = "Split PDFs into multiple documents"
|
||||
tags = "divide,separate,break,split,extract pages,separate pages,divide document,break apart,separate files,unbind,split by page,divide by chapter"
|
||||
@@ -5501,6 +5633,33 @@ title = "Page Ranges"
|
||||
bullet1 = "<strong>all</strong> → selects all pages"
|
||||
title = "Special Keywords"
|
||||
|
||||
[parseDocument]
|
||||
intro = "Reads the document's layout - headings, paragraphs, tables - and turns it into clean structured JSON or Markdown you can feed to other systems."
|
||||
submit = "Parse document"
|
||||
|
||||
[parseDocument.error]
|
||||
failed = "Failed to parse document"
|
||||
|
||||
[parseDocument.mode]
|
||||
advanced = "Advanced"
|
||||
auto = "Auto"
|
||||
basic = "Basic"
|
||||
label = "Mode"
|
||||
|
||||
[parseDocument.outputFormat]
|
||||
json = "JSON"
|
||||
label = "Output format"
|
||||
markdown = "Markdown"
|
||||
|
||||
[parseDocument.results]
|
||||
title = "Parsed output"
|
||||
|
||||
[parseDocument.settings]
|
||||
title = "Parse settings"
|
||||
|
||||
[parseDocument.withOcr]
|
||||
label = "Apply OCR to scanned pages (recommended)"
|
||||
|
||||
[payg.activity]
|
||||
docs = "docs"
|
||||
empty = "No billable activity yet this period."
|
||||
@@ -7764,6 +7923,23 @@ unsavedTitle = "Unsaved changes"
|
||||
uploadUnsupported = "Uploaded files aren't supported in pipelines yet, so these steps can't be saved: {{tools}}."
|
||||
usesDefaults = "Runs with default settings"
|
||||
|
||||
[portal.pipelines.builder.docIntelligence]
|
||||
chunkSize = "Chunk size (characters)"
|
||||
exportChunks = "Export chunks (JSONL)"
|
||||
exportChunksHint = "Adds a .chunks.jsonl file to the step output, ready for external embedding or indexing."
|
||||
exportMarkdown = "Export markdown"
|
||||
exportMarkdownHint = "Adds a .md rendering of the parsed document to the step output."
|
||||
index = "Index into the knowledge base"
|
||||
indexHint = "Embed and store the chunks in the built-in knowledge base. Turn off for export-only ingestion."
|
||||
mode = "Parsing mode"
|
||||
modeAdvanced = "Advanced"
|
||||
modeAuto = "Auto"
|
||||
modeBasic = "Basic"
|
||||
overlap = "Overlap (characters)"
|
||||
ragIngest = "Ingest into knowledge base"
|
||||
ragIngestHint = "Parses, chunks, embeds, and indexes each document in one step. Export toggles add corpus files (markdown, chunks JSONL) to the step output for delivery to external systems."
|
||||
section = "Document intelligence"
|
||||
|
||||
[portal.pipelines.composer]
|
||||
addTool = "Add tool"
|
||||
cancel = "Cancel"
|
||||
@@ -7860,8 +8036,12 @@ label = "Classification"
|
||||
desc = "Enforce HIPAA, GDPR, SOC 2, or FedRAMP requirements on every document."
|
||||
label = "Compliance"
|
||||
|
||||
[portal.policies.categories.docIntelligence]
|
||||
desc = "Extract the structured fields you describe from every document, with confidence scores and citations."
|
||||
label = "Document intelligence"
|
||||
|
||||
[portal.policies.categories.ingestion]
|
||||
desc = "Classify documents, extract structured data, enforce naming conventions, and normalize pages."
|
||||
desc = "Normalize incoming documents - OCR scans, flatten forms - and index them into the searchable knowledge base, or export them as a clean corpus for your own systems."
|
||||
label = "Ingestion"
|
||||
|
||||
[portal.policies.categories.retention]
|
||||
@@ -7900,18 +8080,25 @@ onViolation = "When non-compliant"
|
||||
1 = "Enforce action"
|
||||
2 = "Audit trail"
|
||||
|
||||
[portal.policies.config.ingestion]
|
||||
summary = "Classifies documents, extracts structured data, enforces naming, and normalizes pages."
|
||||
[portal.policies.config.docIntelligence]
|
||||
summary = "Extracts the fields you describe from every document, with confidence and citations."
|
||||
|
||||
[portal.policies.config.ingestion.fields]
|
||||
belowThreshold = "Below threshold"
|
||||
minConfidence = "Min confidence"
|
||||
[portal.policies.config.docIntelligence.rules]
|
||||
0 = "Extract the fields you describe, with confidence and citations"
|
||||
|
||||
[portal.policies.config.extractFields.fields]
|
||||
instructions = "Extraction guidance (optional)"
|
||||
instructionsHelp = "Plain-language hints for tricky fields, e.g. \"the invoice number is top right\"."
|
||||
schema = "Fields to extract (JSON Schema)"
|
||||
schemaHelp = "A JSON Schema object describing the fields. Each extracted value carries a confidence score and a citation."
|
||||
|
||||
[portal.policies.config.ingestion]
|
||||
summary = "Normalizes incoming documents and indexes them into the searchable knowledge base, with optional markdown/JSONL export."
|
||||
|
||||
[portal.policies.config.ingestion.rules]
|
||||
0 = "Classify"
|
||||
1 = "Extract"
|
||||
2 = "Name"
|
||||
3 = "Normalize"
|
||||
0 = "OCR scans and flatten forms so every page is readable"
|
||||
1 = "Parse, chunk, embed, and index into the knowledge base"
|
||||
2 = "Optionally export markdown and chunks JSONL for external systems"
|
||||
|
||||
[portal.policies.config.purview]
|
||||
label = "Apply a Microsoft Purview sensitivity label"
|
||||
@@ -7936,6 +8123,25 @@ label = "Read a Microsoft Purview label"
|
||||
connection = "Purview tenant"
|
||||
connectionHelp = "Reads the label the document already carries, so later steps can act on it."
|
||||
|
||||
[portal.policies.config.ragIngest.fields]
|
||||
chunkSize = "Chunk size (characters)"
|
||||
chunkSizeHelp = "How much text each searchable passage holds. Larger chunks carry more context; smaller ones match more precisely."
|
||||
exportChunks = "Export chunks (JSONL)"
|
||||
exportChunksHelp = "Adds a .chunks.jsonl file to the step output - one chunk per line with page span - ready for external embedding or indexing."
|
||||
exportMarkdown = "Export markdown"
|
||||
exportMarkdownHelp = "Adds a .md rendering of the parsed document to the step output, for delivery to external systems."
|
||||
index = "Index into the knowledge base"
|
||||
indexHelp = "Embed and store the chunks in the built-in knowledge base. Turn off for export-only ingestion."
|
||||
overlap = "Chunk overlap (characters)"
|
||||
overlapHelp = "How much neighbouring chunks share, so answers spanning a boundary aren't cut in half."
|
||||
|
||||
[portal.policies.config.ragIngest.fields.mode]
|
||||
advanced = "Advanced (layout parsing addon)"
|
||||
auto = "Auto (best available)"
|
||||
basic = "Basic (text layer)"
|
||||
help = "Auto uses the best parser available. The advanced layout tier requires the engine's docparse addon."
|
||||
label = "Parse tier"
|
||||
|
||||
[portal.policies.config.retention]
|
||||
summary = "Enforces how long documents are kept, when to archive, and when to delete."
|
||||
|
||||
@@ -8005,8 +8211,10 @@ addWatermark = "Watermark"
|
||||
autoRedact = "Redact PII"
|
||||
classifyAndLabel = "Classify"
|
||||
compressPdf = "Compress"
|
||||
extractFields = "Extract fields"
|
||||
flatten = "Flatten"
|
||||
ocrPdf = "OCR"
|
||||
ragIngest = "Ingest into knowledge base"
|
||||
sanitizePdf = "Remove JavaScript"
|
||||
|
||||
[portal.policies.offline]
|
||||
@@ -8238,6 +8446,10 @@ label = "Classify the document"
|
||||
desc = "Compresses the document to a smaller file size."
|
||||
label = "Reduce file size"
|
||||
|
||||
[portal.policies.wizard.capability.extractFields]
|
||||
desc = "Pulls the values you describe - like invoice numbers or dates - out of every document, each with a confidence score and a citation back to the page."
|
||||
label = "Extract structured fields"
|
||||
|
||||
[portal.policies.wizard.capability.flatten]
|
||||
desc = "Merges form fields and annotations into the page so they can't be edited."
|
||||
label = "Flatten the document"
|
||||
@@ -8246,6 +8458,10 @@ label = "Flatten the document"
|
||||
desc = "Runs OCR so scanned pages become selectable, searchable text."
|
||||
label = "Make text searchable"
|
||||
|
||||
[portal.policies.wizard.capability.ragIngest]
|
||||
desc = "Parses, chunks, and embeds the document so it becomes searchable knowledge, or exports the parsed content (markdown, chunks JSONL) for your own systems."
|
||||
label = "Index into the knowledge base"
|
||||
|
||||
[portal.policies.wizard.capability.redact]
|
||||
desc = "Finds and blacks out sensitive details — like Social Security and card numbers — so they can't be read."
|
||||
label = "Redact sensitive information"
|
||||
@@ -10248,6 +10464,26 @@ medium = "Medium"
|
||||
small = "Small"
|
||||
x-large = "X-Large"
|
||||
|
||||
[smartSplit]
|
||||
intro = "Describe where sub-documents start in plain language and AI reads the content to find those boundaries - no page numbers needed."
|
||||
submit = "Split document"
|
||||
|
||||
[smartSplit.error]
|
||||
failed = "Failed to split document"
|
||||
|
||||
[smartSplit.maxParts]
|
||||
label = "Maximum parts"
|
||||
|
||||
[smartSplit.results]
|
||||
title = "Split documents"
|
||||
|
||||
[smartSplit.rule]
|
||||
label = "Split rule"
|
||||
placeholder = "e.g. Start a new document at every invoice header"
|
||||
|
||||
[smartSplit.settings]
|
||||
title = "Split settings"
|
||||
|
||||
[split]
|
||||
resultsTitle = "Split Results"
|
||||
selectMethod = "Select a split method"
|
||||
@@ -10714,6 +10950,7 @@ standardTools = "Standard Tools"
|
||||
advancedFormatting = "Advanced Formatting"
|
||||
automation = "Automation"
|
||||
developerTools = "Developer Tools"
|
||||
documentIntelligence = "Document intelligence"
|
||||
documentReview = "Document Review"
|
||||
documentSecurity = "Document Security"
|
||||
extraction = "Extraction"
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user