Compare commits

...
71 changed files with 3496 additions and 42 deletions
@@ -433,11 +433,19 @@ public class EndpointConfiguration {
addEndpointToGroup("Automation", "automate"); // Alias for handleData (user-friendly name)
addEndpointToGroup("Automation", "pipeline");
// Adding endpoints to "DocParse" group (ingestion: chunk + index + export)
addEndpointToGroup("DocParse", "rag-ingest");
addEndpointToGroup("DocParse", "extract-tables");
// Adding endpoints to "DocParse" group (parsing, splitting, chunking, extraction,
// templating)
addEndpointToGroup("DocParse", "parse-document");
addEndpointToGroup("DocParse", "extract-fields");
addEndpointToGroup("DocParse", "smart-split");
addEndpointToGroup("DocParse", "chunk-document");
addEndpointToGroup("DocParse", "rag-ingest");
addEndpointToGroup("DocParse", "rag-documents");
addEndpointToGroup("DocParse", "rag-search");
addEndpointToGroup("DocParse", "rag-ask");
addEndpointToGroup("DocParse", "extract-tables");
addEndpointToGroup("DocParse", "suggest-schema");
addEndpointToGroup("DocParse", "fill-template");
// Adding endpoints to "DeveloperTools" group
addEndpointToGroup("DeveloperTools", "show-javascript");
@@ -4,22 +4,32 @@ import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.io.StringWriter;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.StandardCopyOption;
import java.util.Base64;
import java.util.List;
import java.util.Locale;
import java.util.zip.ZipEntry;
import java.util.zip.ZipOutputStream;
import org.apache.commons.csv.CSVFormat;
import org.apache.commons.csv.CSVPrinter;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.springframework.core.io.ByteArrayResource;
import org.springframework.core.io.Resource;
import org.springframework.http.HttpHeaders;
import org.springframework.http.HttpStatus;
import org.springframework.http.MediaType;
import org.springframework.http.ResponseEntity;
import org.springframework.web.bind.annotation.GetMapping;
import org.springframework.web.bind.annotation.ModelAttribute;
import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RequestMapping;
import org.springframework.web.bind.annotation.RequestParam;
import org.springframework.web.bind.annotation.RestController;
import org.springframework.web.multipart.MultipartFile;
import org.springframework.web.server.ResponseStatusException;
import io.swagger.v3.oas.annotations.Operation;
import io.swagger.v3.oas.annotations.tags.Tag;
@@ -29,19 +39,34 @@ import lombok.extern.slf4j.Slf4j;
import stirling.software.common.annotations.AutoJobPostMapping;
import stirling.software.common.enumeration.ResourceWeight;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.util.FormUtils;
import stirling.software.common.util.GeneralUtils;
import stirling.software.common.util.TempFile;
import stirling.software.common.util.TempFileManager;
import stirling.software.common.util.WebResponseUtils;
import stirling.software.proprietary.model.api.docparse.ChunkDocumentApiRequest;
import stirling.software.proprietary.model.api.docparse.ExtractFieldsApiRequest;
import stirling.software.proprietary.model.api.docparse.ExtractTablesApiRequest;
import stirling.software.proprietary.model.api.docparse.ParseDocumentApiRequest;
import stirling.software.proprietary.model.api.docparse.RagAskApiRequest;
import stirling.software.proprietary.model.api.docparse.RagIngestApiRequest;
import stirling.software.proprietary.model.api.docparse.RagSearchApiRequest;
import stirling.software.proprietary.model.api.docparse.SmartSplitApiRequest;
import stirling.software.proprietary.model.api.docparse.SuggestSchemaApiRequest;
import stirling.software.proprietary.model.docparse.ChunkDocumentResponse;
import stirling.software.proprietary.model.docparse.DocChunk;
import stirling.software.proprietary.model.docparse.DocTable;
import stirling.software.proprietary.model.docparse.DocparseCapabilitiesView;
import stirling.software.proprietary.model.docparse.DocparseMode;
import stirling.software.proprietary.model.docparse.ExtractFieldsResponse;
import stirling.software.proprietary.model.docparse.ExtractTablesResponse;
import stirling.software.proprietary.model.docparse.FillDocxResponse;
import stirling.software.proprietary.model.docparse.ParseDocumentResponse;
import stirling.software.proprietary.model.docparse.RagIngestResponse;
import stirling.software.proprietary.model.docparse.RagStatsView;
import stirling.software.proprietary.model.docparse.SmartSplitResponse;
import stirling.software.proprietary.model.docparse.SplitPart;
import stirling.software.proprietary.model.docparse.SuggestSchemaResponse;
import stirling.software.proprietary.service.AiToolResponseHeaders;
import stirling.software.proprietary.service.DocParseService;
@@ -67,7 +92,14 @@ public class DocParseController {
private static final MediaType CSV = MediaType.parseMediaType("text/csv");
private static final MediaType MARKDOWN = MediaType.parseMediaType("text/markdown");
private static final MediaType DOCX =
MediaType.parseMediaType(
"application/vnd.openxmlformats-officedocument.wordprocessingml.document");
private final DocParseService docParseService;
private final CustomPDFDocumentFactory pdfDocumentFactory;
private final TempFileManager tempFileManager;
private final ObjectMapper objectMapper;
@AutoJobPostMapping(
@@ -192,6 +224,123 @@ public class DocParseController {
docParseService.suggestSchema(request.getFileInput(), request.getMaxFields()));
}
@AutoJobPostMapping(
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
value = "/parse-document",
resourceWeight = ResourceWeight.XLARGE_WEIGHT)
@Operation(
summary = "Parse a document into structured blocks, tables, and markdown",
description =
"Parses the PDF into layout blocks, tables, and a markdown rendering. The"
+ " basic tier reads the text layer; the advanced tier (docparse addon)"
+ " adds OCR, real table structure, and bounding boxes."
+ " Input:PDF Output:JSON Type:SISO")
public ResponseEntity<?> parseDocument(@ModelAttribute ParseDocumentApiRequest request)
throws IOException {
ParseDocumentResponse result =
docParseService.parse(
request.getFileInput(),
DocparseMode.fromWire(request.getMode()),
request.isWithOcr());
if ("markdown".equalsIgnoreCase(request.getOutputFormat())) {
return WebResponseUtils.bytesToWebResponse(
result.markdown().getBytes(StandardCharsets.UTF_8),
outputName(request.getFileInput(), "_parsed.md"),
MARKDOWN);
}
return ResponseEntity.ok(result);
}
@AutoJobPostMapping(
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
value = "/smart-split",
resourceWeight = ResourceWeight.LARGE_WEIGHT)
@Operation(
summary = "Split a document at content-derived boundaries",
description =
"Asks the engine where sub-documents start (per the natural-language rule) and"
+ " returns a ZIP with one PDF per part, named from the part labels."
+ " Input:PDF Output:ZIP-PDF Type:SIMO")
public ResponseEntity<Resource> smartSplit(@ModelAttribute SmartSplitApiRequest request)
throws IOException {
MultipartFile file = request.getFileInput();
SmartSplitResponse split =
docParseService.split(file, request.getRule(), request.getMaxParts());
if (split.parts().isEmpty()) {
throw new ResponseStatusException(
HttpStatus.UNPROCESSABLE_ENTITY,
"The split rule produced no parts for this document");
}
TempFile zipTempFile = tempFileManager.createManagedTempFile(".zip");
try {
try (TempFile sourceTempFile = new TempFile(tempFileManager, ".pdf")) {
Files.copy(
file.getInputStream(),
sourceTempFile.getPath(),
StandardCopyOption.REPLACE_EXISTING);
try (ZipOutputStream zipOut =
new ZipOutputStream(Files.newOutputStream(zipTempFile.getPath()))) {
writeParts(sourceTempFile, split.parts(), zipOut);
}
}
return WebResponseUtils.zipFileToWebResponse(
zipTempFile,
GeneralUtils.generateFilename(file.getOriginalFilename(), "_split.zip"));
} catch (Exception e) {
zipTempFile.close();
throw e;
}
}
@AutoJobPostMapping(
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
value = "/chunk-document",
resourceWeight = ResourceWeight.MEDIUM_WEIGHT)
@Operation(
summary = "Chunk a document for RAG",
description =
"Splits the document text into overlapping chunks with page spans and (advanced"
+ " tier) heading breadcrumbs. Input:PDF Output:JSON Type:SISO")
public ResponseEntity<ChunkDocumentResponse> chunkDocument(
@ModelAttribute ChunkDocumentApiRequest request) throws IOException {
return ResponseEntity.ok(
docParseService.chunk(
request.getFileInput(),
request.getChunkSize(),
request.getOverlap(),
DocparseMode.fromWire(request.getMode())));
}
@AutoJobPostMapping(
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
value = "/fill-template",
resourceWeight = ResourceWeight.SMALL_WEIGHT)
@Operation(
summary = "Fill a DOCX template with JSON data",
description =
"Replaces the template's placeholders with values from the JSON object and"
+ " returns the filled DOCX. Replacement counts and missing keys ride"
+ " the X-Stirling-Tool-Report header."
+ " Input:DOCX Output:DOCX Type:SISO")
public ResponseEntity<Resource> fillTemplate(
@RequestParam("templateFile") MultipartFile templateFile,
@RequestParam("data") String data)
throws IOException {
FillDocxResponse result = docParseService.fillDocx(templateFile, data);
byte[] filled = Base64.getDecoder().decode(result.docxBase64());
HttpHeaders headers = new HttpHeaders();
headers.setContentType(DOCX);
headers.setContentDispositionFormData(
"attachment",
GeneralUtils.generateFilename(templateFile.getOriginalFilename(), "_filled.docx"));
headers.setContentLength(filled.length);
headers.set(
AiToolResponseHeaders.TOOL_REPORT,
objectMapper.writeValueAsString(
new FillDocxResponse("", result.replaced(), result.missing())));
return ResponseEntity.ok().headers(headers).body(new ByteArrayResource(filled));
}
@GetMapping("/capabilities")
@Operation(
summary = "DocParse capability summary",
@@ -224,6 +373,52 @@ public class DocParseController {
CSV);
}
@GetMapping("/rag-stats")
@Operation(
summary = "RAG store statistics",
description =
"The engine's document-store totals (backend, documents, chunks, embedding"
+ " model) merged with the DocParse capability fields. Answers with"
+ " zeros and engineReachable=false when the engine is down.")
public ResponseEntity<RagStatsView> ragStats() {
return ResponseEntity.ok(docParseService.ragStats());
}
@GetMapping("/rag-documents")
@Operation(
summary = "List documents in the RAG store",
description =
"Engine passthrough of the caller-visible indexed documents (documentId,"
+ " source, chunk count).")
public ResponseEntity<String> ragDocuments() throws IOException {
return jsonPassthrough(docParseService.ragDocuments());
}
@PostMapping(value = "/rag-search", consumes = MediaType.APPLICATION_JSON_VALUE)
@Operation(
summary = "Semantic search over the RAG store",
description =
"Searches the caller-visible indexed documents and returns the top passages"
+ " with scores, page spans, and heading breadcrumbs.")
public ResponseEntity<String> ragSearch(@RequestBody RagSearchApiRequest request)
throws IOException {
return jsonPassthrough(docParseService.ragSearch(request.getQuery(), request.getTopK()));
}
@PostMapping(value = "/rag-ask", consumes = MediaType.APPLICATION_JSON_VALUE)
@Operation(
summary = "Ask a question over the RAG store",
description =
"Answers the question from the caller-visible indexed documents and returns"
+ " the answer with its supporting passages.")
public ResponseEntity<String> ragAsk(@RequestBody RagAskApiRequest request) throws IOException {
return jsonPassthrough(docParseService.ragAsk(request.getQuestion(), request.getTopK()));
}
private static ResponseEntity<String> jsonPassthrough(String engineJson) {
return ResponseEntity.ok().contentType(MediaType.APPLICATION_JSON).body(engineJson);
}
/** Original + requested corpus files in one ZIP, so destinations receive them together. */
private byte[] exportZip(
String fileName, byte[] original, RagIngestResponse result, RagIngestApiRequest request)
@@ -279,6 +474,40 @@ public class DocParseController {
return dot > 0 ? fileName.substring(0, dot) : fileName;
}
private void writeParts(TempFile sourceTempFile, List<SplitPart> parts, ZipOutputStream zipOut)
throws IOException {
for (int i = 0; i < parts.size(); i++) {
SplitPart part = parts.get(i);
// Load per part and remove pages outside the range: avoids the PDFBox cross-document
// addPage pitfalls while keeping shared resources intact.
try (PDDocument partDoc = pdfDocumentFactory.load(sourceTempFile.getFile())) {
int pageCount = partDoc.getNumberOfPages();
int start = Math.clamp(part.startPage(), 1, pageCount);
int end = Math.clamp(part.endPage(), start, pageCount);
for (int p = pageCount - 1; p >= 0; p--) {
int pageNumber = p + 1;
if (pageNumber < start || pageNumber > end) {
partDoc.removePage(p);
}
}
FormUtils.pruneOrphanedFormFields(partDoc);
zipOut.putNextEntry(new ZipEntry(partEntryName(i, part)));
partDoc.save(zipOut);
zipOut.closeEntry();
}
}
}
private static String partEntryName(int index, SplitPart part) {
String label = part.label() == null ? "" : part.label().trim();
String sanitized = label.replaceAll("[^A-Za-z0-9 ._-]", "_").replaceAll("\\s+", "_");
if (sanitized.isBlank() || sanitized.chars().allMatch(c -> c == '_' || c == '.')) {
sanitized = "part";
}
// Index prefix keeps entries unique even when labels repeat.
return String.format(Locale.ROOT, "%02d_%s.pdf", index + 1, sanitized);
}
private static String tablesToCsv(List<DocTable> tables) throws IOException {
CSVFormat format = CSVFormat.EXCEL.builder().setEscape('"').build();
StringWriter writer = new StringWriter();
@@ -0,0 +1,27 @@
package stirling.software.proprietary.model.api.docparse;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Data;
import lombok.EqualsAndHashCode;
import stirling.software.common.model.api.PDFFile;
@Data
@EqualsAndHashCode(callSuper = true)
public class ChunkDocumentApiRequest extends PDFFile {
@Schema(description = "Target chunk size in characters (64-32768)", defaultValue = "512")
private int chunkSize = 512;
@Schema(
description = "Overlap between adjacent chunks in characters (0-4096)",
defaultValue = "64")
private int overlap = 64;
@Schema(
description = "Tier to use: 'auto' picks per document, or force 'basic'/'advanced'",
allowableValues = {"auto", "basic", "advanced"},
defaultValue = "auto")
private String mode = "auto";
}
@@ -0,0 +1,30 @@
package stirling.software.proprietary.model.api.docparse;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Data;
import lombok.EqualsAndHashCode;
import stirling.software.common.model.api.PDFFile;
@Data
@EqualsAndHashCode(callSuper = true)
public class ParseDocumentApiRequest extends PDFFile {
@Schema(
description = "Tier to use: 'auto' picks per document, or force 'basic'/'advanced'",
allowableValues = {"auto", "basic", "advanced"},
defaultValue = "auto")
private String mode = "auto";
@Schema(
description = "Apply OCR when parsing scanned pages (advanced tier only)",
defaultValue = "true")
private boolean withOcr = true;
@Schema(
description = "Response format: full JSON result or the markdown rendering only",
allowableValues = {"json", "markdown"},
defaultValue = "json")
private String outputFormat = "json";
}
@@ -0,0 +1,17 @@
package stirling.software.proprietary.model.api.docparse;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Data;
@Data
public class RagAskApiRequest {
@Schema(
description = "Question to answer from the indexed documents",
requiredMode = Schema.RequiredMode.REQUIRED)
private String question;
@Schema(description = "Number of passages to ground the answer on (1-20)", defaultValue = "5")
private int topK = 5;
}
@@ -0,0 +1,17 @@
package stirling.software.proprietary.model.api.docparse;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Data;
@Data
public class RagSearchApiRequest {
@Schema(
description = "Natural-language search query",
requiredMode = Schema.RequiredMode.REQUIRED)
private String query;
@Schema(description = "Number of passages to return (1-50)", defaultValue = "10")
private int topK = 10;
}
@@ -0,0 +1,21 @@
package stirling.software.proprietary.model.api.docparse;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Data;
import lombok.EqualsAndHashCode;
import stirling.software.common.model.api.PDFFile;
@Data
@EqualsAndHashCode(callSuper = true)
public class SmartSplitApiRequest extends PDFFile {
@Schema(
description = "Natural-language boundary rule, e.g. 'split where a new invoice starts'",
requiredMode = Schema.RequiredMode.REQUIRED)
private String rule;
@Schema(description = "Maximum number of parts to produce (1-500)", defaultValue = "50")
private int maxParts = 50;
}
@@ -0,0 +1,14 @@
package stirling.software.proprietary.model.docparse;
import java.util.List;
import stirling.software.proprietary.model.api.ai.AiPageText;
/** Engine request for {@code POST /api/v1/docparse/chunk}. */
public record ChunkDocumentRequest(
String fileName,
List<AiPageText> pages,
String contentBase64,
int chunkSize,
int overlap,
DocparseMode mode) {}
@@ -0,0 +1,11 @@
package stirling.software.proprietary.model.docparse;
import java.util.List;
/** Engine response for {@code POST /api/v1/docparse/chunk}. */
public record ChunkDocumentResponse(DocparseTier mode, List<DocChunk> chunks) {
public ChunkDocumentResponse {
chunks = chunks == null ? List.of() : chunks;
}
}
@@ -0,0 +1,9 @@
package stirling.software.proprietary.model.docparse;
import java.util.List;
/**
* One layout block. {@code bbox} is [x0, y0, x1, y1] normalized to 0..1 with a top-left origin;
* {@code null} in basic tier (no layout model ran). Mirrors {@code docparse.py DocBlock}.
*/
public record DocBlock(String type, String text, int page, List<Double> bbox, Double confidence) {}
@@ -0,0 +1,5 @@
package stirling.software.proprietary.model.docparse;
/** Engine response for {@code GET /api/v1/documents/stats}: the RAG document store totals. */
public record DocumentStoreStats(
String backend, long documents, long chunks, String embeddingModel) {}
@@ -0,0 +1,6 @@
package stirling.software.proprietary.model.docparse;
import tools.jackson.databind.JsonNode;
/** Engine request for {@code POST /api/v1/docparse/fill-docx}. */
public record FillDocxRequest(String templateBase64, JsonNode data) {}
@@ -0,0 +1,11 @@
package stirling.software.proprietary.model.docparse;
import java.util.List;
/** Engine response for {@code POST /api/v1/docparse/fill-docx}. */
public record FillDocxResponse(String docxBase64, int replaced, List<String> missing) {
public FillDocxResponse {
missing = missing == null ? List.of() : missing;
}
}
@@ -0,0 +1,4 @@
package stirling.software.proprietary.model.docparse;
/** Engine request for {@code POST /api/v1/docparse/parse}. */
public record ParseDocumentRequest(String fileName, String contentBase64, boolean withOcr) {}
@@ -0,0 +1,21 @@
package stirling.software.proprietary.model.docparse;
import java.util.List;
/**
* Engine response for {@code POST /api/v1/docparse/parse}; also produced by the Java basic tier.
*/
public record ParseDocumentResponse(
DocparseTier mode,
int pages,
List<DocBlock> blocks,
List<DocTable> tables,
String markdown,
boolean ocrApplied) {
public ParseDocumentResponse {
blocks = blocks == null ? List.of() : blocks;
tables = tables == null ? List.of() : tables;
markdown = markdown == null ? "" : markdown;
}
}
@@ -0,0 +1,4 @@
package stirling.software.proprietary.model.docparse;
/** Engine request for {@code POST /api/v1/documents/ask}: grounded Q&A over the RAG store. */
public record RagAskRequest(String question, int topK) {}
@@ -0,0 +1,4 @@
package stirling.software.proprietary.model.docparse;
/** Engine request for {@code POST /api/v1/documents/search}: semantic search over the RAG store. */
public record RagSearchRequest(String query, int topK) {}
@@ -0,0 +1,15 @@
package stirling.software.proprietary.model.docparse;
/**
* Merged RAG store view served by {@code GET /api/v1/docparse/rag-stats} (Java side): the engine's
* document-store totals plus the cached DocParse capability fields. When the engine is unreachable
* the totals are zero and {@code engineReachable} is false.
*/
public record RagStatsView(
String backend,
long documents,
long chunks,
String embeddingModel,
boolean advancedInstalled,
String doclingVersion,
boolean engineReachable) {}
@@ -0,0 +1,9 @@
package stirling.software.proprietary.model.docparse;
import java.util.List;
import stirling.software.proprietary.model.api.ai.AiPageText;
/** Engine request for {@code POST /api/v1/docparse/split}. */
public record SmartSplitRequest(
String fileName, String rule, List<AiPageText> pages, int maxParts) {}
@@ -0,0 +1,11 @@
package stirling.software.proprietary.model.docparse;
import java.util.List;
/** Engine response for {@code POST /api/v1/docparse/split}. */
public record SmartSplitResponse(List<SplitPart> parts) {
public SmartSplitResponse {
parts = parts == null ? List.of() : parts;
}
}
@@ -0,0 +1,4 @@
package stirling.software.proprietary.model.docparse;
/** One sub-document page range (1-based, inclusive). Mirrors {@code docparse.py SplitPart}. */
public record SplitPart(int startPage, int endPage, String label, double confidence) {}
@@ -20,16 +20,29 @@ import stirling.software.common.model.ApplicationProperties;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.service.UserServiceInterface;
import stirling.software.proprietary.model.api.ai.AiPageText;
import stirling.software.proprietary.model.docparse.ChunkDocumentRequest;
import stirling.software.proprietary.model.docparse.ChunkDocumentResponse;
import stirling.software.proprietary.model.docparse.DocBlock;
import stirling.software.proprietary.model.docparse.DocparseCapabilities;
import stirling.software.proprietary.model.docparse.DocparseCapabilitiesView;
import stirling.software.proprietary.model.docparse.DocparseMode;
import stirling.software.proprietary.model.docparse.DocparseTier;
import stirling.software.proprietary.model.docparse.DocumentStoreStats;
import stirling.software.proprietary.model.docparse.ExtractFieldsRequest;
import stirling.software.proprietary.model.docparse.ExtractFieldsResponse;
import stirling.software.proprietary.model.docparse.ExtractTablesRequest;
import stirling.software.proprietary.model.docparse.ExtractTablesResponse;
import stirling.software.proprietary.model.docparse.FillDocxRequest;
import stirling.software.proprietary.model.docparse.FillDocxResponse;
import stirling.software.proprietary.model.docparse.ParseDocumentRequest;
import stirling.software.proprietary.model.docparse.ParseDocumentResponse;
import stirling.software.proprietary.model.docparse.RagAskRequest;
import stirling.software.proprietary.model.docparse.RagIngestRequest;
import stirling.software.proprietary.model.docparse.RagIngestResponse;
import stirling.software.proprietary.model.docparse.RagSearchRequest;
import stirling.software.proprietary.model.docparse.RagStatsView;
import stirling.software.proprietary.model.docparse.SmartSplitRequest;
import stirling.software.proprietary.model.docparse.SmartSplitResponse;
import stirling.software.proprietary.model.docparse.SuggestSchemaRequest;
import stirling.software.proprietary.model.docparse.SuggestSchemaResponse;
@@ -46,10 +59,18 @@ import tools.jackson.databind.ObjectMapper;
@Service
public class DocParseService {
private static final String RAG_INGEST_ENDPOINT = "/api/v1/docparse/rag-ingest";
private static final String TABLES_ENDPOINT = "/api/v1/docparse/tables";
private static final String PARSE_ENDPOINT = "/api/v1/docparse/parse";
private static final String EXTRACT_ENDPOINT = "/api/v1/docparse/extract";
private static final String SPLIT_ENDPOINT = "/api/v1/docparse/split";
private static final String CHUNK_ENDPOINT = "/api/v1/docparse/chunk";
private static final String TABLES_ENDPOINT = "/api/v1/docparse/tables";
private static final String FILL_DOCX_ENDPOINT = "/api/v1/docparse/fill-docx";
private static final String SUGGEST_SCHEMA_ENDPOINT = "/api/v1/docparse/suggest-schema";
private static final String RAG_INGEST_ENDPOINT = "/api/v1/docparse/rag-ingest";
private static final String DOCUMENT_STATS_ENDPOINT = "/api/v1/documents/stats";
private static final String DOCUMENT_LIST_ENDPOINT = "/api/v1/documents/list";
private static final String DOCUMENT_SEARCH_ENDPOINT = "/api/v1/documents/search";
private static final String DOCUMENT_ASK_ENDPOINT = "/api/v1/documents/ask";
/** Below this average of extractable chars per page the document is treated as scanned. */
static final int SCANNED_AVG_CHARS_PER_PAGE = 100;
@@ -104,6 +125,72 @@ public class DocParseService {
capabilities.doclingVersion());
}
public ParseDocumentResponse parse(
MultipartFile file, DocparseMode requestedMode, boolean withOcr) throws IOException {
requireEnabled();
DocparseTier tier;
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
tier =
resolveTier(
requestedMode,
capabilityService.capabilities(),
false,
looksScanned(document));
if (tier == DocparseTier.BASIC) {
return basicParse(document);
}
}
ParseDocumentRequest request =
new ParseDocumentRequest(fileName(file), encodeBase64(file), withOcr);
String responseJson =
aiEngineClient.postLongRunning(
PARSE_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
return objectMapper.readValue(responseJson, ParseDocumentResponse.class);
}
public SmartSplitResponse split(MultipartFile file, String rule, int maxParts)
throws IOException {
requireEnabled();
if (rule == null || rule.isBlank()) {
throw new ResponseStatusException(HttpStatus.BAD_REQUEST, "A split rule is required");
}
List<AiPageText> pages;
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
pages = extractPages(document);
}
SmartSplitRequest request =
new SmartSplitRequest(fileName(file), rule, pages, Math.clamp(maxParts, 1, 500));
String responseJson =
aiEngineClient.post(
SPLIT_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
return objectMapper.readValue(responseJson, SmartSplitResponse.class);
}
public ChunkDocumentResponse chunk(
MultipartFile file, int chunkSize, int overlap, DocparseMode mode) throws IOException {
requireEnabled();
List<AiPageText> pages;
DocparseTier tier;
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
pages = extractPages(document);
tier =
resolveTier(
mode, capabilityService.capabilities(), false, looksScanned(document));
}
ChunkDocumentRequest request =
new ChunkDocumentRequest(
fileName(file),
pages,
tier == DocparseTier.ADVANCED ? encodeBase64(file) : null,
Math.clamp(chunkSize, 64, 32_768),
Math.clamp(overlap, 0, 4_096),
toMode(tier));
String responseJson =
aiEngineClient.postLongRunning(
CHUNK_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
return objectMapper.readValue(responseJson, ChunkDocumentResponse.class);
}
/**
* Chunk, embed, and index the document into the engine's RAG store, and/or echo the parsed
* content back for corpus export. Text extraction and tier routing happen here; the engine
@@ -267,6 +354,68 @@ public class DocParseService {
return node;
}
private static void requireNonBlank(String value, String fieldName) {
if (value == null || value.isBlank()) {
throw new ResponseStatusException(
HttpStatus.BAD_REQUEST, "'" + fieldName + "' is required");
}
}
/** Engine RAG store totals merged with the cached capability fields; graceful when down. */
public RagStatsView ragStats() {
DocparseCapabilities capabilities = capabilityService.capabilities();
try {
// The engine's documents routes are user-gated; an id-less probe 401s
// and would read as "engine offline" in the UI.
String json = aiEngineClient.get(DOCUMENT_STATS_ENDPOINT, currentUserId());
DocumentStoreStats stats = objectMapper.readValue(json, DocumentStoreStats.class);
return new RagStatsView(
stats.backend(),
stats.documents(),
stats.chunks(),
stats.embeddingModel(),
capabilities.advancedInstalled(),
capabilities.doclingVersion(),
true);
} catch (Exception e) {
log.debug("RAG stats probe failed: {}", e.getMessage());
return new RagStatsView(
null,
0,
0,
null,
capabilities.advancedInstalled(),
capabilities.doclingVersion(),
false);
}
}
/** Engine document-list passthrough; X-User-Id scopes it to the caller's ACLs. */
public String ragDocuments() throws IOException {
requireEnabled();
return aiEngineClient.get(DOCUMENT_LIST_ENDPOINT, currentUserId());
}
/** Semantic-search passthrough over the caller-visible RAG documents. */
public String ragSearch(String query, int topK) throws IOException {
requireEnabled();
requireNonBlank(query, "query");
RagSearchRequest request = new RagSearchRequest(query, Math.clamp(topK, 1, 50));
return aiEngineClient.post(
DOCUMENT_SEARCH_ENDPOINT,
objectMapper.writeValueAsString(request),
currentUserId());
}
/** Grounded-answer passthrough; long-running because local models answer slowly. */
public String ragAsk(String question, int topK) throws IOException {
requireEnabled();
requireNonBlank(question, "question");
RagAskRequest request = new RagAskRequest(question, Math.clamp(topK, 1, 20));
return aiEngineClient.postLongRunning(
DOCUMENT_ASK_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
}
/**
* The settings mode wins when stricter: a settings {@code basic} always forces basic, a
* settings {@code advanced} upgrades everything except an explicit basic request.
@@ -327,6 +476,41 @@ public class DocParseService {
return objectMapper.readValue(responseJson, ExtractTablesResponse.class);
}
public FillDocxResponse fillDocx(MultipartFile templateFile, String dataJson)
throws IOException {
requireEnabled();
JsonNode data = parseJsonObject(dataJson, "data");
FillDocxRequest request =
new FillDocxRequest(
Base64.getEncoder().encodeToString(templateFile.getBytes()), data);
String responseJson =
aiEngineClient.post(
FILL_DOCX_ENDPOINT,
objectMapper.writeValueAsString(request),
currentUserId());
return objectMapper.readValue(responseJson, FillDocxResponse.class);
}
/** Basic tier parse: PDFBox text layer only, one paragraph block per non-blank page. */
ParseDocumentResponse basicParse(PDDocument document) throws IOException {
int pageCount = document.getNumberOfPages();
List<DocBlock> blocks = new ArrayList<>();
StringBuilder markdown = new StringBuilder();
for (int page = 1; page <= pageCount; page++) {
String text = pdfContentExtractor.extractPageTextRaw(document, page);
if (text == null || text.isBlank()) {
continue;
}
blocks.add(new DocBlock("paragraph", text, page, null, null));
if (!markdown.isEmpty()) {
markdown.append("\n\n");
}
markdown.append(text);
}
return new ParseDocumentResponse(
DocparseTier.BASIC, pageCount, blocks, List.of(), markdown.toString(), false);
}
/** Extract per-page text for the engine, capped by the shared aiEngine limits. */
List<AiPageText> extractPages(PDDocument document) throws IOException {
ApplicationProperties.AiEngine.Limits limits =
+2
View File
@@ -14,6 +14,8 @@ dependencies = [
"pydantic-ai-slim[voyageai]>=1.99.0,<2.0.0",
"pydantic-settings>=2.0.0",
"python-dotenv>=1.2.1",
# Small (MIT) and always installed: DOCX template filling needs no addon.
"python-docx>=1.1.2",
"sqlite-vec>=0.1.6",
"uvicorn>=0.35.0",
"opentelemetry-sdk>=1.39.0",
+2
View File
@@ -2,6 +2,7 @@
from .document_classifier import DocumentClassifierAgent
from .execution import ExecutionPlanningAgent
from .knowledge_ask import KnowledgeAskAgent
from .orchestrator import OrchestratorAgent
from .pdf_create import PdfCreateAgent
from .pdf_edit import PdfEditAgent, PdfEditParameterSelector, PdfEditPlanSelection
@@ -12,6 +13,7 @@ from .user_spec import UserSpecAgent
__all__ = [
"DocumentClassifierAgent",
"ExecutionPlanningAgent",
"KnowledgeAskAgent",
"OrchestratorAgent",
"PdfCreateAgent",
"PdfEditAgent",
+134
View File
@@ -0,0 +1,134 @@
"""Grounded Q&A over the caller's stored documents.
Retrieval runs the same cross-collection search as ``POST /documents/search``;
one smart-model pass then answers only from the retrieved passages, citing
document and page inline. No retrieval hit means a plain "not found" answer.
"""
from __future__ import annotations
import logging
from pydantic import Field
from pydantic_ai import Agent
from stirling.agents.output_mode import output_retries, structured_output
from stirling.contracts import AskDocumentsRequest, AskDocumentsResponse, DocumentPassage
from stirling.documents import CollectionSearchHit
from stirling.documents.service import PAGE_NUMBER_METADATA_KEY
from stirling.models import ApiModel, PrincipalId
from stirling.services import AppRuntime
logger = logging.getLogger(__name__)
# Metadata keys written by docparse rag-ingest (_chunk_metadata) for structure-aware chunks.
_PAGE_START_KEY = "page_start"
_PAGE_END_KEY = "page_end"
_HEADING_PATH_KEY = "heading_path"
_HEADING_PATH_SEPARATOR = " > "
_SYSTEM_PROMPT = (
"You answer questions using ONLY the numbered passages you are given.\n"
"\n"
"Rules:\n"
"- Every statement must come from the passages. Never use outside knowledge, never guess.\n"
"- Cite the document and page inline right after each fact, "
'e.g. "(invoice.pdf p.2)" or "(report.pdf p.4-6)", using the names and pages '
"shown in each passage header.\n"
"- If the passages do not answer the question, say plainly that the stored "
"documents do not cover it. Do not attempt a partial guess.\n"
"- Answer in the same language as the question."
)
_NO_PASSAGES_ANSWER = "I couldn't find anything relevant to that question in your stored documents."
class _AskOutput(ApiModel):
"""Raw model answer for the single ask pass."""
answer: str = Field(description="The answer grounded in the passages, with inline citations.")
def _meta_int(value: str | None) -> int | None:
if value is None:
return None
try:
return int(value)
except ValueError:
return None
def passage_from_hit(hit: CollectionSearchHit) -> DocumentPassage:
"""Map a store search hit onto the wire passage shape.
Docparse chunks carry page bounds and a heading path; plain page-text
chunks only carry ``page_number``, which maps to both bounds.
"""
meta = hit.result.document.metadata
page_start = _meta_int(meta.get(_PAGE_START_KEY))
page_end = _meta_int(meta.get(_PAGE_END_KEY))
if page_start is None and page_end is None:
page_start = page_end = _meta_int(meta.get(PAGE_NUMBER_METADATA_KEY))
heading = meta.get(_HEADING_PATH_KEY)
source = meta.get("source")
if source and ":page:" in source:
# Page-text chunk sources look like "report.pdf:page:3"; show the file name.
source = source.rsplit(":page:", 1)[0]
return DocumentPassage(
document_id=hit.collection,
text=hit.result.document.text,
score=hit.result.score,
page_start=page_start,
page_end=page_end,
heading_path=heading.split(_HEADING_PATH_SEPARATOR) if heading else [],
source=source or None,
)
def format_passages(passages: list[DocumentPassage]) -> str:
"""Render passages for the prompt with the citation handle in each header."""
return "\n\n".join(_format_passage(i, passage) for i, passage in enumerate(passages, 1))
def _format_passage(index: int, passage: DocumentPassage) -> str:
name = passage.source or passage.document_id
if passage.page_start is None:
pages = ""
elif passage.page_end is not None and passage.page_end != passage.page_start:
pages = f" p.{passage.page_start}-{passage.page_end}"
else:
pages = f" p.{passage.page_start}"
return f"[Passage {index} | {name}{pages}]\n{passage.text}"
class KnowledgeAskAgent:
"""Answers a question from the caller's stored documents.
Retrieves the top passages the caller can read (same path as the search
endpoint), then runs one smart-model pass over just those passages.
"""
def __init__(self, runtime: AppRuntime) -> None:
self.runtime = runtime
# Ollama/custom block tool-calling under native json-schema output; see agents.output_mode.
provider = runtime.settings.chat_provider
self._agent: Agent[None, _AskOutput] = Agent(
model=runtime.smart_model,
output_type=structured_output([_AskOutput], chat_provider=provider),
system_prompt=_SYSTEM_PROMPT,
model_settings=runtime.smart_model_settings,
retries=output_retries(provider),
)
async def ask(self, request: AskDocumentsRequest, principals: list[PrincipalId]) -> AskDocumentsResponse:
hits = await self.runtime.documents.search_with_collections(
request.question, principals=principals, top_k=request.top_k
)
passages = [passage_from_hit(hit) for hit in hits]
if not passages:
logger.info("[knowledge-ask] question=%r -> 0 passages", request.question)
return AskDocumentsResponse(answer=_NO_PASSAGES_ANSWER, passages=[])
prompt = f"Question: {request.question}\n\nPassages:\n{format_passages(passages)}"
logger.debug("[knowledge-ask] prompt:\n%s", prompt)
result = await self._agent.run(prompt)
return AskDocumentsResponse(answer=result.output.answer, passages=passages)
+6 -1
View File
@@ -10,6 +10,7 @@ from pydantic_ai.models import Model
from stirling.agents import (
DocumentClassifierAgent,
ExecutionPlanningAgent,
KnowledgeAskAgent,
OrchestratorAgent,
PdfEditAgent,
PdfQuestionAgent,
@@ -18,7 +19,7 @@ from stirling.agents import (
from stirling.agents.ledger import MathAuditorAgent
from stirling.agents.pdf_comment import PdfCommentAgent
from stirling.config import AppSettings
from stirling.docparse import ExtractFieldsAgent, SuggestSchemaAgent
from stirling.docparse import ExtractFieldsAgent, SmartSplitAgent, SuggestSchemaAgent
from stirling.documents import DocumentService, EmbeddingService
from stirling.services import AppRuntime, build_runtime
@@ -36,7 +37,9 @@ class AppState:
math_auditor_agent: MathAuditorAgent
pdf_comment_agent: PdfCommentAgent
document_classifier_agent: DocumentClassifierAgent
knowledge_ask_agent: KnowledgeAskAgent
extract_fields_agent: ExtractFieldsAgent
smart_split_agent: SmartSplitAgent
suggest_schema_agent: SuggestSchemaAgent
@@ -66,7 +69,9 @@ def build_app_state(
math_auditor_agent=MathAuditorAgent(runtime),
pdf_comment_agent=PdfCommentAgent(runtime),
document_classifier_agent=DocumentClassifierAgent(runtime),
knowledge_ask_agent=KnowledgeAskAgent(runtime),
extract_fields_agent=ExtractFieldsAgent(runtime),
smart_split_agent=SmartSplitAgent(runtime),
suggest_schema_agent=SuggestSchemaAgent(runtime),
)
+10 -1
View File
@@ -7,6 +7,7 @@ from fastapi import Depends, HTTPException, Request, status
from stirling.agents import (
DocumentClassifierAgent,
ExecutionPlanningAgent,
KnowledgeAskAgent,
OrchestratorAgent,
PdfEditAgent,
PdfQuestionAgent,
@@ -15,7 +16,7 @@ from stirling.agents import (
from stirling.agents.ledger import MathAuditorAgent
from stirling.agents.pdf_comment import PdfCommentAgent
from stirling.config import AppSettings, load_settings
from stirling.docparse import ExtractFieldsAgent, SuggestSchemaAgent
from stirling.docparse import ExtractFieldsAgent, SmartSplitAgent, SuggestSchemaAgent
from stirling.documents import DocumentService
from stirling.models import UserId
from stirling.services import AppRuntime, current_user_id
@@ -61,10 +62,18 @@ def get_document_classifier_agent(request: Request) -> DocumentClassifierAgent:
return request.app.state.document_classifier_agent
def get_knowledge_ask_agent(request: Request) -> KnowledgeAskAgent:
return request.app.state.knowledge_ask_agent
def get_extract_fields_agent(request: Request) -> ExtractFieldsAgent:
return request.app.state.extract_fields_agent
def get_smart_split_agent(request: Request) -> SmartSplitAgent:
return request.app.state.smart_split_agent
def get_suggest_schema_agent(request: Request) -> SuggestSchemaAgent:
return request.app.state.suggest_schema_agent
+57 -6
View File
@@ -1,9 +1,9 @@
"""DocParse routes: parse, tables, rag-ingest, capabilities.
"""DocParse routes: parse, extract, split, chunk, tables, fill, capabilities.
Tier routing: requests carrying raw file bytes can use the advanced (Docling)
path when the addon is installed; text-only requests run the basic path.
Forcing ``advanced`` without the addon returns 501 with a machine-readable
``addonRequired`` detail that Java maps onto its own error.
Tier routing happens here: requests carrying raw file bytes can use the
advanced (Docling) path when the addon is installed; text-only requests run
the basic path. Forcing ``advanced`` without the addon returns 501 with a
machine-readable ``addonRequired`` detail that Java maps onto its own error.
"""
from __future__ import annotations
@@ -19,11 +19,14 @@ from fastapi import APIRouter, Depends, HTTPException, status
from stirling.api.dependencies import (
get_document_service,
get_extract_fields_agent,
get_smart_split_agent,
get_suggest_schema_agent,
require_user_id,
)
from stirling.config import AppSettings, load_settings
from stirling.contracts.docparse import (
ChunkDocumentRequest,
ChunkDocumentResponse,
DocChunk,
DocparseCapabilities,
DocparseMode,
@@ -32,17 +35,22 @@ from stirling.contracts.docparse import (
ExtractFieldsResponse,
ExtractTablesRequest,
ExtractTablesResponse,
FillDocxRequest,
FillDocxResponse,
ParseDocumentRequest,
ParseDocumentResponse,
RagIngestRequest,
RagIngestResponse,
SmartSplitRequest,
SmartSplitResponse,
SuggestSchemaRequest,
SuggestSchemaResponse,
)
from stirling.docparse import basic_chunks, probe_capabilities
from stirling.docparse import basic_chunks, fill_docx, probe_capabilities
from stirling.docparse.capability import models_dir
from stirling.docparse.chunking import advanced_chunks
from stirling.docparse.extractor import ExtractFieldsAgent, SchemaError, pages_from_parse
from stirling.docparse.splitter import SmartSplitAgent
from stirling.docparse.suggest_schema import SuggestSchemaAgent
from stirling.documents import DocumentService
from stirling.documents.service import CONTENT_TYPE_METADATA_KEY, DOCPARSE_CHUNK_CONTENT_TYPE
@@ -170,6 +178,39 @@ async def suggest_schema(
return await agent.suggest(request, pages, tier)
@router.post("/split", response_model=SmartSplitResponse)
async def smart_split(
request: SmartSplitRequest,
agent: Annotated[SmartSplitAgent, Depends(get_smart_split_agent)],
) -> SmartSplitResponse:
return await agent.split(request)
@router.post("/chunk", response_model=ChunkDocumentResponse)
async def chunk_document(request: ChunkDocumentRequest) -> ChunkDocumentResponse:
settings = _settings()
caps = _capabilities(settings)
use_advanced = request.mode is DocparseMode.ADVANCED or (
request.mode is DocparseMode.AUTO and caps.advanced_installed and request.content_base64 is not None
)
if use_advanced:
artifacts = _require_advanced(settings)
if request.content_base64 is None:
raise HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
detail="advanced chunking needs contentBase64 (the raw file)",
)
parse = await _parse_advanced(request.content_base64, request.file_name, with_ocr=True, artifacts=artifacts)
return advanced_chunks(parse, request.chunk_size, request.overlap)
if not request.pages:
raise HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
detail="send pages (extracted text) or contentBase64 with the addon installed",
)
return basic_chunks(request.pages, request.chunk_size, request.overlap)
def _chunk_metadata(chunk: DocChunk) -> dict[str, str]:
meta = {CONTENT_TYPE_METADATA_KEY: DOCPARSE_CHUNK_CONTENT_TYPE}
if chunk.page_start is not None:
@@ -261,3 +302,13 @@ async def extract_tables(request: ExtractTablesRequest) -> ExtractTablesResponse
artifacts = _require_advanced(settings)
parse = await _parse_advanced(request.content_base64, request.file_name, with_ocr=True, artifacts=artifacts)
return ExtractTablesResponse(mode=parse.mode, tables=parse.tables)
@router.post("/fill-docx", response_model=FillDocxResponse)
async def fill_docx_template(request: FillDocxRequest) -> FillDocxResponse:
try:
return await anyio.to_thread.run_sync(lambda: fill_docx(request))
except (KeyError, ValueError) as error:
raise HTTPException(
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY, detail=f"invalid docx template: {error}"
) from error
+77 -2
View File
@@ -5,13 +5,24 @@ from typing import Annotated
from fastapi import APIRouter, Depends
from stirling.api.dependencies import get_document_service, require_user_id
from stirling.agents.knowledge_ask import KnowledgeAskAgent, passage_from_hit
from stirling.api.dependencies import get_document_service, get_knowledge_ask_agent, require_user_id
from stirling.config import load_settings
from stirling.contracts import (
DeleteDocumentResponse,
IngestDocumentRequest,
IngestDocumentResponse,
)
from stirling.contracts.documents import PurgeOwnerResponse
from stirling.contracts.documents import (
AskDocumentsRequest,
AskDocumentsResponse,
DocumentStatsResponse,
DocumentSummary,
ListDocumentsResponse,
PurgeOwnerResponse,
SearchDocumentsRequest,
SearchDocumentsResponse,
)
from stirling.documents import DocumentService
from stirling.models import FileId, OwnerId, PrincipalId, UserId
@@ -45,6 +56,70 @@ async def ingest_document(
return IngestDocumentResponse(document_id=request.document_id, chunks_indexed=chunks_indexed)
@router.get("/stats", response_model=DocumentStatsResponse)
async def document_stats(
documents: Annotated[DocumentService, Depends(get_document_service)],
_user_id: Annotated[UserId, Depends(require_user_id)],
) -> DocumentStatsResponse:
"""Deployment-wide store counts for the admin dashboard.
Not tenant-filtered: counts cover every owner's content, so this reports
what the whole store holds, not what the caller can read.
"""
settings = load_settings()
counts = await documents.stats()
return DocumentStatsResponse(
backend=settings.documents_backend.value,
documents=counts.documents,
chunks=counts.chunks,
embedding_model=settings.rag_embedding_model,
)
@router.get("/list", response_model=ListDocumentsResponse)
async def list_documents(
documents: Annotated[DocumentService, Depends(get_document_service)],
user_id: Annotated[UserId, Depends(require_user_id)],
) -> ListDocumentsResponse:
"""Per-document rollup of what the caller can read: distinct document ids
with their stored source label and chunk count. Never shows another
principal's documents.
"""
summaries = await documents.list_documents([PrincipalId(user_id)])
return ListDocumentsResponse(
documents=[
DocumentSummary(document_id=FileId(s.collection), source=s.source, chunks=s.chunks) for s in summaries
]
)
@router.post("/search", response_model=SearchDocumentsResponse)
async def search_documents(
request: SearchDocumentsRequest,
documents: Annotated[DocumentService, Depends(get_document_service)],
user_id: Annotated[UserId, Depends(require_user_id)],
) -> SearchDocumentsResponse:
"""Semantic search across every document the caller can read.
Same retrieval path as the RAG toolset: embed the query, search the
caller's readable collections, merge by score.
"""
hits = await documents.search_with_collections(
request.query, principals=[PrincipalId(user_id)], top_k=request.top_k
)
return SearchDocumentsResponse(passages=[passage_from_hit(hit) for hit in hits])
@router.post("/ask", response_model=AskDocumentsResponse)
async def ask_documents(
request: AskDocumentsRequest,
agent: Annotated[KnowledgeAskAgent, Depends(get_knowledge_ask_agent)],
user_id: Annotated[UserId, Depends(require_user_id)],
) -> AskDocumentsResponse:
"""Answer a question from the caller's stored documents with inline citations."""
return await agent.ask(request, principals=[PrincipalId(user_id)])
@router.delete("/by-id/{document_id}", response_model=DeleteDocumentResponse)
async def delete_document(
document_id: FileId,
+82 -8
View File
@@ -42,6 +42,36 @@ from .contradiction import (
ContradictionReport,
ContradictionSeverity,
)
from .docparse import (
BlockType,
ChunkDocumentRequest,
ChunkDocumentResponse,
DocBlock,
DocChunk,
DocparseCapabilities,
DocparseMode,
DocparseTier,
DocTable,
ExtractedField,
ExtractFieldsRequest,
ExtractFieldsResponse,
ExtractTablesRequest,
ExtractTablesResponse,
FieldCitation,
FillDocxRequest,
FillDocxResponse,
ParseDocumentRequest,
ParseDocumentResponse,
RagIngestRequest,
RagIngestResponse,
SmartSplitRequest,
SmartSplitResponse,
SplitPart,
SuggestedField,
SuggestedFieldType,
SuggestSchemaRequest,
SuggestSchemaResponse,
)
from .document_classifier import (
ClassifyDocumentRequest,
ClassifyDocumentResponse,
@@ -49,13 +79,21 @@ from .document_classifier import (
LabelOption,
)
from .documents import (
AskDocumentsRequest,
AskDocumentsResponse,
DeleteDocumentResponse,
DocumentPassage,
DocumentStatsResponse,
DocumentSummary,
IngestDocumentRequest,
IngestDocumentResponse,
ListDocumentsResponse,
Page,
PageRange,
PageText,
PurgeOwnerResponse,
SearchDocumentsRequest,
SearchDocumentsResponse,
)
from .execution import (
AgentExecutionRequest,
@@ -140,10 +178,15 @@ __all__ = [
"AiFile",
"AiToolAgentStep",
"ArtifactKind",
"AskDocumentsRequest",
"AskDocumentsResponse",
"BlockType",
"CannotContinueExecutionAction",
"ChunkDocumentRequest",
"ChunkDocumentResponse",
"Claim",
"ClassifyDocumentRequest",
"ClassifyDocumentResponse",
"Claim",
"CommentSpec",
"CompletedExecutionAction",
"ConfigApplyResponse",
@@ -156,30 +199,45 @@ __all__ = [
"ContradictionSeverity",
"ConversationMessage",
"DeleteDocumentResponse",
"PurgeOwnerResponse",
"Discrepancy",
"DocumentClassificationResponse",
"LabelOption",
"DocumentMeta",
"DocumentSections",
"DiscrepancyKind",
"DocBlock",
"DocChunk",
"DocTable",
"DocparseCapabilities",
"DocparseMode",
"DocparseTier",
"DocumentClassificationResponse",
"DocumentMeta",
"DocumentPassage",
"DocumentSections",
"DocumentStatsResponse",
"DocumentSummary",
"EditCannotDoResponse",
"EditClarificationRequest",
"EditPlanResponse",
"Evidence",
"ExecutionContext",
"ExecutionStepResult",
"ExtractFieldsRequest",
"ExtractFieldsResponse",
"ExtractTablesRequest",
"ExtractTablesResponse",
"ExtractedField",
"ExtractedFileText",
"ExtractedTextArtifact",
"FieldCitation",
"FillDocxRequest",
"FillDocxResponse",
"Folio",
"FolioManifest",
"FolioType",
"format_conversation_history",
"format_file_names",
"GenerateFileResponse",
"HealthResponse",
"IngestDocumentRequest",
"IngestDocumentResponse",
"LabelOption",
"ListDocumentsResponse",
"MathAuditorToolReportArtifact",
"NeedContentFileRequest",
"NeedContentResponse",
@@ -190,6 +248,8 @@ __all__ = [
"Page",
"PageRange",
"PageText",
"ParseDocumentRequest",
"ParseDocumentResponse",
"PdfCommentInstruction",
"PdfCommentReport",
"PdfCommentRequest",
@@ -212,9 +272,21 @@ __all__ = [
"PdfReviewOrchestrateResponse",
"PdfTextSelection",
"ProgressEvent",
"PurgeOwnerResponse",
"RagIngestRequest",
"RagIngestResponse",
"Requisition",
"SearchDocumentsRequest",
"SearchDocumentsResponse",
"Severity",
"SmartSplitRequest",
"SmartSplitResponse",
"SplitPart",
"StepKind",
"SuggestSchemaRequest",
"SuggestSchemaResponse",
"SuggestedField",
"SuggestedFieldType",
"SupportedCapability",
"TextChunk",
"ToolCallExecutionAction",
@@ -228,4 +300,6 @@ __all__ = [
"WholeDocSliceDone",
"WorkflowArtifact",
"WorkflowOutcome",
"format_conversation_history",
"format_file_names",
]
+40
View File
@@ -206,6 +206,46 @@ class SuggestSchemaResponse(ApiModel):
fields: list[SuggestedField] = Field(default_factory=list)
class SplitPart(ApiModel):
start_page: int = Field(ge=1)
end_page: int = Field(ge=1)
label: str
confidence: float = Field(ge=0.0, le=1.0)
class SmartSplitRequest(ApiModel):
file_name: str = Field(min_length=1)
rule: str = Field(
min_length=1, description="Natural-language boundary rule, e.g. 'split where a new invoice starts'."
)
pages: list[PageText]
max_parts: int = Field(default=50, ge=1, le=500)
class SmartSplitResponse(ApiModel):
parts: list[SplitPart] = Field(default_factory=list)
class ChunkDocumentRequest(ApiModel):
file_name: str = Field(min_length=1)
pages: list[PageText] | None = None
content_base64: str | None = None
chunk_size: int = Field(default=512, ge=64, le=32_768)
overlap: int = Field(default=64, ge=0, le=4_096)
mode: DocparseMode = DocparseMode.AUTO
class FillDocxRequest(ApiModel):
template_base64: str = Field(min_length=1)
data: dict[str, JsonValue]
class FillDocxResponse(ApiModel):
docx_base64: str
replaced: int = Field(ge=0)
missing: list[str] = Field(default_factory=list)
class DocparseCapabilities(ApiModel):
"""What the engine can actually do right now; Java caches and republishes this."""
@@ -75,3 +75,65 @@ class PurgeOwnerResponse(ApiModel):
owner_id: OwnerId
deleted: int = Field(ge=0)
class DocumentStatsResponse(ApiModel):
"""Returned by ``GET /api/v1/documents/stats``. Deployment-wide counts
(every owner's content) powering the admin dashboard."""
backend: str
documents: int = Field(ge=0)
chunks: int = Field(ge=0)
embedding_model: str
class DocumentSummary(ApiModel):
"""One stored document the caller can read: its id, source label, chunk count."""
document_id: FileId
source: str
chunks: int = Field(ge=0)
class ListDocumentsResponse(ApiModel):
"""Returned by ``GET /api/v1/documents/list``. Caller-scoped rollup."""
documents: list[DocumentSummary]
class SearchDocumentsRequest(ApiModel):
"""Semantic search over every document the caller can read."""
query: str = Field(min_length=1)
top_k: int = Field(default=8, ge=1, le=50)
class DocumentPassage(ApiModel):
"""A retrieved chunk on the wire. Page bounds and heading path come from
chunk metadata when present (docparse chunks carry them); nulls otherwise."""
document_id: FileId
text: str
score: float
page_start: int | None = None
page_end: int | None = None
heading_path: list[str] = Field(default_factory=list)
source: str | None = None
class SearchDocumentsResponse(ApiModel):
passages: list[DocumentPassage]
class AskDocumentsRequest(ApiModel):
"""Question answered only from the caller's stored documents."""
question: str = Field(min_length=1)
top_k: int = Field(default=8, ge=1, le=20)
class AskDocumentsResponse(ApiModel):
"""Grounded answer with inline citations plus the passages it drew from."""
answer: str
passages: list[DocumentPassage]
+4
View File
@@ -8,14 +8,18 @@ from __future__ import annotations
from stirling.docparse.capability import activate_site, probe_capabilities
from stirling.docparse.chunking import advanced_chunks, basic_chunks
from stirling.docparse.docxfill import fill_docx
from stirling.docparse.extractor import ExtractFieldsAgent
from stirling.docparse.splitter import SmartSplitAgent
from stirling.docparse.suggest_schema import SuggestSchemaAgent
__all__ = [
"ExtractFieldsAgent",
"SmartSplitAgent",
"SuggestSchemaAgent",
"activate_site",
"advanced_chunks",
"basic_chunks",
"fill_docx",
"probe_capabilities",
]
+178
View File
@@ -0,0 +1,178 @@
"""Fill DOCX templates from JSON data: ``{{ dotted.path }}`` placeholders.
Scalar placeholders are replaced everywhere (body, tables, headers, footers).
A table row whose text contains ``{{#items.field}}`` markers is treated as a
row template: it is cloned once per element of the ``items`` array. Unresolved
placeholders are left in place and reported back so the caller can surface them.
Formatting caveat: a placeholder split across styled runs collapses that
paragraph's text into its first run's style.
"""
from __future__ import annotations
import base64
import copy
import io
import re
from typing import Any
from pydantic import JsonValue
from stirling.contracts.docparse import FillDocxRequest, FillDocxResponse
_PLACEHOLDER = re.compile(r"\{\{\s*(#?[\w.]+)\s*\}\}")
def _resolve(path: str, data: dict[str, Any]) -> Any | None:
node: Any = data
for part in path.split("."):
if isinstance(node, dict) and part in node:
node = node[part]
else:
return None
return node
def _render_scalar(value: Any) -> str:
if value is None:
return ""
if isinstance(value, bool):
return "true" if value else "false"
if isinstance(value, list):
return ", ".join(_render_scalar(v) for v in value)
return str(value)
class _Stats:
def __init__(self) -> None:
self.replaced = 0
self.missing: set[str] = set()
def _fill_paragraph(paragraph: Any, data: dict[str, Any], stats: _Stats) -> None:
text = paragraph.text
if "{{" not in text:
return
def substitute(match: re.Match[str]) -> str:
path = match.group(1)
if path.startswith("#"):
return match.group(0) # row-template marker, handled at table level
value = _resolve(path, data)
if value is None:
stats.missing.add(path)
return match.group(0)
stats.replaced += 1
return _render_scalar(value)
rendered = _PLACEHOLDER.sub(substitute, text)
if rendered == text:
return
# Collapse into the first run to survive placeholders split across runs.
if paragraph.runs:
paragraph.runs[0].text = rendered
for run in paragraph.runs[1:]:
run.text = ""
else:
paragraph.add_run(rendered)
def _row_template_array(row: Any) -> str | None:
"""Return the array name when the row carries ``{{#name.field}}`` markers."""
names = {
match.group(1)[1:].split(".")[0]
for cell in row.cells
for match in _PLACEHOLDER.finditer(cell.text)
if match.group(1).startswith("#")
}
return names.pop() if len(names) == 1 else None
def _fill_table(table: Any, data: dict[str, Any], stats: _Stats) -> None:
for row in list(table.rows):
array_name = _row_template_array(row)
if array_name is None:
continue
items = _resolve(array_name, data)
if not isinstance(items, list):
stats.missing.add(array_name)
continue
for _ in items:
new_row = copy.deepcopy(row._tr)
row._tr.addprevious(new_row)
# Clones sit before the template; rewrite their markers, then drop the template.
_rewrite_cloned_rows(table, row, array_name, items, data, stats)
row._tr.getparent().remove(row._tr)
def _rewrite_cloned_rows(
table: Any, template_row: Any, array_name: str, items: list[Any], data: dict[str, Any], stats: _Stats
) -> None:
marker_prefix = f"#{array_name}"
clones = [
r for r in table.rows if r._tr is not template_row._tr and marker_prefix in "".join(c.text for c in r.cells)
]
for row, item in zip(clones, items, strict=False):
scoped = dict(data)
scoped[array_name] = item if isinstance(item, dict) else {"value": item}
for cell in row.cells:
for paragraph in cell.paragraphs:
text = paragraph.text
def substitute(match: re.Match[str]) -> str:
path = match.group(1)
if not path.startswith(marker_prefix):
return match.group(0)
item_path = path[1:] # "#items.field" -> "items.field"
value = _resolve(item_path, scoped)
if value is None and "." not in item_path:
value = scoped[array_name].get("value") if isinstance(scoped[array_name], dict) else None
if value is None:
stats.missing.add(item_path)
return match.group(0)
stats.replaced += 1
return _render_scalar(value)
rendered = _PLACEHOLDER.sub(substitute, text)
if rendered != text:
if paragraph.runs:
paragraph.runs[0].text = rendered
for run in paragraph.runs[1:]:
run.text = ""
else:
paragraph.add_run(rendered)
def _walk_paragraphs(document: Any) -> list[tuple[Any, Any]]:
"""Yield (paragraph, containing table or None) across body, tables, headers, footers."""
found: list[tuple[Any, Any]] = [(p, None) for p in document.paragraphs]
for table in document.tables:
for row in table.rows:
for cell in row.cells:
found.extend((p, table) for p in cell.paragraphs)
for section in document.sections:
for part in (section.header, section.footer):
found.extend((p, None) for p in part.paragraphs)
return found
def fill_docx(request: FillDocxRequest) -> FillDocxResponse:
import docx # local import: python-docx is small but only needed here
data: dict[str, JsonValue] = dict(request.data)
document = docx.Document(io.BytesIO(base64.b64decode(request.template_base64)))
stats = _Stats()
for table in document.tables:
_fill_table(table, data, stats)
for paragraph, _table in _walk_paragraphs(document):
_fill_paragraph(paragraph, data, stats)
out = io.BytesIO()
document.save(out)
return FillDocxResponse(
docx_base64=base64.b64encode(out.getvalue()).decode("ascii"),
replaced=stats.replaced,
missing=sorted(stats.missing),
)
+105
View File
@@ -0,0 +1,105 @@
"""Content-based document splitting: an LLM finds sub-document boundaries.
Works entirely from caller-supplied page text (basic tier friendly); the fast
model sees a bounded per-page preview and answers with boundary start pages,
which are then validated in code (monotonic, in range, capped)."""
from __future__ import annotations
import logging
from pydantic import Field
from pydantic_ai import Agent
from stirling.agents.output_mode import output_retries, structured_output
from stirling.contracts.docparse import SmartSplitRequest, SmartSplitResponse, SplitPart
from stirling.contracts.documents import PageText
from stirling.models import ApiModel
from stirling.services import AppRuntime
logger = logging.getLogger(__name__)
# Per-page preview budget; boundaries are recognisable from page openings.
PAGE_PREVIEW_CHARS = 600
_SYSTEM_PROMPT = (
"You split a multi-document file into its component documents.\n"
"\n"
"You are shown the beginning of every page. Apply the user's splitting rule and "
"answer with every page where a NEW component document starts.\n"
"Rules:\n"
"- Page 1 always starts the first component.\n"
"- Give each component a short descriptive label (e.g. 'Invoice #4821', 'Cover letter').\n"
"- Give your confidence 0.0-1.0 per boundary.\n"
"- If the rule doesn't match anything, return just the page-1 component spanning the whole file."
)
class _Boundary(ApiModel):
start_page: int = Field(ge=1, description="First page of this component document.")
label: str = Field(description="Short human label for the component.")
confidence: float = Field(ge=0.0, le=1.0)
class _SplitOutput(ApiModel):
boundaries: list[_Boundary] = Field(default_factory=list)
def _format_pages(pages: list[PageText], max_pages: int) -> str:
shown = pages[:max_pages]
parts = [f"[Page {p.page_number}] {p.text[:PAGE_PREVIEW_CHARS]}" for p in shown]
if len(pages) > max_pages:
parts.append(f"({len(pages) - max_pages} further pages omitted)")
return "\n\n".join(parts) if parts else "(no extractable text)"
def validate_boundaries(output: _SplitOutput, page_count: int, max_parts: int) -> list[SplitPart]:
"""Coerce the model's boundaries into a clean, complete partition of 1..page_count."""
starts: dict[int, _Boundary] = {}
for boundary in output.boundaries:
if 1 <= boundary.start_page <= page_count and boundary.start_page not in starts:
starts[boundary.start_page] = boundary
if 1 not in starts:
starts[1] = _Boundary(start_page=1, label="Document", confidence=1.0)
ordered = [starts[k] for k in sorted(starts)][:max_parts]
parts: list[SplitPart] = []
for i, boundary in enumerate(ordered):
end_page = ordered[i + 1].start_page - 1 if i + 1 < len(ordered) else page_count
parts.append(
SplitPart(
start_page=boundary.start_page,
end_page=end_page,
label=boundary.label.strip() or f"Part {i + 1}",
confidence=round(boundary.confidence, 4),
)
)
return parts
class SmartSplitAgent:
def __init__(self, runtime: AppRuntime) -> None:
self.runtime = runtime
provider = runtime.settings.chat_provider
self._agent: Agent[None, _SplitOutput] = Agent(
model=runtime.fast_model,
output_type=structured_output([_SplitOutput], chat_provider=provider),
system_prompt=_SYSTEM_PROMPT,
model_settings=runtime.fast_model_settings,
retries=output_retries(provider),
)
async def split(self, request: SmartSplitRequest) -> SmartSplitResponse:
pages = request.pages
if not pages:
return SmartSplitResponse(parts=[])
page_count = max(p.page_number for p in pages)
prompt = (
f"Splitting rule: {request.rule}\n\n"
f"Document file name: {request.file_name}\n"
f"Pages:\n{_format_pages(pages, self.runtime.settings.max_pages)}"
)
result = await self._agent.run(prompt)
parts = validate_boundaries(result.output, page_count, request.max_parts)
logger.info("docparse: split %s into %d parts", request.file_name, len(parts))
return SmartSplitResponse(parts=parts)
+12 -2
View File
@@ -3,11 +3,20 @@ from __future__ import annotations
from stirling.documents.embedder import EmbeddingService
from stirling.documents.pgvector_store import PgVectorStore
from stirling.documents.rag_capability import RagCapability
from stirling.documents.service import DocumentService
from stirling.documents.service import CollectionSearchHit, DocumentService
from stirling.documents.sqlite_vec_store import SqliteVecStore
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
from stirling.documents.store import (
CollectionSummary,
Document,
DocumentStore,
SearchResult,
StoredPage,
StoreStats,
)
__all__ = [
"CollectionSearchHit",
"CollectionSummary",
"Document",
"DocumentService",
"DocumentStore",
@@ -16,5 +25,6 @@ __all__ = [
"RagCapability",
"SearchResult",
"SqliteVecStore",
"StoreStats",
"StoredPage",
]
@@ -10,7 +10,14 @@ from pgvector.psycopg import register_vector_async
from psycopg_pool import AsyncConnectionPool
from stirling.contracts.documents import Page, PageRange
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
from stirling.documents.store import (
CollectionSummary,
Document,
DocumentStore,
SearchResult,
StoredPage,
StoreStats,
)
from stirling.models import OwnerId, PrincipalId
_READ_PERMISSION = "read"
@@ -411,5 +418,44 @@ class PgVectorStore(DocumentStore):
rows = await cur.fetchall()
return [r[0] for r in rows]
async def list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
if not principals:
return []
await self._ensure_ready()
async with self._pool.connection() as conn:
async with conn.cursor() as cur:
# MIN(owner_id) mirrors _readable_owner_for's ORDER BY owner_id LIMIT 1.
await cur.execute(
"""
SELECT r.collection, m.source, COUNT(d.id)
FROM (
SELECT collection, MIN(owner_id) AS owner_id
FROM document_acl
WHERE permission = %s AND principal_id = ANY(%s)
GROUP BY collection
) r
JOIN documents_meta m ON m.collection = r.collection AND m.owner_id = r.owner_id
LEFT JOIN rag_documents d ON d.collection = r.collection AND d.owner_id = r.owner_id
GROUP BY r.collection, m.source
ORDER BY r.collection
""",
(_READ_PERMISSION, list(principals)),
)
rows = await cur.fetchall()
return [CollectionSummary(collection=r[0], source=r[1], chunks=int(r[2])) for r in rows]
async def stats(self) -> StoreStats:
await self._ensure_ready()
async with self._pool.connection() as conn:
async with conn.cursor() as cur:
await cur.execute("SELECT COUNT(DISTINCT collection) FROM documents_meta")
doc_row = await cur.fetchone()
await cur.execute("SELECT COUNT(*) FROM rag_documents")
chunk_row = await cur.fetchone()
return StoreStats(
documents=int(doc_row[0]) if doc_row else 0,
chunks=int(chunk_row[0]) if chunk_row else 0,
)
async def close(self) -> None:
await self._pool.close()
+52 -1
View File
@@ -1,15 +1,32 @@
from __future__ import annotations
import logging
from dataclasses import dataclass
from datetime import datetime
from stirling.contracts.documents import Page, PageRange, PageText
from stirling.documents.embedder import EmbeddingService
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
from stirling.documents.store import (
CollectionSummary,
Document,
DocumentStore,
SearchResult,
StoredPage,
StoreStats,
)
from stirling.models import FileId, OwnerId, PrincipalId
logger = logging.getLogger(__name__)
@dataclass(frozen=True)
class CollectionSearchHit:
"""A search result tagged with the collection it came from."""
collection: FileId
result: SearchResult
PAGE_NUMBER_METADATA_KEY = "page_number"
CONTENT_TYPE_METADATA_KEY = "content_type"
PAGE_TEXT_CONTENT_TYPE = "page_text"
@@ -187,6 +204,32 @@ class DocumentService:
all_results.sort(key=lambda r: r.score, reverse=True)
return all_results[:k]
async def search_with_collections(
self,
query: str,
principals: list[PrincipalId],
top_k: int | None = None,
) -> list[CollectionSearchHit]:
"""Cross-collection search like :meth:`search`, but every result keeps
the collection it came from. Restricted to what ``principals`` can read.
"""
k = top_k if top_k is not None else self._default_top_k
query_embedding = await self._embedder.embed_query(query)
hits: list[CollectionSearchHit] = []
for col_name in await self._store.list_collections(principals):
try:
results = await self._store.search(col_name, query_embedding, k, principals)
except Exception: # noqa: BLE001 - any backend error on one collection should not stop the others
logger.warning(
"Skipping collection %s during cross-collection search",
col_name,
exc_info=True,
)
continue
hits.extend(CollectionSearchHit(collection=FileId(col_name), result=r) for r in results)
hits.sort(key=lambda hit: hit.result.score, reverse=True)
return hits[:k]
async def read_pages(
self,
collection: FileId,
@@ -223,6 +266,10 @@ class DocumentService:
"""List collections readable by at least one of ``principals``."""
return [FileId(name) for name in await self._store.list_collections(principals)]
async def list_documents(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
"""Per-document rollup (source, chunk count) readable by ``principals``."""
return await self._store.list_collection_summaries(principals)
async def grant_read(
self,
collection: FileId,
@@ -241,6 +288,10 @@ class DocumentService:
"""Revoke a principal's access on an existing doc."""
await self._store.revoke(collection, owner_id, principal)
async def stats(self) -> StoreStats:
"""Deployment-wide document/chunk counts from the backing store."""
return await self._store.stats()
async def close(self) -> None:
"""Release the underlying store's resources."""
await self._store.close()
@@ -11,7 +11,14 @@ from pathlib import Path
import sqlite_vec
from stirling.contracts.documents import Page, PageRange
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
from stirling.documents.store import (
CollectionSummary,
Document,
DocumentStore,
SearchResult,
StoredPage,
StoreStats,
)
from stirling.models import OwnerId, PrincipalId
_READ_PERMISSION = "read"
@@ -538,6 +545,42 @@ class SqliteVecStore(DocumentStore):
).fetchall()
return [r[0] for r in rows]
async def list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
async with self._lock:
return await asyncio.to_thread(self._sync_list_collection_summaries, principals)
def _sync_list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
if not principals:
return []
placeholders = ",".join("?" * len(principals))
# MIN(owner_id) mirrors _readable_owner_for's ORDER BY owner_id LIMIT 1.
rows = self._conn.execute(
f"""
SELECT r.collection, m.source, COUNT(d.id)
FROM (
SELECT collection, MIN(owner_id) AS owner_id
FROM document_acl
WHERE permission = ? AND principal_id IN ({placeholders})
GROUP BY collection
) r
JOIN documents_meta m ON m.collection = r.collection AND m.owner_id = r.owner_id
LEFT JOIN documents d ON d.collection = r.collection AND d.owner_id = r.owner_id
GROUP BY r.collection, m.source
ORDER BY r.collection
""",
(_READ_PERMISSION, *principals),
).fetchall()
return [CollectionSummary(collection=r[0], source=r[1], chunks=int(r[2])) for r in rows]
async def stats(self) -> StoreStats:
async with self._lock:
return await asyncio.to_thread(self._sync_stats)
def _sync_stats(self) -> StoreStats:
documents = self._conn.execute("SELECT COUNT(DISTINCT collection) FROM documents_meta").fetchone()[0]
chunks = self._conn.execute("SELECT COUNT(*) FROM documents").fetchone()[0]
return StoreStats(documents=int(documents), chunks=int(chunks))
async def close(self) -> None:
async with self._lock:
await asyncio.to_thread(self._sync_close)
+31
View File
@@ -34,6 +34,23 @@ class StoredPage:
char_count: int
@dataclass
class StoreStats:
"""Deployment-wide counts: distinct document ids and total vector-chunk rows."""
documents: int
chunks: int
@dataclass
class CollectionSummary:
"""Rollup row for one readable collection: stored source label + chunk count."""
collection: str
source: str
chunks: int
class DocumentStore(ABC):
"""Abstract interface for document storage backends.
@@ -148,6 +165,20 @@ class DocumentStore(ABC):
async def list_collections(self, principals: list[PrincipalId]) -> list[str]:
"""Return collection names readable by at least one of ``principals``."""
@abstractmethod
async def list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
"""Per-collection rollup (source + chunk count) readable by ``principals``.
Counts cover the same owner's copy a read would resolve to, so the
rollup never leaks another tenant's content.
"""
# ── deployment-wide stats (not tenant-scoped) ──────────────────────────
@abstractmethod
async def stats(self) -> StoreStats:
"""Count every owner's content: distinct document ids + total chunk rows."""
# ── lifecycle ──────────────────────────────────────────────────────────
@abstractmethod
+70
View File
@@ -0,0 +1,70 @@
from __future__ import annotations
import base64
import io
from typing import Any
import docx
from stirling.contracts.docparse import FillDocxRequest
from stirling.docparse.docxfill import fill_docx
def _template_base64() -> str:
document = docx.Document()
document.add_paragraph("Dear {{ customer.name }},")
document.add_paragraph("Your total is {{ total }}.")
document.add_paragraph("Unknown: {{ nowhere.field }}")
table = document.add_table(rows=2, cols=2)
table.rows[0].cells[0].text = "Item"
table.rows[0].cells[1].text = "Price"
table.rows[1].cells[0].text = "{{#items.name}}"
table.rows[1].cells[1].text = "{{#items.price}}"
buffer = io.BytesIO()
document.save(buffer)
return base64.b64encode(buffer.getvalue()).decode("ascii")
def _load(response_base64: str) -> Any:
return docx.Document(io.BytesIO(base64.b64decode(response_base64)))
def test_fills_scalars_tables_and_reports_missing() -> None:
request = FillDocxRequest(
template_base64=_template_base64(),
data={
"customer": {"name": "ACME GmbH"},
"total": 12.5,
"items": [
{"name": "Widget", "price": "2.00"},
{"name": "Gadget", "price": "10.50"},
],
},
)
response = fill_docx(request)
filled = _load(response.docx_base64)
paragraphs = [p.text for p in filled.paragraphs]
assert "Dear ACME GmbH," in paragraphs
assert "Your total is 12.5." in paragraphs
# Unresolved placeholders stay put and are reported.
assert any("{{ nowhere.field }}" in p for p in paragraphs)
assert response.missing == ["nowhere.field"]
table = filled.tables[0]
rendered_rows = [[cell.text for cell in row.cells] for row in table.rows]
assert ["Widget", "2.00"] in rendered_rows
assert ["Gadget", "10.50"] in rendered_rows
# The template row is gone.
assert all("{{#" not in cell for row in rendered_rows for cell in row)
assert response.replaced >= 6
def test_empty_items_removes_template_row() -> None:
request = FillDocxRequest(
template_base64=_template_base64(),
data={"customer": {"name": "X"}, "total": 1, "items": []},
)
response = fill_docx(request)
filled = _load(response.docx_base64)
assert len(filled.tables[0].rows) == 1 # only the header remains
+28
View File
@@ -9,6 +9,7 @@ from stirling.docparse.extractor import (
build_output_model,
find_quote,
)
from stirling.docparse.splitter import _Boundary, _SplitOutput, validate_boundaries
def _pages() -> list[PageText]:
@@ -43,6 +44,33 @@ def test_find_quote_missing_returns_none() -> None:
assert find_quote(" ", _pages()) is None
def test_validate_boundaries_partitions_cleanly() -> None:
output = _SplitOutput(
boundaries=[
_Boundary(start_page=4, label="Invoice B", confidence=0.8),
_Boundary(start_page=1, label="Invoice A", confidence=0.9),
_Boundary(start_page=4, label="dup", confidence=0.1),
_Boundary(start_page=99, label="out of range", confidence=0.5),
]
)
parts = validate_boundaries(output, page_count=6, max_parts=10)
assert [(p.start_page, p.end_page) for p in parts] == [(1, 3), (4, 6)]
assert parts[0].label == "Invoice A"
def test_validate_boundaries_inserts_page_one() -> None:
output = _SplitOutput(boundaries=[_Boundary(start_page=3, label="Part", confidence=0.7)])
parts = validate_boundaries(output, page_count=5, max_parts=10)
assert parts[0].start_page == 1
assert parts[1].start_page == 3
assert parts[-1].end_page == 5
def test_validate_boundaries_empty_output_spans_whole_file() -> None:
parts = validate_boundaries(_SplitOutput(), page_count=7, max_parts=10)
assert [(p.start_page, p.end_page) for p in parts] == [(1, 7)]
def _answers(quote: str | None, confidence: float) -> BaseModel:
model = build_output_model({"type": "object", "properties": {"invoice_number": {"type": "string"}}})
return model.model_validate({"invoiceNumber": {"value": "INV-123", "quote": quote, "confidence": confidence}})
+88 -1
View File
@@ -7,7 +7,7 @@ from stirling.documents.chunker import chunk_text
from stirling.documents.rag_capability import RagCapability
from stirling.documents.service import DocumentService
from stirling.documents.sqlite_vec_store import SqliteVecStore
from stirling.documents.store import Document, SearchResult
from stirling.documents.store import CollectionSummary, Document, SearchResult
from stirling.models import FileId, OwnerId, PrincipalId
# Personal-doc tests reuse the same opaque string in all three roles — keeps the
@@ -178,6 +178,27 @@ class TestSqliteVecStore:
assert await store.list_collections(OWNER_PRINCIPALS) == []
assert await store.list_collections(OTHER_OWNER_PRINCIPALS) == ["c.pdf"]
@pytest.mark.anyio
async def test_stats_count_distinct_document_ids_and_chunk_rows(self) -> None:
"""Stats span every owner; a document id shared by two owners counts once."""
store = SqliteVecStore.ephemeral()
empty = await store.stats()
assert (empty.documents, empty.chunks) == (0, 0)
for owner, principals, name, texts in (
(OWNER, OWNER_PRINCIPALS, "doc-a", ["one", "two"]),
(OWNER, OWNER_PRINCIPALS, "doc-b", ["three"]),
(OTHER_OWNER, OTHER_OWNER_PRINCIPALS, "doc-b", ["four"]),
):
await store.ensure_collection(name, f"{name}.pdf", owner, None)
await store.grant_read(name, owner, principals)
docs = [Document(id=str(i), text=t, metadata={}) for i, t in enumerate(texts)]
await store.add_documents(name, docs, [[1.0, 0.0]] * len(docs), owner)
stats = await store.stats()
assert stats.documents == 2
assert stats.chunks == 4
@pytest.mark.anyio
async def test_reap_expired_drops_collections_past_expires_at(self) -> None:
"""TTL backstop: rows with ``expires_at`` in the past go away on reap."""
@@ -239,6 +260,47 @@ class TestSqliteVecStore:
# Owner still can.
assert await store.has_collection("doc", OWNER_PRINCIPALS) is True
@pytest.mark.anyio
async def test_list_collection_summaries_rolls_up_readable_collections(self) -> None:
"""Rollup: one row per readable collection with its source and chunk count."""
store = SqliteVecStore.ephemeral()
await store.ensure_collection("doc-a", "a.pdf", OWNER, None)
await store.grant_read("doc-a", OWNER, OWNER_PRINCIPALS)
docs = [Document(id="1", text="one", metadata={}), Document(id="2", text="two", metadata={})]
await store.add_documents("doc-a", docs, [[1.0, 0.0], [0.0, 1.0]], OWNER)
# Collection with no vector chunks yet: still listed, zero count.
await store.ensure_collection("doc-b", "b.pdf", OWNER, None)
await store.grant_read("doc-b", OWNER, OWNER_PRINCIPALS)
summaries = await store.list_collection_summaries(OWNER_PRINCIPALS)
assert summaries == [
CollectionSummary(collection="doc-a", source="a.pdf", chunks=2),
CollectionSummary(collection="doc-b", source="b.pdf", chunks=0),
]
@pytest.mark.anyio
async def test_list_collection_summaries_scoped_to_principals(self) -> None:
"""One principal's rollup never lists, or counts, another owner's copy."""
store = SqliteVecStore.ephemeral()
await store.ensure_collection("shared-id", "alice.pdf", OWNER, None)
await store.grant_read("shared-id", OWNER, OWNER_PRINCIPALS)
await store.add_documents("shared-id", [Document(id="1", text="alice", metadata={})], [[1.0, 0.0]], OWNER)
await store.ensure_collection("shared-id", "bob.pdf", OTHER_OWNER, None)
await store.grant_read("shared-id", OTHER_OWNER, OTHER_OWNER_PRINCIPALS)
bob_docs = [Document(id="1", text="bob", metadata={}), Document(id="2", text="bob2", metadata={})]
await store.add_documents("shared-id", bob_docs, [[1.0, 0.0], [0.0, 1.0]], OTHER_OWNER)
await store.ensure_collection("bob-only", "bob-only.pdf", OTHER_OWNER, None)
await store.grant_read("bob-only", OTHER_OWNER, OTHER_OWNER_PRINCIPALS)
assert await store.list_collection_summaries(OWNER_PRINCIPALS) == [
CollectionSummary(collection="shared-id", source="alice.pdf", chunks=1)
]
assert await store.list_collection_summaries(OTHER_OWNER_PRINCIPALS) == [
CollectionSummary(collection="bob-only", source="bob-only.pdf", chunks=0),
CollectionSummary(collection="shared-id", source="bob.pdf", chunks=2),
]
assert await store.list_collection_summaries([]) == []
# DocumentService (with stub embedder)
@@ -398,6 +460,31 @@ class TestDocumentService:
multi_results = await documents.search("deploy", principals=[hr_group, eng_group])
assert len(multi_results) > 0
@pytest.mark.anyio
async def test_search_with_collections_tags_results_and_respects_acl(self, documents: DocumentService) -> None:
"""Collection-tagged search only reaches collections the caller can read."""
await documents.ingest(
FileId("col-a"),
_pages("Alpha content."),
source="a.pdf",
owner_id=OWNER,
read_principals=OWNER_PRINCIPALS,
expires_at=None,
)
await documents.ingest(
FileId("col-b"),
_pages("Beta content."),
source="b.pdf",
owner_id=OTHER_OWNER,
read_principals=OTHER_OWNER_PRINCIPALS,
expires_at=None,
)
hits = await documents.search_with_collections("content", principals=OWNER_PRINCIPALS)
assert hits
assert {hit.collection for hit in hits} == {"col-a"}
assert all(hit.result.document.text for hit in hits)
@pytest.mark.anyio
async def test_delete_collection(self, documents: DocumentService) -> None:
await documents.ingest(
+271 -2
View File
@@ -6,9 +6,10 @@ import pytest
from fastapi.testclient import TestClient
from stirling.api import app
from stirling.api.dependencies import get_document_service
from stirling.api.dependencies import get_document_service, get_knowledge_ask_agent
from stirling.contracts import AskDocumentsRequest, AskDocumentsResponse, DocumentPassage
from stirling.documents import Document, DocumentService, SqliteVecStore
from stirling.models import FileId, PrincipalId, UserId
from stirling.models import FileId, OwnerId, PrincipalId, UserId
USER = UserId("test-user")
USER_PRINCIPALS = [PrincipalId("test-user")]
@@ -348,6 +349,274 @@ def test_purge_by_owner_rejects_missing_user_header(client: TestClient) -> None:
assert response.status_code == 401
# ── GET /documents/list ─────────────────────────────────────────────────
def _ingest(client: TestClient, document_id: str, source: str, texts: list[str], owner: str) -> None:
client.post(
"/api/v1/documents",
json={
"documentId": document_id,
"source": source,
"pageText": [{"pageNumber": i, "text": t} for i, t in enumerate(texts, 1)],
"ownerId": owner,
"readPrincipals": [owner],
"expiresAt": None,
},
headers={"X-User-Id": owner},
)
def test_list_documents_returns_caller_rollup(client: TestClient) -> None:
_ingest(client, "list-a", "a.pdf", ["Page one text.", "Page two text."], USER)
_ingest(client, "list-b", "b.pdf", ["Only page."], USER)
response = client.get("/api/v1/documents/list", headers=HEADERS)
assert response.status_code == 200
documents = response.json()["documents"]
assert [d["documentId"] for d in documents] == ["list-a", "list-b"]
by_id = {d["documentId"]: d for d in documents}
assert by_id["list-a"]["source"] == "a.pdf"
assert by_id["list-a"]["chunks"] >= 2
assert by_id["list-b"]["source"] == "b.pdf"
assert by_id["list-b"]["chunks"] >= 1
def test_list_documents_empty_for_new_user(client: TestClient) -> None:
response = client.get("/api/v1/documents/list", headers=HEADERS)
assert response.status_code == 200
assert response.json() == {"documents": []}
def test_list_documents_hides_other_users_documents(client: TestClient) -> None:
"""User A must never see user B's documents in the rollup."""
_ingest(client, "alice-doc", "alice.pdf", ["alice content"], "alice")
_ingest(client, "bob-doc", "bob.pdf", ["bob content"], "bob")
alice_docs = client.get("/api/v1/documents/list", headers={"X-User-Id": "alice"}).json()["documents"]
bob_docs = client.get("/api/v1/documents/list", headers={"X-User-Id": "bob"}).json()["documents"]
assert [d["documentId"] for d in alice_docs] == ["alice-doc"]
assert [d["documentId"] for d in bob_docs] == ["bob-doc"]
def test_list_documents_rejects_missing_user_header(client: TestClient) -> None:
assert client.get("/api/v1/documents/list").status_code == 401
# ── POST /documents/search ──────────────────────────────────────────────
def test_search_documents_maps_page_text_chunks(client: TestClient) -> None:
"""Plain ingested chunks only carry page_number: both bounds map to it and
the ":page:N" suffix is stripped off the source."""
client.post(
"/api/v1/documents",
json={
"documentId": "report",
"source": "report.pdf",
"pageText": [{"pageNumber": 3, "text": "The launch is planned for October."}],
"ownerId": USER,
"readPrincipals": [USER],
"expiresAt": None,
},
headers=HEADERS,
)
response = client.post("/api/v1/documents/search", json={"query": "launch", "topK": 5}, headers=HEADERS)
assert response.status_code == 200
passages = response.json()["passages"]
assert len(passages) >= 1
passage = passages[0]
assert passage["documentId"] == "report"
assert passage["pageStart"] == 3
assert passage["pageEnd"] == 3
assert passage["headingPath"] == []
assert passage["source"] == "report.pdf"
assert "launch" in passage["text"]
assert isinstance(passage["score"], float)
@pytest.mark.anyio
async def test_search_documents_maps_docparse_chunk_metadata(client: TestClient, service: DocumentService) -> None:
"""Docparse chunks carry page bounds + heading path; they map straight onto the wire."""
await service.ingest_prepared(
collection=FileId("dp-doc"),
chunks=[
(
"Revenue grew 12% in Q2.",
{
"content_type": "docparse_chunk",
"page_start": "2",
"page_end": "3",
"heading_path": "Report > Finance",
},
)
],
source="q2.pdf",
owner_id=OwnerId(USER),
read_principals=USER_PRINCIPALS,
expires_at=None,
)
response = client.post("/api/v1/documents/search", json={"query": "revenue"}, headers=HEADERS)
assert response.status_code == 200
passage = response.json()["passages"][0]
assert passage["documentId"] == "dp-doc"
assert passage["pageStart"] == 2
assert passage["pageEnd"] == 3
assert passage["headingPath"] == ["Report", "Finance"]
assert passage["source"] == "q2.pdf"
def test_search_documents_cannot_see_other_users_documents(client: TestClient) -> None:
"""User B searching for user A's content must get nothing back."""
_ingest(client, "alice-doc", "alice.pdf", ["The secret launch code is October."], "alice")
bob = client.post("/api/v1/documents/search", json={"query": "secret launch"}, headers={"X-User-Id": "bob"})
assert bob.status_code == 200
assert bob.json()["passages"] == []
alice = client.post("/api/v1/documents/search", json={"query": "secret launch"}, headers={"X-User-Id": "alice"})
assert alice.json()["passages"] != []
def test_search_documents_rejects_empty_query(client: TestClient) -> None:
response = client.post("/api/v1/documents/search", json={"query": ""}, headers=HEADERS)
assert response.status_code == 422
def test_search_documents_rejects_top_k_above_cap(client: TestClient) -> None:
response = client.post("/api/v1/documents/search", json={"query": "x", "topK": 51}, headers=HEADERS)
assert response.status_code == 422
def test_search_documents_rejects_missing_user_header(client: TestClient) -> None:
assert client.post("/api/v1/documents/search", json={"query": "x"}).status_code == 401
# ── POST /documents/ask ─────────────────────────────────────────────────
class StubKnowledgeAskAgent:
"""Stands in for KnowledgeAskAgent so route tests don't call a model."""
def __init__(self, response: AskDocumentsResponse) -> None:
self._response = response
self.calls: list[tuple[AskDocumentsRequest, list[PrincipalId]]] = []
async def ask(self, request: AskDocumentsRequest, principals: list[PrincipalId]) -> AskDocumentsResponse:
self.calls.append((request, principals))
return self._response
@pytest.fixture
def ask_agent() -> StubKnowledgeAskAgent:
return StubKnowledgeAskAgent(
AskDocumentsResponse(
answer="Revenue grew 12% (q2.pdf p.2).",
passages=[
DocumentPassage(
document_id=FileId("dp-doc"),
text="Revenue grew 12% in Q2.",
score=0.91,
page_start=2,
page_end=3,
heading_path=["Report", "Finance"],
source="q2.pdf",
)
],
)
)
@pytest.fixture
def ask_client(client: TestClient, ask_agent: StubKnowledgeAskAgent) -> Iterator[TestClient]:
app.dependency_overrides[get_knowledge_ask_agent] = lambda: ask_agent
try:
yield client
finally:
app.dependency_overrides.pop(get_knowledge_ask_agent, None)
def test_ask_documents_returns_answer_and_passages(ask_client: TestClient) -> None:
response = ask_client.post("/api/v1/documents/ask", json={"question": "How did revenue do?"}, headers=HEADERS)
assert response.status_code == 200
body = response.json()
assert body["answer"] == "Revenue grew 12% (q2.pdf p.2)."
assert body["passages"] == [
{
"documentId": "dp-doc",
"text": "Revenue grew 12% in Q2.",
"score": 0.91,
"pageStart": 2,
"pageEnd": 3,
"headingPath": ["Report", "Finance"],
"source": "q2.pdf",
}
]
def test_ask_documents_scopes_to_calling_user(ask_client: TestClient, ask_agent: StubKnowledgeAskAgent) -> None:
"""The route hands the agent exactly the caller's principal set."""
ask_client.post("/api/v1/documents/ask", json={"question": "anything"}, headers=HEADERS)
request, principals = ask_agent.calls[0]
assert principals == [PrincipalId(USER)]
assert request.top_k == 8
def test_ask_documents_rejects_empty_question(ask_client: TestClient) -> None:
response = ask_client.post("/api/v1/documents/ask", json={"question": ""}, headers=HEADERS)
assert response.status_code == 422
def test_ask_documents_rejects_top_k_above_cap(ask_client: TestClient) -> None:
response = ask_client.post("/api/v1/documents/ask", json={"question": "x", "topK": 21}, headers=HEADERS)
assert response.status_code == 422
def test_ask_documents_rejects_missing_user_header(ask_client: TestClient) -> None:
assert ask_client.post("/api/v1/documents/ask", json={"question": "x"}).status_code == 401
# ── GET /documents/stats ────────────────────────────────────────────────
def test_stats_on_empty_store_reports_zero(client: TestClient) -> None:
response = client.get("/api/v1/documents/stats", headers=HEADERS)
assert response.status_code == 200
body = response.json()
assert body["documents"] == 0
assert body["chunks"] == 0
assert body["backend"] in ("sqlite", "pgvector")
assert body["embeddingModel"]
def test_stats_counts_seeded_documents_across_owners(client: TestClient) -> None:
"""Stats are deployment-wide: both owners' content is counted."""
for owner, doc in (("alice", "doc-a"), ("bob", "doc-b")):
client.post(
"/api/v1/documents",
json={
"documentId": doc,
"source": f"{doc}.pdf",
"pageText": [{"pageNumber": 1, "text": "Some content for the stats endpoint."}],
"ownerId": owner,
"readPrincipals": [owner],
"expiresAt": None,
},
headers={"X-User-Id": owner},
)
response = client.get("/api/v1/documents/stats", headers=HEADERS)
assert response.status_code == 200
body = response.json()
assert body["documents"] == 2
assert body["chunks"] >= 2
def test_stats_rejects_missing_user_header(client: TestClient) -> None:
assert client.get("/api/v1/documents/stats").status_code == 401
def test_delete_document_only_affects_calling_user(client: TestClient) -> None:
"""Two users with the same document id: one user's delete must not remove the other's."""
alice_body = {
+91
View File
@@ -0,0 +1,91 @@
from __future__ import annotations
import pytest
from stirling.agents.knowledge_ask import KnowledgeAskAgent, format_passages, passage_from_hit
from stirling.config import AppSettings
from stirling.contracts import AskDocumentsRequest, DocumentPassage
from stirling.documents import CollectionSearchHit, Document, DocumentService, SearchResult, SqliteVecStore
from stirling.models import FileId, PrincipalId
from stirling.services import build_runtime
PRINCIPALS = [PrincipalId("test-user")]
def _hit(metadata: dict[str, str], text: str = "chunk text", score: float = 0.8) -> CollectionSearchHit:
return CollectionSearchHit(
collection=FileId("doc-1"),
result=SearchResult(document=Document(id="c1", text=text, metadata=metadata), score=score),
)
# ── passage_from_hit ────────────────────────────────────────────────────
def test_passage_from_hit_maps_docparse_metadata() -> None:
passage = passage_from_hit(
_hit({"source": "q2.pdf", "page_start": "2", "page_end": "3", "heading_path": "Report > Finance"})
)
assert passage.document_id == "doc-1"
assert passage.page_start == 2
assert passage.page_end == 3
assert passage.heading_path == ["Report", "Finance"]
assert passage.source == "q2.pdf"
def test_passage_from_hit_falls_back_to_page_number() -> None:
"""Plain page-text chunks: page_number fills both bounds, source drops the page suffix."""
passage = passage_from_hit(_hit({"source": "report.pdf:page:4", "page_number": "4"}))
assert passage.page_start == 4
assert passage.page_end == 4
assert passage.heading_path == []
assert passage.source == "report.pdf"
def test_passage_from_hit_tolerates_missing_and_bad_metadata() -> None:
passage = passage_from_hit(_hit({"page_start": "not-a-number"}))
assert passage.page_start is None
assert passage.page_end is None
assert passage.heading_path == []
assert passage.source is None
# ── format_passages ─────────────────────────────────────────────────────
def test_format_passages_includes_citation_handles() -> None:
passages = [
DocumentPassage(document_id=FileId("d1"), text="Alpha.", score=0.9, page_start=2, page_end=3, source="a.pdf"),
DocumentPassage(document_id=FileId("d2"), text="Beta.", score=0.5, page_start=7, page_end=7, source="b.pdf"),
DocumentPassage(document_id=FileId("d3"), text="Gamma.", score=0.4),
]
rendered = format_passages(passages)
assert "[Passage 1 | a.pdf p.2-3]\nAlpha." in rendered
assert "[Passage 2 | b.pdf p.7]\nBeta." in rendered
# No source or pages: fall back to the document id alone.
assert "[Passage 3 | d3]\nGamma." in rendered
# ── KnowledgeAskAgent ───────────────────────────────────────────────────
class _StubEmbedder:
"""Deterministic embeddings so the agent test needs no provider."""
async def embed_query(self, text: str) -> list[float]:
return [1.0, 0.0]
async def embed_documents(self, texts: list[str]) -> list[list[float]]:
return [[1.0, 0.0] for _ in texts]
@pytest.mark.anyio
async def test_ask_answers_plainly_when_nothing_retrieved(app_settings: AppSettings) -> None:
"""Empty retrieval short-circuits: no model call, honest not-found answer."""
documents = DocumentService(embedder=_StubEmbedder(), store=SqliteVecStore.ephemeral(), default_top_k=3) # type: ignore[arg-type]
runtime = build_runtime(app_settings, documents=documents)
agent = KnowledgeAskAgent(runtime)
response = await agent.ask(AskDocumentsRequest(question="What is the launch date?"), principals=PRINCIPALS)
assert response.passages == []
assert "couldn't find" in response.answer
+11 -9
View File
@@ -494,14 +494,14 @@ name = "cohere"
version = "7.0.4"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "fastavro" },
{ name = "httpx" },
{ name = "pydantic" },
{ name = "pydantic-core" },
{ name = "requests" },
{ name = "tokenizers" },
{ name = "types-requests" },
{ name = "typing-extensions" },
{ name = "fastavro", marker = "sys_platform != 'emscripten'" },
{ name = "httpx", marker = "sys_platform != 'emscripten'" },
{ name = "pydantic", marker = "sys_platform != 'emscripten'" },
{ name = "pydantic-core", marker = "sys_platform != 'emscripten'" },
{ name = "requests", marker = "sys_platform != 'emscripten'" },
{ name = "tokenizers", marker = "sys_platform != 'emscripten'" },
{ name = "types-requests", marker = "sys_platform != 'emscripten'" },
{ name = "typing-extensions", marker = "sys_platform != 'emscripten'" },
]
sdist = { url = "https://files.pythonhosted.org/packages/cf/3c/670631ee223d7b64d157dc3f309bf93bde65efe0bb1a8341d9b575f407d3/cohere-7.0.4.tar.gz", hash = "sha256:35b6a397d35ae6eafa1a02921f42c2a98309a990874533e5238efaf3426b6a21", size = 208794, upload-time = "2026-06-11T15:17:52.994Z" }
wheels = [
@@ -855,6 +855,7 @@ dependencies = [
{ name = "pydantic-ai" },
{ name = "pydantic-ai-slim", extra = ["voyageai"] },
{ name = "pydantic-settings" },
{ name = "python-docx" },
{ name = "python-dotenv" },
{ name = "sqlite-vec" },
{ name = "uvicorn" },
@@ -892,6 +893,7 @@ requires-dist = [
{ name = "pydantic-ai", specifier = ">=1.99.0,<2.0.0" },
{ name = "pydantic-ai-slim", extras = ["voyageai"], specifier = ">=1.99.0,<2.0.0" },
{ name = "pydantic-settings", specifier = ">=2.0.0" },
{ name = "python-docx", specifier = ">=1.1.2" },
{ name = "python-dotenv", specifier = ">=1.2.1" },
{ name = "sqlite-vec", specifier = ">=0.1.6" },
{ name = "torch", marker = "sys_platform == 'linux' and extra == 'docparse'", specifier = ">=2.6.0", index = "https://download.pytorch.org/whl/cpu" },
@@ -4360,7 +4362,7 @@ name = "types-requests"
version = "2.33.0.20260518"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "urllib3" },
{ name = "urllib3", marker = "sys_platform != 'emscripten'" },
]
sdist = { url = "https://files.pythonhosted.org/packages/e0/01/c5a19253fe1ac159159ddf9a3a07cec8bb5e486ec4d9002ad2821da0e5d2/types_requests-2.33.0.20260518.tar.gz", hash = "sha256:df7bd3bfe0ca8402dfb841e7d9be714bb5578203283d66d7dc4ef69343449a5e", size = 24752, upload-time = "2026-05-18T06:07:37.966Z" }
wheels = [
@@ -2986,6 +2986,38 @@ summary_one = "Ran 1 tool"
summary_other = "Ran {{count}} tools"
unknownTool = "Unknown tool"
[chunkDocument]
intro = "Turns a document into retrieval-ready chunks in three layers, so answers cite the right section instead of a random page."
processorCallout = "To index automatically, add the 'Index into knowledge base' step to an ingestion policy in the"
processorLink = "Processor"
submit = "Prepare chunks"
[chunkDocument.chunkSize]
label = "Chunk size (characters)"
[chunkDocument.error]
failed = "Failed to chunk document"
[chunkDocument.layers]
chunk = "Structure-aware chunks: each carries its heading breadcrumb and page range"
embed = "Ready to embed: exported as JSONL for any vector store"
parse = "Layout-aware parse: headings, paragraphs, and tables are recognized as structure"
[chunkDocument.mode]
advanced = "Advanced"
auto = "Auto"
basic = "Basic"
label = "Mode"
[chunkDocument.overlap]
label = "Overlap (characters)"
[chunkDocument.results]
title = "Chunks (JSONL)"
[chunkDocument.settings]
title = "Chunking settings"
[cloudBadge]
tooltip = "This operation will use your cloud credits"
@@ -4240,6 +4272,24 @@ upload = "Upload"
uploadFile = "Upload File"
uploadFiles = "Upload Files"
[fillTemplate]
hint = "The input file must be a .docx template (not a PDF). Each placeholder in the template is replaced with the matching JSON value."
intro = "Replaces the placeholders in a Word (.docx) template with your JSON data and returns the filled document - deterministic, no AI involved."
submit = "Fill template"
[fillTemplate.data]
invalid = "Enter a valid JSON object"
label = "Data (JSON)"
[fillTemplate.error]
failed = "Failed to fill template"
[fillTemplate.results]
title = "Filled document"
[fillTemplate.settings]
title = "Template data"
[firstLogin]
allFieldsRequired = "All fields are required"
changePassword = "Change Password"
@@ -4579,6 +4629,11 @@ desc = "Change document restrictions and permissions"
tags = "permissions,restrictions,rights,access control,allow,deny,printing,copying,editing,modify permissions,security settings,user rights"
title = "Change Permissions"
[home.chunkDocument]
desc = "Layout-aware parse to structure-aware chunks with heading breadcrumbs and page ranges, ready to embed"
tags = "chunk,RAG,prepare,split text,segments,embedding,vector,ingest,LLM,retrieval,JSONL,overlap,knowledge base,index"
title = "Prepare for RAG"
[home.compare]
desc = "Compares and shows the differences between 2 PDF Documents"
tags = "difference,compare,diff,compare PDFs,compare documents,find differences,show differences,changes,what changed,track changes,revisions,version compare,side by side,contrast,delta"
@@ -4639,6 +4694,11 @@ desc = "Extract specific pages from a PDF document"
tags = "pull,select,copy,extract,extract pages,get pages,pull out,save pages,export pages,copy pages,select pages,specific pages"
title = "Extract Pages"
[home.fillTemplate]
desc = "Fill a DOCX template's placeholders from JSON data"
tags = "template,DOCX,fill,merge fields,mail merge,generate document,placeholders,letters,contracts,Word"
title = "Fill Template"
[home.flatten]
desc = "Remove all interactive elements and forms from a PDF"
tags = "simplify,remove,interactive,flatten,flatten form,remove form fields,make static,finalize form,lock form,disable editing,convert to image,non-editable"
@@ -4692,6 +4752,11 @@ desc = "Merge multiple pages of a PDF document into a single page"
tags = "layout,arrange,combine,N-up,2-up,4-up,multiple per page,pages per sheet,layout pages,tile,grid layout,multi-page layout,combine on page,handout"
title = "Multi-Page Layout"
[home.parseDocument]
desc = "Layout-aware parsing to structured JSON or Markdown, with optional OCR"
tags = "parse,layout,structure,blocks,markdown,JSON,docling,OCR,scan,document understanding,convert"
title = "Parse Document"
[home.pdfCommentAgent]
desc = "Ask AI to annotate a PDF with sticky-note comments based on your prompt"
tags = "AI,agent,comment,annotate,sticky note,review,feedback,notes"
@@ -4805,6 +4870,11 @@ desc = "Adds signature to PDF by drawing, text or image"
tags = "signature,autograph,e-sign,electronic signature,digital signature,sign document,approval,signoff,authorize,endorse,ink signature,handwriting"
title = "Sign"
[home.smartSplit]
desc = "Split a PDF into sub-documents using a natural-language boundary rule"
tags = "split,smart,boundaries,separate,invoices,batches,content split,divide,rules,auto split"
title = "Smart Split"
[home.split]
desc = "Split PDFs into multiple documents"
tags = "divide,separate,break,split,extract pages,separate pages,divide document,break apart,separate files,unbind,split by page,divide by chapter"
@@ -5563,6 +5633,33 @@ title = "Page Ranges"
bullet1 = "<strong>all</strong> → selects all pages"
title = "Special Keywords"
[parseDocument]
intro = "Reads the document's layout - headings, paragraphs, tables - and turns it into clean structured JSON or Markdown you can feed to other systems."
submit = "Parse document"
[parseDocument.error]
failed = "Failed to parse document"
[parseDocument.mode]
advanced = "Advanced"
auto = "Auto"
basic = "Basic"
label = "Mode"
[parseDocument.outputFormat]
json = "JSON"
label = "Output format"
markdown = "Markdown"
[parseDocument.results]
title = "Parsed output"
[parseDocument.settings]
title = "Parse settings"
[parseDocument.withOcr]
label = "Apply OCR to scanned pages (recommended)"
[payg.activity]
docs = "docs"
empty = "No billable activity yet this period."
@@ -10367,6 +10464,26 @@ medium = "Medium"
small = "Small"
x-large = "X-Large"
[smartSplit]
intro = "Describe where sub-documents start in plain language and AI reads the content to find those boundaries - no page numbers needed."
submit = "Split document"
[smartSplit.error]
failed = "Failed to split document"
[smartSplit.maxParts]
label = "Maximum parts"
[smartSplit.results]
title = "Split documents"
[smartSplit.rule]
label = "Split rule"
placeholder = "e.g. Start a new document at every invoice header"
[smartSplit.settings]
title = "Split settings"
[split]
resultsTitle = "Split Results"
selectMethod = "Select a split method"
@@ -0,0 +1,165 @@
import { test, expect } from "@app/tests/helpers/stub-test-base";
import type { Page, Route } from "@playwright/test";
import path from "node:path";
/** DocParse walkthrough: the five workbench tools.
* Dumps PNGs to screenshots/docparse; light + dark per view, RTL spot checks. */
const SCREENSHOTS_DIR = path.resolve(process.cwd(), "screenshots", "docparse");
function shotPath(name: string): string {
return path.join(SCREENSHOTS_DIR, `${name}.png`);
}
async function settle(page: Page, ms = 400): Promise<void> {
await page.waitForTimeout(ms);
}
async function stubApis(page: Page): Promise<void> {
// Narrow fallbacks only: a blanket /api/v1/** would out-rank the stub
// fixture's own /auth/me route (last-registered wins) and break the session.
await page.route("**/api/v1/policies/**", (route: Route) =>
route.fulfill({ json: [] }),
);
await page.route("**/api/v1/proprietary/ui-data/**", (route: Route) =>
route.fulfill({ json: [] }),
);
// enableLogin true: the portal only builds a session when login mode is on
// (matches live behavior; with login off the portal shows its login screen).
const configPayload = {
appVersion: "test",
enableLogin: true,
isAdmin: true,
languages: ["en-US"],
defaultLocale: "en-US",
aiEngineEnabled: true,
docparseEnabled: true,
docparseAdvanced: true,
storageEnabled: false,
premiumEnabled: true,
runningProOrHigher: true,
};
await page.route("**/api/v1/config/app-config", (route: Route) =>
route.fulfill({ json: configPayload }),
);
// The auth layer decides login mode from public-config; keep it in sync.
await page.route("**/api/v1/config/public-config", (route: Route) =>
route.fulfill({
json: { enableLogin: true, languages: ["en-US"], defaultLocale: "en-US" },
}),
);
await page.route(
"**/api/v1/config/endpoints-availability**",
(route: Route) => route.fulfill({ json: {} }),
);
await page.route("**/api/v1/config/endpoint-enabled**", (route: Route) =>
route.fulfill({ json: { enabled: true } }),
);
// DocparseToolIntro probes live capabilities for its tier badges.
await page.route("**/api/v1/docparse/capabilities", (route: Route) =>
route.fulfill({
json: {
enabled: true,
mode: "auto",
advancedInstalled: true,
engineReachable: true,
doclingVersion: "2.116.0",
},
}),
);
}
async function enableDarkMode(page: Page): Promise<void> {
await page.addInitScript(() => {
localStorage.setItem("mantine-color-scheme", "dark");
localStorage.setItem("mantine-color-scheme-value", "dark");
});
await page.emulateMedia({ colorScheme: "dark" });
}
async function enableRtl(page: Page): Promise<void> {
await page.addInitScript(() => {
localStorage.setItem("i18nextLng", "ar-AR");
localStorage.setItem("stirling-language", "ar-AR");
localStorage.setItem("stirling-language-source", "user");
const applyDir = () => {
document.documentElement.setAttribute("dir", "rtl");
document.documentElement.setAttribute("lang", "ar-AR");
};
if (document.documentElement) applyDir();
else document.addEventListener("DOMContentLoaded", applyDir);
});
}
const TOOLS = [
{ id: "parseDocument", url: "/parse-document", waitText: /Parse/i },
{ id: "extractFields", url: "/extract-fields", waitText: /Extract/i },
{ id: "smartSplit", url: "/smart-split", waitText: /Split/i },
{ id: "chunkDocument", url: "/chunk-document", waitText: /Chunk/i },
{ id: "fillTemplate", url: "/fill-template", waitText: /Template|Fill/i },
];
async function openTool(page: Page, url: string): Promise<void> {
await page.goto(url, { waitUntil: "domcontentloaded" });
await expect(page.locator("body").first()).not.toBeEmpty();
// The tool panel is the left rail; give lazy chunks a moment.
await settle(page, 900);
}
test.describe("DocParse walkthrough", () => {
test.use({
autoGoto: false,
viewport: { width: 1600, height: 900 },
seedJwt: true,
});
// ─── Editor tools, light ──────────────────────────────────────────────────
for (const [i, tool] of TOOLS.entries()) {
test(`t${i}_${tool.id}_light`, async ({ page }) => {
await stubApis(page);
await openTool(page, tool.url);
await page.screenshot({ path: shotPath(`0${i + 1}_${tool.id}_light`) });
});
test(`t${i}_${tool.id}_dark`, async ({ page }) => {
await enableDarkMode(page);
await stubApis(page);
await openTool(page, tool.url);
await page.screenshot({ path: shotPath(`0${i + 1}_${tool.id}_dark`) });
});
}
// ─── Extract Fields with builder rows filled ─────────────────────────────
for (const theme of ["light", "dark"] as const) {
test(`extract_fields_populated_${theme}`, async ({ page }) => {
if (theme === "dark") await enableDarkMode(page);
await stubApis(page);
await openTool(page, "/extract-fields");
// Fill the first schema-builder row when present; tolerate layout drift.
const nameInput = page.getByPlaceholder(/name/i).first();
if (await nameInput.isVisible().catch(() => false)) {
await nameInput.fill("invoice_number");
const addButton = page.getByRole("button", { name: /add/i }).first();
if (await addButton.isVisible().catch(() => false)) {
await addButton.click();
const second = page.getByPlaceholder(/name/i).nth(1);
if (await second.isVisible().catch(() => false)) {
await second.fill("total_due");
}
}
}
await settle(page);
await page.screenshot({
path: shotPath(`06_extract_fields_populated_${theme}`),
});
});
}
// ─── RTL spot checks ──────────────────────────────────────────────────────
test("rtl_extract_fields", async ({ page }) => {
await enableRtl(page);
await stubApis(page);
await openTool(page, "/extract-fields");
await page.screenshot({ path: shotPath("11_extract_fields_rtl") });
});
});
@@ -0,0 +1,91 @@
import { useTranslation } from "react-i18next";
import { Anchor, List, NumberInput, Select, Stack, Text } from "@mantine/core";
import type { ToolAutomationSettingsProps } from "@app/hooks/tools/shared/toolOperationTypes";
import type { ChunkDocumentParameters } from "@app/hooks/tools/chunkDocument/useChunkDocumentParameters";
import type { DocparseMode } from "@app/hooks/tools/parseDocument/useParseDocumentParameters";
import DocparseToolIntro from "@app/components/tools/docparse/DocparseToolIntro";
const ChunkDocumentSettings = ({
parameters,
onParameterChange,
disabled,
}: ToolAutomationSettingsProps<ChunkDocumentParameters>) => {
const { t } = useTranslation();
return (
<Stack gap="sm">
<DocparseToolIntro
description={t(
"chunkDocument.intro",
"Turns a document into retrieval-ready chunks in three layers, so answers cite the right section instead of a random page.",
)}
aiBadge="layout"
/>
<List type="ordered" size="sm" spacing={4}>
<List.Item>
{t(
"chunkDocument.layers.parse",
"Layout-aware parse: headings, paragraphs, and tables are recognized as structure",
)}
</List.Item>
<List.Item>
{t(
"chunkDocument.layers.chunk",
"Structure-aware chunks: each carries its heading breadcrumb and page range",
)}
</List.Item>
<List.Item>
{t(
"chunkDocument.layers.embed",
"Ready to embed: exported as JSONL for any vector store",
)}
</List.Item>
</List>
<NumberInput
label={t("chunkDocument.chunkSize.label", "Chunk size (characters)")}
value={parameters.chunkSize}
onChange={(value) =>
onParameterChange("chunkSize", typeof value === "number" ? value : 0)
}
min={1}
disabled={disabled}
/>
<NumberInput
label={t("chunkDocument.overlap.label", "Overlap (characters)")}
value={parameters.overlap}
onChange={(value) =>
onParameterChange("overlap", typeof value === "number" ? value : 0)
}
min={0}
disabled={disabled}
/>
<Select
label={t("chunkDocument.mode.label", "Mode")}
value={parameters.mode}
onChange={(value) =>
onParameterChange("mode", (value ?? "auto") as DocparseMode)
}
data={[
{ value: "auto", label: t("chunkDocument.mode.auto", "Auto") },
{ value: "basic", label: t("chunkDocument.mode.basic", "Basic") },
{
value: "advanced",
label: t("chunkDocument.mode.advanced", "Advanced"),
},
]}
disabled={disabled}
/>
<Text size="xs" c="dimmed">
{t(
"chunkDocument.processorCallout",
"To index automatically, add the 'Index into knowledge base' step to an ingestion policy in the",
)}{" "}
<Anchor href="/processor/policies" size="xs">
{t("chunkDocument.processorLink", "Processor")}
</Anchor>
</Text>
</Stack>
);
};
export default ChunkDocumentSettings;
@@ -0,0 +1,54 @@
import { useTranslation } from "react-i18next";
import { Stack, Text, Textarea } from "@mantine/core";
import type { ToolAutomationSettingsProps } from "@app/hooks/tools/shared/toolOperationTypes";
import {
isJsonObjectString,
type FillTemplateParameters,
} from "@app/hooks/tools/fillTemplate/useFillTemplateParameters";
import DocparseToolIntro from "@app/components/tools/docparse/DocparseToolIntro";
const FillTemplateSettings = ({
parameters,
onParameterChange,
disabled,
}: ToolAutomationSettingsProps<FillTemplateParameters>) => {
const { t } = useTranslation();
const dataJson = parameters.dataJson;
const jsonError =
dataJson.trim().length > 0 && !isJsonObjectString(dataJson)
? t("fillTemplate.data.invalid", "Enter a valid JSON object")
: null;
return (
<Stack gap="sm">
<DocparseToolIntro
description={t(
"fillTemplate.intro",
"Replaces the placeholders in a Word (.docx) template with your JSON data and returns the filled document - deterministic, no AI involved.",
)}
showFallbackNote={false}
/>
<Text size="sm" c="dimmed">
{t(
"fillTemplate.hint",
"The input file must be a .docx template (not a PDF). Each placeholder in the template is replaced with the matching JSON value.",
)}
</Text>
<Textarea
label={t("fillTemplate.data.label", "Data (JSON)")}
placeholder='{"customer": "ACME Corp", "total": "128.00"}'
value={dataJson}
onChange={(event) =>
onParameterChange("dataJson", event.currentTarget.value)
}
minRows={5}
autosize
error={jsonError}
disabled={disabled}
/>
</Stack>
);
};
export default FillTemplateSettings;
@@ -0,0 +1,78 @@
import { useTranslation } from "react-i18next";
import { Checkbox, Select, Stack } from "@mantine/core";
import type { ToolAutomationSettingsProps } from "@app/hooks/tools/shared/toolOperationTypes";
import type {
DocparseMode,
ParseDocumentParameters,
} from "@app/hooks/tools/parseDocument/useParseDocumentParameters";
import DocparseToolIntro from "@app/components/tools/docparse/DocparseToolIntro";
const ParseDocumentSettings = ({
parameters,
onParameterChange,
disabled,
}: ToolAutomationSettingsProps<ParseDocumentParameters>) => {
const { t } = useTranslation();
return (
<Stack gap="sm">
<DocparseToolIntro
description={t(
"parseDocument.intro",
"Reads the document's layout - headings, paragraphs, tables - and turns it into clean structured JSON or Markdown you can feed to other systems.",
)}
aiBadge="layout"
/>
<Select
label={t("parseDocument.mode.label", "Mode")}
value={parameters.mode}
onChange={(value) =>
onParameterChange("mode", (value ?? "auto") as DocparseMode)
}
data={[
{ value: "auto", label: t("parseDocument.mode.auto", "Auto") },
{ value: "basic", label: t("parseDocument.mode.basic", "Basic") },
{
value: "advanced",
label: t("parseDocument.mode.advanced", "Advanced"),
},
]}
disabled={disabled}
/>
<Select
label={t("parseDocument.outputFormat.label", "Output format")}
value={parameters.outputFormat}
onChange={(value) =>
onParameterChange(
"outputFormat",
(value ?? "json") as ParseDocumentParameters["outputFormat"],
)
}
data={[
{
value: "json",
label: t("parseDocument.outputFormat.json", "JSON"),
},
{
value: "markdown",
label: t("parseDocument.outputFormat.markdown", "Markdown"),
},
]}
disabled={disabled}
/>
<Checkbox
label={t(
"parseDocument.withOcr.label",
"Apply OCR to scanned pages (recommended)",
)}
checked={parameters.withOcr}
onChange={(event) =>
onParameterChange("withOcr", event.currentTarget.checked)
}
disabled={disabled}
/>
</Stack>
);
};
export default ParseDocumentSettings;
@@ -0,0 +1,51 @@
import { useTranslation } from "react-i18next";
import { NumberInput, Stack, Textarea } from "@mantine/core";
import type { ToolAutomationSettingsProps } from "@app/hooks/tools/shared/toolOperationTypes";
import type { SmartSplitParameters } from "@app/hooks/tools/smartSplit/useSmartSplitParameters";
import DocparseToolIntro from "@app/components/tools/docparse/DocparseToolIntro";
const SmartSplitSettings = ({
parameters,
onParameterChange,
disabled,
}: ToolAutomationSettingsProps<SmartSplitParameters>) => {
const { t } = useTranslation();
return (
<Stack gap="sm">
<DocparseToolIntro
description={t(
"smartSplit.intro",
"Describe where sub-documents start in plain language and AI reads the content to find those boundaries - no page numbers needed.",
)}
aiBadge="llm"
/>
<Textarea
label={t("smartSplit.rule.label", "Split rule")}
placeholder={t(
"smartSplit.rule.placeholder",
"e.g. Start a new document at every invoice header",
)}
value={parameters.rule}
onChange={(event) =>
onParameterChange("rule", event.currentTarget.value)
}
minRows={3}
autosize
disabled={disabled}
/>
<NumberInput
label={t("smartSplit.maxParts.label", "Maximum parts")}
value={parameters.maxParts}
onChange={(value) =>
onParameterChange("maxParts", typeof value === "number" ? value : 1)
}
min={1}
max={100}
disabled={disabled}
/>
</Stack>
);
};
export default SmartSplitSettings;
@@ -9,8 +9,16 @@ import {
} from "@app/data/toolsTaxonomy";
import { asRegistryConfig } from "@app/hooks/tools/shared/toolOperationTypes";
import { useDocparseEnabled } from "@app/hooks/useDocparseEnabled";
import { parseDocumentOperationConfig } from "@app/hooks/tools/parseDocument/parseDocumentOperationConfig";
import { extractFieldsOperationConfig } from "@app/hooks/tools/extractFields/extractFieldsOperationConfig";
import { smartSplitOperationConfig } from "@app/hooks/tools/smartSplit/smartSplitOperationConfig";
import { chunkDocumentOperationConfig } from "@app/hooks/tools/chunkDocument/chunkDocumentOperationConfig";
import { fillTemplateOperationConfig } from "@app/hooks/tools/fillTemplate/fillTemplateOperationConfig";
import ParseDocument from "@app/tools/ParseDocument";
import ExtractFields from "@app/tools/ExtractFields";
import SmartSplit from "@app/tools/SmartSplit";
import ChunkDocument from "@app/tools/ChunkDocument";
import FillTemplate from "@app/tools/FillTemplate";
import { getSynonyms } from "@app/utils/toolSynonyms";
const toolIcon = (icon: string) => (
@@ -29,6 +37,23 @@ export function useProprietaryToolRegistry(): ProprietaryToolRegistry {
return useMemo(() => {
if (!docparseEnabled) return {} as ProprietaryToolRegistry;
return {
parseDocument: {
icon: toolIcon("quick-reference-all-outline-rounded"),
name: t("home.parseDocument.title", "Parse Document"),
component: ParseDocument,
description: t(
"home.parseDocument.desc",
"Layout-aware parsing to structured JSON or Markdown, with optional OCR",
),
categoryId: ToolCategoryId.STANDARD_TOOLS,
subcategoryId: SubcategoryId.DOCUMENT_INTELLIGENCE,
maxFiles: 1,
endpoints: ["parse-document"],
operationConfig: asRegistryConfig(parseDocumentOperationConfig),
automationSettings: null,
synonyms: getSynonyms(t, "parseDocument"),
versionStatus: "beta",
},
extractFields: {
icon: toolIcon("fact-check-outline-rounded"),
name: t("home.extractFields.title", "Extract Fields"),
@@ -46,6 +71,58 @@ export function useProprietaryToolRegistry(): ProprietaryToolRegistry {
synonyms: getSynonyms(t, "extractFields"),
versionStatus: "beta",
},
smartSplit: {
icon: toolIcon("content-cut-rounded"),
name: t("home.smartSplit.title", "Smart Split"),
component: SmartSplit,
description: t(
"home.smartSplit.desc",
"Split a PDF into sub-documents using a natural-language boundary rule",
),
categoryId: ToolCategoryId.STANDARD_TOOLS,
subcategoryId: SubcategoryId.DOCUMENT_INTELLIGENCE,
maxFiles: 1,
endpoints: ["smart-split"],
operationConfig: asRegistryConfig(smartSplitOperationConfig),
automationSettings: null,
synonyms: getSynonyms(t, "smartSplit"),
versionStatus: "beta",
},
chunkDocument: {
icon: toolIcon("layers-outline-rounded"),
name: t("home.chunkDocument.title", "Prepare for RAG"),
component: ChunkDocument,
description: t(
"home.chunkDocument.desc",
"Layout-aware parse to structure-aware chunks with heading breadcrumbs and page ranges, ready to embed",
),
categoryId: ToolCategoryId.STANDARD_TOOLS,
subcategoryId: SubcategoryId.DOCUMENT_INTELLIGENCE,
maxFiles: 1,
endpoints: ["chunk-document"],
operationConfig: asRegistryConfig(chunkDocumentOperationConfig),
automationSettings: null,
synonyms: getSynonyms(t, "chunkDocument"),
versionStatus: "beta",
},
fillTemplate: {
icon: toolIcon("assignment-outline-rounded"),
name: t("home.fillTemplate.title", "Fill Template"),
component: FillTemplate,
description: t(
"home.fillTemplate.desc",
"Fill a DOCX template's placeholders from JSON data",
),
categoryId: ToolCategoryId.STANDARD_TOOLS,
subcategoryId: SubcategoryId.DOCUMENT_INTELLIGENCE,
maxFiles: 1,
supportedFormats: ["docx"],
endpoints: ["fill-template"],
operationConfig: asRegistryConfig(fillTemplateOperationConfig),
automationSettings: null,
synonyms: getSynonyms(t, "fillTemplate"),
versionStatus: "beta",
},
} as ProprietaryToolRegistry;
}, [docparseEnabled, t]);
}, [t, docparseEnabled]);
}
@@ -0,0 +1,42 @@
import { describe, expect, test } from "vitest";
import {
buildChunkDocumentFormData,
chunksFromResponse,
chunksToJsonl,
} from "@app/hooks/tools/chunkDocument/chunkDocumentOperationConfig";
import { defaultParameters } from "@app/hooks/tools/chunkDocument/useChunkDocumentParameters";
describe("chunkDocument operation helpers", () => {
test("accepts both bare-array and wrapped chunk responses", () => {
const chunks = [{ text: "a" }, { text: "b" }];
expect(chunksFromResponse(chunks)).toEqual(chunks);
expect(chunksFromResponse({ chunks })).toEqual(chunks);
expect(chunksFromResponse({ nope: true })).toEqual([]);
expect(chunksFromResponse(null)).toEqual([]);
});
test("emits one JSON document per JSONL line", () => {
const jsonl = chunksToJsonl([
{ text: "a", page: 1 },
{ text: "b", page: 2 },
]);
const lines = jsonl.split("\n");
expect(lines).toHaveLength(2);
expect(lines.map((line) => JSON.parse(line))).toEqual([
{ text: "a", page: 1 },
{ text: "b", page: 2 },
]);
});
test("builds the multipart form the backend contract expects", () => {
const file = new File(["pdf"], "doc.pdf", { type: "application/pdf" });
const form = buildChunkDocumentFormData(
{ ...defaultParameters, chunkSize: 800, overlap: 50, mode: "advanced" },
file,
);
expect(form.get("fileInput")).toBe(file);
expect(form.get("chunkSize")).toBe("800");
expect(form.get("overlap")).toBe("50");
expect(form.get("mode")).toBe("advanced");
});
});
@@ -0,0 +1,64 @@
import apiClient from "@app/services/apiClient";
import {
defineCustomTool,
CustomProcessorResult,
} from "@app/hooks/tools/shared/toolOperationTypes";
import { deriveName } from "@app/hooks/tools/shared/docparseFilenames";
import {
ChunkDocumentParameters,
defaultParameters,
} from "@app/hooks/tools/chunkDocument/useChunkDocumentParameters";
export const CHUNK_DOCUMENT_ENDPOINT = "/api/v1/docparse/chunk-document";
export const buildChunkDocumentFormData = (
parameters: ChunkDocumentParameters,
file: File,
): FormData => {
const formData = new FormData();
formData.append("fileInput", file);
formData.append("chunkSize", String(parameters.chunkSize));
formData.append("overlap", String(parameters.overlap));
formData.append("mode", parameters.mode);
return formData;
};
/** The chunks array, whether the backend returns it bare or wrapped. */
export function chunksFromResponse(data: unknown): unknown[] {
if (Array.isArray(data)) return data;
const wrapped = (data as { chunks?: unknown[] } | null)?.chunks;
return Array.isArray(wrapped) ? wrapped : [];
}
/** JSON chunks -> one JSONL line per chunk, the standard RAG-ingest shape. */
export function chunksToJsonl(chunks: unknown[]): string {
return chunks.map((chunk) => JSON.stringify(chunk)).join("\n");
}
const processChunkDocument = async (
parameters: ChunkDocumentParameters,
files: File[],
): Promise<CustomProcessorResult> => {
if (files.length === 0) return { files: [] };
const [inputFile] = files;
const response = await apiClient.post<unknown>(
CHUNK_DOCUMENT_ENDPOINT,
buildChunkDocumentFormData(parameters, inputFile),
);
const resultFile = new File(
[chunksToJsonl(chunksFromResponse(response.data))],
deriveName(inputFile.name, ".chunks.jsonl"),
{ type: "application/x-ndjson" },
);
return { files: [resultFile] };
};
export const chunkDocumentOperationConfig =
defineCustomTool<ChunkDocumentParameters>({
operationType: "chunkDocument",
endpoint: CHUNK_DOCUMENT_ENDPOINT,
customProcessor: processChunkDocument,
defaultParameters,
});
@@ -0,0 +1,15 @@
import { useTranslation } from "react-i18next";
import { useToolOperation } from "@app/hooks/tools/shared/useToolOperation";
import { createStandardErrorHandler } from "@app/utils/toolErrorHandler";
import { chunkDocumentOperationConfig } from "@app/hooks/tools/chunkDocument/chunkDocumentOperationConfig";
export const useChunkDocumentOperation = () => {
const { t } = useTranslation();
return useToolOperation({
...chunkDocumentOperationConfig,
getErrorMessage: createStandardErrorHandler(
t("chunkDocument.error.failed", "Failed to chunk document"),
),
});
};
@@ -0,0 +1,34 @@
import { BaseParameters } from "@app/types/parameters";
import {
useBaseParameters,
BaseParametersHook,
} from "@app/hooks/tools/shared/useBaseParameters";
import type { DocparseMode } from "@app/hooks/tools/parseDocument/useParseDocumentParameters";
export interface ChunkDocumentParameters extends BaseParameters {
/** Target chunk size in characters. */
chunkSize: number;
/** Characters of overlap carried between neighbouring chunks. */
overlap: number;
mode: DocparseMode;
}
export const defaultParameters: ChunkDocumentParameters = {
chunkSize: 1000,
overlap: 100,
mode: "auto",
};
export type ChunkDocumentParametersHook =
BaseParametersHook<ChunkDocumentParameters>;
export const useChunkDocumentParameters = (): ChunkDocumentParametersHook => {
return useBaseParameters({
defaultParameters,
endpointName: "chunk-document",
validateFn: (params) =>
params.chunkSize > 0 &&
params.overlap >= 0 &&
params.overlap < params.chunkSize,
});
};
@@ -0,0 +1,55 @@
import apiClient from "@app/services/apiClient";
import {
defineCustomTool,
CustomProcessorResult,
} from "@app/hooks/tools/shared/toolOperationTypes";
import { deriveName } from "@app/hooks/tools/shared/docparseFilenames";
import {
FillTemplateParameters,
defaultParameters,
} from "@app/hooks/tools/fillTemplate/useFillTemplateParameters";
export const FILL_TEMPLATE_ENDPOINT = "/api/v1/docparse/fill-template";
const DOCX_TYPE =
"application/vnd.openxmlformats-officedocument.wordprocessingml.document";
export const buildFillTemplateFormData = (
parameters: FillTemplateParameters,
file: File,
): FormData => {
const formData = new FormData();
formData.append("templateFile", file);
formData.append("data", parameters.dataJson.trim());
return formData;
};
/** POST the DOCX template + data; the filled DOCX comes straight back. */
const processFillTemplate = async (
parameters: FillTemplateParameters,
files: File[],
): Promise<CustomProcessorResult> => {
if (files.length === 0) return { files: [] };
const [templateFile] = files;
const response = await apiClient.post<Blob>(
FILL_TEMPLATE_ENDPOINT,
buildFillTemplateFormData(parameters, templateFile),
{ responseType: "blob" },
);
const resultFile = new File(
[response.data],
deriveName(templateFile.name, "-filled.docx"),
{ type: DOCX_TYPE },
);
return { files: [resultFile] };
};
export const fillTemplateOperationConfig =
defineCustomTool<FillTemplateParameters>({
operationType: "fillTemplate",
endpoint: FILL_TEMPLATE_ENDPOINT,
customProcessor: processFillTemplate,
defaultParameters,
});
@@ -0,0 +1,15 @@
import { useTranslation } from "react-i18next";
import { useToolOperation } from "@app/hooks/tools/shared/useToolOperation";
import { createStandardErrorHandler } from "@app/utils/toolErrorHandler";
import { fillTemplateOperationConfig } from "@app/hooks/tools/fillTemplate/fillTemplateOperationConfig";
export const useFillTemplateOperation = () => {
const { t } = useTranslation();
return useToolOperation({
...fillTemplateOperationConfig,
getErrorMessage: createStandardErrorHandler(
t("fillTemplate.error.failed", "Failed to fill template"),
),
});
};
@@ -0,0 +1,37 @@
import { BaseParameters } from "@app/types/parameters";
import {
useBaseParameters,
BaseParametersHook,
} from "@app/hooks/tools/shared/useBaseParameters";
export interface FillTemplateParameters extends BaseParameters {
/** JSON object whose keys fill the template's placeholders. */
dataJson: string;
}
export const defaultParameters: FillTemplateParameters = {
dataJson: "",
};
/** True when the text parses to a plain JSON object. */
export function isJsonObjectString(text: string): boolean {
try {
const parsed = JSON.parse(text);
return (
typeof parsed === "object" && parsed !== null && !Array.isArray(parsed)
);
} catch {
return false;
}
}
export type FillTemplateParametersHook =
BaseParametersHook<FillTemplateParameters>;
export const useFillTemplateParameters = (): FillTemplateParametersHook => {
return useBaseParameters({
defaultParameters,
endpointName: "fill-template",
validateFn: (params) => isJsonObjectString(params.dataJson),
});
};
@@ -0,0 +1,56 @@
import apiClient from "@app/services/apiClient";
import {
defineCustomTool,
CustomProcessorResult,
} from "@app/hooks/tools/shared/toolOperationTypes";
import { deriveName } from "@app/hooks/tools/shared/docparseFilenames";
import {
ParseDocumentParameters,
defaultParameters,
} from "@app/hooks/tools/parseDocument/useParseDocumentParameters";
// Not part of the generated ToolEndpoint union; DocParse is an optional addon.
export const PARSE_DOCUMENT_ENDPOINT = "/api/v1/docparse/parse-document";
export const buildParseDocumentFormData = (
parameters: ParseDocumentParameters,
file: File,
): FormData => {
const formData = new FormData();
formData.append("fileInput", file);
formData.append("mode", parameters.mode);
formData.append("withOcr", String(parameters.withOcr));
formData.append("outputFormat", parameters.outputFormat);
return formData;
};
/** POST the PDF; wrap the JSON or markdown result as a downloadable file. */
const processParseDocument = async (
parameters: ParseDocumentParameters,
files: File[],
): Promise<CustomProcessorResult> => {
if (files.length === 0) return { files: [] };
const [inputFile] = files;
const response = await apiClient.post<Blob>(
PARSE_DOCUMENT_ENDPOINT,
buildParseDocumentFormData(parameters, inputFile),
{ responseType: "blob" },
);
const isMarkdown = parameters.outputFormat === "markdown";
const resultFile = new File(
[response.data],
deriveName(inputFile.name, isMarkdown ? ".md" : ".parsed.json"),
{ type: isMarkdown ? "text/markdown" : "application/json" },
);
return { files: [resultFile] };
};
export const parseDocumentOperationConfig =
defineCustomTool<ParseDocumentParameters>({
operationType: "parseDocument",
endpoint: PARSE_DOCUMENT_ENDPOINT,
customProcessor: processParseDocument,
defaultParameters,
});
@@ -0,0 +1,15 @@
import { useTranslation } from "react-i18next";
import { useToolOperation } from "@app/hooks/tools/shared/useToolOperation";
import { createStandardErrorHandler } from "@app/utils/toolErrorHandler";
import { parseDocumentOperationConfig } from "@app/hooks/tools/parseDocument/parseDocumentOperationConfig";
export const useParseDocumentOperation = () => {
const { t } = useTranslation();
return useToolOperation({
...parseDocumentOperationConfig,
getErrorMessage: createStandardErrorHandler(
t("parseDocument.error.failed", "Failed to parse document"),
),
});
};
@@ -0,0 +1,30 @@
import { BaseParameters } from "@app/types/parameters";
import {
useBaseParameters,
BaseParametersHook,
} from "@app/hooks/tools/shared/useBaseParameters";
export type DocparseMode = "auto" | "basic" | "advanced";
export interface ParseDocumentParameters extends BaseParameters {
mode: DocparseMode;
outputFormat: "json" | "markdown";
withOcr: boolean;
}
// withOcr defaults true to match the API's default behaviour.
export const defaultParameters: ParseDocumentParameters = {
mode: "auto",
outputFormat: "json",
withOcr: true,
};
export type ParseDocumentParametersHook =
BaseParametersHook<ParseDocumentParameters>;
export const useParseDocumentParameters = (): ParseDocumentParametersHook => {
return useBaseParameters({
defaultParameters,
endpointName: "parse-document",
});
};
@@ -0,0 +1,60 @@
import JSZip from "jszip";
import apiClient from "@app/services/apiClient";
import {
defineCustomTool,
CustomProcessorResult,
} from "@app/hooks/tools/shared/toolOperationTypes";
import {
SmartSplitParameters,
defaultParameters,
} from "@app/hooks/tools/smartSplit/useSmartSplitParameters";
export const SMART_SPLIT_ENDPOINT = "/api/v1/docparse/smart-split";
export const buildSmartSplitFormData = (
parameters: SmartSplitParameters,
file: File,
): FormData => {
const formData = new FormData();
formData.append("fileInput", file);
formData.append("rule", parameters.rule.trim());
formData.append("maxParts", String(parameters.maxParts));
return formData;
};
/** POST the PDF + rule; unpack the returned ZIP into the sub-PDFs. */
const processSmartSplit = async (
parameters: SmartSplitParameters,
files: File[],
): Promise<CustomProcessorResult> => {
if (files.length === 0) return { files: [] };
const [inputFile] = files;
const response = await apiClient.post<Blob>(
SMART_SPLIT_ENDPOINT,
buildSmartSplitFormData(parameters, inputFile),
{ responseType: "blob" },
);
const zip = await JSZip.loadAsync(response.data);
const entries = Object.values(zip.files).filter((entry) => !entry.dir);
const parts = await Promise.all(
entries.map(
async (entry) =>
new File([await entry.async("blob")], entry.name.split("/").pop()!, {
type: "application/pdf",
}),
),
);
// One input becomes N parts, so filename-based input mapping cannot apply.
return { files: parts, consumedAllInputs: true };
};
export const smartSplitOperationConfig = defineCustomTool<SmartSplitParameters>(
{
operationType: "smartSplit",
endpoint: SMART_SPLIT_ENDPOINT,
customProcessor: processSmartSplit,
defaultParameters,
},
);
@@ -0,0 +1,15 @@
import { useTranslation } from "react-i18next";
import { useToolOperation } from "@app/hooks/tools/shared/useToolOperation";
import { createStandardErrorHandler } from "@app/utils/toolErrorHandler";
import { smartSplitOperationConfig } from "@app/hooks/tools/smartSplit/smartSplitOperationConfig";
export const useSmartSplitOperation = () => {
const { t } = useTranslation();
return useToolOperation({
...smartSplitOperationConfig,
getErrorMessage: createStandardErrorHandler(
t("smartSplit.error.failed", "Failed to split document"),
),
});
};
@@ -0,0 +1,27 @@
import { BaseParameters } from "@app/types/parameters";
import {
useBaseParameters,
BaseParametersHook,
} from "@app/hooks/tools/shared/useBaseParameters";
export interface SmartSplitParameters extends BaseParameters {
/** Natural-language boundary rule, e.g. "split at each new invoice". */
rule: string;
maxParts: number;
}
export const defaultParameters: SmartSplitParameters = {
rule: "",
maxParts: 10,
};
export type SmartSplitParametersHook = BaseParametersHook<SmartSplitParameters>;
export const useSmartSplitParameters = (): SmartSplitParametersHook => {
return useBaseParameters({
defaultParameters,
endpointName: "smart-split",
validateFn: (params) =>
params.rule.trim().length > 0 && params.maxParts > 0,
});
};
@@ -0,0 +1,55 @@
import { useTranslation } from "react-i18next";
import { createToolFlow } from "@app/components/tools/shared/createToolFlow";
import { useBaseTool } from "@app/hooks/tools/shared/useBaseTool";
import type { BaseToolProps } from "@app/types/tool";
import ChunkDocumentSettings from "@app/components/tools/docparse/ChunkDocumentSettings";
import { useChunkDocumentParameters } from "@app/hooks/tools/chunkDocument/useChunkDocumentParameters";
import { useChunkDocumentOperation } from "@app/hooks/tools/chunkDocument/useChunkDocumentOperation";
const ChunkDocument = (props: BaseToolProps) => {
const { t } = useTranslation();
const base = useBaseTool(
"chunkDocument",
useChunkDocumentParameters,
useChunkDocumentOperation,
props,
);
return createToolFlow({
files: {
selectedFiles: base.selectedFiles,
isCollapsed: base.hasResults,
},
steps: [
{
title: t("chunkDocument.settings.title", "Chunking settings"),
isCollapsed: false,
content: (
<ChunkDocumentSettings
parameters={base.params.parameters}
onParameterChange={base.params.updateParameter}
disabled={base.endpointLoading}
/>
),
},
],
executeButton: {
text: t("chunkDocument.submit", "Prepare chunks"),
isVisible: !base.hasResults,
loadingText: t("loading"),
onClick: base.handleExecute,
endpointEnabled: base.endpointEnabled,
paramsValid: base.params.validateParameters(),
},
review: {
isVisible: base.hasResults,
operation: base.operation,
title: t("chunkDocument.results.title", "Chunks (JSONL)"),
onFileClick: base.handleThumbnailClick,
onUndo: base.handleUndo,
},
});
};
export default ChunkDocument;
@@ -0,0 +1,55 @@
import { useTranslation } from "react-i18next";
import { createToolFlow } from "@app/components/tools/shared/createToolFlow";
import { useBaseTool } from "@app/hooks/tools/shared/useBaseTool";
import type { BaseToolProps } from "@app/types/tool";
import FillTemplateSettings from "@app/components/tools/docparse/FillTemplateSettings";
import { useFillTemplateParameters } from "@app/hooks/tools/fillTemplate/useFillTemplateParameters";
import { useFillTemplateOperation } from "@app/hooks/tools/fillTemplate/useFillTemplateOperation";
const FillTemplate = (props: BaseToolProps) => {
const { t } = useTranslation();
const base = useBaseTool(
"fillTemplate",
useFillTemplateParameters,
useFillTemplateOperation,
props,
);
return createToolFlow({
files: {
selectedFiles: base.selectedFiles,
isCollapsed: base.hasResults,
},
steps: [
{
title: t("fillTemplate.settings.title", "Template data"),
isCollapsed: false,
content: (
<FillTemplateSettings
parameters={base.params.parameters}
onParameterChange={base.params.updateParameter}
disabled={base.endpointLoading}
/>
),
},
],
executeButton: {
text: t("fillTemplate.submit", "Fill template"),
isVisible: !base.hasResults,
loadingText: t("loading"),
onClick: base.handleExecute,
endpointEnabled: base.endpointEnabled,
paramsValid: base.params.validateParameters(),
},
review: {
isVisible: base.hasResults,
operation: base.operation,
title: t("fillTemplate.results.title", "Filled document"),
onFileClick: base.handleThumbnailClick,
onUndo: base.handleUndo,
},
});
};
export default FillTemplate;
@@ -0,0 +1,55 @@
import { useTranslation } from "react-i18next";
import { createToolFlow } from "@app/components/tools/shared/createToolFlow";
import { useBaseTool } from "@app/hooks/tools/shared/useBaseTool";
import type { BaseToolProps } from "@app/types/tool";
import ParseDocumentSettings from "@app/components/tools/docparse/ParseDocumentSettings";
import { useParseDocumentParameters } from "@app/hooks/tools/parseDocument/useParseDocumentParameters";
import { useParseDocumentOperation } from "@app/hooks/tools/parseDocument/useParseDocumentOperation";
const ParseDocument = (props: BaseToolProps) => {
const { t } = useTranslation();
const base = useBaseTool(
"parseDocument",
useParseDocumentParameters,
useParseDocumentOperation,
props,
);
return createToolFlow({
files: {
selectedFiles: base.selectedFiles,
isCollapsed: base.hasResults,
},
steps: [
{
title: t("parseDocument.settings.title", "Parse settings"),
isCollapsed: false,
content: (
<ParseDocumentSettings
parameters={base.params.parameters}
onParameterChange={base.params.updateParameter}
disabled={base.endpointLoading}
/>
),
},
],
executeButton: {
text: t("parseDocument.submit", "Parse document"),
isVisible: !base.hasResults,
loadingText: t("loading"),
onClick: base.handleExecute,
endpointEnabled: base.endpointEnabled,
paramsValid: base.params.validateParameters(),
},
review: {
isVisible: base.hasResults,
operation: base.operation,
title: t("parseDocument.results.title", "Parsed output"),
onFileClick: base.handleThumbnailClick,
onUndo: base.handleUndo,
},
});
};
export default ParseDocument;
@@ -0,0 +1,55 @@
import { useTranslation } from "react-i18next";
import { createToolFlow } from "@app/components/tools/shared/createToolFlow";
import { useBaseTool } from "@app/hooks/tools/shared/useBaseTool";
import type { BaseToolProps } from "@app/types/tool";
import SmartSplitSettings from "@app/components/tools/docparse/SmartSplitSettings";
import { useSmartSplitParameters } from "@app/hooks/tools/smartSplit/useSmartSplitParameters";
import { useSmartSplitOperation } from "@app/hooks/tools/smartSplit/useSmartSplitOperation";
const SmartSplit = (props: BaseToolProps) => {
const { t } = useTranslation();
const base = useBaseTool(
"smartSplit",
useSmartSplitParameters,
useSmartSplitOperation,
props,
);
return createToolFlow({
files: {
selectedFiles: base.selectedFiles,
isCollapsed: base.hasResults,
},
steps: [
{
title: t("smartSplit.settings.title", "Split settings"),
isCollapsed: false,
content: (
<SmartSplitSettings
parameters={base.params.parameters}
onParameterChange={base.params.updateParameter}
disabled={base.endpointLoading}
/>
),
},
],
executeButton: {
text: t("smartSplit.submit", "Split document"),
isVisible: !base.hasResults,
loadingText: t("loading"),
onClick: base.handleExecute,
endpointEnabled: base.endpointEnabled,
paramsValid: base.params.validateParameters(),
},
review: {
isVisible: base.hasResults,
operation: base.operation,
title: t("smartSplit.results.title", "Split documents"),
onFileClick: base.handleThumbnailClick,
onUndo: base.handleUndo,
},
});
};
export default SmartSplit;
@@ -5,7 +5,13 @@
*/
// The DocParse tool family; visible only when the backend reports docparseEnabled.
export const PROPRIETARY_REGULAR_TOOL_IDS = ["extractFields"] as const;
export const PROPRIETARY_REGULAR_TOOL_IDS = [
"parseDocument",
"extractFields",
"smartSplit",
"chunkDocument",
"fillTemplate",
] as const;
// "ai-workflow" is a generic marker stamped onto files produced by the agents
// chat orchestrator (which may invoke one or more underlying tools). Lives here