mirror of
https://github.com/Stirling-Tools/Stirling-PDF.git
synced 2026-09-03 21:30:14 +03:00
Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0d70133e64 | ||
|
|
07de12746c | ||
|
|
469e7c499c |
@@ -433,11 +433,19 @@ public class EndpointConfiguration {
|
||||
addEndpointToGroup("Automation", "automate"); // Alias for handleData (user-friendly name)
|
||||
addEndpointToGroup("Automation", "pipeline");
|
||||
|
||||
// Adding endpoints to "DocParse" group (ingestion: chunk + index + export)
|
||||
addEndpointToGroup("DocParse", "rag-ingest");
|
||||
addEndpointToGroup("DocParse", "extract-tables");
|
||||
// Adding endpoints to "DocParse" group (parsing, splitting, chunking, extraction,
|
||||
// templating)
|
||||
addEndpointToGroup("DocParse", "parse-document");
|
||||
addEndpointToGroup("DocParse", "extract-fields");
|
||||
addEndpointToGroup("DocParse", "smart-split");
|
||||
addEndpointToGroup("DocParse", "chunk-document");
|
||||
addEndpointToGroup("DocParse", "rag-ingest");
|
||||
addEndpointToGroup("DocParse", "rag-documents");
|
||||
addEndpointToGroup("DocParse", "rag-search");
|
||||
addEndpointToGroup("DocParse", "rag-ask");
|
||||
addEndpointToGroup("DocParse", "extract-tables");
|
||||
addEndpointToGroup("DocParse", "suggest-schema");
|
||||
addEndpointToGroup("DocParse", "fill-template");
|
||||
|
||||
// Adding endpoints to "DeveloperTools" group
|
||||
addEndpointToGroup("DeveloperTools", "show-javascript");
|
||||
|
||||
+229
@@ -4,22 +4,32 @@ import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.StringWriter;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.util.Base64;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.zip.ZipEntry;
|
||||
import java.util.zip.ZipOutputStream;
|
||||
|
||||
import org.apache.commons.csv.CSVFormat;
|
||||
import org.apache.commons.csv.CSVPrinter;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.springframework.core.io.ByteArrayResource;
|
||||
import org.springframework.core.io.Resource;
|
||||
import org.springframework.http.HttpHeaders;
|
||||
import org.springframework.http.HttpStatus;
|
||||
import org.springframework.http.MediaType;
|
||||
import org.springframework.http.ResponseEntity;
|
||||
import org.springframework.web.bind.annotation.GetMapping;
|
||||
import org.springframework.web.bind.annotation.ModelAttribute;
|
||||
import org.springframework.web.bind.annotation.PostMapping;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
import org.springframework.web.bind.annotation.RequestMapping;
|
||||
import org.springframework.web.bind.annotation.RequestParam;
|
||||
import org.springframework.web.bind.annotation.RestController;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
import org.springframework.web.server.ResponseStatusException;
|
||||
|
||||
import io.swagger.v3.oas.annotations.Operation;
|
||||
import io.swagger.v3.oas.annotations.tags.Tag;
|
||||
@@ -29,19 +39,34 @@ import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.common.annotations.AutoJobPostMapping;
|
||||
import stirling.software.common.enumeration.ResourceWeight;
|
||||
import stirling.software.common.service.CustomPDFDocumentFactory;
|
||||
import stirling.software.common.util.FormUtils;
|
||||
import stirling.software.common.util.GeneralUtils;
|
||||
import stirling.software.common.util.TempFile;
|
||||
import stirling.software.common.util.TempFileManager;
|
||||
import stirling.software.common.util.WebResponseUtils;
|
||||
import stirling.software.proprietary.model.api.docparse.ChunkDocumentApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.ExtractFieldsApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.ExtractTablesApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.ParseDocumentApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.RagAskApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.RagIngestApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.RagSearchApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.SmartSplitApiRequest;
|
||||
import stirling.software.proprietary.model.api.docparse.SuggestSchemaApiRequest;
|
||||
import stirling.software.proprietary.model.docparse.ChunkDocumentResponse;
|
||||
import stirling.software.proprietary.model.docparse.DocChunk;
|
||||
import stirling.software.proprietary.model.docparse.DocTable;
|
||||
import stirling.software.proprietary.model.docparse.DocparseCapabilitiesView;
|
||||
import stirling.software.proprietary.model.docparse.DocparseMode;
|
||||
import stirling.software.proprietary.model.docparse.ExtractFieldsResponse;
|
||||
import stirling.software.proprietary.model.docparse.ExtractTablesResponse;
|
||||
import stirling.software.proprietary.model.docparse.FillDocxResponse;
|
||||
import stirling.software.proprietary.model.docparse.ParseDocumentResponse;
|
||||
import stirling.software.proprietary.model.docparse.RagIngestResponse;
|
||||
import stirling.software.proprietary.model.docparse.RagStatsView;
|
||||
import stirling.software.proprietary.model.docparse.SmartSplitResponse;
|
||||
import stirling.software.proprietary.model.docparse.SplitPart;
|
||||
import stirling.software.proprietary.model.docparse.SuggestSchemaResponse;
|
||||
import stirling.software.proprietary.service.AiToolResponseHeaders;
|
||||
import stirling.software.proprietary.service.DocParseService;
|
||||
@@ -67,7 +92,14 @@ public class DocParseController {
|
||||
|
||||
private static final MediaType CSV = MediaType.parseMediaType("text/csv");
|
||||
|
||||
private static final MediaType MARKDOWN = MediaType.parseMediaType("text/markdown");
|
||||
private static final MediaType DOCX =
|
||||
MediaType.parseMediaType(
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document");
|
||||
|
||||
private final DocParseService docParseService;
|
||||
private final CustomPDFDocumentFactory pdfDocumentFactory;
|
||||
private final TempFileManager tempFileManager;
|
||||
private final ObjectMapper objectMapper;
|
||||
|
||||
@AutoJobPostMapping(
|
||||
@@ -192,6 +224,123 @@ public class DocParseController {
|
||||
docParseService.suggestSchema(request.getFileInput(), request.getMaxFields()));
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/parse-document",
|
||||
resourceWeight = ResourceWeight.XLARGE_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Parse a document into structured blocks, tables, and markdown",
|
||||
description =
|
||||
"Parses the PDF into layout blocks, tables, and a markdown rendering. The"
|
||||
+ " basic tier reads the text layer; the advanced tier (docparse addon)"
|
||||
+ " adds OCR, real table structure, and bounding boxes."
|
||||
+ " Input:PDF Output:JSON Type:SISO")
|
||||
public ResponseEntity<?> parseDocument(@ModelAttribute ParseDocumentApiRequest request)
|
||||
throws IOException {
|
||||
ParseDocumentResponse result =
|
||||
docParseService.parse(
|
||||
request.getFileInput(),
|
||||
DocparseMode.fromWire(request.getMode()),
|
||||
request.isWithOcr());
|
||||
if ("markdown".equalsIgnoreCase(request.getOutputFormat())) {
|
||||
return WebResponseUtils.bytesToWebResponse(
|
||||
result.markdown().getBytes(StandardCharsets.UTF_8),
|
||||
outputName(request.getFileInput(), "_parsed.md"),
|
||||
MARKDOWN);
|
||||
}
|
||||
return ResponseEntity.ok(result);
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/smart-split",
|
||||
resourceWeight = ResourceWeight.LARGE_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Split a document at content-derived boundaries",
|
||||
description =
|
||||
"Asks the engine where sub-documents start (per the natural-language rule) and"
|
||||
+ " returns a ZIP with one PDF per part, named from the part labels."
|
||||
+ " Input:PDF Output:ZIP-PDF Type:SIMO")
|
||||
public ResponseEntity<Resource> smartSplit(@ModelAttribute SmartSplitApiRequest request)
|
||||
throws IOException {
|
||||
MultipartFile file = request.getFileInput();
|
||||
SmartSplitResponse split =
|
||||
docParseService.split(file, request.getRule(), request.getMaxParts());
|
||||
if (split.parts().isEmpty()) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.UNPROCESSABLE_ENTITY,
|
||||
"The split rule produced no parts for this document");
|
||||
}
|
||||
TempFile zipTempFile = tempFileManager.createManagedTempFile(".zip");
|
||||
try {
|
||||
try (TempFile sourceTempFile = new TempFile(tempFileManager, ".pdf")) {
|
||||
Files.copy(
|
||||
file.getInputStream(),
|
||||
sourceTempFile.getPath(),
|
||||
StandardCopyOption.REPLACE_EXISTING);
|
||||
try (ZipOutputStream zipOut =
|
||||
new ZipOutputStream(Files.newOutputStream(zipTempFile.getPath()))) {
|
||||
writeParts(sourceTempFile, split.parts(), zipOut);
|
||||
}
|
||||
}
|
||||
return WebResponseUtils.zipFileToWebResponse(
|
||||
zipTempFile,
|
||||
GeneralUtils.generateFilename(file.getOriginalFilename(), "_split.zip"));
|
||||
} catch (Exception e) {
|
||||
zipTempFile.close();
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/chunk-document",
|
||||
resourceWeight = ResourceWeight.MEDIUM_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Chunk a document for RAG",
|
||||
description =
|
||||
"Splits the document text into overlapping chunks with page spans and (advanced"
|
||||
+ " tier) heading breadcrumbs. Input:PDF Output:JSON Type:SISO")
|
||||
public ResponseEntity<ChunkDocumentResponse> chunkDocument(
|
||||
@ModelAttribute ChunkDocumentApiRequest request) throws IOException {
|
||||
return ResponseEntity.ok(
|
||||
docParseService.chunk(
|
||||
request.getFileInput(),
|
||||
request.getChunkSize(),
|
||||
request.getOverlap(),
|
||||
DocparseMode.fromWire(request.getMode())));
|
||||
}
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/fill-template",
|
||||
resourceWeight = ResourceWeight.SMALL_WEIGHT)
|
||||
@Operation(
|
||||
summary = "Fill a DOCX template with JSON data",
|
||||
description =
|
||||
"Replaces the template's placeholders with values from the JSON object and"
|
||||
+ " returns the filled DOCX. Replacement counts and missing keys ride"
|
||||
+ " the X-Stirling-Tool-Report header."
|
||||
+ " Input:DOCX Output:DOCX Type:SISO")
|
||||
public ResponseEntity<Resource> fillTemplate(
|
||||
@RequestParam("templateFile") MultipartFile templateFile,
|
||||
@RequestParam("data") String data)
|
||||
throws IOException {
|
||||
FillDocxResponse result = docParseService.fillDocx(templateFile, data);
|
||||
byte[] filled = Base64.getDecoder().decode(result.docxBase64());
|
||||
HttpHeaders headers = new HttpHeaders();
|
||||
headers.setContentType(DOCX);
|
||||
headers.setContentDispositionFormData(
|
||||
"attachment",
|
||||
GeneralUtils.generateFilename(templateFile.getOriginalFilename(), "_filled.docx"));
|
||||
headers.setContentLength(filled.length);
|
||||
headers.set(
|
||||
AiToolResponseHeaders.TOOL_REPORT,
|
||||
objectMapper.writeValueAsString(
|
||||
new FillDocxResponse("", result.replaced(), result.missing())));
|
||||
return ResponseEntity.ok().headers(headers).body(new ByteArrayResource(filled));
|
||||
}
|
||||
|
||||
@GetMapping("/capabilities")
|
||||
@Operation(
|
||||
summary = "DocParse capability summary",
|
||||
@@ -224,6 +373,52 @@ public class DocParseController {
|
||||
CSV);
|
||||
}
|
||||
|
||||
@GetMapping("/rag-stats")
|
||||
@Operation(
|
||||
summary = "RAG store statistics",
|
||||
description =
|
||||
"The engine's document-store totals (backend, documents, chunks, embedding"
|
||||
+ " model) merged with the DocParse capability fields. Answers with"
|
||||
+ " zeros and engineReachable=false when the engine is down.")
|
||||
public ResponseEntity<RagStatsView> ragStats() {
|
||||
return ResponseEntity.ok(docParseService.ragStats());
|
||||
}
|
||||
|
||||
@GetMapping("/rag-documents")
|
||||
@Operation(
|
||||
summary = "List documents in the RAG store",
|
||||
description =
|
||||
"Engine passthrough of the caller-visible indexed documents (documentId,"
|
||||
+ " source, chunk count).")
|
||||
public ResponseEntity<String> ragDocuments() throws IOException {
|
||||
return jsonPassthrough(docParseService.ragDocuments());
|
||||
}
|
||||
|
||||
@PostMapping(value = "/rag-search", consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
@Operation(
|
||||
summary = "Semantic search over the RAG store",
|
||||
description =
|
||||
"Searches the caller-visible indexed documents and returns the top passages"
|
||||
+ " with scores, page spans, and heading breadcrumbs.")
|
||||
public ResponseEntity<String> ragSearch(@RequestBody RagSearchApiRequest request)
|
||||
throws IOException {
|
||||
return jsonPassthrough(docParseService.ragSearch(request.getQuery(), request.getTopK()));
|
||||
}
|
||||
|
||||
@PostMapping(value = "/rag-ask", consumes = MediaType.APPLICATION_JSON_VALUE)
|
||||
@Operation(
|
||||
summary = "Ask a question over the RAG store",
|
||||
description =
|
||||
"Answers the question from the caller-visible indexed documents and returns"
|
||||
+ " the answer with its supporting passages.")
|
||||
public ResponseEntity<String> ragAsk(@RequestBody RagAskApiRequest request) throws IOException {
|
||||
return jsonPassthrough(docParseService.ragAsk(request.getQuestion(), request.getTopK()));
|
||||
}
|
||||
|
||||
private static ResponseEntity<String> jsonPassthrough(String engineJson) {
|
||||
return ResponseEntity.ok().contentType(MediaType.APPLICATION_JSON).body(engineJson);
|
||||
}
|
||||
|
||||
/** Original + requested corpus files in one ZIP, so destinations receive them together. */
|
||||
private byte[] exportZip(
|
||||
String fileName, byte[] original, RagIngestResponse result, RagIngestApiRequest request)
|
||||
@@ -279,6 +474,40 @@ public class DocParseController {
|
||||
return dot > 0 ? fileName.substring(0, dot) : fileName;
|
||||
}
|
||||
|
||||
private void writeParts(TempFile sourceTempFile, List<SplitPart> parts, ZipOutputStream zipOut)
|
||||
throws IOException {
|
||||
for (int i = 0; i < parts.size(); i++) {
|
||||
SplitPart part = parts.get(i);
|
||||
// Load per part and remove pages outside the range: avoids the PDFBox cross-document
|
||||
// addPage pitfalls while keeping shared resources intact.
|
||||
try (PDDocument partDoc = pdfDocumentFactory.load(sourceTempFile.getFile())) {
|
||||
int pageCount = partDoc.getNumberOfPages();
|
||||
int start = Math.clamp(part.startPage(), 1, pageCount);
|
||||
int end = Math.clamp(part.endPage(), start, pageCount);
|
||||
for (int p = pageCount - 1; p >= 0; p--) {
|
||||
int pageNumber = p + 1;
|
||||
if (pageNumber < start || pageNumber > end) {
|
||||
partDoc.removePage(p);
|
||||
}
|
||||
}
|
||||
FormUtils.pruneOrphanedFormFields(partDoc);
|
||||
zipOut.putNextEntry(new ZipEntry(partEntryName(i, part)));
|
||||
partDoc.save(zipOut);
|
||||
zipOut.closeEntry();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private static String partEntryName(int index, SplitPart part) {
|
||||
String label = part.label() == null ? "" : part.label().trim();
|
||||
String sanitized = label.replaceAll("[^A-Za-z0-9 ._-]", "_").replaceAll("\\s+", "_");
|
||||
if (sanitized.isBlank() || sanitized.chars().allMatch(c -> c == '_' || c == '.')) {
|
||||
sanitized = "part";
|
||||
}
|
||||
// Index prefix keeps entries unique even when labels repeat.
|
||||
return String.format(Locale.ROOT, "%02d_%s.pdf", index + 1, sanitized);
|
||||
}
|
||||
|
||||
private static String tablesToCsv(List<DocTable> tables) throws IOException {
|
||||
CSVFormat format = CSVFormat.EXCEL.builder().setEscape('"').build();
|
||||
StringWriter writer = new StringWriter();
|
||||
|
||||
+27
@@ -0,0 +1,27 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class ChunkDocumentApiRequest extends PDFFile {
|
||||
|
||||
@Schema(description = "Target chunk size in characters (64-32768)", defaultValue = "512")
|
||||
private int chunkSize = 512;
|
||||
|
||||
@Schema(
|
||||
description = "Overlap between adjacent chunks in characters (0-4096)",
|
||||
defaultValue = "64")
|
||||
private int overlap = 64;
|
||||
|
||||
@Schema(
|
||||
description = "Tier to use: 'auto' picks per document, or force 'basic'/'advanced'",
|
||||
allowableValues = {"auto", "basic", "advanced"},
|
||||
defaultValue = "auto")
|
||||
private String mode = "auto";
|
||||
}
|
||||
+30
@@ -0,0 +1,30 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class ParseDocumentApiRequest extends PDFFile {
|
||||
|
||||
@Schema(
|
||||
description = "Tier to use: 'auto' picks per document, or force 'basic'/'advanced'",
|
||||
allowableValues = {"auto", "basic", "advanced"},
|
||||
defaultValue = "auto")
|
||||
private String mode = "auto";
|
||||
|
||||
@Schema(
|
||||
description = "Apply OCR when parsing scanned pages (advanced tier only)",
|
||||
defaultValue = "true")
|
||||
private boolean withOcr = true;
|
||||
|
||||
@Schema(
|
||||
description = "Response format: full JSON result or the markdown rendering only",
|
||||
allowableValues = {"json", "markdown"},
|
||||
defaultValue = "json")
|
||||
private String outputFormat = "json";
|
||||
}
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
|
||||
@Data
|
||||
public class RagAskApiRequest {
|
||||
|
||||
@Schema(
|
||||
description = "Question to answer from the indexed documents",
|
||||
requiredMode = Schema.RequiredMode.REQUIRED)
|
||||
private String question;
|
||||
|
||||
@Schema(description = "Number of passages to ground the answer on (1-20)", defaultValue = "5")
|
||||
private int topK = 5;
|
||||
}
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
|
||||
@Data
|
||||
public class RagSearchApiRequest {
|
||||
|
||||
@Schema(
|
||||
description = "Natural-language search query",
|
||||
requiredMode = Schema.RequiredMode.REQUIRED)
|
||||
private String query;
|
||||
|
||||
@Schema(description = "Number of passages to return (1-50)", defaultValue = "10")
|
||||
private int topK = 10;
|
||||
}
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
package stirling.software.proprietary.model.api.docparse;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class SmartSplitApiRequest extends PDFFile {
|
||||
|
||||
@Schema(
|
||||
description = "Natural-language boundary rule, e.g. 'split where a new invoice starts'",
|
||||
requiredMode = Schema.RequiredMode.REQUIRED)
|
||||
private String rule;
|
||||
|
||||
@Schema(description = "Maximum number of parts to produce (1-500)", defaultValue = "50")
|
||||
private int maxParts = 50;
|
||||
}
|
||||
+14
@@ -0,0 +1,14 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import stirling.software.proprietary.model.api.ai.AiPageText;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/docparse/chunk}. */
|
||||
public record ChunkDocumentRequest(
|
||||
String fileName,
|
||||
List<AiPageText> pages,
|
||||
String contentBase64,
|
||||
int chunkSize,
|
||||
int overlap,
|
||||
DocparseMode mode) {}
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Engine response for {@code POST /api/v1/docparse/chunk}. */
|
||||
public record ChunkDocumentResponse(DocparseTier mode, List<DocChunk> chunks) {
|
||||
|
||||
public ChunkDocumentResponse {
|
||||
chunks = chunks == null ? List.of() : chunks;
|
||||
}
|
||||
}
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* One layout block. {@code bbox} is [x0, y0, x1, y1] normalized to 0..1 with a top-left origin;
|
||||
* {@code null} in basic tier (no layout model ran). Mirrors {@code docparse.py DocBlock}.
|
||||
*/
|
||||
public record DocBlock(String type, String text, int page, List<Double> bbox, Double confidence) {}
|
||||
+5
@@ -0,0 +1,5 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** Engine response for {@code GET /api/v1/documents/stats}: the RAG document store totals. */
|
||||
public record DocumentStoreStats(
|
||||
String backend, long documents, long chunks, String embeddingModel) {}
|
||||
+6
@@ -0,0 +1,6 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import tools.jackson.databind.JsonNode;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/docparse/fill-docx}. */
|
||||
public record FillDocxRequest(String templateBase64, JsonNode data) {}
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Engine response for {@code POST /api/v1/docparse/fill-docx}. */
|
||||
public record FillDocxResponse(String docxBase64, int replaced, List<String> missing) {
|
||||
|
||||
public FillDocxResponse {
|
||||
missing = missing == null ? List.of() : missing;
|
||||
}
|
||||
}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/docparse/parse}. */
|
||||
public record ParseDocumentRequest(String fileName, String contentBase64, boolean withOcr) {}
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Engine response for {@code POST /api/v1/docparse/parse}; also produced by the Java basic tier.
|
||||
*/
|
||||
public record ParseDocumentResponse(
|
||||
DocparseTier mode,
|
||||
int pages,
|
||||
List<DocBlock> blocks,
|
||||
List<DocTable> tables,
|
||||
String markdown,
|
||||
boolean ocrApplied) {
|
||||
|
||||
public ParseDocumentResponse {
|
||||
blocks = blocks == null ? List.of() : blocks;
|
||||
tables = tables == null ? List.of() : tables;
|
||||
markdown = markdown == null ? "" : markdown;
|
||||
}
|
||||
}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/documents/ask}: grounded Q&A over the RAG store. */
|
||||
public record RagAskRequest(String question, int topK) {}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/documents/search}: semantic search over the RAG store. */
|
||||
public record RagSearchRequest(String query, int topK) {}
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/**
|
||||
* Merged RAG store view served by {@code GET /api/v1/docparse/rag-stats} (Java side): the engine's
|
||||
* document-store totals plus the cached DocParse capability fields. When the engine is unreachable
|
||||
* the totals are zero and {@code engineReachable} is false.
|
||||
*/
|
||||
public record RagStatsView(
|
||||
String backend,
|
||||
long documents,
|
||||
long chunks,
|
||||
String embeddingModel,
|
||||
boolean advancedInstalled,
|
||||
String doclingVersion,
|
||||
boolean engineReachable) {}
|
||||
+9
@@ -0,0 +1,9 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import stirling.software.proprietary.model.api.ai.AiPageText;
|
||||
|
||||
/** Engine request for {@code POST /api/v1/docparse/split}. */
|
||||
public record SmartSplitRequest(
|
||||
String fileName, String rule, List<AiPageText> pages, int maxParts) {}
|
||||
+11
@@ -0,0 +1,11 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Engine response for {@code POST /api/v1/docparse/split}. */
|
||||
public record SmartSplitResponse(List<SplitPart> parts) {
|
||||
|
||||
public SmartSplitResponse {
|
||||
parts = parts == null ? List.of() : parts;
|
||||
}
|
||||
}
|
||||
+4
@@ -0,0 +1,4 @@
|
||||
package stirling.software.proprietary.model.docparse;
|
||||
|
||||
/** One sub-document page range (1-based, inclusive). Mirrors {@code docparse.py SplitPart}. */
|
||||
public record SplitPart(int startPage, int endPage, String label, double confidence) {}
|
||||
+186
-2
@@ -20,16 +20,29 @@ import stirling.software.common.model.ApplicationProperties;
|
||||
import stirling.software.common.service.CustomPDFDocumentFactory;
|
||||
import stirling.software.common.service.UserServiceInterface;
|
||||
import stirling.software.proprietary.model.api.ai.AiPageText;
|
||||
import stirling.software.proprietary.model.docparse.ChunkDocumentRequest;
|
||||
import stirling.software.proprietary.model.docparse.ChunkDocumentResponse;
|
||||
import stirling.software.proprietary.model.docparse.DocBlock;
|
||||
import stirling.software.proprietary.model.docparse.DocparseCapabilities;
|
||||
import stirling.software.proprietary.model.docparse.DocparseCapabilitiesView;
|
||||
import stirling.software.proprietary.model.docparse.DocparseMode;
|
||||
import stirling.software.proprietary.model.docparse.DocparseTier;
|
||||
import stirling.software.proprietary.model.docparse.DocumentStoreStats;
|
||||
import stirling.software.proprietary.model.docparse.ExtractFieldsRequest;
|
||||
import stirling.software.proprietary.model.docparse.ExtractFieldsResponse;
|
||||
import stirling.software.proprietary.model.docparse.ExtractTablesRequest;
|
||||
import stirling.software.proprietary.model.docparse.ExtractTablesResponse;
|
||||
import stirling.software.proprietary.model.docparse.FillDocxRequest;
|
||||
import stirling.software.proprietary.model.docparse.FillDocxResponse;
|
||||
import stirling.software.proprietary.model.docparse.ParseDocumentRequest;
|
||||
import stirling.software.proprietary.model.docparse.ParseDocumentResponse;
|
||||
import stirling.software.proprietary.model.docparse.RagAskRequest;
|
||||
import stirling.software.proprietary.model.docparse.RagIngestRequest;
|
||||
import stirling.software.proprietary.model.docparse.RagIngestResponse;
|
||||
import stirling.software.proprietary.model.docparse.RagSearchRequest;
|
||||
import stirling.software.proprietary.model.docparse.RagStatsView;
|
||||
import stirling.software.proprietary.model.docparse.SmartSplitRequest;
|
||||
import stirling.software.proprietary.model.docparse.SmartSplitResponse;
|
||||
import stirling.software.proprietary.model.docparse.SuggestSchemaRequest;
|
||||
import stirling.software.proprietary.model.docparse.SuggestSchemaResponse;
|
||||
|
||||
@@ -46,10 +59,18 @@ import tools.jackson.databind.ObjectMapper;
|
||||
@Service
|
||||
public class DocParseService {
|
||||
|
||||
private static final String RAG_INGEST_ENDPOINT = "/api/v1/docparse/rag-ingest";
|
||||
private static final String TABLES_ENDPOINT = "/api/v1/docparse/tables";
|
||||
private static final String PARSE_ENDPOINT = "/api/v1/docparse/parse";
|
||||
private static final String EXTRACT_ENDPOINT = "/api/v1/docparse/extract";
|
||||
private static final String SPLIT_ENDPOINT = "/api/v1/docparse/split";
|
||||
private static final String CHUNK_ENDPOINT = "/api/v1/docparse/chunk";
|
||||
private static final String TABLES_ENDPOINT = "/api/v1/docparse/tables";
|
||||
private static final String FILL_DOCX_ENDPOINT = "/api/v1/docparse/fill-docx";
|
||||
private static final String SUGGEST_SCHEMA_ENDPOINT = "/api/v1/docparse/suggest-schema";
|
||||
private static final String RAG_INGEST_ENDPOINT = "/api/v1/docparse/rag-ingest";
|
||||
private static final String DOCUMENT_STATS_ENDPOINT = "/api/v1/documents/stats";
|
||||
private static final String DOCUMENT_LIST_ENDPOINT = "/api/v1/documents/list";
|
||||
private static final String DOCUMENT_SEARCH_ENDPOINT = "/api/v1/documents/search";
|
||||
private static final String DOCUMENT_ASK_ENDPOINT = "/api/v1/documents/ask";
|
||||
|
||||
/** Below this average of extractable chars per page the document is treated as scanned. */
|
||||
static final int SCANNED_AVG_CHARS_PER_PAGE = 100;
|
||||
@@ -104,6 +125,72 @@ public class DocParseService {
|
||||
capabilities.doclingVersion());
|
||||
}
|
||||
|
||||
public ParseDocumentResponse parse(
|
||||
MultipartFile file, DocparseMode requestedMode, boolean withOcr) throws IOException {
|
||||
requireEnabled();
|
||||
DocparseTier tier;
|
||||
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
|
||||
tier =
|
||||
resolveTier(
|
||||
requestedMode,
|
||||
capabilityService.capabilities(),
|
||||
false,
|
||||
looksScanned(document));
|
||||
if (tier == DocparseTier.BASIC) {
|
||||
return basicParse(document);
|
||||
}
|
||||
}
|
||||
ParseDocumentRequest request =
|
||||
new ParseDocumentRequest(fileName(file), encodeBase64(file), withOcr);
|
||||
String responseJson =
|
||||
aiEngineClient.postLongRunning(
|
||||
PARSE_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
|
||||
return objectMapper.readValue(responseJson, ParseDocumentResponse.class);
|
||||
}
|
||||
|
||||
public SmartSplitResponse split(MultipartFile file, String rule, int maxParts)
|
||||
throws IOException {
|
||||
requireEnabled();
|
||||
if (rule == null || rule.isBlank()) {
|
||||
throw new ResponseStatusException(HttpStatus.BAD_REQUEST, "A split rule is required");
|
||||
}
|
||||
List<AiPageText> pages;
|
||||
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
|
||||
pages = extractPages(document);
|
||||
}
|
||||
SmartSplitRequest request =
|
||||
new SmartSplitRequest(fileName(file), rule, pages, Math.clamp(maxParts, 1, 500));
|
||||
String responseJson =
|
||||
aiEngineClient.post(
|
||||
SPLIT_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
|
||||
return objectMapper.readValue(responseJson, SmartSplitResponse.class);
|
||||
}
|
||||
|
||||
public ChunkDocumentResponse chunk(
|
||||
MultipartFile file, int chunkSize, int overlap, DocparseMode mode) throws IOException {
|
||||
requireEnabled();
|
||||
List<AiPageText> pages;
|
||||
DocparseTier tier;
|
||||
try (PDDocument document = pdfDocumentFactory.load(file, true)) {
|
||||
pages = extractPages(document);
|
||||
tier =
|
||||
resolveTier(
|
||||
mode, capabilityService.capabilities(), false, looksScanned(document));
|
||||
}
|
||||
ChunkDocumentRequest request =
|
||||
new ChunkDocumentRequest(
|
||||
fileName(file),
|
||||
pages,
|
||||
tier == DocparseTier.ADVANCED ? encodeBase64(file) : null,
|
||||
Math.clamp(chunkSize, 64, 32_768),
|
||||
Math.clamp(overlap, 0, 4_096),
|
||||
toMode(tier));
|
||||
String responseJson =
|
||||
aiEngineClient.postLongRunning(
|
||||
CHUNK_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
|
||||
return objectMapper.readValue(responseJson, ChunkDocumentResponse.class);
|
||||
}
|
||||
|
||||
/**
|
||||
* Chunk, embed, and index the document into the engine's RAG store, and/or echo the parsed
|
||||
* content back for corpus export. Text extraction and tier routing happen here; the engine
|
||||
@@ -267,6 +354,68 @@ public class DocParseService {
|
||||
return node;
|
||||
}
|
||||
|
||||
private static void requireNonBlank(String value, String fieldName) {
|
||||
if (value == null || value.isBlank()) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.BAD_REQUEST, "'" + fieldName + "' is required");
|
||||
}
|
||||
}
|
||||
|
||||
/** Engine RAG store totals merged with the cached capability fields; graceful when down. */
|
||||
public RagStatsView ragStats() {
|
||||
DocparseCapabilities capabilities = capabilityService.capabilities();
|
||||
try {
|
||||
// The engine's documents routes are user-gated; an id-less probe 401s
|
||||
// and would read as "engine offline" in the UI.
|
||||
String json = aiEngineClient.get(DOCUMENT_STATS_ENDPOINT, currentUserId());
|
||||
DocumentStoreStats stats = objectMapper.readValue(json, DocumentStoreStats.class);
|
||||
return new RagStatsView(
|
||||
stats.backend(),
|
||||
stats.documents(),
|
||||
stats.chunks(),
|
||||
stats.embeddingModel(),
|
||||
capabilities.advancedInstalled(),
|
||||
capabilities.doclingVersion(),
|
||||
true);
|
||||
} catch (Exception e) {
|
||||
log.debug("RAG stats probe failed: {}", e.getMessage());
|
||||
return new RagStatsView(
|
||||
null,
|
||||
0,
|
||||
0,
|
||||
null,
|
||||
capabilities.advancedInstalled(),
|
||||
capabilities.doclingVersion(),
|
||||
false);
|
||||
}
|
||||
}
|
||||
|
||||
/** Engine document-list passthrough; X-User-Id scopes it to the caller's ACLs. */
|
||||
public String ragDocuments() throws IOException {
|
||||
requireEnabled();
|
||||
return aiEngineClient.get(DOCUMENT_LIST_ENDPOINT, currentUserId());
|
||||
}
|
||||
|
||||
/** Semantic-search passthrough over the caller-visible RAG documents. */
|
||||
public String ragSearch(String query, int topK) throws IOException {
|
||||
requireEnabled();
|
||||
requireNonBlank(query, "query");
|
||||
RagSearchRequest request = new RagSearchRequest(query, Math.clamp(topK, 1, 50));
|
||||
return aiEngineClient.post(
|
||||
DOCUMENT_SEARCH_ENDPOINT,
|
||||
objectMapper.writeValueAsString(request),
|
||||
currentUserId());
|
||||
}
|
||||
|
||||
/** Grounded-answer passthrough; long-running because local models answer slowly. */
|
||||
public String ragAsk(String question, int topK) throws IOException {
|
||||
requireEnabled();
|
||||
requireNonBlank(question, "question");
|
||||
RagAskRequest request = new RagAskRequest(question, Math.clamp(topK, 1, 20));
|
||||
return aiEngineClient.postLongRunning(
|
||||
DOCUMENT_ASK_ENDPOINT, objectMapper.writeValueAsString(request), currentUserId());
|
||||
}
|
||||
|
||||
/**
|
||||
* The settings mode wins when stricter: a settings {@code basic} always forces basic, a
|
||||
* settings {@code advanced} upgrades everything except an explicit basic request.
|
||||
@@ -327,6 +476,41 @@ public class DocParseService {
|
||||
return objectMapper.readValue(responseJson, ExtractTablesResponse.class);
|
||||
}
|
||||
|
||||
public FillDocxResponse fillDocx(MultipartFile templateFile, String dataJson)
|
||||
throws IOException {
|
||||
requireEnabled();
|
||||
JsonNode data = parseJsonObject(dataJson, "data");
|
||||
FillDocxRequest request =
|
||||
new FillDocxRequest(
|
||||
Base64.getEncoder().encodeToString(templateFile.getBytes()), data);
|
||||
String responseJson =
|
||||
aiEngineClient.post(
|
||||
FILL_DOCX_ENDPOINT,
|
||||
objectMapper.writeValueAsString(request),
|
||||
currentUserId());
|
||||
return objectMapper.readValue(responseJson, FillDocxResponse.class);
|
||||
}
|
||||
|
||||
/** Basic tier parse: PDFBox text layer only, one paragraph block per non-blank page. */
|
||||
ParseDocumentResponse basicParse(PDDocument document) throws IOException {
|
||||
int pageCount = document.getNumberOfPages();
|
||||
List<DocBlock> blocks = new ArrayList<>();
|
||||
StringBuilder markdown = new StringBuilder();
|
||||
for (int page = 1; page <= pageCount; page++) {
|
||||
String text = pdfContentExtractor.extractPageTextRaw(document, page);
|
||||
if (text == null || text.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
blocks.add(new DocBlock("paragraph", text, page, null, null));
|
||||
if (!markdown.isEmpty()) {
|
||||
markdown.append("\n\n");
|
||||
}
|
||||
markdown.append(text);
|
||||
}
|
||||
return new ParseDocumentResponse(
|
||||
DocparseTier.BASIC, pageCount, blocks, List.of(), markdown.toString(), false);
|
||||
}
|
||||
|
||||
/** Extract per-page text for the engine, capped by the shared aiEngine limits. */
|
||||
List<AiPageText> extractPages(PDDocument document) throws IOException {
|
||||
ApplicationProperties.AiEngine.Limits limits =
|
||||
|
||||
@@ -14,6 +14,8 @@ dependencies = [
|
||||
"pydantic-ai-slim[voyageai]>=1.99.0,<2.0.0",
|
||||
"pydantic-settings>=2.0.0",
|
||||
"python-dotenv>=1.2.1",
|
||||
# Small (MIT) and always installed: DOCX template filling needs no addon.
|
||||
"python-docx>=1.1.2",
|
||||
"sqlite-vec>=0.1.6",
|
||||
"uvicorn>=0.35.0",
|
||||
"opentelemetry-sdk>=1.39.0",
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from .document_classifier import DocumentClassifierAgent
|
||||
from .execution import ExecutionPlanningAgent
|
||||
from .knowledge_ask import KnowledgeAskAgent
|
||||
from .orchestrator import OrchestratorAgent
|
||||
from .pdf_create import PdfCreateAgent
|
||||
from .pdf_edit import PdfEditAgent, PdfEditParameterSelector, PdfEditPlanSelection
|
||||
@@ -12,6 +13,7 @@ from .user_spec import UserSpecAgent
|
||||
__all__ = [
|
||||
"DocumentClassifierAgent",
|
||||
"ExecutionPlanningAgent",
|
||||
"KnowledgeAskAgent",
|
||||
"OrchestratorAgent",
|
||||
"PdfCreateAgent",
|
||||
"PdfEditAgent",
|
||||
|
||||
@@ -0,0 +1,134 @@
|
||||
"""Grounded Q&A over the caller's stored documents.
|
||||
|
||||
Retrieval runs the same cross-collection search as ``POST /documents/search``;
|
||||
one smart-model pass then answers only from the retrieved passages, citing
|
||||
document and page inline. No retrieval hit means a plain "not found" answer.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from pydantic import Field
|
||||
from pydantic_ai import Agent
|
||||
|
||||
from stirling.agents.output_mode import output_retries, structured_output
|
||||
from stirling.contracts import AskDocumentsRequest, AskDocumentsResponse, DocumentPassage
|
||||
from stirling.documents import CollectionSearchHit
|
||||
from stirling.documents.service import PAGE_NUMBER_METADATA_KEY
|
||||
from stirling.models import ApiModel, PrincipalId
|
||||
from stirling.services import AppRuntime
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Metadata keys written by docparse rag-ingest (_chunk_metadata) for structure-aware chunks.
|
||||
_PAGE_START_KEY = "page_start"
|
||||
_PAGE_END_KEY = "page_end"
|
||||
_HEADING_PATH_KEY = "heading_path"
|
||||
_HEADING_PATH_SEPARATOR = " > "
|
||||
|
||||
_SYSTEM_PROMPT = (
|
||||
"You answer questions using ONLY the numbered passages you are given.\n"
|
||||
"\n"
|
||||
"Rules:\n"
|
||||
"- Every statement must come from the passages. Never use outside knowledge, never guess.\n"
|
||||
"- Cite the document and page inline right after each fact, "
|
||||
'e.g. "(invoice.pdf p.2)" or "(report.pdf p.4-6)", using the names and pages '
|
||||
"shown in each passage header.\n"
|
||||
"- If the passages do not answer the question, say plainly that the stored "
|
||||
"documents do not cover it. Do not attempt a partial guess.\n"
|
||||
"- Answer in the same language as the question."
|
||||
)
|
||||
|
||||
_NO_PASSAGES_ANSWER = "I couldn't find anything relevant to that question in your stored documents."
|
||||
|
||||
|
||||
class _AskOutput(ApiModel):
|
||||
"""Raw model answer for the single ask pass."""
|
||||
|
||||
answer: str = Field(description="The answer grounded in the passages, with inline citations.")
|
||||
|
||||
|
||||
def _meta_int(value: str | None) -> int | None:
|
||||
if value is None:
|
||||
return None
|
||||
try:
|
||||
return int(value)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def passage_from_hit(hit: CollectionSearchHit) -> DocumentPassage:
|
||||
"""Map a store search hit onto the wire passage shape.
|
||||
|
||||
Docparse chunks carry page bounds and a heading path; plain page-text
|
||||
chunks only carry ``page_number``, which maps to both bounds.
|
||||
"""
|
||||
meta = hit.result.document.metadata
|
||||
page_start = _meta_int(meta.get(_PAGE_START_KEY))
|
||||
page_end = _meta_int(meta.get(_PAGE_END_KEY))
|
||||
if page_start is None and page_end is None:
|
||||
page_start = page_end = _meta_int(meta.get(PAGE_NUMBER_METADATA_KEY))
|
||||
heading = meta.get(_HEADING_PATH_KEY)
|
||||
source = meta.get("source")
|
||||
if source and ":page:" in source:
|
||||
# Page-text chunk sources look like "report.pdf:page:3"; show the file name.
|
||||
source = source.rsplit(":page:", 1)[0]
|
||||
return DocumentPassage(
|
||||
document_id=hit.collection,
|
||||
text=hit.result.document.text,
|
||||
score=hit.result.score,
|
||||
page_start=page_start,
|
||||
page_end=page_end,
|
||||
heading_path=heading.split(_HEADING_PATH_SEPARATOR) if heading else [],
|
||||
source=source or None,
|
||||
)
|
||||
|
||||
|
||||
def format_passages(passages: list[DocumentPassage]) -> str:
|
||||
"""Render passages for the prompt with the citation handle in each header."""
|
||||
return "\n\n".join(_format_passage(i, passage) for i, passage in enumerate(passages, 1))
|
||||
|
||||
|
||||
def _format_passage(index: int, passage: DocumentPassage) -> str:
|
||||
name = passage.source or passage.document_id
|
||||
if passage.page_start is None:
|
||||
pages = ""
|
||||
elif passage.page_end is not None and passage.page_end != passage.page_start:
|
||||
pages = f" p.{passage.page_start}-{passage.page_end}"
|
||||
else:
|
||||
pages = f" p.{passage.page_start}"
|
||||
return f"[Passage {index} | {name}{pages}]\n{passage.text}"
|
||||
|
||||
|
||||
class KnowledgeAskAgent:
|
||||
"""Answers a question from the caller's stored documents.
|
||||
|
||||
Retrieves the top passages the caller can read (same path as the search
|
||||
endpoint), then runs one smart-model pass over just those passages.
|
||||
"""
|
||||
|
||||
def __init__(self, runtime: AppRuntime) -> None:
|
||||
self.runtime = runtime
|
||||
# Ollama/custom block tool-calling under native json-schema output; see agents.output_mode.
|
||||
provider = runtime.settings.chat_provider
|
||||
self._agent: Agent[None, _AskOutput] = Agent(
|
||||
model=runtime.smart_model,
|
||||
output_type=structured_output([_AskOutput], chat_provider=provider),
|
||||
system_prompt=_SYSTEM_PROMPT,
|
||||
model_settings=runtime.smart_model_settings,
|
||||
retries=output_retries(provider),
|
||||
)
|
||||
|
||||
async def ask(self, request: AskDocumentsRequest, principals: list[PrincipalId]) -> AskDocumentsResponse:
|
||||
hits = await self.runtime.documents.search_with_collections(
|
||||
request.question, principals=principals, top_k=request.top_k
|
||||
)
|
||||
passages = [passage_from_hit(hit) for hit in hits]
|
||||
if not passages:
|
||||
logger.info("[knowledge-ask] question=%r -> 0 passages", request.question)
|
||||
return AskDocumentsResponse(answer=_NO_PASSAGES_ANSWER, passages=[])
|
||||
prompt = f"Question: {request.question}\n\nPassages:\n{format_passages(passages)}"
|
||||
logger.debug("[knowledge-ask] prompt:\n%s", prompt)
|
||||
result = await self._agent.run(prompt)
|
||||
return AskDocumentsResponse(answer=result.output.answer, passages=passages)
|
||||
@@ -10,6 +10,7 @@ from pydantic_ai.models import Model
|
||||
from stirling.agents import (
|
||||
DocumentClassifierAgent,
|
||||
ExecutionPlanningAgent,
|
||||
KnowledgeAskAgent,
|
||||
OrchestratorAgent,
|
||||
PdfEditAgent,
|
||||
PdfQuestionAgent,
|
||||
@@ -18,7 +19,7 @@ from stirling.agents import (
|
||||
from stirling.agents.ledger import MathAuditorAgent
|
||||
from stirling.agents.pdf_comment import PdfCommentAgent
|
||||
from stirling.config import AppSettings
|
||||
from stirling.docparse import ExtractFieldsAgent, SuggestSchemaAgent
|
||||
from stirling.docparse import ExtractFieldsAgent, SmartSplitAgent, SuggestSchemaAgent
|
||||
from stirling.documents import DocumentService, EmbeddingService
|
||||
from stirling.services import AppRuntime, build_runtime
|
||||
|
||||
@@ -36,7 +37,9 @@ class AppState:
|
||||
math_auditor_agent: MathAuditorAgent
|
||||
pdf_comment_agent: PdfCommentAgent
|
||||
document_classifier_agent: DocumentClassifierAgent
|
||||
knowledge_ask_agent: KnowledgeAskAgent
|
||||
extract_fields_agent: ExtractFieldsAgent
|
||||
smart_split_agent: SmartSplitAgent
|
||||
suggest_schema_agent: SuggestSchemaAgent
|
||||
|
||||
|
||||
@@ -66,7 +69,9 @@ def build_app_state(
|
||||
math_auditor_agent=MathAuditorAgent(runtime),
|
||||
pdf_comment_agent=PdfCommentAgent(runtime),
|
||||
document_classifier_agent=DocumentClassifierAgent(runtime),
|
||||
knowledge_ask_agent=KnowledgeAskAgent(runtime),
|
||||
extract_fields_agent=ExtractFieldsAgent(runtime),
|
||||
smart_split_agent=SmartSplitAgent(runtime),
|
||||
suggest_schema_agent=SuggestSchemaAgent(runtime),
|
||||
)
|
||||
|
||||
|
||||
@@ -7,6 +7,7 @@ from fastapi import Depends, HTTPException, Request, status
|
||||
from stirling.agents import (
|
||||
DocumentClassifierAgent,
|
||||
ExecutionPlanningAgent,
|
||||
KnowledgeAskAgent,
|
||||
OrchestratorAgent,
|
||||
PdfEditAgent,
|
||||
PdfQuestionAgent,
|
||||
@@ -15,7 +16,7 @@ from stirling.agents import (
|
||||
from stirling.agents.ledger import MathAuditorAgent
|
||||
from stirling.agents.pdf_comment import PdfCommentAgent
|
||||
from stirling.config import AppSettings, load_settings
|
||||
from stirling.docparse import ExtractFieldsAgent, SuggestSchemaAgent
|
||||
from stirling.docparse import ExtractFieldsAgent, SmartSplitAgent, SuggestSchemaAgent
|
||||
from stirling.documents import DocumentService
|
||||
from stirling.models import UserId
|
||||
from stirling.services import AppRuntime, current_user_id
|
||||
@@ -61,10 +62,18 @@ def get_document_classifier_agent(request: Request) -> DocumentClassifierAgent:
|
||||
return request.app.state.document_classifier_agent
|
||||
|
||||
|
||||
def get_knowledge_ask_agent(request: Request) -> KnowledgeAskAgent:
|
||||
return request.app.state.knowledge_ask_agent
|
||||
|
||||
|
||||
def get_extract_fields_agent(request: Request) -> ExtractFieldsAgent:
|
||||
return request.app.state.extract_fields_agent
|
||||
|
||||
|
||||
def get_smart_split_agent(request: Request) -> SmartSplitAgent:
|
||||
return request.app.state.smart_split_agent
|
||||
|
||||
|
||||
def get_suggest_schema_agent(request: Request) -> SuggestSchemaAgent:
|
||||
return request.app.state.suggest_schema_agent
|
||||
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
"""DocParse routes: parse, tables, rag-ingest, capabilities.
|
||||
"""DocParse routes: parse, extract, split, chunk, tables, fill, capabilities.
|
||||
|
||||
Tier routing: requests carrying raw file bytes can use the advanced (Docling)
|
||||
path when the addon is installed; text-only requests run the basic path.
|
||||
Forcing ``advanced`` without the addon returns 501 with a machine-readable
|
||||
``addonRequired`` detail that Java maps onto its own error.
|
||||
Tier routing happens here: requests carrying raw file bytes can use the
|
||||
advanced (Docling) path when the addon is installed; text-only requests run
|
||||
the basic path. Forcing ``advanced`` without the addon returns 501 with a
|
||||
machine-readable ``addonRequired`` detail that Java maps onto its own error.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -19,11 +19,14 @@ from fastapi import APIRouter, Depends, HTTPException, status
|
||||
from stirling.api.dependencies import (
|
||||
get_document_service,
|
||||
get_extract_fields_agent,
|
||||
get_smart_split_agent,
|
||||
get_suggest_schema_agent,
|
||||
require_user_id,
|
||||
)
|
||||
from stirling.config import AppSettings, load_settings
|
||||
from stirling.contracts.docparse import (
|
||||
ChunkDocumentRequest,
|
||||
ChunkDocumentResponse,
|
||||
DocChunk,
|
||||
DocparseCapabilities,
|
||||
DocparseMode,
|
||||
@@ -32,17 +35,22 @@ from stirling.contracts.docparse import (
|
||||
ExtractFieldsResponse,
|
||||
ExtractTablesRequest,
|
||||
ExtractTablesResponse,
|
||||
FillDocxRequest,
|
||||
FillDocxResponse,
|
||||
ParseDocumentRequest,
|
||||
ParseDocumentResponse,
|
||||
RagIngestRequest,
|
||||
RagIngestResponse,
|
||||
SmartSplitRequest,
|
||||
SmartSplitResponse,
|
||||
SuggestSchemaRequest,
|
||||
SuggestSchemaResponse,
|
||||
)
|
||||
from stirling.docparse import basic_chunks, probe_capabilities
|
||||
from stirling.docparse import basic_chunks, fill_docx, probe_capabilities
|
||||
from stirling.docparse.capability import models_dir
|
||||
from stirling.docparse.chunking import advanced_chunks
|
||||
from stirling.docparse.extractor import ExtractFieldsAgent, SchemaError, pages_from_parse
|
||||
from stirling.docparse.splitter import SmartSplitAgent
|
||||
from stirling.docparse.suggest_schema import SuggestSchemaAgent
|
||||
from stirling.documents import DocumentService
|
||||
from stirling.documents.service import CONTENT_TYPE_METADATA_KEY, DOCPARSE_CHUNK_CONTENT_TYPE
|
||||
@@ -170,6 +178,39 @@ async def suggest_schema(
|
||||
return await agent.suggest(request, pages, tier)
|
||||
|
||||
|
||||
@router.post("/split", response_model=SmartSplitResponse)
|
||||
async def smart_split(
|
||||
request: SmartSplitRequest,
|
||||
agent: Annotated[SmartSplitAgent, Depends(get_smart_split_agent)],
|
||||
) -> SmartSplitResponse:
|
||||
return await agent.split(request)
|
||||
|
||||
|
||||
@router.post("/chunk", response_model=ChunkDocumentResponse)
|
||||
async def chunk_document(request: ChunkDocumentRequest) -> ChunkDocumentResponse:
|
||||
settings = _settings()
|
||||
caps = _capabilities(settings)
|
||||
use_advanced = request.mode is DocparseMode.ADVANCED or (
|
||||
request.mode is DocparseMode.AUTO and caps.advanced_installed and request.content_base64 is not None
|
||||
)
|
||||
if use_advanced:
|
||||
artifacts = _require_advanced(settings)
|
||||
if request.content_base64 is None:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
||||
detail="advanced chunking needs contentBase64 (the raw file)",
|
||||
)
|
||||
parse = await _parse_advanced(request.content_base64, request.file_name, with_ocr=True, artifacts=artifacts)
|
||||
return advanced_chunks(parse, request.chunk_size, request.overlap)
|
||||
|
||||
if not request.pages:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY,
|
||||
detail="send pages (extracted text) or contentBase64 with the addon installed",
|
||||
)
|
||||
return basic_chunks(request.pages, request.chunk_size, request.overlap)
|
||||
|
||||
|
||||
def _chunk_metadata(chunk: DocChunk) -> dict[str, str]:
|
||||
meta = {CONTENT_TYPE_METADATA_KEY: DOCPARSE_CHUNK_CONTENT_TYPE}
|
||||
if chunk.page_start is not None:
|
||||
@@ -261,3 +302,13 @@ async def extract_tables(request: ExtractTablesRequest) -> ExtractTablesResponse
|
||||
artifacts = _require_advanced(settings)
|
||||
parse = await _parse_advanced(request.content_base64, request.file_name, with_ocr=True, artifacts=artifacts)
|
||||
return ExtractTablesResponse(mode=parse.mode, tables=parse.tables)
|
||||
|
||||
|
||||
@router.post("/fill-docx", response_model=FillDocxResponse)
|
||||
async def fill_docx_template(request: FillDocxRequest) -> FillDocxResponse:
|
||||
try:
|
||||
return await anyio.to_thread.run_sync(lambda: fill_docx(request))
|
||||
except (KeyError, ValueError) as error:
|
||||
raise HTTPException(
|
||||
status_code=status.HTTP_422_UNPROCESSABLE_ENTITY, detail=f"invalid docx template: {error}"
|
||||
) from error
|
||||
|
||||
@@ -5,13 +5,24 @@ from typing import Annotated
|
||||
|
||||
from fastapi import APIRouter, Depends
|
||||
|
||||
from stirling.api.dependencies import get_document_service, require_user_id
|
||||
from stirling.agents.knowledge_ask import KnowledgeAskAgent, passage_from_hit
|
||||
from stirling.api.dependencies import get_document_service, get_knowledge_ask_agent, require_user_id
|
||||
from stirling.config import load_settings
|
||||
from stirling.contracts import (
|
||||
DeleteDocumentResponse,
|
||||
IngestDocumentRequest,
|
||||
IngestDocumentResponse,
|
||||
)
|
||||
from stirling.contracts.documents import PurgeOwnerResponse
|
||||
from stirling.contracts.documents import (
|
||||
AskDocumentsRequest,
|
||||
AskDocumentsResponse,
|
||||
DocumentStatsResponse,
|
||||
DocumentSummary,
|
||||
ListDocumentsResponse,
|
||||
PurgeOwnerResponse,
|
||||
SearchDocumentsRequest,
|
||||
SearchDocumentsResponse,
|
||||
)
|
||||
from stirling.documents import DocumentService
|
||||
from stirling.models import FileId, OwnerId, PrincipalId, UserId
|
||||
|
||||
@@ -45,6 +56,70 @@ async def ingest_document(
|
||||
return IngestDocumentResponse(document_id=request.document_id, chunks_indexed=chunks_indexed)
|
||||
|
||||
|
||||
@router.get("/stats", response_model=DocumentStatsResponse)
|
||||
async def document_stats(
|
||||
documents: Annotated[DocumentService, Depends(get_document_service)],
|
||||
_user_id: Annotated[UserId, Depends(require_user_id)],
|
||||
) -> DocumentStatsResponse:
|
||||
"""Deployment-wide store counts for the admin dashboard.
|
||||
|
||||
Not tenant-filtered: counts cover every owner's content, so this reports
|
||||
what the whole store holds, not what the caller can read.
|
||||
"""
|
||||
settings = load_settings()
|
||||
counts = await documents.stats()
|
||||
return DocumentStatsResponse(
|
||||
backend=settings.documents_backend.value,
|
||||
documents=counts.documents,
|
||||
chunks=counts.chunks,
|
||||
embedding_model=settings.rag_embedding_model,
|
||||
)
|
||||
|
||||
|
||||
@router.get("/list", response_model=ListDocumentsResponse)
|
||||
async def list_documents(
|
||||
documents: Annotated[DocumentService, Depends(get_document_service)],
|
||||
user_id: Annotated[UserId, Depends(require_user_id)],
|
||||
) -> ListDocumentsResponse:
|
||||
"""Per-document rollup of what the caller can read: distinct document ids
|
||||
with their stored source label and chunk count. Never shows another
|
||||
principal's documents.
|
||||
"""
|
||||
summaries = await documents.list_documents([PrincipalId(user_id)])
|
||||
return ListDocumentsResponse(
|
||||
documents=[
|
||||
DocumentSummary(document_id=FileId(s.collection), source=s.source, chunks=s.chunks) for s in summaries
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
@router.post("/search", response_model=SearchDocumentsResponse)
|
||||
async def search_documents(
|
||||
request: SearchDocumentsRequest,
|
||||
documents: Annotated[DocumentService, Depends(get_document_service)],
|
||||
user_id: Annotated[UserId, Depends(require_user_id)],
|
||||
) -> SearchDocumentsResponse:
|
||||
"""Semantic search across every document the caller can read.
|
||||
|
||||
Same retrieval path as the RAG toolset: embed the query, search the
|
||||
caller's readable collections, merge by score.
|
||||
"""
|
||||
hits = await documents.search_with_collections(
|
||||
request.query, principals=[PrincipalId(user_id)], top_k=request.top_k
|
||||
)
|
||||
return SearchDocumentsResponse(passages=[passage_from_hit(hit) for hit in hits])
|
||||
|
||||
|
||||
@router.post("/ask", response_model=AskDocumentsResponse)
|
||||
async def ask_documents(
|
||||
request: AskDocumentsRequest,
|
||||
agent: Annotated[KnowledgeAskAgent, Depends(get_knowledge_ask_agent)],
|
||||
user_id: Annotated[UserId, Depends(require_user_id)],
|
||||
) -> AskDocumentsResponse:
|
||||
"""Answer a question from the caller's stored documents with inline citations."""
|
||||
return await agent.ask(request, principals=[PrincipalId(user_id)])
|
||||
|
||||
|
||||
@router.delete("/by-id/{document_id}", response_model=DeleteDocumentResponse)
|
||||
async def delete_document(
|
||||
document_id: FileId,
|
||||
|
||||
@@ -42,6 +42,36 @@ from .contradiction import (
|
||||
ContradictionReport,
|
||||
ContradictionSeverity,
|
||||
)
|
||||
from .docparse import (
|
||||
BlockType,
|
||||
ChunkDocumentRequest,
|
||||
ChunkDocumentResponse,
|
||||
DocBlock,
|
||||
DocChunk,
|
||||
DocparseCapabilities,
|
||||
DocparseMode,
|
||||
DocparseTier,
|
||||
DocTable,
|
||||
ExtractedField,
|
||||
ExtractFieldsRequest,
|
||||
ExtractFieldsResponse,
|
||||
ExtractTablesRequest,
|
||||
ExtractTablesResponse,
|
||||
FieldCitation,
|
||||
FillDocxRequest,
|
||||
FillDocxResponse,
|
||||
ParseDocumentRequest,
|
||||
ParseDocumentResponse,
|
||||
RagIngestRequest,
|
||||
RagIngestResponse,
|
||||
SmartSplitRequest,
|
||||
SmartSplitResponse,
|
||||
SplitPart,
|
||||
SuggestedField,
|
||||
SuggestedFieldType,
|
||||
SuggestSchemaRequest,
|
||||
SuggestSchemaResponse,
|
||||
)
|
||||
from .document_classifier import (
|
||||
ClassifyDocumentRequest,
|
||||
ClassifyDocumentResponse,
|
||||
@@ -49,13 +79,21 @@ from .document_classifier import (
|
||||
LabelOption,
|
||||
)
|
||||
from .documents import (
|
||||
AskDocumentsRequest,
|
||||
AskDocumentsResponse,
|
||||
DeleteDocumentResponse,
|
||||
DocumentPassage,
|
||||
DocumentStatsResponse,
|
||||
DocumentSummary,
|
||||
IngestDocumentRequest,
|
||||
IngestDocumentResponse,
|
||||
ListDocumentsResponse,
|
||||
Page,
|
||||
PageRange,
|
||||
PageText,
|
||||
PurgeOwnerResponse,
|
||||
SearchDocumentsRequest,
|
||||
SearchDocumentsResponse,
|
||||
)
|
||||
from .execution import (
|
||||
AgentExecutionRequest,
|
||||
@@ -140,10 +178,15 @@ __all__ = [
|
||||
"AiFile",
|
||||
"AiToolAgentStep",
|
||||
"ArtifactKind",
|
||||
"AskDocumentsRequest",
|
||||
"AskDocumentsResponse",
|
||||
"BlockType",
|
||||
"CannotContinueExecutionAction",
|
||||
"ChunkDocumentRequest",
|
||||
"ChunkDocumentResponse",
|
||||
"Claim",
|
||||
"ClassifyDocumentRequest",
|
||||
"ClassifyDocumentResponse",
|
||||
"Claim",
|
||||
"CommentSpec",
|
||||
"CompletedExecutionAction",
|
||||
"ConfigApplyResponse",
|
||||
@@ -156,30 +199,45 @@ __all__ = [
|
||||
"ContradictionSeverity",
|
||||
"ConversationMessage",
|
||||
"DeleteDocumentResponse",
|
||||
"PurgeOwnerResponse",
|
||||
"Discrepancy",
|
||||
"DocumentClassificationResponse",
|
||||
"LabelOption",
|
||||
"DocumentMeta",
|
||||
"DocumentSections",
|
||||
"DiscrepancyKind",
|
||||
"DocBlock",
|
||||
"DocChunk",
|
||||
"DocTable",
|
||||
"DocparseCapabilities",
|
||||
"DocparseMode",
|
||||
"DocparseTier",
|
||||
"DocumentClassificationResponse",
|
||||
"DocumentMeta",
|
||||
"DocumentPassage",
|
||||
"DocumentSections",
|
||||
"DocumentStatsResponse",
|
||||
"DocumentSummary",
|
||||
"EditCannotDoResponse",
|
||||
"EditClarificationRequest",
|
||||
"EditPlanResponse",
|
||||
"Evidence",
|
||||
"ExecutionContext",
|
||||
"ExecutionStepResult",
|
||||
"ExtractFieldsRequest",
|
||||
"ExtractFieldsResponse",
|
||||
"ExtractTablesRequest",
|
||||
"ExtractTablesResponse",
|
||||
"ExtractedField",
|
||||
"ExtractedFileText",
|
||||
"ExtractedTextArtifact",
|
||||
"FieldCitation",
|
||||
"FillDocxRequest",
|
||||
"FillDocxResponse",
|
||||
"Folio",
|
||||
"FolioManifest",
|
||||
"FolioType",
|
||||
"format_conversation_history",
|
||||
"format_file_names",
|
||||
"GenerateFileResponse",
|
||||
"HealthResponse",
|
||||
"IngestDocumentRequest",
|
||||
"IngestDocumentResponse",
|
||||
"LabelOption",
|
||||
"ListDocumentsResponse",
|
||||
"MathAuditorToolReportArtifact",
|
||||
"NeedContentFileRequest",
|
||||
"NeedContentResponse",
|
||||
@@ -190,6 +248,8 @@ __all__ = [
|
||||
"Page",
|
||||
"PageRange",
|
||||
"PageText",
|
||||
"ParseDocumentRequest",
|
||||
"ParseDocumentResponse",
|
||||
"PdfCommentInstruction",
|
||||
"PdfCommentReport",
|
||||
"PdfCommentRequest",
|
||||
@@ -212,9 +272,21 @@ __all__ = [
|
||||
"PdfReviewOrchestrateResponse",
|
||||
"PdfTextSelection",
|
||||
"ProgressEvent",
|
||||
"PurgeOwnerResponse",
|
||||
"RagIngestRequest",
|
||||
"RagIngestResponse",
|
||||
"Requisition",
|
||||
"SearchDocumentsRequest",
|
||||
"SearchDocumentsResponse",
|
||||
"Severity",
|
||||
"SmartSplitRequest",
|
||||
"SmartSplitResponse",
|
||||
"SplitPart",
|
||||
"StepKind",
|
||||
"SuggestSchemaRequest",
|
||||
"SuggestSchemaResponse",
|
||||
"SuggestedField",
|
||||
"SuggestedFieldType",
|
||||
"SupportedCapability",
|
||||
"TextChunk",
|
||||
"ToolCallExecutionAction",
|
||||
@@ -228,4 +300,6 @@ __all__ = [
|
||||
"WholeDocSliceDone",
|
||||
"WorkflowArtifact",
|
||||
"WorkflowOutcome",
|
||||
"format_conversation_history",
|
||||
"format_file_names",
|
||||
]
|
||||
|
||||
@@ -206,6 +206,46 @@ class SuggestSchemaResponse(ApiModel):
|
||||
fields: list[SuggestedField] = Field(default_factory=list)
|
||||
|
||||
|
||||
class SplitPart(ApiModel):
|
||||
start_page: int = Field(ge=1)
|
||||
end_page: int = Field(ge=1)
|
||||
label: str
|
||||
confidence: float = Field(ge=0.0, le=1.0)
|
||||
|
||||
|
||||
class SmartSplitRequest(ApiModel):
|
||||
file_name: str = Field(min_length=1)
|
||||
rule: str = Field(
|
||||
min_length=1, description="Natural-language boundary rule, e.g. 'split where a new invoice starts'."
|
||||
)
|
||||
pages: list[PageText]
|
||||
max_parts: int = Field(default=50, ge=1, le=500)
|
||||
|
||||
|
||||
class SmartSplitResponse(ApiModel):
|
||||
parts: list[SplitPart] = Field(default_factory=list)
|
||||
|
||||
|
||||
class ChunkDocumentRequest(ApiModel):
|
||||
file_name: str = Field(min_length=1)
|
||||
pages: list[PageText] | None = None
|
||||
content_base64: str | None = None
|
||||
chunk_size: int = Field(default=512, ge=64, le=32_768)
|
||||
overlap: int = Field(default=64, ge=0, le=4_096)
|
||||
mode: DocparseMode = DocparseMode.AUTO
|
||||
|
||||
|
||||
class FillDocxRequest(ApiModel):
|
||||
template_base64: str = Field(min_length=1)
|
||||
data: dict[str, JsonValue]
|
||||
|
||||
|
||||
class FillDocxResponse(ApiModel):
|
||||
docx_base64: str
|
||||
replaced: int = Field(ge=0)
|
||||
missing: list[str] = Field(default_factory=list)
|
||||
|
||||
|
||||
class DocparseCapabilities(ApiModel):
|
||||
"""What the engine can actually do right now; Java caches and republishes this."""
|
||||
|
||||
|
||||
@@ -75,3 +75,65 @@ class PurgeOwnerResponse(ApiModel):
|
||||
|
||||
owner_id: OwnerId
|
||||
deleted: int = Field(ge=0)
|
||||
|
||||
|
||||
class DocumentStatsResponse(ApiModel):
|
||||
"""Returned by ``GET /api/v1/documents/stats``. Deployment-wide counts
|
||||
(every owner's content) powering the admin dashboard."""
|
||||
|
||||
backend: str
|
||||
documents: int = Field(ge=0)
|
||||
chunks: int = Field(ge=0)
|
||||
embedding_model: str
|
||||
|
||||
|
||||
class DocumentSummary(ApiModel):
|
||||
"""One stored document the caller can read: its id, source label, chunk count."""
|
||||
|
||||
document_id: FileId
|
||||
source: str
|
||||
chunks: int = Field(ge=0)
|
||||
|
||||
|
||||
class ListDocumentsResponse(ApiModel):
|
||||
"""Returned by ``GET /api/v1/documents/list``. Caller-scoped rollup."""
|
||||
|
||||
documents: list[DocumentSummary]
|
||||
|
||||
|
||||
class SearchDocumentsRequest(ApiModel):
|
||||
"""Semantic search over every document the caller can read."""
|
||||
|
||||
query: str = Field(min_length=1)
|
||||
top_k: int = Field(default=8, ge=1, le=50)
|
||||
|
||||
|
||||
class DocumentPassage(ApiModel):
|
||||
"""A retrieved chunk on the wire. Page bounds and heading path come from
|
||||
chunk metadata when present (docparse chunks carry them); nulls otherwise."""
|
||||
|
||||
document_id: FileId
|
||||
text: str
|
||||
score: float
|
||||
page_start: int | None = None
|
||||
page_end: int | None = None
|
||||
heading_path: list[str] = Field(default_factory=list)
|
||||
source: str | None = None
|
||||
|
||||
|
||||
class SearchDocumentsResponse(ApiModel):
|
||||
passages: list[DocumentPassage]
|
||||
|
||||
|
||||
class AskDocumentsRequest(ApiModel):
|
||||
"""Question answered only from the caller's stored documents."""
|
||||
|
||||
question: str = Field(min_length=1)
|
||||
top_k: int = Field(default=8, ge=1, le=20)
|
||||
|
||||
|
||||
class AskDocumentsResponse(ApiModel):
|
||||
"""Grounded answer with inline citations plus the passages it drew from."""
|
||||
|
||||
answer: str
|
||||
passages: list[DocumentPassage]
|
||||
|
||||
@@ -8,14 +8,18 @@ from __future__ import annotations
|
||||
|
||||
from stirling.docparse.capability import activate_site, probe_capabilities
|
||||
from stirling.docparse.chunking import advanced_chunks, basic_chunks
|
||||
from stirling.docparse.docxfill import fill_docx
|
||||
from stirling.docparse.extractor import ExtractFieldsAgent
|
||||
from stirling.docparse.splitter import SmartSplitAgent
|
||||
from stirling.docparse.suggest_schema import SuggestSchemaAgent
|
||||
|
||||
__all__ = [
|
||||
"ExtractFieldsAgent",
|
||||
"SmartSplitAgent",
|
||||
"SuggestSchemaAgent",
|
||||
"activate_site",
|
||||
"advanced_chunks",
|
||||
"basic_chunks",
|
||||
"fill_docx",
|
||||
"probe_capabilities",
|
||||
]
|
||||
|
||||
@@ -0,0 +1,178 @@
|
||||
"""Fill DOCX templates from JSON data: ``{{ dotted.path }}`` placeholders.
|
||||
|
||||
Scalar placeholders are replaced everywhere (body, tables, headers, footers).
|
||||
A table row whose text contains ``{{#items.field}}`` markers is treated as a
|
||||
row template: it is cloned once per element of the ``items`` array. Unresolved
|
||||
placeholders are left in place and reported back so the caller can surface them.
|
||||
|
||||
Formatting caveat: a placeholder split across styled runs collapses that
|
||||
paragraph's text into its first run's style.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import copy
|
||||
import io
|
||||
import re
|
||||
from typing import Any
|
||||
|
||||
from pydantic import JsonValue
|
||||
|
||||
from stirling.contracts.docparse import FillDocxRequest, FillDocxResponse
|
||||
|
||||
_PLACEHOLDER = re.compile(r"\{\{\s*(#?[\w.]+)\s*\}\}")
|
||||
|
||||
|
||||
def _resolve(path: str, data: dict[str, Any]) -> Any | None:
|
||||
node: Any = data
|
||||
for part in path.split("."):
|
||||
if isinstance(node, dict) and part in node:
|
||||
node = node[part]
|
||||
else:
|
||||
return None
|
||||
return node
|
||||
|
||||
|
||||
def _render_scalar(value: Any) -> str:
|
||||
if value is None:
|
||||
return ""
|
||||
if isinstance(value, bool):
|
||||
return "true" if value else "false"
|
||||
if isinstance(value, list):
|
||||
return ", ".join(_render_scalar(v) for v in value)
|
||||
return str(value)
|
||||
|
||||
|
||||
class _Stats:
|
||||
def __init__(self) -> None:
|
||||
self.replaced = 0
|
||||
self.missing: set[str] = set()
|
||||
|
||||
|
||||
def _fill_paragraph(paragraph: Any, data: dict[str, Any], stats: _Stats) -> None:
|
||||
text = paragraph.text
|
||||
if "{{" not in text:
|
||||
return
|
||||
|
||||
def substitute(match: re.Match[str]) -> str:
|
||||
path = match.group(1)
|
||||
if path.startswith("#"):
|
||||
return match.group(0) # row-template marker, handled at table level
|
||||
value = _resolve(path, data)
|
||||
if value is None:
|
||||
stats.missing.add(path)
|
||||
return match.group(0)
|
||||
stats.replaced += 1
|
||||
return _render_scalar(value)
|
||||
|
||||
rendered = _PLACEHOLDER.sub(substitute, text)
|
||||
if rendered == text:
|
||||
return
|
||||
# Collapse into the first run to survive placeholders split across runs.
|
||||
if paragraph.runs:
|
||||
paragraph.runs[0].text = rendered
|
||||
for run in paragraph.runs[1:]:
|
||||
run.text = ""
|
||||
else:
|
||||
paragraph.add_run(rendered)
|
||||
|
||||
|
||||
def _row_template_array(row: Any) -> str | None:
|
||||
"""Return the array name when the row carries ``{{#name.field}}`` markers."""
|
||||
names = {
|
||||
match.group(1)[1:].split(".")[0]
|
||||
for cell in row.cells
|
||||
for match in _PLACEHOLDER.finditer(cell.text)
|
||||
if match.group(1).startswith("#")
|
||||
}
|
||||
return names.pop() if len(names) == 1 else None
|
||||
|
||||
|
||||
def _fill_table(table: Any, data: dict[str, Any], stats: _Stats) -> None:
|
||||
for row in list(table.rows):
|
||||
array_name = _row_template_array(row)
|
||||
if array_name is None:
|
||||
continue
|
||||
items = _resolve(array_name, data)
|
||||
if not isinstance(items, list):
|
||||
stats.missing.add(array_name)
|
||||
continue
|
||||
for _ in items:
|
||||
new_row = copy.deepcopy(row._tr)
|
||||
row._tr.addprevious(new_row)
|
||||
# Clones sit before the template; rewrite their markers, then drop the template.
|
||||
_rewrite_cloned_rows(table, row, array_name, items, data, stats)
|
||||
row._tr.getparent().remove(row._tr)
|
||||
|
||||
|
||||
def _rewrite_cloned_rows(
|
||||
table: Any, template_row: Any, array_name: str, items: list[Any], data: dict[str, Any], stats: _Stats
|
||||
) -> None:
|
||||
marker_prefix = f"#{array_name}"
|
||||
clones = [
|
||||
r for r in table.rows if r._tr is not template_row._tr and marker_prefix in "".join(c.text for c in r.cells)
|
||||
]
|
||||
for row, item in zip(clones, items, strict=False):
|
||||
scoped = dict(data)
|
||||
scoped[array_name] = item if isinstance(item, dict) else {"value": item}
|
||||
for cell in row.cells:
|
||||
for paragraph in cell.paragraphs:
|
||||
text = paragraph.text
|
||||
|
||||
def substitute(match: re.Match[str]) -> str:
|
||||
path = match.group(1)
|
||||
if not path.startswith(marker_prefix):
|
||||
return match.group(0)
|
||||
item_path = path[1:] # "#items.field" -> "items.field"
|
||||
value = _resolve(item_path, scoped)
|
||||
if value is None and "." not in item_path:
|
||||
value = scoped[array_name].get("value") if isinstance(scoped[array_name], dict) else None
|
||||
if value is None:
|
||||
stats.missing.add(item_path)
|
||||
return match.group(0)
|
||||
stats.replaced += 1
|
||||
return _render_scalar(value)
|
||||
|
||||
rendered = _PLACEHOLDER.sub(substitute, text)
|
||||
if rendered != text:
|
||||
if paragraph.runs:
|
||||
paragraph.runs[0].text = rendered
|
||||
for run in paragraph.runs[1:]:
|
||||
run.text = ""
|
||||
else:
|
||||
paragraph.add_run(rendered)
|
||||
|
||||
|
||||
def _walk_paragraphs(document: Any) -> list[tuple[Any, Any]]:
|
||||
"""Yield (paragraph, containing table or None) across body, tables, headers, footers."""
|
||||
found: list[tuple[Any, Any]] = [(p, None) for p in document.paragraphs]
|
||||
for table in document.tables:
|
||||
for row in table.rows:
|
||||
for cell in row.cells:
|
||||
found.extend((p, table) for p in cell.paragraphs)
|
||||
for section in document.sections:
|
||||
for part in (section.header, section.footer):
|
||||
found.extend((p, None) for p in part.paragraphs)
|
||||
return found
|
||||
|
||||
|
||||
def fill_docx(request: FillDocxRequest) -> FillDocxResponse:
|
||||
import docx # local import: python-docx is small but only needed here
|
||||
|
||||
data: dict[str, JsonValue] = dict(request.data)
|
||||
document = docx.Document(io.BytesIO(base64.b64decode(request.template_base64)))
|
||||
stats = _Stats()
|
||||
|
||||
for table in document.tables:
|
||||
_fill_table(table, data, stats)
|
||||
for paragraph, _table in _walk_paragraphs(document):
|
||||
_fill_paragraph(paragraph, data, stats)
|
||||
|
||||
out = io.BytesIO()
|
||||
document.save(out)
|
||||
return FillDocxResponse(
|
||||
docx_base64=base64.b64encode(out.getvalue()).decode("ascii"),
|
||||
replaced=stats.replaced,
|
||||
missing=sorted(stats.missing),
|
||||
)
|
||||
@@ -0,0 +1,105 @@
|
||||
"""Content-based document splitting: an LLM finds sub-document boundaries.
|
||||
|
||||
Works entirely from caller-supplied page text (basic tier friendly); the fast
|
||||
model sees a bounded per-page preview and answers with boundary start pages,
|
||||
which are then validated in code (monotonic, in range, capped)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from pydantic import Field
|
||||
from pydantic_ai import Agent
|
||||
|
||||
from stirling.agents.output_mode import output_retries, structured_output
|
||||
from stirling.contracts.docparse import SmartSplitRequest, SmartSplitResponse, SplitPart
|
||||
from stirling.contracts.documents import PageText
|
||||
from stirling.models import ApiModel
|
||||
from stirling.services import AppRuntime
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Per-page preview budget; boundaries are recognisable from page openings.
|
||||
PAGE_PREVIEW_CHARS = 600
|
||||
|
||||
_SYSTEM_PROMPT = (
|
||||
"You split a multi-document file into its component documents.\n"
|
||||
"\n"
|
||||
"You are shown the beginning of every page. Apply the user's splitting rule and "
|
||||
"answer with every page where a NEW component document starts.\n"
|
||||
"Rules:\n"
|
||||
"- Page 1 always starts the first component.\n"
|
||||
"- Give each component a short descriptive label (e.g. 'Invoice #4821', 'Cover letter').\n"
|
||||
"- Give your confidence 0.0-1.0 per boundary.\n"
|
||||
"- If the rule doesn't match anything, return just the page-1 component spanning the whole file."
|
||||
)
|
||||
|
||||
|
||||
class _Boundary(ApiModel):
|
||||
start_page: int = Field(ge=1, description="First page of this component document.")
|
||||
label: str = Field(description="Short human label for the component.")
|
||||
confidence: float = Field(ge=0.0, le=1.0)
|
||||
|
||||
|
||||
class _SplitOutput(ApiModel):
|
||||
boundaries: list[_Boundary] = Field(default_factory=list)
|
||||
|
||||
|
||||
def _format_pages(pages: list[PageText], max_pages: int) -> str:
|
||||
shown = pages[:max_pages]
|
||||
parts = [f"[Page {p.page_number}] {p.text[:PAGE_PREVIEW_CHARS]}" for p in shown]
|
||||
if len(pages) > max_pages:
|
||||
parts.append(f"({len(pages) - max_pages} further pages omitted)")
|
||||
return "\n\n".join(parts) if parts else "(no extractable text)"
|
||||
|
||||
|
||||
def validate_boundaries(output: _SplitOutput, page_count: int, max_parts: int) -> list[SplitPart]:
|
||||
"""Coerce the model's boundaries into a clean, complete partition of 1..page_count."""
|
||||
starts: dict[int, _Boundary] = {}
|
||||
for boundary in output.boundaries:
|
||||
if 1 <= boundary.start_page <= page_count and boundary.start_page not in starts:
|
||||
starts[boundary.start_page] = boundary
|
||||
if 1 not in starts:
|
||||
starts[1] = _Boundary(start_page=1, label="Document", confidence=1.0)
|
||||
|
||||
ordered = [starts[k] for k in sorted(starts)][:max_parts]
|
||||
parts: list[SplitPart] = []
|
||||
for i, boundary in enumerate(ordered):
|
||||
end_page = ordered[i + 1].start_page - 1 if i + 1 < len(ordered) else page_count
|
||||
parts.append(
|
||||
SplitPart(
|
||||
start_page=boundary.start_page,
|
||||
end_page=end_page,
|
||||
label=boundary.label.strip() or f"Part {i + 1}",
|
||||
confidence=round(boundary.confidence, 4),
|
||||
)
|
||||
)
|
||||
return parts
|
||||
|
||||
|
||||
class SmartSplitAgent:
|
||||
def __init__(self, runtime: AppRuntime) -> None:
|
||||
self.runtime = runtime
|
||||
provider = runtime.settings.chat_provider
|
||||
self._agent: Agent[None, _SplitOutput] = Agent(
|
||||
model=runtime.fast_model,
|
||||
output_type=structured_output([_SplitOutput], chat_provider=provider),
|
||||
system_prompt=_SYSTEM_PROMPT,
|
||||
model_settings=runtime.fast_model_settings,
|
||||
retries=output_retries(provider),
|
||||
)
|
||||
|
||||
async def split(self, request: SmartSplitRequest) -> SmartSplitResponse:
|
||||
pages = request.pages
|
||||
if not pages:
|
||||
return SmartSplitResponse(parts=[])
|
||||
page_count = max(p.page_number for p in pages)
|
||||
prompt = (
|
||||
f"Splitting rule: {request.rule}\n\n"
|
||||
f"Document file name: {request.file_name}\n"
|
||||
f"Pages:\n{_format_pages(pages, self.runtime.settings.max_pages)}"
|
||||
)
|
||||
result = await self._agent.run(prompt)
|
||||
parts = validate_boundaries(result.output, page_count, request.max_parts)
|
||||
logger.info("docparse: split %s into %d parts", request.file_name, len(parts))
|
||||
return SmartSplitResponse(parts=parts)
|
||||
@@ -3,11 +3,20 @@ from __future__ import annotations
|
||||
from stirling.documents.embedder import EmbeddingService
|
||||
from stirling.documents.pgvector_store import PgVectorStore
|
||||
from stirling.documents.rag_capability import RagCapability
|
||||
from stirling.documents.service import DocumentService
|
||||
from stirling.documents.service import CollectionSearchHit, DocumentService
|
||||
from stirling.documents.sqlite_vec_store import SqliteVecStore
|
||||
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
|
||||
from stirling.documents.store import (
|
||||
CollectionSummary,
|
||||
Document,
|
||||
DocumentStore,
|
||||
SearchResult,
|
||||
StoredPage,
|
||||
StoreStats,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"CollectionSearchHit",
|
||||
"CollectionSummary",
|
||||
"Document",
|
||||
"DocumentService",
|
||||
"DocumentStore",
|
||||
@@ -16,5 +25,6 @@ __all__ = [
|
||||
"RagCapability",
|
||||
"SearchResult",
|
||||
"SqliteVecStore",
|
||||
"StoreStats",
|
||||
"StoredPage",
|
||||
]
|
||||
|
||||
@@ -10,7 +10,14 @@ from pgvector.psycopg import register_vector_async
|
||||
from psycopg_pool import AsyncConnectionPool
|
||||
|
||||
from stirling.contracts.documents import Page, PageRange
|
||||
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
|
||||
from stirling.documents.store import (
|
||||
CollectionSummary,
|
||||
Document,
|
||||
DocumentStore,
|
||||
SearchResult,
|
||||
StoredPage,
|
||||
StoreStats,
|
||||
)
|
||||
from stirling.models import OwnerId, PrincipalId
|
||||
|
||||
_READ_PERMISSION = "read"
|
||||
@@ -411,5 +418,44 @@ class PgVectorStore(DocumentStore):
|
||||
rows = await cur.fetchall()
|
||||
return [r[0] for r in rows]
|
||||
|
||||
async def list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
|
||||
if not principals:
|
||||
return []
|
||||
await self._ensure_ready()
|
||||
async with self._pool.connection() as conn:
|
||||
async with conn.cursor() as cur:
|
||||
# MIN(owner_id) mirrors _readable_owner_for's ORDER BY owner_id LIMIT 1.
|
||||
await cur.execute(
|
||||
"""
|
||||
SELECT r.collection, m.source, COUNT(d.id)
|
||||
FROM (
|
||||
SELECT collection, MIN(owner_id) AS owner_id
|
||||
FROM document_acl
|
||||
WHERE permission = %s AND principal_id = ANY(%s)
|
||||
GROUP BY collection
|
||||
) r
|
||||
JOIN documents_meta m ON m.collection = r.collection AND m.owner_id = r.owner_id
|
||||
LEFT JOIN rag_documents d ON d.collection = r.collection AND d.owner_id = r.owner_id
|
||||
GROUP BY r.collection, m.source
|
||||
ORDER BY r.collection
|
||||
""",
|
||||
(_READ_PERMISSION, list(principals)),
|
||||
)
|
||||
rows = await cur.fetchall()
|
||||
return [CollectionSummary(collection=r[0], source=r[1], chunks=int(r[2])) for r in rows]
|
||||
|
||||
async def stats(self) -> StoreStats:
|
||||
await self._ensure_ready()
|
||||
async with self._pool.connection() as conn:
|
||||
async with conn.cursor() as cur:
|
||||
await cur.execute("SELECT COUNT(DISTINCT collection) FROM documents_meta")
|
||||
doc_row = await cur.fetchone()
|
||||
await cur.execute("SELECT COUNT(*) FROM rag_documents")
|
||||
chunk_row = await cur.fetchone()
|
||||
return StoreStats(
|
||||
documents=int(doc_row[0]) if doc_row else 0,
|
||||
chunks=int(chunk_row[0]) if chunk_row else 0,
|
||||
)
|
||||
|
||||
async def close(self) -> None:
|
||||
await self._pool.close()
|
||||
|
||||
@@ -1,15 +1,32 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime
|
||||
|
||||
from stirling.contracts.documents import Page, PageRange, PageText
|
||||
from stirling.documents.embedder import EmbeddingService
|
||||
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
|
||||
from stirling.documents.store import (
|
||||
CollectionSummary,
|
||||
Document,
|
||||
DocumentStore,
|
||||
SearchResult,
|
||||
StoredPage,
|
||||
StoreStats,
|
||||
)
|
||||
from stirling.models import FileId, OwnerId, PrincipalId
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CollectionSearchHit:
|
||||
"""A search result tagged with the collection it came from."""
|
||||
|
||||
collection: FileId
|
||||
result: SearchResult
|
||||
|
||||
|
||||
PAGE_NUMBER_METADATA_KEY = "page_number"
|
||||
CONTENT_TYPE_METADATA_KEY = "content_type"
|
||||
PAGE_TEXT_CONTENT_TYPE = "page_text"
|
||||
@@ -187,6 +204,32 @@ class DocumentService:
|
||||
all_results.sort(key=lambda r: r.score, reverse=True)
|
||||
return all_results[:k]
|
||||
|
||||
async def search_with_collections(
|
||||
self,
|
||||
query: str,
|
||||
principals: list[PrincipalId],
|
||||
top_k: int | None = None,
|
||||
) -> list[CollectionSearchHit]:
|
||||
"""Cross-collection search like :meth:`search`, but every result keeps
|
||||
the collection it came from. Restricted to what ``principals`` can read.
|
||||
"""
|
||||
k = top_k if top_k is not None else self._default_top_k
|
||||
query_embedding = await self._embedder.embed_query(query)
|
||||
hits: list[CollectionSearchHit] = []
|
||||
for col_name in await self._store.list_collections(principals):
|
||||
try:
|
||||
results = await self._store.search(col_name, query_embedding, k, principals)
|
||||
except Exception: # noqa: BLE001 - any backend error on one collection should not stop the others
|
||||
logger.warning(
|
||||
"Skipping collection %s during cross-collection search",
|
||||
col_name,
|
||||
exc_info=True,
|
||||
)
|
||||
continue
|
||||
hits.extend(CollectionSearchHit(collection=FileId(col_name), result=r) for r in results)
|
||||
hits.sort(key=lambda hit: hit.result.score, reverse=True)
|
||||
return hits[:k]
|
||||
|
||||
async def read_pages(
|
||||
self,
|
||||
collection: FileId,
|
||||
@@ -223,6 +266,10 @@ class DocumentService:
|
||||
"""List collections readable by at least one of ``principals``."""
|
||||
return [FileId(name) for name in await self._store.list_collections(principals)]
|
||||
|
||||
async def list_documents(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
|
||||
"""Per-document rollup (source, chunk count) readable by ``principals``."""
|
||||
return await self._store.list_collection_summaries(principals)
|
||||
|
||||
async def grant_read(
|
||||
self,
|
||||
collection: FileId,
|
||||
@@ -241,6 +288,10 @@ class DocumentService:
|
||||
"""Revoke a principal's access on an existing doc."""
|
||||
await self._store.revoke(collection, owner_id, principal)
|
||||
|
||||
async def stats(self) -> StoreStats:
|
||||
"""Deployment-wide document/chunk counts from the backing store."""
|
||||
return await self._store.stats()
|
||||
|
||||
async def close(self) -> None:
|
||||
"""Release the underlying store's resources."""
|
||||
await self._store.close()
|
||||
|
||||
@@ -11,7 +11,14 @@ from pathlib import Path
|
||||
import sqlite_vec
|
||||
|
||||
from stirling.contracts.documents import Page, PageRange
|
||||
from stirling.documents.store import Document, DocumentStore, SearchResult, StoredPage
|
||||
from stirling.documents.store import (
|
||||
CollectionSummary,
|
||||
Document,
|
||||
DocumentStore,
|
||||
SearchResult,
|
||||
StoredPage,
|
||||
StoreStats,
|
||||
)
|
||||
from stirling.models import OwnerId, PrincipalId
|
||||
|
||||
_READ_PERMISSION = "read"
|
||||
@@ -538,6 +545,42 @@ class SqliteVecStore(DocumentStore):
|
||||
).fetchall()
|
||||
return [r[0] for r in rows]
|
||||
|
||||
async def list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
|
||||
async with self._lock:
|
||||
return await asyncio.to_thread(self._sync_list_collection_summaries, principals)
|
||||
|
||||
def _sync_list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
|
||||
if not principals:
|
||||
return []
|
||||
placeholders = ",".join("?" * len(principals))
|
||||
# MIN(owner_id) mirrors _readable_owner_for's ORDER BY owner_id LIMIT 1.
|
||||
rows = self._conn.execute(
|
||||
f"""
|
||||
SELECT r.collection, m.source, COUNT(d.id)
|
||||
FROM (
|
||||
SELECT collection, MIN(owner_id) AS owner_id
|
||||
FROM document_acl
|
||||
WHERE permission = ? AND principal_id IN ({placeholders})
|
||||
GROUP BY collection
|
||||
) r
|
||||
JOIN documents_meta m ON m.collection = r.collection AND m.owner_id = r.owner_id
|
||||
LEFT JOIN documents d ON d.collection = r.collection AND d.owner_id = r.owner_id
|
||||
GROUP BY r.collection, m.source
|
||||
ORDER BY r.collection
|
||||
""",
|
||||
(_READ_PERMISSION, *principals),
|
||||
).fetchall()
|
||||
return [CollectionSummary(collection=r[0], source=r[1], chunks=int(r[2])) for r in rows]
|
||||
|
||||
async def stats(self) -> StoreStats:
|
||||
async with self._lock:
|
||||
return await asyncio.to_thread(self._sync_stats)
|
||||
|
||||
def _sync_stats(self) -> StoreStats:
|
||||
documents = self._conn.execute("SELECT COUNT(DISTINCT collection) FROM documents_meta").fetchone()[0]
|
||||
chunks = self._conn.execute("SELECT COUNT(*) FROM documents").fetchone()[0]
|
||||
return StoreStats(documents=int(documents), chunks=int(chunks))
|
||||
|
||||
async def close(self) -> None:
|
||||
async with self._lock:
|
||||
await asyncio.to_thread(self._sync_close)
|
||||
|
||||
@@ -34,6 +34,23 @@ class StoredPage:
|
||||
char_count: int
|
||||
|
||||
|
||||
@dataclass
|
||||
class StoreStats:
|
||||
"""Deployment-wide counts: distinct document ids and total vector-chunk rows."""
|
||||
|
||||
documents: int
|
||||
chunks: int
|
||||
|
||||
|
||||
@dataclass
|
||||
class CollectionSummary:
|
||||
"""Rollup row for one readable collection: stored source label + chunk count."""
|
||||
|
||||
collection: str
|
||||
source: str
|
||||
chunks: int
|
||||
|
||||
|
||||
class DocumentStore(ABC):
|
||||
"""Abstract interface for document storage backends.
|
||||
|
||||
@@ -148,6 +165,20 @@ class DocumentStore(ABC):
|
||||
async def list_collections(self, principals: list[PrincipalId]) -> list[str]:
|
||||
"""Return collection names readable by at least one of ``principals``."""
|
||||
|
||||
@abstractmethod
|
||||
async def list_collection_summaries(self, principals: list[PrincipalId]) -> list[CollectionSummary]:
|
||||
"""Per-collection rollup (source + chunk count) readable by ``principals``.
|
||||
|
||||
Counts cover the same owner's copy a read would resolve to, so the
|
||||
rollup never leaks another tenant's content.
|
||||
"""
|
||||
|
||||
# ── deployment-wide stats (not tenant-scoped) ──────────────────────────
|
||||
|
||||
@abstractmethod
|
||||
async def stats(self) -> StoreStats:
|
||||
"""Count every owner's content: distinct document ids + total chunk rows."""
|
||||
|
||||
# ── lifecycle ──────────────────────────────────────────────────────────
|
||||
|
||||
@abstractmethod
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import io
|
||||
from typing import Any
|
||||
|
||||
import docx
|
||||
|
||||
from stirling.contracts.docparse import FillDocxRequest
|
||||
from stirling.docparse.docxfill import fill_docx
|
||||
|
||||
|
||||
def _template_base64() -> str:
|
||||
document = docx.Document()
|
||||
document.add_paragraph("Dear {{ customer.name }},")
|
||||
document.add_paragraph("Your total is {{ total }}.")
|
||||
document.add_paragraph("Unknown: {{ nowhere.field }}")
|
||||
table = document.add_table(rows=2, cols=2)
|
||||
table.rows[0].cells[0].text = "Item"
|
||||
table.rows[0].cells[1].text = "Price"
|
||||
table.rows[1].cells[0].text = "{{#items.name}}"
|
||||
table.rows[1].cells[1].text = "{{#items.price}}"
|
||||
buffer = io.BytesIO()
|
||||
document.save(buffer)
|
||||
return base64.b64encode(buffer.getvalue()).decode("ascii")
|
||||
|
||||
|
||||
def _load(response_base64: str) -> Any:
|
||||
return docx.Document(io.BytesIO(base64.b64decode(response_base64)))
|
||||
|
||||
|
||||
def test_fills_scalars_tables_and_reports_missing() -> None:
|
||||
request = FillDocxRequest(
|
||||
template_base64=_template_base64(),
|
||||
data={
|
||||
"customer": {"name": "ACME GmbH"},
|
||||
"total": 12.5,
|
||||
"items": [
|
||||
{"name": "Widget", "price": "2.00"},
|
||||
{"name": "Gadget", "price": "10.50"},
|
||||
],
|
||||
},
|
||||
)
|
||||
response = fill_docx(request)
|
||||
filled = _load(response.docx_base64)
|
||||
|
||||
paragraphs = [p.text for p in filled.paragraphs]
|
||||
assert "Dear ACME GmbH," in paragraphs
|
||||
assert "Your total is 12.5." in paragraphs
|
||||
# Unresolved placeholders stay put and are reported.
|
||||
assert any("{{ nowhere.field }}" in p for p in paragraphs)
|
||||
assert response.missing == ["nowhere.field"]
|
||||
|
||||
table = filled.tables[0]
|
||||
rendered_rows = [[cell.text for cell in row.cells] for row in table.rows]
|
||||
assert ["Widget", "2.00"] in rendered_rows
|
||||
assert ["Gadget", "10.50"] in rendered_rows
|
||||
# The template row is gone.
|
||||
assert all("{{#" not in cell for row in rendered_rows for cell in row)
|
||||
assert response.replaced >= 6
|
||||
|
||||
|
||||
def test_empty_items_removes_template_row() -> None:
|
||||
request = FillDocxRequest(
|
||||
template_base64=_template_base64(),
|
||||
data={"customer": {"name": "X"}, "total": 1, "items": []},
|
||||
)
|
||||
response = fill_docx(request)
|
||||
filled = _load(response.docx_base64)
|
||||
assert len(filled.tables[0].rows) == 1 # only the header remains
|
||||
@@ -9,6 +9,7 @@ from stirling.docparse.extractor import (
|
||||
build_output_model,
|
||||
find_quote,
|
||||
)
|
||||
from stirling.docparse.splitter import _Boundary, _SplitOutput, validate_boundaries
|
||||
|
||||
|
||||
def _pages() -> list[PageText]:
|
||||
@@ -43,6 +44,33 @@ def test_find_quote_missing_returns_none() -> None:
|
||||
assert find_quote(" ", _pages()) is None
|
||||
|
||||
|
||||
def test_validate_boundaries_partitions_cleanly() -> None:
|
||||
output = _SplitOutput(
|
||||
boundaries=[
|
||||
_Boundary(start_page=4, label="Invoice B", confidence=0.8),
|
||||
_Boundary(start_page=1, label="Invoice A", confidence=0.9),
|
||||
_Boundary(start_page=4, label="dup", confidence=0.1),
|
||||
_Boundary(start_page=99, label="out of range", confidence=0.5),
|
||||
]
|
||||
)
|
||||
parts = validate_boundaries(output, page_count=6, max_parts=10)
|
||||
assert [(p.start_page, p.end_page) for p in parts] == [(1, 3), (4, 6)]
|
||||
assert parts[0].label == "Invoice A"
|
||||
|
||||
|
||||
def test_validate_boundaries_inserts_page_one() -> None:
|
||||
output = _SplitOutput(boundaries=[_Boundary(start_page=3, label="Part", confidence=0.7)])
|
||||
parts = validate_boundaries(output, page_count=5, max_parts=10)
|
||||
assert parts[0].start_page == 1
|
||||
assert parts[1].start_page == 3
|
||||
assert parts[-1].end_page == 5
|
||||
|
||||
|
||||
def test_validate_boundaries_empty_output_spans_whole_file() -> None:
|
||||
parts = validate_boundaries(_SplitOutput(), page_count=7, max_parts=10)
|
||||
assert [(p.start_page, p.end_page) for p in parts] == [(1, 7)]
|
||||
|
||||
|
||||
def _answers(quote: str | None, confidence: float) -> BaseModel:
|
||||
model = build_output_model({"type": "object", "properties": {"invoice_number": {"type": "string"}}})
|
||||
return model.model_validate({"invoiceNumber": {"value": "INV-123", "quote": quote, "confidence": confidence}})
|
||||
|
||||
@@ -7,7 +7,7 @@ from stirling.documents.chunker import chunk_text
|
||||
from stirling.documents.rag_capability import RagCapability
|
||||
from stirling.documents.service import DocumentService
|
||||
from stirling.documents.sqlite_vec_store import SqliteVecStore
|
||||
from stirling.documents.store import Document, SearchResult
|
||||
from stirling.documents.store import CollectionSummary, Document, SearchResult
|
||||
from stirling.models import FileId, OwnerId, PrincipalId
|
||||
|
||||
# Personal-doc tests reuse the same opaque string in all three roles — keeps the
|
||||
@@ -178,6 +178,27 @@ class TestSqliteVecStore:
|
||||
assert await store.list_collections(OWNER_PRINCIPALS) == []
|
||||
assert await store.list_collections(OTHER_OWNER_PRINCIPALS) == ["c.pdf"]
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_stats_count_distinct_document_ids_and_chunk_rows(self) -> None:
|
||||
"""Stats span every owner; a document id shared by two owners counts once."""
|
||||
store = SqliteVecStore.ephemeral()
|
||||
empty = await store.stats()
|
||||
assert (empty.documents, empty.chunks) == (0, 0)
|
||||
|
||||
for owner, principals, name, texts in (
|
||||
(OWNER, OWNER_PRINCIPALS, "doc-a", ["one", "two"]),
|
||||
(OWNER, OWNER_PRINCIPALS, "doc-b", ["three"]),
|
||||
(OTHER_OWNER, OTHER_OWNER_PRINCIPALS, "doc-b", ["four"]),
|
||||
):
|
||||
await store.ensure_collection(name, f"{name}.pdf", owner, None)
|
||||
await store.grant_read(name, owner, principals)
|
||||
docs = [Document(id=str(i), text=t, metadata={}) for i, t in enumerate(texts)]
|
||||
await store.add_documents(name, docs, [[1.0, 0.0]] * len(docs), owner)
|
||||
|
||||
stats = await store.stats()
|
||||
assert stats.documents == 2
|
||||
assert stats.chunks == 4
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_reap_expired_drops_collections_past_expires_at(self) -> None:
|
||||
"""TTL backstop: rows with ``expires_at`` in the past go away on reap."""
|
||||
@@ -239,6 +260,47 @@ class TestSqliteVecStore:
|
||||
# Owner still can.
|
||||
assert await store.has_collection("doc", OWNER_PRINCIPALS) is True
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_list_collection_summaries_rolls_up_readable_collections(self) -> None:
|
||||
"""Rollup: one row per readable collection with its source and chunk count."""
|
||||
store = SqliteVecStore.ephemeral()
|
||||
await store.ensure_collection("doc-a", "a.pdf", OWNER, None)
|
||||
await store.grant_read("doc-a", OWNER, OWNER_PRINCIPALS)
|
||||
docs = [Document(id="1", text="one", metadata={}), Document(id="2", text="two", metadata={})]
|
||||
await store.add_documents("doc-a", docs, [[1.0, 0.0], [0.0, 1.0]], OWNER)
|
||||
# Collection with no vector chunks yet: still listed, zero count.
|
||||
await store.ensure_collection("doc-b", "b.pdf", OWNER, None)
|
||||
await store.grant_read("doc-b", OWNER, OWNER_PRINCIPALS)
|
||||
|
||||
summaries = await store.list_collection_summaries(OWNER_PRINCIPALS)
|
||||
assert summaries == [
|
||||
CollectionSummary(collection="doc-a", source="a.pdf", chunks=2),
|
||||
CollectionSummary(collection="doc-b", source="b.pdf", chunks=0),
|
||||
]
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_list_collection_summaries_scoped_to_principals(self) -> None:
|
||||
"""One principal's rollup never lists, or counts, another owner's copy."""
|
||||
store = SqliteVecStore.ephemeral()
|
||||
await store.ensure_collection("shared-id", "alice.pdf", OWNER, None)
|
||||
await store.grant_read("shared-id", OWNER, OWNER_PRINCIPALS)
|
||||
await store.add_documents("shared-id", [Document(id="1", text="alice", metadata={})], [[1.0, 0.0]], OWNER)
|
||||
await store.ensure_collection("shared-id", "bob.pdf", OTHER_OWNER, None)
|
||||
await store.grant_read("shared-id", OTHER_OWNER, OTHER_OWNER_PRINCIPALS)
|
||||
bob_docs = [Document(id="1", text="bob", metadata={}), Document(id="2", text="bob2", metadata={})]
|
||||
await store.add_documents("shared-id", bob_docs, [[1.0, 0.0], [0.0, 1.0]], OTHER_OWNER)
|
||||
await store.ensure_collection("bob-only", "bob-only.pdf", OTHER_OWNER, None)
|
||||
await store.grant_read("bob-only", OTHER_OWNER, OTHER_OWNER_PRINCIPALS)
|
||||
|
||||
assert await store.list_collection_summaries(OWNER_PRINCIPALS) == [
|
||||
CollectionSummary(collection="shared-id", source="alice.pdf", chunks=1)
|
||||
]
|
||||
assert await store.list_collection_summaries(OTHER_OWNER_PRINCIPALS) == [
|
||||
CollectionSummary(collection="bob-only", source="bob-only.pdf", chunks=0),
|
||||
CollectionSummary(collection="shared-id", source="bob.pdf", chunks=2),
|
||||
]
|
||||
assert await store.list_collection_summaries([]) == []
|
||||
|
||||
|
||||
# DocumentService (with stub embedder)
|
||||
|
||||
@@ -398,6 +460,31 @@ class TestDocumentService:
|
||||
multi_results = await documents.search("deploy", principals=[hr_group, eng_group])
|
||||
assert len(multi_results) > 0
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_search_with_collections_tags_results_and_respects_acl(self, documents: DocumentService) -> None:
|
||||
"""Collection-tagged search only reaches collections the caller can read."""
|
||||
await documents.ingest(
|
||||
FileId("col-a"),
|
||||
_pages("Alpha content."),
|
||||
source="a.pdf",
|
||||
owner_id=OWNER,
|
||||
read_principals=OWNER_PRINCIPALS,
|
||||
expires_at=None,
|
||||
)
|
||||
await documents.ingest(
|
||||
FileId("col-b"),
|
||||
_pages("Beta content."),
|
||||
source="b.pdf",
|
||||
owner_id=OTHER_OWNER,
|
||||
read_principals=OTHER_OWNER_PRINCIPALS,
|
||||
expires_at=None,
|
||||
)
|
||||
|
||||
hits = await documents.search_with_collections("content", principals=OWNER_PRINCIPALS)
|
||||
assert hits
|
||||
assert {hit.collection for hit in hits} == {"col-a"}
|
||||
assert all(hit.result.document.text for hit in hits)
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_delete_collection(self, documents: DocumentService) -> None:
|
||||
await documents.ingest(
|
||||
|
||||
@@ -6,9 +6,10 @@ import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from stirling.api import app
|
||||
from stirling.api.dependencies import get_document_service
|
||||
from stirling.api.dependencies import get_document_service, get_knowledge_ask_agent
|
||||
from stirling.contracts import AskDocumentsRequest, AskDocumentsResponse, DocumentPassage
|
||||
from stirling.documents import Document, DocumentService, SqliteVecStore
|
||||
from stirling.models import FileId, PrincipalId, UserId
|
||||
from stirling.models import FileId, OwnerId, PrincipalId, UserId
|
||||
|
||||
USER = UserId("test-user")
|
||||
USER_PRINCIPALS = [PrincipalId("test-user")]
|
||||
@@ -348,6 +349,274 @@ def test_purge_by_owner_rejects_missing_user_header(client: TestClient) -> None:
|
||||
assert response.status_code == 401
|
||||
|
||||
|
||||
# ── GET /documents/list ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _ingest(client: TestClient, document_id: str, source: str, texts: list[str], owner: str) -> None:
|
||||
client.post(
|
||||
"/api/v1/documents",
|
||||
json={
|
||||
"documentId": document_id,
|
||||
"source": source,
|
||||
"pageText": [{"pageNumber": i, "text": t} for i, t in enumerate(texts, 1)],
|
||||
"ownerId": owner,
|
||||
"readPrincipals": [owner],
|
||||
"expiresAt": None,
|
||||
},
|
||||
headers={"X-User-Id": owner},
|
||||
)
|
||||
|
||||
|
||||
def test_list_documents_returns_caller_rollup(client: TestClient) -> None:
|
||||
_ingest(client, "list-a", "a.pdf", ["Page one text.", "Page two text."], USER)
|
||||
_ingest(client, "list-b", "b.pdf", ["Only page."], USER)
|
||||
|
||||
response = client.get("/api/v1/documents/list", headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
documents = response.json()["documents"]
|
||||
assert [d["documentId"] for d in documents] == ["list-a", "list-b"]
|
||||
by_id = {d["documentId"]: d for d in documents}
|
||||
assert by_id["list-a"]["source"] == "a.pdf"
|
||||
assert by_id["list-a"]["chunks"] >= 2
|
||||
assert by_id["list-b"]["source"] == "b.pdf"
|
||||
assert by_id["list-b"]["chunks"] >= 1
|
||||
|
||||
|
||||
def test_list_documents_empty_for_new_user(client: TestClient) -> None:
|
||||
response = client.get("/api/v1/documents/list", headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
assert response.json() == {"documents": []}
|
||||
|
||||
|
||||
def test_list_documents_hides_other_users_documents(client: TestClient) -> None:
|
||||
"""User A must never see user B's documents in the rollup."""
|
||||
_ingest(client, "alice-doc", "alice.pdf", ["alice content"], "alice")
|
||||
_ingest(client, "bob-doc", "bob.pdf", ["bob content"], "bob")
|
||||
|
||||
alice_docs = client.get("/api/v1/documents/list", headers={"X-User-Id": "alice"}).json()["documents"]
|
||||
bob_docs = client.get("/api/v1/documents/list", headers={"X-User-Id": "bob"}).json()["documents"]
|
||||
assert [d["documentId"] for d in alice_docs] == ["alice-doc"]
|
||||
assert [d["documentId"] for d in bob_docs] == ["bob-doc"]
|
||||
|
||||
|
||||
def test_list_documents_rejects_missing_user_header(client: TestClient) -> None:
|
||||
assert client.get("/api/v1/documents/list").status_code == 401
|
||||
|
||||
|
||||
# ── POST /documents/search ──────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_search_documents_maps_page_text_chunks(client: TestClient) -> None:
|
||||
"""Plain ingested chunks only carry page_number: both bounds map to it and
|
||||
the ":page:N" suffix is stripped off the source."""
|
||||
client.post(
|
||||
"/api/v1/documents",
|
||||
json={
|
||||
"documentId": "report",
|
||||
"source": "report.pdf",
|
||||
"pageText": [{"pageNumber": 3, "text": "The launch is planned for October."}],
|
||||
"ownerId": USER,
|
||||
"readPrincipals": [USER],
|
||||
"expiresAt": None,
|
||||
},
|
||||
headers=HEADERS,
|
||||
)
|
||||
|
||||
response = client.post("/api/v1/documents/search", json={"query": "launch", "topK": 5}, headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
passages = response.json()["passages"]
|
||||
assert len(passages) >= 1
|
||||
passage = passages[0]
|
||||
assert passage["documentId"] == "report"
|
||||
assert passage["pageStart"] == 3
|
||||
assert passage["pageEnd"] == 3
|
||||
assert passage["headingPath"] == []
|
||||
assert passage["source"] == "report.pdf"
|
||||
assert "launch" in passage["text"]
|
||||
assert isinstance(passage["score"], float)
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_search_documents_maps_docparse_chunk_metadata(client: TestClient, service: DocumentService) -> None:
|
||||
"""Docparse chunks carry page bounds + heading path; they map straight onto the wire."""
|
||||
await service.ingest_prepared(
|
||||
collection=FileId("dp-doc"),
|
||||
chunks=[
|
||||
(
|
||||
"Revenue grew 12% in Q2.",
|
||||
{
|
||||
"content_type": "docparse_chunk",
|
||||
"page_start": "2",
|
||||
"page_end": "3",
|
||||
"heading_path": "Report > Finance",
|
||||
},
|
||||
)
|
||||
],
|
||||
source="q2.pdf",
|
||||
owner_id=OwnerId(USER),
|
||||
read_principals=USER_PRINCIPALS,
|
||||
expires_at=None,
|
||||
)
|
||||
|
||||
response = client.post("/api/v1/documents/search", json={"query": "revenue"}, headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
passage = response.json()["passages"][0]
|
||||
assert passage["documentId"] == "dp-doc"
|
||||
assert passage["pageStart"] == 2
|
||||
assert passage["pageEnd"] == 3
|
||||
assert passage["headingPath"] == ["Report", "Finance"]
|
||||
assert passage["source"] == "q2.pdf"
|
||||
|
||||
|
||||
def test_search_documents_cannot_see_other_users_documents(client: TestClient) -> None:
|
||||
"""User B searching for user A's content must get nothing back."""
|
||||
_ingest(client, "alice-doc", "alice.pdf", ["The secret launch code is October."], "alice")
|
||||
|
||||
bob = client.post("/api/v1/documents/search", json={"query": "secret launch"}, headers={"X-User-Id": "bob"})
|
||||
assert bob.status_code == 200
|
||||
assert bob.json()["passages"] == []
|
||||
|
||||
alice = client.post("/api/v1/documents/search", json={"query": "secret launch"}, headers={"X-User-Id": "alice"})
|
||||
assert alice.json()["passages"] != []
|
||||
|
||||
|
||||
def test_search_documents_rejects_empty_query(client: TestClient) -> None:
|
||||
response = client.post("/api/v1/documents/search", json={"query": ""}, headers=HEADERS)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_search_documents_rejects_top_k_above_cap(client: TestClient) -> None:
|
||||
response = client.post("/api/v1/documents/search", json={"query": "x", "topK": 51}, headers=HEADERS)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_search_documents_rejects_missing_user_header(client: TestClient) -> None:
|
||||
assert client.post("/api/v1/documents/search", json={"query": "x"}).status_code == 401
|
||||
|
||||
|
||||
# ── POST /documents/ask ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
class StubKnowledgeAskAgent:
|
||||
"""Stands in for KnowledgeAskAgent so route tests don't call a model."""
|
||||
|
||||
def __init__(self, response: AskDocumentsResponse) -> None:
|
||||
self._response = response
|
||||
self.calls: list[tuple[AskDocumentsRequest, list[PrincipalId]]] = []
|
||||
|
||||
async def ask(self, request: AskDocumentsRequest, principals: list[PrincipalId]) -> AskDocumentsResponse:
|
||||
self.calls.append((request, principals))
|
||||
return self._response
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ask_agent() -> StubKnowledgeAskAgent:
|
||||
return StubKnowledgeAskAgent(
|
||||
AskDocumentsResponse(
|
||||
answer="Revenue grew 12% (q2.pdf p.2).",
|
||||
passages=[
|
||||
DocumentPassage(
|
||||
document_id=FileId("dp-doc"),
|
||||
text="Revenue grew 12% in Q2.",
|
||||
score=0.91,
|
||||
page_start=2,
|
||||
page_end=3,
|
||||
heading_path=["Report", "Finance"],
|
||||
source="q2.pdf",
|
||||
)
|
||||
],
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def ask_client(client: TestClient, ask_agent: StubKnowledgeAskAgent) -> Iterator[TestClient]:
|
||||
app.dependency_overrides[get_knowledge_ask_agent] = lambda: ask_agent
|
||||
try:
|
||||
yield client
|
||||
finally:
|
||||
app.dependency_overrides.pop(get_knowledge_ask_agent, None)
|
||||
|
||||
|
||||
def test_ask_documents_returns_answer_and_passages(ask_client: TestClient) -> None:
|
||||
response = ask_client.post("/api/v1/documents/ask", json={"question": "How did revenue do?"}, headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["answer"] == "Revenue grew 12% (q2.pdf p.2)."
|
||||
assert body["passages"] == [
|
||||
{
|
||||
"documentId": "dp-doc",
|
||||
"text": "Revenue grew 12% in Q2.",
|
||||
"score": 0.91,
|
||||
"pageStart": 2,
|
||||
"pageEnd": 3,
|
||||
"headingPath": ["Report", "Finance"],
|
||||
"source": "q2.pdf",
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
def test_ask_documents_scopes_to_calling_user(ask_client: TestClient, ask_agent: StubKnowledgeAskAgent) -> None:
|
||||
"""The route hands the agent exactly the caller's principal set."""
|
||||
ask_client.post("/api/v1/documents/ask", json={"question": "anything"}, headers=HEADERS)
|
||||
request, principals = ask_agent.calls[0]
|
||||
assert principals == [PrincipalId(USER)]
|
||||
assert request.top_k == 8
|
||||
|
||||
|
||||
def test_ask_documents_rejects_empty_question(ask_client: TestClient) -> None:
|
||||
response = ask_client.post("/api/v1/documents/ask", json={"question": ""}, headers=HEADERS)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_ask_documents_rejects_top_k_above_cap(ask_client: TestClient) -> None:
|
||||
response = ask_client.post("/api/v1/documents/ask", json={"question": "x", "topK": 21}, headers=HEADERS)
|
||||
assert response.status_code == 422
|
||||
|
||||
|
||||
def test_ask_documents_rejects_missing_user_header(ask_client: TestClient) -> None:
|
||||
assert ask_client.post("/api/v1/documents/ask", json={"question": "x"}).status_code == 401
|
||||
|
||||
|
||||
# ── GET /documents/stats ────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_stats_on_empty_store_reports_zero(client: TestClient) -> None:
|
||||
response = client.get("/api/v1/documents/stats", headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["documents"] == 0
|
||||
assert body["chunks"] == 0
|
||||
assert body["backend"] in ("sqlite", "pgvector")
|
||||
assert body["embeddingModel"]
|
||||
|
||||
|
||||
def test_stats_counts_seeded_documents_across_owners(client: TestClient) -> None:
|
||||
"""Stats are deployment-wide: both owners' content is counted."""
|
||||
for owner, doc in (("alice", "doc-a"), ("bob", "doc-b")):
|
||||
client.post(
|
||||
"/api/v1/documents",
|
||||
json={
|
||||
"documentId": doc,
|
||||
"source": f"{doc}.pdf",
|
||||
"pageText": [{"pageNumber": 1, "text": "Some content for the stats endpoint."}],
|
||||
"ownerId": owner,
|
||||
"readPrincipals": [owner],
|
||||
"expiresAt": None,
|
||||
},
|
||||
headers={"X-User-Id": owner},
|
||||
)
|
||||
response = client.get("/api/v1/documents/stats", headers=HEADERS)
|
||||
assert response.status_code == 200
|
||||
body = response.json()
|
||||
assert body["documents"] == 2
|
||||
assert body["chunks"] >= 2
|
||||
|
||||
|
||||
def test_stats_rejects_missing_user_header(client: TestClient) -> None:
|
||||
assert client.get("/api/v1/documents/stats").status_code == 401
|
||||
|
||||
|
||||
def test_delete_document_only_affects_calling_user(client: TestClient) -> None:
|
||||
"""Two users with the same document id: one user's delete must not remove the other's."""
|
||||
alice_body = {
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from stirling.agents.knowledge_ask import KnowledgeAskAgent, format_passages, passage_from_hit
|
||||
from stirling.config import AppSettings
|
||||
from stirling.contracts import AskDocumentsRequest, DocumentPassage
|
||||
from stirling.documents import CollectionSearchHit, Document, DocumentService, SearchResult, SqliteVecStore
|
||||
from stirling.models import FileId, PrincipalId
|
||||
from stirling.services import build_runtime
|
||||
|
||||
PRINCIPALS = [PrincipalId("test-user")]
|
||||
|
||||
|
||||
def _hit(metadata: dict[str, str], text: str = "chunk text", score: float = 0.8) -> CollectionSearchHit:
|
||||
return CollectionSearchHit(
|
||||
collection=FileId("doc-1"),
|
||||
result=SearchResult(document=Document(id="c1", text=text, metadata=metadata), score=score),
|
||||
)
|
||||
|
||||
|
||||
# ── passage_from_hit ────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_passage_from_hit_maps_docparse_metadata() -> None:
|
||||
passage = passage_from_hit(
|
||||
_hit({"source": "q2.pdf", "page_start": "2", "page_end": "3", "heading_path": "Report > Finance"})
|
||||
)
|
||||
assert passage.document_id == "doc-1"
|
||||
assert passage.page_start == 2
|
||||
assert passage.page_end == 3
|
||||
assert passage.heading_path == ["Report", "Finance"]
|
||||
assert passage.source == "q2.pdf"
|
||||
|
||||
|
||||
def test_passage_from_hit_falls_back_to_page_number() -> None:
|
||||
"""Plain page-text chunks: page_number fills both bounds, source drops the page suffix."""
|
||||
passage = passage_from_hit(_hit({"source": "report.pdf:page:4", "page_number": "4"}))
|
||||
assert passage.page_start == 4
|
||||
assert passage.page_end == 4
|
||||
assert passage.heading_path == []
|
||||
assert passage.source == "report.pdf"
|
||||
|
||||
|
||||
def test_passage_from_hit_tolerates_missing_and_bad_metadata() -> None:
|
||||
passage = passage_from_hit(_hit({"page_start": "not-a-number"}))
|
||||
assert passage.page_start is None
|
||||
assert passage.page_end is None
|
||||
assert passage.heading_path == []
|
||||
assert passage.source is None
|
||||
|
||||
|
||||
# ── format_passages ─────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_format_passages_includes_citation_handles() -> None:
|
||||
passages = [
|
||||
DocumentPassage(document_id=FileId("d1"), text="Alpha.", score=0.9, page_start=2, page_end=3, source="a.pdf"),
|
||||
DocumentPassage(document_id=FileId("d2"), text="Beta.", score=0.5, page_start=7, page_end=7, source="b.pdf"),
|
||||
DocumentPassage(document_id=FileId("d3"), text="Gamma.", score=0.4),
|
||||
]
|
||||
rendered = format_passages(passages)
|
||||
assert "[Passage 1 | a.pdf p.2-3]\nAlpha." in rendered
|
||||
assert "[Passage 2 | b.pdf p.7]\nBeta." in rendered
|
||||
# No source or pages: fall back to the document id alone.
|
||||
assert "[Passage 3 | d3]\nGamma." in rendered
|
||||
|
||||
|
||||
# ── KnowledgeAskAgent ───────────────────────────────────────────────────
|
||||
|
||||
|
||||
class _StubEmbedder:
|
||||
"""Deterministic embeddings so the agent test needs no provider."""
|
||||
|
||||
async def embed_query(self, text: str) -> list[float]:
|
||||
return [1.0, 0.0]
|
||||
|
||||
async def embed_documents(self, texts: list[str]) -> list[list[float]]:
|
||||
return [[1.0, 0.0] for _ in texts]
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_ask_answers_plainly_when_nothing_retrieved(app_settings: AppSettings) -> None:
|
||||
"""Empty retrieval short-circuits: no model call, honest not-found answer."""
|
||||
documents = DocumentService(embedder=_StubEmbedder(), store=SqliteVecStore.ephemeral(), default_top_k=3) # type: ignore[arg-type]
|
||||
runtime = build_runtime(app_settings, documents=documents)
|
||||
agent = KnowledgeAskAgent(runtime)
|
||||
|
||||
response = await agent.ask(AskDocumentsRequest(question="What is the launch date?"), principals=PRINCIPALS)
|
||||
assert response.passages == []
|
||||
assert "couldn't find" in response.answer
|
||||
Generated
+11
-9
@@ -494,14 +494,14 @@ name = "cohere"
|
||||
version = "7.0.4"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
dependencies = [
|
||||
{ name = "fastavro" },
|
||||
{ name = "httpx" },
|
||||
{ name = "pydantic" },
|
||||
{ name = "pydantic-core" },
|
||||
{ name = "requests" },
|
||||
{ name = "tokenizers" },
|
||||
{ name = "types-requests" },
|
||||
{ name = "typing-extensions" },
|
||||
{ name = "fastavro", marker = "sys_platform != 'emscripten'" },
|
||||
{ name = "httpx", marker = "sys_platform != 'emscripten'" },
|
||||
{ name = "pydantic", marker = "sys_platform != 'emscripten'" },
|
||||
{ name = "pydantic-core", marker = "sys_platform != 'emscripten'" },
|
||||
{ name = "requests", marker = "sys_platform != 'emscripten'" },
|
||||
{ name = "tokenizers", marker = "sys_platform != 'emscripten'" },
|
||||
{ name = "types-requests", marker = "sys_platform != 'emscripten'" },
|
||||
{ name = "typing-extensions", marker = "sys_platform != 'emscripten'" },
|
||||
]
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/cf/3c/670631ee223d7b64d157dc3f309bf93bde65efe0bb1a8341d9b575f407d3/cohere-7.0.4.tar.gz", hash = "sha256:35b6a397d35ae6eafa1a02921f42c2a98309a990874533e5238efaf3426b6a21", size = 208794, upload-time = "2026-06-11T15:17:52.994Z" }
|
||||
wheels = [
|
||||
@@ -855,6 +855,7 @@ dependencies = [
|
||||
{ name = "pydantic-ai" },
|
||||
{ name = "pydantic-ai-slim", extra = ["voyageai"] },
|
||||
{ name = "pydantic-settings" },
|
||||
{ name = "python-docx" },
|
||||
{ name = "python-dotenv" },
|
||||
{ name = "sqlite-vec" },
|
||||
{ name = "uvicorn" },
|
||||
@@ -892,6 +893,7 @@ requires-dist = [
|
||||
{ name = "pydantic-ai", specifier = ">=1.99.0,<2.0.0" },
|
||||
{ name = "pydantic-ai-slim", extras = ["voyageai"], specifier = ">=1.99.0,<2.0.0" },
|
||||
{ name = "pydantic-settings", specifier = ">=2.0.0" },
|
||||
{ name = "python-docx", specifier = ">=1.1.2" },
|
||||
{ name = "python-dotenv", specifier = ">=1.2.1" },
|
||||
{ name = "sqlite-vec", specifier = ">=0.1.6" },
|
||||
{ name = "torch", marker = "sys_platform == 'linux' and extra == 'docparse'", specifier = ">=2.6.0", index = "https://download.pytorch.org/whl/cpu" },
|
||||
@@ -4360,7 +4362,7 @@ name = "types-requests"
|
||||
version = "2.33.0.20260518"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
dependencies = [
|
||||
{ name = "urllib3" },
|
||||
{ name = "urllib3", marker = "sys_platform != 'emscripten'" },
|
||||
]
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/e0/01/c5a19253fe1ac159159ddf9a3a07cec8bb5e486ec4d9002ad2821da0e5d2/types_requests-2.33.0.20260518.tar.gz", hash = "sha256:df7bd3bfe0ca8402dfb841e7d9be714bb5578203283d66d7dc4ef69343449a5e", size = 24752, upload-time = "2026-05-18T06:07:37.966Z" }
|
||||
wheels = [
|
||||
|
||||
@@ -2986,6 +2986,38 @@ summary_one = "Ran 1 tool"
|
||||
summary_other = "Ran {{count}} tools"
|
||||
unknownTool = "Unknown tool"
|
||||
|
||||
[chunkDocument]
|
||||
intro = "Turns a document into retrieval-ready chunks in three layers, so answers cite the right section instead of a random page."
|
||||
processorCallout = "To index automatically, add the 'Index into knowledge base' step to an ingestion policy in the"
|
||||
processorLink = "Processor"
|
||||
submit = "Prepare chunks"
|
||||
|
||||
[chunkDocument.chunkSize]
|
||||
label = "Chunk size (characters)"
|
||||
|
||||
[chunkDocument.error]
|
||||
failed = "Failed to chunk document"
|
||||
|
||||
[chunkDocument.layers]
|
||||
chunk = "Structure-aware chunks: each carries its heading breadcrumb and page range"
|
||||
embed = "Ready to embed: exported as JSONL for any vector store"
|
||||
parse = "Layout-aware parse: headings, paragraphs, and tables are recognized as structure"
|
||||
|
||||
[chunkDocument.mode]
|
||||
advanced = "Advanced"
|
||||
auto = "Auto"
|
||||
basic = "Basic"
|
||||
label = "Mode"
|
||||
|
||||
[chunkDocument.overlap]
|
||||
label = "Overlap (characters)"
|
||||
|
||||
[chunkDocument.results]
|
||||
title = "Chunks (JSONL)"
|
||||
|
||||
[chunkDocument.settings]
|
||||
title = "Chunking settings"
|
||||
|
||||
[cloudBadge]
|
||||
tooltip = "This operation will use your cloud credits"
|
||||
|
||||
@@ -4240,6 +4272,24 @@ upload = "Upload"
|
||||
uploadFile = "Upload File"
|
||||
uploadFiles = "Upload Files"
|
||||
|
||||
[fillTemplate]
|
||||
hint = "The input file must be a .docx template (not a PDF). Each placeholder in the template is replaced with the matching JSON value."
|
||||
intro = "Replaces the placeholders in a Word (.docx) template with your JSON data and returns the filled document - deterministic, no AI involved."
|
||||
submit = "Fill template"
|
||||
|
||||
[fillTemplate.data]
|
||||
invalid = "Enter a valid JSON object"
|
||||
label = "Data (JSON)"
|
||||
|
||||
[fillTemplate.error]
|
||||
failed = "Failed to fill template"
|
||||
|
||||
[fillTemplate.results]
|
||||
title = "Filled document"
|
||||
|
||||
[fillTemplate.settings]
|
||||
title = "Template data"
|
||||
|
||||
[firstLogin]
|
||||
allFieldsRequired = "All fields are required"
|
||||
changePassword = "Change Password"
|
||||
@@ -4579,6 +4629,11 @@ desc = "Change document restrictions and permissions"
|
||||
tags = "permissions,restrictions,rights,access control,allow,deny,printing,copying,editing,modify permissions,security settings,user rights"
|
||||
title = "Change Permissions"
|
||||
|
||||
[home.chunkDocument]
|
||||
desc = "Layout-aware parse to structure-aware chunks with heading breadcrumbs and page ranges, ready to embed"
|
||||
tags = "chunk,RAG,prepare,split text,segments,embedding,vector,ingest,LLM,retrieval,JSONL,overlap,knowledge base,index"
|
||||
title = "Prepare for RAG"
|
||||
|
||||
[home.compare]
|
||||
desc = "Compares and shows the differences between 2 PDF Documents"
|
||||
tags = "difference,compare,diff,compare PDFs,compare documents,find differences,show differences,changes,what changed,track changes,revisions,version compare,side by side,contrast,delta"
|
||||
@@ -4639,6 +4694,11 @@ desc = "Extract specific pages from a PDF document"
|
||||
tags = "pull,select,copy,extract,extract pages,get pages,pull out,save pages,export pages,copy pages,select pages,specific pages"
|
||||
title = "Extract Pages"
|
||||
|
||||
[home.fillTemplate]
|
||||
desc = "Fill a DOCX template's placeholders from JSON data"
|
||||
tags = "template,DOCX,fill,merge fields,mail merge,generate document,placeholders,letters,contracts,Word"
|
||||
title = "Fill Template"
|
||||
|
||||
[home.flatten]
|
||||
desc = "Remove all interactive elements and forms from a PDF"
|
||||
tags = "simplify,remove,interactive,flatten,flatten form,remove form fields,make static,finalize form,lock form,disable editing,convert to image,non-editable"
|
||||
@@ -4692,6 +4752,11 @@ desc = "Merge multiple pages of a PDF document into a single page"
|
||||
tags = "layout,arrange,combine,N-up,2-up,4-up,multiple per page,pages per sheet,layout pages,tile,grid layout,multi-page layout,combine on page,handout"
|
||||
title = "Multi-Page Layout"
|
||||
|
||||
[home.parseDocument]
|
||||
desc = "Layout-aware parsing to structured JSON or Markdown, with optional OCR"
|
||||
tags = "parse,layout,structure,blocks,markdown,JSON,docling,OCR,scan,document understanding,convert"
|
||||
title = "Parse Document"
|
||||
|
||||
[home.pdfCommentAgent]
|
||||
desc = "Ask AI to annotate a PDF with sticky-note comments based on your prompt"
|
||||
tags = "AI,agent,comment,annotate,sticky note,review,feedback,notes"
|
||||
@@ -4805,6 +4870,11 @@ desc = "Adds signature to PDF by drawing, text or image"
|
||||
tags = "signature,autograph,e-sign,electronic signature,digital signature,sign document,approval,signoff,authorize,endorse,ink signature,handwriting"
|
||||
title = "Sign"
|
||||
|
||||
[home.smartSplit]
|
||||
desc = "Split a PDF into sub-documents using a natural-language boundary rule"
|
||||
tags = "split,smart,boundaries,separate,invoices,batches,content split,divide,rules,auto split"
|
||||
title = "Smart Split"
|
||||
|
||||
[home.split]
|
||||
desc = "Split PDFs into multiple documents"
|
||||
tags = "divide,separate,break,split,extract pages,separate pages,divide document,break apart,separate files,unbind,split by page,divide by chapter"
|
||||
@@ -5563,6 +5633,33 @@ title = "Page Ranges"
|
||||
bullet1 = "<strong>all</strong> → selects all pages"
|
||||
title = "Special Keywords"
|
||||
|
||||
[parseDocument]
|
||||
intro = "Reads the document's layout - headings, paragraphs, tables - and turns it into clean structured JSON or Markdown you can feed to other systems."
|
||||
submit = "Parse document"
|
||||
|
||||
[parseDocument.error]
|
||||
failed = "Failed to parse document"
|
||||
|
||||
[parseDocument.mode]
|
||||
advanced = "Advanced"
|
||||
auto = "Auto"
|
||||
basic = "Basic"
|
||||
label = "Mode"
|
||||
|
||||
[parseDocument.outputFormat]
|
||||
json = "JSON"
|
||||
label = "Output format"
|
||||
markdown = "Markdown"
|
||||
|
||||
[parseDocument.results]
|
||||
title = "Parsed output"
|
||||
|
||||
[parseDocument.settings]
|
||||
title = "Parse settings"
|
||||
|
||||
[parseDocument.withOcr]
|
||||
label = "Apply OCR to scanned pages (recommended)"
|
||||
|
||||
[payg.activity]
|
||||
docs = "docs"
|
||||
empty = "No billable activity yet this period."
|
||||
@@ -10367,6 +10464,26 @@ medium = "Medium"
|
||||
small = "Small"
|
||||
x-large = "X-Large"
|
||||
|
||||
[smartSplit]
|
||||
intro = "Describe where sub-documents start in plain language and AI reads the content to find those boundaries - no page numbers needed."
|
||||
submit = "Split document"
|
||||
|
||||
[smartSplit.error]
|
||||
failed = "Failed to split document"
|
||||
|
||||
[smartSplit.maxParts]
|
||||
label = "Maximum parts"
|
||||
|
||||
[smartSplit.results]
|
||||
title = "Split documents"
|
||||
|
||||
[smartSplit.rule]
|
||||
label = "Split rule"
|
||||
placeholder = "e.g. Start a new document at every invoice header"
|
||||
|
||||
[smartSplit.settings]
|
||||
title = "Split settings"
|
||||
|
||||
[split]
|
||||
resultsTitle = "Split Results"
|
||||
selectMethod = "Select a split method"
|
||||
|
||||
@@ -0,0 +1,165 @@
|
||||
import { test, expect } from "@app/tests/helpers/stub-test-base";
|
||||
import type { Page, Route } from "@playwright/test";
|
||||
import path from "node:path";
|
||||
|
||||
/** DocParse walkthrough: the five workbench tools.
|
||||
* Dumps PNGs to screenshots/docparse; light + dark per view, RTL spot checks. */
|
||||
|
||||
const SCREENSHOTS_DIR = path.resolve(process.cwd(), "screenshots", "docparse");
|
||||
|
||||
function shotPath(name: string): string {
|
||||
return path.join(SCREENSHOTS_DIR, `${name}.png`);
|
||||
}
|
||||
|
||||
async function settle(page: Page, ms = 400): Promise<void> {
|
||||
await page.waitForTimeout(ms);
|
||||
}
|
||||
|
||||
async function stubApis(page: Page): Promise<void> {
|
||||
// Narrow fallbacks only: a blanket /api/v1/** would out-rank the stub
|
||||
// fixture's own /auth/me route (last-registered wins) and break the session.
|
||||
await page.route("**/api/v1/policies/**", (route: Route) =>
|
||||
route.fulfill({ json: [] }),
|
||||
);
|
||||
await page.route("**/api/v1/proprietary/ui-data/**", (route: Route) =>
|
||||
route.fulfill({ json: [] }),
|
||||
);
|
||||
// enableLogin true: the portal only builds a session when login mode is on
|
||||
// (matches live behavior; with login off the portal shows its login screen).
|
||||
const configPayload = {
|
||||
appVersion: "test",
|
||||
enableLogin: true,
|
||||
isAdmin: true,
|
||||
languages: ["en-US"],
|
||||
defaultLocale: "en-US",
|
||||
aiEngineEnabled: true,
|
||||
docparseEnabled: true,
|
||||
docparseAdvanced: true,
|
||||
storageEnabled: false,
|
||||
premiumEnabled: true,
|
||||
runningProOrHigher: true,
|
||||
};
|
||||
await page.route("**/api/v1/config/app-config", (route: Route) =>
|
||||
route.fulfill({ json: configPayload }),
|
||||
);
|
||||
// The auth layer decides login mode from public-config; keep it in sync.
|
||||
await page.route("**/api/v1/config/public-config", (route: Route) =>
|
||||
route.fulfill({
|
||||
json: { enableLogin: true, languages: ["en-US"], defaultLocale: "en-US" },
|
||||
}),
|
||||
);
|
||||
await page.route(
|
||||
"**/api/v1/config/endpoints-availability**",
|
||||
(route: Route) => route.fulfill({ json: {} }),
|
||||
);
|
||||
await page.route("**/api/v1/config/endpoint-enabled**", (route: Route) =>
|
||||
route.fulfill({ json: { enabled: true } }),
|
||||
);
|
||||
// DocparseToolIntro probes live capabilities for its tier badges.
|
||||
await page.route("**/api/v1/docparse/capabilities", (route: Route) =>
|
||||
route.fulfill({
|
||||
json: {
|
||||
enabled: true,
|
||||
mode: "auto",
|
||||
advancedInstalled: true,
|
||||
engineReachable: true,
|
||||
doclingVersion: "2.116.0",
|
||||
},
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
async function enableDarkMode(page: Page): Promise<void> {
|
||||
await page.addInitScript(() => {
|
||||
localStorage.setItem("mantine-color-scheme", "dark");
|
||||
localStorage.setItem("mantine-color-scheme-value", "dark");
|
||||
});
|
||||
await page.emulateMedia({ colorScheme: "dark" });
|
||||
}
|
||||
|
||||
async function enableRtl(page: Page): Promise<void> {
|
||||
await page.addInitScript(() => {
|
||||
localStorage.setItem("i18nextLng", "ar-AR");
|
||||
localStorage.setItem("stirling-language", "ar-AR");
|
||||
localStorage.setItem("stirling-language-source", "user");
|
||||
const applyDir = () => {
|
||||
document.documentElement.setAttribute("dir", "rtl");
|
||||
document.documentElement.setAttribute("lang", "ar-AR");
|
||||
};
|
||||
if (document.documentElement) applyDir();
|
||||
else document.addEventListener("DOMContentLoaded", applyDir);
|
||||
});
|
||||
}
|
||||
|
||||
const TOOLS = [
|
||||
{ id: "parseDocument", url: "/parse-document", waitText: /Parse/i },
|
||||
{ id: "extractFields", url: "/extract-fields", waitText: /Extract/i },
|
||||
{ id: "smartSplit", url: "/smart-split", waitText: /Split/i },
|
||||
{ id: "chunkDocument", url: "/chunk-document", waitText: /Chunk/i },
|
||||
{ id: "fillTemplate", url: "/fill-template", waitText: /Template|Fill/i },
|
||||
];
|
||||
|
||||
async function openTool(page: Page, url: string): Promise<void> {
|
||||
await page.goto(url, { waitUntil: "domcontentloaded" });
|
||||
await expect(page.locator("body").first()).not.toBeEmpty();
|
||||
// The tool panel is the left rail; give lazy chunks a moment.
|
||||
await settle(page, 900);
|
||||
}
|
||||
|
||||
test.describe("DocParse walkthrough", () => {
|
||||
test.use({
|
||||
autoGoto: false,
|
||||
viewport: { width: 1600, height: 900 },
|
||||
seedJwt: true,
|
||||
});
|
||||
|
||||
// ─── Editor tools, light ──────────────────────────────────────────────────
|
||||
for (const [i, tool] of TOOLS.entries()) {
|
||||
test(`t${i}_${tool.id}_light`, async ({ page }) => {
|
||||
await stubApis(page);
|
||||
await openTool(page, tool.url);
|
||||
await page.screenshot({ path: shotPath(`0${i + 1}_${tool.id}_light`) });
|
||||
});
|
||||
|
||||
test(`t${i}_${tool.id}_dark`, async ({ page }) => {
|
||||
await enableDarkMode(page);
|
||||
await stubApis(page);
|
||||
await openTool(page, tool.url);
|
||||
await page.screenshot({ path: shotPath(`0${i + 1}_${tool.id}_dark`) });
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Extract Fields with builder rows filled ─────────────────────────────
|
||||
for (const theme of ["light", "dark"] as const) {
|
||||
test(`extract_fields_populated_${theme}`, async ({ page }) => {
|
||||
if (theme === "dark") await enableDarkMode(page);
|
||||
await stubApis(page);
|
||||
await openTool(page, "/extract-fields");
|
||||
// Fill the first schema-builder row when present; tolerate layout drift.
|
||||
const nameInput = page.getByPlaceholder(/name/i).first();
|
||||
if (await nameInput.isVisible().catch(() => false)) {
|
||||
await nameInput.fill("invoice_number");
|
||||
const addButton = page.getByRole("button", { name: /add/i }).first();
|
||||
if (await addButton.isVisible().catch(() => false)) {
|
||||
await addButton.click();
|
||||
const second = page.getByPlaceholder(/name/i).nth(1);
|
||||
if (await second.isVisible().catch(() => false)) {
|
||||
await second.fill("total_due");
|
||||
}
|
||||
}
|
||||
}
|
||||
await settle(page);
|
||||
await page.screenshot({
|
||||
path: shotPath(`06_extract_fields_populated_${theme}`),
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
// ─── RTL spot checks ──────────────────────────────────────────────────────
|
||||
test("rtl_extract_fields", async ({ page }) => {
|
||||
await enableRtl(page);
|
||||
await stubApis(page);
|
||||
await openTool(page, "/extract-fields");
|
||||
await page.screenshot({ path: shotPath("11_extract_fields_rtl") });
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,91 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { Anchor, List, NumberInput, Select, Stack, Text } from "@mantine/core";
|
||||
import type { ToolAutomationSettingsProps } from "@app/hooks/tools/shared/toolOperationTypes";
|
||||
import type { ChunkDocumentParameters } from "@app/hooks/tools/chunkDocument/useChunkDocumentParameters";
|
||||
import type { DocparseMode } from "@app/hooks/tools/parseDocument/useParseDocumentParameters";
|
||||
import DocparseToolIntro from "@app/components/tools/docparse/DocparseToolIntro";
|
||||
|
||||
const ChunkDocumentSettings = ({
|
||||
parameters,
|
||||
onParameterChange,
|
||||
disabled,
|
||||
}: ToolAutomationSettingsProps<ChunkDocumentParameters>) => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
return (
|
||||
<Stack gap="sm">
|
||||
<DocparseToolIntro
|
||||
description={t(
|
||||
"chunkDocument.intro",
|
||||
"Turns a document into retrieval-ready chunks in three layers, so answers cite the right section instead of a random page.",
|
||||
)}
|
||||
aiBadge="layout"
|
||||
/>
|
||||
<List type="ordered" size="sm" spacing={4}>
|
||||
<List.Item>
|
||||
{t(
|
||||
"chunkDocument.layers.parse",
|
||||
"Layout-aware parse: headings, paragraphs, and tables are recognized as structure",
|
||||
)}
|
||||
</List.Item>
|
||||
<List.Item>
|
||||
{t(
|
||||
"chunkDocument.layers.chunk",
|
||||
"Structure-aware chunks: each carries its heading breadcrumb and page range",
|
||||
)}
|
||||
</List.Item>
|
||||
<List.Item>
|
||||
{t(
|
||||
"chunkDocument.layers.embed",
|
||||
"Ready to embed: exported as JSONL for any vector store",
|
||||
)}
|
||||
</List.Item>
|
||||
</List>
|
||||
<NumberInput
|
||||
label={t("chunkDocument.chunkSize.label", "Chunk size (characters)")}
|
||||
value={parameters.chunkSize}
|
||||
onChange={(value) =>
|
||||
onParameterChange("chunkSize", typeof value === "number" ? value : 0)
|
||||
}
|
||||
min={1}
|
||||
disabled={disabled}
|
||||
/>
|
||||
<NumberInput
|
||||
label={t("chunkDocument.overlap.label", "Overlap (characters)")}
|
||||
value={parameters.overlap}
|
||||
onChange={(value) =>
|
||||
onParameterChange("overlap", typeof value === "number" ? value : 0)
|
||||
}
|
||||
min={0}
|
||||
disabled={disabled}
|
||||
/>
|
||||
<Select
|
||||
label={t("chunkDocument.mode.label", "Mode")}
|
||||
value={parameters.mode}
|
||||
onChange={(value) =>
|
||||
onParameterChange("mode", (value ?? "auto") as DocparseMode)
|
||||
}
|
||||
data={[
|
||||
{ value: "auto", label: t("chunkDocument.mode.auto", "Auto") },
|
||||
{ value: "basic", label: t("chunkDocument.mode.basic", "Basic") },
|
||||
{
|
||||
value: "advanced",
|
||||
label: t("chunkDocument.mode.advanced", "Advanced"),
|
||||
},
|
||||
]}
|
||||
disabled={disabled}
|
||||
/>
|
||||
<Text size="xs" c="dimmed">
|
||||
{t(
|
||||
"chunkDocument.processorCallout",
|
||||
"To index automatically, add the 'Index into knowledge base' step to an ingestion policy in the",
|
||||
)}{" "}
|
||||
<Anchor href="/processor/policies" size="xs">
|
||||
{t("chunkDocument.processorLink", "Processor")}
|
||||
</Anchor>
|
||||
</Text>
|
||||
</Stack>
|
||||
);
|
||||
};
|
||||
|
||||
export default ChunkDocumentSettings;
|
||||
@@ -0,0 +1,54 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { Stack, Text, Textarea } from "@mantine/core";
|
||||
import type { ToolAutomationSettingsProps } from "@app/hooks/tools/shared/toolOperationTypes";
|
||||
import {
|
||||
isJsonObjectString,
|
||||
type FillTemplateParameters,
|
||||
} from "@app/hooks/tools/fillTemplate/useFillTemplateParameters";
|
||||
import DocparseToolIntro from "@app/components/tools/docparse/DocparseToolIntro";
|
||||
|
||||
const FillTemplateSettings = ({
|
||||
parameters,
|
||||
onParameterChange,
|
||||
disabled,
|
||||
}: ToolAutomationSettingsProps<FillTemplateParameters>) => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
const dataJson = parameters.dataJson;
|
||||
const jsonError =
|
||||
dataJson.trim().length > 0 && !isJsonObjectString(dataJson)
|
||||
? t("fillTemplate.data.invalid", "Enter a valid JSON object")
|
||||
: null;
|
||||
|
||||
return (
|
||||
<Stack gap="sm">
|
||||
<DocparseToolIntro
|
||||
description={t(
|
||||
"fillTemplate.intro",
|
||||
"Replaces the placeholders in a Word (.docx) template with your JSON data and returns the filled document - deterministic, no AI involved.",
|
||||
)}
|
||||
showFallbackNote={false}
|
||||
/>
|
||||
<Text size="sm" c="dimmed">
|
||||
{t(
|
||||
"fillTemplate.hint",
|
||||
"The input file must be a .docx template (not a PDF). Each placeholder in the template is replaced with the matching JSON value.",
|
||||
)}
|
||||
</Text>
|
||||
<Textarea
|
||||
label={t("fillTemplate.data.label", "Data (JSON)")}
|
||||
placeholder='{"customer": "ACME Corp", "total": "128.00"}'
|
||||
value={dataJson}
|
||||
onChange={(event) =>
|
||||
onParameterChange("dataJson", event.currentTarget.value)
|
||||
}
|
||||
minRows={5}
|
||||
autosize
|
||||
error={jsonError}
|
||||
disabled={disabled}
|
||||
/>
|
||||
</Stack>
|
||||
);
|
||||
};
|
||||
|
||||
export default FillTemplateSettings;
|
||||
@@ -0,0 +1,78 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { Checkbox, Select, Stack } from "@mantine/core";
|
||||
import type { ToolAutomationSettingsProps } from "@app/hooks/tools/shared/toolOperationTypes";
|
||||
import type {
|
||||
DocparseMode,
|
||||
ParseDocumentParameters,
|
||||
} from "@app/hooks/tools/parseDocument/useParseDocumentParameters";
|
||||
import DocparseToolIntro from "@app/components/tools/docparse/DocparseToolIntro";
|
||||
|
||||
const ParseDocumentSettings = ({
|
||||
parameters,
|
||||
onParameterChange,
|
||||
disabled,
|
||||
}: ToolAutomationSettingsProps<ParseDocumentParameters>) => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
return (
|
||||
<Stack gap="sm">
|
||||
<DocparseToolIntro
|
||||
description={t(
|
||||
"parseDocument.intro",
|
||||
"Reads the document's layout - headings, paragraphs, tables - and turns it into clean structured JSON or Markdown you can feed to other systems.",
|
||||
)}
|
||||
aiBadge="layout"
|
||||
/>
|
||||
<Select
|
||||
label={t("parseDocument.mode.label", "Mode")}
|
||||
value={parameters.mode}
|
||||
onChange={(value) =>
|
||||
onParameterChange("mode", (value ?? "auto") as DocparseMode)
|
||||
}
|
||||
data={[
|
||||
{ value: "auto", label: t("parseDocument.mode.auto", "Auto") },
|
||||
{ value: "basic", label: t("parseDocument.mode.basic", "Basic") },
|
||||
{
|
||||
value: "advanced",
|
||||
label: t("parseDocument.mode.advanced", "Advanced"),
|
||||
},
|
||||
]}
|
||||
disabled={disabled}
|
||||
/>
|
||||
<Select
|
||||
label={t("parseDocument.outputFormat.label", "Output format")}
|
||||
value={parameters.outputFormat}
|
||||
onChange={(value) =>
|
||||
onParameterChange(
|
||||
"outputFormat",
|
||||
(value ?? "json") as ParseDocumentParameters["outputFormat"],
|
||||
)
|
||||
}
|
||||
data={[
|
||||
{
|
||||
value: "json",
|
||||
label: t("parseDocument.outputFormat.json", "JSON"),
|
||||
},
|
||||
{
|
||||
value: "markdown",
|
||||
label: t("parseDocument.outputFormat.markdown", "Markdown"),
|
||||
},
|
||||
]}
|
||||
disabled={disabled}
|
||||
/>
|
||||
<Checkbox
|
||||
label={t(
|
||||
"parseDocument.withOcr.label",
|
||||
"Apply OCR to scanned pages (recommended)",
|
||||
)}
|
||||
checked={parameters.withOcr}
|
||||
onChange={(event) =>
|
||||
onParameterChange("withOcr", event.currentTarget.checked)
|
||||
}
|
||||
disabled={disabled}
|
||||
/>
|
||||
</Stack>
|
||||
);
|
||||
};
|
||||
|
||||
export default ParseDocumentSettings;
|
||||
@@ -0,0 +1,51 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { NumberInput, Stack, Textarea } from "@mantine/core";
|
||||
import type { ToolAutomationSettingsProps } from "@app/hooks/tools/shared/toolOperationTypes";
|
||||
import type { SmartSplitParameters } from "@app/hooks/tools/smartSplit/useSmartSplitParameters";
|
||||
import DocparseToolIntro from "@app/components/tools/docparse/DocparseToolIntro";
|
||||
|
||||
const SmartSplitSettings = ({
|
||||
parameters,
|
||||
onParameterChange,
|
||||
disabled,
|
||||
}: ToolAutomationSettingsProps<SmartSplitParameters>) => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
return (
|
||||
<Stack gap="sm">
|
||||
<DocparseToolIntro
|
||||
description={t(
|
||||
"smartSplit.intro",
|
||||
"Describe where sub-documents start in plain language and AI reads the content to find those boundaries - no page numbers needed.",
|
||||
)}
|
||||
aiBadge="llm"
|
||||
/>
|
||||
<Textarea
|
||||
label={t("smartSplit.rule.label", "Split rule")}
|
||||
placeholder={t(
|
||||
"smartSplit.rule.placeholder",
|
||||
"e.g. Start a new document at every invoice header",
|
||||
)}
|
||||
value={parameters.rule}
|
||||
onChange={(event) =>
|
||||
onParameterChange("rule", event.currentTarget.value)
|
||||
}
|
||||
minRows={3}
|
||||
autosize
|
||||
disabled={disabled}
|
||||
/>
|
||||
<NumberInput
|
||||
label={t("smartSplit.maxParts.label", "Maximum parts")}
|
||||
value={parameters.maxParts}
|
||||
onChange={(value) =>
|
||||
onParameterChange("maxParts", typeof value === "number" ? value : 1)
|
||||
}
|
||||
min={1}
|
||||
max={100}
|
||||
disabled={disabled}
|
||||
/>
|
||||
</Stack>
|
||||
);
|
||||
};
|
||||
|
||||
export default SmartSplitSettings;
|
||||
@@ -9,8 +9,16 @@ import {
|
||||
} from "@app/data/toolsTaxonomy";
|
||||
import { asRegistryConfig } from "@app/hooks/tools/shared/toolOperationTypes";
|
||||
import { useDocparseEnabled } from "@app/hooks/useDocparseEnabled";
|
||||
import { parseDocumentOperationConfig } from "@app/hooks/tools/parseDocument/parseDocumentOperationConfig";
|
||||
import { extractFieldsOperationConfig } from "@app/hooks/tools/extractFields/extractFieldsOperationConfig";
|
||||
import { smartSplitOperationConfig } from "@app/hooks/tools/smartSplit/smartSplitOperationConfig";
|
||||
import { chunkDocumentOperationConfig } from "@app/hooks/tools/chunkDocument/chunkDocumentOperationConfig";
|
||||
import { fillTemplateOperationConfig } from "@app/hooks/tools/fillTemplate/fillTemplateOperationConfig";
|
||||
import ParseDocument from "@app/tools/ParseDocument";
|
||||
import ExtractFields from "@app/tools/ExtractFields";
|
||||
import SmartSplit from "@app/tools/SmartSplit";
|
||||
import ChunkDocument from "@app/tools/ChunkDocument";
|
||||
import FillTemplate from "@app/tools/FillTemplate";
|
||||
import { getSynonyms } from "@app/utils/toolSynonyms";
|
||||
|
||||
const toolIcon = (icon: string) => (
|
||||
@@ -29,6 +37,23 @@ export function useProprietaryToolRegistry(): ProprietaryToolRegistry {
|
||||
return useMemo(() => {
|
||||
if (!docparseEnabled) return {} as ProprietaryToolRegistry;
|
||||
return {
|
||||
parseDocument: {
|
||||
icon: toolIcon("quick-reference-all-outline-rounded"),
|
||||
name: t("home.parseDocument.title", "Parse Document"),
|
||||
component: ParseDocument,
|
||||
description: t(
|
||||
"home.parseDocument.desc",
|
||||
"Layout-aware parsing to structured JSON or Markdown, with optional OCR",
|
||||
),
|
||||
categoryId: ToolCategoryId.STANDARD_TOOLS,
|
||||
subcategoryId: SubcategoryId.DOCUMENT_INTELLIGENCE,
|
||||
maxFiles: 1,
|
||||
endpoints: ["parse-document"],
|
||||
operationConfig: asRegistryConfig(parseDocumentOperationConfig),
|
||||
automationSettings: null,
|
||||
synonyms: getSynonyms(t, "parseDocument"),
|
||||
versionStatus: "beta",
|
||||
},
|
||||
extractFields: {
|
||||
icon: toolIcon("fact-check-outline-rounded"),
|
||||
name: t("home.extractFields.title", "Extract Fields"),
|
||||
@@ -46,6 +71,58 @@ export function useProprietaryToolRegistry(): ProprietaryToolRegistry {
|
||||
synonyms: getSynonyms(t, "extractFields"),
|
||||
versionStatus: "beta",
|
||||
},
|
||||
smartSplit: {
|
||||
icon: toolIcon("content-cut-rounded"),
|
||||
name: t("home.smartSplit.title", "Smart Split"),
|
||||
component: SmartSplit,
|
||||
description: t(
|
||||
"home.smartSplit.desc",
|
||||
"Split a PDF into sub-documents using a natural-language boundary rule",
|
||||
),
|
||||
categoryId: ToolCategoryId.STANDARD_TOOLS,
|
||||
subcategoryId: SubcategoryId.DOCUMENT_INTELLIGENCE,
|
||||
maxFiles: 1,
|
||||
endpoints: ["smart-split"],
|
||||
operationConfig: asRegistryConfig(smartSplitOperationConfig),
|
||||
automationSettings: null,
|
||||
synonyms: getSynonyms(t, "smartSplit"),
|
||||
versionStatus: "beta",
|
||||
},
|
||||
chunkDocument: {
|
||||
icon: toolIcon("layers-outline-rounded"),
|
||||
name: t("home.chunkDocument.title", "Prepare for RAG"),
|
||||
component: ChunkDocument,
|
||||
description: t(
|
||||
"home.chunkDocument.desc",
|
||||
"Layout-aware parse to structure-aware chunks with heading breadcrumbs and page ranges, ready to embed",
|
||||
),
|
||||
categoryId: ToolCategoryId.STANDARD_TOOLS,
|
||||
subcategoryId: SubcategoryId.DOCUMENT_INTELLIGENCE,
|
||||
maxFiles: 1,
|
||||
endpoints: ["chunk-document"],
|
||||
operationConfig: asRegistryConfig(chunkDocumentOperationConfig),
|
||||
automationSettings: null,
|
||||
synonyms: getSynonyms(t, "chunkDocument"),
|
||||
versionStatus: "beta",
|
||||
},
|
||||
fillTemplate: {
|
||||
icon: toolIcon("assignment-outline-rounded"),
|
||||
name: t("home.fillTemplate.title", "Fill Template"),
|
||||
component: FillTemplate,
|
||||
description: t(
|
||||
"home.fillTemplate.desc",
|
||||
"Fill a DOCX template's placeholders from JSON data",
|
||||
),
|
||||
categoryId: ToolCategoryId.STANDARD_TOOLS,
|
||||
subcategoryId: SubcategoryId.DOCUMENT_INTELLIGENCE,
|
||||
maxFiles: 1,
|
||||
supportedFormats: ["docx"],
|
||||
endpoints: ["fill-template"],
|
||||
operationConfig: asRegistryConfig(fillTemplateOperationConfig),
|
||||
automationSettings: null,
|
||||
synonyms: getSynonyms(t, "fillTemplate"),
|
||||
versionStatus: "beta",
|
||||
},
|
||||
} as ProprietaryToolRegistry;
|
||||
}, [docparseEnabled, t]);
|
||||
}, [t, docparseEnabled]);
|
||||
}
|
||||
|
||||
+42
@@ -0,0 +1,42 @@
|
||||
import { describe, expect, test } from "vitest";
|
||||
import {
|
||||
buildChunkDocumentFormData,
|
||||
chunksFromResponse,
|
||||
chunksToJsonl,
|
||||
} from "@app/hooks/tools/chunkDocument/chunkDocumentOperationConfig";
|
||||
import { defaultParameters } from "@app/hooks/tools/chunkDocument/useChunkDocumentParameters";
|
||||
|
||||
describe("chunkDocument operation helpers", () => {
|
||||
test("accepts both bare-array and wrapped chunk responses", () => {
|
||||
const chunks = [{ text: "a" }, { text: "b" }];
|
||||
expect(chunksFromResponse(chunks)).toEqual(chunks);
|
||||
expect(chunksFromResponse({ chunks })).toEqual(chunks);
|
||||
expect(chunksFromResponse({ nope: true })).toEqual([]);
|
||||
expect(chunksFromResponse(null)).toEqual([]);
|
||||
});
|
||||
|
||||
test("emits one JSON document per JSONL line", () => {
|
||||
const jsonl = chunksToJsonl([
|
||||
{ text: "a", page: 1 },
|
||||
{ text: "b", page: 2 },
|
||||
]);
|
||||
const lines = jsonl.split("\n");
|
||||
expect(lines).toHaveLength(2);
|
||||
expect(lines.map((line) => JSON.parse(line))).toEqual([
|
||||
{ text: "a", page: 1 },
|
||||
{ text: "b", page: 2 },
|
||||
]);
|
||||
});
|
||||
|
||||
test("builds the multipart form the backend contract expects", () => {
|
||||
const file = new File(["pdf"], "doc.pdf", { type: "application/pdf" });
|
||||
const form = buildChunkDocumentFormData(
|
||||
{ ...defaultParameters, chunkSize: 800, overlap: 50, mode: "advanced" },
|
||||
file,
|
||||
);
|
||||
expect(form.get("fileInput")).toBe(file);
|
||||
expect(form.get("chunkSize")).toBe("800");
|
||||
expect(form.get("overlap")).toBe("50");
|
||||
expect(form.get("mode")).toBe("advanced");
|
||||
});
|
||||
});
|
||||
+64
@@ -0,0 +1,64 @@
|
||||
import apiClient from "@app/services/apiClient";
|
||||
import {
|
||||
defineCustomTool,
|
||||
CustomProcessorResult,
|
||||
} from "@app/hooks/tools/shared/toolOperationTypes";
|
||||
import { deriveName } from "@app/hooks/tools/shared/docparseFilenames";
|
||||
import {
|
||||
ChunkDocumentParameters,
|
||||
defaultParameters,
|
||||
} from "@app/hooks/tools/chunkDocument/useChunkDocumentParameters";
|
||||
|
||||
export const CHUNK_DOCUMENT_ENDPOINT = "/api/v1/docparse/chunk-document";
|
||||
|
||||
export const buildChunkDocumentFormData = (
|
||||
parameters: ChunkDocumentParameters,
|
||||
file: File,
|
||||
): FormData => {
|
||||
const formData = new FormData();
|
||||
formData.append("fileInput", file);
|
||||
formData.append("chunkSize", String(parameters.chunkSize));
|
||||
formData.append("overlap", String(parameters.overlap));
|
||||
formData.append("mode", parameters.mode);
|
||||
return formData;
|
||||
};
|
||||
|
||||
/** The chunks array, whether the backend returns it bare or wrapped. */
|
||||
export function chunksFromResponse(data: unknown): unknown[] {
|
||||
if (Array.isArray(data)) return data;
|
||||
const wrapped = (data as { chunks?: unknown[] } | null)?.chunks;
|
||||
return Array.isArray(wrapped) ? wrapped : [];
|
||||
}
|
||||
|
||||
/** JSON chunks -> one JSONL line per chunk, the standard RAG-ingest shape. */
|
||||
export function chunksToJsonl(chunks: unknown[]): string {
|
||||
return chunks.map((chunk) => JSON.stringify(chunk)).join("\n");
|
||||
}
|
||||
|
||||
const processChunkDocument = async (
|
||||
parameters: ChunkDocumentParameters,
|
||||
files: File[],
|
||||
): Promise<CustomProcessorResult> => {
|
||||
if (files.length === 0) return { files: [] };
|
||||
|
||||
const [inputFile] = files;
|
||||
const response = await apiClient.post<unknown>(
|
||||
CHUNK_DOCUMENT_ENDPOINT,
|
||||
buildChunkDocumentFormData(parameters, inputFile),
|
||||
);
|
||||
|
||||
const resultFile = new File(
|
||||
[chunksToJsonl(chunksFromResponse(response.data))],
|
||||
deriveName(inputFile.name, ".chunks.jsonl"),
|
||||
{ type: "application/x-ndjson" },
|
||||
);
|
||||
return { files: [resultFile] };
|
||||
};
|
||||
|
||||
export const chunkDocumentOperationConfig =
|
||||
defineCustomTool<ChunkDocumentParameters>({
|
||||
operationType: "chunkDocument",
|
||||
endpoint: CHUNK_DOCUMENT_ENDPOINT,
|
||||
customProcessor: processChunkDocument,
|
||||
defaultParameters,
|
||||
});
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { useToolOperation } from "@app/hooks/tools/shared/useToolOperation";
|
||||
import { createStandardErrorHandler } from "@app/utils/toolErrorHandler";
|
||||
import { chunkDocumentOperationConfig } from "@app/hooks/tools/chunkDocument/chunkDocumentOperationConfig";
|
||||
|
||||
export const useChunkDocumentOperation = () => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
return useToolOperation({
|
||||
...chunkDocumentOperationConfig,
|
||||
getErrorMessage: createStandardErrorHandler(
|
||||
t("chunkDocument.error.failed", "Failed to chunk document"),
|
||||
),
|
||||
});
|
||||
};
|
||||
+34
@@ -0,0 +1,34 @@
|
||||
import { BaseParameters } from "@app/types/parameters";
|
||||
import {
|
||||
useBaseParameters,
|
||||
BaseParametersHook,
|
||||
} from "@app/hooks/tools/shared/useBaseParameters";
|
||||
import type { DocparseMode } from "@app/hooks/tools/parseDocument/useParseDocumentParameters";
|
||||
|
||||
export interface ChunkDocumentParameters extends BaseParameters {
|
||||
/** Target chunk size in characters. */
|
||||
chunkSize: number;
|
||||
/** Characters of overlap carried between neighbouring chunks. */
|
||||
overlap: number;
|
||||
mode: DocparseMode;
|
||||
}
|
||||
|
||||
export const defaultParameters: ChunkDocumentParameters = {
|
||||
chunkSize: 1000,
|
||||
overlap: 100,
|
||||
mode: "auto",
|
||||
};
|
||||
|
||||
export type ChunkDocumentParametersHook =
|
||||
BaseParametersHook<ChunkDocumentParameters>;
|
||||
|
||||
export const useChunkDocumentParameters = (): ChunkDocumentParametersHook => {
|
||||
return useBaseParameters({
|
||||
defaultParameters,
|
||||
endpointName: "chunk-document",
|
||||
validateFn: (params) =>
|
||||
params.chunkSize > 0 &&
|
||||
params.overlap >= 0 &&
|
||||
params.overlap < params.chunkSize,
|
||||
});
|
||||
};
|
||||
+55
@@ -0,0 +1,55 @@
|
||||
import apiClient from "@app/services/apiClient";
|
||||
import {
|
||||
defineCustomTool,
|
||||
CustomProcessorResult,
|
||||
} from "@app/hooks/tools/shared/toolOperationTypes";
|
||||
import { deriveName } from "@app/hooks/tools/shared/docparseFilenames";
|
||||
import {
|
||||
FillTemplateParameters,
|
||||
defaultParameters,
|
||||
} from "@app/hooks/tools/fillTemplate/useFillTemplateParameters";
|
||||
|
||||
export const FILL_TEMPLATE_ENDPOINT = "/api/v1/docparse/fill-template";
|
||||
|
||||
const DOCX_TYPE =
|
||||
"application/vnd.openxmlformats-officedocument.wordprocessingml.document";
|
||||
|
||||
export const buildFillTemplateFormData = (
|
||||
parameters: FillTemplateParameters,
|
||||
file: File,
|
||||
): FormData => {
|
||||
const formData = new FormData();
|
||||
formData.append("templateFile", file);
|
||||
formData.append("data", parameters.dataJson.trim());
|
||||
return formData;
|
||||
};
|
||||
|
||||
/** POST the DOCX template + data; the filled DOCX comes straight back. */
|
||||
const processFillTemplate = async (
|
||||
parameters: FillTemplateParameters,
|
||||
files: File[],
|
||||
): Promise<CustomProcessorResult> => {
|
||||
if (files.length === 0) return { files: [] };
|
||||
|
||||
const [templateFile] = files;
|
||||
const response = await apiClient.post<Blob>(
|
||||
FILL_TEMPLATE_ENDPOINT,
|
||||
buildFillTemplateFormData(parameters, templateFile),
|
||||
{ responseType: "blob" },
|
||||
);
|
||||
|
||||
const resultFile = new File(
|
||||
[response.data],
|
||||
deriveName(templateFile.name, "-filled.docx"),
|
||||
{ type: DOCX_TYPE },
|
||||
);
|
||||
return { files: [resultFile] };
|
||||
};
|
||||
|
||||
export const fillTemplateOperationConfig =
|
||||
defineCustomTool<FillTemplateParameters>({
|
||||
operationType: "fillTemplate",
|
||||
endpoint: FILL_TEMPLATE_ENDPOINT,
|
||||
customProcessor: processFillTemplate,
|
||||
defaultParameters,
|
||||
});
|
||||
@@ -0,0 +1,15 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { useToolOperation } from "@app/hooks/tools/shared/useToolOperation";
|
||||
import { createStandardErrorHandler } from "@app/utils/toolErrorHandler";
|
||||
import { fillTemplateOperationConfig } from "@app/hooks/tools/fillTemplate/fillTemplateOperationConfig";
|
||||
|
||||
export const useFillTemplateOperation = () => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
return useToolOperation({
|
||||
...fillTemplateOperationConfig,
|
||||
getErrorMessage: createStandardErrorHandler(
|
||||
t("fillTemplate.error.failed", "Failed to fill template"),
|
||||
),
|
||||
});
|
||||
};
|
||||
@@ -0,0 +1,37 @@
|
||||
import { BaseParameters } from "@app/types/parameters";
|
||||
import {
|
||||
useBaseParameters,
|
||||
BaseParametersHook,
|
||||
} from "@app/hooks/tools/shared/useBaseParameters";
|
||||
|
||||
export interface FillTemplateParameters extends BaseParameters {
|
||||
/** JSON object whose keys fill the template's placeholders. */
|
||||
dataJson: string;
|
||||
}
|
||||
|
||||
export const defaultParameters: FillTemplateParameters = {
|
||||
dataJson: "",
|
||||
};
|
||||
|
||||
/** True when the text parses to a plain JSON object. */
|
||||
export function isJsonObjectString(text: string): boolean {
|
||||
try {
|
||||
const parsed = JSON.parse(text);
|
||||
return (
|
||||
typeof parsed === "object" && parsed !== null && !Array.isArray(parsed)
|
||||
);
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
export type FillTemplateParametersHook =
|
||||
BaseParametersHook<FillTemplateParameters>;
|
||||
|
||||
export const useFillTemplateParameters = (): FillTemplateParametersHook => {
|
||||
return useBaseParameters({
|
||||
defaultParameters,
|
||||
endpointName: "fill-template",
|
||||
validateFn: (params) => isJsonObjectString(params.dataJson),
|
||||
});
|
||||
};
|
||||
+56
@@ -0,0 +1,56 @@
|
||||
import apiClient from "@app/services/apiClient";
|
||||
import {
|
||||
defineCustomTool,
|
||||
CustomProcessorResult,
|
||||
} from "@app/hooks/tools/shared/toolOperationTypes";
|
||||
import { deriveName } from "@app/hooks/tools/shared/docparseFilenames";
|
||||
import {
|
||||
ParseDocumentParameters,
|
||||
defaultParameters,
|
||||
} from "@app/hooks/tools/parseDocument/useParseDocumentParameters";
|
||||
|
||||
// Not part of the generated ToolEndpoint union; DocParse is an optional addon.
|
||||
export const PARSE_DOCUMENT_ENDPOINT = "/api/v1/docparse/parse-document";
|
||||
|
||||
export const buildParseDocumentFormData = (
|
||||
parameters: ParseDocumentParameters,
|
||||
file: File,
|
||||
): FormData => {
|
||||
const formData = new FormData();
|
||||
formData.append("fileInput", file);
|
||||
formData.append("mode", parameters.mode);
|
||||
formData.append("withOcr", String(parameters.withOcr));
|
||||
formData.append("outputFormat", parameters.outputFormat);
|
||||
return formData;
|
||||
};
|
||||
|
||||
/** POST the PDF; wrap the JSON or markdown result as a downloadable file. */
|
||||
const processParseDocument = async (
|
||||
parameters: ParseDocumentParameters,
|
||||
files: File[],
|
||||
): Promise<CustomProcessorResult> => {
|
||||
if (files.length === 0) return { files: [] };
|
||||
|
||||
const [inputFile] = files;
|
||||
const response = await apiClient.post<Blob>(
|
||||
PARSE_DOCUMENT_ENDPOINT,
|
||||
buildParseDocumentFormData(parameters, inputFile),
|
||||
{ responseType: "blob" },
|
||||
);
|
||||
|
||||
const isMarkdown = parameters.outputFormat === "markdown";
|
||||
const resultFile = new File(
|
||||
[response.data],
|
||||
deriveName(inputFile.name, isMarkdown ? ".md" : ".parsed.json"),
|
||||
{ type: isMarkdown ? "text/markdown" : "application/json" },
|
||||
);
|
||||
return { files: [resultFile] };
|
||||
};
|
||||
|
||||
export const parseDocumentOperationConfig =
|
||||
defineCustomTool<ParseDocumentParameters>({
|
||||
operationType: "parseDocument",
|
||||
endpoint: PARSE_DOCUMENT_ENDPOINT,
|
||||
customProcessor: processParseDocument,
|
||||
defaultParameters,
|
||||
});
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { useToolOperation } from "@app/hooks/tools/shared/useToolOperation";
|
||||
import { createStandardErrorHandler } from "@app/utils/toolErrorHandler";
|
||||
import { parseDocumentOperationConfig } from "@app/hooks/tools/parseDocument/parseDocumentOperationConfig";
|
||||
|
||||
export const useParseDocumentOperation = () => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
return useToolOperation({
|
||||
...parseDocumentOperationConfig,
|
||||
getErrorMessage: createStandardErrorHandler(
|
||||
t("parseDocument.error.failed", "Failed to parse document"),
|
||||
),
|
||||
});
|
||||
};
|
||||
+30
@@ -0,0 +1,30 @@
|
||||
import { BaseParameters } from "@app/types/parameters";
|
||||
import {
|
||||
useBaseParameters,
|
||||
BaseParametersHook,
|
||||
} from "@app/hooks/tools/shared/useBaseParameters";
|
||||
|
||||
export type DocparseMode = "auto" | "basic" | "advanced";
|
||||
|
||||
export interface ParseDocumentParameters extends BaseParameters {
|
||||
mode: DocparseMode;
|
||||
outputFormat: "json" | "markdown";
|
||||
withOcr: boolean;
|
||||
}
|
||||
|
||||
// withOcr defaults true to match the API's default behaviour.
|
||||
export const defaultParameters: ParseDocumentParameters = {
|
||||
mode: "auto",
|
||||
outputFormat: "json",
|
||||
withOcr: true,
|
||||
};
|
||||
|
||||
export type ParseDocumentParametersHook =
|
||||
BaseParametersHook<ParseDocumentParameters>;
|
||||
|
||||
export const useParseDocumentParameters = (): ParseDocumentParametersHook => {
|
||||
return useBaseParameters({
|
||||
defaultParameters,
|
||||
endpointName: "parse-document",
|
||||
});
|
||||
};
|
||||
@@ -0,0 +1,60 @@
|
||||
import JSZip from "jszip";
|
||||
import apiClient from "@app/services/apiClient";
|
||||
import {
|
||||
defineCustomTool,
|
||||
CustomProcessorResult,
|
||||
} from "@app/hooks/tools/shared/toolOperationTypes";
|
||||
import {
|
||||
SmartSplitParameters,
|
||||
defaultParameters,
|
||||
} from "@app/hooks/tools/smartSplit/useSmartSplitParameters";
|
||||
|
||||
export const SMART_SPLIT_ENDPOINT = "/api/v1/docparse/smart-split";
|
||||
|
||||
export const buildSmartSplitFormData = (
|
||||
parameters: SmartSplitParameters,
|
||||
file: File,
|
||||
): FormData => {
|
||||
const formData = new FormData();
|
||||
formData.append("fileInput", file);
|
||||
formData.append("rule", parameters.rule.trim());
|
||||
formData.append("maxParts", String(parameters.maxParts));
|
||||
return formData;
|
||||
};
|
||||
|
||||
/** POST the PDF + rule; unpack the returned ZIP into the sub-PDFs. */
|
||||
const processSmartSplit = async (
|
||||
parameters: SmartSplitParameters,
|
||||
files: File[],
|
||||
): Promise<CustomProcessorResult> => {
|
||||
if (files.length === 0) return { files: [] };
|
||||
|
||||
const [inputFile] = files;
|
||||
const response = await apiClient.post<Blob>(
|
||||
SMART_SPLIT_ENDPOINT,
|
||||
buildSmartSplitFormData(parameters, inputFile),
|
||||
{ responseType: "blob" },
|
||||
);
|
||||
|
||||
const zip = await JSZip.loadAsync(response.data);
|
||||
const entries = Object.values(zip.files).filter((entry) => !entry.dir);
|
||||
const parts = await Promise.all(
|
||||
entries.map(
|
||||
async (entry) =>
|
||||
new File([await entry.async("blob")], entry.name.split("/").pop()!, {
|
||||
type: "application/pdf",
|
||||
}),
|
||||
),
|
||||
);
|
||||
// One input becomes N parts, so filename-based input mapping cannot apply.
|
||||
return { files: parts, consumedAllInputs: true };
|
||||
};
|
||||
|
||||
export const smartSplitOperationConfig = defineCustomTool<SmartSplitParameters>(
|
||||
{
|
||||
operationType: "smartSplit",
|
||||
endpoint: SMART_SPLIT_ENDPOINT,
|
||||
customProcessor: processSmartSplit,
|
||||
defaultParameters,
|
||||
},
|
||||
);
|
||||
@@ -0,0 +1,15 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { useToolOperation } from "@app/hooks/tools/shared/useToolOperation";
|
||||
import { createStandardErrorHandler } from "@app/utils/toolErrorHandler";
|
||||
import { smartSplitOperationConfig } from "@app/hooks/tools/smartSplit/smartSplitOperationConfig";
|
||||
|
||||
export const useSmartSplitOperation = () => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
return useToolOperation({
|
||||
...smartSplitOperationConfig,
|
||||
getErrorMessage: createStandardErrorHandler(
|
||||
t("smartSplit.error.failed", "Failed to split document"),
|
||||
),
|
||||
});
|
||||
};
|
||||
@@ -0,0 +1,27 @@
|
||||
import { BaseParameters } from "@app/types/parameters";
|
||||
import {
|
||||
useBaseParameters,
|
||||
BaseParametersHook,
|
||||
} from "@app/hooks/tools/shared/useBaseParameters";
|
||||
|
||||
export interface SmartSplitParameters extends BaseParameters {
|
||||
/** Natural-language boundary rule, e.g. "split at each new invoice". */
|
||||
rule: string;
|
||||
maxParts: number;
|
||||
}
|
||||
|
||||
export const defaultParameters: SmartSplitParameters = {
|
||||
rule: "",
|
||||
maxParts: 10,
|
||||
};
|
||||
|
||||
export type SmartSplitParametersHook = BaseParametersHook<SmartSplitParameters>;
|
||||
|
||||
export const useSmartSplitParameters = (): SmartSplitParametersHook => {
|
||||
return useBaseParameters({
|
||||
defaultParameters,
|
||||
endpointName: "smart-split",
|
||||
validateFn: (params) =>
|
||||
params.rule.trim().length > 0 && params.maxParts > 0,
|
||||
});
|
||||
};
|
||||
@@ -0,0 +1,55 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { createToolFlow } from "@app/components/tools/shared/createToolFlow";
|
||||
import { useBaseTool } from "@app/hooks/tools/shared/useBaseTool";
|
||||
import type { BaseToolProps } from "@app/types/tool";
|
||||
import ChunkDocumentSettings from "@app/components/tools/docparse/ChunkDocumentSettings";
|
||||
import { useChunkDocumentParameters } from "@app/hooks/tools/chunkDocument/useChunkDocumentParameters";
|
||||
import { useChunkDocumentOperation } from "@app/hooks/tools/chunkDocument/useChunkDocumentOperation";
|
||||
|
||||
const ChunkDocument = (props: BaseToolProps) => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
const base = useBaseTool(
|
||||
"chunkDocument",
|
||||
useChunkDocumentParameters,
|
||||
useChunkDocumentOperation,
|
||||
props,
|
||||
);
|
||||
|
||||
return createToolFlow({
|
||||
files: {
|
||||
selectedFiles: base.selectedFiles,
|
||||
isCollapsed: base.hasResults,
|
||||
},
|
||||
steps: [
|
||||
{
|
||||
title: t("chunkDocument.settings.title", "Chunking settings"),
|
||||
isCollapsed: false,
|
||||
content: (
|
||||
<ChunkDocumentSettings
|
||||
parameters={base.params.parameters}
|
||||
onParameterChange={base.params.updateParameter}
|
||||
disabled={base.endpointLoading}
|
||||
/>
|
||||
),
|
||||
},
|
||||
],
|
||||
executeButton: {
|
||||
text: t("chunkDocument.submit", "Prepare chunks"),
|
||||
isVisible: !base.hasResults,
|
||||
loadingText: t("loading"),
|
||||
onClick: base.handleExecute,
|
||||
endpointEnabled: base.endpointEnabled,
|
||||
paramsValid: base.params.validateParameters(),
|
||||
},
|
||||
review: {
|
||||
isVisible: base.hasResults,
|
||||
operation: base.operation,
|
||||
title: t("chunkDocument.results.title", "Chunks (JSONL)"),
|
||||
onFileClick: base.handleThumbnailClick,
|
||||
onUndo: base.handleUndo,
|
||||
},
|
||||
});
|
||||
};
|
||||
|
||||
export default ChunkDocument;
|
||||
@@ -0,0 +1,55 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { createToolFlow } from "@app/components/tools/shared/createToolFlow";
|
||||
import { useBaseTool } from "@app/hooks/tools/shared/useBaseTool";
|
||||
import type { BaseToolProps } from "@app/types/tool";
|
||||
import FillTemplateSettings from "@app/components/tools/docparse/FillTemplateSettings";
|
||||
import { useFillTemplateParameters } from "@app/hooks/tools/fillTemplate/useFillTemplateParameters";
|
||||
import { useFillTemplateOperation } from "@app/hooks/tools/fillTemplate/useFillTemplateOperation";
|
||||
|
||||
const FillTemplate = (props: BaseToolProps) => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
const base = useBaseTool(
|
||||
"fillTemplate",
|
||||
useFillTemplateParameters,
|
||||
useFillTemplateOperation,
|
||||
props,
|
||||
);
|
||||
|
||||
return createToolFlow({
|
||||
files: {
|
||||
selectedFiles: base.selectedFiles,
|
||||
isCollapsed: base.hasResults,
|
||||
},
|
||||
steps: [
|
||||
{
|
||||
title: t("fillTemplate.settings.title", "Template data"),
|
||||
isCollapsed: false,
|
||||
content: (
|
||||
<FillTemplateSettings
|
||||
parameters={base.params.parameters}
|
||||
onParameterChange={base.params.updateParameter}
|
||||
disabled={base.endpointLoading}
|
||||
/>
|
||||
),
|
||||
},
|
||||
],
|
||||
executeButton: {
|
||||
text: t("fillTemplate.submit", "Fill template"),
|
||||
isVisible: !base.hasResults,
|
||||
loadingText: t("loading"),
|
||||
onClick: base.handleExecute,
|
||||
endpointEnabled: base.endpointEnabled,
|
||||
paramsValid: base.params.validateParameters(),
|
||||
},
|
||||
review: {
|
||||
isVisible: base.hasResults,
|
||||
operation: base.operation,
|
||||
title: t("fillTemplate.results.title", "Filled document"),
|
||||
onFileClick: base.handleThumbnailClick,
|
||||
onUndo: base.handleUndo,
|
||||
},
|
||||
});
|
||||
};
|
||||
|
||||
export default FillTemplate;
|
||||
@@ -0,0 +1,55 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { createToolFlow } from "@app/components/tools/shared/createToolFlow";
|
||||
import { useBaseTool } from "@app/hooks/tools/shared/useBaseTool";
|
||||
import type { BaseToolProps } from "@app/types/tool";
|
||||
import ParseDocumentSettings from "@app/components/tools/docparse/ParseDocumentSettings";
|
||||
import { useParseDocumentParameters } from "@app/hooks/tools/parseDocument/useParseDocumentParameters";
|
||||
import { useParseDocumentOperation } from "@app/hooks/tools/parseDocument/useParseDocumentOperation";
|
||||
|
||||
const ParseDocument = (props: BaseToolProps) => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
const base = useBaseTool(
|
||||
"parseDocument",
|
||||
useParseDocumentParameters,
|
||||
useParseDocumentOperation,
|
||||
props,
|
||||
);
|
||||
|
||||
return createToolFlow({
|
||||
files: {
|
||||
selectedFiles: base.selectedFiles,
|
||||
isCollapsed: base.hasResults,
|
||||
},
|
||||
steps: [
|
||||
{
|
||||
title: t("parseDocument.settings.title", "Parse settings"),
|
||||
isCollapsed: false,
|
||||
content: (
|
||||
<ParseDocumentSettings
|
||||
parameters={base.params.parameters}
|
||||
onParameterChange={base.params.updateParameter}
|
||||
disabled={base.endpointLoading}
|
||||
/>
|
||||
),
|
||||
},
|
||||
],
|
||||
executeButton: {
|
||||
text: t("parseDocument.submit", "Parse document"),
|
||||
isVisible: !base.hasResults,
|
||||
loadingText: t("loading"),
|
||||
onClick: base.handleExecute,
|
||||
endpointEnabled: base.endpointEnabled,
|
||||
paramsValid: base.params.validateParameters(),
|
||||
},
|
||||
review: {
|
||||
isVisible: base.hasResults,
|
||||
operation: base.operation,
|
||||
title: t("parseDocument.results.title", "Parsed output"),
|
||||
onFileClick: base.handleThumbnailClick,
|
||||
onUndo: base.handleUndo,
|
||||
},
|
||||
});
|
||||
};
|
||||
|
||||
export default ParseDocument;
|
||||
@@ -0,0 +1,55 @@
|
||||
import { useTranslation } from "react-i18next";
|
||||
import { createToolFlow } from "@app/components/tools/shared/createToolFlow";
|
||||
import { useBaseTool } from "@app/hooks/tools/shared/useBaseTool";
|
||||
import type { BaseToolProps } from "@app/types/tool";
|
||||
import SmartSplitSettings from "@app/components/tools/docparse/SmartSplitSettings";
|
||||
import { useSmartSplitParameters } from "@app/hooks/tools/smartSplit/useSmartSplitParameters";
|
||||
import { useSmartSplitOperation } from "@app/hooks/tools/smartSplit/useSmartSplitOperation";
|
||||
|
||||
const SmartSplit = (props: BaseToolProps) => {
|
||||
const { t } = useTranslation();
|
||||
|
||||
const base = useBaseTool(
|
||||
"smartSplit",
|
||||
useSmartSplitParameters,
|
||||
useSmartSplitOperation,
|
||||
props,
|
||||
);
|
||||
|
||||
return createToolFlow({
|
||||
files: {
|
||||
selectedFiles: base.selectedFiles,
|
||||
isCollapsed: base.hasResults,
|
||||
},
|
||||
steps: [
|
||||
{
|
||||
title: t("smartSplit.settings.title", "Split settings"),
|
||||
isCollapsed: false,
|
||||
content: (
|
||||
<SmartSplitSettings
|
||||
parameters={base.params.parameters}
|
||||
onParameterChange={base.params.updateParameter}
|
||||
disabled={base.endpointLoading}
|
||||
/>
|
||||
),
|
||||
},
|
||||
],
|
||||
executeButton: {
|
||||
text: t("smartSplit.submit", "Split document"),
|
||||
isVisible: !base.hasResults,
|
||||
loadingText: t("loading"),
|
||||
onClick: base.handleExecute,
|
||||
endpointEnabled: base.endpointEnabled,
|
||||
paramsValid: base.params.validateParameters(),
|
||||
},
|
||||
review: {
|
||||
isVisible: base.hasResults,
|
||||
operation: base.operation,
|
||||
title: t("smartSplit.results.title", "Split documents"),
|
||||
onFileClick: base.handleThumbnailClick,
|
||||
onUndo: base.handleUndo,
|
||||
},
|
||||
});
|
||||
};
|
||||
|
||||
export default SmartSplit;
|
||||
@@ -5,7 +5,13 @@
|
||||
*/
|
||||
|
||||
// The DocParse tool family; visible only when the backend reports docparseEnabled.
|
||||
export const PROPRIETARY_REGULAR_TOOL_IDS = ["extractFields"] as const;
|
||||
export const PROPRIETARY_REGULAR_TOOL_IDS = [
|
||||
"parseDocument",
|
||||
"extractFields",
|
||||
"smartSplit",
|
||||
"chunkDocument",
|
||||
"fillTemplate",
|
||||
] as const;
|
||||
|
||||
// "ai-workflow" is a generic marker stamped onto files produced by the agents
|
||||
// chat orchestrator (which may invoke one or more underlying tools). Lives here
|
||||
|
||||
Reference in New Issue
Block a user