mirror of
https://github.com/Stirling-Tools/Stirling-PDF.git
synced 2026-09-03 05:10:16 +03:00
Compare commits
32
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b6e19c9a96 | ||
|
|
ec67f7e431 | ||
|
|
2f1fe2c80c | ||
|
|
1051f28c52 | ||
|
|
29ccbf7ae6 | ||
|
|
007c8e17de | ||
|
|
81dbb29be3 | ||
|
|
caef0477a9 | ||
|
|
7682a0dd54 | ||
|
|
f90ed4657a | ||
|
|
a6c7a68242 | ||
|
|
a184b394d6 | ||
|
|
302d04c201 | ||
|
|
68ae9d52fb | ||
|
|
6552ba905c | ||
|
|
ee36b6f616 | ||
|
|
627091f8df | ||
|
|
67a37d3291 | ||
|
|
4b6d4885f4 | ||
|
|
2a151b65f7 | ||
|
|
9458fcd0e2 | ||
|
|
1e8c41425b | ||
|
|
b3277a18c8 | ||
|
|
5ca1586976 | ||
|
|
ce47a4e3af | ||
|
|
859a2d97c2 | ||
|
|
a422deecdb | ||
|
|
7f8f09c899 | ||
|
|
96920b1186 | ||
|
|
3626319685 | ||
|
|
755f270a31 | ||
|
|
6e1a7454ca |
@@ -1,7 +1,7 @@
|
||||
name: Backend build, format check, and coverage
|
||||
|
||||
# Reusable workflow called from build.yml. Runs the full backend build matrix
|
||||
# (JDK 21/25 × spring-security on/off), Spotless formatting check, JUnit, and
|
||||
# Reusable workflow called from build.yml. Runs the backend build matrix
|
||||
# (JDK 25 × spring-security on/off), Spotless formatting check, JUnit, and
|
||||
# posts Jacoco coverage to PRs.
|
||||
on:
|
||||
workflow_call:
|
||||
@@ -18,7 +18,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
jdk-version: [21, 25]
|
||||
jdk-version: [25]
|
||||
spring-security: [true, false]
|
||||
steps:
|
||||
- name: Harden Runner
|
||||
|
||||
@@ -90,21 +90,21 @@ jobs:
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
case "${{ github.event.inputs.platform }}" in
|
||||
"windows")
|
||||
echo 'matrix={"include":[{"platform":"windows-latest","args":"--target x86_64-pc-windows-msvc","name":"windows-x86_64"}]}' >> $GITHUB_OUTPUT
|
||||
echo 'matrix={"include":[{"platform":"windows-latest","args":"--target x86_64-pc-windows-msvc","name":"windows-x86_64","jpdfium_platforms":"windows-x64"}]}' >> $GITHUB_OUTPUT
|
||||
;;
|
||||
"macos")
|
||||
echo 'matrix={"include":[{"platform":"macos-15","args":"--target universal-apple-darwin","name":"macos-universal"}]}' >> $GITHUB_OUTPUT
|
||||
echo 'matrix={"include":[{"platform":"macos-15","args":"--target universal-apple-darwin","name":"macos-universal","jpdfium_platforms":"darwin-arm64,darwin-x64"}]}' >> $GITHUB_OUTPUT
|
||||
;;
|
||||
"linux")
|
||||
echo 'matrix={"include":[{"platform":"ubuntu-22.04","args":"","name":"linux-x86_64"}]}' >> $GITHUB_OUTPUT
|
||||
echo 'matrix={"include":[{"platform":"ubuntu-22.04","args":"","name":"linux-x86_64","jpdfium_platforms":"linux-x64"}]}' >> $GITHUB_OUTPUT
|
||||
;;
|
||||
*)
|
||||
echo 'matrix={"include":[{"platform":"windows-latest","args":"--target x86_64-pc-windows-msvc","name":"windows-x86_64"},{"platform":"macos-15","args":"--target universal-apple-darwin","name":"macos-universal"},{"platform":"ubuntu-22.04","args":"","name":"linux-x86_64"}]}' >> $GITHUB_OUTPUT
|
||||
echo 'matrix={"include":[{"platform":"windows-latest","args":"--target x86_64-pc-windows-msvc","name":"windows-x86_64","jpdfium_platforms":"windows-x64"},{"platform":"macos-15","args":"--target universal-apple-darwin","name":"macos-universal","jpdfium_platforms":"darwin-arm64,darwin-x64"},{"platform":"ubuntu-22.04","args":"","name":"linux-x86_64","jpdfium_platforms":"linux-x64"}]}' >> $GITHUB_OUTPUT
|
||||
;;
|
||||
esac
|
||||
else
|
||||
# For push/release events, build all platforms
|
||||
echo 'matrix={"include":[{"platform":"windows-latest","args":"--target x86_64-pc-windows-msvc","name":"windows-x86_64"},{"platform":"macos-15","args":"--target universal-apple-darwin","name":"macos-universal"},{"platform":"ubuntu-22.04","args":"","name":"linux-x86_64"}]}' >> $GITHUB_OUTPUT
|
||||
echo 'matrix={"include":[{"platform":"windows-latest","args":"--target x86_64-pc-windows-msvc","name":"windows-x86_64","jpdfium_platforms":"windows-x64"},{"platform":"macos-15","args":"--target universal-apple-darwin","name":"macos-universal","jpdfium_platforms":"darwin-arm64,darwin-x64"},{"platform":"ubuntu-22.04","args":"","name":"linux-x86_64","jpdfium_platforms":"linux-x64"}]}' >> $GITHUB_OUTPUT
|
||||
fi
|
||||
|
||||
build-jars:
|
||||
@@ -256,15 +256,17 @@ jobs:
|
||||
if: matrix.platform == 'macos-15'
|
||||
env:
|
||||
AARCH64_JAVA_HOME: ${{ env.JAVA_HOME }}
|
||||
JPDFIUM_PLATFORMS: ${{ matrix.jpdfium_platforms }}
|
||||
run: task desktop:jlink:universal-mac
|
||||
|
||||
- name: Prepare desktop build
|
||||
run: task desktop:prepare
|
||||
env:
|
||||
MAVEN_USER: ${{ secrets.MAVEN_USER }}
|
||||
MAVEN_PASSWORD: ${{ secrets.MAVEN_PASSWORD }}
|
||||
MAVEN_PUBLIC_URL: ${{ secrets.MAVEN_PUBLIC_URL }}
|
||||
DISABLE_ADDITIONAL_FEATURES: true
|
||||
JPDFIUM_PLATFORMS: ${{ matrix.jpdfium_platforms }}
|
||||
run: task desktop:prepare
|
||||
|
||||
# DigiCert KeyLocker Setup (Cloud HSM)
|
||||
- name: Setup DigiCert KeyLocker
|
||||
|
||||
@@ -47,12 +47,10 @@ jobs:
|
||||
APPLE_CERTIFICATE: ${{ secrets.APPLE_CERTIFICATE }}
|
||||
PLATFORM: ${{ inputs.platform }}
|
||||
run: |
|
||||
WINDOWS='{"platform":"windows-latest","args":"--target x86_64-pc-windows-msvc","name":"windows-x86_64"}'
|
||||
MACOS='{"platform":"macos-15","args":"--target universal-apple-darwin","name":"macos-universal"}'
|
||||
LINUX='{"platform":"ubuntu-22.04","args":"","name":"linux-x86_64"}'
|
||||
WINDOWS='{"platform":"windows-latest","args":"--target x86_64-pc-windows-msvc","name":"windows-x86_64","jpdfium_platforms":"windows-x64"}'
|
||||
MACOS='{"platform":"macos-15","args":"--target universal-apple-darwin","name":"macos-universal","jpdfium_platforms":"darwin-arm64,darwin-x64"}'
|
||||
LINUX='{"platform":"ubuntu-22.04","args":"","name":"linux-x86_64","jpdfium_platforms":"linux-x64"}'
|
||||
|
||||
# Resolve requested platform — populated by either workflow_dispatch
|
||||
# or workflow_call inputs; both paths default to "all".
|
||||
case "$PLATFORM" in
|
||||
windows) ENTRIES=("$WINDOWS") ;;
|
||||
macos) ENTRIES=("$MACOS") ;;
|
||||
@@ -112,10 +110,6 @@ jobs:
|
||||
toolchain: stable
|
||||
targets: ${{ matrix.platform == 'macos-15' && 'aarch64-apple-darwin,x86_64-apple-darwin' || '' }}
|
||||
|
||||
# x86_64 JDK is set up first so the aarch64 step below can leave its
|
||||
# JAVA_HOME as the active one. The macOS universal JRE build needs
|
||||
# jmods from both arches; the x64 path is captured into the env
|
||||
# before the second setup-java overwrites JAVA_HOME.
|
||||
- name: Set up x86_64 JDK 25 (macOS universal JRE)
|
||||
if: matrix.platform == 'macos-15'
|
||||
uses: actions/setup-java@be666c2fcd27ec809703dec50e508c2fdc7f6654 # v5.2.0
|
||||
@@ -142,21 +136,21 @@ jobs:
|
||||
- name: Setup Task
|
||||
uses: go-task/setup-task@3be4020d41929789a01026e0e427a4321ce0ad44 # v2.0.0
|
||||
|
||||
# Build the universal JRE before desktop:prepare so the jlink:runtime
|
||||
# task short-circuits on its `test -d runtime/jre` status check.
|
||||
- name: Build universal macOS JRE
|
||||
if: matrix.platform == 'macos-15'
|
||||
env:
|
||||
AARCH64_JAVA_HOME: ${{ env.JAVA_HOME }}
|
||||
JPDFIUM_PLATFORMS: ${{ matrix.jpdfium_platforms }}
|
||||
run: task desktop:jlink:universal-mac
|
||||
|
||||
- name: Prepare desktop build
|
||||
run: task desktop:prepare
|
||||
env:
|
||||
MAVEN_USER: ${{ secrets.MAVEN_USER }}
|
||||
MAVEN_PASSWORD: ${{ secrets.MAVEN_PASSWORD }}
|
||||
MAVEN_PUBLIC_URL: ${{ secrets.MAVEN_PUBLIC_URL }}
|
||||
DISABLE_ADDITIONAL_FEATURES: true
|
||||
JPDFIUM_PLATFORMS: ${{ matrix.jpdfium_platforms }}
|
||||
run: task desktop:prepare
|
||||
|
||||
# DigiCert KeyLocker Setup (Cloud HSM)
|
||||
- name: Setup DigiCert KeyLocker
|
||||
@@ -269,6 +263,10 @@ jobs:
|
||||
echo "APPLE_SIGNING_IDENTITY=$CERT_ID" >> $GITHUB_ENV
|
||||
echo "Certificate imported successfully."
|
||||
|
||||
- name: Sign JPDFium dylibs inside bootJar (macOS only)
|
||||
if: matrix.platform == 'macos-15' && env.APPLE_CERTIFICATE != ''
|
||||
run: bash frontend/scripts/sign-jpdfium-dylibs-in-bootjar.sh
|
||||
|
||||
- name: Check DMG creation dependencies (macOS only)
|
||||
if: matrix.platform == 'macos-15'
|
||||
run: |
|
||||
|
||||
+20
-3
@@ -3,6 +3,22 @@ version: '3'
|
||||
vars:
|
||||
JLINK_MODULES: "java.base,java.compiler,java.desktop,java.instrument,java.logging,java.management,java.naming,java.net.http,java.prefs,java.rmi,java.scripting,java.security.jgss,java.security.sasl,java.sql,java.transaction.xa,java.xml,java.xml.crypto,jdk.crypto.ec,jdk.crypto.cryptoki,jdk.unsupported"
|
||||
|
||||
# Override via JPDFIUM_PLATFORMS env (csv of platform keys, or 'all').
|
||||
JPDFIUM_PLATFORMS:
|
||||
sh: |
|
||||
if [ -n "${JPDFIUM_PLATFORMS:-}" ]; then
|
||||
echo "$JPDFIUM_PLATFORMS"
|
||||
else
|
||||
case "{{OS}}-{{ARCH}}" in
|
||||
darwin-arm64) echo "darwin-arm64";;
|
||||
darwin-amd64) echo "darwin-x64";;
|
||||
linux-amd64) echo "linux-x64";;
|
||||
linux-arm64) echo "linux-arm64";;
|
||||
windows-amd64) echo "windows-x64";;
|
||||
*) echo "all";;
|
||||
esac
|
||||
fi
|
||||
|
||||
tasks:
|
||||
prepare:
|
||||
desc: "Prepare desktop build dependencies"
|
||||
@@ -71,15 +87,16 @@ tasks:
|
||||
deps: [jlink:jar, jlink:runtime]
|
||||
|
||||
jlink:jar:
|
||||
desc: "Build backend JAR for Tauri bundling"
|
||||
desc: "Build backend JAR for Tauri bundling (host-OS natives only by default)"
|
||||
run: once
|
||||
dir: ..
|
||||
env:
|
||||
DISABLE_ADDITIONAL_FEATURES: "true"
|
||||
cmds:
|
||||
- cmd: cmd /c gradlew.bat bootJar --no-daemon
|
||||
- echo "Building bootJar with JPDFium natives for {{.JPDFIUM_PLATFORMS}}"
|
||||
- cmd: cmd /c gradlew.bat bootJar --no-daemon -PjpdfiumPlatforms={{.JPDFIUM_PLATFORMS}}
|
||||
platforms: [windows]
|
||||
- cmd: ./gradlew bootJar --no-daemon
|
||||
- cmd: ./gradlew bootJar --no-daemon -PjpdfiumPlatforms={{.JPDFIUM_PLATFORMS}}
|
||||
platforms: [linux, darwin]
|
||||
- mkdir -p frontend/src-tauri/libs
|
||||
- cp app/core/build/libs/stirling-pdf-*.jar frontend/src-tauri/libs/
|
||||
|
||||
@@ -431,7 +431,7 @@ The frontend is organized with a clear separation of concerns:
|
||||
|
||||
## Important Notes
|
||||
|
||||
- **Java Version**: Minimum JDK 21, supports and recommends JDK 25
|
||||
- **Java Version**: Requires JDK 25.
|
||||
- **Lombok**: Used extensively - ensure IDE plugin is installed
|
||||
- **File Persistence**:
|
||||
- **Backend**: Designed to be stateless - files are processed in memory/temp locations only
|
||||
|
||||
+3
-3
@@ -11,7 +11,7 @@ This guide focuses on developing for Stirling 2.0, including both the React fron
|
||||
**Stirling 2.0** is built using:
|
||||
|
||||
**Backend:**
|
||||
- Spring Boot (Java 21+, JDK 25 recommended)
|
||||
- Spring Boot (requires JDK 25)
|
||||
- PDFBox for core PDF operations
|
||||
- LibreOffice for document conversions
|
||||
- qpdf for PDF optimization
|
||||
@@ -45,7 +45,7 @@ This guide focuses on developing for Stirling 2.0, including both the React fron
|
||||
- [Task](https://taskfile.dev/installation/) — unified command runner (recommended)
|
||||
- Docker
|
||||
- Git
|
||||
- Java JDK 21 or later (JDK 25 recommended)
|
||||
- Java JDK 25
|
||||
- Node.js 18+ and npm (required for frontend development)
|
||||
- Gradle 7.0 or later (Included within the repo)
|
||||
- [uv](https://docs.astral.sh/uv/) — Python package manager (required for engine development)
|
||||
@@ -61,7 +61,7 @@ This guide focuses on developing for Stirling 2.0, including both the React fron
|
||||
cd Stirling-PDF
|
||||
```
|
||||
|
||||
2. Install Docker and JDK 21 (or JDK 25 recommended) if not already installed.
|
||||
2. Install Docker and JDK 25 if not already installed.
|
||||
|
||||
3. Install a recommended Java IDE such as Eclipse, IntelliJ, or VSCode
|
||||
1. Only VSCode
|
||||
|
||||
@@ -60,6 +60,24 @@ dependencies {
|
||||
exclude group: 'com.google.code.gson', module: 'gson'
|
||||
}
|
||||
|
||||
api 'com.stirling:jpdfium:1.0.0'
|
||||
|
||||
// -PjpdfiumPlatforms=all|<csv of linux-x64,linux-arm64,darwin-x64,darwin-arm64,windows-x64>
|
||||
def jpdfiumPlatformsProp = (project.findProperty('jpdfiumPlatforms') ?: 'all').toString().trim()
|
||||
def jpdfiumAllPlatforms = ['linux-x64', 'linux-arm64', 'darwin-x64', 'darwin-arm64', 'windows-x64']
|
||||
def jpdfiumPlatforms = jpdfiumPlatformsProp == 'all'
|
||||
? jpdfiumAllPlatforms
|
||||
: jpdfiumPlatformsProp.split(',').collect { it.trim() }.findAll { it }
|
||||
def jpdfiumInvalid = jpdfiumPlatforms.findAll { !jpdfiumAllPlatforms.contains(it) }
|
||||
if (jpdfiumInvalid) {
|
||||
throw new GradleException("Unknown jpdfiumPlatforms value(s): ${jpdfiumInvalid.join(', ')}. " +
|
||||
"Valid: ${jpdfiumAllPlatforms.join(', ')} or 'all'.")
|
||||
}
|
||||
logger.lifecycle("JPDFium native platforms: ${jpdfiumPlatforms.join(', ')}")
|
||||
jpdfiumPlatforms.each { platform ->
|
||||
runtimeOnly "com.stirling:jpdfium-natives-${platform}:1.0.0"
|
||||
}
|
||||
|
||||
// ArchUnit: enforces module dependency direction (see ArchitectureTest)
|
||||
testImplementation 'com.tngtech.archunit:archunit-junit5:1.4.2'
|
||||
}
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
package stirling.software.common.jpdfium;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import stirling.software.jpdfium.PdfDocument;
|
||||
|
||||
class JPDFiumSmokeTest {
|
||||
|
||||
@Test
|
||||
void opensExamplePdfAndReadsPageCount(@TempDir Path tmp) throws IOException {
|
||||
Path pdf = tmp.resolve("example.pdf");
|
||||
try (InputStream in = getClass().getResourceAsStream("/example.pdf")) {
|
||||
assertNotNull(in, "example.pdf must exist under src/test/resources");
|
||||
Files.copy(in, pdf);
|
||||
}
|
||||
|
||||
try (PdfDocument doc = PdfDocument.open(pdf)) {
|
||||
assertTrue(
|
||||
doc.pageCount() >= 1,
|
||||
"PdfDocument should report at least one page for example.pdf");
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -137,6 +137,13 @@ sourceSets {
|
||||
}
|
||||
|
||||
|
||||
// Forward jpdfium.bench to the forked test JVM so EnabledIfSystemProperty fires
|
||||
tasks.named('test') {
|
||||
if (System.getProperty('jpdfium.bench')) {
|
||||
systemProperty 'jpdfium.bench', System.getProperty('jpdfium.bench')
|
||||
}
|
||||
}
|
||||
|
||||
// Disable regular jar
|
||||
jar {
|
||||
enabled = false
|
||||
@@ -162,7 +169,8 @@ bootJar {
|
||||
manifest {
|
||||
attributes(
|
||||
'Implementation-Title': 'Stirling-PDF',
|
||||
'Implementation-Version': project.version
|
||||
'Implementation-Version': project.version,
|
||||
'Enable-Native-Access': 'ALL-UNNAMED'
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
+166
-64
@@ -3,13 +3,14 @@ package stirling.software.SPDF.controller.api;
|
||||
import java.io.File;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Arrays;
|
||||
import java.util.Comparator;
|
||||
import java.util.List;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.apache.pdfbox.multipdf.PDFMergerUtility;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDDocumentCatalog;
|
||||
import org.apache.pdfbox.pdmodel.PDDocumentInformation;
|
||||
@@ -47,6 +48,11 @@ import stirling.software.common.util.PdfErrorUtils;
|
||||
import stirling.software.common.util.TempFile;
|
||||
import stirling.software.common.util.TempFileManager;
|
||||
import stirling.software.common.util.WebResponseUtils;
|
||||
import stirling.software.jpdfium.PdfDocument;
|
||||
import stirling.software.jpdfium.PdfMerge;
|
||||
import stirling.software.jpdfium.doc.Bookmark;
|
||||
import stirling.software.jpdfium.doc.PdfBookmarkEditor;
|
||||
import stirling.software.jpdfium.doc.PdfBookmarkEditor.BookmarkTree;
|
||||
|
||||
@GeneralApi
|
||||
@Slf4j
|
||||
@@ -57,7 +63,6 @@ public class MergeController {
|
||||
private final CustomPDFDocumentFactory pdfDocumentFactory;
|
||||
private final TempFileManager tempFileManager;
|
||||
|
||||
// Merges a list of PDDocument objects into a single PDDocument
|
||||
public PDDocument mergeDocuments(List<PDDocument> documents) throws IOException {
|
||||
PDDocument mergedDoc = pdfDocumentFactory.createNewDocument();
|
||||
boolean success = false;
|
||||
@@ -76,11 +81,8 @@ public class MergeController {
|
||||
}
|
||||
}
|
||||
|
||||
// Re-order files to match the explicit order provided by the front-end.
|
||||
// fileOrder is newline-delimited original filenames in the desired order.
|
||||
private static MultipartFile[] reorderFilesByProvidedOrder(
|
||||
MultipartFile[] files, String fileOrder) {
|
||||
// Split by various line endings and trim each entry
|
||||
String[] desired =
|
||||
stirling.software.common.util.RegexPatternUtils.getInstance()
|
||||
.getNewlineSplitPattern()
|
||||
@@ -107,7 +109,6 @@ public class MergeController {
|
||||
return ordered.toArray(new MultipartFile[0]);
|
||||
}
|
||||
|
||||
// Returns a comparator for sorting MultipartFile arrays based on the given sort type
|
||||
private Comparator<MultipartFile> getSortComparator(String sortType) {
|
||||
return switch (sortType) {
|
||||
case "byFileName" ->
|
||||
@@ -155,18 +156,16 @@ public class MergeController {
|
||||
return 0;
|
||||
}
|
||||
};
|
||||
case "orderProvided" -> (file1, file2) -> 0; // Default is the order provided
|
||||
default -> (file1, file2) -> 0; // Default is the order provided
|
||||
case "orderProvided" -> (file1, file2) -> 0;
|
||||
default -> (file1, file2) -> 0;
|
||||
};
|
||||
}
|
||||
|
||||
// Parse client file IDs from JSON string
|
||||
private String[] parseClientFileIds(String clientFileIds) {
|
||||
if (clientFileIds == null || clientFileIds.trim().isEmpty()) {
|
||||
return new String[0];
|
||||
}
|
||||
try {
|
||||
// Simple JSON array parsing - remove brackets and split by comma
|
||||
String trimmed = clientFileIds.trim();
|
||||
if (trimmed.startsWith("[") && trimmed.endsWith("]")) {
|
||||
String inside = trimmed.substring(1, trimmed.length() - 1).trim();
|
||||
@@ -186,39 +185,29 @@ public class MergeController {
|
||||
return new String[0];
|
||||
}
|
||||
|
||||
// Adds a table of contents to the merged document using filenames as chapter titles
|
||||
private void addTableOfContents(PDDocument mergedDocument, MultipartFile[] files) {
|
||||
// Create the document outline
|
||||
PDDocumentOutline outline = new PDDocumentOutline();
|
||||
mergedDocument.getDocumentCatalog().setDocumentOutline(outline);
|
||||
|
||||
int pageIndex = 0; // Current page index in the merged document
|
||||
|
||||
// Iterate through the original files
|
||||
int pageIndex = 0;
|
||||
for (MultipartFile file : files) {
|
||||
// Get the filename without extension to use as bookmark title
|
||||
String filename = file.getOriginalFilename();
|
||||
String title = GeneralUtils.removeExtension(filename);
|
||||
|
||||
// Create an outline item for this file
|
||||
PDOutlineItem item = new PDOutlineItem();
|
||||
item.setTitle(title);
|
||||
|
||||
// Set the destination to the first page of this file in the merged document
|
||||
if (pageIndex < mergedDocument.getNumberOfPages()) {
|
||||
PDPage page = mergedDocument.getPage(pageIndex);
|
||||
item.setDestination(page);
|
||||
}
|
||||
|
||||
// Add the item to the outline
|
||||
outline.addLast(item);
|
||||
|
||||
// Increment page index for the next file
|
||||
try (PDDocument doc = pdfDocumentFactory.load(file)) {
|
||||
pageIndex += doc.getNumberOfPages();
|
||||
} catch (IOException e) {
|
||||
ExceptionUtils.logException("document loading for TOC generation", e);
|
||||
pageIndex++; // Increment by at least one if we can't determine page count
|
||||
pageIndex++;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -236,7 +225,6 @@ public class MergeController {
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback to XMP metadata if Info dates are missing
|
||||
PDMetadata metadata = doc.getDocumentCatalog().getMetadata();
|
||||
if (metadata != null) {
|
||||
try (InputStream is = metadata.createInputStream()) {
|
||||
@@ -287,7 +275,7 @@ public class MergeController {
|
||||
@ModelAttribute MergePdfsRequest request,
|
||||
@RequestParam(value = "fileOrder", required = false) String fileOrder)
|
||||
throws IOException {
|
||||
List<File> filesToDelete = new ArrayList<>(); // List of temporary files to delete
|
||||
List<File> filesToDelete = new ArrayList<>();
|
||||
TempFile outputTempFile = null;
|
||||
|
||||
boolean removeCertSign = Boolean.TRUE.equals(request.getRemoveCertSign());
|
||||
@@ -298,48 +286,35 @@ public class MergeController {
|
||||
files = new MultipartFile[0];
|
||||
}
|
||||
|
||||
// If front-end provided explicit visible order, honor it and override backend sorting
|
||||
if (fileOrder != null && !fileOrder.isBlank()) {
|
||||
log.info("Reordering files based on fileOrder parameter");
|
||||
files = reorderFilesByProvidedOrder(files, fileOrder);
|
||||
} else {
|
||||
log.info("Sorting files based on sortType: {}", request.getSortType());
|
||||
Arrays.sort(
|
||||
files,
|
||||
getSortComparator(
|
||||
request.getSortType())); // Sort files based on requested sort type
|
||||
Arrays.sort(files, getSortComparator(request.getSortType()));
|
||||
}
|
||||
|
||||
try (TempFile mt = new TempFile(tempFileManager, ".pdf")) {
|
||||
|
||||
PDFMergerUtility mergerUtility = new PDFMergerUtility();
|
||||
long totalSize = 0;
|
||||
List<Path> inputPaths = new ArrayList<>(files.length);
|
||||
List<Integer> invalidIndexes = new ArrayList<>();
|
||||
for (int index = 0; index < files.length; index++) {
|
||||
MultipartFile multipartFile = files[index];
|
||||
totalSize += multipartFile.getSize();
|
||||
File tempFile =
|
||||
tempFileManager.convertMultipartFileToFile(
|
||||
multipartFile); // Convert MultipartFile to File
|
||||
filesToDelete.add(tempFile); // Add temp file to the list for later deletion
|
||||
File tempFile = tempFileManager.convertMultipartFileToFile(multipartFile);
|
||||
filesToDelete.add(tempFile);
|
||||
inputPaths.add(tempFile.toPath());
|
||||
|
||||
// Pre-validate each PDF so we can report which one(s) are broken
|
||||
// Use the original MultipartFile to avoid deleting the tempFile during validation
|
||||
try (PDDocument ignored = pdfDocumentFactory.load(multipartFile)) {
|
||||
// OK
|
||||
} catch (IOException e) {
|
||||
try (PdfDocument ignored = PdfDocument.open(tempFile.toPath())) {
|
||||
} catch (Exception e) {
|
||||
ExceptionUtils.logException("PDF pre-validate", e);
|
||||
invalidIndexes.add(index);
|
||||
}
|
||||
mergerUtility.addSource(tempFile); // Add source file to the merger utility
|
||||
}
|
||||
|
||||
mergerUtility.setDestinationFileName(mt.getFile().getAbsolutePath());
|
||||
|
||||
int[] pageCounts;
|
||||
try {
|
||||
mergerUtility.mergeDocuments(
|
||||
pdfDocumentFactory.getStreamCacheFunction(
|
||||
totalSize)); // Merge the documents
|
||||
pageCounts =
|
||||
mergeWithJpdfium(inputPaths, files, generateToc, mt.getFile().toPath());
|
||||
} catch (IOException e) {
|
||||
ExceptionUtils.logException("PDF merge", e);
|
||||
if (PdfErrorUtils.isCorruptedPdfError(e)) {
|
||||
@@ -348,10 +323,26 @@ public class MergeController {
|
||||
throw e;
|
||||
}
|
||||
|
||||
// Load the merged PDF document and operate on it inside try-with-resources
|
||||
try (PDDocument mergedDocument = pdfDocumentFactory.load(mt.getFile())) {
|
||||
// Remove signatures if removeCertSign is true
|
||||
if (removeCertSign) {
|
||||
boolean sigFlattenNeeded = false;
|
||||
if (removeCertSign) {
|
||||
try (PdfDocument check = PdfDocument.open(mt.getFile().toPath())) {
|
||||
sigFlattenNeeded = !check.signatures().isEmpty();
|
||||
} catch (Exception e) {
|
||||
log.debug(
|
||||
"JPDFium signature pre-check failed; falling back to PDFBox flatten:"
|
||||
+ " {}",
|
||||
e.getMessage());
|
||||
sigFlattenNeeded = true;
|
||||
}
|
||||
if (!sigFlattenNeeded) {
|
||||
log.info(
|
||||
"removeCertSign requested but merged document has no signature"
|
||||
+ " fields; skipping PDFBox flatten pass");
|
||||
}
|
||||
}
|
||||
|
||||
if (sigFlattenNeeded) {
|
||||
try (PDDocument mergedDocument = pdfDocumentFactory.load(mt.getFile())) {
|
||||
PDDocumentCatalog catalog = mergedDocument.getDocumentCatalog();
|
||||
PDAcroForm acroForm = catalog.getAcroForm();
|
||||
if (acroForm != null) {
|
||||
@@ -359,24 +350,26 @@ public class MergeController {
|
||||
acroForm.getFields().stream()
|
||||
.filter(PDSignatureField.class::isInstance)
|
||||
.toList();
|
||||
|
||||
if (!fieldsToRemove.isEmpty()) {
|
||||
acroForm.flatten(
|
||||
fieldsToRemove,
|
||||
false); // Flatten the fields, effectively removing them
|
||||
acroForm.flatten(fieldsToRemove, false);
|
||||
}
|
||||
}
|
||||
outputTempFile = new TempFile(tempFileManager, ".pdf");
|
||||
try {
|
||||
mergedDocument.save(outputTempFile.getFile());
|
||||
} catch (Exception e) {
|
||||
outputTempFile.close();
|
||||
outputTempFile = null;
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
// Add table of contents if generateToc is true
|
||||
if (generateToc && files.length > 0) {
|
||||
addTableOfContents(mergedDocument, files);
|
||||
}
|
||||
|
||||
// Save the modified document to a temporary file
|
||||
} else {
|
||||
outputTempFile = new TempFile(tempFileManager, ".pdf");
|
||||
try {
|
||||
mergedDocument.save(outputTempFile.getFile());
|
||||
Files.copy(
|
||||
mt.getFile().toPath(),
|
||||
outputTempFile.getFile().toPath(),
|
||||
java.nio.file.StandardCopyOption.REPLACE_EXISTING);
|
||||
} catch (Exception e) {
|
||||
outputTempFile.close();
|
||||
outputTempFile = null;
|
||||
@@ -395,7 +388,7 @@ public class MergeController {
|
||||
throw ex;
|
||||
} finally {
|
||||
for (File file : filesToDelete) {
|
||||
tempFileManager.deleteTempFile(file); // Delete temporary files
|
||||
tempFileManager.deleteTempFile(file);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -405,4 +398,113 @@ public class MergeController {
|
||||
|
||||
return WebResponseUtils.pdfFileToWebResponse(outputTempFile, mergedFileName);
|
||||
}
|
||||
|
||||
private int[] mergeWithJpdfium(
|
||||
List<Path> inputPaths, MultipartFile[] files, boolean generateToc, Path outputPath)
|
||||
throws IOException {
|
||||
if (inputPaths.isEmpty()) {
|
||||
try (PdfDocument empty = PdfDocument.open(new byte[0])) {
|
||||
empty.save(outputPath);
|
||||
} catch (Exception ignored) {
|
||||
Files.write(outputPath, new byte[0]);
|
||||
}
|
||||
return new int[0];
|
||||
}
|
||||
|
||||
List<PdfDocument> docs = new ArrayList<>(inputPaths.size());
|
||||
int[] pageCounts = new int[inputPaths.size()];
|
||||
int[] pageOffsets = new int[inputPaths.size()];
|
||||
List<List<Bookmark>> sourceBookmarks = new ArrayList<>(inputPaths.size());
|
||||
int runningOffset = 0;
|
||||
try {
|
||||
for (int i = 0; i < inputPaths.size(); i++) {
|
||||
Path p = inputPaths.get(i);
|
||||
PdfDocument doc = PdfDocument.open(p);
|
||||
docs.add(doc);
|
||||
pageCounts[i] = doc.pageCount();
|
||||
pageOffsets[i] = runningOffset;
|
||||
sourceBookmarks.add(doc.bookmarks());
|
||||
runningOffset += pageCounts[i];
|
||||
}
|
||||
|
||||
BookmarkTree combinedTree =
|
||||
buildCombinedBookmarkTree(files, pageOffsets, sourceBookmarks, generateToc);
|
||||
|
||||
try (PdfDocument merged = PdfMerge.merge(docs)) {
|
||||
if (combinedTree.entries().isEmpty()) {
|
||||
merged.save(outputPath);
|
||||
} else {
|
||||
PdfBookmarkEditor.setBookmarks(merged, combinedTree, outputPath);
|
||||
}
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
throw new IOException("JPDFium merge failed", e);
|
||||
} finally {
|
||||
for (PdfDocument doc : docs) {
|
||||
try {
|
||||
doc.close();
|
||||
} catch (Exception ignored) {
|
||||
}
|
||||
}
|
||||
}
|
||||
return pageCounts;
|
||||
}
|
||||
|
||||
private BookmarkTree buildCombinedBookmarkTree(
|
||||
MultipartFile[] files,
|
||||
int[] pageOffsets,
|
||||
List<List<Bookmark>> sourceBookmarks,
|
||||
boolean generateToc) {
|
||||
BookmarkTree.Builder builder = BookmarkTree.builder();
|
||||
|
||||
if (generateToc) {
|
||||
for (int i = 0; i < files.length; i++) {
|
||||
String filename = files[i].getOriginalFilename();
|
||||
String title = GeneralUtils.removeExtension(filename);
|
||||
if (title == null || title.isBlank()) {
|
||||
title = "Document " + (i + 1);
|
||||
}
|
||||
builder.add(title, pageOffsets[i]);
|
||||
}
|
||||
}
|
||||
|
||||
for (int i = 0; i < sourceBookmarks.size(); i++) {
|
||||
int offset = pageOffsets[i];
|
||||
for (Bookmark bm : sourceBookmarks.get(i)) {
|
||||
addBookmarkFlat(builder, bm, offset);
|
||||
}
|
||||
}
|
||||
|
||||
return builder.build();
|
||||
}
|
||||
|
||||
private void addBookmarkFlat(BookmarkTree.Builder builder, Bookmark root, int offset) {
|
||||
final int maxNodes = 100_000;
|
||||
java.util.Deque<Bookmark> stack = new java.util.ArrayDeque<>();
|
||||
java.util.Set<Bookmark> visited =
|
||||
java.util.Collections.newSetFromMap(new java.util.IdentityHashMap<>());
|
||||
stack.push(root);
|
||||
int processed = 0;
|
||||
while (!stack.isEmpty() && processed < maxNodes) {
|
||||
Bookmark bm = stack.pop();
|
||||
if (!visited.add(bm)) {
|
||||
continue;
|
||||
}
|
||||
processed++;
|
||||
if (bm.isInternal() && bm.title() != null) {
|
||||
builder.add(bm.title(), offset + bm.pageIndex());
|
||||
}
|
||||
if (bm.hasChildren()) {
|
||||
List<Bookmark> children = bm.children();
|
||||
for (int i = children.size() - 1; i >= 0; i--) {
|
||||
stack.push(children.get(i));
|
||||
}
|
||||
}
|
||||
}
|
||||
if (processed >= maxNodes) {
|
||||
log.warn(
|
||||
"Source bookmark traversal hit {}-node cap; remaining bookmarks dropped",
|
||||
maxNodes);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+35
-26
@@ -1,5 +1,6 @@
|
||||
package stirling.software.SPDF.controller.api.misc;
|
||||
|
||||
import java.io.File;
|
||||
import java.io.IOException;
|
||||
import java.io.OutputStream;
|
||||
import java.util.HashSet;
|
||||
@@ -30,6 +31,7 @@ import stirling.software.common.util.GeneralUtils;
|
||||
import stirling.software.common.util.TempFile;
|
||||
import stirling.software.common.util.TempFileManager;
|
||||
import stirling.software.common.util.WebResponseUtils;
|
||||
import stirling.software.jpdfium.PdfDocument;
|
||||
|
||||
@MiscApi
|
||||
@Slf4j
|
||||
@@ -51,23 +53,41 @@ public class DecompressPdfController {
|
||||
|
||||
MultipartFile file = request.getFileInput();
|
||||
|
||||
try (PDDocument document = pdfDocumentFactory.load(file)) {
|
||||
// Process all objects in document
|
||||
processAllObjects(document);
|
||||
|
||||
// Save with explicit no compression to a temp file
|
||||
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
|
||||
try {
|
||||
document.save(tempOut.getFile(), CompressParameters.NO_COMPRESSION);
|
||||
} catch (IOException e) {
|
||||
tempOut.close();
|
||||
throw e;
|
||||
// JPDFium fast pre-validate: catches corrupt PDFs cheaply before the expensive PDFBox walk.
|
||||
// JPDFium's FPDF_SaveAsCopy has no "uncompress streams" flag, so PDFBox does the actual
|
||||
// work.
|
||||
File inputTemp = null;
|
||||
try {
|
||||
inputTemp = tempFileManager.convertMultipartFileToFile(file);
|
||||
try (PdfDocument ignored = PdfDocument.open(inputTemp.toPath())) {
|
||||
// pre-validate only
|
||||
} catch (Exception e) {
|
||||
log.debug(
|
||||
"JPDFium pre-validate failed; proceeding with PDFBox: {}", e.getMessage());
|
||||
}
|
||||
|
||||
// Return the PDF as a streaming response
|
||||
return WebResponseUtils.pdfFileToWebResponse(
|
||||
tempOut,
|
||||
GeneralUtils.generateFilename(file.getOriginalFilename(), "_decompressed.pdf"));
|
||||
try (PDDocument document = pdfDocumentFactory.load(file)) {
|
||||
// Walk every object and strip stream filters so PDFBox writes raw bytes
|
||||
processAllObjects(document);
|
||||
|
||||
// Hybrid fallback: PDFium cannot save uncompressed, so use PDFBox NO_COMPRESSION
|
||||
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
|
||||
try {
|
||||
document.save(tempOut.getFile(), CompressParameters.NO_COMPRESSION);
|
||||
} catch (IOException e) {
|
||||
tempOut.close();
|
||||
throw e;
|
||||
}
|
||||
|
||||
return WebResponseUtils.pdfFileToWebResponse(
|
||||
tempOut,
|
||||
GeneralUtils.generateFilename(
|
||||
file.getOriginalFilename(), "_decompressed.pdf"));
|
||||
}
|
||||
} finally {
|
||||
if (inputTemp != null) {
|
||||
tempFileManager.deleteTempFile(inputTemp);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -75,7 +95,6 @@ public class DecompressPdfController {
|
||||
Set<COSBase> processed = new HashSet<>();
|
||||
COSDocument cosDoc = document.getDocument();
|
||||
|
||||
// Process all objects in the document
|
||||
for (COSObjectKey key : cosDoc.getXrefTable().keySet()) {
|
||||
COSObject obj = cosDoc.getObjectFromPool(key);
|
||||
processObject(obj, processed);
|
||||
@@ -83,7 +102,6 @@ public class DecompressPdfController {
|
||||
}
|
||||
|
||||
private void processObject(COSBase obj, Set<COSBase> processed) {
|
||||
// Skip null objects or already processed objects to avoid infinite recursion
|
||||
if (obj == null || processed.contains(obj)) return;
|
||||
processed.add(obj);
|
||||
|
||||
@@ -97,19 +115,16 @@ public class DecompressPdfController {
|
||||
}
|
||||
|
||||
private void processDictionary(COSDictionary dict, Set<COSBase> processed) {
|
||||
// Process all dictionary entries
|
||||
for (COSName key : dict.keySet()) {
|
||||
processObject(dict.getDictionaryObject(key), processed);
|
||||
}
|
||||
|
||||
// If this is a stream, decompress it
|
||||
if (dict instanceof COSStream stream) {
|
||||
decompressStream(stream);
|
||||
}
|
||||
}
|
||||
|
||||
private void processArray(COSArray array, Set<COSBase> processed) {
|
||||
// Process all array elements
|
||||
for (int i = 0; i < array.size(); i++) {
|
||||
processObject(array.get(i), processed);
|
||||
}
|
||||
@@ -119,33 +134,27 @@ public class DecompressPdfController {
|
||||
try {
|
||||
log.debug("Processing stream: {}", stream);
|
||||
|
||||
// Only remove filter information if it exists
|
||||
if (stream.containsKey(COSName.FILTER)
|
||||
|| stream.containsKey(COSName.DECODE_PARMS)
|
||||
|| stream.containsKey(COSName.D)) {
|
||||
|
||||
// Read the decompressed content first
|
||||
byte[] decompressedBytes;
|
||||
try (COSInputStream is = stream.createInputStream()) {
|
||||
decompressedBytes = IOUtils.toByteArray(is);
|
||||
}
|
||||
|
||||
// Now remove filter information
|
||||
stream.removeItem(COSName.FILTER);
|
||||
stream.removeItem(COSName.DECODE_PARMS);
|
||||
stream.removeItem(COSName.D);
|
||||
|
||||
// Write the raw content back
|
||||
try (OutputStream out = stream.createRawOutputStream()) {
|
||||
out.write(decompressedBytes);
|
||||
}
|
||||
|
||||
// Set the Length to reflect the new stream size
|
||||
stream.setInt(COSName.LENGTH, decompressedBytes.length);
|
||||
}
|
||||
} catch (IOException e) {
|
||||
ExceptionUtils.logException("stream decompression", e);
|
||||
// Continue processing other streams even if this one fails
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+258
@@ -0,0 +1,258 @@
|
||||
package stirling.software.SPDF.controller.api.misc;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.OutputStream;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.cos.COSArray;
|
||||
import org.apache.pdfbox.cos.COSBase;
|
||||
import org.apache.pdfbox.cos.COSDictionary;
|
||||
import org.apache.pdfbox.cos.COSDocument;
|
||||
import org.apache.pdfbox.cos.COSInputStream;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.cos.COSObject;
|
||||
import org.apache.pdfbox.cos.COSObjectKey;
|
||||
import org.apache.pdfbox.cos.COSStream;
|
||||
import org.apache.pdfbox.io.IOUtils;
|
||||
import org.apache.pdfbox.pdfwriter.compress.CompressParameters;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.PDPageContentStream;
|
||||
import org.apache.pdfbox.pdmodel.common.PDRectangle;
|
||||
import org.apache.pdfbox.pdmodel.font.PDType1Font;
|
||||
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.condition.EnabledIfSystemProperty;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import stirling.software.jpdfium.PdfDocument;
|
||||
|
||||
/**
|
||||
* Bench gated on -Djpdfium.bench=true. Measures whether the JPDFium pre-validate hop in
|
||||
* DecompressPdfController earns its keep over a plain PDFBox-only run.
|
||||
*
|
||||
* <p>VERDICT: DROP. Measured on 500-page / 110 KB synthetic PDF, JDK 25, Windows x64.
|
||||
*
|
||||
* <ul>
|
||||
* <li>Happy path: JPDFium open+close costs ~1.1 ms; peak heap is unchanged (PDFBox still loads
|
||||
* the full document afterwards, so the pre-validate cannot lower memory).
|
||||
* <li>Corrupt input: JPDFium rejects in ~4.6 ms vs PDFBox in ~2.1 ms - JPDFium is 2x SLOWER at
|
||||
* reject, so the pre-validate adds latency in the very case it was meant to short-circuit.
|
||||
* <li>Hidden I/O cost not measured here: the controller calls convertMultipartFileToFile before
|
||||
* PdfDocument.open, which writes the full upload to disk and reads it back; on big PDFs that
|
||||
* extra round-trip dwarfs the 1 ms native call.
|
||||
* </ul>
|
||||
*
|
||||
* <p>Recommendation: drop the pre-validate hop. PDFBox already validates on load, and PDFium 1.0.0
|
||||
* has no FPDF_SaveAsCopy flag that strips /Filter or rewrites ObjStm, so it cannot replace the
|
||||
* PDFBox decompress walk either. Re-evaluate only if jpdfium ships an uncompressed-save API.
|
||||
*/
|
||||
@EnabledIfSystemProperty(named = "jpdfium.bench", matches = "true")
|
||||
class DecompressPdfBench {
|
||||
|
||||
private static final int WARMUPS = 5;
|
||||
private static final int RUNS = 15;
|
||||
private static final int PAGES = 500;
|
||||
|
||||
@Test
|
||||
void compareJpdfiumPrevalidateVsPdfboxOnly(@TempDir Path tmp) throws IOException {
|
||||
Path pdf = tmp.resolve("bench.pdf");
|
||||
buildLargePdf(pdf, PAGES);
|
||||
long sizeBytes = Files.size(pdf);
|
||||
byte[] bytes = Files.readAllBytes(pdf);
|
||||
|
||||
for (int i = 0; i < WARMUPS; i++) {
|
||||
runPdfboxOnly(bytes);
|
||||
runJpdfiumThenPdfbox(pdf, bytes);
|
||||
runJpdfiumOpenOnly(pdf);
|
||||
}
|
||||
|
||||
Stats pdfboxOnly = bench(() -> runPdfboxOnly(bytes));
|
||||
Stats hybrid = bench(() -> runJpdfiumThenPdfbox(pdf, bytes));
|
||||
Stats jpdfiumOnly = bench(() -> runJpdfiumOpenOnly(pdf));
|
||||
|
||||
// Corrupt-input short-circuit scenario
|
||||
byte[] junk = new byte[(int) sizeBytes];
|
||||
for (int i = 0; i < junk.length; i++) junk[i] = (byte) (i & 0x7F);
|
||||
Path junkPath = tmp.resolve("junk.pdf");
|
||||
Files.write(junkPath, junk);
|
||||
Stats corruptJpdfium = bench(() -> tryJpdfiumOnly(junkPath));
|
||||
Stats corruptPdfbox = bench(() -> tryPdfboxOnly(junk));
|
||||
|
||||
System.out.println("=== DecompressPdfBench ===");
|
||||
System.out.printf("input size: %d KB (%d pages)%n", sizeBytes / 1024, PAGES);
|
||||
System.out.printf("runs: %d (after %d warmups)%n", RUNS, WARMUPS);
|
||||
System.out.println("--- happy path ---");
|
||||
System.out.printf(
|
||||
"PDFBox-only wall=%.1fms peakHeap=%.1fMB%n",
|
||||
pdfboxOnly.avgMillis, pdfboxOnly.avgPeakHeapMb);
|
||||
System.out.printf(
|
||||
"JPDFium+PDFBox wall=%.1fms peakHeap=%.1fMB%n",
|
||||
hybrid.avgMillis, hybrid.avgPeakHeapMb);
|
||||
System.out.printf(
|
||||
"JPDFium open only wall=%.1fms peakHeap=%.1fMB%n",
|
||||
jpdfiumOnly.avgMillis, jpdfiumOnly.avgPeakHeapMb);
|
||||
double wallDelta = hybrid.avgMillis - pdfboxOnly.avgMillis;
|
||||
double heapDelta = hybrid.avgPeakHeapMb - pdfboxOnly.avgPeakHeapMb;
|
||||
System.out.printf(
|
||||
"delta (hybrid - pdfboxOnly): wall=%+.1fms (%+.1f%%) peakHeap=%+.1fMB%n",
|
||||
wallDelta, 100.0 * wallDelta / pdfboxOnly.avgMillis, heapDelta);
|
||||
System.out.println("--- corrupt input ---");
|
||||
System.out.printf(
|
||||
"JPDFium reject wall=%.2fms peakHeap=%.2fMB%n",
|
||||
corruptJpdfium.avgMillis, corruptJpdfium.avgPeakHeapMb);
|
||||
System.out.printf(
|
||||
"PDFBox reject wall=%.2fms peakHeap=%.2fMB%n",
|
||||
corruptPdfbox.avgMillis, corruptPdfbox.avgPeakHeapMb);
|
||||
}
|
||||
|
||||
private Stats bench(IoRunnable r) throws IOException {
|
||||
double sumMs = 0;
|
||||
double sumHeapMb = 0;
|
||||
for (int i = 0; i < RUNS; i++) {
|
||||
System.gc();
|
||||
long before = usedHeap();
|
||||
long t0 = System.nanoTime();
|
||||
r.run();
|
||||
long t1 = System.nanoTime();
|
||||
long after = usedHeap();
|
||||
sumMs += (t1 - t0) / 1_000_000.0;
|
||||
sumHeapMb += Math.max(0, after - before) / (1024.0 * 1024.0);
|
||||
}
|
||||
Stats s = new Stats();
|
||||
s.avgMillis = sumMs / RUNS;
|
||||
s.avgPeakHeapMb = sumHeapMb / RUNS;
|
||||
return s;
|
||||
}
|
||||
|
||||
private static long usedHeap() {
|
||||
Runtime rt = Runtime.getRuntime();
|
||||
return rt.totalMemory() - rt.freeMemory();
|
||||
}
|
||||
|
||||
private void runPdfboxOnly(byte[] bytes) throws IOException {
|
||||
try (PDDocument doc = Loader.loadPDF(bytes)) {
|
||||
processAllObjects(doc);
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
doc.save(out, CompressParameters.NO_COMPRESSION);
|
||||
}
|
||||
}
|
||||
|
||||
private void runJpdfiumThenPdfbox(Path inputPath, byte[] bytes) throws IOException {
|
||||
try (PdfDocument ignored = PdfDocument.open(inputPath)) {
|
||||
// pre-validate only
|
||||
} catch (Exception ignored) {
|
||||
// mirror controller fallback
|
||||
}
|
||||
try (PDDocument doc = Loader.loadPDF(bytes)) {
|
||||
processAllObjects(doc);
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
doc.save(out, CompressParameters.NO_COMPRESSION);
|
||||
}
|
||||
}
|
||||
|
||||
private void runJpdfiumOpenOnly(Path inputPath) throws IOException {
|
||||
try (PdfDocument ignored = PdfDocument.open(inputPath)) {
|
||||
// open and close only
|
||||
} catch (Exception ignored) {
|
||||
}
|
||||
}
|
||||
|
||||
private void tryJpdfiumOnly(Path inputPath) throws IOException {
|
||||
try (PdfDocument ignored = PdfDocument.open(inputPath)) {
|
||||
} catch (Exception ignored) {
|
||||
}
|
||||
}
|
||||
|
||||
private void tryPdfboxOnly(byte[] bytes) throws IOException {
|
||||
try (PDDocument doc = Loader.loadPDF(bytes)) {
|
||||
} catch (Exception ignored) {
|
||||
}
|
||||
}
|
||||
|
||||
private void processAllObjects(PDDocument document) {
|
||||
Set<COSBase> processed = new HashSet<>();
|
||||
COSDocument cosDoc = document.getDocument();
|
||||
for (COSObjectKey key : cosDoc.getXrefTable().keySet()) {
|
||||
COSObject obj = cosDoc.getObjectFromPool(key);
|
||||
processObject(obj, processed);
|
||||
}
|
||||
}
|
||||
|
||||
private void processObject(COSBase obj, Set<COSBase> processed) {
|
||||
if (obj == null || processed.contains(obj)) return;
|
||||
processed.add(obj);
|
||||
if (obj instanceof COSObject cosObj) {
|
||||
processObject(cosObj.getObject(), processed);
|
||||
} else if (obj instanceof COSDictionary dict) {
|
||||
for (COSName key : dict.keySet()) {
|
||||
processObject(dict.getDictionaryObject(key), processed);
|
||||
}
|
||||
if (dict instanceof COSStream stream) {
|
||||
decompressStream(stream);
|
||||
}
|
||||
} else if (obj instanceof COSArray array) {
|
||||
for (int i = 0; i < array.size(); i++) {
|
||||
processObject(array.get(i), processed);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private void decompressStream(COSStream stream) {
|
||||
try {
|
||||
if (stream.containsKey(COSName.FILTER)
|
||||
|| stream.containsKey(COSName.DECODE_PARMS)
|
||||
|| stream.containsKey(COSName.D)) {
|
||||
byte[] bytes;
|
||||
try (COSInputStream is = stream.createInputStream()) {
|
||||
bytes = IOUtils.toByteArray(is);
|
||||
}
|
||||
stream.removeItem(COSName.FILTER);
|
||||
stream.removeItem(COSName.DECODE_PARMS);
|
||||
stream.removeItem(COSName.D);
|
||||
try (OutputStream out = stream.createRawOutputStream()) {
|
||||
out.write(bytes);
|
||||
}
|
||||
stream.setInt(COSName.LENGTH, bytes.length);
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// mirror controller swallow
|
||||
}
|
||||
}
|
||||
|
||||
private static void buildLargePdf(Path path, int pages) throws IOException {
|
||||
try (PDDocument doc = new PDDocument()) {
|
||||
for (int i = 0; i < pages; i++) {
|
||||
PDPage page = new PDPage(PDRectangle.LETTER);
|
||||
doc.addPage(page);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(doc, page)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 10);
|
||||
cs.newLineAtOffset(50, 750);
|
||||
StringBuilder line = new StringBuilder("Page ").append(i).append(": ");
|
||||
for (int j = 0; j < 60; j++) {
|
||||
line.append("Lorem ipsum dolor sit amet consectetur ");
|
||||
}
|
||||
cs.showText(line.toString());
|
||||
cs.endText();
|
||||
}
|
||||
}
|
||||
doc.save(path.toFile());
|
||||
}
|
||||
}
|
||||
|
||||
private static class Stats {
|
||||
double avgMillis;
|
||||
double avgPeakHeapMb;
|
||||
}
|
||||
|
||||
@FunctionalInterface
|
||||
private interface IoRunnable {
|
||||
void run() throws IOException;
|
||||
}
|
||||
}
|
||||
+38
@@ -1,6 +1,7 @@
|
||||
package stirling.software.SPDF.controller.api.misc;
|
||||
|
||||
import static org.assertj.core.api.Assertions.*;
|
||||
import static org.mockito.ArgumentMatchers.any;
|
||||
import static org.mockito.ArgumentMatchers.anyString;
|
||||
import static org.mockito.Mockito.*;
|
||||
|
||||
@@ -68,6 +69,16 @@ class DecompressPdfControllerTest {
|
||||
lenient().when(tf.getPath()).thenReturn(f.toPath());
|
||||
return tf;
|
||||
});
|
||||
// Stub multipart-to-file conversion for the JPDFium pre-validate step
|
||||
lenient()
|
||||
.when(tempFileManager.convertMultipartFileToFile(any()))
|
||||
.thenAnswer(
|
||||
inv -> {
|
||||
org.springframework.web.multipart.MultipartFile mf = inv.getArgument(0);
|
||||
File f = Files.createTempFile("input", ".pdf").toFile();
|
||||
java.nio.file.Files.write(f.toPath(), mf.getBytes());
|
||||
return f;
|
||||
});
|
||||
}
|
||||
|
||||
private MockMultipartFile createRealPdf(String content) throws IOException {
|
||||
@@ -209,6 +220,33 @@ class DecompressPdfControllerTest {
|
||||
assertThat(drainBody(response).length).isGreaterThan(0);
|
||||
}
|
||||
|
||||
@Test
|
||||
void decompressPdf_oracleNoFilterEntries() throws IOException {
|
||||
// PDFBox oracle: load the decompressed output and verify every COSStream has no /Filter
|
||||
MockMultipartFile file = createRealPdf("Oracle decompression check");
|
||||
PDFFile request = new PDFFile();
|
||||
request.setFileInput(file);
|
||||
|
||||
PDDocument doc = Loader.loadPDF(file.getBytes());
|
||||
when(pdfDocumentFactory.load(file)).thenReturn(doc);
|
||||
|
||||
ResponseEntity<Resource> response = controller.decompressPdf(request);
|
||||
byte[] body = drainBody(response);
|
||||
|
||||
try (PDDocument result = Loader.loadPDF(body)) {
|
||||
org.apache.pdfbox.cos.COSDocument cosDoc = result.getDocument();
|
||||
for (org.apache.pdfbox.cos.COSObjectKey key : cosDoc.getXrefTable().keySet()) {
|
||||
org.apache.pdfbox.cos.COSObject obj = cosDoc.getObjectFromPool(key);
|
||||
org.apache.pdfbox.cos.COSBase base = obj.getObject();
|
||||
if (base instanceof org.apache.pdfbox.cos.COSStream stream) {
|
||||
assertThat(stream.containsKey(org.apache.pdfbox.cos.COSName.FILTER))
|
||||
.as("stream %s should have no /Filter", key)
|
||||
.isFalse();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void decompressPdf_returnsOkContentType() throws IOException {
|
||||
MockMultipartFile file = createRealPdf("test");
|
||||
|
||||
+7
-6
@@ -31,12 +31,12 @@ ext {
|
||||
googleJavaFormatVersion = "1.28.0"
|
||||
logback = "1.5.32"
|
||||
// junit-platform-launcher version managed by Spring Boot BOM
|
||||
modernJavaVersion = 21
|
||||
modernJavaVersion = 25
|
||||
}
|
||||
|
||||
java {
|
||||
sourceCompatibility = JavaVersion.VERSION_21
|
||||
targetCompatibility = JavaVersion.VERSION_21
|
||||
sourceCompatibility = JavaVersion.VERSION_25
|
||||
targetCompatibility = JavaVersion.VERSION_25
|
||||
toolchain {
|
||||
languageVersion = JavaLanguageVersion.of(project.findProperty('javaVersion')?.toString() ?: '25')
|
||||
}
|
||||
@@ -158,8 +158,8 @@ subprojects {
|
||||
apply plugin: 'jacoco'
|
||||
|
||||
java {
|
||||
sourceCompatibility = JavaVersion.VERSION_21
|
||||
targetCompatibility = JavaVersion.VERSION_21
|
||||
sourceCompatibility = JavaVersion.VERSION_25
|
||||
targetCompatibility = JavaVersion.VERSION_25
|
||||
toolchain {
|
||||
languageVersion = JavaLanguageVersion.of(25)
|
||||
}
|
||||
@@ -444,7 +444,8 @@ subprojects {
|
||||
"-XX:G1HeapRegionSize=4m",
|
||||
"-XX:+ExplicitGCInvokesConcurrent",
|
||||
"-XX:+UseStringDeduplication",
|
||||
"-XX:+UseCompactObjectHeaders"
|
||||
"-XX:+UseCompactObjectHeaders",
|
||||
"--enable-native-access=ALL-UNNAMED"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,111 @@
|
||||
#!/usr/bin/env bash
|
||||
# Sign every .dylib inside the bootJar's JPDFium native jars.
|
||||
# Requires APPLE_SIGNING_IDENTITY set to a Developer ID identity in the keychain.
|
||||
#
|
||||
# Usage: sign-jpdfium-dylibs-in-bootjar.sh [path/to/stirling-pdf-*.jar]
|
||||
|
||||
set -u
|
||||
|
||||
echo "sign-jpdfium-dylibs-in-bootjar.sh: start ($(uname -s) $(uname -m))"
|
||||
|
||||
case "$(uname -s)" in
|
||||
Darwin*) ;;
|
||||
*) echo "Not macOS, skipping"; exit 0;;
|
||||
esac
|
||||
|
||||
if [ -z "${APPLE_SIGNING_IDENTITY:-}" ]; then
|
||||
echo "APPLE_SIGNING_IDENTITY not set; skipping"
|
||||
exit 0
|
||||
fi
|
||||
if ! command -v codesign >/dev/null 2>&1; then
|
||||
echo "codesign not on PATH; skipping"
|
||||
exit 0
|
||||
fi
|
||||
if ! command -v jar >/dev/null 2>&1; then
|
||||
echo "jar not on PATH (need a JDK setup-action earlier); skipping"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
BOOTJARS=()
|
||||
if [ -n "${1:-}" ]; then
|
||||
BOOTJARS+=("$1")
|
||||
else
|
||||
for cand in app/core/build/libs/stirling-pdf-*.jar \
|
||||
frontend/src-tauri/libs/stirling-pdf-*.jar; do
|
||||
[ -f "$cand" ] || continue
|
||||
BOOTJARS+=("$cand")
|
||||
done
|
||||
fi
|
||||
if [ "${#BOOTJARS[@]:-0}" = 0 ]; then
|
||||
echo "bootJar not found (expected app/core/build/libs/stirling-pdf-*.jar" \
|
||||
"or frontend/src-tauri/libs/stirling-pdf-*.jar)"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
for BOOTJAR in "${BOOTJARS[@]}"; do
|
||||
BOOTJAR=$(cd "$(dirname "$BOOTJAR")" && pwd)/$(basename "$BOOTJAR")
|
||||
echo ""
|
||||
echo "=== Target bootJar: $BOOTJAR ($(du -h "$BOOTJAR" | cut -f1)) ==="
|
||||
|
||||
WORK=$(mktemp -d)
|
||||
# shellcheck disable=SC2064
|
||||
trap "rm -rf '$WORK'" EXIT
|
||||
|
||||
NATIVE_JAR_PATHS=()
|
||||
while IFS= read -r line; do
|
||||
[ -n "$line" ] || continue
|
||||
NATIVE_JAR_PATHS+=("$line")
|
||||
done < <(jar tf "$BOOTJAR" \
|
||||
| grep -E '^BOOT-INF/lib/jpdfium-natives-darwin-(x64|arm64)-.*\.jar$' || true)
|
||||
|
||||
if [ "${#NATIVE_JAR_PATHS[@]:-0}" = 0 ]; then
|
||||
echo " No JPDFium darwin natives in this bootJar; skipping"
|
||||
rm -rf "$WORK"
|
||||
continue
|
||||
fi
|
||||
|
||||
( cd "$WORK" && jar xf "$BOOTJAR" ${NATIVE_JAR_PATHS[@]+"${NATIVE_JAR_PATHS[@]}"} ) \
|
||||
|| { echo "jar xf failed to extract natives jars" >&2; exit 1; }
|
||||
|
||||
ANY_SIGNED=0
|
||||
for nat_jar in "$WORK/BOOT-INF/lib"/jpdfium-natives-darwin-*.jar; do
|
||||
[ -f "$nat_jar" ] || continue
|
||||
base=$(basename "$nat_jar")
|
||||
echo " Processing $base"
|
||||
|
||||
exp_dir="$WORK/${base%.jar}.expanded"
|
||||
mkdir -p "$exp_dir"
|
||||
( cd "$exp_dir" && jar xf "$nat_jar" )
|
||||
|
||||
signed=0
|
||||
while IFS= read -r dylib; do
|
||||
codesign --force --sign "$APPLE_SIGNING_IDENTITY" \
|
||||
--options runtime --timestamp "$dylib" 2>&1 | sed 's/^/ /'
|
||||
signed=$((signed + 1))
|
||||
done < <(find "$exp_dir" -name '*.dylib' -type f)
|
||||
|
||||
if [ "$signed" = 0 ]; then
|
||||
echo " (no .dylibs found)"
|
||||
continue
|
||||
fi
|
||||
echo " signed $signed dylib(s)"
|
||||
|
||||
rm -f "$nat_jar"
|
||||
( cd "$exp_dir" && jar cfM0 "$nat_jar" . )
|
||||
ANY_SIGNED=1
|
||||
done
|
||||
|
||||
if [ "$ANY_SIGNED" = 0 ]; then
|
||||
echo " No .dylibs signed; skipping update"
|
||||
rm -rf "$WORK"
|
||||
continue
|
||||
fi
|
||||
|
||||
( cd "$WORK" && jar uf "$BOOTJAR" \
|
||||
BOOT-INF/lib/jpdfium-natives-darwin-x64-*.jar \
|
||||
BOOT-INF/lib/jpdfium-natives-darwin-arm64-*.jar ) \
|
||||
2>/dev/null || { echo "jar uf failed" >&2; exit 1; }
|
||||
|
||||
echo " Updated: $BOOTJAR ($(du -h "$BOOTJAR" | cut -f1))"
|
||||
rm -rf "$WORK"
|
||||
done
|
||||
+4
-2
@@ -442,8 +442,10 @@ compare_file_lists() {
|
||||
echo "New files created during test:"
|
||||
cat "${diff_file}.added" | sed 's/^> //'
|
||||
|
||||
# Check for tmp files
|
||||
grep -i "tmp\|temp" "${diff_file}.added" > "${diff_file}.tmp" || true
|
||||
# Exclude JPDFium native cache (deleteOnExit-registered, not a leak).
|
||||
grep -i "tmp\|temp" "${diff_file}.added" \
|
||||
| grep -v '/jpdfium-[0-9]\+/' \
|
||||
> "${diff_file}.tmp" || true
|
||||
if [ -s "${diff_file}.tmp" ]; then
|
||||
echo "WARNING: Temporary files detected:"
|
||||
cat "${diff_file}.tmp"
|
||||
|
||||
Reference in New Issue
Block a user