mirror of
https://github.com/Stirling-Tools/Stirling-PDF.git
synced 2026-09-02 21:03:34 +03:00
Merge branch 'main' into fix_css_20260729
This commit is contained in:
@@ -1,6 +1,6 @@
|
||||
# Maintainer: Stirling PDF Inc <contact@stirlingpdf.com>
|
||||
pkgname=stirling-pdf-desktop
|
||||
pkgver=2.14.2
|
||||
pkgver=2.14.3
|
||||
pkgrel=1
|
||||
pkgdesc="Locally hosted, web-based PDF manipulation tool (Tauri desktop app, official Stirling PDF Inc build)"
|
||||
arch=('x86_64')
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Maintainer: Stirling PDF Inc <contact@stirlingpdf.com>
|
||||
pkgname=stirling-pdf-server-bin
|
||||
pkgver=2.14.2
|
||||
pkgver=2.14.3
|
||||
pkgrel=1
|
||||
pkgdesc="Locally hosted, web-based PDF manipulation tool (server JAR, prebuilt)"
|
||||
arch=('any')
|
||||
|
||||
@@ -4,10 +4,12 @@
|
||||
# instead of matching only the project filter.
|
||||
ci: &ci
|
||||
- .github/workflows/build.yml
|
||||
- .github/workflows/gradle-cache-prime.yml
|
||||
- .github/config/.files.yaml
|
||||
|
||||
build: &build
|
||||
- *ci
|
||||
- buildSrc/**
|
||||
- build.gradle
|
||||
- gradle/spotless.gradle
|
||||
- app/(common|core|proprietary|saas)/build.gradle
|
||||
@@ -15,6 +17,22 @@ build: &build
|
||||
- .taskfiles/backend.yml
|
||||
- .github/workflows/check-licence.yml
|
||||
|
||||
# Backend build inputs. This is intentionally broader than `build`: Java and
|
||||
# backend resource changes must exercise the backend matrix even when Gradle
|
||||
# build scripts themselves are unchanged.
|
||||
backend: &backend
|
||||
- *ci
|
||||
- *build
|
||||
- gradle/**
|
||||
- gradle.properties
|
||||
- gradlew
|
||||
- gradlew.bat
|
||||
- settings.gradle
|
||||
- app/(common|core|proprietary|saas)/src/(main|test)/java/**
|
||||
- "app/(common|core|proprietary|saas)/src/(main|test)/resources/**/!(messages_*.properties|*.md)*"
|
||||
- scripts/db-migration/**
|
||||
- .github/workflows/backend-build.yml
|
||||
|
||||
openapi: &openapi
|
||||
- *ci
|
||||
- *build
|
||||
|
||||
@@ -65,6 +65,7 @@ updates:
|
||||
directories:
|
||||
- /devTools
|
||||
- /frontend
|
||||
- /testing/compose/mcp-client-check
|
||||
schedule:
|
||||
interval: "weekly"
|
||||
cooldown:
|
||||
@@ -93,6 +94,13 @@ updates:
|
||||
- "react-dom"
|
||||
- "@types/react"
|
||||
- "@types/react-dom"
|
||||
tanstack:
|
||||
patterns:
|
||||
- "@tanstack/*"
|
||||
typescript:
|
||||
patterns:
|
||||
- "typescript"
|
||||
- "@typescript/*"
|
||||
vite:
|
||||
patterns:
|
||||
- "vite"
|
||||
@@ -171,14 +179,6 @@ updates:
|
||||
- "tokio"
|
||||
- "tokio-*"
|
||||
|
||||
- package-ecosystem: pip
|
||||
directory: /testing/cucumber
|
||||
schedule:
|
||||
interval: "weekly"
|
||||
cooldown:
|
||||
default-days: 7
|
||||
rebase-strategy: "auto"
|
||||
|
||||
- package-ecosystem: "uv"
|
||||
directory: "/engine"
|
||||
schedule:
|
||||
|
||||
@@ -462,7 +462,10 @@ jobs:
|
||||
});
|
||||
|
||||
cleanup-v2-deployment:
|
||||
environment: pr-preview
|
||||
# Tearing a preview down is not a deployment - no deployment object.
|
||||
environment:
|
||||
name: pr-preview
|
||||
deployment: false
|
||||
if: github.event.action == 'closed'
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
@@ -557,5 +560,5 @@ jobs:
|
||||
- name: Cleanup temporary files
|
||||
if: always()
|
||||
run: |
|
||||
rm -f ../private.key
|
||||
rm -f ../private.key docker-compose.yml storybook.tgz
|
||||
continue-on-error: true
|
||||
|
||||
@@ -9,7 +9,10 @@ permissions:
|
||||
|
||||
jobs:
|
||||
cleanup:
|
||||
environment: pr-preview
|
||||
# Tearing a preview down is not a deployment - no deployment object.
|
||||
environment:
|
||||
name: pr-preview
|
||||
deployment: false
|
||||
if: github.event.action == 'closed'
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
|
||||
@@ -28,7 +28,7 @@ jobs:
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -20,7 +20,9 @@ permissions:
|
||||
|
||||
jobs:
|
||||
build:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
@@ -195,7 +197,7 @@ jobs:
|
||||
|
||||
- name: Install uv
|
||||
if: always() && matrix.flavor == 'saas'
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -42,7 +42,9 @@ jobs:
|
||||
uses: ./.github/workflows/_runner-pick.yml
|
||||
|
||||
playwright-e2e-enterprise:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
needs: pick
|
||||
# Skip on fork PRs / untrusted authors: they have no PREMIUM_KEY_ENTERPRISE,
|
||||
# so the suite can't boot premium and would fail. See the header comment.
|
||||
@@ -322,10 +324,18 @@ jobs:
|
||||
path: frontend/playwright-report/
|
||||
retention-days: 7
|
||||
|
||||
- name: Cleanup temporary files
|
||||
if: always()
|
||||
run: |
|
||||
rm -f /tmp/helpers.sh /tmp/backend.log /tmp/backend.pid
|
||||
continue-on-error: true
|
||||
|
||||
# Multi-node regression: builds + seeds the clustered stack (testing/compose/docker-compose-multinode.yml)
|
||||
# and runs behave features/multinode. Licence-gated, so it runs after the Playwright job (not in parallel).
|
||||
multinode-e2e:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
needs: [pick, playwright-e2e-enterprise]
|
||||
# Nightly cron + manual dispatch only (heavy build), fork-gated for the licence secret.
|
||||
if: >-
|
||||
@@ -347,7 +357,7 @@ jobs:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
+15
-54
@@ -37,6 +37,7 @@ jobs:
|
||||
timeout-minutes: 3
|
||||
outputs:
|
||||
build: ${{ steps.changes.outputs.build }}
|
||||
backend: ${{ steps.changes.outputs.backend }}
|
||||
project: ${{ steps.changes.outputs.project }}
|
||||
openapi: ${{ steps.changes.outputs.openapi }}
|
||||
frontend: ${{ steps.changes.outputs.frontend }}
|
||||
@@ -61,61 +62,12 @@ jobs:
|
||||
filters: .github/config/.files.yaml
|
||||
|
||||
gradle-cache-prime:
|
||||
environment: ci-unsigned
|
||||
name: Prime shared Gradle cache
|
||||
needs: [files-changed]
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- name: Harden the runner (Audit all outbound calls)
|
||||
uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1
|
||||
with:
|
||||
egress-policy: audit
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Calculate Gradle cache key
|
||||
id: gradle-cache-key
|
||||
shell: bash
|
||||
run: |
|
||||
echo "key=gradle-v1-${{ runner.os }}-${{ runner.arch }}-jdk-25-${{ hashFiles('gradle/wrapper/gradle-wrapper.properties', 'gradle/libs.versions.toml', 'buildSrc/**', 'settings.gradle', 'build.gradle', 'app/**/build.gradle', 'gradle/**/*.gradle') }}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Cache Gradle (lookup-only)
|
||||
id: cache-gradle-restore
|
||||
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
path: |
|
||||
~/.gradle/caches
|
||||
~/.gradle/wrapper
|
||||
key: ${{ steps.gradle-cache-key.outputs.key }}
|
||||
lookup-only: true
|
||||
|
||||
- name: Set up JDK 25
|
||||
if: steps.cache-gradle-restore.outputs.cache-hit != 'true'
|
||||
uses: actions/setup-java@b6effb05e454b25005698d916606bdc6ffcbf961 # v5.7.0
|
||||
with:
|
||||
java-version: "25"
|
||||
distribution: "temurin"
|
||||
|
||||
- name: Resolve backend dependencies
|
||||
if: steps.cache-gradle-restore.outputs.cache-hit != 'true'
|
||||
run: ./gradlew :stirling-pdf:classes --no-daemon
|
||||
env:
|
||||
STIRLING_FLAVOR: saas
|
||||
MAVEN_USER: ${{ secrets.MAVEN_USER }}
|
||||
MAVEN_PASSWORD: ${{ secrets.MAVEN_PASSWORD }}
|
||||
MAVEN_PUBLIC_URL: ${{ secrets.MAVEN_PUBLIC_URL }}
|
||||
|
||||
- name: Save cache Gradle User Home
|
||||
if: steps.cache-gradle-restore.outputs.cache-hit != 'true'
|
||||
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
path: |
|
||||
~/.gradle/caches
|
||||
~/.gradle/wrapper
|
||||
key: ${{ steps.gradle-cache-key.outputs.key }}
|
||||
uses: ./.github/workflows/gradle-cache-prime.yml
|
||||
secrets: inherit
|
||||
|
||||
build:
|
||||
if: needs.files-changed.outputs.backend == 'true'
|
||||
needs: [files-changed, gradle-cache-prime]
|
||||
permissions:
|
||||
actions: read
|
||||
@@ -194,7 +146,7 @@ jobs:
|
||||
|
||||
check-licence:
|
||||
if: needs.files-changed.outputs.build == 'true'
|
||||
needs: [files-changed, build, gradle-cache-prime]
|
||||
needs: [files-changed, gradle-cache-prime]
|
||||
permissions:
|
||||
contents: read
|
||||
uses: ./.github/workflows/check-licence.yml
|
||||
@@ -213,7 +165,14 @@ jobs:
|
||||
docker-base-changed: ${{ needs.files-changed.outputs.docker-base }}
|
||||
|
||||
test-build-docker-images:
|
||||
if: github.event_name == 'pull_request' && needs.files-changed.outputs.project == 'true'
|
||||
if: |
|
||||
always() &&
|
||||
github.event_name == 'pull_request' &&
|
||||
needs.files-changed.outputs.project == 'true' &&
|
||||
contains(fromJSON('["success", "skipped"]'), needs.gradle-cache-prime.result) &&
|
||||
contains(fromJSON('["success", "skipped"]'), needs.build.result) &&
|
||||
contains(fromJSON('["success", "skipped"]'), needs.check-generateOpenApiDocs.result) &&
|
||||
contains(fromJSON('["success", "skipped"]'), needs.check-licence.result)
|
||||
needs:
|
||||
[
|
||||
files-changed,
|
||||
@@ -319,6 +278,7 @@ jobs:
|
||||
if: always()
|
||||
needs:
|
||||
- files-changed
|
||||
- gradle-cache-prime
|
||||
- build
|
||||
- db-migration-test
|
||||
- check-generateOpenApiDocs
|
||||
@@ -346,6 +306,7 @@ jobs:
|
||||
env:
|
||||
RESULTS: |
|
||||
files-changed=${{ needs.files-changed.result }}
|
||||
gradle-cache-prime=${{ needs.gradle-cache-prime.result }}
|
||||
build=${{ needs.build.result }}
|
||||
db-migration-test=${{ needs.db-migration-test.result }}
|
||||
check-generateOpenApiDocs=${{ needs.check-generateOpenApiDocs.result }}
|
||||
|
||||
@@ -36,7 +36,7 @@ jobs:
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -10,7 +10,9 @@ permissions:
|
||||
|
||||
jobs:
|
||||
check-licence:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Harden Runner
|
||||
|
||||
@@ -11,7 +11,9 @@ permissions:
|
||||
|
||||
jobs:
|
||||
check-generate-openapi-docs:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Harden Runner
|
||||
|
||||
@@ -195,7 +195,7 @@ jobs:
|
||||
core.exportVariable("REFERENCE_FILE", referenceFilePath);
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -55,7 +55,7 @@ jobs:
|
||||
distribution: "temurin"
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -13,7 +13,9 @@ permissions:
|
||||
|
||||
jobs:
|
||||
migration-test:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
@@ -79,3 +81,8 @@ jobs:
|
||||
path: /tmp/stirling-migration-failed-*/app.log
|
||||
retention-days: 7
|
||||
if-no-files-found: warn
|
||||
|
||||
- name: Cleanup temporary files
|
||||
if: always()
|
||||
run: rm -rf /tmp/stirling-migration-failed-*
|
||||
continue-on-error: true
|
||||
|
||||
@@ -17,7 +17,9 @@ permissions:
|
||||
|
||||
jobs:
|
||||
docker-compose-tests:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
actions: write
|
||||
@@ -64,11 +66,11 @@ jobs:
|
||||
|
||||
- name: Install Docker Compose
|
||||
run: |
|
||||
sudo curl -SL "https://github.com/docker/compose/releases/download/v2.39.4/docker-compose-$(uname -s)-$(uname -m)" -o /usr/local/bin/docker-compose
|
||||
sudo curl -SL "https://github.com/docker/compose/releases/download/v5.4.0/docker-compose-$(uname -s)-$(uname -m)" -o /usr/local/bin/docker-compose
|
||||
sudo chmod +x /usr/local/bin/docker-compose
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -11,7 +11,9 @@ permissions:
|
||||
|
||||
jobs:
|
||||
playwright-e2e-live:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
@@ -104,7 +106,7 @@ jobs:
|
||||
fi
|
||||
- name: Install uv
|
||||
if: always()
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -14,6 +14,11 @@ jobs:
|
||||
playwright-e2e:
|
||||
name: playwright-e2e (${{ matrix.browser }})
|
||||
runs-on: ubuntu-latest
|
||||
# The image already contains the Playwright browsers and all Linux
|
||||
# dependencies. This keeps the matrix for per-browser reporting while
|
||||
# avoiding three concurrent `playwright install --with-deps` runs.
|
||||
container:
|
||||
image: mcr.microsoft.com/playwright:v1.58.2-noble@sha256:6446946a1d9fd62d9ae501312a2d76a43ee688542b21622056a372959b65d63d
|
||||
strategy:
|
||||
# One browser breaking must not mask a failure in another - report all.
|
||||
fail-fast: false
|
||||
@@ -40,15 +45,23 @@ jobs:
|
||||
cache-dependency-path: frontend/package-lock.json
|
||||
- name: Install Task
|
||||
uses: go-task/setup-task@01a4adf9db2d14c1de7a560f09170b6e0df736aa # v2.1.0
|
||||
- name: Install Playwright (${{ matrix.browser }})
|
||||
run: task e2e:install -- ${{ matrix.browser }}
|
||||
- name: Build frontend (production bundle for vite preview)
|
||||
env:
|
||||
VITE_BUILD_FOR_PREVIEW: "1"
|
||||
run: task frontend:build
|
||||
- name: Run stubbed E2E tests (${{ matrix.browser }})
|
||||
env:
|
||||
# The official Playwright image expects its browser runtime under
|
||||
# the root home directory. Keep this scoped to Playwright and use a
|
||||
# neutral Docker config path so Docker does not read /root/.docker.
|
||||
HOME: /root
|
||||
DOCKER_CONFIG: /tmp/playwright-docker-config
|
||||
PLAYWRIGHT_JSON_OUTPUT_FILE: ${{ github.workspace }}/frontend/playwright-report/results.json
|
||||
NPM_CONFIG_PREFER_OFFLINE: "true"
|
||||
NPM_CONFIG_FETCH_RETRIES: "5"
|
||||
NPM_CONFIG_FETCH_RETRY_FACTOR: "2"
|
||||
NPM_CONFIG_FETCH_RETRY_MINTIMEOUT: "1000"
|
||||
NPM_CONFIG_FETCH_RETRY_MAXTIMEOUT: "120000"
|
||||
run: task e2e:stubbed-project PROJECT=${{ matrix.project }} -- --workers=3
|
||||
- name: Flag flaky tests
|
||||
# Runs regardless of the test outcome: a flaky test (passed on retry)
|
||||
|
||||
@@ -43,7 +43,9 @@ jobs:
|
||||
|
||||
generate-frontend-license-report:
|
||||
# ci-bot, not bot-identity: this job runs on PRs too, and bot-identity is main-only.
|
||||
environment: ci-bot
|
||||
environment:
|
||||
name: ci-bot
|
||||
deployment: false
|
||||
if: needs.files-changed.outputs.licenses-frontend == 'true'
|
||||
name: Generate Frontend License Report
|
||||
needs: files-changed
|
||||
@@ -319,7 +321,9 @@ jobs:
|
||||
|
||||
generate-backend-license-report:
|
||||
# ci-bot, not bot-identity: this job runs on PRs too, and bot-identity is main-only.
|
||||
environment: ci-bot
|
||||
environment:
|
||||
name: ci-bot
|
||||
deployment: false
|
||||
if: needs.files-changed.outputs.licenses-backend == 'true'
|
||||
needs: files-changed
|
||||
name: Generate Backend License Report
|
||||
|
||||
@@ -107,7 +107,7 @@ jobs:
|
||||
}
|
||||
- name: Install uv
|
||||
if: always()
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
name: Prime Gradle Cache
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
push:
|
||||
branches: ["main"]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
gradle-cache-prime:
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
name: Prime shared Gradle cache
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- name: Harden the runner (Audit all outbound calls)
|
||||
uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1
|
||||
with:
|
||||
egress-policy: audit
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Calculate Gradle cache key
|
||||
id: gradle-cache-key
|
||||
shell: bash
|
||||
run: |
|
||||
echo "key=gradle-v1-${{ runner.os }}-${{ runner.arch }}-jdk-25-${{ hashFiles('gradle/wrapper/gradle-wrapper.properties', 'gradle/libs.versions.toml', 'buildSrc/**', 'settings.gradle', 'build.gradle', 'app/**/build.gradle', 'gradle/**/*.gradle') }}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Cache Gradle (lookup-only)
|
||||
id: cache-gradle-restore
|
||||
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
path: |
|
||||
~/.gradle/caches
|
||||
~/.gradle/wrapper
|
||||
key: ${{ steps.gradle-cache-key.outputs.key }}
|
||||
lookup-only: true
|
||||
|
||||
- name: Set up JDK 25
|
||||
if: steps.cache-gradle-restore.outputs.cache-hit != 'true'
|
||||
uses: actions/setup-java@b6effb05e454b25005698d916606bdc6ffcbf961 # v5.7.0
|
||||
with:
|
||||
java-version: "25"
|
||||
distribution: "temurin"
|
||||
|
||||
- name: Resolve backend dependencies
|
||||
if: steps.cache-gradle-restore.outputs.cache-hit != 'true'
|
||||
run: ./gradlew :stirling-pdf:classes --no-daemon
|
||||
env:
|
||||
STIRLING_FLAVOR: saas
|
||||
MAVEN_USER: ${{ secrets.MAVEN_USER }}
|
||||
MAVEN_PASSWORD: ${{ secrets.MAVEN_PASSWORD }}
|
||||
MAVEN_PUBLIC_URL: ${{ secrets.MAVEN_PUBLIC_URL }}
|
||||
|
||||
- name: Save cache Gradle User Home
|
||||
if: steps.cache-gradle-restore.outputs.cache-hit != 'true'
|
||||
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
path: |
|
||||
~/.gradle/caches
|
||||
~/.gradle/wrapper
|
||||
key: ${{ steps.gradle-cache-key.outputs.key }}
|
||||
@@ -38,7 +38,9 @@ permissions:
|
||||
|
||||
jobs:
|
||||
determine-matrix:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
if: ${{ vars.CI_PROFILE != 'lite' }}
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
@@ -116,7 +118,9 @@ jobs:
|
||||
env:
|
||||
INPUT_PLATFORM: ${{ github.event.inputs.platform }}
|
||||
build-jars:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
needs: determine-matrix
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
@@ -693,6 +697,17 @@ jobs:
|
||||
path: ./dist/*
|
||||
retention-days: 1
|
||||
|
||||
- name: Cleanup temporary files
|
||||
if: always()
|
||||
shell: bash
|
||||
run: |
|
||||
rm -f certificate.p12
|
||||
rm -rf "$RUNNER_TEMP/msi-verify"
|
||||
if [ "${{ matrix.platform }}" = "macos-15" ]; then
|
||||
security delete-keychain "$RUNNER_TEMP/app-signing.keychain-db" 2>/dev/null || true
|
||||
fi
|
||||
continue-on-error: true
|
||||
|
||||
collect-and-release:
|
||||
needs: [determine-matrix, build, build-jars]
|
||||
runs-on: ubuntu-latest
|
||||
@@ -857,7 +872,7 @@ jobs:
|
||||
# Gate publish on valid updater sigs. Runs after the review upload (so
|
||||
# artifacts survive for debugging) and before action-gh-release.
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -127,7 +127,9 @@ jobs:
|
||||
# Runs the @nightly tag (conversion scenarios) plus a 10-shard concurrency run
|
||||
# of every other feature.
|
||||
cucumber-nightly:
|
||||
environment: ci-unsigned
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
name: Cucumber (nightly scenarios + full concurrency)
|
||||
runs-on: ubuntu-latest
|
||||
# Fork pull requests get no MAVEN_* secrets, so the image build cannot work.
|
||||
@@ -146,13 +148,13 @@ jobs:
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@be666c2fcd27ec809703dec50e508c2fdc7f6654 # v5.2.0
|
||||
uses: actions/setup-java@b6effb05e454b25005698d916606bdc6ffcbf961 # v5.7.0
|
||||
with:
|
||||
java-version: "25"
|
||||
distribution: "temurin"
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -25,7 +25,7 @@ jobs:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -75,6 +75,6 @@ jobs:
|
||||
|
||||
# Upload the results to GitHub's code scanning dashboard.
|
||||
- name: "Upload to code-scanning"
|
||||
uses: github/codeql-action/upload-sarif@5595ccaf912efad79be6eef63a5619ff05969be3 # v4.37.6
|
||||
uses: github/codeql-action/upload-sarif@ff2f1c621b7f889edc0d3c761ac2e6a3f8cdb0dd # v4.37.7
|
||||
with:
|
||||
sarif_file: results.sarif
|
||||
|
||||
@@ -53,7 +53,7 @@ jobs:
|
||||
private-key: ${{ secrets.GH_APP_PRIVATE_KEY }}
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
|
||||
uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
|
||||
with:
|
||||
enable-cache: true
|
||||
cache-dependency-glob: |
|
||||
|
||||
@@ -63,7 +63,9 @@ jobs:
|
||||
determine-matrix:
|
||||
# Only probes APPLE_CERTIFICATE for presence, so it stays on the unrestricted
|
||||
# signing environment - release-signing would block every PR run.
|
||||
environment: ci-signing
|
||||
environment:
|
||||
name: ci-signing
|
||||
deployment: false
|
||||
if: ${{ vars.CI_PROFILE != 'lite' }}
|
||||
runs-on: ubuntu-latest
|
||||
outputs:
|
||||
@@ -675,6 +677,17 @@ jobs:
|
||||
fi
|
||||
done
|
||||
|
||||
- name: Cleanup temporary files
|
||||
if: always()
|
||||
shell: bash
|
||||
run: |
|
||||
rm -f certificate.p12
|
||||
rm -rf "$RUNNER_TEMP/msi-verify"
|
||||
if [ "${{ matrix.platform }}" = "macos-15" ]; then
|
||||
security delete-keychain "$RUNNER_TEMP/app-signing.keychain-db" 2>/dev/null || true
|
||||
fi
|
||||
continue-on-error: true
|
||||
|
||||
pr-comment:
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
@@ -22,22 +22,45 @@ permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
# TODO: extract a pre-matrix `prepare` job that runs once and produces
|
||||
# shared artifacts for the three matrix entries below to consume:
|
||||
# 1. `task backend:build` — currently runs 3× in parallel with
|
||||
# identical env (DISABLE_ADDITIONAL_FEATURES=true,
|
||||
# STIRLING_PDF_DESKTOP_UI=false). Build once, upload the JAR as an
|
||||
# artifact, matrix entries download.
|
||||
# 2. The base-image `docker build` (gated on docker-base-changed) —
|
||||
# currently runs 3× in parallel against the same Dockerfile and
|
||||
# context. Build once, `docker save` to an artifact, matrix entries
|
||||
# `docker load` before the embedded build.
|
||||
# Saves ~2 full backend builds + 2 base-image builds per PR that touches
|
||||
# docker. May also be reusable from backend-build.yml's jdk-25 +
|
||||
# spring-security=true matrix entry if `task backend:build` and
|
||||
# `task backend:build:ci` produce equivalent JARs (verify before wiring).
|
||||
# A changed base image is shared by all three embedded-image builds. Build
|
||||
# it once and transfer it as an artifact; the matrix jobs use the local
|
||||
# Docker driver so the loaded image is visible to the build.
|
||||
prepare-base-image:
|
||||
if: github.event_name == 'pull_request' && inputs.docker-base-changed == 'true'
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Harden Runner
|
||||
uses: step-security/harden-runner@b09bb98e06d4d774595224525879c09bc6e98c40 # v2.20.1
|
||||
with:
|
||||
egress-policy: audit
|
||||
|
||||
- name: Checkout Repository
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Build base image locally
|
||||
run: docker build --platform linux/amd64 -t stirling-pdf-base:pr-test -f docker/base/Dockerfile docker/base
|
||||
|
||||
- name: Export base image
|
||||
run: docker save stirling-pdf-base:pr-test | gzip -1 > stirling-pdf-base-pr-test.tar.gz
|
||||
|
||||
- name: Upload base image
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: docker-base-pr-test
|
||||
path: stirling-pdf-base-pr-test.tar.gz
|
||||
retention-days: 1
|
||||
if-no-files-found: error
|
||||
|
||||
test-build-docker-images:
|
||||
environment: ci-unsigned
|
||||
if: always() && (needs.prepare-base-image.result == 'success' || needs.prepare-base-image.result == 'skipped')
|
||||
needs: [prepare-base-image]
|
||||
environment:
|
||||
name: ci-unsigned
|
||||
deployment: false
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
@@ -79,6 +102,16 @@ jobs:
|
||||
docker system prune -af || true
|
||||
echo "Disk space after cleanup:" && df -h
|
||||
|
||||
- name: Download prepared base image
|
||||
if: github.event_name == 'pull_request' && inputs.docker-base-changed == 'true'
|
||||
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
|
||||
with:
|
||||
name: docker-base-pr-test
|
||||
|
||||
- name: Load prepared base image
|
||||
if: github.event_name == 'pull_request' && inputs.docker-base-changed == 'true'
|
||||
run: gzip -dc stirling-pdf-base-pr-test.tar.gz | docker load
|
||||
|
||||
- name: Restore cache Gradle User Home
|
||||
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
|
||||
with:
|
||||
@@ -111,11 +144,6 @@ jobs:
|
||||
id: buildx
|
||||
uses: docker/setup-buildx-action@bb05f3f5519dd87d3ba754cc423b652a5edd6d2c # v4.2.0
|
||||
|
||||
- name: Build base image locally (PR base change only)
|
||||
if: github.event_name == 'pull_request' && inputs.docker-base-changed == 'true'
|
||||
run: |
|
||||
docker build -t stirling-pdf-base:pr-test -f docker/base/Dockerfile docker/base
|
||||
|
||||
- name: Set base image and platform for this build
|
||||
id: build-params
|
||||
# Pass workflow inputs through env vars rather than expanding `${{ }}`
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
name: Update Gradle
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: "0 3 * * 1"
|
||||
|
||||
concurrency:
|
||||
group: update-gradle
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
update-gradle:
|
||||
name: Update Gradle and Docker images
|
||||
permissions:
|
||||
contents: write
|
||||
pull-requests: write
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: Harden runner
|
||||
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
|
||||
with:
|
||||
egress-policy: audit
|
||||
|
||||
- name: Check out repository
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- name: Set up Java
|
||||
uses: actions/setup-java@b6effb05e454b25005698d916606bdc6ffcbf961 # v5.7.0
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: "25"
|
||||
|
||||
- name: Find latest Gradle release
|
||||
id: gradle
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
version=$(curl --fail --silent --show-error --retry 3 \
|
||||
https://services.gradle.org/versions/current | jq -r '.version')
|
||||
[[ "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]] || {
|
||||
echo "Could not determine a stable Gradle version: $version" >&2
|
||||
exit 1
|
||||
}
|
||||
echo "version=$version" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Find matching Docker image digest
|
||||
id: docker
|
||||
env:
|
||||
GRADLE_VERSION: ${{ steps.gradle.outputs.version }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
tag="${GRADLE_VERSION}-jdk25"
|
||||
digest=$(curl --fail --silent --show-error --retry 3 \
|
||||
"https://hub.docker.com/v2/repositories/library/gradle/tags/${tag}" \
|
||||
| jq -r '.digest // empty')
|
||||
[[ "$digest" =~ ^sha256:[0-9a-f]{64}$ ]] || {
|
||||
echo "Docker image gradle:${tag} was not found" >&2
|
||||
exit 1
|
||||
}
|
||||
echo "tag=$tag" >> "$GITHUB_OUTPUT"
|
||||
echo "digest=$digest" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Update Gradle wrapper
|
||||
env:
|
||||
GRADLE_VERSION: ${{ steps.gradle.outputs.version }}
|
||||
run: ./gradlew wrapper --gradle-version "$GRADLE_VERSION" --distribution-type bin
|
||||
|
||||
- name: Update Gradle Docker images
|
||||
env:
|
||||
DOCKER_TAG: ${{ steps.docker.outputs.tag }}
|
||||
DOCKER_DIGEST: ${{ steps.docker.outputs.digest }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
find docker -type f -name 'Dockerfile*' -print0 |
|
||||
xargs -0 sed -E -i \
|
||||
"s#gradle:[^@[:space:]]+-jdk25(@sha256:[^[:space:]]+)?#gradle:${DOCKER_TAG}@${DOCKER_DIGEST}#g"
|
||||
|
||||
- name: Verify Gradle update
|
||||
env:
|
||||
EXPECTED_VERSION: ${{ steps.gradle.outputs.version }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
actual=$(./gradlew --version | sed -n 's/^Gradle \([0-9.]*\)$/\1/p')
|
||||
[[ "$actual" == "$EXPECTED_VERSION" ]] || {
|
||||
echo "Wrapper resolved Gradle $actual, expected $EXPECTED_VERSION" >&2
|
||||
exit 1
|
||||
}
|
||||
if git diff --quiet; then
|
||||
echo "Gradle is already up to date."
|
||||
exit 0
|
||||
fi
|
||||
git diff --check
|
||||
|
||||
- name: Create pull request
|
||||
uses: peter-evans/create-pull-request@5f6978faf089d4d20b00c7766989d076bb2fc7f1 # v8.1.1
|
||||
with:
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
branch: automation/update-gradle
|
||||
delete-branch: true
|
||||
commit-message: "chore: update Gradle"
|
||||
title: "chore: update Gradle to ${{ steps.gradle.outputs.version }}"
|
||||
body: |
|
||||
Automated update of the Gradle wrapper and Gradle Docker build images.
|
||||
|
||||
Gradle version: `${{ steps.gradle.outputs.version }}`
|
||||
Docker image: `gradle:${{ steps.docker.outputs.tag }}`
|
||||
labels: dependencies
|
||||
@@ -27,3 +27,8 @@ app/core/src/main/java/stirling/software/SPDF/pdf/signature/CreateSignatureBase.
|
||||
# Supabase publishable key (public by design, RLS-protected) used as a CI fallback
|
||||
# default in the tauri-build workflow when the GitHub secret is unset - not a real secret.
|
||||
.github/workflows/tauri-build.yml:generic-api-key:402
|
||||
|
||||
# Staging Supabase publishable key (public by design). Ignored here rather than with an
|
||||
# inline gitleaks:allow because a trailing comment in a .properties file is part of the
|
||||
# value, so the pragma would end up inside the key.
|
||||
app/saas/src/main/resources/application-staging.properties:generic-api-key:16
|
||||
|
||||
+3
-1
@@ -1,5 +1,7 @@
|
||||
{
|
||||
"ignoredFiles": [
|
||||
"frontend/editor/src-tauri/icons/icon.png"
|
||||
"frontend/editor/src-tauri/icons/macos/*",
|
||||
"frontend/editor/src-tauri/icons/linux/*",
|
||||
"frontend/editor/src-tauri/icons/windows/*"
|
||||
]
|
||||
}
|
||||
|
||||
+49
-6
@@ -57,16 +57,57 @@ tasks:
|
||||
- cmd: ./gradlew clean bootRun -PbuildWithFrontend=true
|
||||
platforms: [linux, darwin]
|
||||
|
||||
# SaaS backend. dev:saas -> the PR's preview branch, staging:saas -> shared v3,
|
||||
# PROFILES=none -> production against your own SAAS_DB_*. Production has no named
|
||||
# task on purpose. Use `none`, not an empty value: Go template `default` treats ""
|
||||
# as absent and would resolve back to dev.
|
||||
|
||||
dev:saas:
|
||||
desc: "Start backend in SaaS flavor against Supabase"
|
||||
# `dotenv:` reads from the root Taskfile's directory (".") because this
|
||||
# subtaskfile is included with `dir: .`.
|
||||
desc: "Start SaaS backend against the current PR's Supabase preview branch"
|
||||
dotenv: ['app/.env.saas.local', 'app/.env.saas']
|
||||
vars:
|
||||
PROFILES: '{{.PROFILES | default "dev"}}'
|
||||
cmds:
|
||||
# Don't move this check into a `sh:` var: dotenv is visible in cmds but not
|
||||
# during var evaluation, so the test would always see an empty value.
|
||||
- cmd: |
|
||||
if [ "{{.PROFILES}}" = "dev" ] && [ -z "${SAAS_DEV_PROJECT_REF:-}" ]; then
|
||||
echo ">> SAAS_DEV_PROJECT_REF is not set."
|
||||
echo ">> Testing a SaaS PR? Put its ref, DB password and publishable key in app/.env.saas.local."
|
||||
echo ">> Wanted the shared v3 project? Use 'task backend:staging:saas' instead."
|
||||
exit 1
|
||||
fi
|
||||
- task: _run:saas
|
||||
vars:
|
||||
PORT: '{{.PORT}}'
|
||||
PROFILES: '{{.PROFILES}}'
|
||||
AIENGINE_URL: '{{.AIENGINE_URL}}'
|
||||
AIENGINE_ENABLED: '{{.AIENGINE_ENABLED}}'
|
||||
AIENGINE_TIMEOUTSECONDS: '{{.AIENGINE_TIMEOUTSECONDS}}'
|
||||
|
||||
staging:saas:
|
||||
desc: "Start SaaS backend against the shared v3 staging project"
|
||||
cmds:
|
||||
- task: _run:saas
|
||||
vars:
|
||||
PORT: '{{.PORT}}'
|
||||
PROFILES: staging
|
||||
AIENGINE_URL: '{{.AIENGINE_URL}}'
|
||||
AIENGINE_ENABLED: '{{.AIENGINE_ENABLED}}'
|
||||
AIENGINE_TIMEOUTSECONDS: '{{.AIENGINE_TIMEOUTSECONDS}}'
|
||||
|
||||
_run:saas:
|
||||
internal: true
|
||||
dotenv: ['app/.env.saas.local', 'app/.env.saas']
|
||||
ignore_error: true
|
||||
vars:
|
||||
PORT: '{{.PORT | default "8080"}}'
|
||||
# Override to "" to run the pure `saas` profile against your own SAAS_DB_*.
|
||||
PROFILES: '{{.PROFILES | default "dev"}}'
|
||||
# Built here rather than inline in the cmds below: the Windows line is an
|
||||
# unquoted YAML scalar wrapping a cmd.exe string, so a nested {{if ne .X
|
||||
# "none"}} needs escaped quotes that reach the Go template as literal
|
||||
# backslashes and fail with `unexpected "\" in operand`.
|
||||
PROFILE_ARGS: '{{if ne .PROFILES "none"}}--spring.profiles.include={{.PROFILES}}{{end}}'
|
||||
AIENGINE_URL: '{{.AIENGINE_URL | default ""}}'
|
||||
AIENGINE_ENABLED: '{{.AIENGINE_ENABLED | default "false"}}'
|
||||
AIENGINE_TIMEOUTSECONDS: '{{.AIENGINE_TIMEOUTSECONDS | default "120"}}'
|
||||
@@ -77,9 +118,11 @@ tasks:
|
||||
AIENGINE_ENABLED: '{{.AIENGINE_ENABLED}}'
|
||||
AIENGINE_TIMEOUTSECONDS: '{{.AIENGINE_TIMEOUTSECONDS}}'
|
||||
cmds:
|
||||
- cmd: cmd /c ".\gradlew.bat :stirling-pdf:bootRun {{if .PROFILES}}--args=\"--spring.profiles.include={{.PROFILES}}\"{{end}}"
|
||||
# PROFILE_ARGS is empty when PROFILES=none, i.e. the bare `saas` profile
|
||||
# against SAAS_DB_* (production).
|
||||
- cmd: cmd /c ".\gradlew.bat :stirling-pdf:bootRun {{if .PROFILE_ARGS}}--args=\"{{.PROFILE_ARGS}}\"{{end}}"
|
||||
platforms: [windows]
|
||||
- cmd: ./gradlew :stirling-pdf:bootRun {{if .PROFILES}}--args='--spring.profiles.include={{.PROFILES}}'{{end}}
|
||||
- cmd: ./gradlew :stirling-pdf:bootRun {{if .PROFILE_ARGS}}--args='{{.PROFILE_ARGS}}'{{end}}
|
||||
platforms: [linux, darwin]
|
||||
|
||||
build:
|
||||
|
||||
+64
-10
@@ -5,6 +5,14 @@ version: '3'
|
||||
# mode flag) or use `--project editor/...` for tsc — so the editor lives
|
||||
# under frontend/editor/ without each task needing a cd.
|
||||
|
||||
vars:
|
||||
# Dev-only browser-tab label so concurrent worktrees are distinguishable. Only
|
||||
# the worktree folder basename (e.g. "wt1") is exposed — never the full path,
|
||||
# hostname, or user. Dropped from production builds.
|
||||
DEV_LABEL:
|
||||
sh: >-
|
||||
{{if eq OS "windows"}}powershell -NoProfile -Command '$root = git rev-parse --show-toplevel 2>$null; if (-not $root) { $root = (Get-Location).Path }; Split-Path -Leaf $root'{{else}}basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)"{{end}}
|
||||
|
||||
tasks:
|
||||
install:
|
||||
desc: "Install dependencies"
|
||||
@@ -80,16 +88,52 @@ tasks:
|
||||
OPEN: '{{.OPEN | default ""}}'
|
||||
env:
|
||||
BACKEND_URL: '{{.BACKEND_URL}}'
|
||||
# Dev-only browser-tab label so concurrent worktrees are distinguishable.
|
||||
# Only the worktree folder basename (e.g. "wt1") is exposed — never the
|
||||
# full path, hostname, or user. Consumed at dev-serve time by vite.config
|
||||
# and dropped from production builds.
|
||||
STIRLING_DEV_LABEL:
|
||||
sh: >-
|
||||
{{if eq OS "windows"}}powershell -NoProfile -Command '$root = git rev-parse --show-toplevel 2>$null; if (-not $root) { $root = (Get-Location).Path }; Split-Path -Leaf $root'{{else}}basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)"{{end}}
|
||||
STIRLING_DEV_LABEL: '{{.DEV_LABEL}}'
|
||||
cmds:
|
||||
- npx vite editor --mode {{.MODE}} --port {{.PORT}}{{if .OPEN}} --open{{end}}
|
||||
|
||||
# Separate from dev:_run rather than a flag on it: Task sets an `env:` key even
|
||||
# when its value resolves to empty, and Vite treats an empty process.env VITE_* as
|
||||
# authoritative over the committed editor/.env, so folding these in blanks Supabase
|
||||
# config for the core, proprietary and desktop dev servers.
|
||||
dev:_run:saas:
|
||||
internal: true
|
||||
ignore_error: true
|
||||
# The backend's own env files, so both halves target one project. Paths are
|
||||
# relative to this taskfile's dir, `frontend`.
|
||||
dotenv: ['../app/.env.saas.local', '../app/.env.saas']
|
||||
vars:
|
||||
PORT: '{{.PORT | default "5173"}}'
|
||||
BACKEND_URL: '{{.BACKEND_URL | default "http://localhost:8080"}}'
|
||||
OPEN: '{{.OPEN | default ""}}'
|
||||
SAAS_ENV: '{{.SAAS_ENV | default "dev"}}'
|
||||
env:
|
||||
BACKEND_URL: '{{.BACKEND_URL}}'
|
||||
STIRLING_DEV_LABEL: '{{.DEV_LABEL}}'
|
||||
SAAS_ENV: '{{.SAAS_ENV}}'
|
||||
# A real process.env VITE_* beats a committed .env in Vite (loadEnv applies
|
||||
# process.env last), which is what lets this override editor/.env.
|
||||
#
|
||||
# These must stay `sh:`, not Go templates: dotenv values are visible to Task's
|
||||
# embedded shell but not to templates, where {{.SAAS_DEV_PROJECT_REF}} is
|
||||
# always empty.
|
||||
VITE_SUPABASE_URL:
|
||||
sh: |
|
||||
case "${SAAS_ENV:-dev}" in
|
||||
staging) ref="${SAAS_STAGING_PROJECT_REF:?set it in app/.env.saas.local}" ;;
|
||||
*) ref="${SAAS_DEV_PROJECT_REF:?set it in app/.env.saas.local, or run task staging:saas}" ;;
|
||||
esac
|
||||
echo "https://${ref}.supabase.co"
|
||||
VITE_SUPABASE_PUBLISHABLE_DEFAULT_KEY:
|
||||
sh: |
|
||||
case "${SAAS_ENV:-dev}" in
|
||||
staging) echo "${SAAS_STAGING_PUBLISHABLE_KEY:?set it in app/.env.saas.local}" ;;
|
||||
*) echo "${SAAS_DEV_PUBLISHABLE_KEY:?set it in app/.env.saas.local}" ;;
|
||||
esac
|
||||
cmds:
|
||||
- 'echo ">> frontend Supabase target: $VITE_SUPABASE_URL"'
|
||||
- npx vite editor --mode saas --port {{.PORT}}{{if .OPEN}} --open{{end}}
|
||||
|
||||
dev:
|
||||
desc: "Start frontend dev server"
|
||||
cmds:
|
||||
@@ -111,13 +155,23 @@ tasks:
|
||||
vars: { MODE: proprietary, PORT: '{{.PORT}}', BACKEND_URL: '{{.BACKEND_URL}}', OPEN: '{{.OPEN}}' }
|
||||
|
||||
dev:saas:
|
||||
desc: "Start frontend dev server in SaaS mode"
|
||||
desc: "Start frontend dev server in SaaS mode (SAAS_ENV=dev|staging|prod)"
|
||||
deps:
|
||||
- task: prepare
|
||||
vars: { MODE: saas }
|
||||
vars:
|
||||
SAAS_ENV: '{{.SAAS_ENV | default "dev"}}'
|
||||
# prod routes to the plain runner, which sets no VITE_SUPABASE_* and so leaves
|
||||
# the committed editor/.env alone.
|
||||
RUNNER: '{{if eq .SAAS_ENV "prod"}}dev:_run{{else}}dev:_run:saas{{end}}'
|
||||
cmds:
|
||||
- task: dev:_run
|
||||
vars: { MODE: saas, PORT: '{{.PORT}}', BACKEND_URL: '{{.BACKEND_URL}}', OPEN: '{{.OPEN}}' }
|
||||
- task: '{{.RUNNER}}'
|
||||
vars:
|
||||
MODE: saas
|
||||
PORT: '{{.PORT}}'
|
||||
BACKEND_URL: '{{.BACKEND_URL}}'
|
||||
OPEN: '{{.OPEN}}'
|
||||
SAAS_ENV: '{{.SAAS_ENV}}'
|
||||
|
||||
dev:desktop:
|
||||
desc: "Start frontend dev server in desktop mode"
|
||||
|
||||
+2
-2
@@ -46,8 +46,8 @@ This guide focuses on developing for Stirling 2.0, including both the React fron
|
||||
- Docker
|
||||
- Git
|
||||
- Java JDK 25
|
||||
- Node.js 18+ and npm (required for frontend development)
|
||||
- Gradle 7.0 or later (Included within the repo)
|
||||
- Node.js 22+ and npm (required for frontend development)
|
||||
- Gradle 9.0 or later (Included within the repo)
|
||||
- [uv](https://docs.astral.sh/uv/) — Python package manager (required for engine development)
|
||||
- Rust and Cargo (required for Tauri desktop app development)
|
||||
- Tauri CLI (install with `cargo install tauri-cli`)
|
||||
|
||||
+18
-3
@@ -99,11 +99,22 @@ tasks:
|
||||
BACKEND_URL: 'http://localhost:{{.BACKEND_PORT}}'
|
||||
OPEN: "true"
|
||||
|
||||
# Set SAAS_DEV_PROJECT_REF in app/.env.saas.local to pick the PR.
|
||||
dev:saas:
|
||||
desc: "Start SaaS backend + frontend concurrently on free ports"
|
||||
desc: "Start SaaS backend + frontend + engine against the current PR's preview branch"
|
||||
cmds:
|
||||
- task: dev:_all
|
||||
vars: { FRONTEND: saas, BACKEND: saas }
|
||||
vars: { FRONTEND: saas, BACKEND: saas, SAAS_ENV: dev }
|
||||
|
||||
staging:saas:
|
||||
desc: "Start SaaS backend + frontend + engine against the shared v3 staging project"
|
||||
cmds:
|
||||
- task: dev:_all
|
||||
vars:
|
||||
FRONTEND: saas
|
||||
BACKEND: saas
|
||||
BACKEND_TASK: backend:staging:saas
|
||||
SAAS_ENV: staging
|
||||
|
||||
dev:all:
|
||||
desc: "Start backend + frontend + engine concurrently on free ports"
|
||||
@@ -115,6 +126,9 @@ tasks:
|
||||
vars:
|
||||
FRONTEND: '{{.FRONTEND | default "proprietary"}}'
|
||||
BACKEND: '{{.BACKEND | default "proprietary"}}'
|
||||
BACKEND_TASK: '{{.BACKEND_TASK | default (printf "backend:dev:%s" .BACKEND)}}'
|
||||
# Only meaningful to the saas frontend; every other flavor ignores it.
|
||||
SAAS_ENV: '{{.SAAS_ENV | default ""}}'
|
||||
PORTS:
|
||||
sh: '{{if eq OS "windows"}}{{.FIND_FREE_PORT_PS}} 8080 5173 5001{{else}}{{.FIND_FREE_PORT_SH}} 8080 5173 5001{{end}}'
|
||||
BACKEND_PORT: '{{index (splitList "\n" .PORTS) 0}}'
|
||||
@@ -124,7 +138,7 @@ tasks:
|
||||
- task: engine:dev
|
||||
vars:
|
||||
PORT: '{{.ENGINE_PORT}}'
|
||||
- task: 'backend:dev:{{.BACKEND}}'
|
||||
- task: '{{.BACKEND_TASK}}'
|
||||
vars:
|
||||
PORT: '{{.BACKEND_PORT}}'
|
||||
AIENGINE_URL: 'http://localhost:{{.ENGINE_PORT}}'
|
||||
@@ -134,6 +148,7 @@ tasks:
|
||||
PORT: '{{.FRONTEND_PORT}}'
|
||||
BACKEND_URL: 'http://localhost:{{.BACKEND_PORT}}'
|
||||
OPEN: "true"
|
||||
SAAS_ENV: '{{.SAAS_ENV}}'
|
||||
|
||||
# ============================================================
|
||||
# Build
|
||||
|
||||
+35
-17
@@ -1,15 +1,16 @@
|
||||
###############################################################################
|
||||
# Stirling-PDF SaaS environment defaults.
|
||||
# Stirling-PDF SaaS environment defaults. Committed, non-secret. Real values for secrets go in
|
||||
# .env.saas.local, which is loaded first and wins. Do not commit that file.
|
||||
#
|
||||
# This file is committed and provides non-secret defaults loaded by
|
||||
# `task backend:dev:saas`. Put real values for secrets (passwords, project
|
||||
# refs, edge function secrets) in `.env.saas.local` - any variable set there
|
||||
# takes precedence over what's defined here.
|
||||
# Three environments, each deriving its Supabase URLs, JWT issuer and JWKS from one project ref:
|
||||
#
|
||||
# DO NOT commit `.env.saas.local`. Only `.env.saas` is checked in.
|
||||
###############################################################################
|
||||
# prod PROFILES=none SAAS_DB_* the live project
|
||||
# staging PROFILES=staging SAAS_STAGING_* pinned to v3, always there
|
||||
# dev PROFILES=dev SAAS_DEV_* follows a SaaS PR's preview branch
|
||||
#
|
||||
# dev is the default for `task backend:dev:saas`. Use staging for somewhere stable; use dev when
|
||||
# testing an open SaaS PR, since its preview branch is the only place those migrations are applied.
|
||||
|
||||
# ---------- Supabase project ----------
|
||||
# ---------- Supabase project (prod / no-profile) ----------
|
||||
# Project reference (the subdomain part of <ref>.supabase.co). Required.
|
||||
# Set in .env.saas.local.
|
||||
SAAS_DB_PROJECT_REF=
|
||||
@@ -17,18 +18,35 @@ SAAS_DB_PROJECT_REF=
|
||||
# Edge function secret used by billing/license rollup calls. Set in .env.saas.local.
|
||||
SUPABASE_EDGE_FUNCTION_SECRET=
|
||||
|
||||
# ---------- Database (saas profile) ----------
|
||||
# Direct JDBC URL to the Supabase Postgres. Required when running the plain
|
||||
# `saas` profile (i.e. without `--spring.profiles.include=dev`).
|
||||
# ---------- Database (no profile) ----------
|
||||
# Direct JDBC URL to the Supabase Postgres. Required when running without
|
||||
# `--spring.profiles.include=...`.
|
||||
# Example: jdbc:postgresql://db.<project-ref>.supabase.co:5432/postgres
|
||||
SAAS_DB_URL=
|
||||
SAAS_DB_USERNAME=postgres
|
||||
SAAS_DB_PASSWORD=
|
||||
|
||||
# ---------- Database (dev profile overrides) ----------
|
||||
# Used when `--spring.profiles.include=dev` is active. The dev profile
|
||||
# defaults the URL/username to the shared dev Supabase project, but the
|
||||
# password must still be provided in .env.saas.local.
|
||||
SAAS_DEV_DB_URL=
|
||||
# ---------- staging profile ----------
|
||||
# The shared long-lived v3 project. application-staging.properties defaults the ref,
|
||||
# URL, database host and meter endpoint, so staging needs only the password, in
|
||||
# .env.saas.local. Set SAAS_STAGING_PROJECT_REF to repoint it; everything derives.
|
||||
#
|
||||
# The ref and publishable key are duplicated here because the task derives the
|
||||
# frontend's VITE_SUPABASE_* from them and a shell cannot read a Spring default.
|
||||
# Neither is secret: the ref is a public subdomain, the key ships in the bundle.
|
||||
SAAS_STAGING_PROJECT_REF=qacaivhsjtftfwtgjvva
|
||||
SAAS_STAGING_PUBLISHABLE_KEY=sb_publishable_nIM8y-9ARPE7EzQwAQHKMg_40fCN6kY # gitleaks:allow
|
||||
SAAS_STAGING_DB_USERNAME=postgres
|
||||
SAAS_STAGING_DB_PASSWORD=
|
||||
|
||||
# ---------- dev profile ----------
|
||||
# The SaaS PR's Supabase preview branch. Take the ref from that PR's "Supabase
|
||||
# Preview" check; the profile derives URL, JWT issuer, JWKS, meter endpoint and
|
||||
# database host from it, so this one value follows a different PR.
|
||||
#
|
||||
# A preview branch has its own password and keys; the parent project's will not
|
||||
# authenticate. Both go in .env.saas.local, along with the ref.
|
||||
SAAS_DEV_PROJECT_REF=
|
||||
SAAS_DEV_PUBLISHABLE_KEY=
|
||||
SAAS_DEV_DB_USERNAME=postgres
|
||||
SAAS_DEV_DB_PASSWORD=
|
||||
|
||||
@@ -21,8 +21,8 @@ dependencies {
|
||||
api 'org.snakeyaml:snakeyaml-engine:3.0.1'
|
||||
api "org.springdoc:springdoc-openapi-starter-webmvc-ui:3.0.3"
|
||||
// Simple Java Mail for EML/MSG parsing (replaces direct Angus Mail usage)
|
||||
api 'org.simplejavamail:simple-java-mail:9.2.0'
|
||||
api 'org.simplejavamail:outlook-module:9.2.0' // MSG file support
|
||||
api 'org.simplejavamail:simple-java-mail:9.3.2'
|
||||
api 'org.simplejavamail:outlook-module:9.3.2' // MSG file support
|
||||
api 'jakarta.mail:jakarta.mail-api:2.1.5'
|
||||
runtimeOnly 'org.eclipse.angus:angus-mail:2.0.5'
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.beans.factory.annotation.Qualifier;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
@@ -13,6 +14,7 @@ import lombok.Getter;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.common.model.ApplicationProperties;
|
||||
import stirling.software.common.service.PdfaLevelAServiceInterface;
|
||||
|
||||
@Service
|
||||
@Slf4j
|
||||
@@ -51,12 +53,16 @@ public class EndpointConfiguration {
|
||||
private Map<String, DisableReason> groupDisableReasons = new ConcurrentHashMap<>();
|
||||
private Map<String, Set<String>> endpointAlternatives = new ConcurrentHashMap<>();
|
||||
private final boolean runningProOrHigher;
|
||||
private final boolean pdfUaAvailable;
|
||||
|
||||
public EndpointConfiguration(
|
||||
ApplicationProperties applicationProperties,
|
||||
@Qualifier("runningProOrHigher") boolean runningProOrHigher) {
|
||||
@Qualifier("runningProOrHigher") boolean runningProOrHigher,
|
||||
@Autowired(required = false) PdfaLevelAServiceInterface pdfaLevelAService) {
|
||||
this.applicationProperties = applicationProperties;
|
||||
this.runningProOrHigher = runningProOrHigher;
|
||||
// The PDF/UA tagger ships in the proprietary module, and so do its endpoints.
|
||||
this.pdfUaAvailable = pdfaLevelAService != null;
|
||||
init();
|
||||
processEnvironmentConfigs();
|
||||
}
|
||||
@@ -356,6 +362,7 @@ public class EndpointConfiguration {
|
||||
addEndpointToGroup("Convert", "pdf-to-img");
|
||||
addEndpointToGroup("Convert", "img-to-pdf");
|
||||
addEndpointToGroup("Convert", "pdf-to-pdfa");
|
||||
addEndpointToGroup("Convert", "pdf-to-ua");
|
||||
addEndpointToGroup("Convert", "file-to-pdf");
|
||||
addEndpointToGroup("Convert", "pdf-to-word");
|
||||
addEndpointToGroup("Convert", "pdf-to-presentation");
|
||||
@@ -395,6 +402,7 @@ public class EndpointConfiguration {
|
||||
// Backend-only endpoints (not in frontend tool registry endpoints)
|
||||
addEndpointToGroup("Security", "redact");
|
||||
addEndpointToGroup("Security", "verify-pdf");
|
||||
addEndpointToGroup("Security", "accessibility-report");
|
||||
addEndpointToGroup("Security", "sign");
|
||||
|
||||
// Adding endpoints to "Other" group
|
||||
@@ -529,6 +537,8 @@ public class EndpointConfiguration {
|
||||
addEndpointToGroup("Java", "json-to-pdf");
|
||||
addEndpointToGroup("Java", "pdf-to-video");
|
||||
addEndpointToGroup("Java", "verify-pdf");
|
||||
addEndpointToGroup("Java", "pdf-to-ua");
|
||||
addEndpointToGroup("Java", "accessibility-report");
|
||||
addEndpointToGroup("Java", "flatten");
|
||||
addEndpointToGroup("Java", "unlock-pdf-forms");
|
||||
addEndpointToGroup("Java", "validate-signature");
|
||||
@@ -600,6 +610,8 @@ public class EndpointConfiguration {
|
||||
|
||||
// veraPDF dependent endpoints
|
||||
addEndpointToGroup("veraPDF", "verify-pdf");
|
||||
addEndpointToGroup("veraPDF", "pdf-to-ua");
|
||||
addEndpointToGroup("veraPDF", "accessibility-report");
|
||||
|
||||
// Pdftohtml dependent endpoints
|
||||
addEndpointToGroup("Pdftohtml", "pdf-to-html");
|
||||
@@ -630,6 +642,11 @@ public class EndpointConfiguration {
|
||||
disableGroup("enterprise");
|
||||
}
|
||||
|
||||
if (!pdfUaAvailable) {
|
||||
disableEndpoint("pdf-to-ua");
|
||||
disableEndpoint("accessibility-report");
|
||||
}
|
||||
|
||||
if (!applicationProperties.getSystem().isEnableUrlToPDF()) {
|
||||
disableEndpoint("url-to-pdf");
|
||||
}
|
||||
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
package stirling.software.common.service;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Raises a converted PDF/A file from conformance level B to level A, which needs the tagging the
|
||||
* PDF/UA tagger does. Implemented only in the proprietary module; core builds convert at level B.
|
||||
*/
|
||||
public interface PdfaLevelAServiceInterface {
|
||||
|
||||
/**
|
||||
* @param levelA true only when the file was tagged and validated, so the claim is never a guess
|
||||
*/
|
||||
record Result(byte[] pdfBytes, boolean levelA, List<String> warnings) {}
|
||||
|
||||
/**
|
||||
* @param part PDF/A part, 1 to 3; part 1 keeps its PDF 1.4 version
|
||||
* @param alsoDeclareUa additionally claim PDF/UA, but only if it validates
|
||||
*/
|
||||
Result upgradeToLevelA(
|
||||
byte[] pdfBytes, int part, String language, String title, boolean alsoDeclareUa);
|
||||
}
|
||||
@@ -4,7 +4,14 @@ import java.util.regex.Pattern;
|
||||
|
||||
public class RequestUriUtils {
|
||||
|
||||
private static final Pattern SHARE_LINK_PATTERN = Pattern.compile("^/share/[^/]+/?$");
|
||||
// Share tokens are 36-char lowercase UUIDs (UUID.randomUUID().toString()); match exactly
|
||||
private static final Pattern SHARE_LINK_PATTERN =
|
||||
Pattern.compile(
|
||||
"^/share/[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}/?$");
|
||||
// Invite tokens are 36-char lowercase UUIDs (UUID.randomUUID().toString()); match exactly
|
||||
private static final Pattern INVITE_LINK_PATTERN =
|
||||
Pattern.compile(
|
||||
"^/invite/[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}/?$");
|
||||
|
||||
public static boolean isStaticResource(String requestURI) {
|
||||
return isStaticResource("", requestURI);
|
||||
@@ -69,7 +76,7 @@ public class RequestUriUtils {
|
||||
// cookie, so the server can't authenticate the navigation itself). The
|
||||
// portal gates access via its own auth gate + RequirePortalAccess, and its
|
||||
// data APIs stay protected, so serving the shell pre-auth is safe.
|
||||
if (normalizedUri.equals("/processor") || normalizedUri.startsWith("/processor/")) {
|
||||
if ("/processor".equals(normalizedUri) || normalizedUri.startsWith("/processor/")) {
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -209,7 +216,9 @@ public class RequestUriUtils {
|
||||
// Workflow participant endpoints - access controlled by share tokens, not login
|
||||
|| trimmedUri.startsWith("/api/v1/workflow/participant/")
|
||||
// Share-link SPA bootstrap; data APIs remain protected
|
||||
|| SHARE_LINK_PATTERN.matcher(trimmedUri).matches();
|
||||
|| SHARE_LINK_PATTERN.matcher(trimmedUri).matches()
|
||||
// Invite-accept SPA bootstrap; data APIs remain protected
|
||||
|| INVITE_LINK_PATTERN.matcher(trimmedUri).matches();
|
||||
}
|
||||
|
||||
private static String stripContextPath(String contextPath, String requestURI) {
|
||||
|
||||
+31
-1
@@ -17,6 +17,7 @@ import org.junit.jupiter.api.Test;
|
||||
import stirling.software.SPDF.config.EndpointConfiguration.DisableReason;
|
||||
import stirling.software.SPDF.config.EndpointConfiguration.EndpointAvailability;
|
||||
import stirling.software.common.model.ApplicationProperties;
|
||||
import stirling.software.common.service.PdfaLevelAServiceInterface;
|
||||
|
||||
/**
|
||||
* Unit tests for {@link EndpointConfiguration}. The class wires up its endpoint/group registry in
|
||||
@@ -32,7 +33,14 @@ class EndpointConfigurationGapTest {
|
||||
* Construct an EndpointConfiguration with the given pro flag and current applicationProperties.
|
||||
*/
|
||||
private EndpointConfiguration build(boolean runningProOrHigher) {
|
||||
return new EndpointConfiguration(applicationProperties, runningProOrHigher);
|
||||
return build(runningProOrHigher, null);
|
||||
}
|
||||
|
||||
/** The PDF/UA service is only present in proprietary builds, so it is injected separately. */
|
||||
private EndpointConfiguration build(
|
||||
boolean runningProOrHigher, PdfaLevelAServiceInterface pdfaLevelAService) {
|
||||
return new EndpointConfiguration(
|
||||
applicationProperties, runningProOrHigher, pdfaLevelAService);
|
||||
}
|
||||
|
||||
/** Default config: not pro, no removals, url-to-pdf disabled (default System flag is false). */
|
||||
@@ -177,6 +185,28 @@ class EndpointConfigurationGapTest {
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("PDF/UA availability")
|
||||
class PdfUaTests {
|
||||
|
||||
@Test
|
||||
@DisplayName("the PDF/UA endpoints are off when the proprietary tagger is absent")
|
||||
void disabledWithoutTagger() {
|
||||
EndpointConfiguration config = build(false, null);
|
||||
assertFalse(config.isEndpointEnabled("pdf-to-ua"));
|
||||
assertFalse(config.isEndpointEnabled("accessibility-report"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("they are on once the tagger is on the classpath")
|
||||
void enabledWithTagger() {
|
||||
EndpointConfiguration config =
|
||||
build(false, (pdfBytes, part, language, title, alsoDeclareUa) -> null);
|
||||
assertTrue(config.isEndpointEnabled("pdf-to-ua"));
|
||||
assertTrue(config.isEndpointEnabled("accessibility-report"));
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("group enable / disable")
|
||||
class GroupTests {
|
||||
|
||||
@@ -206,12 +206,24 @@ class RequestUriUtilsTest {
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_shareLinkTokenTrailingSlash() {
|
||||
assertTrue(RequestUriUtils.isPublicAuthEndpoint("/share/abc123/", ""));
|
||||
assertTrue(
|
||||
RequestUriUtils.isPublicAuthEndpoint(
|
||||
"/share/00dcac3a-fc7a-4989-9c4f-97745484d62f/", ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_shareLinkWithContextPath() {
|
||||
assertTrue(RequestUriUtils.isPublicAuthEndpoint("/app/share/abc123", "/app"));
|
||||
assertTrue(
|
||||
RequestUriUtils.isPublicAuthEndpoint(
|
||||
"/app/share/00dcac3a-fc7a-4989-9c4f-97745484d62f", "/app"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_shareLinkWithInvalidTokenLength() {
|
||||
assertFalse(RequestUriUtils.isPublicAuthEndpoint("/share/abc123", ""));
|
||||
assertFalse(
|
||||
RequestUriUtils.isPublicAuthEndpoint(
|
||||
"/share/00dcac3a-fc7a-4989-9c4f-97745484d62fa", ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -236,4 +248,86 @@ class RequestUriUtilsTest {
|
||||
RequestUriUtils.isPublicAuthEndpoint(
|
||||
"/api/v1/storage/share-links/abc123/metadata", ""));
|
||||
}
|
||||
|
||||
// --- invite-accept SPA bootstrap ---
|
||||
|
||||
private static final String INVITE_TOKEN = "06a20e7e-2e35-4e26-be7d-2dce14f28f12";
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_inviteLinkToken() {
|
||||
assertTrue(RequestUriUtils.isPublicAuthEndpoint("/invite/" + INVITE_TOKEN, ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_inviteLinkTokenTrailingSlash() {
|
||||
assertTrue(RequestUriUtils.isPublicAuthEndpoint("/invite/" + INVITE_TOKEN + "/", ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_inviteLinkWithContextPath() {
|
||||
assertTrue(RequestUriUtils.isPublicAuthEndpoint("/app/invite/" + INVITE_TOKEN, "/app"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_inviteRootNotPublic() {
|
||||
// Avoid matching bare "/invite" or "/invite/" - must have a token segment
|
||||
assertFalse(RequestUriUtils.isPublicAuthEndpoint("/invite", ""));
|
||||
assertFalse(RequestUriUtils.isPublicAuthEndpoint("/invite/", ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_inviteNestedPathNotPublic() {
|
||||
// Guard against future additions like /invite/<token>/foo becoming accidentally public
|
||||
assertFalse(RequestUriUtils.isPublicAuthEndpoint("/invite/" + INVITE_TOKEN + "/foo", ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_invitePrefixDoesNotOvermatch() {
|
||||
// "/inviteX" must not match the invite pattern
|
||||
assertFalse(RequestUriUtils.isPublicAuthEndpoint("/inviteX", ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_inviteNonUuidTokenNotPublic() {
|
||||
// Only exactly-shaped 36-char lowercase UUID tokens are treated as invite links
|
||||
assertFalse(RequestUriUtils.isPublicAuthEndpoint("/invite/abc123", ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_inviteUppercaseUuidNotPublic() {
|
||||
// Tokens are generated lowercase by UUID.randomUUID().toString()
|
||||
assertFalse(
|
||||
RequestUriUtils.isPublicAuthEndpoint(
|
||||
"/invite/06A20E7E-2E35-4E26-BE7D-2DCE14F28F12", ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_inviteWrongLengthNotPublic() {
|
||||
// 35-char and 37-char UUID-like tokens are not valid UUIDs
|
||||
assertFalse(
|
||||
RequestUriUtils.isPublicAuthEndpoint(
|
||||
"/invite/06a20e7e-2e35-4e26-be7d-2dce14f28f1", ""));
|
||||
assertFalse(
|
||||
RequestUriUtils.isPublicAuthEndpoint(
|
||||
"/invite/06a20e7e-2e35-4e26-be7d-2dce14f28f122", ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_inviteWrongGroupingNotPublic() {
|
||||
// Groups of 8-4-4-4-4 must not be shifted around (e.g. 4-4-4-4-8)
|
||||
assertFalse(
|
||||
RequestUriUtils.isPublicAuthEndpoint(
|
||||
"/invite/06a2-0e7e-2e35-4e26-be7d2dce14f28f12", ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void testIsPublicAuthEndpoint_inviteTokenInvalidCharsNotPublic() {
|
||||
// Hex-only; anything outside [0-9a-f] or the UUID hyphens is rejected
|
||||
assertFalse(RequestUriUtils.isPublicAuthEndpoint("/invite/abc$123", ""));
|
||||
assertFalse(RequestUriUtils.isPublicAuthEndpoint("/invite/abc..123", ""));
|
||||
assertFalse(RequestUriUtils.isPublicAuthEndpoint("/invite/abc%2F123", ""));
|
||||
assertFalse(
|
||||
RequestUriUtils.isPublicAuthEndpoint(
|
||||
"/invite/06a20e7e-2e35-4e26-be7d-2dce14f28f1g", ""));
|
||||
}
|
||||
}
|
||||
|
||||
+155
-16
@@ -11,6 +11,7 @@ import java.time.Instant;
|
||||
import java.time.ZoneId;
|
||||
import java.time.ZonedDateTime;
|
||||
import java.util.*;
|
||||
import java.util.Locale;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
import java.util.stream.Stream;
|
||||
@@ -71,6 +72,7 @@ import org.apache.xmpbox.schema.PDFAIdentificationSchema;
|
||||
import org.apache.xmpbox.schema.XMPBasicSchema;
|
||||
import org.apache.xmpbox.xml.DomXmpParser;
|
||||
import org.apache.xmpbox.xml.XmpSerializer;
|
||||
import org.springframework.beans.factory.annotation.Autowired;
|
||||
import org.springframework.core.io.Resource;
|
||||
import org.springframework.http.HttpStatus;
|
||||
import org.springframework.http.MediaType;
|
||||
@@ -83,7 +85,6 @@ import io.github.pixee.security.Filenames;
|
||||
import io.swagger.v3.oas.annotations.Operation;
|
||||
|
||||
import lombok.Getter;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.SPDF.model.api.converters.PdfToPdfARequest;
|
||||
@@ -93,6 +94,7 @@ import stirling.software.common.configuration.RuntimePathConfig;
|
||||
import stirling.software.common.enumeration.ResourceWeight;
|
||||
import stirling.software.common.model.tool.ToolFormat;
|
||||
import stirling.software.common.model.tool.ToolIO;
|
||||
import stirling.software.common.service.PdfaLevelAServiceInterface;
|
||||
import stirling.software.common.util.ExceptionUtils;
|
||||
import stirling.software.common.util.ProcessExecutor;
|
||||
import stirling.software.common.util.ProcessExecutor.ProcessExecutorResult;
|
||||
@@ -102,14 +104,26 @@ import stirling.software.common.util.WebResponseUtils;
|
||||
|
||||
@ConvertApi
|
||||
@Slf4j
|
||||
@RequiredArgsConstructor
|
||||
public class ConvertPDFToPDFA {
|
||||
|
||||
private static final Pattern NON_PRINTABLE_ASCII = Pattern.compile("[^\\x20-\\x7E]");
|
||||
private final RuntimePathConfig runtimePathConfig;
|
||||
private final stirling.software.SPDF.service.VeraPDFService veraPDFService;
|
||||
// Level A needs the proprietary tagger; core builds convert at level B instead.
|
||||
private final PdfaLevelAServiceInterface pdfaLevelAService;
|
||||
private final TempFileManager tempFileManager;
|
||||
|
||||
public ConvertPDFToPDFA(
|
||||
RuntimePathConfig runtimePathConfig,
|
||||
stirling.software.SPDF.service.VeraPDFService veraPDFService,
|
||||
@Autowired(required = false) PdfaLevelAServiceInterface pdfaLevelAService,
|
||||
TempFileManager tempFileManager) {
|
||||
this.runtimePathConfig = runtimePathConfig;
|
||||
this.veraPDFService = veraPDFService;
|
||||
this.pdfaLevelAService = pdfaLevelAService;
|
||||
this.tempFileManager = tempFileManager;
|
||||
}
|
||||
|
||||
private static final String ICC_RESOURCE_PATH = "/icc/sRGB2014.icc";
|
||||
private static final int PDFA_COMPATIBILITY_POLICY = 1;
|
||||
|
||||
@@ -604,7 +618,10 @@ public class ConvertPDFToPDFA {
|
||||
return handlePdfXConversion(inputFile, outputFormat);
|
||||
} else {
|
||||
return handlePdfAConversion(
|
||||
inputFile, outputFormat, request.getStrict() != null && request.getStrict());
|
||||
inputFile,
|
||||
outputFormat,
|
||||
request.getStrict() != null && request.getStrict(),
|
||||
request.getPdfUa() != null && request.getPdfUa());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1815,8 +1832,64 @@ public class ConvertPDFToPDFA {
|
||||
return Files.readAllBytes(outputPdf);
|
||||
}
|
||||
|
||||
/** Tags a converted PDF/A for level A; must run after Ghostscript, which discards tags. */
|
||||
private PdfaLevelAServiceInterface.Result applyLevelA(
|
||||
byte[] converted,
|
||||
Path original,
|
||||
PdfaProfile profile,
|
||||
String baseFileName,
|
||||
boolean declarePdfUa) {
|
||||
if (!profile.requiresTagging()) {
|
||||
return new PdfaLevelAServiceInterface.Result(converted, true, List.of());
|
||||
}
|
||||
if (pdfaLevelAService == null) {
|
||||
return new PdfaLevelAServiceInterface.Result(
|
||||
converted,
|
||||
false,
|
||||
List.of(
|
||||
"Level A tagging is not available in this build, so the file was left"
|
||||
+ " at conformance level B."));
|
||||
}
|
||||
// Prefer the document's own title/language; hardcoding "en" mislabelled German reports.
|
||||
// Read the original, not the converted bytes: Ghostscript discards /Lang, so probing its
|
||||
// output always yields null and every document would be relabelled with the default.
|
||||
String language = null;
|
||||
String title = null;
|
||||
try (PDDocument probe = Loader.loadPDF(original.toFile())) {
|
||||
language = probe.getDocumentCatalog().getLanguage();
|
||||
title = probe.getDocumentInformation().getTitle();
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not read original title/language: {}", e.getMessage());
|
||||
}
|
||||
if (language == null || language.isBlank()) {
|
||||
try (PDDocument probe = Loader.loadPDF(converted)) {
|
||||
language = probe.getDocumentCatalog().getLanguage();
|
||||
if (title == null || title.isBlank()) {
|
||||
title = probe.getDocumentInformation().getTitle();
|
||||
}
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not read converted title/language: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
PdfaLevelAServiceInterface.Result result =
|
||||
pdfaLevelAService.upgradeToLevelA(
|
||||
converted,
|
||||
profile.getPart(),
|
||||
language,
|
||||
title != null && !title.isBlank() ? title : baseFileName,
|
||||
declarePdfUa);
|
||||
result.warnings().forEach(warning -> log.info("PDF/A level A: {}", warning));
|
||||
if (!result.levelA()) {
|
||||
log.warn(
|
||||
"{} requested but the document could not be tagged; returning level B",
|
||||
profile.getDisplayName());
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
private ResponseEntity<Resource> handlePdfAConversion(
|
||||
MultipartFile inputFile, String outputFormat, boolean strict) throws Exception {
|
||||
MultipartFile inputFile, String outputFormat, boolean strict, boolean declarePdfUa)
|
||||
throws Exception {
|
||||
PdfaProfile profile = PdfaProfile.fromRequest(outputFormat);
|
||||
|
||||
// Get the original filename without extension
|
||||
@@ -1841,12 +1914,15 @@ public class ConvertPDFToPDFA {
|
||||
log.info("Using Ghostscript for PDF/A conversion to {}", profile.getDisplayName());
|
||||
try {
|
||||
converted = convertWithGhostscript(inputPath, workingDir, profile);
|
||||
String outputFilename = baseFileName + profile.outputSuffix();
|
||||
var levelA =
|
||||
applyLevelA(converted, inputPath, profile, baseFileName, declarePdfUa);
|
||||
converted = levelA.pdfBytes();
|
||||
String outputFilename = baseFileName + profile.outputSuffix(levelA.levelA());
|
||||
|
||||
validateAndWarnPdfA(converted, profile, "Ghostscript");
|
||||
|
||||
if (strict) {
|
||||
verifyStrictCompliance(converted);
|
||||
verifyStrictCompliance(converted, profile, levelA.levelA());
|
||||
}
|
||||
|
||||
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
|
||||
@@ -1867,13 +1943,15 @@ public class ConvertPDFToPDFA {
|
||||
}
|
||||
|
||||
converted = convertWithPdfBoxMethod(inputPath, profile);
|
||||
String outputFilename = baseFileName + profile.outputSuffix();
|
||||
var levelA = applyLevelA(converted, inputPath, profile, baseFileName, declarePdfUa);
|
||||
converted = levelA.pdfBytes();
|
||||
String outputFilename = baseFileName + profile.outputSuffix(levelA.levelA());
|
||||
|
||||
// Validate with PDFBox preflight and warn if issues found
|
||||
validateAndWarnPdfA(converted, profile, "PDFBox/LibreOffice");
|
||||
|
||||
if (strict) {
|
||||
verifyStrictCompliance(converted);
|
||||
verifyStrictCompliance(converted, profile, levelA.levelA());
|
||||
}
|
||||
|
||||
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
|
||||
@@ -1889,11 +1967,56 @@ public class ConvertPDFToPDFA {
|
||||
}
|
||||
}
|
||||
|
||||
private void verifyStrictCompliance(byte[] pdfBytes) throws IOException {
|
||||
/** True for a PDF/UA or WCAG result, which says nothing about archival conformance. */
|
||||
private static boolean isAccessibilityProfile(
|
||||
stirling.software.SPDF.model.api.security.PDFVerificationResult result) {
|
||||
String profile = result.getValidationProfile();
|
||||
if (profile == null) {
|
||||
return false;
|
||||
}
|
||||
String normalised = profile.toLowerCase(Locale.ROOT);
|
||||
return normalised.contains("ua") || normalised.contains("wcag");
|
||||
}
|
||||
|
||||
/**
|
||||
* True when a result speaks for the requested profile. Only archival results count, and a level
|
||||
* B pass must never satisfy a level A request.
|
||||
*/
|
||||
private static boolean answersRequest(
|
||||
PdfaProfile profile,
|
||||
stirling.software.SPDF.model.api.security.PDFVerificationResult result) {
|
||||
if (isAccessibilityProfile(result)) {
|
||||
return false;
|
||||
}
|
||||
String standard = result.getStandard();
|
||||
if (standard == null || standard.length() < 2) {
|
||||
return false;
|
||||
}
|
||||
if (standard.charAt(0) != Character.forDigit(profile.getPart(), 10)) {
|
||||
return false;
|
||||
}
|
||||
return !profile.requiresTagging() || Character.toLowerCase(standard.charAt(1)) == 'a';
|
||||
}
|
||||
|
||||
private void verifyStrictCompliance(byte[] pdfBytes, PdfaProfile profile, boolean levelAReached)
|
||||
throws IOException {
|
||||
// Tagging is the only route to level A, so an untagged file cannot answer a strict request.
|
||||
if (!levelAReached) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.BAD_REQUEST,
|
||||
"Strict PDF/A mode enabled: the document could not be tagged, so "
|
||||
+ profile.getDisplayName()
|
||||
+ " was not reached. It is valid at level B.");
|
||||
}
|
||||
try (InputStream is = new ByteArrayInputStream(pdfBytes)) {
|
||||
List<stirling.software.SPDF.model.api.security.PDFVerificationResult> results =
|
||||
veraPDFService.validatePDF(is);
|
||||
boolean isCompliant = results.stream().anyMatch(result -> result.isCompliant());
|
||||
boolean isCompliant =
|
||||
results.stream()
|
||||
.filter(result -> answersRequest(profile, result))
|
||||
.anyMatch(
|
||||
stirling.software.SPDF.model.api.security.PDFVerificationResult
|
||||
::isCompliant);
|
||||
if (!isCompliant) {
|
||||
String details =
|
||||
results.stream()
|
||||
@@ -1901,7 +2024,9 @@ public class ConvertPDFToPDFA {
|
||||
.collect(Collectors.joining("; "));
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.BAD_REQUEST,
|
||||
"Strict PDF/A mode enabled: Conversion is not perfectly compliant. Details: "
|
||||
"Strict PDF/A mode enabled: the output is not perfectly compliant with "
|
||||
+ profile.getDisplayName()
|
||||
+ ". Details: "
|
||||
+ details);
|
||||
}
|
||||
} catch (Exception e) {
|
||||
@@ -2466,11 +2591,16 @@ public class ConvertPDFToPDFA {
|
||||
|
||||
@Getter
|
||||
private enum PdfaProfile {
|
||||
PDF_A_1B(1, "PDF/A-1b", "_PDFA-1b.pdf", "1.4", Format.PDF_A1B, "pdfa-1"),
|
||||
PDF_A_2B(2, "PDF/A-2b", "_PDFA-2b.pdf", "1.7", null, "pdfa", "pdfa-2", "pdfa-2b"),
|
||||
PDF_A_3B(3, "PDF/A-3b", "_PDFA-3b.pdf", "1.7", null, "pdfa-3", "pdfa-3b");
|
||||
PDF_A_1B(1, "B", "PDF/A-1b", "_PDFA-1b.pdf", "1.4", Format.PDF_A1B, "pdfa-1"),
|
||||
PDF_A_2B(2, "B", "PDF/A-2b", "_PDFA-2b.pdf", "1.7", null, "pdfa", "pdfa-2", "pdfa-2b"),
|
||||
PDF_A_3B(3, "B", "PDF/A-3b", "_PDFA-3b.pdf", "1.7", null, "pdfa-3", "pdfa-3b"),
|
||||
// Level A = level B plus tagging, declared language and Unicode text; tagged post-convert.
|
||||
PDF_A_1A(1, "A", "PDF/A-1a", "_PDFA-1a.pdf", "1.4", Format.PDF_A1B, "pdfa-1a"),
|
||||
PDF_A_2A(2, "A", "PDF/A-2a", "_PDFA-2a.pdf", "1.7", null, "pdfa-2a"),
|
||||
PDF_A_3A(3, "A", "PDF/A-3a", "_PDFA-3a.pdf", "1.7", null, "pdfa-3a");
|
||||
|
||||
private final int part;
|
||||
private final String conformanceLevel;
|
||||
private final String displayName;
|
||||
private final String suffix;
|
||||
private final String compatibilityLevel;
|
||||
@@ -2479,12 +2609,14 @@ public class ConvertPDFToPDFA {
|
||||
|
||||
PdfaProfile(
|
||||
int part,
|
||||
String conformanceLevel,
|
||||
String displayName,
|
||||
String suffix,
|
||||
String compatibilityLevel,
|
||||
Format preflightFormat,
|
||||
String... requestTokens) {
|
||||
this.part = part;
|
||||
this.conformanceLevel = conformanceLevel;
|
||||
this.displayName = displayName;
|
||||
this.suffix = suffix;
|
||||
this.compatibilityLevel = compatibilityLevel;
|
||||
@@ -2495,6 +2627,10 @@ public class ConvertPDFToPDFA {
|
||||
.toList();
|
||||
}
|
||||
|
||||
boolean requiresTagging() {
|
||||
return "A".equals(conformanceLevel);
|
||||
}
|
||||
|
||||
static PdfaProfile fromRequest(String requestToken) {
|
||||
if (requestToken == null) {
|
||||
return PDF_A_2B;
|
||||
@@ -2508,8 +2644,11 @@ public class ConvertPDFToPDFA {
|
||||
return match.orElse(PDF_A_2B);
|
||||
}
|
||||
|
||||
String outputSuffix() {
|
||||
return suffix;
|
||||
/**
|
||||
* Names the file at the level actually reached; a level A name over level B content lies.
|
||||
*/
|
||||
String outputSuffix(boolean levelAReached) {
|
||||
return levelAReached ? suffix : "_PDFA-" + part + "b.pdf";
|
||||
}
|
||||
|
||||
Optional<Format> preflightFormat() {
|
||||
|
||||
+11
-1
@@ -14,9 +14,19 @@ public class PdfToPdfARequest extends PDFFile {
|
||||
@Schema(
|
||||
description = "The output format type (PDF/A or PDF/X)",
|
||||
requiredMode = Schema.RequiredMode.REQUIRED,
|
||||
allowableValues = {"pdfa", "pdfa-1", "pdfa-2", "pdfa-2b", "pdfa-3", "pdfa-3b", "pdfx"})
|
||||
allowableValues = {
|
||||
"pdfa", "pdfa-1", "pdfa-2", "pdfa-2b", "pdfa-3", "pdfa-3b", "pdfa-1a", "pdfa-2a",
|
||||
"pdfa-3a", "pdfx"
|
||||
})
|
||||
private String outputFormat;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"Also declare PDF/UA accessibility alongside PDF/A. Only applies to the level A"
|
||||
+ " formats, and the claim is written only if it validates.",
|
||||
defaultValue = "false")
|
||||
private Boolean pdfUa;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"If true, the conversion will fail if the output is not perfectly compliant")
|
||||
|
||||
@@ -285,6 +285,8 @@ public class VeraPDFService {
|
||||
}
|
||||
}
|
||||
|
||||
// Never force PDF/UA here - it flags every ordinary document as non-compliant and doubles
|
||||
// verify cost; /accessibility-report checks PDF/UA on demand.
|
||||
if (!hasPdfaDeclaration) {
|
||||
results.add(createNoPdfaDeclarationResult());
|
||||
}
|
||||
|
||||
@@ -14,6 +14,13 @@
|
||||
"moduleLicense": "GNU Lesser General Public License",
|
||||
"moduleLicenseUrl": "https://www.gnu.org/licenses/old-licenses/lgpl-2.1.html"
|
||||
},
|
||||
{
|
||||
"moduleName": "ch.qos.logback:logback-classic",
|
||||
"moduleUrl": "http://www.qos.ch",
|
||||
"moduleVersion": "1.6.1",
|
||||
"moduleLicense": "LGPL-2.1-only",
|
||||
"moduleLicenseUrl": "https://www.gnu.org/licenses/old-licenses/lgpl-2.1.html"
|
||||
},
|
||||
{
|
||||
"moduleName": "ch.qos.logback:logback-core",
|
||||
"moduleUrl": "http://www.qos.ch",
|
||||
@@ -21,6 +28,13 @@
|
||||
"moduleLicense": "GNU Lesser General Public License",
|
||||
"moduleLicenseUrl": "https://www.gnu.org/licenses/old-licenses/lgpl-2.1.html"
|
||||
},
|
||||
{
|
||||
"moduleName": "ch.qos.logback:logback-core",
|
||||
"moduleUrl": "http://www.qos.ch",
|
||||
"moduleVersion": "1.6.1",
|
||||
"moduleLicense": "LGPL-2.1-only",
|
||||
"moduleLicenseUrl": "https://www.gnu.org/licenses/old-licenses/lgpl-2.1.html"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.adobe.xmp:xmpcore",
|
||||
"moduleUrl": "https://www.adobe.com/devnet/xmp/library/eula-xmp-library-java.html",
|
||||
@@ -182,7 +196,7 @@
|
||||
{
|
||||
"moduleName": "com.github.mwiede:jsch",
|
||||
"moduleUrl": "https://github.com/mwiede/jsch",
|
||||
"moduleVersion": "0.2.23",
|
||||
"moduleVersion": "2.28.6",
|
||||
"moduleLicense": "Revised BSD",
|
||||
"moduleLicenseUrl": "https://github.com/mwiede/jsch/blob/master/LICENSE.txt"
|
||||
},
|
||||
@@ -513,27 +527,45 @@
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.common:common-image",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.common:common-io",
|
||||
"moduleVersion": "3.13.1",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.common:common-io",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.common:common-lang",
|
||||
"moduleVersion": "3.13.1",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.common:common-lang",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.imageio:imageio-batik",
|
||||
"moduleVersion": "3.13.1",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.imageio:imageio-bmp",
|
||||
"moduleVersion": "3.13.1",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
@@ -543,9 +575,15 @@
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.imageio:imageio-core",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.imageio:imageio-jpeg",
|
||||
"moduleVersion": "3.13.1",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
@@ -555,9 +593,15 @@
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.imageio:imageio-metadata",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.imageio:imageio-psd",
|
||||
"moduleVersion": "3.13.1",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
@@ -567,12 +611,24 @@
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.imageio:imageio-tiff",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.imageio:imageio-webp",
|
||||
"moduleVersion": "3.13.1",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.twelvemonkeys.imageio:imageio-webp",
|
||||
"moduleVersion": "3.14.0",
|
||||
"moduleLicense": "The BSD License",
|
||||
"moduleLicenseUrl": "https://github.com/haraldk/TwelveMonkeys#license"
|
||||
},
|
||||
{
|
||||
"moduleName": "com.vladsch.flexmark:flexmark",
|
||||
"moduleVersion": "0.64.8",
|
||||
@@ -758,7 +814,7 @@
|
||||
{
|
||||
"moduleName": "commons-net:commons-net",
|
||||
"moduleUrl": "https://commons.apache.org/proper/commons-net/",
|
||||
"moduleVersion": "3.11.1",
|
||||
"moduleVersion": "3.13.0",
|
||||
"moduleLicense": "Apache-2.0",
|
||||
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
@@ -1213,24 +1269,48 @@
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "net.shibboleth:shib-networking",
|
||||
"moduleVersion": "9.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "net.shibboleth:shib-security",
|
||||
"moduleVersion": "9.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "net.shibboleth:shib-security",
|
||||
"moduleVersion": "9.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "net.shibboleth:shib-support",
|
||||
"moduleVersion": "9.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "net.shibboleth:shib-support",
|
||||
"moduleVersion": "9.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "net.shibboleth:shib-velocity",
|
||||
"moduleVersion": "9.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "net.shibboleth:shib-velocity",
|
||||
"moduleVersion": "9.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.antlr:antlr4-runtime",
|
||||
"moduleUrl": "https://www.antlr.org/",
|
||||
@@ -1325,13 +1405,6 @@
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.apache.httpcomponents:httpclient",
|
||||
"moduleUrl": "http://hc.apache.org/httpcomponents-client",
|
||||
"moduleVersion": "4.5.13",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.apache.httpcomponents:httpclient",
|
||||
"moduleUrl": "http://hc.apache.org/httpcomponents-client-ga",
|
||||
@@ -1456,6 +1529,13 @@
|
||||
"moduleLicense": "Apache-2.0",
|
||||
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.apache.santuario:xmlsec",
|
||||
"moduleUrl": "https://www.apache.org/",
|
||||
"moduleVersion": "3.0.6",
|
||||
"moduleLicense": "Apache-2.0",
|
||||
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.apache.tomcat.embed:tomcat-embed-el",
|
||||
"moduleUrl": "https://tomcat.apache.org/",
|
||||
@@ -1470,6 +1550,13 @@
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.apache.velocity:velocity-engine-core",
|
||||
"moduleUrl": "https://www.apache.org/",
|
||||
"moduleVersion": "2.4.1",
|
||||
"moduleLicense": "Apache-2.0",
|
||||
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.apache.xmlbeans:xmlbeans",
|
||||
"moduleUrl": "https://xmlbeans.apache.org/",
|
||||
@@ -1647,6 +1734,13 @@
|
||||
"moduleLicense": "GNU Lesser General Public License",
|
||||
"moduleLicenseUrl": "http://www.gnu.org/licenses/lgpl-3.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.cryptacular:cryptacular",
|
||||
"moduleUrl": "https://www.cryptacular.org",
|
||||
"moduleVersion": "1.3.0",
|
||||
"moduleLicense": "GNU Lesser General Public License",
|
||||
"moduleLicenseUrl": "https://www.gnu.org/licenses/lgpl-3.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.eclipse.angus:angus-activation",
|
||||
"moduleUrl": "https://www.eclipse.org",
|
||||
@@ -2016,78 +2110,156 @@
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-core-api",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-core-impl",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-core-impl",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-messaging-api",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-messaging-api",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-profile-api",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-profile-api",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-saml-api",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-saml-api",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-saml-impl",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-saml-impl",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-security-api",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-security-api",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-security-impl",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-security-impl",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-soap-api",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-soap-api",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-soap-impl",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-soap-impl",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-storage-api",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-storage-api",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-xmlsec-api",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-xmlsec-api",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-xmlsec-impl",
|
||||
"moduleVersion": "5.1.6",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.opensaml:opensaml-xmlsec-impl",
|
||||
"moduleVersion": "5.2.2",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.ow2.asm:asm",
|
||||
"moduleUrl": "http://asm.ow2.org",
|
||||
@@ -2132,7 +2304,7 @@
|
||||
},
|
||||
{
|
||||
"moduleName": "org.simplejavamail:core-module",
|
||||
"moduleVersion": "9.2.0",
|
||||
"moduleVersion": "9.3.1",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
@@ -2145,13 +2317,13 @@
|
||||
},
|
||||
{
|
||||
"moduleName": "org.simplejavamail:outlook-module",
|
||||
"moduleVersion": "9.2.0",
|
||||
"moduleVersion": "9.3.1",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.simplejavamail:simple-java-mail",
|
||||
"moduleVersion": "9.2.0",
|
||||
"moduleVersion": "9.3.1",
|
||||
"moduleLicense": "The Apache Software License, Version 2.0",
|
||||
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
|
||||
},
|
||||
@@ -2564,6 +2736,13 @@
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.springframework.security:spring-security-core",
|
||||
"moduleUrl": "https://spring.io/projects/spring-security",
|
||||
"moduleVersion": "7.1.0",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.springframework.security:spring-security-crypto",
|
||||
"moduleUrl": "https://spring.io/projects/spring-security",
|
||||
@@ -2606,6 +2785,13 @@
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.springframework.security:spring-security-saml2-service-provider",
|
||||
"moduleUrl": "https://spring.io/projects/spring-security",
|
||||
"moduleVersion": "7.1.0",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "org.springframework.security:spring-security-web",
|
||||
"moduleUrl": "https://spring.io/projects/spring-security",
|
||||
@@ -2805,207 +2991,207 @@
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:annotations",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:apache-client",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleName": "software.amazon.awssdk:apache5-client",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:arns",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:auth",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:aws-core",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:aws-query-protocol",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:aws-xml-protocol",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:checksums",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:checksums-spi",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:crt-core",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:endpoints-spi",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:http-auth",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:http-auth-aws",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:http-auth-aws-eventstream",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:http-auth-spi",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:http-client-spi",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:identity-spi",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:json-utils",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:metrics-spi",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:netty-nio-client",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:profiles",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:protocol-core",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:regions",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:retries",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:retries-spi",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:s3",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:sdk-core",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:third-party-jackson-core",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:url-connection-client",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:utils",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
{
|
||||
"moduleName": "software.amazon.awssdk:utils-lite",
|
||||
"moduleUrl": "https://aws.amazon.com/sdkforjava",
|
||||
"moduleVersion": "2.44.12",
|
||||
"moduleVersion": "2.51.3",
|
||||
"moduleLicense": "Apache License, Version 2.0",
|
||||
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
|
||||
},
|
||||
|
||||
+98
-38
@@ -46,6 +46,7 @@ import stirling.software.SPDF.model.api.converters.PdfToPdfARequest;
|
||||
import stirling.software.SPDF.model.api.security.PDFVerificationResult;
|
||||
import stirling.software.SPDF.service.VeraPDFService;
|
||||
import stirling.software.common.configuration.RuntimePathConfig;
|
||||
import stirling.software.common.service.PdfaLevelAServiceInterface;
|
||||
import stirling.software.common.util.TempFileManager;
|
||||
|
||||
/**
|
||||
@@ -62,10 +63,12 @@ class ConvertPDFToPDFAGapTest {
|
||||
|
||||
@Mock private RuntimePathConfig runtimePathConfig;
|
||||
@Mock private VeraPDFService veraPDFService;
|
||||
@Mock private PdfaLevelAServiceInterface pdfaLevelAService;
|
||||
@Mock private TempFileManager tempFileManager;
|
||||
|
||||
private ConvertPDFToPDFA newController() {
|
||||
return new ConvertPDFToPDFA(runtimePathConfig, veraPDFService, tempFileManager);
|
||||
return new ConvertPDFToPDFA(
|
||||
runtimePathConfig, veraPDFService, pdfaLevelAService, tempFileManager);
|
||||
}
|
||||
|
||||
// ---- reflection helpers ----------------------------------------------------------------
|
||||
@@ -161,9 +164,21 @@ class ConvertPDFToPDFAGapTest {
|
||||
}
|
||||
|
||||
private String suffixOf(Object profile) throws Exception {
|
||||
Method m = profile.getClass().getDeclaredMethod("outputSuffix");
|
||||
return suffixOf(profile, true);
|
||||
}
|
||||
|
||||
private String suffixOf(Object profile, boolean levelAReached) throws Exception {
|
||||
Method m = profile.getClass().getDeclaredMethod("outputSuffix", boolean.class);
|
||||
m.setAccessible(true);
|
||||
return (String) m.invoke(profile);
|
||||
return (String) m.invoke(profile, levelAReached);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a level A profile falls back to the level B name when tagging failed")
|
||||
void levelANotReachedIsNamedLevelB() throws Exception {
|
||||
assertThat(suffixOf(resolveProfile("pdfa-1a"), false)).isEqualTo("_PDFA-1b.pdf");
|
||||
assertThat(suffixOf(resolveProfile("pdfa-2a"), false)).isEqualTo("_PDFA-2b.pdf");
|
||||
assertThat(suffixOf(resolveProfile("pdfa-3a"), true)).isEqualTo("_PDFA-3a.pdf");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -717,6 +732,30 @@ class ConvertPDFToPDFAGapTest {
|
||||
@DisplayName("verifyStrictCompliance (VeraPDFService mocked)")
|
||||
class StrictCompliance {
|
||||
|
||||
private Object profile(String token) throws Exception {
|
||||
Class<?> enumClass = null;
|
||||
for (Class<?> inner : ConvertPDFToPDFA.class.getDeclaredClasses()) {
|
||||
if (inner.getSimpleName().equals("PdfaProfile")) {
|
||||
enumClass = inner;
|
||||
}
|
||||
}
|
||||
Method m = enumClass.getDeclaredMethod("fromRequest", String.class);
|
||||
m.setAccessible(true);
|
||||
return m.invoke(null, token);
|
||||
}
|
||||
|
||||
private Throwable verify(String token, boolean levelAReached) throws Exception {
|
||||
ConvertPDFToPDFA controller = newController();
|
||||
return catchThrowable(
|
||||
() ->
|
||||
invokeInstance(
|
||||
controller,
|
||||
"verifyStrictCompliance",
|
||||
(Object) "dummy".getBytes(),
|
||||
profile(token),
|
||||
levelAReached));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("compliant result passes without throwing")
|
||||
void compliantPasses() throws Exception {
|
||||
@@ -726,14 +765,7 @@ class ConvertPDFToPDFAGapTest {
|
||||
ok.setComplianceSummary("PDF/A-1b compliant");
|
||||
when(veraPDFService.validatePDF(any())).thenReturn(List.of(ok));
|
||||
|
||||
ConvertPDFToPDFA controller = newController();
|
||||
assertThatCode(
|
||||
() ->
|
||||
invokeInstance(
|
||||
controller,
|
||||
"verifyStrictCompliance",
|
||||
(Object) "dummy".getBytes()))
|
||||
.doesNotThrowAnyException();
|
||||
assertThat(verify("pdfa-1", true)).isNull();
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -745,34 +777,70 @@ class ConvertPDFToPDFAGapTest {
|
||||
bad.setComplianceSummary("PDF/A-1b with errors");
|
||||
when(veraPDFService.validatePDF(any())).thenReturn(List.of(bad));
|
||||
|
||||
ConvertPDFToPDFA controller = newController();
|
||||
ResponseStatusException ex =
|
||||
(ResponseStatusException)
|
||||
catchThrowable(
|
||||
() ->
|
||||
invokeInstance(
|
||||
controller,
|
||||
"verifyStrictCompliance",
|
||||
(Object) "dummy".getBytes()));
|
||||
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-1", true);
|
||||
assertThat(ex).isNotNull();
|
||||
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
|
||||
assertThat(ex.getReason()).contains("PDF/A-1b with errors");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a level B pass does not satisfy a level A request")
|
||||
void levelBDoesNotSatisfyLevelA() throws Exception {
|
||||
PDFVerificationResult ok = new PDFVerificationResult();
|
||||
ok.setCompliant(true);
|
||||
ok.setStandard("1b");
|
||||
ok.setComplianceSummary("PDF/A-1b compliant");
|
||||
when(veraPDFService.validatePDF(any())).thenReturn(List.of(ok));
|
||||
|
||||
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-1a", true);
|
||||
assertThat(ex).isNotNull();
|
||||
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
|
||||
assertThat(ex.getReason()).contains("PDF/A-1a");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a level A result satisfies a level A request")
|
||||
void levelASatisfiesLevelA() throws Exception {
|
||||
PDFVerificationResult ok = new PDFVerificationResult();
|
||||
ok.setCompliant(true);
|
||||
ok.setStandard("2a");
|
||||
ok.setComplianceSummary("PDF/A-2a compliant");
|
||||
when(veraPDFService.validatePDF(any())).thenReturn(List.of(ok));
|
||||
|
||||
assertThat(verify("pdfa-2a", true)).isNull();
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("untagged output fails a level A request before validation runs")
|
||||
void untaggedLevelARequestFails() throws Exception {
|
||||
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-2a", false);
|
||||
assertThat(ex).isNotNull();
|
||||
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
|
||||
assertThat(ex.getReason()).contains("could not be tagged");
|
||||
verifyNoInteractions(veraPDFService);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a compliant PDF/UA result never satisfies a strict PDF/A request")
|
||||
void accessibilityResultIsIgnored() throws Exception {
|
||||
PDFVerificationResult ua = new PDFVerificationResult();
|
||||
ua.setCompliant(true);
|
||||
ua.setStandard("ua1");
|
||||
ua.setValidationProfile("ua1");
|
||||
ua.setComplianceSummary("PDF/UA-1 compliant");
|
||||
when(veraPDFService.validatePDF(any())).thenReturn(List.of(ua));
|
||||
|
||||
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-2b", true);
|
||||
assertThat(ex).isNotNull();
|
||||
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("empty result list is treated as non-compliant -> 400")
|
||||
void emptyResultsTreatedNonCompliant() throws Exception {
|
||||
when(veraPDFService.validatePDF(any())).thenReturn(Collections.emptyList());
|
||||
|
||||
ConvertPDFToPDFA controller = newController();
|
||||
ResponseStatusException ex =
|
||||
(ResponseStatusException)
|
||||
catchThrowable(
|
||||
() ->
|
||||
invokeInstance(
|
||||
controller,
|
||||
"verifyStrictCompliance",
|
||||
(Object) "dummy".getBytes()));
|
||||
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-1", true);
|
||||
assertThat(ex).isNotNull();
|
||||
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
|
||||
}
|
||||
@@ -782,15 +850,7 @@ class ConvertPDFToPDFAGapTest {
|
||||
void serviceErrorWrappedAs500() throws Exception {
|
||||
when(veraPDFService.validatePDF(any())).thenThrow(new IOException("boom"));
|
||||
|
||||
ConvertPDFToPDFA controller = newController();
|
||||
ResponseStatusException ex =
|
||||
(ResponseStatusException)
|
||||
catchThrowable(
|
||||
() ->
|
||||
invokeInstance(
|
||||
controller,
|
||||
"verifyStrictCompliance",
|
||||
(Object) "dummy".getBytes()));
|
||||
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-1", true);
|
||||
assertThat(ex).isNotNull();
|
||||
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.INTERNAL_SERVER_ERROR);
|
||||
}
|
||||
|
||||
+4
-1
@@ -42,6 +42,7 @@ import org.springframework.mock.web.MockMultipartFile;
|
||||
import stirling.software.SPDF.model.api.converters.PdfToPdfARequest;
|
||||
import stirling.software.SPDF.service.VeraPDFService;
|
||||
import stirling.software.common.configuration.RuntimePathConfig;
|
||||
import stirling.software.common.service.PdfaLevelAServiceInterface;
|
||||
import stirling.software.common.util.ProcessExecutor;
|
||||
import stirling.software.common.util.ProcessExecutor.ProcessExecutorResult;
|
||||
import stirling.software.common.util.TempFile;
|
||||
@@ -63,10 +64,12 @@ class ConvertPDFToPDFAMoreTest {
|
||||
|
||||
@Mock private RuntimePathConfig runtimePathConfig;
|
||||
@Mock private VeraPDFService veraPDFService;
|
||||
@Mock private PdfaLevelAServiceInterface pdfaLevelAService;
|
||||
@Mock private TempFileManager tempFileManager;
|
||||
|
||||
private ConvertPDFToPDFA newController() {
|
||||
return new ConvertPDFToPDFA(runtimePathConfig, veraPDFService, tempFileManager);
|
||||
return new ConvertPDFToPDFA(
|
||||
runtimePathConfig, veraPDFService, pdfaLevelAService, tempFileManager);
|
||||
}
|
||||
|
||||
private static ResponseEntity<Resource> streamingOk(byte[] bytes) {
|
||||
|
||||
+14
-3
@@ -90,7 +90,9 @@ class VeraPDFServicePdfaFixtureTest {
|
||||
() -> service.validatePDF(new ByteArrayInputStream(pdfBytes)),
|
||||
"Empty veraPDF flavour list must not surface as IndexOutOfBoundsException");
|
||||
|
||||
assertEquals(1, results.size());
|
||||
// One result: PDF/UA is checked by the dedicated accessibility-report endpoint, not here.
|
||||
assertEquals(1, results.size(), () -> "Expected a single PDF/A result, got: " + results);
|
||||
|
||||
PDFVerificationResult result = results.get(0);
|
||||
assertEquals("not-pdfa", result.getStandard());
|
||||
assertFalse(result.isDeclaredPdfa());
|
||||
@@ -161,13 +163,22 @@ class VeraPDFServicePdfaFixtureTest {
|
||||
}
|
||||
}
|
||||
|
||||
/** The PDF/A result; every document is also checked against PDF/UA, so filter that one out. */
|
||||
private PDFVerificationResult onlyResult(byte[] pdfBytes) throws Exception {
|
||||
List<PDFVerificationResult> results =
|
||||
service.validatePDF(new ByteArrayInputStream(pdfBytes));
|
||||
|
||||
assertNotNull(results);
|
||||
assertEquals(1, results.size(), () -> "Expected a single result, got: " + results);
|
||||
return results.get(0);
|
||||
List<PDFVerificationResult> pdfaResults =
|
||||
results.stream().filter(r -> !isUaResult(r)).toList();
|
||||
assertEquals(
|
||||
1, pdfaResults.size(), () -> "Expected a single PDF/A result, got: " + results);
|
||||
return pdfaResults.get(0);
|
||||
}
|
||||
|
||||
private static boolean isUaResult(PDFVerificationResult result) {
|
||||
String profile = result.getValidationProfile();
|
||||
return profile != null && profile.toLowerCase().contains("ua");
|
||||
}
|
||||
|
||||
private static String messages(PDFVerificationResult result) {
|
||||
|
||||
@@ -27,7 +27,7 @@ dependencies {
|
||||
api 'org.springframework.boot:spring-boot-starter-cache'
|
||||
api 'com.github.ben-manes.caffeine:caffeine'
|
||||
implementation 'org.springframework.boot:spring-boot-starter-data-redis'
|
||||
api 'io.swagger.core.v3:swagger-core-jakarta:2.2.46'
|
||||
api 'io.swagger.core.v3:swagger-core-jakarta:2.2.53'
|
||||
implementation "com.bucket4j:bucket4j_jdk17-core:${bucket4jVersion}"
|
||||
// Lettuce-backed Bucket4j ProxyManager used by ValkeyRateLimitStore for cluster-wide
|
||||
// token-bucket rate limiting (parity with in-process Bucket4j semantics; no fixed-window
|
||||
@@ -37,6 +37,15 @@ dependencies {
|
||||
// https://mvnrepository.com/artifact/com.bucket4j/bucket4j_jdk17
|
||||
implementation "org.bouncycastle:bcprov-jdk18on:$bouncycastleVersion"
|
||||
|
||||
// PDF/UA tagging and its validation oracle.
|
||||
implementation 'org.verapdf:validation-model:1.30.2'
|
||||
// CVE-2025-66453: Explicit rhino 1.7.15 to override verapdf's 1.7.13
|
||||
implementation "org.mozilla:rhino:${rhinoVersion}"
|
||||
// veraPDF still uses javax.xml.bind, not the new jakarta namespace
|
||||
implementation 'javax.xml.bind:jaxb-api:2.3.1'
|
||||
runtimeOnly 'com.sun.xml.bind:jaxb-impl:2.3.9'
|
||||
runtimeOnly 'com.sun.xml.bind:jaxb-core:4.0.9'
|
||||
|
||||
implementation "com.google.code.gson:gson:${gsonVersion}"
|
||||
|
||||
// jinjava/jjwt transitively request older Jackson 2 versions; declare the current
|
||||
|
||||
+176
@@ -0,0 +1,176 @@
|
||||
package stirling.software.proprietary.controller.api.converters;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import org.springframework.core.io.Resource;
|
||||
import org.springframework.http.MediaType;
|
||||
import org.springframework.http.ResponseEntity;
|
||||
import org.springframework.web.bind.annotation.ModelAttribute;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
|
||||
import io.github.pixee.security.Filenames;
|
||||
import io.swagger.v3.oas.annotations.Operation;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.common.annotations.AutoJobPostMapping;
|
||||
import stirling.software.common.annotations.api.ConvertApi;
|
||||
import stirling.software.common.enumeration.ResourceWeight;
|
||||
import stirling.software.common.model.tool.ToolFormat;
|
||||
import stirling.software.common.model.tool.ToolIO;
|
||||
import stirling.software.common.util.ExceptionUtils;
|
||||
import stirling.software.common.util.TempFile;
|
||||
import stirling.software.common.util.TempFileManager;
|
||||
import stirling.software.common.util.WebResponseUtils;
|
||||
import stirling.software.proprietary.model.api.converters.PdfToPdfUaRequest;
|
||||
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
|
||||
import stirling.software.proprietary.pdf.ua.TaggingOptions;
|
||||
import stirling.software.proprietary.service.ua.PdfUaConversionService;
|
||||
|
||||
/** Converts a PDF to PDF/UA; response headers say whether the result actually conforms. */
|
||||
@ConvertApi
|
||||
@Slf4j
|
||||
@RequiredArgsConstructor
|
||||
public class ConvertPdfToPdfUa {
|
||||
|
||||
private static final String HEADER_DECLARED = "X-Stirling-UA-Declared";
|
||||
private static final String HEADER_FAILURES = "X-Stirling-UA-Failures";
|
||||
private static final String HEADER_ALT_NEEDED = "X-Stirling-UA-Figures-Needing-Alt";
|
||||
private static final String HEADER_WARNINGS = "X-Stirling-UA-Warnings";
|
||||
|
||||
/** Any line ending, so descriptions pasted from any platform parse the same. */
|
||||
private static final Pattern NEWLINE = Pattern.compile("\\R");
|
||||
|
||||
private final PdfUaConversionService conversionService;
|
||||
private final TempFileManager tempFileManager;
|
||||
|
||||
@AutoJobPostMapping(
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
value = "/pdf/ua",
|
||||
resourceWeight = ResourceWeight.LARGE_WEIGHT)
|
||||
@ToolIO(produces = ToolFormat.PDF)
|
||||
@Operation(
|
||||
summary = "Convert a PDF to PDF/UA-1 or PDF/UA-2",
|
||||
description =
|
||||
"Tags the document, marks decorative content as artifacts, embeds fonts and"
|
||||
+ " applies the document-level requirements of PDF/UA, then validates"
|
||||
+ " the result. A conformance declaration is written only if validation"
|
||||
+ " passes, so the returned file never claims more than it delivers.")
|
||||
public ResponseEntity<Resource> pdfToPdfUa(@ModelAttribute PdfToPdfUaRequest request)
|
||||
throws IOException {
|
||||
|
||||
MultipartFile input = request.getFileInput();
|
||||
if (input == null || input.isEmpty()) {
|
||||
throw ExceptionUtils.createPdfFileRequiredException();
|
||||
}
|
||||
|
||||
String originalName = Filenames.toSimpleFileName(input.getOriginalFilename());
|
||||
String stem = stripExtension(originalName == null ? "document" : originalName);
|
||||
PdfUaProfile profile = PdfUaProfile.fromRequest(request.getProfile());
|
||||
|
||||
TaggingOptions options =
|
||||
TaggingOptions.builder()
|
||||
.profile(profile)
|
||||
.title(request.getTitle())
|
||||
.fallbackTitle(stem)
|
||||
// Only used when the document declares no language of its own.
|
||||
.language(
|
||||
request.getLanguage() == null || request.getLanguage().isBlank()
|
||||
? "en-GB"
|
||||
: request.getLanguage())
|
||||
.overrideLanguage(
|
||||
request.getOverrideLanguage() != null
|
||||
&& request.getOverrideLanguage())
|
||||
.existingTags(existingTags(request.getExistingTags()))
|
||||
.figurePolicy(figurePolicy(request.getFigurePolicy()))
|
||||
.embedFonts(request.getEmbedFonts() == null || request.getEmbedFonts())
|
||||
.altTextByFigure(parseAltText(request.getAltText()))
|
||||
.build();
|
||||
|
||||
PdfUaConversionOutcome outcome = conversionService.convert(input.getBytes(), options);
|
||||
|
||||
log.info(
|
||||
"Converted '{}' to {}: declared={}, {} remaining failure(s)",
|
||||
originalName,
|
||||
profile.displayName(),
|
||||
outcome.declared(),
|
||||
outcome.validation().totalFailures());
|
||||
|
||||
outcome.warnings().forEach(warning -> log.info("PDF/UA warning: {}", warning));
|
||||
|
||||
// Streamed from a temp file so a large conversion does not hold a second heap copy.
|
||||
String suffix = outcome.declared() ? "_pdfua" + profile.part() : "_tagged";
|
||||
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
|
||||
try {
|
||||
Files.write(tempOut.getPath(), outcome.pdfBytes());
|
||||
} catch (IOException e) {
|
||||
tempOut.close();
|
||||
throw e;
|
||||
}
|
||||
ResponseEntity<Resource> response =
|
||||
WebResponseUtils.pdfFileToWebResponse(tempOut, stem + suffix + ".pdf");
|
||||
|
||||
return ResponseEntity.status(response.getStatusCode())
|
||||
.headers(response.getHeaders())
|
||||
.header(HEADER_DECLARED, String.valueOf(outcome.declared()))
|
||||
.header(HEADER_FAILURES, String.valueOf(outcome.validation().totalFailures()))
|
||||
.header(
|
||||
HEADER_ALT_NEEDED,
|
||||
String.valueOf(outcome.tagging().figuresNeedingAltText()))
|
||||
// Count only: warning text is multi-line prose, which HTTP headers mangle.
|
||||
.header(HEADER_WARNINGS, String.valueOf(outcome.warnings().size()))
|
||||
.body(response.getBody());
|
||||
}
|
||||
|
||||
/**
|
||||
* Parses newline-separated {@code key=description} pairs, keyed as the report hands them out.
|
||||
* Only the first "=" splits, since a description may contain one.
|
||||
*/
|
||||
public static Map<String, String> parseAltText(String raw) {
|
||||
if (raw == null || raw.isBlank()) {
|
||||
return Map.of();
|
||||
}
|
||||
Map<String, String> parsed = new LinkedHashMap<>();
|
||||
for (String line : NEWLINE.split(raw)) {
|
||||
int split = line.indexOf('=');
|
||||
if (split <= 0) {
|
||||
continue;
|
||||
}
|
||||
String key = line.substring(0, split).strip();
|
||||
String description = line.substring(split + 1).strip();
|
||||
if (!key.isEmpty() && !description.isEmpty()) {
|
||||
parsed.put(key, description);
|
||||
}
|
||||
}
|
||||
return parsed;
|
||||
}
|
||||
|
||||
private static TaggingOptions.ExistingTags existingTags(String value) {
|
||||
if (value == null) {
|
||||
return TaggingOptions.ExistingTags.AUTO;
|
||||
}
|
||||
return switch (value.trim().toLowerCase()) {
|
||||
case "keep" -> TaggingOptions.ExistingTags.KEEP;
|
||||
case "rebuild" -> TaggingOptions.ExistingTags.REBUILD;
|
||||
default -> TaggingOptions.ExistingTags.AUTO;
|
||||
};
|
||||
}
|
||||
|
||||
private static TaggingOptions.FigurePolicy figurePolicy(String value) {
|
||||
if (value != null && value.trim().equalsIgnoreCase("mark-decorative")) {
|
||||
return TaggingOptions.FigurePolicy.MARK_DECORATIVE;
|
||||
}
|
||||
return TaggingOptions.FigurePolicy.REQUIRE_ALT;
|
||||
}
|
||||
|
||||
private static String stripExtension(String filename) {
|
||||
int dot = filename.lastIndexOf('.');
|
||||
return dot > 0 ? filename.substring(0, dot) : filename;
|
||||
}
|
||||
}
|
||||
+67
@@ -0,0 +1,67 @@
|
||||
package stirling.software.proprietary.controller.api.security;
|
||||
|
||||
import java.io.IOException;
|
||||
|
||||
import org.springframework.http.MediaType;
|
||||
import org.springframework.http.ResponseEntity;
|
||||
import org.springframework.web.bind.annotation.ModelAttribute;
|
||||
import org.springframework.web.multipart.MultipartFile;
|
||||
|
||||
import io.swagger.v3.oas.annotations.Operation;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.common.annotations.AutoJobPostMapping;
|
||||
import stirling.software.common.annotations.api.SecurityApi;
|
||||
import stirling.software.common.enumeration.ResourceWeight;
|
||||
import stirling.software.common.model.tool.ToolFormat;
|
||||
import stirling.software.common.model.tool.ToolIO;
|
||||
import stirling.software.common.util.ExceptionUtils;
|
||||
import stirling.software.proprietary.model.api.ua.AccessibilityReport;
|
||||
import stirling.software.proprietary.model.api.ua.AccessibilityReportRequest;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
|
||||
import stirling.software.proprietary.service.ua.AccessibilityAuditService;
|
||||
|
||||
/** Reports how accessible a document is, without modifying it. */
|
||||
@SecurityApi
|
||||
@RequiredArgsConstructor
|
||||
@Slf4j
|
||||
public class AccessibilityReportController {
|
||||
|
||||
private final AccessibilityAuditService auditService;
|
||||
|
||||
@ToolIO(produces = ToolFormat.JSON)
|
||||
@Operation(
|
||||
summary = "Report a document's accessibility standing",
|
||||
description =
|
||||
"Validates the document against PDF/UA and reports what fails, which failures"
|
||||
+ " can be fixed automatically, and which checks still need a person."
|
||||
+ " Does not modify the file.")
|
||||
// Costs a full veraPDF pass plus the converter's own layout analysis over every page.
|
||||
@AutoJobPostMapping(
|
||||
value = "/accessibility-report",
|
||||
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
|
||||
resourceWeight = ResourceWeight.LARGE_WEIGHT)
|
||||
public ResponseEntity<AccessibilityReport> report(
|
||||
@ModelAttribute AccessibilityReportRequest request) {
|
||||
|
||||
MultipartFile file = request.getFileInput();
|
||||
if (file == null || file.isEmpty()) {
|
||||
throw ExceptionUtils.createPdfFileRequiredException();
|
||||
}
|
||||
PdfUaProfile profile = PdfUaProfile.fromRequest(request.getProfile());
|
||||
try {
|
||||
AccessibilityReport report = auditService.audit(file.getBytes(), profile);
|
||||
log.info(
|
||||
"Accessibility report for '{}': tagged={}, {} issue(s)",
|
||||
file.getOriginalFilename(),
|
||||
report.isTagged(),
|
||||
report.getIssues().size());
|
||||
return ResponseEntity.ok(report);
|
||||
} catch (IOException e) {
|
||||
throw ExceptionUtils.createRuntimeException(
|
||||
"error.ioException", "Could not read the PDF: {0}", e, e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
+73
@@ -0,0 +1,73 @@
|
||||
package stirling.software.proprietary.model.api.converters;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class PdfToPdfUaRequest extends PDFFile {
|
||||
|
||||
@Schema(
|
||||
description = "PDF/UA conformance level to target",
|
||||
defaultValue = "ua1",
|
||||
allowableValues = {"ua1", "ua2"})
|
||||
private String profile;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"Document title, required by PDF/UA. Falls back to the first heading, then the"
|
||||
+ " filename.")
|
||||
private String title;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"Document language as a BCP-47 tag, for example en-GB. Applied only when the"
|
||||
+ " document does not already declare one, unless overrideLanguage is"
|
||||
+ " set.",
|
||||
defaultValue = "en-GB")
|
||||
private String language;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"Replace the language the document already declares. Off by default, so a"
|
||||
+ " document is never relabelled into a language it is not written in.",
|
||||
defaultValue = "false")
|
||||
private Boolean overrideLanguage;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"What to do with an existing structure tree: keep it, rebuild it, or decide"
|
||||
+ " automatically",
|
||||
defaultValue = "auto",
|
||||
allowableValues = {"auto", "keep", "rebuild"})
|
||||
private String existingTags;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"How to treat images with no description. require-alt leaves them undescribed so"
|
||||
+ " the report asks for input; mark-decorative treats every image as"
|
||||
+ " decoration.",
|
||||
defaultValue = "require-alt",
|
||||
allowableValues = {"require-alt", "mark-decorative"})
|
||||
private String figurePolicy;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"Embed fonts the document references but does not carry. Required for"
|
||||
+ " conformance and needs Ghostscript.",
|
||||
defaultValue = "true")
|
||||
private Boolean embedFonts;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"Alternative descriptions for figures, as key=text pairs separated by newlines."
|
||||
+ " Keys come from the accessibility-report endpoint's"
|
||||
+ " figuresNeedingDescription list, for example \"0:12=Bar chart of"
|
||||
+ " quarterly revenue\". Descriptions are never invented, so without"
|
||||
+ " these an illustrated document cannot claim conformance.")
|
||||
private String altText;
|
||||
}
|
||||
+38
@@ -0,0 +1,38 @@
|
||||
package stirling.software.proprietary.model.api.ua;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
|
||||
/** One accessibility problem, grouped across all of its occurrences. */
|
||||
@Data
|
||||
@Schema(description = "A single accessibility issue found in a document")
|
||||
public class AccessibilityIssue {
|
||||
|
||||
@Schema(description = "ISO 14289 clause, e.g. 7.3")
|
||||
private String clause;
|
||||
|
||||
@Schema(description = "Test number within the clause")
|
||||
private String testNumber;
|
||||
|
||||
@Schema(description = "Plain-English description of the problem")
|
||||
private String message;
|
||||
|
||||
@Schema(description = "The validator's own wording, for support and debugging")
|
||||
private String technicalMessage;
|
||||
|
||||
@Schema(description = "error or warning")
|
||||
private String severity = "error";
|
||||
|
||||
@Schema(description = "Standard the check came from, e.g. PDF/UA-1")
|
||||
private String specification;
|
||||
|
||||
@Schema(description = "Where the problem was found, when the validator reports it")
|
||||
private String location;
|
||||
|
||||
@Schema(description = "How many times this issue occurs")
|
||||
private int occurrences;
|
||||
|
||||
@Schema(description = "True when the converter can fix this without human input")
|
||||
private boolean autoFixable;
|
||||
}
|
||||
+63
@@ -0,0 +1,63 @@
|
||||
package stirling.software.proprietary.model.api.ua;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
|
||||
/**
|
||||
* A document's accessibility standing. The machine/human split is load-bearing: veraPDF covers only
|
||||
* about half of the Matterhorn Protocol, so a clean automated pass is not "accessible".
|
||||
*/
|
||||
@Data
|
||||
@Schema(description = "Accessibility standing of a document")
|
||||
public class AccessibilityReport {
|
||||
|
||||
@Schema(description = "Profile the document was checked against, e.g. PDF/UA-1")
|
||||
private String profile;
|
||||
|
||||
@Schema(description = "Whether the document has a structure tree at all")
|
||||
private boolean tagged;
|
||||
|
||||
@Schema(description = "Whether the document declares PDF/UA conformance in its metadata")
|
||||
private boolean declaresConformance;
|
||||
|
||||
@Schema(description = "Whether every automated check passed")
|
||||
private boolean passesAutomatedChecks;
|
||||
|
||||
@Schema(description = "Automated checks that failed, grouped by rule")
|
||||
private List<AccessibilityIssue> issues = List.of();
|
||||
|
||||
@Schema(description = "Things a person still has to verify; automation cannot decide these")
|
||||
private List<String> humanChecks = List.of();
|
||||
|
||||
@Schema(description = "How many of the failing checks the converter can fix on its own")
|
||||
private int automaticallyFixable;
|
||||
|
||||
@Schema(description = "How many need information from the user, such as alternative text")
|
||||
private int needsInput;
|
||||
|
||||
@Schema(
|
||||
description =
|
||||
"Figures that need an alternative description. Each carries the key to pass"
|
||||
+ " back in the conversion request's altTextByFigure map, so a caller"
|
||||
+ " can enumerate what is missing and then supply it.")
|
||||
private List<FigureDescriptor> figuresNeedingDescription = List.of();
|
||||
|
||||
@Schema(description = "Document-level facts that drive most failures")
|
||||
private Summary summary = new Summary();
|
||||
|
||||
@Data
|
||||
@Schema(description = "Quick document-level facts")
|
||||
public static class Summary {
|
||||
private int pages;
|
||||
private boolean hasTitle;
|
||||
private boolean displaysDocTitle;
|
||||
private boolean hasLanguage;
|
||||
private boolean allFontsEmbedded;
|
||||
private int unembeddedFonts;
|
||||
private int figures;
|
||||
private boolean encrypted;
|
||||
}
|
||||
}
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
package stirling.software.proprietary.model.api.ua;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
import lombok.Data;
|
||||
import lombok.EqualsAndHashCode;
|
||||
|
||||
import stirling.software.common.model.api.PDFFile;
|
||||
|
||||
@Data
|
||||
@EqualsAndHashCode(callSuper = true)
|
||||
public class AccessibilityReportRequest extends PDFFile {
|
||||
|
||||
@Schema(
|
||||
description = "Profile to check against",
|
||||
defaultValue = "ua1",
|
||||
allowableValues = {"ua1", "ua2"})
|
||||
private String profile;
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
package stirling.software.proprietary.model.api.ua;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
/**
|
||||
* One figure needing an alternative description, which is never invented. key is the
|
||||
* altTextByFigure key "pageIndex:ordinal"; page is 1-based; kind is "figure" or "formula".
|
||||
*/
|
||||
@Schema(description = "A figure that needs an alternative description")
|
||||
public record FigureDescriptor(
|
||||
String key,
|
||||
int page,
|
||||
String kind,
|
||||
float x,
|
||||
float y,
|
||||
float width,
|
||||
float height,
|
||||
String existingAlt) {}
|
||||
+26
@@ -0,0 +1,26 @@
|
||||
package stirling.software.proprietary.model.api.ua;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
/**
|
||||
* Result of a PDF/UA conversion.
|
||||
*
|
||||
* @param declared whether a {@code pdfuaid} conformance claim was written into {@code pdfBytes}
|
||||
*/
|
||||
@Schema(description = "Result of converting a document to PDF/UA")
|
||||
public record PdfUaConversionOutcome(
|
||||
byte[] pdfBytes,
|
||||
boolean declared,
|
||||
UaValidationResult validation,
|
||||
TaggingSummary tagging,
|
||||
List<String> warnings) {
|
||||
|
||||
@Schema(description = "What the tagging pass produced")
|
||||
public record TaggingSummary(
|
||||
boolean rebuiltStructure,
|
||||
int taggedElements,
|
||||
int artifacts,
|
||||
int figuresNeedingAltText) {}
|
||||
}
|
||||
+18
@@ -0,0 +1,18 @@
|
||||
package stirling.software.proprietary.model.api.ua;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
import io.swagger.v3.oas.annotations.media.Schema;
|
||||
|
||||
/**
|
||||
* Outcome of validating against one PDF/UA profile. compliant means every automated check passed,
|
||||
* which is not the same as usable by assistive technology; totalFailures is ungrouped.
|
||||
*/
|
||||
@Schema(description = "Result of validating a document against a PDF/UA profile")
|
||||
public record UaValidationResult(
|
||||
String profile, boolean compliant, List<AccessibilityIssue> issues, int totalFailures) {
|
||||
|
||||
public boolean hasIssues() {
|
||||
return !issues.isEmpty();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
/** Artifact subtypes (ISO 32000-1 14.8.2.2). Artifacts are excluded from the structure tree. */
|
||||
public enum ArtifactType {
|
||||
/** Running heads, folios, page numbers. Required by PDF/UA-1 clause 7.8. */
|
||||
PAGINATION("Pagination"),
|
||||
/** Rules, boxes, and other layout ornamentation. */
|
||||
LAYOUT("Layout"),
|
||||
/** Cut marks and colour bars. */
|
||||
PAGE("Page"),
|
||||
/** Background graphics with no informational content. */
|
||||
BACKGROUND("Background");
|
||||
|
||||
private final String subtype;
|
||||
|
||||
ArtifactType(String subtype) {
|
||||
this.subtype = subtype;
|
||||
}
|
||||
|
||||
public String subtype() {
|
||||
return subtype;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
/** An axis-aligned rectangle in PDF user space, with y increasing upwards. */
|
||||
public record BBox(float x0, float y0, float x1, float y1) {
|
||||
|
||||
public static final BBox EMPTY = new BBox(0, 0, 0, 0);
|
||||
|
||||
public static BBox of(float x, float y, float width, float height) {
|
||||
return new BBox(x, y, x + width, y + height);
|
||||
}
|
||||
|
||||
public float width() {
|
||||
return x1 - x0;
|
||||
}
|
||||
|
||||
public float height() {
|
||||
return y1 - y0;
|
||||
}
|
||||
|
||||
public float centreX() {
|
||||
return (x0 + x1) / 2f;
|
||||
}
|
||||
|
||||
public BBox union(BBox other) {
|
||||
if (other == null || other.isEmpty()) {
|
||||
return this;
|
||||
}
|
||||
if (isEmpty()) {
|
||||
return other;
|
||||
}
|
||||
return new BBox(
|
||||
Math.min(x0, other.x0),
|
||||
Math.min(y0, other.y0),
|
||||
Math.max(x1, other.x1),
|
||||
Math.max(y1, other.y1));
|
||||
}
|
||||
|
||||
public boolean isEmpty() {
|
||||
return x1 <= x0 || y1 <= y0;
|
||||
}
|
||||
|
||||
/** Horizontal overlap with another box as a fraction of the narrower box's width. */
|
||||
public float horizontalOverlap(BBox other) {
|
||||
float overlap = Math.min(x1, other.x1) - Math.max(x0, other.x0);
|
||||
float narrower = Math.min(width(), other.width());
|
||||
return narrower <= 0 ? 0 : Math.max(0, overlap) / narrower;
|
||||
}
|
||||
}
|
||||
+84
@@ -0,0 +1,84 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.function.Consumer;
|
||||
|
||||
import lombok.Getter;
|
||||
import lombok.Setter;
|
||||
|
||||
/** The derived logical structure of a document, ready for serialisation into a structure tree. */
|
||||
@Getter
|
||||
@Setter
|
||||
public class DocumentStructure {
|
||||
|
||||
/** Top-level blocks in document reading order. */
|
||||
private final List<StructBlock> blocks = new ArrayList<>();
|
||||
|
||||
/** Warnings raised during analysis, surfaced in the conversion report. */
|
||||
private final List<String> warnings = new ArrayList<>();
|
||||
|
||||
private String title;
|
||||
private String language;
|
||||
|
||||
/** True when real text was wrapped as artifacts, which blocks any conformance claim. */
|
||||
private boolean textSuppressed;
|
||||
|
||||
/** Body text size used as the baseline for heading detection, in points. */
|
||||
private float bodyFontSize;
|
||||
|
||||
public void add(StructBlock block) {
|
||||
blocks.add(block);
|
||||
}
|
||||
|
||||
public void warn(String message) {
|
||||
if (!warnings.contains(message)) {
|
||||
warnings.add(message);
|
||||
}
|
||||
}
|
||||
|
||||
public void visit(Consumer<StructBlock> visitor) {
|
||||
blocks.forEach(block -> block.visit(visitor));
|
||||
}
|
||||
|
||||
public int count(StructType type) {
|
||||
int[] total = {0};
|
||||
visit(
|
||||
block -> {
|
||||
if (block.getType() == type) {
|
||||
total[0]++;
|
||||
}
|
||||
});
|
||||
return total[0];
|
||||
}
|
||||
|
||||
public int artifactCount() {
|
||||
int[] total = {0};
|
||||
visit(
|
||||
block -> {
|
||||
if (block.isArtifact()) {
|
||||
total[0]++;
|
||||
}
|
||||
});
|
||||
return total[0];
|
||||
}
|
||||
|
||||
/** Figures with no alternative description, the most common PDF/UA failure. */
|
||||
public List<StructBlock> figuresWithoutAlt() {
|
||||
List<StructBlock> missing = new ArrayList<>();
|
||||
visit(
|
||||
block -> {
|
||||
if ((block.getType() == StructType.FIGURE
|
||||
|| block.getType() == StructType.FORMULA)
|
||||
&& (block.getAlt() == null || block.getAlt().isBlank())
|
||||
&& (block.getActualText() == null || block.getActualText().isBlank())) {
|
||||
missing.add(block);
|
||||
}
|
||||
});
|
||||
return missing;
|
||||
}
|
||||
|
||||
public boolean isEmpty() {
|
||||
return blocks.isEmpty();
|
||||
}
|
||||
}
|
||||
+831
@@ -0,0 +1,831 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.Comparator;
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.IdentityHashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
/**
|
||||
* Derives a logical structure from extracted lines and graphics, reusing {@code HeadingDetector}'s
|
||||
* heuristics. Degrades to paragraphs rather than guessing, since a wrong tag misleads readers.
|
||||
*/
|
||||
@Slf4j
|
||||
public class LayoutAnalyzer {
|
||||
|
||||
private static final Pattern BULLET = Pattern.compile("^[•‣◦⁃∙·▪●■o\\-\\*\\+]\\s+.*");
|
||||
private static final Pattern ORDERED =
|
||||
Pattern.compile("^(\\d{1,3}|[a-zA-Z]|[ivxlcIVXLC]{1,5})[\\.\\)]\\s+.*");
|
||||
private static final Pattern PAGE_NUMBER =
|
||||
Pattern.compile(
|
||||
"^(page\\s+)?\\d{1,4}(\\s*(of|/)\\s*\\d{1,4})?$", Pattern.CASE_INSENSITIVE);
|
||||
private static final Pattern DIGITS = Pattern.compile("\\d+");
|
||||
|
||||
/** Fraction of page height treated as the running head / foot band. */
|
||||
private static final float MARGIN_BAND = 0.10f;
|
||||
|
||||
/** A line must exceed the body size by this ratio before it can be a heading. */
|
||||
private static final float HEADING_RATIO = 1.10f;
|
||||
|
||||
/** Sizes within this many points are treated as the same heading tier. */
|
||||
private static final float TIER_TOLERANCE = 0.4f;
|
||||
|
||||
private static final int MAX_HEADING_WORDS = 12;
|
||||
|
||||
/** Word gap beyond this multiple of the font size separates table cells. */
|
||||
private static final float CELL_GAP_RATIO = 1.2f;
|
||||
|
||||
/** Images smaller than this in either dimension are decoration, not content. */
|
||||
private static final float MIN_FIGURE_SIZE = 12f;
|
||||
|
||||
/** A size used by more than this share of lines is body text, however large the median says. */
|
||||
private static final float MAX_HEADING_LINE_SHARE = 0.2f;
|
||||
|
||||
/** Consecutive lines sharing a size are a text block; headings appear alone. */
|
||||
private static final int MAX_HEADING_RUN = 3;
|
||||
|
||||
/** A vector thinner than this in either dimension is a rule or border, not a drawing. */
|
||||
private static final float MIN_VECTOR_THICKNESS = 3f;
|
||||
|
||||
/** Vector clusters smaller than this are ornament; larger ones are probably a chart. */
|
||||
private static final float MIN_VECTOR_FIGURE_SIZE = 40f;
|
||||
|
||||
/** A drawing is built from several strokes; one big rectangle is a panel, not a chart. */
|
||||
private static final int MIN_VECTOR_FIGURE_OPS = 4;
|
||||
|
||||
/** More text than this inside the region means shading behind content, not a drawing. */
|
||||
private static final int MAX_LINES_INSIDE_FIGURE = 2;
|
||||
|
||||
public DocumentStructure analyse(List<PageContent> pages) {
|
||||
DocumentStructure structure = new DocumentStructure();
|
||||
float bodySize = bodyFontSize(pages);
|
||||
structure.setBodyFontSize(bodySize);
|
||||
Map<Float, Integer> tiers = headingTiers(pages, bodySize);
|
||||
Map<Integer, List<TextLineInfo>> artifactLines = repeatedMarginLines(pages, bodySize);
|
||||
|
||||
for (PageContent page : pages) {
|
||||
analysePage(
|
||||
page,
|
||||
structure,
|
||||
bodySize,
|
||||
tiers,
|
||||
artifactLines.getOrDefault(page.pageIndex(), List.of()));
|
||||
}
|
||||
|
||||
List<Integer> suppressedPages =
|
||||
pages.stream()
|
||||
.filter(PageContent::linesDropped)
|
||||
.map(PageContent::pageIndex)
|
||||
.toList();
|
||||
if (!suppressedPages.isEmpty()) {
|
||||
structure.setTextSuppressed(true);
|
||||
structure.warn(
|
||||
"Text on page(s) "
|
||||
+ suppressedPages.stream()
|
||||
.map(i -> String.valueOf(i + 1))
|
||||
.collect(Collectors.joining(", "))
|
||||
+ " could not be tagged reliably and was marked as artifacts. The"
|
||||
+ " converter will not claim conformance while real text is hidden"
|
||||
+ " from assistive technology.");
|
||||
}
|
||||
|
||||
normaliseHeadingLevels(structure);
|
||||
structure.setTitle(deriveTitle(structure));
|
||||
return structure;
|
||||
}
|
||||
|
||||
// --- Document-wide statistics -----------------------------------------
|
||||
|
||||
/** Character-weighted median line size, which is far more stable than a plain median. */
|
||||
static float bodyFontSize(List<PageContent> pages) {
|
||||
Map<Float, Integer> weights = new HashMap<>();
|
||||
for (PageContent page : pages) {
|
||||
for (TextLineInfo line : page.lines()) {
|
||||
if (line.dominantFontSize() > 0 && !line.isBlank()) {
|
||||
weights.merge(line.dominantFontSize(), line.charCount(), Integer::sum);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (weights.isEmpty()) {
|
||||
return 0f;
|
||||
}
|
||||
int total = weights.values().stream().mapToInt(Integer::intValue).sum();
|
||||
List<Map.Entry<Float, Integer>> sorted =
|
||||
weights.entrySet().stream().sorted(Map.Entry.comparingByKey()).toList();
|
||||
int seen = 0;
|
||||
for (Map.Entry<Float, Integer> entry : sorted) {
|
||||
seen += entry.getValue();
|
||||
if (seen >= total / 2) {
|
||||
return entry.getKey();
|
||||
}
|
||||
}
|
||||
return sorted.get(sorted.size() - 1).getKey();
|
||||
}
|
||||
|
||||
/** Maps each distinct heading size to a 1-based level, largest size first. */
|
||||
static Map<Float, Integer> headingTiers(List<PageContent> pages, float bodySize) {
|
||||
if (bodySize <= 0) {
|
||||
return Map.of();
|
||||
}
|
||||
// A size used by a large share of the lines is body text, whatever the median says.
|
||||
Map<Float, Integer> lineCounts = new HashMap<>();
|
||||
int totalLines = 0;
|
||||
for (PageContent page : pages) {
|
||||
for (TextLineInfo line : page.lines()) {
|
||||
if (!line.isBlank()) {
|
||||
lineCounts.merge(line.dominantFontSize(), 1, Integer::sum);
|
||||
totalLines++;
|
||||
}
|
||||
}
|
||||
}
|
||||
int headingLineCeiling = Math.max(1, (int) (totalLines * MAX_HEADING_LINE_SHARE));
|
||||
|
||||
// Headings do not cluster; a run of same-size lines is a text block, not headings.
|
||||
Map<Float, Integer> longestRun = new HashMap<>();
|
||||
for (PageContent page : pages) {
|
||||
Float runSize = null;
|
||||
int runLength = 0;
|
||||
for (TextLineInfo line : page.lines()) {
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
float size = line.dominantFontSize();
|
||||
if (runSize != null && Float.compare(size, runSize) == 0) {
|
||||
runLength++;
|
||||
} else {
|
||||
runSize = size;
|
||||
runLength = 1;
|
||||
}
|
||||
int seen = longestRun.getOrDefault(size, 0);
|
||||
if (runLength > seen) {
|
||||
longestRun.put(size, runLength);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
List<Float> sizes = new ArrayList<>();
|
||||
for (PageContent page : pages) {
|
||||
for (TextLineInfo line : page.lines()) {
|
||||
if (isHeadingCandidate(line)
|
||||
&& line.dominantFontSize() > bodySize * HEADING_RATIO
|
||||
&& lineCounts.getOrDefault(line.dominantFontSize(), 0) <= headingLineCeiling
|
||||
&& longestRun.getOrDefault(line.dominantFontSize(), 0) < MAX_HEADING_RUN) {
|
||||
sizes.add(line.dominantFontSize());
|
||||
}
|
||||
}
|
||||
}
|
||||
List<Float> distinct = sizes.stream().distinct().sorted(Comparator.reverseOrder()).toList();
|
||||
|
||||
Map<Float, Integer> tiers = new LinkedHashMap<>();
|
||||
int level = 0;
|
||||
Float previous = null;
|
||||
for (Float size : distinct) {
|
||||
if (previous == null || previous - size > TIER_TOLERANCE) {
|
||||
level = Math.min(level + 1, 6);
|
||||
previous = size;
|
||||
}
|
||||
tiers.put(size, level);
|
||||
}
|
||||
return tiers;
|
||||
}
|
||||
|
||||
/**
|
||||
* Claims a line's operators word run by word run; claiming the whole ordinal interval would
|
||||
* swallow anything drawn between them, an image included.
|
||||
*/
|
||||
private static void claimLine(StructBlock block, TextLineInfo line) {
|
||||
// Sort by ordinal, not position: merging out-of-order runs silently drops them to
|
||||
// /Artifact, hiding them from assistive technology while the file still validates.
|
||||
List<WordInfo> words =
|
||||
line.words().stream()
|
||||
.filter(w -> !w.isBlank())
|
||||
.sorted(Comparator.comparingInt(WordInfo::startOrdinal))
|
||||
.toList();
|
||||
if (words.isEmpty()) {
|
||||
block.addRange(line.startOrdinal(), line.endOrdinal());
|
||||
return;
|
||||
}
|
||||
int start = words.get(0).startOrdinal();
|
||||
int end = words.get(0).endOrdinal();
|
||||
for (int i = 1; i < words.size(); i++) {
|
||||
WordInfo word = words.get(i);
|
||||
if (word.startOrdinal() <= end + 1) {
|
||||
end = Math.max(end, word.endOrdinal());
|
||||
} else {
|
||||
block.addRange(start, end);
|
||||
start = word.startOrdinal();
|
||||
end = word.endOrdinal();
|
||||
}
|
||||
}
|
||||
block.addRange(start, end);
|
||||
}
|
||||
|
||||
static boolean isHeadingCandidate(TextLineInfo line) {
|
||||
String text = line.text().strip();
|
||||
if (text.isEmpty() || line.wordCount() > MAX_HEADING_WORDS) {
|
||||
return false;
|
||||
}
|
||||
char last = text.charAt(text.length() - 1);
|
||||
return last != '.' && last != '!' && last != '?';
|
||||
}
|
||||
|
||||
/**
|
||||
* Finds lines in the head/foot bands whose text repeats across pages. Digits are masked first
|
||||
* so that "Page 4" and "Page 5" count as the same running foot.
|
||||
*/
|
||||
static Map<Integer, List<TextLineInfo>> repeatedMarginLines(List<PageContent> pages) {
|
||||
return repeatedMarginLines(pages, bodyFontSize(pages));
|
||||
}
|
||||
|
||||
static Map<Integer, List<TextLineInfo>> repeatedMarginLines(
|
||||
List<PageContent> pages, float bodySize) {
|
||||
Map<Integer, List<TextLineInfo>> result = new HashMap<>();
|
||||
if (pages.isEmpty()) {
|
||||
return result;
|
||||
}
|
||||
Map<String, Integer> counts = new HashMap<>();
|
||||
Map<Integer, List<TextLineInfo>> candidates = new HashMap<>();
|
||||
|
||||
for (PageContent page : pages) {
|
||||
float height = page.mediaBox().height();
|
||||
if (height <= 0) {
|
||||
continue;
|
||||
}
|
||||
float topEdge = page.mediaBox().y1() - height * MARGIN_BAND;
|
||||
float bottomEdge = page.mediaBox().y0() + height * MARGIN_BAND;
|
||||
List<TextLineInfo> inBand = new ArrayList<>();
|
||||
for (TextLineInfo line : page.lines()) {
|
||||
if (line.bbox().y0() >= topEdge || line.bbox().y1() <= bottomEdge) {
|
||||
inBand.add(line);
|
||||
counts.merge(mask(line.text()), 1, Integer::sum);
|
||||
}
|
||||
}
|
||||
candidates.put(page.pageIndex(), inBand);
|
||||
}
|
||||
|
||||
int threshold = Math.max(2, pages.size() / 2);
|
||||
for (Map.Entry<Integer, List<TextLineInfo>> entry : candidates.entrySet()) {
|
||||
List<TextLineInfo> artifacts = new ArrayList<>();
|
||||
for (TextLineInfo line : entry.getValue()) {
|
||||
boolean repeats =
|
||||
pages.size() >= 3 && counts.getOrDefault(mask(line.text()), 0) >= threshold;
|
||||
boolean pageNumber = PAGE_NUMBER.matcher(line.text().strip()).matches();
|
||||
// Masked digits merge "Section 1" and "Section 2"; size is the tie-break that stops
|
||||
// a real heading being demoted, as running heads are never larger than body text.
|
||||
boolean looksLikeChrome =
|
||||
bodySize <= 0 || line.dominantFontSize() <= bodySize * 1.05f;
|
||||
if (pageNumber || (repeats && looksLikeChrome)) {
|
||||
artifacts.add(line);
|
||||
}
|
||||
}
|
||||
result.put(entry.getKey(), artifacts);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
private static String mask(String text) {
|
||||
return DIGITS.matcher(text.strip().toLowerCase()).replaceAll("#").replaceAll("\\s+", " ");
|
||||
}
|
||||
|
||||
// --- Per-page analysis -------------------------------------------------
|
||||
|
||||
private void analysePage(
|
||||
PageContent page,
|
||||
DocumentStructure structure,
|
||||
float bodySize,
|
||||
Map<Float, Integer> tiers,
|
||||
List<TextLineInfo> marginArtifacts) {
|
||||
|
||||
for (TextLineInfo line : marginArtifacts) {
|
||||
StructBlock artifact = StructBlock.artifact(ArtifactType.PAGINATION, page.pageIndex());
|
||||
claimLine(artifact, line);
|
||||
artifact.setBbox(line.bbox());
|
||||
artifact.setText(line.text());
|
||||
structure.add(artifact);
|
||||
}
|
||||
|
||||
// Identity set, not List.contains: TextLineInfo is a record whose equals walks its word
|
||||
// list, so a linear scan per line is quadratic with a deep comparison inside it.
|
||||
java.util.Set<TextLineInfo> marginSet = Collections.newSetFromMap(new IdentityHashMap<>());
|
||||
marginSet.addAll(marginArtifacts);
|
||||
List<TextLineInfo> body =
|
||||
page.lines().stream()
|
||||
.filter(line -> !line.isBlank() && !marginSet.contains(line))
|
||||
.sorted(readingOrder(page))
|
||||
.toList();
|
||||
|
||||
List<StructBlock> blocks = new ArrayList<>();
|
||||
int index = 0;
|
||||
while (index < body.size()) {
|
||||
TextLineInfo line = body.get(index);
|
||||
|
||||
int tableEnd = tableRunEnd(body, index);
|
||||
if (tableEnd > index) {
|
||||
StructBlock table = buildTable(body.subList(index, tableEnd + 1), page.pageIndex());
|
||||
if (table != null) {
|
||||
blocks.add(table);
|
||||
index = tableEnd + 1;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
int listEnd = listRunEnd(body, index);
|
||||
if (listEnd > index) {
|
||||
blocks.add(buildList(body.subList(index, listEnd + 1), page.pageIndex()));
|
||||
index = listEnd + 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
Integer level = headingLevel(line, tiers);
|
||||
if (level != null) {
|
||||
StructBlock heading = new StructBlock(StructType.heading(level), page.pageIndex());
|
||||
claimLine(heading, line);
|
||||
heading.setBbox(line.bbox());
|
||||
heading.setText(line.text());
|
||||
blocks.add(heading);
|
||||
index++;
|
||||
continue;
|
||||
}
|
||||
|
||||
int paragraphEnd = paragraphRunEnd(body, index, tiers, bodySize);
|
||||
blocks.add(buildParagraph(body.subList(index, paragraphEnd + 1), page.pageIndex()));
|
||||
index = paragraphEnd + 1;
|
||||
}
|
||||
|
||||
// Form XObject text is attributed to its Do, so a Figure too would double-claim it.
|
||||
Set<Integer> claimed = new HashSet<>();
|
||||
for (StructBlock block : blocks) {
|
||||
block.visit(
|
||||
node ->
|
||||
node.getRanges()
|
||||
.forEach(
|
||||
range -> {
|
||||
for (int i = range.start(); i <= range.end(); i++) {
|
||||
claimed.add(i);
|
||||
}
|
||||
}));
|
||||
}
|
||||
blocks.addAll(buildGraphics(page, structure, claimed));
|
||||
blocks.forEach(structure::add);
|
||||
}
|
||||
|
||||
/**
|
||||
* Orders lines top-to-bottom, splitting into columns first when the page is clearly
|
||||
* multi-column. Without this, a two-column page reads as interleaved half-sentences.
|
||||
*/
|
||||
private Comparator<TextLineInfo> readingOrder(PageContent page) {
|
||||
Float gutter = detectGutter(page);
|
||||
if (gutter == null) {
|
||||
return Comparator.comparingDouble((TextLineInfo l) -> -l.bbox().y1())
|
||||
.thenComparingDouble(l -> l.bbox().x0());
|
||||
}
|
||||
return Comparator.comparingInt((TextLineInfo l) -> l.bbox().centreX() < gutter ? 0 : 1)
|
||||
.thenComparingDouble(l -> -l.bbox().y1())
|
||||
.thenComparingDouble(l -> l.bbox().x0());
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns the x of a vertical gutter when the page is two-column, else null. A gutter must sit
|
||||
* near the middle, be crossed by almost no line, and have substantial text on both sides.
|
||||
*/
|
||||
static Float detectGutter(PageContent page) {
|
||||
List<TextLineInfo> lines = page.lines().stream().filter(line -> !line.isBlank()).toList();
|
||||
if (lines.size() < 8) {
|
||||
return null;
|
||||
}
|
||||
float pageWidth = page.mediaBox().width();
|
||||
if (pageWidth <= 0) {
|
||||
return null;
|
||||
}
|
||||
float centre = page.mediaBox().x0() + pageWidth / 2f;
|
||||
long crossing =
|
||||
lines.stream()
|
||||
.filter(
|
||||
line ->
|
||||
line.bbox().x0() < centre - 5
|
||||
&& line.bbox().x1() > centre + 5)
|
||||
.count();
|
||||
if (crossing > lines.size() * 0.1) {
|
||||
return null;
|
||||
}
|
||||
long left = lines.stream().filter(line -> line.bbox().centreX() < centre).count();
|
||||
long right = lines.size() - left;
|
||||
boolean balanced = left > lines.size() * 0.25 && right > lines.size() * 0.25;
|
||||
return balanced ? centre : null;
|
||||
}
|
||||
|
||||
private static Integer headingLevel(TextLineInfo line, Map<Float, Integer> tiers) {
|
||||
if (!isHeadingCandidate(line)) {
|
||||
return null;
|
||||
}
|
||||
return tiers.get(line.dominantFontSize());
|
||||
}
|
||||
|
||||
// --- Paragraphs --------------------------------------------------------
|
||||
|
||||
private static int paragraphRunEnd(
|
||||
List<TextLineInfo> lines, int start, Map<Float, Integer> tiers, float bodySize) {
|
||||
int end = start;
|
||||
for (int i = start + 1; i < lines.size(); i++) {
|
||||
TextLineInfo previous = lines.get(i - 1);
|
||||
TextLineInfo current = lines.get(i);
|
||||
if (headingLevel(current, tiers) != null || startsListItem(current)) {
|
||||
break;
|
||||
}
|
||||
float gap = previous.bbox().y0() - current.bbox().y1();
|
||||
float leading = Math.max(bodySize, current.bbox().height());
|
||||
boolean sameBlock = gap < leading * 0.8f && gap > -leading;
|
||||
boolean sentenceEnded = endsSentence(previous.text());
|
||||
if (!sameBlock || (sentenceEnded && gap > leading * 0.4f)) {
|
||||
break;
|
||||
}
|
||||
end = i;
|
||||
}
|
||||
return end;
|
||||
}
|
||||
|
||||
private static boolean endsSentence(String text) {
|
||||
String stripped = text.strip();
|
||||
if (stripped.isEmpty()) {
|
||||
return false;
|
||||
}
|
||||
char last = stripped.charAt(stripped.length() - 1);
|
||||
return last == '.' || last == '!' || last == '?';
|
||||
}
|
||||
|
||||
private static StructBlock buildParagraph(List<TextLineInfo> lines, int pageIndex) {
|
||||
StructBlock paragraph = new StructBlock(StructType.P, pageIndex);
|
||||
BBox box = BBox.EMPTY;
|
||||
StringBuilder text = new StringBuilder();
|
||||
for (TextLineInfo line : lines) {
|
||||
claimLine(paragraph, line);
|
||||
box = box.union(line.bbox());
|
||||
if (text.length() > 0) {
|
||||
text.append(' ');
|
||||
}
|
||||
text.append(line.text().strip());
|
||||
}
|
||||
paragraph.setBbox(box);
|
||||
paragraph.setText(text.toString());
|
||||
return paragraph;
|
||||
}
|
||||
|
||||
// --- Lists -------------------------------------------------------------
|
||||
|
||||
static boolean startsListItem(TextLineInfo line) {
|
||||
String text = line.text().strip();
|
||||
return BULLET.matcher(text).matches() || ORDERED.matcher(text).matches();
|
||||
}
|
||||
|
||||
private static int listRunEnd(List<TextLineInfo> lines, int start) {
|
||||
if (!startsListItem(lines.get(start))) {
|
||||
return start;
|
||||
}
|
||||
float indent = lines.get(start).bbox().x0();
|
||||
int end = start;
|
||||
for (int i = start + 1; i < lines.size(); i++) {
|
||||
TextLineInfo line = lines.get(i);
|
||||
boolean isItem = startsListItem(line) && Math.abs(line.bbox().x0() - indent) < 6f;
|
||||
boolean isContinuation = !startsListItem(line) && line.bbox().x0() > indent + 2f;
|
||||
if (!isItem && !isContinuation) {
|
||||
break;
|
||||
}
|
||||
end = i;
|
||||
}
|
||||
// A single marker is a stray character, not a list.
|
||||
long items =
|
||||
lines.subList(start, end + 1).stream()
|
||||
.filter(LayoutAnalyzer::startsListItem)
|
||||
.count();
|
||||
return items >= 2 ? end : start;
|
||||
}
|
||||
|
||||
private static StructBlock buildList(List<TextLineInfo> lines, int pageIndex) {
|
||||
StructBlock list = new StructBlock(StructType.L, pageIndex);
|
||||
list.setListNumbering(listNumbering(lines.get(0)));
|
||||
BBox box = BBox.EMPTY;
|
||||
StructBlock currentBody = null;
|
||||
|
||||
for (TextLineInfo line : lines) {
|
||||
box = box.union(line.bbox());
|
||||
if (startsListItem(line) || currentBody == null) {
|
||||
StructBlock item = new StructBlock(StructType.LI, pageIndex);
|
||||
StructBlock body = new StructBlock(StructType.LBODY, pageIndex);
|
||||
claimLine(body, line);
|
||||
body.setBbox(line.bbox());
|
||||
body.setText(line.text());
|
||||
item.addChild(body);
|
||||
item.setBbox(line.bbox());
|
||||
list.addChild(item);
|
||||
currentBody = body;
|
||||
} else {
|
||||
claimLine(currentBody, line);
|
||||
currentBody.setBbox(currentBody.getBbox().union(line.bbox()));
|
||||
currentBody.setText(currentBody.getText() + " " + line.text().strip());
|
||||
}
|
||||
}
|
||||
list.setBbox(box);
|
||||
return list;
|
||||
}
|
||||
|
||||
private static String listNumbering(TextLineInfo first) {
|
||||
String text = first.text().strip();
|
||||
if (BULLET.matcher(text).matches()) {
|
||||
return "Disc";
|
||||
}
|
||||
char c = text.charAt(0);
|
||||
if (Character.isDigit(c)) {
|
||||
return "Decimal";
|
||||
}
|
||||
if ("ivxlc".indexOf(Character.toLowerCase(c)) >= 0 && text.length() > 1) {
|
||||
return Character.isUpperCase(c) ? "UpperRoman" : "LowerRoman";
|
||||
}
|
||||
return Character.isUpperCase(c) ? "UpperAlpha" : "LowerAlpha";
|
||||
}
|
||||
|
||||
// --- Tables ------------------------------------------------------------
|
||||
|
||||
/** Splits a line into cells wherever the gap between words exceeds the cell threshold. */
|
||||
static List<List<WordInfo>> splitCells(TextLineInfo line) {
|
||||
List<WordInfo> words = line.words().stream().filter(w -> !w.isBlank()).toList();
|
||||
List<List<WordInfo>> cells = new ArrayList<>();
|
||||
if (words.isEmpty()) {
|
||||
return cells;
|
||||
}
|
||||
float threshold = Math.max(line.dominantFontSize(), 1f) * CELL_GAP_RATIO;
|
||||
List<WordInfo> current = new ArrayList<>();
|
||||
current.add(words.get(0));
|
||||
for (int i = 1; i < words.size(); i++) {
|
||||
float gap = words.get(i).bbox().x0() - words.get(i - 1).bbox().x1();
|
||||
if (gap > threshold) {
|
||||
cells.add(List.copyOf(current));
|
||||
current = new ArrayList<>();
|
||||
}
|
||||
current.add(words.get(i));
|
||||
}
|
||||
cells.add(List.copyOf(current));
|
||||
return cells;
|
||||
}
|
||||
|
||||
/**
|
||||
* Index of the last line of a table run starting at {@code start}, or {@code start} if none.
|
||||
*/
|
||||
private static int tableRunEnd(List<TextLineInfo> lines, int start) {
|
||||
int end = start;
|
||||
for (int i = start; i < lines.size(); i++) {
|
||||
if (splitCells(lines.get(i)).size() < 2) {
|
||||
break;
|
||||
}
|
||||
end = i;
|
||||
}
|
||||
return end > start ? end : start;
|
||||
}
|
||||
|
||||
/**
|
||||
* Builds a Table when the run really looks tabular and each cell owns its own operators.
|
||||
* Returns null when it does not, so the caller falls back to paragraphs.
|
||||
*/
|
||||
private static StructBlock buildTable(List<TextLineInfo> rows, int pageIndex) {
|
||||
if (rows.size() < 2) {
|
||||
return null;
|
||||
}
|
||||
List<List<List<WordInfo>>> grid = new ArrayList<>();
|
||||
for (TextLineInfo row : rows) {
|
||||
if (!row.wordsAreSeparable()) {
|
||||
log.debug("Table row shares operators between cells; falling back to paragraphs");
|
||||
return null;
|
||||
}
|
||||
grid.add(splitCells(row));
|
||||
}
|
||||
int columns = grid.get(0).size();
|
||||
long consistent = grid.stream().filter(row -> row.size() == columns).count();
|
||||
if (columns < 2 || consistent < Math.max(2, grid.size() * 0.6)) {
|
||||
return null;
|
||||
}
|
||||
|
||||
boolean headerRow = looksLikeHeader(rows, grid);
|
||||
StructBlock table = new StructBlock(StructType.TABLE, pageIndex);
|
||||
BBox box = BBox.EMPTY;
|
||||
|
||||
for (int r = 0; r < grid.size(); r++) {
|
||||
List<List<WordInfo>> cells = grid.get(r);
|
||||
if (cells.size() != columns) {
|
||||
continue;
|
||||
}
|
||||
StructBlock tr = new StructBlock(StructType.TR, pageIndex);
|
||||
boolean isHeader = headerRow && r == 0;
|
||||
for (List<WordInfo> cell : cells) {
|
||||
StructBlock td =
|
||||
new StructBlock(isHeader ? StructType.TH : StructType.TD, pageIndex);
|
||||
if (isHeader) {
|
||||
td.setScope("Column");
|
||||
}
|
||||
BBox cellBox = BBox.EMPTY;
|
||||
StringBuilder text = new StringBuilder();
|
||||
int from = cell.get(0).startOrdinal();
|
||||
int to = cell.get(cell.size() - 1).endOrdinal();
|
||||
for (WordInfo word : cell) {
|
||||
cellBox = cellBox.union(word.bbox());
|
||||
if (text.length() > 0) {
|
||||
text.append(' ');
|
||||
}
|
||||
text.append(word.text());
|
||||
}
|
||||
td.addRange(from, to);
|
||||
td.setBbox(cellBox);
|
||||
td.setText(text.toString());
|
||||
tr.addChild(td);
|
||||
box = box.union(cellBox);
|
||||
}
|
||||
tr.setBbox(box);
|
||||
table.addChild(tr);
|
||||
}
|
||||
table.setBbox(box);
|
||||
if (table.getChildren().size() < 2) {
|
||||
return null;
|
||||
}
|
||||
// Clause 7.5 needs equal cell counts per row; a ragged table fails validation outright.
|
||||
long distinctWidths =
|
||||
table.getChildren().stream()
|
||||
.map(row -> row.getChildren().size())
|
||||
.distinct()
|
||||
.count();
|
||||
if (distinctWidths != 1) {
|
||||
log.debug("Discarding a table whose rows have different cell counts");
|
||||
return null;
|
||||
}
|
||||
return table;
|
||||
}
|
||||
|
||||
/** The first row is a header when it is bold, or when only later rows carry numbers. */
|
||||
private static boolean looksLikeHeader(
|
||||
List<TextLineInfo> rows, List<List<List<WordInfo>>> grid) {
|
||||
if (rows.get(0).bold()) {
|
||||
return true;
|
||||
}
|
||||
boolean firstHasDigits = DIGITS.matcher(rows.get(0).text()).find();
|
||||
boolean laterHasDigits =
|
||||
rows.subList(1, rows.size()).stream()
|
||||
.anyMatch(row -> DIGITS.matcher(row.text()).find());
|
||||
return !firstHasDigits && laterHasDigits;
|
||||
}
|
||||
|
||||
// --- Graphics ----------------------------------------------------------
|
||||
|
||||
private List<StructBlock> buildGraphics(
|
||||
PageContent page, DocumentStructure structure, java.util.Set<Integer> claimed) {
|
||||
List<StructBlock> blocks = new ArrayList<>();
|
||||
boolean warnedForms = false;
|
||||
|
||||
// Vectors cluster: a chart is many strokes in one region, a rule is a single thin one.
|
||||
java.util.Set<Integer> vectorFigureOrdinals = vectorFigureOrdinals(page, claimed);
|
||||
|
||||
for (MarkableOp op : page.ops()) {
|
||||
if (op.kind() == MarkableOp.Kind.TEXT || claimed.contains(op.ordinal())) {
|
||||
continue;
|
||||
}
|
||||
BBox box = op.bbox();
|
||||
|
||||
if (op.kind() == MarkableOp.Kind.VECTOR) {
|
||||
StructBlock block;
|
||||
if (vectorFigureOrdinals.contains(op.ordinal())) {
|
||||
block = new StructBlock(StructType.FIGURE, page.pageIndex());
|
||||
} else {
|
||||
block = StructBlock.artifact(ArtifactType.LAYOUT, page.pageIndex());
|
||||
}
|
||||
block.addRange(op.ordinal(), op.ordinal());
|
||||
block.setBbox(box);
|
||||
blocks.add(block);
|
||||
continue;
|
||||
}
|
||||
|
||||
boolean decorative = box.width() < MIN_FIGURE_SIZE || box.height() < MIN_FIGURE_SIZE;
|
||||
if (decorative) {
|
||||
StructBlock artifact = StructBlock.artifact(ArtifactType.LAYOUT, page.pageIndex());
|
||||
artifact.addRange(op.ordinal(), op.ordinal());
|
||||
artifact.setBbox(box);
|
||||
blocks.add(artifact);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (op.kind() == MarkableOp.Kind.FORM && !warnedForms) {
|
||||
structure.warn(
|
||||
"Content inside form XObjects was tagged as a single region because its"
|
||||
+ " text is not separately addressable; review those areas.");
|
||||
warnedForms = true;
|
||||
}
|
||||
|
||||
StructBlock figure = new StructBlock(StructType.FIGURE, page.pageIndex());
|
||||
figure.addRange(op.ordinal(), op.ordinal());
|
||||
figure.setBbox(box);
|
||||
blocks.add(figure);
|
||||
}
|
||||
return blocks;
|
||||
}
|
||||
|
||||
/**
|
||||
* Finds vector operators belonging to a substantial drawing rather than page furniture; thin
|
||||
* paths are rules and table borders, and a short run is ornament.
|
||||
*/
|
||||
private static Set<Integer> vectorFigureOrdinals(
|
||||
PageContent page, java.util.Set<Integer> claimed) {
|
||||
// A chart's plot area is mostly empty, while shading sits behind the text it decorates.
|
||||
Set<Integer> result = new HashSet<>();
|
||||
List<MarkableOp> run = new ArrayList<>();
|
||||
BBox extent = BBox.EMPTY;
|
||||
|
||||
for (MarkableOp op : page.ops()) {
|
||||
boolean substantial =
|
||||
op.kind() == MarkableOp.Kind.VECTOR
|
||||
&& !claimed.contains(op.ordinal())
|
||||
&& !op.bbox().isEmpty()
|
||||
&& op.bbox().width() >= MIN_VECTOR_THICKNESS
|
||||
&& op.bbox().height() >= MIN_VECTOR_THICKNESS;
|
||||
if (substantial) {
|
||||
run.add(op);
|
||||
extent = extent.isEmpty() ? op.bbox() : extent.union(op.bbox());
|
||||
continue;
|
||||
}
|
||||
flushVectorRun(run, extent, page.lines(), result);
|
||||
run = new ArrayList<>();
|
||||
extent = BBox.EMPTY;
|
||||
}
|
||||
flushVectorRun(run, extent, page.lines(), result);
|
||||
return result;
|
||||
}
|
||||
|
||||
private static void flushVectorRun(
|
||||
List<MarkableOp> run,
|
||||
BBox extent,
|
||||
List<TextLineInfo> lines,
|
||||
java.util.Set<Integer> result) {
|
||||
if (run.size() < MIN_VECTOR_FIGURE_OPS
|
||||
|| extent.width() < MIN_VECTOR_FIGURE_SIZE
|
||||
|| extent.height() < MIN_VECTOR_FIGURE_SIZE) {
|
||||
return;
|
||||
}
|
||||
if (overlappingLines(extent, lines) > MAX_LINES_INSIDE_FIGURE) {
|
||||
return;
|
||||
}
|
||||
run.forEach(op -> result.add(op.ordinal()));
|
||||
}
|
||||
|
||||
/** How many text lines sit within the region a vector cluster covers. */
|
||||
private static int overlappingLines(BBox extent, List<TextLineInfo> lines) {
|
||||
int count = 0;
|
||||
for (TextLineInfo line : lines) {
|
||||
BBox box = line.bbox();
|
||||
boolean inside =
|
||||
box.x0() >= extent.x0() - 2
|
||||
&& box.x1() <= extent.x1() + 2
|
||||
&& box.y0() >= extent.y0() - 2
|
||||
&& box.y1() <= extent.y1() + 2;
|
||||
if (inside) {
|
||||
count++;
|
||||
}
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
// --- Post-processing ---------------------------------------------------
|
||||
|
||||
/**
|
||||
* Rewrites heading levels so no level is skipped, which PDF/UA-1 clause 7.4 requires. A
|
||||
* document that jumps H1 to H3 is remapped to H1, H2 while preserving relative depth.
|
||||
*/
|
||||
static void normaliseHeadingLevels(DocumentStructure structure) {
|
||||
List<StructBlock> headings = new ArrayList<>();
|
||||
structure.visit(
|
||||
block -> {
|
||||
if (block.getType().isHeading()) {
|
||||
headings.add(block);
|
||||
}
|
||||
});
|
||||
int previous = 0;
|
||||
for (StructBlock heading : headings) {
|
||||
int level = heading.getType().headingLevel();
|
||||
int adjusted = level > previous + 1 ? previous + 1 : level;
|
||||
heading.setType(StructType.heading(adjusted));
|
||||
previous = adjusted;
|
||||
}
|
||||
}
|
||||
|
||||
/** Uses the first top-level heading as the title when the document has no metadata title. */
|
||||
private static String deriveTitle(DocumentStructure structure) {
|
||||
for (StructBlock block : structure.getBlocks()) {
|
||||
if (block.getType().isHeading() && !block.getText().isBlank()) {
|
||||
return block.getText().strip();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
/**
|
||||
* One operator in a page content stream that may be wrapped in a marked-content sequence. The
|
||||
* ordinal counts only markable operators, joining text extraction to token rewriting.
|
||||
*/
|
||||
public record MarkableOp(int ordinal, Kind kind, BBox bbox, String resourceName) {
|
||||
|
||||
public enum Kind {
|
||||
/** Tj, TJ, ' or " */
|
||||
TEXT,
|
||||
/** Do referencing an image XObject */
|
||||
IMAGE,
|
||||
/** Do referencing a form XObject */
|
||||
FORM,
|
||||
/** BI ... ID ... EI */
|
||||
INLINE_IMAGE,
|
||||
/** A path-painting or shading operator: rules, borders, fills, logos */
|
||||
VECTOR;
|
||||
|
||||
public boolean isGraphic() {
|
||||
return this == IMAGE || this == INLINE_IMAGE;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Operator names counted as markable; both passes must agree on this set. Path painting is
|
||||
* included because clause 7.1 needs visible rules and borders tagged or artifacted.
|
||||
*/
|
||||
public static boolean isMarkableOperator(String name) {
|
||||
return switch (name) {
|
||||
case "Tj", "TJ", "'", "\"", "Do", "BI" -> true;
|
||||
default -> isPathPainting(name);
|
||||
};
|
||||
}
|
||||
|
||||
/** Painting operators only: {@code n} ends a path without marking the page. */
|
||||
public static boolean isPathPainting(String name) {
|
||||
return switch (name) {
|
||||
case "S", "s", "f", "F", "f*", "B", "B*", "b", "b*", "sh" -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
}
|
||||
+284
@@ -0,0 +1,284 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.OutputStream;
|
||||
import java.util.ArrayDeque;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Deque;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import org.apache.pdfbox.contentstream.operator.Operator;
|
||||
import org.apache.pdfbox.cos.COSBase;
|
||||
import org.apache.pdfbox.cos.COSDictionary;
|
||||
import org.apache.pdfbox.cos.COSInteger;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.pdfparser.PDFStreamParser;
|
||||
import org.apache.pdfbox.pdfwriter.ContentStreamWriter;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.common.PDStream;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
/**
|
||||
* Rewrites a page stream so every markable operator sits inside a marked-content sequence: claimed
|
||||
* content gets an MCID, everything else /Artifact, satisfying PDF/UA-1 clause 7.1 by construction.
|
||||
*/
|
||||
@Slf4j
|
||||
public class MarkedContentInjector {
|
||||
|
||||
private static final COSName ARTIFACT = COSName.getPDFName("Artifact");
|
||||
private static final COSName MCID = COSName.getPDFName("MCID");
|
||||
private static final COSName ACTUAL_TEXT = COSName.getPDFName("ActualText");
|
||||
private static final COSName ALT = COSName.getPDFName("Alt");
|
||||
|
||||
/** Operators that force an open sequence to close so nesting stays legal. */
|
||||
private static boolean isBoundary(String name) {
|
||||
return "BT".equals(name) || "ET".equals(name) || "q".equals(name) || "Q".equals(name);
|
||||
}
|
||||
|
||||
private static boolean isMarkedContentOperator(String name) {
|
||||
return "BDC".equals(name) || "BMC".equals(name) || "EMC".equals(name);
|
||||
}
|
||||
|
||||
/**
|
||||
* Path-construction operators; ISO 32000-1 forbids marked content inside a path object, so a
|
||||
* sequence wrapping a fill or stroke must open before the path starts.
|
||||
*/
|
||||
private static boolean isPathConstruction(String name) {
|
||||
return switch (name) {
|
||||
case "m", "l", "c", "v", "y", "h", "re" -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
private static boolean opensMarkedContent(String name) {
|
||||
return "BDC".equals(name) || "BMC".equals(name);
|
||||
}
|
||||
|
||||
/**
|
||||
* True for an optional-content sequence; stripping an {@code /OC} wrapper would make hidden
|
||||
* layers such as watermarks or redaction overlays visible.
|
||||
*/
|
||||
private static boolean isOptionalContent(String name, List<COSBase> operands) {
|
||||
return opensMarkedContent(name)
|
||||
&& !operands.isEmpty()
|
||||
&& operands.get(0) instanceof COSName tag
|
||||
&& "OC".equals(tag.getName());
|
||||
}
|
||||
|
||||
/**
|
||||
* True when a sequence supplies replacement text for its glyphs; dropping it leaves a screen
|
||||
* reader with the font's own mapping, which for a ligature says nothing useful.
|
||||
*/
|
||||
private static boolean carriesReplacementText(String name, List<COSBase> operands) {
|
||||
if (!opensMarkedContent(name)) {
|
||||
return false;
|
||||
}
|
||||
for (COSBase operand : operands) {
|
||||
if (operand instanceof COSDictionary properties
|
||||
&& (properties.containsKey(ACTUAL_TEXT)
|
||||
|| properties.containsKey(ALT)
|
||||
|| properties.containsKey(COSName.E))) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/** The source's own ids mean nothing once the tree is rebuilt, so they are dropped. */
|
||||
private static void stripStaleMcid(List<COSBase> operands) {
|
||||
for (COSBase operand : operands) {
|
||||
if (operand instanceof COSDictionary properties) {
|
||||
properties.removeItem(MCID);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Wraps every markable operator on the page; returns the next unused marked content id. */
|
||||
public int inject(
|
||||
PDDocument document,
|
||||
PDPage page,
|
||||
List<StructBlock> blocks,
|
||||
int nextMcid,
|
||||
boolean stripExisting)
|
||||
throws IOException {
|
||||
|
||||
Map<Integer, StructBlock> owners = ownersByOrdinal(blocks);
|
||||
List<Object> tokens = parse(page);
|
||||
List<Object> output = new ArrayList<>(tokens.size() + owners.size() * 4);
|
||||
|
||||
List<COSBase> operands = new ArrayList<>();
|
||||
// Tracks, for each surviving source sequence, whether its closer should be kept.
|
||||
Deque<Boolean> keptSequences = new ArrayDeque<>();
|
||||
StructBlock openBlock = null;
|
||||
boolean open = false;
|
||||
int ordinal = -1;
|
||||
int mcid = nextMcid;
|
||||
int pathStart = -1;
|
||||
|
||||
for (Object token : tokens) {
|
||||
if (!(token instanceof Operator operator)) {
|
||||
operands.add((COSBase) token);
|
||||
continue;
|
||||
}
|
||||
String name = operator.getName();
|
||||
|
||||
if (stripExisting && isMarkedContentOperator(name)) {
|
||||
boolean keep;
|
||||
if (opensMarkedContent(name)) {
|
||||
keep =
|
||||
isOptionalContent(name, operands)
|
||||
|| carriesReplacementText(name, operands);
|
||||
if (keep) {
|
||||
stripStaleMcid(operands);
|
||||
}
|
||||
keptSequences.push(keep);
|
||||
} else {
|
||||
// A closer is kept exactly when its matching opener was.
|
||||
keep = !keptSequences.isEmpty() && keptSequences.pop();
|
||||
}
|
||||
if (!keep) {
|
||||
operands.clear();
|
||||
continue;
|
||||
}
|
||||
// Close our own sequence first so the two never interleave illegally.
|
||||
if (open) {
|
||||
output.add(Operator.getOperator("EMC"));
|
||||
open = false;
|
||||
openBlock = null;
|
||||
}
|
||||
output.addAll(operands);
|
||||
output.add(operator);
|
||||
operands.clear();
|
||||
continue;
|
||||
}
|
||||
|
||||
if (isBoundary(name) && open) {
|
||||
output.add(Operator.getOperator("EMC"));
|
||||
open = false;
|
||||
openBlock = null;
|
||||
}
|
||||
|
||||
// Remember where the current path object began so a sequence wrapping its painting
|
||||
// operator can be opened before it rather than inside it.
|
||||
if (isPathConstruction(name)) {
|
||||
if (pathStart < 0) {
|
||||
pathStart = output.size();
|
||||
}
|
||||
} else if (!MarkableOp.isPathPainting(name) && !"n".equals(name)) {
|
||||
pathStart = -1;
|
||||
}
|
||||
|
||||
if (MarkableOp.isMarkableOperator(name)) {
|
||||
ordinal++;
|
||||
StructBlock owner = owners.get(ordinal);
|
||||
if (!open || owner != openBlock) {
|
||||
boolean insidePath = MarkableOp.isPathPainting(name) && pathStart >= 0;
|
||||
if (open) {
|
||||
// Close before the path began, so the EMC also stays outside the path.
|
||||
output.add(
|
||||
insidePath ? pathStart : output.size(),
|
||||
Operator.getOperator("EMC"));
|
||||
if (insidePath) {
|
||||
pathStart++;
|
||||
}
|
||||
}
|
||||
int at = insidePath ? pathStart : output.size();
|
||||
mcid = openSequenceAt(output, at, owner, mcid);
|
||||
open = true;
|
||||
openBlock = owner;
|
||||
}
|
||||
}
|
||||
|
||||
output.addAll(operands);
|
||||
output.add(operator);
|
||||
operands.clear();
|
||||
|
||||
if (MarkableOp.isPathPainting(name) || "n".equals(name)) {
|
||||
pathStart = -1;
|
||||
}
|
||||
}
|
||||
|
||||
if (open) {
|
||||
output.add(Operator.getOperator("EMC"));
|
||||
}
|
||||
|
||||
write(document, page, output);
|
||||
return mcid;
|
||||
}
|
||||
|
||||
/** Emits the opening BDC/BMC at a given position and records the id on the owning block. */
|
||||
private int openSequenceAt(List<Object> output, int at, StructBlock owner, int mcid) {
|
||||
List<Object> opening = new ArrayList<>(3);
|
||||
if (owner == null) {
|
||||
opening.add(ARTIFACT);
|
||||
opening.add(Operator.getOperator("BMC"));
|
||||
} else if (owner.isArtifact()) {
|
||||
COSDictionary properties = new COSDictionary();
|
||||
if (owner.getArtifactType() != null) {
|
||||
properties.setName(COSName.TYPE, owner.getArtifactType().subtype());
|
||||
}
|
||||
opening.add(ARTIFACT);
|
||||
opening.add(properties);
|
||||
opening.add(Operator.getOperator("BDC"));
|
||||
} else {
|
||||
COSDictionary properties = new COSDictionary();
|
||||
properties.setItem(MCID, COSInteger.get(mcid));
|
||||
opening.add(COSName.getPDFName(owner.getType().tag()));
|
||||
opening.add(properties);
|
||||
opening.add(Operator.getOperator("BDC"));
|
||||
owner.getMcids().add(mcid);
|
||||
mcid++;
|
||||
}
|
||||
output.addAll(at, opening);
|
||||
return mcid;
|
||||
}
|
||||
|
||||
/**
|
||||
* Maps each claimed ordinal to its block. Overlapping claims are dropped rather than merged:
|
||||
* two structure elements sharing content would make the reading order ambiguous.
|
||||
*/
|
||||
static Map<Integer, StructBlock> ownersByOrdinal(List<StructBlock> blocks) {
|
||||
Map<Integer, StructBlock> owners = new HashMap<>();
|
||||
for (StructBlock block : blocks) {
|
||||
block.visit(
|
||||
node -> {
|
||||
for (StructBlock.OrdinalRange range : node.getRanges()) {
|
||||
for (int i = range.start(); i <= range.end(); i++) {
|
||||
StructBlock existing = owners.putIfAbsent(i, node);
|
||||
if (existing != null && existing != node) {
|
||||
log.debug(
|
||||
"Ordinal {} claimed by both {} and {}; keeping the first",
|
||||
i,
|
||||
existing,
|
||||
node);
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
return owners;
|
||||
}
|
||||
|
||||
private static List<Object> parse(PDPage page) throws IOException {
|
||||
PDFStreamParser parser = new PDFStreamParser(page);
|
||||
List<Object> tokens = new ArrayList<>();
|
||||
Object token;
|
||||
while ((token = parser.parseNextToken()) != null) {
|
||||
tokens.add(token);
|
||||
}
|
||||
return tokens;
|
||||
}
|
||||
|
||||
private static void write(PDDocument document, PDPage page, List<Object> tokens)
|
||||
throws IOException {
|
||||
PDStream stream = new PDStream(document);
|
||||
try (OutputStream out = stream.createOutputStream(COSName.FLATE_DECODE)) {
|
||||
new ContentStreamWriter(out).writeTokens(tokens);
|
||||
}
|
||||
page.setContents(stream);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Everything the layout analyser needs about one page. carriesTextSemantics: existing marked
|
||||
* content has ActualText/Alt/expansion a rebuild would discard. linesDropped: text became
|
||||
* artifacts.
|
||||
*/
|
||||
public record PageContent(
|
||||
int pageIndex,
|
||||
List<TextLineInfo> lines,
|
||||
List<MarkableOp> ops,
|
||||
int markableCount,
|
||||
boolean preExistingMarkedContent,
|
||||
boolean carriesTextSemantics,
|
||||
boolean linesDropped,
|
||||
BBox mediaBox) {
|
||||
|
||||
public boolean hasText() {
|
||||
return lines.stream().anyMatch(line -> !line.isBlank());
|
||||
}
|
||||
|
||||
/** Markable operators that draw graphics rather than text. */
|
||||
public List<MarkableOp> graphics() {
|
||||
return ops.stream().filter(op -> op.kind().isGraphic()).toList();
|
||||
}
|
||||
|
||||
/** Form XObject invocations, which are tagged as a unit because their text is opaque here. */
|
||||
public List<MarkableOp> forms() {
|
||||
return ops.stream().filter(op -> op.kind() == MarkableOp.Kind.FORM).toList();
|
||||
}
|
||||
}
|
||||
+47
@@ -0,0 +1,47 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import org.apache.xmpbox.XMPMetadata;
|
||||
import org.apache.xmpbox.schema.XMPSchema;
|
||||
import org.apache.xmpbox.type.IntegerType;
|
||||
import org.apache.xmpbox.type.StructuredType;
|
||||
|
||||
/**
|
||||
* The {@code pdfuaid} XMP conformance schema, which XMPBox does not ship. Only write it once
|
||||
* validation has passed - it is a compliance claim.
|
||||
*/
|
||||
@StructuredType(
|
||||
preferedPrefix = PdfUaIdentificationSchema.PREFERRED_PREFIX,
|
||||
namespace = PdfUaIdentificationSchema.NAMESPACE)
|
||||
public class PdfUaIdentificationSchema extends XMPSchema {
|
||||
|
||||
public static final String PREFERRED_PREFIX = "pdfuaid";
|
||||
public static final String NAMESPACE = "http://www.aiim.org/pdfua/ns/id/";
|
||||
|
||||
public static final String PART = "part";
|
||||
public static final String REV = "rev";
|
||||
|
||||
public PdfUaIdentificationSchema(XMPMetadata metadata) {
|
||||
super(metadata);
|
||||
}
|
||||
|
||||
public PdfUaIdentificationSchema(XMPMetadata metadata, String prefix) {
|
||||
super(metadata, prefix);
|
||||
}
|
||||
|
||||
/** Sets {@code pdfuaid:part}, the conformance level (1 or 2). */
|
||||
public void setPart(int part) {
|
||||
addProperty(new IntegerType(getMetadata(), getNamespace(), getPrefix(), PART, part));
|
||||
}
|
||||
|
||||
/** Sets {@code pdfuaid:rev}, the four-digit revision year used by PDF/UA-2. */
|
||||
public void setRevision(int year) {
|
||||
addProperty(new IntegerType(getMetadata(), getNamespace(), getPrefix(), REV, year));
|
||||
}
|
||||
|
||||
public Integer getPart() {
|
||||
if (getProperty(PART) instanceof IntegerType part) {
|
||||
return part.getValue();
|
||||
}
|
||||
return null;
|
||||
}
|
||||
}
|
||||
+224
@@ -0,0 +1,224 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDDocumentCatalog;
|
||||
import org.apache.pdfbox.pdmodel.PDDocumentInformation;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.common.PDMetadata;
|
||||
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
|
||||
import org.apache.pdfbox.pdmodel.interactive.form.PDField;
|
||||
import org.apache.pdfbox.pdmodel.interactive.viewerpreferences.PDViewerPreferences;
|
||||
import org.apache.xmpbox.XMPMetadata;
|
||||
import org.apache.xmpbox.schema.DublinCoreSchema;
|
||||
import org.apache.xmpbox.schema.XMPSchema;
|
||||
import org.apache.xmpbox.xml.DomXmpParser;
|
||||
import org.apache.xmpbox.xml.XmpSerializer;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
/** Applies the document-level PDF/UA requirements: title, language, tab order, declaration. */
|
||||
@Slf4j
|
||||
public class PdfUaMetadataWriter {
|
||||
|
||||
private static final COSName TABS = COSName.getPDFName("Tabs");
|
||||
private static final COSName SUSPECTS = COSName.getPDFName("Suspects");
|
||||
|
||||
/**
|
||||
* Applies everything except the conformance declaration. Clause 7.1 requires a title, so a
|
||||
* blank one falls back to the existing metadata title.
|
||||
*/
|
||||
public List<String> applyDocumentRequirements(
|
||||
PDDocument document, String title, String language, PdfUaProfile profile)
|
||||
throws IOException {
|
||||
return applyDocumentRequirements(document, title, language, profile, false);
|
||||
}
|
||||
|
||||
public List<String> applyDocumentRequirements(
|
||||
PDDocument document,
|
||||
String title,
|
||||
String language,
|
||||
PdfUaProfile profile,
|
||||
boolean preserveVersion)
|
||||
throws IOException {
|
||||
|
||||
List<String> warnings = new ArrayList<>();
|
||||
PDDocumentCatalog catalog = document.getDocumentCatalog();
|
||||
|
||||
if (language != null && !language.isBlank()) {
|
||||
catalog.setLanguage(language);
|
||||
}
|
||||
|
||||
String effectiveTitle = resolveTitle(document, title);
|
||||
if (effectiveTitle != null) {
|
||||
PDDocumentInformation info = document.getDocumentInformation();
|
||||
info.setTitle(effectiveTitle);
|
||||
document.setDocumentInformation(info);
|
||||
}
|
||||
|
||||
// Without this a viewer shows the filename instead of the title, which defeats the point.
|
||||
PDViewerPreferences preferences = catalog.getViewerPreferences();
|
||||
if (preferences == null) {
|
||||
preferences = new PDViewerPreferences(catalog.getCOSObject());
|
||||
}
|
||||
preferences.setDisplayDocTitle(true);
|
||||
catalog.setViewerPreferences(preferences);
|
||||
|
||||
// Clause 7.18.1: every page needs an explicit tab order.
|
||||
for (PDPage page : document.getPages()) {
|
||||
page.getCOSObject().setName(TABS, "S");
|
||||
}
|
||||
|
||||
// A structure tree flagged as suspect is not conforming.
|
||||
if (catalog.getMarkInfo() != null) {
|
||||
catalog.getMarkInfo().getCOSObject().removeItem(SUSPECTS);
|
||||
}
|
||||
|
||||
if (!preserveVersion && document.getVersion() < profile.pdfVersion()) {
|
||||
document.setVersion(profile.pdfVersion());
|
||||
}
|
||||
|
||||
warnings.addAll(describeFormFields(document));
|
||||
writeXmp(document, effectiveTitle, language, null);
|
||||
return warnings;
|
||||
}
|
||||
|
||||
/**
|
||||
* Gives every form field the {@code /TU} description clause 7.18.1 requires, reusing its
|
||||
* authored partial name. Unnamed fields are reported, never given a useless placeholder.
|
||||
*/
|
||||
private static List<String> describeFormFields(PDDocument document) {
|
||||
List<String> warnings = new ArrayList<>();
|
||||
PDAcroForm form = document.getDocumentCatalog().getAcroForm();
|
||||
if (form == null) {
|
||||
return warnings;
|
||||
}
|
||||
int unnamed = 0;
|
||||
for (PDField field : form.getFieldTree()) {
|
||||
String existing = field.getAlternateFieldName();
|
||||
if (existing != null && !existing.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
String partialName = field.getPartialName();
|
||||
if (partialName == null || partialName.isBlank()) {
|
||||
unnamed++;
|
||||
continue;
|
||||
}
|
||||
field.setAlternateFieldName(partialName);
|
||||
}
|
||||
if (unnamed > 0) {
|
||||
warnings.add(
|
||||
unnamed
|
||||
+ " form field(s) have neither a description nor a name, so no tooltip"
|
||||
+ " could be derived. Add one for each before claiming conformance.");
|
||||
}
|
||||
return warnings;
|
||||
}
|
||||
|
||||
/**
|
||||
* Strips the {@code pdfuaid} declaration when validation fails after it was written, so the
|
||||
* returned file does not assert conformance it lacks.
|
||||
*/
|
||||
public void removeConformanceDeclaration(PDDocument document) throws IOException {
|
||||
PDDocumentCatalog catalog = document.getDocumentCatalog();
|
||||
XMPMetadata metadata = loadOrCreate(catalog);
|
||||
XMPSchema identification = metadata.getSchema(PdfUaIdentificationSchema.NAMESPACE);
|
||||
if (identification == null) {
|
||||
return;
|
||||
}
|
||||
metadata.removeSchema(identification);
|
||||
serialiseInto(document, metadata);
|
||||
}
|
||||
|
||||
/** Writes the {@code pdfuaid:part} declaration. Only call this after validation has passed. */
|
||||
public void declareConformance(PDDocument document, PdfUaProfile profile) throws IOException {
|
||||
writeXmp(document, resolveTitle(document, null), documentLanguage(document), profile);
|
||||
}
|
||||
|
||||
private String resolveTitle(PDDocument document, String preferred) {
|
||||
if (preferred != null && !preferred.isBlank()) {
|
||||
return preferred.strip();
|
||||
}
|
||||
String existing = document.getDocumentInformation().getTitle();
|
||||
return existing != null && !existing.isBlank() ? existing.strip() : null;
|
||||
}
|
||||
|
||||
private static String documentLanguage(PDDocument document) {
|
||||
return document.getDocumentCatalog().getLanguage();
|
||||
}
|
||||
|
||||
/**
|
||||
* Rewrites the XMP packet, preserving what was there. A malformed packet is replaced, since an
|
||||
* unparseable one fails validation on its own.
|
||||
*/
|
||||
private void writeXmp(PDDocument document, String title, String language, PdfUaProfile profile)
|
||||
throws IOException {
|
||||
|
||||
PDDocumentCatalog catalog = document.getDocumentCatalog();
|
||||
XMPMetadata metadata = loadOrCreate(catalog);
|
||||
|
||||
if (title != null) {
|
||||
DublinCoreSchema dublinCore = metadata.getDublinCoreSchema();
|
||||
if (dublinCore == null) {
|
||||
dublinCore = metadata.createAndAddDublinCoreSchema();
|
||||
}
|
||||
dublinCore.setTitle(title);
|
||||
if (language != null
|
||||
&& !language.isBlank()
|
||||
&& (dublinCore.getLanguages() == null
|
||||
|| !dublinCore.getLanguages().contains(language))) {
|
||||
dublinCore.addLanguage(language);
|
||||
}
|
||||
}
|
||||
|
||||
if (profile != null) {
|
||||
// Re-converting an already-declared file must not leave two pdfuaid schemas.
|
||||
XMPSchema stale = metadata.getSchema(PdfUaIdentificationSchema.NAMESPACE);
|
||||
if (stale != null) {
|
||||
metadata.removeSchema(stale);
|
||||
}
|
||||
PdfUaIdentificationSchema identification = new PdfUaIdentificationSchema(metadata);
|
||||
identification.setPart(profile.part());
|
||||
if (profile.revision() > 0) {
|
||||
identification.setRevision(profile.revision());
|
||||
}
|
||||
metadata.addSchema(identification);
|
||||
}
|
||||
|
||||
serialiseInto(document, metadata);
|
||||
}
|
||||
|
||||
private static void serialiseInto(PDDocument document, XMPMetadata metadata)
|
||||
throws IOException {
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
try {
|
||||
new XmpSerializer().serialize(metadata, out, true);
|
||||
} catch (javax.xml.transform.TransformerException e) {
|
||||
throw new IOException("Could not serialise XMP metadata", e);
|
||||
}
|
||||
PDMetadata pdMetadata = new PDMetadata(document);
|
||||
pdMetadata.importXMPMetadata(out.toByteArray());
|
||||
document.getDocumentCatalog().setMetadata(pdMetadata);
|
||||
}
|
||||
|
||||
private XMPMetadata loadOrCreate(PDDocumentCatalog catalog) {
|
||||
PDMetadata existing = catalog.getMetadata();
|
||||
if (existing != null) {
|
||||
try {
|
||||
DomXmpParser parser = new DomXmpParser();
|
||||
// Strict parsing rejects pdfuaid, silently discarding a packet we just wrote.
|
||||
parser.setStrictParsing(false);
|
||||
return parser.parse(new ByteArrayInputStream(existing.toByteArray()));
|
||||
} catch (Exception e) {
|
||||
log.debug("Replacing unparseable XMP packet: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
return XMPMetadata.createXMPMetadata();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,47 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
/** The PDF/UA conformance level a conversion targets. */
|
||||
public enum PdfUaProfile {
|
||||
/** ISO 14289-1, layered on PDF 1.7. */
|
||||
UA1(1, 1.7f, 0),
|
||||
/** ISO 14289-2: needs PDF 2.0, namespaced structure types and a revision year. */
|
||||
UA2(2, 2.0f, 2024);
|
||||
|
||||
private final int part;
|
||||
private final float pdfVersion;
|
||||
private final int revision;
|
||||
|
||||
PdfUaProfile(int part, float pdfVersion, int revision) {
|
||||
this.part = part;
|
||||
this.pdfVersion = pdfVersion;
|
||||
this.revision = revision;
|
||||
}
|
||||
|
||||
public int part() {
|
||||
return part;
|
||||
}
|
||||
|
||||
public float pdfVersion() {
|
||||
return pdfVersion;
|
||||
}
|
||||
|
||||
/** The {@code pdfuaid:rev} year, or 0 when the profile does not use one. */
|
||||
public int revision() {
|
||||
return revision;
|
||||
}
|
||||
|
||||
public String displayName() {
|
||||
return "PDF/UA-" + part;
|
||||
}
|
||||
|
||||
public static PdfUaProfile fromRequest(String value) {
|
||||
if (value == null || value.isBlank()) {
|
||||
return UA1;
|
||||
}
|
||||
String normalised = value.trim().toLowerCase().replace("/", "").replace("-", "");
|
||||
return switch (normalised) {
|
||||
case "ua2", "pdfua2", "2" -> UA2;
|
||||
default -> UA1;
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,303 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDDocumentCatalog;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureElement;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureTreeRoot;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
/**
|
||||
* Tags an untagged PDF and applies the document-level PDF/UA requirements. Content must be marked
|
||||
* before the tree can reference it, and conformance is declared elsewhere, only after validation.
|
||||
*/
|
||||
@Slf4j
|
||||
public class PdfUaTagger {
|
||||
|
||||
private final TaggedContentExtractor extractor = new TaggedContentExtractor();
|
||||
private final LayoutAnalyzer analyzer = new LayoutAnalyzer();
|
||||
private final MarkedContentInjector injector = new MarkedContentInjector();
|
||||
private final PdfUaMetadataWriter metadataWriter = new PdfUaMetadataWriter();
|
||||
|
||||
public TaggingResult tag(PDDocument document, TaggingOptions options) throws IOException {
|
||||
boolean alreadyTagged = hasUsableStructureTree(document);
|
||||
boolean rebuild =
|
||||
switch (options.getExistingTags()) {
|
||||
case KEEP -> false;
|
||||
case REBUILD -> true;
|
||||
case AUTO -> !alreadyTagged;
|
||||
};
|
||||
|
||||
List<String> languageWarnings = new ArrayList<>();
|
||||
String language = resolveLanguage(document, options, languageWarnings);
|
||||
|
||||
if (!rebuild) {
|
||||
log.info("Keeping existing structure tree; applying document requirements only");
|
||||
DocumentStructure kept = new DocumentStructure();
|
||||
languageWarnings.forEach(kept::warn);
|
||||
metadataWriter
|
||||
.applyDocumentRequirements(
|
||||
document,
|
||||
options.getTitle(),
|
||||
language,
|
||||
options.getProfile(),
|
||||
options.isPreservePdfVersion())
|
||||
.forEach(kept::warn);
|
||||
return new TaggingResult(kept, false);
|
||||
}
|
||||
|
||||
// Types the old tree carried, so a rebuild that cannot reproduce them can say so. Font
|
||||
// embedding may already have deleted the tree, so fall back to what the source had.
|
||||
Set<String> discardedTypes =
|
||||
alreadyTagged
|
||||
? structureTypes(document)
|
||||
: options.getSourceFacts().structureTypes();
|
||||
|
||||
if (alreadyTagged) {
|
||||
stripStructure(document);
|
||||
}
|
||||
|
||||
List<PageContent> pages = extractor.extract(document);
|
||||
DocumentStructure structure = analyzer.analyse(pages);
|
||||
structure.setLanguage(language);
|
||||
languageWarnings.forEach(structure::warn);
|
||||
applyFigurePolicy(structure, options);
|
||||
|
||||
if (structure.isEmpty()) {
|
||||
structure.warn(
|
||||
"No taggable content was found; the document may be a scan with no text layer.");
|
||||
}
|
||||
|
||||
injectMarkedContent(document, structure, pages);
|
||||
new StructTreeWriter().write(document, structure, options.getProfile());
|
||||
// Losing the tree to the embedder is a different problem from a requested rebuild, and
|
||||
// the advice that helps differs too, so tell them apart.
|
||||
boolean lostToEmbedder = !alreadyTagged && options.getSourceFacts().hasUsableTree();
|
||||
warnAboutFlattenedStructure(
|
||||
discardedTypes, structureTypes(document), structure, lostToEmbedder);
|
||||
|
||||
String title = resolveTitle(options, structure);
|
||||
if (title == null) {
|
||||
structure.warn(
|
||||
"No document title could be derived. PDF/UA requires one, so supply a title.");
|
||||
}
|
||||
metadataWriter
|
||||
.applyDocumentRequirements(
|
||||
document,
|
||||
title,
|
||||
language,
|
||||
options.getProfile(),
|
||||
options.isPreservePdfVersion())
|
||||
.forEach(structure::warn);
|
||||
|
||||
return new TaggingResult(structure, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Keeps the language the document already declares. Overwriting it relabels, say, a French file
|
||||
* as English, and no validator can catch that.
|
||||
*/
|
||||
private static String resolveLanguage(
|
||||
PDDocument document, TaggingOptions options, List<String> warnings) {
|
||||
String existing = document.getDocumentCatalog().getLanguage();
|
||||
if (existing == null || existing.isBlank()) {
|
||||
// Font embedding discards /Lang, so without this a rewritten French document would
|
||||
// silently take the caller's default language.
|
||||
existing = options.getSourceFacts().language();
|
||||
}
|
||||
String requested = options.getLanguage();
|
||||
if (existing == null || existing.isBlank() || options.isOverrideLanguage()) {
|
||||
return requested;
|
||||
}
|
||||
if (requested != null && !requested.isBlank() && !requested.equalsIgnoreCase(existing)) {
|
||||
warnings.add(
|
||||
"The document already declares its language as '"
|
||||
+ existing
|
||||
+ "', so the requested '"
|
||||
+ requested
|
||||
+ "' was ignored. Ask to override the language to change it.");
|
||||
}
|
||||
return existing;
|
||||
}
|
||||
|
||||
/** Explicit title first, then the first heading, then the caller's fallback. */
|
||||
private static String resolveTitle(TaggingOptions options, DocumentStructure structure) {
|
||||
for (String candidate :
|
||||
new String[] {
|
||||
options.getTitle(), structure.getTitle(), options.getFallbackTitle()
|
||||
}) {
|
||||
if (candidate != null && !candidate.isBlank()) {
|
||||
return candidate.strip();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Writes the conformance declaration. Separate from tagging so validation can gate it. */
|
||||
public void declareConformance(PDDocument document, PdfUaProfile profile) throws IOException {
|
||||
metadataWriter.declareConformance(document, profile);
|
||||
}
|
||||
|
||||
/** Withdraws the conformance claim, for a document that turned out not to validate. */
|
||||
public void withdrawConformance(PDDocument document) throws IOException {
|
||||
metadataWriter.removeConformanceDeclaration(document);
|
||||
}
|
||||
|
||||
/** Wraps content page by page; marked content ids restart on each page. */
|
||||
private void injectMarkedContent(
|
||||
PDDocument document, DocumentStructure structure, List<PageContent> pages)
|
||||
throws IOException {
|
||||
Map<Integer, Integer> markableCounts = new LinkedHashMap<>();
|
||||
pages.forEach(page -> markableCounts.put(page.pageIndex(), page.markableCount()));
|
||||
Map<Integer, List<StructBlock>> byPage = new LinkedHashMap<>();
|
||||
for (StructBlock block : structure.getBlocks()) {
|
||||
byPage.computeIfAbsent(block.getPageIndex(), k -> new ArrayList<>()).add(block);
|
||||
}
|
||||
for (int pageIndex = 0; pageIndex < document.getNumberOfPages(); pageIndex++) {
|
||||
List<StructBlock> blocks = byPage.getOrDefault(pageIndex, List.of());
|
||||
// Nothing to wrap, and rewriting costs a parse and recompress for an identical stream.
|
||||
if (blocks.isEmpty() && markableCounts.getOrDefault(pageIndex, 0) == 0) {
|
||||
continue;
|
||||
}
|
||||
injector.inject(document, document.getPage(pageIndex), blocks, 0, true);
|
||||
}
|
||||
}
|
||||
|
||||
/** Applies alt text supplied by the caller, or demotes images to artifacts on request. */
|
||||
private static void applyFigurePolicy(DocumentStructure structure, TaggingOptions options) {
|
||||
int[] suppressed = {0};
|
||||
structure.visit(
|
||||
block -> {
|
||||
if (block.getType() != StructType.FIGURE) {
|
||||
return;
|
||||
}
|
||||
if (options.getFigurePolicy() == TaggingOptions.FigurePolicy.MARK_DECORATIVE) {
|
||||
block.setType(StructType.ARTIFACT);
|
||||
block.setArtifactType(ArtifactType.LAYOUT);
|
||||
suppressed[0]++;
|
||||
return;
|
||||
}
|
||||
int ordinal =
|
||||
block.getRanges().isEmpty() ? -1 : block.getRanges().get(0).start();
|
||||
String alt = options.altTextFor(block.getPageIndex(), ordinal);
|
||||
if (alt != null && !alt.isBlank()) {
|
||||
block.setAlt(alt);
|
||||
}
|
||||
});
|
||||
// Marking images decorative validates by hiding content, so never report it as clean.
|
||||
if (suppressed[0] > 0) {
|
||||
structure.warn(
|
||||
suppressed[0]
|
||||
+ " image(s) were marked as decoration and are now hidden from"
|
||||
+ " assistive technology. Confirm none of them carried meaning.");
|
||||
}
|
||||
int missing = structure.figuresWithoutAlt().size();
|
||||
if (missing > 0) {
|
||||
structure.warn(
|
||||
missing
|
||||
+ " figure(s) have no alternative description. PDF/UA requires one for"
|
||||
+ " every image that carries meaning.");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A tree is only worth keeping when wired up: kids, a parent tree, and a marked catalog.
|
||||
* Keeping one that fails any of those leaves the document permanently unfixable.
|
||||
*/
|
||||
public static boolean hasUsableStructureTree(PDDocument document) {
|
||||
PDDocumentCatalog catalog = document.getDocumentCatalog();
|
||||
PDStructureTreeRoot root = catalog.getStructureTreeRoot();
|
||||
if (root == null) {
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
boolean hasKids = root.getKids() != null && !root.getKids().isEmpty();
|
||||
boolean hasParentTree = root.getParentTree() != null;
|
||||
boolean marked = catalog.getMarkInfo() != null && catalog.getMarkInfo().isMarked();
|
||||
return hasKids && hasParentTree && marked;
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("Unreadable structure tree, treating as absent: {}", e.getMessage());
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A rebuild derives structure from layout, so semantics the old tree carried can vanish - a
|
||||
* table becomes loose paragraphs. Validators cannot see that loss, so it has to be reported.
|
||||
*/
|
||||
private static void warnAboutFlattenedStructure(
|
||||
Set<String> before,
|
||||
Set<String> after,
|
||||
DocumentStructure structure,
|
||||
boolean lostToEmbedder) {
|
||||
List<String> lost =
|
||||
MEANINGFUL_TYPES.stream()
|
||||
.filter(type -> before.contains(type) && !after.contains(type))
|
||||
.toList();
|
||||
if (lost.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
// Keeping the tags cannot help once the embedder has deleted them, so do not suggest it.
|
||||
String remedy =
|
||||
lostToEmbedder
|
||||
? " Embedding the missing fonts rewrote the document and deleted its"
|
||||
+ " original tags. Turn off font embedding to keep them."
|
||||
: " Keep the existing tags instead to preserve it.";
|
||||
structure.warn(
|
||||
"Rebuilding the tags could not reproduce "
|
||||
+ String.join(", ", lost)
|
||||
+ " structure, so that content is now plain paragraphs."
|
||||
+ remedy);
|
||||
}
|
||||
|
||||
/** Structure whose loss changes what a screen reader conveys, not just how it is nested. */
|
||||
private static final List<String> MEANINGFUL_TYPES =
|
||||
List.of("Table", "TH", "Formula", "L", "LI", "TOC", "Note");
|
||||
|
||||
private static Set<String> structureTypes(PDDocument document) {
|
||||
Set<String> types = new HashSet<>();
|
||||
try {
|
||||
PDStructureTreeRoot root = document.getDocumentCatalog().getStructureTreeRoot();
|
||||
if (root != null) {
|
||||
collectTypes(root.getKids(), types, 0);
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("Could not read structure types: {}", e.getMessage());
|
||||
}
|
||||
return types;
|
||||
}
|
||||
|
||||
private static void collectTypes(Object node, Set<String> types, int depth) {
|
||||
// Structure trees can be deep or, in damaged files, cyclic; cap rather than overflow.
|
||||
if (node == null || depth > 64) {
|
||||
return;
|
||||
}
|
||||
if (node instanceof List<?> list) {
|
||||
list.forEach(child -> collectTypes(child, types, depth + 1));
|
||||
} else if (node instanceof PDStructureElement element) {
|
||||
types.add(element.getStructureType());
|
||||
collectTypes(element.getKids(), types, depth + 1);
|
||||
}
|
||||
}
|
||||
|
||||
private static void stripStructure(PDDocument document) {
|
||||
PDDocumentCatalog catalog = document.getDocumentCatalog();
|
||||
catalog.getCOSObject().removeItem(COSName.getPDFName("StructTreeRoot"));
|
||||
catalog.getCOSObject().removeItem(COSName.getPDFName("MarkInfo"));
|
||||
document.getPages()
|
||||
.forEach(
|
||||
page ->
|
||||
page.getCOSObject()
|
||||
.removeItem(COSName.getPDFName("StructParents")));
|
||||
log.info("Removed existing structure tree before rebuilding");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.Set;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureElement;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureTreeRoot;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
/**
|
||||
* What the document said about itself before anything rewrote it. Font embedding shells out to
|
||||
* Ghostscript, which returns a file with no structure tree, no {@code /Lang} and no XMP, so a
|
||||
* tagger reading the rewritten document sees an untagged, language-less file and cannot tell that
|
||||
* anything was lost. These facts are captured from the original and carried past that stage.
|
||||
*
|
||||
* @param language the catalog {@code /Lang} the author declared, or null
|
||||
* @param structureTypes every structure element type the original tree contained
|
||||
* @param hasUsableTree whether the original had a structure tree worth preserving
|
||||
*/
|
||||
@Slf4j
|
||||
public record SourceFacts(String language, Set<String> structureTypes, boolean hasUsableTree) {
|
||||
|
||||
private static final int MAX_DEPTH = 64;
|
||||
|
||||
/** Facts for a document nothing has rewritten, used when font embedding did not run. */
|
||||
public static final SourceFacts NONE = new SourceFacts(null, Set.of(), false);
|
||||
|
||||
public static SourceFacts of(PDDocument document) {
|
||||
String language = null;
|
||||
Set<String> types = new HashSet<>();
|
||||
boolean usable = false;
|
||||
try {
|
||||
language = document.getDocumentCatalog().getLanguage();
|
||||
usable = PdfUaTagger.hasUsableStructureTree(document);
|
||||
PDStructureTreeRoot root = document.getDocumentCatalog().getStructureTreeRoot();
|
||||
if (root != null) {
|
||||
collect(root.getKids(), types, 0);
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("Could not read source facts: {}", e.getMessage());
|
||||
}
|
||||
return new SourceFacts(language, Set.copyOf(types), usable);
|
||||
}
|
||||
|
||||
private static void collect(Object node, Set<String> types, int depth) {
|
||||
// Damaged files can present a cyclic tree; cap rather than overflow the stack.
|
||||
if (node == null || depth > MAX_DEPTH) {
|
||||
return;
|
||||
}
|
||||
if (node instanceof java.util.List<?> list) {
|
||||
list.forEach(child -> collect(child, types, depth + 1));
|
||||
} else if (node instanceof PDStructureElement element) {
|
||||
types.add(element.getStructureType());
|
||||
collect(element.getKids(), types, depth + 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,134 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.function.Consumer;
|
||||
|
||||
import lombok.Getter;
|
||||
import lombok.Setter;
|
||||
|
||||
/**
|
||||
* One node of the derived logical structure: either page content (ranges of markable operator
|
||||
* ordinals) or child blocks. Containers with no content are pruned before serialisation.
|
||||
*/
|
||||
@Getter
|
||||
@Setter
|
||||
public class StructBlock {
|
||||
|
||||
/** A contiguous, inclusive run of markable operator ordinals within one page stream. */
|
||||
public record OrdinalRange(int start, int end) {
|
||||
public boolean contains(int ordinal) {
|
||||
return ordinal >= start && ordinal <= end;
|
||||
}
|
||||
|
||||
public int size() {
|
||||
return end - start + 1;
|
||||
}
|
||||
}
|
||||
|
||||
private StructType type;
|
||||
private ArtifactType artifactType;
|
||||
private int pageIndex;
|
||||
private BBox bbox = BBox.EMPTY;
|
||||
private String text = "";
|
||||
|
||||
private final List<OrdinalRange> ranges = new ArrayList<>();
|
||||
private final List<StructBlock> children = new ArrayList<>();
|
||||
|
||||
/** {@code /Alt} - required on Figure and Formula for PDF/UA. */
|
||||
private String alt;
|
||||
|
||||
/** {@code /ActualText} - replacement text for content whose glyphs do not spell the word. */
|
||||
private String actualText;
|
||||
|
||||
/** {@code /Lang} - set only where it differs from the document default. */
|
||||
private String lang;
|
||||
|
||||
/** {@code /Scope} on a TH: Row, Column or Both. */
|
||||
private String scope;
|
||||
|
||||
/** {@code /ListNumbering} on an L. */
|
||||
private String listNumbering;
|
||||
|
||||
/** Unique {@code /ID}, required on Note and FENote elements. */
|
||||
private String id;
|
||||
|
||||
/**
|
||||
* Marked content ids assigned during injection; one block yields several when split, since a
|
||||
* sequence must nest inside BT/ET and q/Q rather than straddle them.
|
||||
*/
|
||||
private final List<Integer> mcids = new ArrayList<>();
|
||||
|
||||
/** True when the source content was already inside a marked-content sequence. */
|
||||
private boolean preMarked;
|
||||
|
||||
public StructBlock(StructType type, int pageIndex) {
|
||||
this.type = type;
|
||||
this.pageIndex = pageIndex;
|
||||
}
|
||||
|
||||
public static StructBlock artifact(ArtifactType artifactType, int pageIndex) {
|
||||
StructBlock block = new StructBlock(StructType.ARTIFACT, pageIndex);
|
||||
block.artifactType = artifactType;
|
||||
return block;
|
||||
}
|
||||
|
||||
public StructBlock addChild(StructBlock child) {
|
||||
children.add(child);
|
||||
return this;
|
||||
}
|
||||
|
||||
public StructBlock addRange(int start, int end) {
|
||||
ranges.add(new OrdinalRange(start, end));
|
||||
return this;
|
||||
}
|
||||
|
||||
public boolean isArtifact() {
|
||||
return type == StructType.ARTIFACT;
|
||||
}
|
||||
|
||||
/** Depth-first walk over this block and all descendants. */
|
||||
public void visit(Consumer<StructBlock> visitor) {
|
||||
visitor.accept(this);
|
||||
for (StructBlock child : children) {
|
||||
child.visit(visitor);
|
||||
}
|
||||
}
|
||||
|
||||
/** Total number of ordinals owned by this block and its descendants. */
|
||||
public int contentCount() {
|
||||
int total = ranges.stream().mapToInt(OrdinalRange::size).sum();
|
||||
for (StructBlock child : children) {
|
||||
total += child.contentCount();
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
/** Concatenated text of this block and its descendants, in tree order. */
|
||||
public String collectText() {
|
||||
StringBuilder sb = new StringBuilder();
|
||||
visit(
|
||||
block -> {
|
||||
if (!block.text.isBlank()) {
|
||||
if (sb.length() > 0) {
|
||||
sb.append(' ');
|
||||
}
|
||||
sb.append(block.text.strip());
|
||||
}
|
||||
});
|
||||
return sb.toString();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String toString() {
|
||||
return type.tag()
|
||||
+ (artifactType != null ? "[" + artifactType.subtype() + "]" : "")
|
||||
+ "(p"
|
||||
+ pageIndex
|
||||
+ ", "
|
||||
+ ranges.size()
|
||||
+ " ranges, "
|
||||
+ children.size()
|
||||
+ " kids)";
|
||||
}
|
||||
}
|
||||
+295
@@ -0,0 +1,295 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import org.apache.pdfbox.cos.COSArray;
|
||||
import org.apache.pdfbox.cos.COSBase;
|
||||
import org.apache.pdfbox.cos.COSDictionary;
|
||||
import org.apache.pdfbox.cos.COSInteger;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.common.PDNumberTreeNode;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDMarkInfo;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDObjectReference;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureElement;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureTreeRoot;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.taggedpdf.PDListAttributeObject;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.taggedpdf.PDTableAttributeObject;
|
||||
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
|
||||
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationLink;
|
||||
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationWidget;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
/**
|
||||
* Serialises a {@link DocumentStructure} into a PDF structure tree. Must run after {@link
|
||||
* MarkedContentInjector}, which assigns the marked content ids this writer references.
|
||||
*/
|
||||
@Slf4j
|
||||
public class StructTreeWriter {
|
||||
|
||||
private static final COSName STRUCT_PARENT = COSName.getPDFName("StructParent");
|
||||
private static final COSName NUMS = COSName.getPDFName("Nums");
|
||||
private static final String PDF2_STANDARD_NAMESPACE = "http://iso.org/pdf2/ssn";
|
||||
|
||||
/** Per-page marked content id to owning element, built while walking the tree. */
|
||||
private final Map<Integer, Map<Integer, PDStructureElement>> mcidOwners = new LinkedHashMap<>();
|
||||
|
||||
private COSDictionary standardNamespace;
|
||||
private int nextParentKey;
|
||||
|
||||
public void write(PDDocument document, DocumentStructure structure, PdfUaProfile profile)
|
||||
throws IOException {
|
||||
PDStructureTreeRoot root = new PDStructureTreeRoot();
|
||||
PDStructureElement documentElement =
|
||||
new PDStructureElement(StructType.DOCUMENT.tag(), root);
|
||||
if (structure.getLanguage() != null) {
|
||||
documentElement.setLanguage(structure.getLanguage());
|
||||
}
|
||||
if (profile == PdfUaProfile.UA2) {
|
||||
applyNamespace(documentElement, document);
|
||||
}
|
||||
|
||||
for (StructBlock block : structure.getBlocks()) {
|
||||
if (block.isArtifact()) {
|
||||
continue;
|
||||
}
|
||||
PDStructureElement child = buildElement(document, block, documentElement, profile);
|
||||
if (child != null) {
|
||||
documentElement.appendKid(child);
|
||||
}
|
||||
}
|
||||
|
||||
root.appendKid(documentElement);
|
||||
buildParentTree(document, root);
|
||||
registerNamespaces(root);
|
||||
|
||||
PDMarkInfo markInfo = new PDMarkInfo();
|
||||
markInfo.setMarked(true);
|
||||
document.getDocumentCatalog().setMarkInfo(markInfo);
|
||||
document.getDocumentCatalog().setStructureTreeRoot(root);
|
||||
}
|
||||
|
||||
/** Recursively builds an element, returning null when the block carries no content at all. */
|
||||
private PDStructureElement buildElement(
|
||||
PDDocument document,
|
||||
StructBlock block,
|
||||
PDStructureElement parent,
|
||||
PdfUaProfile profile) {
|
||||
|
||||
// Prune on assigned MCIDs, not claimed ranges: form-XObject lines all resolve to one Do,
|
||||
// and emitting the losers would announce empty paragraphs to a screen reader.
|
||||
if (!carriesContent(block)) {
|
||||
return null;
|
||||
}
|
||||
StructType type = effectiveType(block, profile);
|
||||
PDStructureElement element = new PDStructureElement(type.tag(), parent);
|
||||
PDPage page = document.getPage(block.getPageIndex());
|
||||
element.setPage(page);
|
||||
|
||||
if (profile == PdfUaProfile.UA2) {
|
||||
applyNamespace(element, document);
|
||||
}
|
||||
applyAttributes(block, element);
|
||||
|
||||
for (int mcid : block.getMcids()) {
|
||||
element.appendKid(mcid);
|
||||
mcidOwners
|
||||
.computeIfAbsent(block.getPageIndex(), k -> new LinkedHashMap<>())
|
||||
.put(mcid, element);
|
||||
}
|
||||
|
||||
for (StructBlock child : block.getChildren()) {
|
||||
PDStructureElement childElement = buildElement(document, child, element, profile);
|
||||
if (childElement != null) {
|
||||
element.appendKid(childElement);
|
||||
}
|
||||
}
|
||||
return element;
|
||||
}
|
||||
|
||||
/** True when this block, or something beneath it, was actually given marked content. */
|
||||
private static boolean carriesContent(StructBlock block) {
|
||||
if (!block.getMcids().isEmpty()) {
|
||||
return true;
|
||||
}
|
||||
return block.getChildren().stream().anyMatch(StructTreeWriter::carriesContent);
|
||||
}
|
||||
|
||||
/** PDF/UA-2 replaces Note with FENote for footnotes. */
|
||||
private static StructType effectiveType(StructBlock block, PdfUaProfile profile) {
|
||||
if (profile == PdfUaProfile.UA2 && block.getType() == StructType.NOTE) {
|
||||
return StructType.FENOTE;
|
||||
}
|
||||
return block.getType();
|
||||
}
|
||||
|
||||
private static void applyAttributes(StructBlock block, PDStructureElement element) {
|
||||
if (block.getAlt() != null && !block.getAlt().isBlank()) {
|
||||
element.setAlternateDescription(block.getAlt());
|
||||
}
|
||||
if (block.getActualText() != null && !block.getActualText().isBlank()) {
|
||||
element.setActualText(block.getActualText());
|
||||
}
|
||||
if (block.getLang() != null && !block.getLang().isBlank()) {
|
||||
element.setLanguage(block.getLang());
|
||||
}
|
||||
if (block.getId() != null && !block.getId().isBlank()) {
|
||||
element.setElementIdentifier(block.getId());
|
||||
}
|
||||
if (block.getScope() != null) {
|
||||
PDTableAttributeObject table = new PDTableAttributeObject();
|
||||
table.setScope(block.getScope());
|
||||
element.addAttribute(table);
|
||||
}
|
||||
if (block.getListNumbering() != null) {
|
||||
PDListAttributeObject list = new PDListAttributeObject();
|
||||
list.setListNumbering(block.getListNumbering());
|
||||
element.addAttribute(list);
|
||||
}
|
||||
}
|
||||
|
||||
/** PDF/UA-2 requires every element to declare the standard structure namespace. */
|
||||
private void applyNamespace(PDStructureElement element, PDDocument document) {
|
||||
element.getCOSObject().setItem(COSName.getPDFName("NS"), standardNamespace());
|
||||
}
|
||||
|
||||
/** The PDF 2.0 standard structure namespace, created once per document. */
|
||||
private COSDictionary standardNamespace() {
|
||||
if (standardNamespace == null) {
|
||||
standardNamespace = new COSDictionary();
|
||||
standardNamespace.setName(COSName.TYPE, "Namespace");
|
||||
standardNamespace.setString(COSName.getPDFName("NS"), PDF2_STANDARD_NAMESPACE);
|
||||
}
|
||||
return standardNamespace;
|
||||
}
|
||||
|
||||
private void registerNamespaces(PDStructureTreeRoot root) {
|
||||
if (standardNamespace == null) {
|
||||
return;
|
||||
}
|
||||
COSArray namespaces = new COSArray();
|
||||
namespaces.add(standardNamespace);
|
||||
root.getCOSObject().setItem(COSName.getPDFName("Namespaces"), namespaces);
|
||||
}
|
||||
|
||||
/**
|
||||
* Builds {@code /ParentTree}: per page, an array indexed by marked content id keyed on {@code
|
||||
* /StructParents}, plus one entry per annotation keyed on {@code /StructParent}.
|
||||
*/
|
||||
private void buildParentTree(PDDocument document, PDStructureTreeRoot root) {
|
||||
COSArray nums = new COSArray();
|
||||
nextParentKey = 0;
|
||||
|
||||
for (int pageIndex = 0; pageIndex < document.getNumberOfPages(); pageIndex++) {
|
||||
Map<Integer, PDStructureElement> owners = mcidOwners.get(pageIndex);
|
||||
if (owners == null || owners.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
PDPage page = document.getPage(pageIndex);
|
||||
int key = nextParentKey++;
|
||||
page.setStructParents(key);
|
||||
|
||||
int maxMcid = owners.keySet().stream().mapToInt(Integer::intValue).max().orElse(-1);
|
||||
COSArray entries = new COSArray();
|
||||
for (int mcid = 0; mcid <= maxMcid; mcid++) {
|
||||
PDStructureElement owner = owners.get(mcid);
|
||||
entries.add(
|
||||
owner != null ? owner.getCOSObject() : org.apache.pdfbox.cos.COSNull.NULL);
|
||||
}
|
||||
nums.add(COSInteger.get(key));
|
||||
nums.add(entries);
|
||||
}
|
||||
|
||||
List<COSBase> annotationEntries = tagAnnotations(document, root);
|
||||
for (int i = 0; i + 1 < annotationEntries.size(); i += 2) {
|
||||
nums.add(annotationEntries.get(i));
|
||||
nums.add(annotationEntries.get(i + 1));
|
||||
}
|
||||
|
||||
COSDictionary parentTreeDict = new COSDictionary();
|
||||
parentTreeDict.setItem(NUMS, nums);
|
||||
root.setParentTree(new PDNumberTreeNode(parentTreeDict, PDStructureElement.class));
|
||||
root.setParentTreeNextKey(nextParentKey);
|
||||
}
|
||||
|
||||
/**
|
||||
* Clause 7.18: every visible annotation needs a structure element so it is reachable from the
|
||||
* tree. Links become Link elements, anything else an Annot.
|
||||
*/
|
||||
private List<COSBase> tagAnnotations(PDDocument document, PDStructureTreeRoot root) {
|
||||
List<COSBase> entries = new ArrayList<>();
|
||||
PDStructureElement documentElement = firstDocumentElement(root);
|
||||
if (documentElement == null) {
|
||||
return entries;
|
||||
}
|
||||
for (int pageIndex = 0; pageIndex < document.getNumberOfPages(); pageIndex++) {
|
||||
PDPage page = document.getPage(pageIndex);
|
||||
List<PDAnnotation> annotations;
|
||||
try {
|
||||
annotations = page.getAnnotations();
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not read annotations on page {}: {}", pageIndex, e.getMessage());
|
||||
continue;
|
||||
}
|
||||
for (PDAnnotation annotation : annotations) {
|
||||
if (annotation == null
|
||||
|| annotation.isHidden()
|
||||
|| annotation.isNoView()
|
||||
|| "Popup".equals(annotation.getSubtype())) {
|
||||
continue;
|
||||
}
|
||||
PDStructureElement element =
|
||||
new PDStructureElement(annotationType(annotation), documentElement);
|
||||
element.setPage(page);
|
||||
|
||||
PDObjectReference reference = new PDObjectReference();
|
||||
reference.setReferencedObject(annotation);
|
||||
element.appendKid(reference);
|
||||
documentElement.appendKid(element);
|
||||
|
||||
int key = nextParentKey++;
|
||||
annotation.getCOSObject().setInt(STRUCT_PARENT, key);
|
||||
entries.add(COSInteger.get(key));
|
||||
entries.add(element.getCOSObject());
|
||||
|
||||
if (annotation.getContents() == null || annotation.getContents().isBlank()) {
|
||||
annotation.setContents(defaultContents(annotation));
|
||||
}
|
||||
}
|
||||
}
|
||||
return entries;
|
||||
}
|
||||
|
||||
/** Clause 7.18.4: widgets need a Form element, links a Link element, everything else Annot. */
|
||||
private static String annotationType(PDAnnotation annotation) {
|
||||
if (annotation instanceof PDAnnotationWidget) {
|
||||
return StructType.FORM.tag();
|
||||
}
|
||||
if (annotation instanceof PDAnnotationLink) {
|
||||
return StructType.LINK.tag();
|
||||
}
|
||||
return "Annot";
|
||||
}
|
||||
|
||||
private static String defaultContents(PDAnnotation annotation) {
|
||||
if (annotation instanceof PDAnnotationLink link && link.getAction() != null) {
|
||||
return "Link";
|
||||
}
|
||||
return annotation.getSubtype() == null ? "Annotation" : annotation.getSubtype();
|
||||
}
|
||||
|
||||
private static PDStructureElement firstDocumentElement(PDStructureTreeRoot root) {
|
||||
for (Object kid : root.getKids()) {
|
||||
if (kid instanceof PDStructureElement element) {
|
||||
return element;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
/**
|
||||
* PDF standard structure types emitted by the tagger (ISO 32000-1 14.8.4), limited to the PDF/UA
|
||||
* subset. {@link #ARTIFACT} is not one: it marks content in the stream and stays out of the tree.
|
||||
*/
|
||||
public enum StructType {
|
||||
DOCUMENT("Document"),
|
||||
PART("Part"),
|
||||
SECT("Sect"),
|
||||
H1("H1"),
|
||||
H2("H2"),
|
||||
H3("H3"),
|
||||
H4("H4"),
|
||||
H5("H5"),
|
||||
H6("H6"),
|
||||
P("P"),
|
||||
L("L"),
|
||||
LI("LI"),
|
||||
LBL("Lbl"),
|
||||
LBODY("LBody"),
|
||||
TABLE("Table"),
|
||||
TR("TR"),
|
||||
TH("TH"),
|
||||
TD("TD"),
|
||||
FIGURE("Figure"),
|
||||
CAPTION("Caption"),
|
||||
FORMULA("Formula"),
|
||||
NOTE("Note"),
|
||||
FENOTE("FENote"),
|
||||
LINK("Link"),
|
||||
/** Wraps a widget annotation; PDF/UA-1 clause 7.18.4 requires widgets to sit inside one. */
|
||||
FORM("Form"),
|
||||
SPAN("Span"),
|
||||
ARTIFACT("Artifact");
|
||||
|
||||
private final String tag;
|
||||
|
||||
StructType(String tag) {
|
||||
this.tag = tag;
|
||||
}
|
||||
|
||||
/** The name written into the PDF {@code /S} entry. */
|
||||
public String tag() {
|
||||
return tag;
|
||||
}
|
||||
|
||||
public boolean isHeading() {
|
||||
return this == H1 || this == H2 || this == H3 || this == H4 || this == H5 || this == H6;
|
||||
}
|
||||
|
||||
/** Heading level 1-6, or 0 when this is not a heading. */
|
||||
public int headingLevel() {
|
||||
return isHeading() ? ordinal() - H1.ordinal() + 1 : 0;
|
||||
}
|
||||
|
||||
/** The heading type for a 1-based level, clamped to the H1-H6 range. */
|
||||
public static StructType heading(int level) {
|
||||
int clamped = Math.max(1, Math.min(6, level));
|
||||
return values()[H1.ordinal() + clamped - 1];
|
||||
}
|
||||
}
|
||||
+630
@@ -0,0 +1,630 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.Writer;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.IdentityHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import org.apache.pdfbox.contentstream.operator.Operator;
|
||||
import org.apache.pdfbox.cos.COSBase;
|
||||
import org.apache.pdfbox.cos.COSDictionary;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.pdfparser.PDFStreamParser;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.PDResources;
|
||||
import org.apache.pdfbox.pdmodel.common.PDRectangle;
|
||||
import org.apache.pdfbox.pdmodel.font.PDFontDescriptor;
|
||||
import org.apache.pdfbox.pdmodel.font.PDType3Font;
|
||||
import org.apache.pdfbox.pdmodel.graphics.PDXObject;
|
||||
import org.apache.pdfbox.pdmodel.graphics.form.PDFormXObject;
|
||||
import org.apache.pdfbox.pdmodel.graphics.form.PDTransparencyGroup;
|
||||
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
|
||||
import org.apache.pdfbox.text.PDFTextStripper;
|
||||
import org.apache.pdfbox.text.TextPosition;
|
||||
import org.apache.pdfbox.util.Matrix;
|
||||
import org.apache.pdfbox.util.Vector;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
/**
|
||||
* Extracts text lines and graphic ops from page streams, tagging each with its operator ordinal.
|
||||
* Both passes count the same operators in the same order, so ordinals cross-reference.
|
||||
*/
|
||||
@Slf4j
|
||||
public class TaggedContentExtractor {
|
||||
|
||||
/** Glyph size below which a run is treated as noise rather than a line. */
|
||||
private static final float MIN_FONT_SIZE = 0.5f;
|
||||
|
||||
public List<PageContent> extract(PDDocument document) throws IOException {
|
||||
LineCollector collector = new LineCollector();
|
||||
collector.setSortByPosition(true);
|
||||
collector.setStartPage(1);
|
||||
collector.setEndPage(document.getNumberOfPages());
|
||||
collector.writeText(document, Writer.nullWriter());
|
||||
|
||||
List<PageContent> pages = new ArrayList<>(document.getNumberOfPages());
|
||||
for (int i = 0; i < document.getNumberOfPages(); i++) {
|
||||
PDPage page = document.getPage(i);
|
||||
List<MarkableOp> ops = collector.opsFor(i);
|
||||
List<TextLineInfo> lines = collector.linesFor(i);
|
||||
boolean dropped = false;
|
||||
if (ops.size() < maxOrdinal(lines) + 1) {
|
||||
// Untrusted ordinals: drop the lines so the page is untaggable rather than
|
||||
// mis-tagged, and flag it so the caller refuses to declare conformance.
|
||||
log.warn(
|
||||
"Ordinal mismatch on page {} (ops={}, text={}); skipping page",
|
||||
i,
|
||||
ops.size(),
|
||||
maxOrdinal(lines) + 1);
|
||||
dropped = !lines.isEmpty();
|
||||
lines = List.of();
|
||||
}
|
||||
pages.add(
|
||||
new PageContent(
|
||||
i,
|
||||
lines,
|
||||
ops,
|
||||
ops.size(),
|
||||
collector.preMarkedOn(i),
|
||||
collector.textSemanticsOn(i),
|
||||
dropped,
|
||||
normalisedBox(page)));
|
||||
}
|
||||
return pages;
|
||||
}
|
||||
|
||||
/**
|
||||
* The page box in the space of extracted line coordinates: origin-zero, width and height
|
||||
* swapped for 90/270 rotations, because the text engine reports in the rotated frame.
|
||||
*/
|
||||
static BBox normalisedBox(PDPage page) {
|
||||
PDRectangle mediaBox = page.getMediaBox();
|
||||
boolean sideways = page.getRotation() % 180 != 0;
|
||||
float width = sideways ? mediaBox.getHeight() : mediaBox.getWidth();
|
||||
float height = sideways ? mediaBox.getWidth() : mediaBox.getHeight();
|
||||
return new BBox(0, 0, width, height);
|
||||
}
|
||||
|
||||
/** Counts images with the token scan alone, skipping the expensive text pass. */
|
||||
public int countGraphics(PDDocument document) {
|
||||
int total = 0;
|
||||
for (int i = 0; i < document.getNumberOfPages(); i++) {
|
||||
try {
|
||||
PDResources resources = document.getPage(i).getResources();
|
||||
PDFStreamParser parser = new PDFStreamParser(document.getPage(i));
|
||||
List<COSBase> operands = new ArrayList<>();
|
||||
Object token;
|
||||
while ((token = parser.parseNextToken()) != null) {
|
||||
if (!(token instanceof Operator operator)) {
|
||||
operands.add((COSBase) token);
|
||||
continue;
|
||||
}
|
||||
if (isGraphicOperator(operator.getName(), operands, resources)) {
|
||||
total++;
|
||||
}
|
||||
operands.clear();
|
||||
}
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not scan page {} for graphics: {}", i, e.getMessage());
|
||||
}
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
/** True for an inline image, or a Do that resolves to an image XObject. */
|
||||
private static boolean isGraphicOperator(
|
||||
String name, List<COSBase> operands, PDResources resources) {
|
||||
if ("BI".equals(name)) {
|
||||
return true;
|
||||
}
|
||||
if (!"Do".equals(name) || resources == null || operands.size() != 1) {
|
||||
return false;
|
||||
}
|
||||
if (!(operands.get(0) instanceof COSName resourceName)) {
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
return resources.getXObject(resourceName) instanceof PDImageXObject;
|
||||
} catch (IOException e) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
private static int maxOrdinal(List<TextLineInfo> lines) {
|
||||
return lines.stream().mapToInt(TextLineInfo::endOrdinal).max().orElse(-1);
|
||||
}
|
||||
|
||||
static BBox toBBox(PDRectangle rect) {
|
||||
return new BBox(
|
||||
rect.getLowerLeftX(),
|
||||
rect.getLowerLeftY(),
|
||||
rect.getUpperRightX(),
|
||||
rect.getUpperRightY());
|
||||
}
|
||||
|
||||
// --- Operator classification -------------------------------------------
|
||||
|
||||
private static boolean isPathConstruction(String name) {
|
||||
return switch (name) {
|
||||
case "m", "l", "c", "v", "y", "re" -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
/** True when a sequence carries replacement or alternative text, which a rebuild would drop. */
|
||||
private static boolean carriesTextSemantics(List<COSBase> operands) {
|
||||
for (COSBase operand : operands) {
|
||||
if (operand instanceof COSDictionary dictionary
|
||||
&& (dictionary.containsKey(COSName.getPDFName("ActualText"))
|
||||
|| dictionary.containsKey(COSName.getPDFName("Alt"))
|
||||
|| dictionary.containsKey(COSName.E))) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Describes one markable operator, placed with the engine's own matrix rather than a
|
||||
* hand-rolled q/Q/cm stack that would get nesting and form matrices wrong.
|
||||
*/
|
||||
private static MarkableOp classify(
|
||||
String name,
|
||||
List<COSBase> operands,
|
||||
PDResources resources,
|
||||
Matrix ctm,
|
||||
BBox pathBox,
|
||||
int ordinal) {
|
||||
|
||||
if ("BI".equals(name)) {
|
||||
return new MarkableOp(ordinal, MarkableOp.Kind.INLINE_IMAGE, unitSquare(ctm), null);
|
||||
}
|
||||
if (MarkableOp.isPathPainting(name)) {
|
||||
return new MarkableOp(ordinal, MarkableOp.Kind.VECTOR, pathBox, null);
|
||||
}
|
||||
if (!"Do".equals(name)) {
|
||||
return new MarkableOp(ordinal, MarkableOp.Kind.TEXT, BBox.EMPTY, null);
|
||||
}
|
||||
COSName resourceName =
|
||||
operands.size() == 1 && operands.get(0) instanceof COSName n ? n : null;
|
||||
if (resourceName == null || resources == null) {
|
||||
return new MarkableOp(ordinal, MarkableOp.Kind.FORM, unitSquare(ctm), null);
|
||||
}
|
||||
try {
|
||||
PDXObject xobject = resources.getXObject(resourceName);
|
||||
if (xobject instanceof PDImageXObject) {
|
||||
return new MarkableOp(
|
||||
ordinal, MarkableOp.Kind.IMAGE, unitSquare(ctm), resourceName.getName());
|
||||
}
|
||||
if (xobject instanceof PDFormXObject form) {
|
||||
return new MarkableOp(
|
||||
ordinal, MarkableOp.Kind.FORM, formBox(form, ctm), resourceName.getName());
|
||||
}
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not resolve XObject {}: {}", resourceName.getName(), e.getMessage());
|
||||
}
|
||||
return new MarkableOp(
|
||||
ordinal, MarkableOp.Kind.FORM, unitSquare(ctm), resourceName.getName());
|
||||
}
|
||||
|
||||
/**
|
||||
* Extends the running path box with one path-construction operator's points; without it every
|
||||
* vector had an empty box and charts and vector logos vanished from the structure tree.
|
||||
*/
|
||||
private static BBox extendPath(BBox current, String name, List<COSBase> operands, Matrix ctm) {
|
||||
int pairs =
|
||||
switch (name) {
|
||||
case "m", "l" -> 1;
|
||||
case "re" -> 2;
|
||||
case "v", "y" -> 2;
|
||||
case "c" -> 3;
|
||||
default -> 0;
|
||||
};
|
||||
if (pairs == 0 || operands.size() < pairs * 2) {
|
||||
return current;
|
||||
}
|
||||
|
||||
// Deliberately allocation-free; the obvious version cost a third of the extraction budget.
|
||||
float minX = current.isEmpty() ? Float.MAX_VALUE : current.x0();
|
||||
float minY = current.isEmpty() ? Float.MAX_VALUE : current.y0();
|
||||
float maxX = current.isEmpty() ? -Float.MAX_VALUE : current.x1();
|
||||
float maxY = current.isEmpty() ? -Float.MAX_VALUE : current.y1();
|
||||
|
||||
for (int pair = 0; pair < pairs; pair++) {
|
||||
Float x = numberAt(operands, pair * 2);
|
||||
Float y = numberAt(operands, pair * 2 + 1);
|
||||
if (x == null || y == null) {
|
||||
continue;
|
||||
}
|
||||
float px = x;
|
||||
float py = y;
|
||||
// "re" gives origin plus size, so the second pair is a corner offset from the first.
|
||||
if ("re".equals(name) && pair == 1) {
|
||||
Float ox = numberAt(operands, 0);
|
||||
Float oy = numberAt(operands, 1);
|
||||
if (ox == null || oy == null) {
|
||||
continue;
|
||||
}
|
||||
px = ox + x;
|
||||
py = oy + y;
|
||||
}
|
||||
float tx = ctm.getScaleX() * px + ctm.getShearX() * py + ctm.getTranslateX();
|
||||
float ty = ctm.getShearY() * px + ctm.getScaleY() * py + ctm.getTranslateY();
|
||||
minX = Math.min(minX, tx);
|
||||
minY = Math.min(minY, ty);
|
||||
maxX = Math.max(maxX, tx);
|
||||
maxY = Math.max(maxY, ty);
|
||||
}
|
||||
return maxX < minX ? current : new BBox(minX, minY, maxX, maxY);
|
||||
}
|
||||
|
||||
private static Float numberAt(List<COSBase> operands, int index) {
|
||||
return index < operands.size()
|
||||
&& operands.get(index) instanceof org.apache.pdfbox.cos.COSNumber number
|
||||
? number.floatValue()
|
||||
: null;
|
||||
}
|
||||
|
||||
/** The unit square mapped through the CTM, which is how images are placed. */
|
||||
private static BBox unitSquare(Matrix ctm) {
|
||||
return transformBox(new BBox(0, 0, 1, 1), ctm);
|
||||
}
|
||||
|
||||
private static BBox formBox(PDFormXObject form, Matrix ctm) {
|
||||
PDRectangle box = form.getBBox();
|
||||
if (box == null) {
|
||||
return unitSquare(ctm);
|
||||
}
|
||||
Matrix combined = form.getMatrix() != null ? form.getMatrix().multiply(ctm) : ctm;
|
||||
return transformBox(toBBox(box), combined);
|
||||
}
|
||||
|
||||
private static BBox transformBox(BBox box, Matrix m) {
|
||||
float[] xs = new float[4];
|
||||
float[] ys = new float[4];
|
||||
float[][] corners = {
|
||||
{box.x0(), box.y0()}, {box.x1(), box.y0()},
|
||||
{box.x0(), box.y1()}, {box.x1(), box.y1()}
|
||||
};
|
||||
for (int i = 0; i < 4; i++) {
|
||||
Vector v = m.transform(new Vector(corners[i][0], corners[i][1]));
|
||||
xs[i] = v.getX();
|
||||
ys[i] = v.getY();
|
||||
}
|
||||
float minX = Math.min(Math.min(xs[0], xs[1]), Math.min(xs[2], xs[3]));
|
||||
float maxX = Math.max(Math.max(xs[0], xs[1]), Math.max(xs[2], xs[3]));
|
||||
float minY = Math.min(Math.min(ys[0], ys[1]), Math.min(ys[2], ys[3]));
|
||||
float maxY = Math.max(Math.max(ys[0], ys[1]), Math.max(ys[2], ys[3]));
|
||||
return new BBox(minX, minY, maxX, maxY);
|
||||
}
|
||||
|
||||
// --- Text pass ---------------------------------------------------------
|
||||
|
||||
/** Marker recorded for each glyph so a finished line knows where it came from. */
|
||||
private record GlyphOrigin(int ordinal, boolean marked) {}
|
||||
|
||||
private static final class LineCollector extends PDFTextStripper {
|
||||
|
||||
private final Map<Integer, List<TextLineInfo>> byPage = new HashMap<>();
|
||||
private final Map<Integer, List<MarkableOp>> opsByPage = new HashMap<>();
|
||||
private final Map<Integer, Boolean> preMarkedByPage = new HashMap<>();
|
||||
private final Map<Integer, Boolean> textSemanticsByPage = new HashMap<>();
|
||||
private final Map<TextPosition, GlyphOrigin> origins = new IdentityHashMap<>();
|
||||
private final List<TextPosition> lineBuffer = new ArrayList<>();
|
||||
private final List<WordInfo> lineWords = new ArrayList<>();
|
||||
private final StringBuilder lineText = new StringBuilder();
|
||||
|
||||
private int ordinal = -1;
|
||||
private int markedDepth;
|
||||
private int nestedDepth;
|
||||
private BBox pathBox = BBox.EMPTY;
|
||||
private int syntheticDepth;
|
||||
private float pageHeight;
|
||||
private int pageIndex;
|
||||
|
||||
LineCollector() throws IOException {
|
||||
super();
|
||||
}
|
||||
|
||||
List<TextLineInfo> linesFor(int index) {
|
||||
return byPage.getOrDefault(index, List.of());
|
||||
}
|
||||
|
||||
List<MarkableOp> opsFor(int index) {
|
||||
return opsByPage.getOrDefault(index, List.of());
|
||||
}
|
||||
|
||||
boolean preMarkedOn(int index) {
|
||||
return preMarkedByPage.getOrDefault(index, false);
|
||||
}
|
||||
|
||||
boolean textSemanticsOn(int index) {
|
||||
return textSemanticsByPage.getOrDefault(index, false);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void startPage(PDPage page) throws IOException {
|
||||
ordinal = -1;
|
||||
markedDepth = 0;
|
||||
nestedDepth = 0;
|
||||
syntheticDepth = 0;
|
||||
pathBox = BBox.EMPTY;
|
||||
origins.clear();
|
||||
lineBuffer.clear();
|
||||
lineWords.clear();
|
||||
lineText.setLength(0);
|
||||
// Dir-adjusted glyph coordinates live in the rotated frame, so the flip must too.
|
||||
pageHeight = normalisedBox(page).height();
|
||||
pageIndex = getCurrentPageNo() - 1;
|
||||
super.startPage(page);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void endPage(PDPage page) throws IOException {
|
||||
flushLine();
|
||||
super.endPage(page);
|
||||
}
|
||||
|
||||
/**
|
||||
* Counts only operators physically present in the page's own stream: PDFBox re-enters here
|
||||
* with synthetic calls for {@code '} and {@code "}, and descends into form XObjects.
|
||||
*/
|
||||
@Override
|
||||
protected void processOperator(Operator operator, List<COSBase> operands)
|
||||
throws IOException {
|
||||
String name = operator.getName();
|
||||
if (nestedDepth == 0 && syntheticDepth == 0) {
|
||||
if (isPathConstruction(name)) {
|
||||
pathBox =
|
||||
extendPath(
|
||||
pathBox,
|
||||
name,
|
||||
operands,
|
||||
getGraphicsState().getCurrentTransformationMatrix());
|
||||
}
|
||||
if (MarkableOp.isMarkableOperator(name)) {
|
||||
ordinal++;
|
||||
// Classified here rather than in a second parse of the same stream: the engine
|
||||
// already has the operands and the live transformation matrix.
|
||||
opsByPage
|
||||
.computeIfAbsent(pageIndex, k -> new ArrayList<>())
|
||||
.add(
|
||||
classify(
|
||||
name,
|
||||
operands,
|
||||
getResources(),
|
||||
getGraphicsState().getCurrentTransformationMatrix(),
|
||||
pathBox,
|
||||
ordinal));
|
||||
if (MarkableOp.isPathPainting(name)) {
|
||||
pathBox = BBox.EMPTY;
|
||||
}
|
||||
} else if ("n".equals(name)) {
|
||||
pathBox = BBox.EMPTY;
|
||||
} else if ("BDC".equals(name) || "BMC".equals(name)) {
|
||||
markedDepth++;
|
||||
preMarkedByPage.put(pageIndex, true);
|
||||
if (carriesTextSemantics(operands)) {
|
||||
textSemanticsByPage.put(pageIndex, true);
|
||||
}
|
||||
} else if ("EMC".equals(name) && markedDepth > 0) {
|
||||
markedDepth--;
|
||||
}
|
||||
}
|
||||
boolean synthesises = "'".equals(name) || "\"".equals(name);
|
||||
if (synthesises) {
|
||||
syntheticDepth++;
|
||||
}
|
||||
try {
|
||||
super.processOperator(operator, operands);
|
||||
} finally {
|
||||
if (synthesises) {
|
||||
syntheticDepth--;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void showForm(PDFormXObject form) throws IOException {
|
||||
nestedDepth++;
|
||||
try {
|
||||
super.showForm(form);
|
||||
} finally {
|
||||
nestedDepth--;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void showTransparencyGroup(PDTransparencyGroup group) throws IOException {
|
||||
nestedDepth++;
|
||||
try {
|
||||
super.showTransparencyGroup(group);
|
||||
} finally {
|
||||
nestedDepth--;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void showType3Glyph(
|
||||
Matrix textRenderingMatrix,
|
||||
PDType3Font font,
|
||||
int code,
|
||||
org.apache.pdfbox.util.Vector displacement)
|
||||
throws IOException {
|
||||
nestedDepth++;
|
||||
try {
|
||||
super.showType3Glyph(textRenderingMatrix, font, code, displacement);
|
||||
} finally {
|
||||
nestedDepth--;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void processChildStream(
|
||||
org.apache.pdfbox.contentstream.PDContentStream contentStream, PDPage page)
|
||||
throws IOException {
|
||||
nestedDepth++;
|
||||
try {
|
||||
super.processChildStream(contentStream, page);
|
||||
} finally {
|
||||
nestedDepth--;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void processTextPosition(TextPosition text) {
|
||||
origins.put(text, new GlyphOrigin(ordinal, markedDepth > 0));
|
||||
super.processTextPosition(text);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void writeString(String text, List<TextPosition> positions) {
|
||||
lineText.append(text);
|
||||
lineBuffer.addAll(positions);
|
||||
WordInfo word = buildWord(text, positions);
|
||||
if (word != null) {
|
||||
lineWords.add(word);
|
||||
}
|
||||
}
|
||||
|
||||
private WordInfo buildWord(String text, List<TextPosition> positions) {
|
||||
if (text == null || text.isBlank() || positions.isEmpty()) {
|
||||
return null;
|
||||
}
|
||||
Bounds bounds = new Bounds();
|
||||
for (TextPosition tp : positions) {
|
||||
bounds.accept(tp, pageHeight, origins.get(tp));
|
||||
}
|
||||
if (bounds.end < 0) {
|
||||
return null;
|
||||
}
|
||||
return new WordInfo(
|
||||
text,
|
||||
bounds.box(),
|
||||
bounds.start,
|
||||
bounds.end,
|
||||
bounds.dominantSize(),
|
||||
bounds.bold);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void writeWordSeparator() {
|
||||
lineText.append(' ');
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void writeLineSeparator() {
|
||||
flushLine();
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void writeParagraphSeparator() {
|
||||
flushLine();
|
||||
}
|
||||
|
||||
private void flushLine() {
|
||||
if (lineBuffer.isEmpty()) {
|
||||
lineText.setLength(0);
|
||||
lineWords.clear();
|
||||
return;
|
||||
}
|
||||
TextLineInfo line = buildLine();
|
||||
lineBuffer.clear();
|
||||
lineWords.clear();
|
||||
lineText.setLength(0);
|
||||
if (line != null) {
|
||||
byPage.computeIfAbsent(pageIndex, k -> new ArrayList<>()).add(line);
|
||||
}
|
||||
}
|
||||
|
||||
private TextLineInfo buildLine() {
|
||||
String text = lineText.toString();
|
||||
if (text.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
Bounds bounds = new Bounds();
|
||||
for (TextPosition tp : lineBuffer) {
|
||||
bounds.accept(tp, pageHeight, origins.get(tp));
|
||||
}
|
||||
if (bounds.end < 0) {
|
||||
return null;
|
||||
}
|
||||
return new TextLineInfo(
|
||||
pageIndex,
|
||||
text,
|
||||
bounds.box(),
|
||||
bounds.dominantSize(),
|
||||
bounds.bold,
|
||||
bounds.start,
|
||||
bounds.end,
|
||||
bounds.marked,
|
||||
List.copyOf(lineWords));
|
||||
}
|
||||
}
|
||||
|
||||
/** Accumulates glyph geometry, ordinals and font signals for a word or a line. */
|
||||
private static final class Bounds {
|
||||
private float minX = Float.MAX_VALUE;
|
||||
private float maxX = -Float.MAX_VALUE;
|
||||
private float minY = Float.MAX_VALUE;
|
||||
private float maxY = -Float.MAX_VALUE;
|
||||
private int start = Integer.MAX_VALUE;
|
||||
private int end = -1;
|
||||
private boolean marked;
|
||||
private boolean bold;
|
||||
private final Map<Float, Integer> sizeCounts = new HashMap<>();
|
||||
|
||||
void accept(TextPosition tp, float pageHeight, GlyphOrigin origin) {
|
||||
float top = pageHeight - tp.getYDirAdj();
|
||||
float bottom = top - Math.max(tp.getHeightDir(), 0);
|
||||
minX = Math.min(minX, tp.getXDirAdj());
|
||||
maxX = Math.max(maxX, tp.getXDirAdj() + tp.getWidthDirAdj());
|
||||
minY = Math.min(minY, bottom);
|
||||
maxY = Math.max(maxY, top);
|
||||
|
||||
if (origin != null) {
|
||||
start = Math.min(start, origin.ordinal());
|
||||
end = Math.max(end, origin.ordinal());
|
||||
marked |= origin.marked();
|
||||
}
|
||||
float size = tp.getFontSizeInPt();
|
||||
if (size > MIN_FONT_SIZE) {
|
||||
sizeCounts.merge(round(size), 1, Integer::sum);
|
||||
}
|
||||
bold |= isBold(tp);
|
||||
}
|
||||
|
||||
BBox box() {
|
||||
return new BBox(minX, minY, maxX, maxY);
|
||||
}
|
||||
|
||||
float dominantSize() {
|
||||
return sizeCounts.entrySet().stream()
|
||||
.max(Map.Entry.comparingByValue())
|
||||
.map(Map.Entry::getKey)
|
||||
.orElse(0f);
|
||||
}
|
||||
|
||||
private static float round(float value) {
|
||||
return Math.round(value * 10f) / 10f;
|
||||
}
|
||||
|
||||
private static boolean isBold(TextPosition tp) {
|
||||
if (tp.getFont() == null) {
|
||||
return false;
|
||||
}
|
||||
String name = tp.getFont().getName();
|
||||
if (name != null && name.toLowerCase().contains("bold")) {
|
||||
return true;
|
||||
}
|
||||
PDFontDescriptor descriptor = tp.getFont().getFontDescriptor();
|
||||
return descriptor != null
|
||||
&& (descriptor.getFontWeight() >= 600 || descriptor.isForceBold());
|
||||
}
|
||||
}
|
||||
}
|
||||
+63
@@ -0,0 +1,63 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
import lombok.Builder;
|
||||
import lombok.Getter;
|
||||
|
||||
/** Inputs that change how a document is tagged. */
|
||||
@Getter
|
||||
@Builder(toBuilder = true)
|
||||
public class TaggingOptions {
|
||||
|
||||
/** What to do when the source already has a structure tree. */
|
||||
public enum ExistingTags {
|
||||
/** Leave the tree alone and fix only document-level requirements. */
|
||||
KEEP,
|
||||
/** Discard the tree and derive a new one. */
|
||||
REBUILD,
|
||||
/** Keep a usable tree, rebuild an empty or trivially broken one. */
|
||||
AUTO
|
||||
}
|
||||
|
||||
/** How images with no alternative description are handled. */
|
||||
public enum FigurePolicy {
|
||||
/** Leave undescribed so validation fails honestly; a faked {@code /Alt} helps nobody. */
|
||||
REQUIRE_ALT,
|
||||
/** Treat every image as decoration and mark it as an artifact. */
|
||||
MARK_DECORATIVE
|
||||
}
|
||||
|
||||
@Builder.Default private PdfUaProfile profile = PdfUaProfile.UA1;
|
||||
|
||||
/** BCP-47 language tag for the document, for example {@code en-GB}. */
|
||||
private String language;
|
||||
|
||||
/** Replace a language the document already declares. Off, so a French file stays French. */
|
||||
@Builder.Default private boolean overrideLanguage = false;
|
||||
|
||||
private String title;
|
||||
|
||||
/** Last resort when no title is given and none can be derived; pass the uploaded filename. */
|
||||
private String fallbackTitle;
|
||||
|
||||
/** Embed any font the document references but does not carry, which clause 7.21 requires. */
|
||||
@Builder.Default private boolean embedFonts = true;
|
||||
|
||||
/** Leave the PDF version alone; raising it would break PDF/A-1, defined on PDF 1.4. */
|
||||
@Builder.Default private boolean preservePdfVersion = false;
|
||||
|
||||
@Builder.Default private ExistingTags existingTags = ExistingTags.AUTO;
|
||||
|
||||
@Builder.Default private FigurePolicy figurePolicy = FigurePolicy.REQUIRE_ALT;
|
||||
|
||||
/** Alternative descriptions supplied by the caller, keyed by "pageIndex:ordinal". */
|
||||
@Builder.Default private Map<String, String> altTextByFigure = Map.of();
|
||||
|
||||
/** What the document said before font embedding rewrote it; see {@link SourceFacts}. */
|
||||
@Builder.Default private SourceFacts sourceFacts = SourceFacts.NONE;
|
||||
|
||||
public String altTextFor(int pageIndex, int ordinal) {
|
||||
return altTextByFigure.get(pageIndex + ":" + ordinal);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import lombok.Getter;
|
||||
|
||||
/** What a tagging run produced, for the conversion report. */
|
||||
@Getter
|
||||
public class TaggingResult {
|
||||
|
||||
private final List<String> warnings = new ArrayList<>();
|
||||
private final DocumentStructure structure;
|
||||
private final boolean rebuilt;
|
||||
private final int taggedElements;
|
||||
private final int artifacts;
|
||||
private final int figuresNeedingAlt;
|
||||
|
||||
/** True when text was hidden as artifacts; the caller must not declare conformance. */
|
||||
private final boolean contentSuppressed;
|
||||
|
||||
public TaggingResult(DocumentStructure structure, boolean rebuilt) {
|
||||
this.structure = structure;
|
||||
this.rebuilt = rebuilt;
|
||||
this.warnings.addAll(structure.getWarnings());
|
||||
int[] elements = {0};
|
||||
structure.visit(
|
||||
block -> {
|
||||
if (!block.isArtifact()) {
|
||||
elements[0]++;
|
||||
}
|
||||
});
|
||||
this.taggedElements = elements[0];
|
||||
this.artifacts = structure.artifactCount();
|
||||
this.figuresNeedingAlt = structure.figuresWithoutAlt().size();
|
||||
this.contentSuppressed = structure.isTextSuppressed();
|
||||
}
|
||||
|
||||
public boolean needsHumanReview() {
|
||||
return figuresNeedingAlt > 0 || !warnings.isEmpty();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* A run of text on one baseline, with the operator ordinals that produced it. {@code preMarked}
|
||||
* means the source stream already wrapped this text in BDC/EMC.
|
||||
*/
|
||||
public record TextLineInfo(
|
||||
int pageIndex,
|
||||
String text,
|
||||
BBox bbox,
|
||||
float dominantFontSize,
|
||||
boolean bold,
|
||||
int startOrdinal,
|
||||
int endOrdinal,
|
||||
boolean preMarked,
|
||||
List<WordInfo> words) {
|
||||
|
||||
public boolean isBlank() {
|
||||
return text == null || text.isBlank();
|
||||
}
|
||||
|
||||
public int charCount() {
|
||||
return text == null ? 0 : text.strip().length();
|
||||
}
|
||||
|
||||
public int wordCount() {
|
||||
return (int) words.stream().filter(w -> !w.isBlank()).count();
|
||||
}
|
||||
|
||||
/** True when every word occupies its own operator run, so cells can be tagged separately. */
|
||||
public boolean wordsAreSeparable() {
|
||||
List<WordInfo> real = words.stream().filter(w -> !w.isBlank()).toList();
|
||||
for (int i = 1; i < real.size(); i++) {
|
||||
if (!real.get(i - 1).isSeparableFrom(real.get(i))) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
/**
|
||||
* A whitespace-delimited run of glyphs, with the operator ordinals that produced it. Cell detection
|
||||
* needs both: geometry to find cells, ordinals to tell whether they can be tagged separately.
|
||||
*/
|
||||
public record WordInfo(
|
||||
String text, BBox bbox, int startOrdinal, int endOrdinal, float fontSize, boolean bold) {
|
||||
|
||||
public boolean isBlank() {
|
||||
return text == null || text.isBlank();
|
||||
}
|
||||
|
||||
/** True when this word shares no operator with the other, so both can carry their own MCID. */
|
||||
public boolean isSeparableFrom(WordInfo other) {
|
||||
return endOrdinal < other.startOrdinal || other.endOrdinal < startOrdinal;
|
||||
}
|
||||
}
|
||||
+5
@@ -26,6 +26,11 @@ public class AdminPolicyManagementAuthority implements PolicyManagementAuthority
|
||||
return userService.isCurrentUserAdmin();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean canTriggerPolicies() {
|
||||
return userService.isCurrentUserAdmin();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Long currentUserTeamId() {
|
||||
String username = userService.getCurrentUsername();
|
||||
|
||||
+11
@@ -12,6 +12,17 @@ public interface PolicyManagementAuthority {
|
||||
/** Whether the current user may create, edit, or delete policies (for their own team). */
|
||||
boolean canEditPolicies();
|
||||
|
||||
/**
|
||||
* Whether the current user may run a policy against its <em>configured sources</em> (the manual
|
||||
* "run now" sweep). Kept separate from {@link #canEditPolicies()} because the two are distinct
|
||||
* capabilities, even where a deployment grants both to the same people: a sweep operates on the
|
||||
* team's configured sources using the server's stored connection credentials, which makes it a
|
||||
* policy-management capability rather than ordinary use. Running a policy over the caller's
|
||||
* <em>own</em> uploaded files is not covered by this and stays open to every team member — that
|
||||
* is ordinary editor enforcement.
|
||||
*/
|
||||
boolean canTriggerPolicies();
|
||||
|
||||
/**
|
||||
* The team that scopes the current user's policies — the team a new policy is stamped with and
|
||||
* the only team whose policies the user may see/run/edit. {@code null} when it can't be
|
||||
|
||||
+62
-10
@@ -21,6 +21,7 @@ import org.springframework.web.bind.annotation.PostMapping;
|
||||
import org.springframework.web.bind.annotation.PutMapping;
|
||||
import org.springframework.web.bind.annotation.RequestBody;
|
||||
import org.springframework.web.bind.annotation.RequestMapping;
|
||||
import org.springframework.web.bind.annotation.RequestParam;
|
||||
import org.springframework.web.bind.annotation.RequestPart;
|
||||
import org.springframework.web.bind.annotation.RestController;
|
||||
import org.springframework.web.context.request.RequestContextHolder;
|
||||
@@ -51,6 +52,7 @@ import stirling.software.common.util.TempFile;
|
||||
import stirling.software.common.util.TempFileManager;
|
||||
import stirling.software.proprietary.audit.AuditContext;
|
||||
import stirling.software.proprietary.policy.asset.PolicyAssetCleaner;
|
||||
import stirling.software.proprietary.policy.asset.PolicyAssetResolver;
|
||||
import stirling.software.proprietary.policy.config.PolicyAccessGuard;
|
||||
import stirling.software.proprietary.policy.config.PolicyManagementAuthority;
|
||||
import stirling.software.proprietary.policy.engine.PolicyRunHandle;
|
||||
@@ -106,6 +108,7 @@ public class PolicyController {
|
||||
private final PolicyTriggerManager policyTriggerManager;
|
||||
private final PolicyOverviewService policyOverviewService;
|
||||
private final PolicyAssetCleaner assetCleaner;
|
||||
private final PolicyAssetResolver assetResolver;
|
||||
private final ProcessedLedger processedLedger;
|
||||
private final List<PolicyTrigger> policyTriggers;
|
||||
private final ApplicationProperties applicationProperties;
|
||||
@@ -125,12 +128,13 @@ public class PolicyController {
|
||||
+ " endpoint and download outputs via /api/v1/general/files/{id}.")
|
||||
public ResponseEntity<JobResponse<Void>> run(
|
||||
@RequestPart("json") PipelineDefinition definition,
|
||||
@RequestParam(value = "policyId", required = false) String policyId,
|
||||
@Valid @ModelAttribute PolicyRunFiles files)
|
||||
throws IOException {
|
||||
stampPolicyAudit(definition);
|
||||
requireRunnable(definition);
|
||||
validateAdHocRun(definition);
|
||||
PolicyInputs inputs = toInputs(files);
|
||||
PolicyInputs inputs = resolveStoredAssets(policyId, toInputs(files));
|
||||
PolicyRunHandle handle =
|
||||
policyRunner.runAdHoc(definition, inputs, PolicyProgressListener.NOOP);
|
||||
recordEditorDocs(inputs);
|
||||
@@ -146,12 +150,13 @@ public class PolicyController {
|
||||
+ " 'cancelled', or 'waiting' event carrying the final run view.")
|
||||
public SseEmitter runStream(
|
||||
@RequestPart("json") PipelineDefinition definition,
|
||||
@RequestParam(value = "policyId", required = false) String policyId,
|
||||
@Valid @ModelAttribute PolicyRunFiles files)
|
||||
throws IOException {
|
||||
stampPolicyAudit(definition);
|
||||
requireRunnable(definition);
|
||||
validateAdHocRun(definition);
|
||||
PolicyInputs inputs = toInputs(files);
|
||||
PolicyInputs inputs = resolveStoredAssets(policyId, toInputs(files));
|
||||
|
||||
SseEmitter emitter =
|
||||
new SseEmitter(applicationProperties.getPolicies().getStreamTimeoutMs());
|
||||
@@ -432,21 +437,47 @@ public class PolicyController {
|
||||
* admin gets no say on SaaS. Team scoping (which team's policies) is enforced separately by
|
||||
* {@link PolicyAccessGuard}. Every mutation routes through {@link #savePolicy} (pause/resume
|
||||
* re-save with a flipped {@code enabled} flag) or {@link #deletePolicy}, so gating those two
|
||||
* covers them all; runs ({@code /run}) stay open to the team. Single-user deployments (login
|
||||
* disabled) have no such role, so they trust the local operator. The path allowlist for folder
|
||||
* sources/outputs is enforced separately by {@link PolicyValidator} at validation time.
|
||||
* covers them all; runs over the caller's own files ({@code /{id}/run}) stay open to the team,
|
||||
* while source sweeps are gated by {@link #requirePolicySweepAllowed}. Single-user deployments
|
||||
* (login disabled) have no such role, so they trust the local operator. The path allowlist for
|
||||
* folder sources/outputs is enforced separately by {@link PolicyValidator} at validation time.
|
||||
*/
|
||||
private void requirePolicyEditingAllowed() {
|
||||
if (!applicationProperties.getSecurity().isEnableLogin()) {
|
||||
return;
|
||||
}
|
||||
if (!policyManagementAuthority.canEditPolicies()) {
|
||||
if (!policyEditingAllowed()) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.FORBIDDEN,
|
||||
"Policies may only be created or modified by a team leader");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Sweeping a policy's configured sources requires the same role as managing policies: the sweep
|
||||
* operates on the team's configured sources using the server's stored connection credentials,
|
||||
* which makes it a policy-management capability rather than ordinary use, and team scoping on
|
||||
* its own does not express that. Deliberately narrower than it looks: it gates only the sweep,
|
||||
* not {@link #runStoredPolicy}, because running a policy over documents the caller supplied is
|
||||
* ordinary editor enforcement that every member performs on upload and export.
|
||||
*/
|
||||
private void requirePolicySweepAllowed() {
|
||||
if (!applicationProperties.getSecurity().isEnableLogin()) {
|
||||
return;
|
||||
}
|
||||
if (!policyManagementAuthority.canTriggerPolicies()) {
|
||||
throw new ResponseStatusException(
|
||||
HttpStatus.FORBIDDEN,
|
||||
"Not permitted to run this policy against its configured sources");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the caller may create/modify policies (a team leader, or any operator when login is
|
||||
* off).
|
||||
*/
|
||||
private boolean policyEditingAllowed() {
|
||||
return !applicationProperties.getSecurity().isEnableLogin()
|
||||
|| policyManagementAuthority.canEditPolicies();
|
||||
}
|
||||
|
||||
@GetMapping
|
||||
@Operation(
|
||||
summary = "List policies",
|
||||
@@ -571,8 +602,10 @@ public class PolicyController {
|
||||
+ " the enabled flag (which only gates automatic triggering). Returns"
|
||||
+ " the ids of the runs started (poll the run-status endpoint for each)"
|
||||
+ " plus what the sweep skipped - already-processed, parked-by-failure,"
|
||||
+ " and in-flight counts - so an empty result explains itself.")
|
||||
+ " and in-flight counts - so an empty result explains itself. Requires"
|
||||
+ " the policy-management role.")
|
||||
public ResponseEntity<SweepOutcome> trigger(@PathVariable String policyId) {
|
||||
requirePolicySweepAllowed();
|
||||
Policy policy =
|
||||
policyStore
|
||||
.get(policyId)
|
||||
@@ -650,6 +683,25 @@ public class PolicyController {
|
||||
inputs.primary().size());
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a test run's stored {@code asset:<id>} bindings from the saved policy the builder is
|
||||
* editing, so their bytes need not be re-uploaded. Scoped to that policy (the resolver loads
|
||||
* only the assets it references, in its own team) and gated to policy editors - the same
|
||||
* authority that can read asset bytes - so a member can't rebind a policy's stored asset into
|
||||
* an ad-hoc step to read it back. A blank id (an unsaved pipeline has no stored bindings) or an
|
||||
* inaccessible policy leaves the run-supplied inputs untouched.
|
||||
*/
|
||||
private PolicyInputs resolveStoredAssets(String policyId, PolicyInputs inputs) {
|
||||
if (policyId == null || policyId.isBlank() || !policyEditingAllowed()) {
|
||||
return inputs;
|
||||
}
|
||||
return policyStore
|
||||
.get(policyId)
|
||||
.filter(policyAccessGuard::canAccess)
|
||||
.map(policy -> assetResolver.resolve(policy, inputs))
|
||||
.orElse(inputs);
|
||||
}
|
||||
|
||||
/**
|
||||
* Turn the typed run files into engine {@link PolicyInputs}: the primary documents plus the
|
||||
* named supporting-file store, where each asset's {@code key} is the name a step references
|
||||
|
||||
+187
@@ -0,0 +1,187 @@
|
||||
package stirling.software.proprietary.service.ua;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDDocumentCatalog;
|
||||
import org.apache.pdfbox.pdmodel.interactive.viewerpreferences.PDViewerPreferences;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.common.util.ExceptionUtils;
|
||||
import stirling.software.proprietary.model.api.ua.AccessibilityIssue;
|
||||
import stirling.software.proprietary.model.api.ua.AccessibilityReport;
|
||||
import stirling.software.proprietary.model.api.ua.FigureDescriptor;
|
||||
import stirling.software.proprietary.model.api.ua.UaValidationResult;
|
||||
import stirling.software.proprietary.pdf.ua.BBox;
|
||||
import stirling.software.proprietary.pdf.ua.DocumentStructure;
|
||||
import stirling.software.proprietary.pdf.ua.LayoutAnalyzer;
|
||||
import stirling.software.proprietary.pdf.ua.PageContent;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
|
||||
import stirling.software.proprietary.pdf.ua.StructBlock;
|
||||
import stirling.software.proprietary.pdf.ua.StructType;
|
||||
import stirling.software.proprietary.pdf.ua.TaggedContentExtractor;
|
||||
|
||||
/** Produces an accessibility report without changing the document. */
|
||||
@Service
|
||||
@Slf4j
|
||||
@RequiredArgsConstructor
|
||||
public class AccessibilityAuditService {
|
||||
|
||||
/** Checks no validator can make; omitting them implies the work does not exist. */
|
||||
private static final List<String> HUMAN_CHECKS =
|
||||
List.of(
|
||||
"Is the reading order correct for someone who cannot see the layout?",
|
||||
"Does each alternative description convey what the image is for, not just what"
|
||||
+ " it looks like?",
|
||||
"Are headings used for structure rather than for visual emphasis?",
|
||||
"Is any information conveyed by colour alone also available another way?",
|
||||
"Do tables have headers that identify the right rows and columns?",
|
||||
"Is the document language correct, including for quoted passages?",
|
||||
"Do links describe their destination rather than saying 'click here'?");
|
||||
|
||||
/** The report walks every page and validates, so it carries the conversion's own caps. */
|
||||
private static final long MAX_INPUT_BYTES = 100L * 1024 * 1024;
|
||||
|
||||
private static final int MAX_PAGES = 2000;
|
||||
|
||||
private final PdfUaValidationService validationService;
|
||||
|
||||
public AccessibilityReport audit(byte[] pdfBytes, PdfUaProfile profile) throws IOException {
|
||||
enforceLimits(pdfBytes);
|
||||
AccessibilityReport report = new AccessibilityReport();
|
||||
report.setProfile(profile.displayName());
|
||||
|
||||
UaValidationResult validation = validationService.validate(pdfBytes, profile);
|
||||
report.setIssues(validation.issues());
|
||||
report.setPassesAutomatedChecks(validation.compliant());
|
||||
report.setHumanChecks(HUMAN_CHECKS);
|
||||
|
||||
int fixable = 0;
|
||||
int needsInput = 0;
|
||||
for (AccessibilityIssue issue : validation.issues()) {
|
||||
if (issue.isAutoFixable()) {
|
||||
fixable++;
|
||||
} else {
|
||||
needsInput++;
|
||||
}
|
||||
}
|
||||
report.setAutomaticallyFixable(fixable);
|
||||
report.setNeedsInput(needsInput);
|
||||
|
||||
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
|
||||
populateSummary(document, report);
|
||||
report.setFiguresNeedingDescription(figuresNeedingDescription(document));
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not inspect document for the summary: {}", e.getMessage());
|
||||
}
|
||||
return report;
|
||||
}
|
||||
|
||||
/**
|
||||
* Rejects before the expensive pass. An unreadable file is left to the report itself to say.
|
||||
*/
|
||||
private static void enforceLimits(byte[] pdfBytes) {
|
||||
if (pdfBytes.length > MAX_INPUT_BYTES) {
|
||||
throw ExceptionUtils.createIllegalArgumentException(
|
||||
"error.fileTooLarge",
|
||||
"This PDF is {0} MB. The accessibility report is limited to {1} MB.",
|
||||
pdfBytes.length / (1024 * 1024),
|
||||
MAX_INPUT_BYTES / (1024 * 1024));
|
||||
}
|
||||
int pages;
|
||||
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
|
||||
pages = document.getNumberOfPages();
|
||||
} catch (IOException e) {
|
||||
return;
|
||||
}
|
||||
if (pages > MAX_PAGES) {
|
||||
throw ExceptionUtils.createIllegalArgumentException(
|
||||
"error.tooManyPages",
|
||||
"This PDF has {0} pages. The accessibility report is limited to {1} pages;"
|
||||
+ " split it first.",
|
||||
pages,
|
||||
MAX_PAGES);
|
||||
}
|
||||
}
|
||||
|
||||
private void populateSummary(PDDocument document, AccessibilityReport report)
|
||||
throws IOException {
|
||||
PDDocumentCatalog catalog = document.getDocumentCatalog();
|
||||
AccessibilityReport.Summary summary = report.getSummary();
|
||||
|
||||
report.setTagged(catalog.getStructureTreeRoot() != null);
|
||||
report.setDeclaresConformance(declaresUa(document));
|
||||
|
||||
summary.setPages(document.getNumberOfPages());
|
||||
summary.setEncrypted(document.isEncrypted());
|
||||
summary.setHasLanguage(catalog.getLanguage() != null && !catalog.getLanguage().isBlank());
|
||||
|
||||
String title = document.getDocumentInformation().getTitle();
|
||||
summary.setHasTitle(title != null && !title.isBlank());
|
||||
|
||||
PDViewerPreferences preferences = catalog.getViewerPreferences();
|
||||
summary.setDisplaysDocTitle(preferences != null && preferences.displayDocTitle());
|
||||
|
||||
Set<String> unembedded = FontEmbeddingService.findUnembeddedFonts(document);
|
||||
summary.setUnembeddedFonts(unembedded.size());
|
||||
summary.setAllFontsEmbedded(unembedded.isEmpty());
|
||||
|
||||
try {
|
||||
summary.setFigures(new TaggedContentExtractor().countGraphics(document));
|
||||
} catch (Exception e) {
|
||||
log.debug("Could not count figures: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Lists the figures a conversion would leave undescribed, running the converter's own analysis
|
||||
* because counting raster images would miss vector charts and existing descriptions.
|
||||
*/
|
||||
private List<FigureDescriptor> figuresNeedingDescription(PDDocument document) {
|
||||
try {
|
||||
List<PageContent> pages = new TaggedContentExtractor().extract(document);
|
||||
DocumentStructure structure = new LayoutAnalyzer().analyse(pages);
|
||||
List<FigureDescriptor> figures = new ArrayList<>();
|
||||
for (StructBlock block : structure.figuresWithoutAlt()) {
|
||||
int ordinal = block.getRanges().isEmpty() ? -1 : block.getRanges().get(0).start();
|
||||
BBox box = block.getBbox();
|
||||
figures.add(
|
||||
new FigureDescriptor(
|
||||
block.getPageIndex() + ":" + ordinal,
|
||||
block.getPageIndex() + 1,
|
||||
block.getType() == StructType.FORMULA ? "formula" : "figure",
|
||||
box.x0(),
|
||||
box.y0(),
|
||||
box.width(),
|
||||
box.height(),
|
||||
block.getAlt()));
|
||||
}
|
||||
return figures;
|
||||
} catch (Exception e) {
|
||||
log.debug("Could not enumerate figures: {}", e.getMessage());
|
||||
return List.of();
|
||||
}
|
||||
}
|
||||
|
||||
/** True when the XMP packet carries a pdfuaid identifier. */
|
||||
private static boolean declaresUa(PDDocument document) {
|
||||
try {
|
||||
var metadata = document.getDocumentCatalog().getMetadata();
|
||||
if (metadata == null) {
|
||||
return false;
|
||||
}
|
||||
String xmp =
|
||||
new String(metadata.toByteArray(), java.nio.charset.StandardCharsets.UTF_8);
|
||||
return xmp.contains("pdfuaid");
|
||||
} catch (IOException e) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
+254
@@ -0,0 +1,254 @@
|
||||
package stirling.software.proprietary.service.ua;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Comparator;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.PDResources;
|
||||
import org.apache.pdfbox.pdmodel.font.PDFont;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.common.util.ProcessExecutor;
|
||||
import stirling.software.common.util.ProcessExecutor.ProcessExecutorResult;
|
||||
|
||||
/**
|
||||
* Embeds any font the document references but does not carry, as PDF/UA-1 clause 7.21 requires.
|
||||
* Ghostscript does the embedding and discards the structure tree, so this must run before tagging.
|
||||
*/
|
||||
@Service
|
||||
@Slf4j
|
||||
public class FontEmbeddingService {
|
||||
|
||||
public boolean hasUnembeddedFonts(byte[] pdfBytes) {
|
||||
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
|
||||
return !findUnembeddedFonts(document).isEmpty();
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not inspect fonts: {}", e.getMessage());
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
public static Set<String> findUnembeddedFonts(PDDocument document) {
|
||||
Set<String> missing = new HashSet<>();
|
||||
for (PDPage page : document.getPages()) {
|
||||
PDResources resources = page.getResources();
|
||||
if (resources == null) {
|
||||
continue;
|
||||
}
|
||||
for (COSName name : resources.getFontNames()) {
|
||||
try {
|
||||
PDFont font = resources.getFont(name);
|
||||
if (font != null && !font.isEmbedded()) {
|
||||
missing.add(font.getName());
|
||||
}
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not read font {}: {}", name.getName(), e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
return missing;
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns the document with all fonts embedded, or the input unchanged. Never throws: failing
|
||||
* to embed is a reportable shortfall, not a reason to abandon the conversion.
|
||||
*/
|
||||
public Result embedFonts(byte[] pdfBytes) {
|
||||
Set<String> missing;
|
||||
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
|
||||
missing = findUnembeddedFonts(document);
|
||||
} catch (IOException e) {
|
||||
return new Result(
|
||||
pdfBytes, false, Set.of(), "Could not inspect fonts: " + e.getMessage());
|
||||
}
|
||||
if (missing.isEmpty()) {
|
||||
return new Result(pdfBytes, false, Set.of(), null);
|
||||
}
|
||||
if (!isGhostscriptAvailable()) {
|
||||
return new Result(
|
||||
pdfBytes,
|
||||
false,
|
||||
missing,
|
||||
"Ghostscript is not installed, so "
|
||||
+ missing.size()
|
||||
+ " unembedded font(s) could not be embedded. PDF/UA requires every font"
|
||||
+ " to be embedded.");
|
||||
}
|
||||
|
||||
Path workingDir = null;
|
||||
try {
|
||||
workingDir = Files.createTempDirectory("pdfua_fonts_");
|
||||
Path input = workingDir.resolve("input.pdf");
|
||||
Path output = workingDir.resolve("output.pdf");
|
||||
Files.write(input, pdfBytes);
|
||||
|
||||
ProcessExecutorResult result =
|
||||
ProcessExecutor.getInstance(ProcessExecutor.Processes.GHOSTSCRIPT)
|
||||
.runCommandWithOutputHandling(command(input, output, workingDir));
|
||||
|
||||
if (result.getRc() != 0 || !Files.exists(output)) {
|
||||
return new Result(
|
||||
pdfBytes,
|
||||
false,
|
||||
missing,
|
||||
"Font embedding failed with code " + result.getRc());
|
||||
}
|
||||
byte[] embedded = Files.readAllBytes(output);
|
||||
|
||||
// Ghostscript can exit 0 having written a blank page, so keep the original rather than
|
||||
// return an empty document.
|
||||
if (!survived(pdfBytes, embedded)) {
|
||||
log.warn("Ghostscript produced a degenerate document; keeping the original");
|
||||
return new Result(
|
||||
pdfBytes,
|
||||
false,
|
||||
missing,
|
||||
"Font embedding was skipped because the embedder returned a document that"
|
||||
+ " had lost content. "
|
||||
+ missing.size()
|
||||
+ " font(s) remain unembedded.");
|
||||
}
|
||||
|
||||
// It can also exit 0 while simply leaving fonts unembedded.
|
||||
Set<String> remaining;
|
||||
try (PDDocument check = Loader.loadPDF(embedded)) {
|
||||
remaining = findUnembeddedFonts(check);
|
||||
}
|
||||
if (!remaining.isEmpty()) {
|
||||
return new Result(
|
||||
embedded,
|
||||
true,
|
||||
remaining,
|
||||
remaining.size()
|
||||
+ " font(s) could not be embedded ("
|
||||
+ String.join(", ", remaining)
|
||||
+ "). PDF/UA requires every font to be embedded.");
|
||||
}
|
||||
log.info("Embedded {} previously unembedded font(s)", missing.size());
|
||||
return new Result(embedded, true, missing, null);
|
||||
|
||||
} catch (Exception e) {
|
||||
log.warn("Font embedding failed: {}", e.getMessage());
|
||||
return new Result(pdfBytes, false, missing, "Font embedding failed: " + e.getMessage());
|
||||
} finally {
|
||||
deleteQuietly(workingDir);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the rewritten document still holds the original's content. A collapse in page count
|
||||
* or content-stream size is the only signature of a failed rewrite the exit code hides.
|
||||
*/
|
||||
private static boolean survived(byte[] original, byte[] rewritten) {
|
||||
try (PDDocument before = Loader.loadPDF(original);
|
||||
PDDocument after = Loader.loadPDF(rewritten)) {
|
||||
if (after.getNumberOfPages() != before.getNumberOfPages()) {
|
||||
return false;
|
||||
}
|
||||
long beforeBytes = contentBytes(before);
|
||||
long afterBytes = contentBytes(after);
|
||||
if (beforeBytes == 0) {
|
||||
return true;
|
||||
}
|
||||
return afterBytes * 20L >= beforeBytes;
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not compare documents after embedding: {}", e.getMessage());
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
private static long contentBytes(PDDocument document) {
|
||||
long total = 0;
|
||||
for (PDPage page : document.getPages()) {
|
||||
try (InputStream in = page.getContents()) {
|
||||
if (in != null) {
|
||||
byte[] buffer = new byte[8192];
|
||||
int read;
|
||||
while ((read = in.read(buffer)) > 0) {
|
||||
total += read;
|
||||
}
|
||||
}
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not measure page content: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
private static List<String> command(Path input, Path output, Path workingDir) {
|
||||
List<String> command = new ArrayList<>();
|
||||
command.add("gs");
|
||||
command.add("--permit-file-read=" + workingDir.toAbsolutePath());
|
||||
command.add("--permit-file-write=" + workingDir.toAbsolutePath());
|
||||
command.add("-sDEVICE=pdfwrite");
|
||||
command.add("-dEmbedAllFonts=true");
|
||||
command.add("-dSubsetFonts=true");
|
||||
command.add("-dCompressFonts=true");
|
||||
command.add("-dNOSUBSTFONTS=false");
|
||||
command.add("-dPDFSETTINGS=/prepress");
|
||||
command.add("-dNOPAUSE");
|
||||
command.add("-dBATCH");
|
||||
command.add("-sOutputFile=" + output.toAbsolutePath());
|
||||
command.add(input.toAbsolutePath().toString());
|
||||
return command;
|
||||
}
|
||||
|
||||
/** Cached after the first probe: availability does not change mid-process. */
|
||||
private volatile Boolean ghostscriptAvailable;
|
||||
|
||||
private boolean isGhostscriptAvailable() {
|
||||
Boolean cached = ghostscriptAvailable;
|
||||
if (cached != null) {
|
||||
return cached;
|
||||
}
|
||||
boolean available;
|
||||
try {
|
||||
ProcessExecutorResult result =
|
||||
ProcessExecutor.getInstance(ProcessExecutor.Processes.GHOSTSCRIPT)
|
||||
.runCommandWithOutputHandling(List.of("gs", "--version"));
|
||||
available = result.getRc() == 0;
|
||||
} catch (Exception e) {
|
||||
log.debug("Ghostscript availability check failed: {}", e.getMessage());
|
||||
available = false;
|
||||
}
|
||||
ghostscriptAvailable = available;
|
||||
return available;
|
||||
}
|
||||
|
||||
private static void deleteQuietly(Path directory) {
|
||||
if (directory == null) {
|
||||
return;
|
||||
}
|
||||
try (Stream<Path> stream = Files.walk(directory)) {
|
||||
stream.sorted(Comparator.reverseOrder())
|
||||
.forEach(
|
||||
path -> {
|
||||
try {
|
||||
Files.deleteIfExists(path);
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not delete {}", path);
|
||||
}
|
||||
});
|
||||
} catch (IOException e) {
|
||||
log.debug("Could not clean {}", directory);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @param warning non-null when fonts remain unembedded, for the conversion report
|
||||
*/
|
||||
public record Result(byte[] pdfBytes, boolean changed, Set<String> fonts, String warning) {}
|
||||
}
|
||||
+251
@@ -0,0 +1,251 @@
|
||||
package stirling.software.proprietary.service.ua;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.encryption.InvalidPasswordException;
|
||||
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
|
||||
import stirling.software.proprietary.model.api.ua.UaValidationResult;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaTagger;
|
||||
import stirling.software.proprietary.pdf.ua.SourceFacts;
|
||||
import stirling.software.proprietary.pdf.ua.TaggingOptions;
|
||||
import stirling.software.proprietary.pdf.ua.TaggingResult;
|
||||
|
||||
/**
|
||||
* Converts a PDF to PDF/UA. The declaration is written first and withdrawn unless validation
|
||||
* passes, so a returned file either conforms or does not claim to.
|
||||
*/
|
||||
@Service
|
||||
@Slf4j
|
||||
@RequiredArgsConstructor
|
||||
public class PdfUaConversionService {
|
||||
|
||||
private final PdfUaValidationService validationService;
|
||||
private final FontEmbeddingService fontEmbeddingService;
|
||||
private final stirling.software.common.service.CustomPDFDocumentFactory pdfDocumentFactory;
|
||||
|
||||
/** Matches the cap GetInfoOnPDF already applies to comparable whole-document work. */
|
||||
private static final long MAX_INPUT_BYTES = 100L * 1024 * 1024;
|
||||
|
||||
/** Beyond this the structure model alone runs to hundreds of megabytes. */
|
||||
private static final int MAX_PAGES = 2000;
|
||||
|
||||
public PdfUaConversionOutcome convert(byte[] input, TaggingOptions options) throws IOException {
|
||||
if (input.length > MAX_INPUT_BYTES) {
|
||||
throw new IOException(
|
||||
"This PDF is "
|
||||
+ (input.length / (1024 * 1024))
|
||||
+ " MB. PDF/UA conversion is limited to "
|
||||
+ (MAX_INPUT_BYTES / (1024 * 1024))
|
||||
+ " MB.");
|
||||
}
|
||||
PdfUaProfile profile = options.getProfile();
|
||||
List<String> warnings = new ArrayList<>();
|
||||
|
||||
// Read the document's own facts before anything rewrites it. Font embedding runs
|
||||
// Ghostscript over the whole file, which discards the structure tree, /Lang and XFA, so
|
||||
// every guard and every "what did the source say" question must be answered from here.
|
||||
SourceFacts facts;
|
||||
try (PDDocument original = load(input)) {
|
||||
rejectUnsupportedSource(original);
|
||||
warnSignatures(original, warnings);
|
||||
facts = SourceFacts.of(original);
|
||||
}
|
||||
|
||||
byte[] source = input;
|
||||
if (options.isEmbedFonts()) {
|
||||
// Must precede tagging: the embedder rewrites the file and drops any structure tree.
|
||||
FontEmbeddingService.Result fonts = fontEmbeddingService.embedFonts(input);
|
||||
source = fonts.pdfBytes();
|
||||
if (fonts.warning() != null) {
|
||||
warnings.add(fonts.warning());
|
||||
}
|
||||
source = keepTagsOverFonts(input, source, facts, options, warnings);
|
||||
}
|
||||
|
||||
TaggingOptions effective = options.toBuilder().sourceFacts(facts).build();
|
||||
|
||||
// Tag and declare in one pass; the claim is withdrawn below if validation disagrees.
|
||||
byte[] declared;
|
||||
TaggingResult taggingResult;
|
||||
PdfUaTagger tagger = new PdfUaTagger();
|
||||
try (PDDocument document = load(source)) {
|
||||
rejectEncrypted(document);
|
||||
taggingResult = tagger.tag(document, effective);
|
||||
warnings.addAll(taggingResult.getWarnings());
|
||||
tagger.declareConformance(document, profile);
|
||||
declared = save(document);
|
||||
}
|
||||
|
||||
UaValidationResult validation = validationService.validate(declared, profile);
|
||||
|
||||
// A validator cannot see text hidden behind artifact markers, so a clean verdict over
|
||||
// suppressed content would be a false claim.
|
||||
boolean honest = !taggingResult.isContentSuppressed();
|
||||
|
||||
if (validation.compliant() && honest) {
|
||||
log.info("{} conversion passed validation", profile.displayName());
|
||||
return new PdfUaConversionOutcome(
|
||||
declared, true, validation, summary(taggingResult), warnings);
|
||||
}
|
||||
|
||||
byte[] undeclared;
|
||||
try (PDDocument document = load(declared)) {
|
||||
tagger.withdrawConformance(document);
|
||||
undeclared = save(document);
|
||||
}
|
||||
|
||||
if (!validation.compliant()) {
|
||||
warnings.add(
|
||||
"The document could not be made "
|
||||
+ profile.displayName()
|
||||
+ " conformant, so no conformance claim was written. "
|
||||
+ validation.totalFailures()
|
||||
+ " automated check(s) still fail.");
|
||||
}
|
||||
log.info(
|
||||
"{} conversion left undeclared: {} failures, suppressedText={}",
|
||||
profile.displayName(),
|
||||
validation.totalFailures(),
|
||||
!honest);
|
||||
return new PdfUaConversionOutcome(
|
||||
undeclared, false, validation, summary(taggingResult), warnings);
|
||||
}
|
||||
|
||||
private static PdfUaConversionOutcome.TaggingSummary summary(TaggingResult result) {
|
||||
return new PdfUaConversionOutcome.TaggingSummary(
|
||||
result.isRebuilt(),
|
||||
result.getTaggedElements(),
|
||||
result.getArtifacts(),
|
||||
result.getFiguresNeedingAlt());
|
||||
}
|
||||
|
||||
/**
|
||||
* Tagging rewrites the content streams a signature covers, so the conversion still runs but the
|
||||
* caller has to know the signature will no longer verify.
|
||||
*/
|
||||
private static void warnSignatures(PDDocument document, List<String> warnings) {
|
||||
int signatures = document.getSignatureDictionaries().size();
|
||||
if (signatures > 0) {
|
||||
warnings.add(
|
||||
signatures
|
||||
+ " digital signature(s) will stop verifying: tagging rewrites the"
|
||||
+ " content streams they cover. Convert first, then re-sign.");
|
||||
}
|
||||
}
|
||||
|
||||
/** Replaces PDFBox's "incorrect password" wording, baffling when the caller supplied none. */
|
||||
private PDDocument load(byte[] bytes) throws IOException {
|
||||
try {
|
||||
// The factory spills large documents to a temp-file cache instead of the heap.
|
||||
return pdfDocumentFactory.load(bytes);
|
||||
} catch (IOException | RuntimeException e) {
|
||||
// The factory wraps the parse failure, so check the cause chain rather than the type.
|
||||
if (mentionsPassword(e)) {
|
||||
throw new IOException(
|
||||
"This PDF is encrypted. Remove the password before converting it to"
|
||||
+ " PDF/UA.",
|
||||
e);
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
private static boolean mentionsPassword(Throwable error) {
|
||||
for (Throwable cause = error; cause != null; cause = cause.getCause()) {
|
||||
if (cause instanceof InvalidPasswordException) {
|
||||
return true;
|
||||
}
|
||||
String message = cause.getMessage();
|
||||
if (message != null) {
|
||||
String lower = message.toLowerCase(Locale.ROOT);
|
||||
if (lower.contains("password") || lower.contains("decrypt")) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/** XFA is forbidden by PDF/UA-1 clause 7.15; encrypted or huge files cannot be restructured. */
|
||||
/**
|
||||
* Under KEEP nothing rebuilds a tree, so if the embedder deleted one we would hand back an
|
||||
* untagged document. Fonts are not worth the whole structure; give the tags back instead.
|
||||
*/
|
||||
private byte[] keepTagsOverFonts(
|
||||
byte[] input,
|
||||
byte[] embedded,
|
||||
SourceFacts facts,
|
||||
TaggingOptions options,
|
||||
List<String> warnings)
|
||||
throws IOException {
|
||||
if (options.getExistingTags() != TaggingOptions.ExistingTags.KEEP
|
||||
|| !facts.hasUsableTree()
|
||||
|| embedded == input) {
|
||||
return embedded;
|
||||
}
|
||||
boolean survived;
|
||||
try (PDDocument rewritten = load(embedded)) {
|
||||
survived = PdfUaTagger.hasUsableStructureTree(rewritten);
|
||||
}
|
||||
if (survived) {
|
||||
return embedded;
|
||||
}
|
||||
warnings.add(
|
||||
"Embedding the missing fonts would have deleted the document's existing tags, so"
|
||||
+ " the tags were kept and the fonts left unembedded. Turn off font"
|
||||
+ " embedding to silence this, or rebuild the tags to embed them.");
|
||||
return input;
|
||||
}
|
||||
|
||||
/**
|
||||
* Checks that must see the document as the author wrote it. Font embedding strips XFA, so
|
||||
* running this afterwards would let a dynamic form through unnoticed, and it would push a
|
||||
* document we are about to reject through the whole embedder first.
|
||||
*/
|
||||
private static void rejectUnsupportedSource(PDDocument document) throws IOException {
|
||||
if (document.getNumberOfPages() > MAX_PAGES) {
|
||||
throw new IOException(
|
||||
"This PDF has "
|
||||
+ document.getNumberOfPages()
|
||||
+ " pages. PDF/UA conversion is limited to "
|
||||
+ MAX_PAGES
|
||||
+ " pages; split it first.");
|
||||
}
|
||||
PDAcroForm form = document.getDocumentCatalog().getAcroForm();
|
||||
if (form != null && form.xfaIsDynamic()) {
|
||||
throw new IOException(
|
||||
"Dynamic XFA forms are not permitted by PDF/UA. Flatten the form first.");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliberately checked on the working document rather than the source. Permissions-only
|
||||
* encryption with an empty user password is common in published documents, the embedder
|
||||
* resolves it, and those files convert usefully; rejecting them up front would fail a document
|
||||
* for a password its author never set.
|
||||
*/
|
||||
private static void rejectEncrypted(PDDocument document) throws IOException {
|
||||
if (document.isEncrypted()) {
|
||||
throw new IOException(
|
||||
"Encrypted PDFs cannot be converted to PDF/UA. Remove the password first.");
|
||||
}
|
||||
}
|
||||
|
||||
private static byte[] save(PDDocument document) throws IOException {
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
document.save(out);
|
||||
return out.toByteArray();
|
||||
}
|
||||
}
|
||||
+245
@@ -0,0 +1,245 @@
|
||||
package stirling.software.proprietary.service.ua;
|
||||
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import org.springframework.stereotype.Service;
|
||||
import org.verapdf.gf.foundry.VeraGreenfieldFoundryProvider;
|
||||
import org.verapdf.pdfa.Foundries;
|
||||
import org.verapdf.pdfa.PDFAParser;
|
||||
import org.verapdf.pdfa.PDFAValidator;
|
||||
import org.verapdf.pdfa.flavours.PDFAFlavour;
|
||||
import org.verapdf.pdfa.results.TestAssertion;
|
||||
import org.verapdf.pdfa.results.ValidationResult;
|
||||
|
||||
import jakarta.annotation.PostConstruct;
|
||||
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.proprietary.model.api.ua.AccessibilityIssue;
|
||||
import stirling.software.proprietary.model.api.ua.UaValidationResult;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
|
||||
|
||||
/**
|
||||
* Validates a document against a PDF/UA profile using veraPDF, the oracle a conversion is declared
|
||||
* against. It checks only the machine-verifiable subset, so a clean result is not "accessible".
|
||||
*/
|
||||
@Service
|
||||
@Slf4j
|
||||
public class PdfUaValidationService {
|
||||
|
||||
/** Plain-English text and remediability for the clauses users actually hit. */
|
||||
private static final Map<String, ClauseInfo> CLAUSES = buildClauseTable();
|
||||
|
||||
record ClauseInfo(String message, boolean autoFixable) {}
|
||||
|
||||
@PostConstruct
|
||||
public void initialise() {
|
||||
try {
|
||||
VeraGreenfieldFoundryProvider.initialise();
|
||||
} catch (Exception e) {
|
||||
log.error("Failed to initialise veraPDF for PDF/UA validation", e);
|
||||
}
|
||||
}
|
||||
|
||||
public UaValidationResult validate(byte[] pdfBytes, PdfUaProfile profile) {
|
||||
PDFAFlavour flavour = flavourFor(profile);
|
||||
try (PDFAParser parser =
|
||||
Foundries.defaultInstance()
|
||||
.createParser(new ByteArrayInputStream(pdfBytes), flavour)) {
|
||||
|
||||
PDFAValidator validator = Foundries.defaultInstance().createValidator(flavour, false);
|
||||
ValidationResult result = validator.validate(parser);
|
||||
return toResult(profile, result);
|
||||
|
||||
} catch (Exception e) {
|
||||
log.warn("PDF/UA validation failed for {}: {}", profile.displayName(), e.getMessage());
|
||||
AccessibilityIssue issue = new AccessibilityIssue();
|
||||
issue.setMessage("Validation could not run: " + e.getMessage());
|
||||
issue.setSeverity("error");
|
||||
issue.setClause("n/a");
|
||||
return new UaValidationResult(profile.displayName(), false, List.of(issue), 0);
|
||||
}
|
||||
}
|
||||
|
||||
public static PDFAFlavour flavourFor(PdfUaProfile profile) {
|
||||
return profile == PdfUaProfile.UA2 ? PDFAFlavour.PDFUA_2 : PDFAFlavour.PDFUA_1;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the bytes really validate as PDF/A level A for the given part. Tagging is necessary
|
||||
* for level A but not sufficient, so the claim is only written once veraPDF agrees.
|
||||
*/
|
||||
public boolean validatesAsPdfaLevelA(byte[] pdfBytes, int part) {
|
||||
PDFAFlavour flavour =
|
||||
switch (part) {
|
||||
case 1 -> PDFAFlavour.PDFA_1_A;
|
||||
case 2 -> PDFAFlavour.PDFA_2_A;
|
||||
case 3 -> PDFAFlavour.PDFA_3_A;
|
||||
default -> null;
|
||||
};
|
||||
if (flavour == null) {
|
||||
log.warn("No PDF/A level A flavour for part {}", part);
|
||||
return false;
|
||||
}
|
||||
try (PDFAParser parser =
|
||||
Foundries.defaultInstance()
|
||||
.createParser(new ByteArrayInputStream(pdfBytes), flavour)) {
|
||||
PDFAValidator validator = Foundries.defaultInstance().createValidator(flavour, false);
|
||||
return validator.validate(parser).isCompliant();
|
||||
} catch (Exception e) {
|
||||
log.warn("Level A validation could not run: {}", e.getMessage());
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Groups repeated failures of the same rule so a report lists issues, not thousands of lines.
|
||||
*/
|
||||
private static UaValidationResult toResult(PdfUaProfile profile, ValidationResult result) {
|
||||
Map<String, AccessibilityIssue> grouped = new LinkedHashMap<>();
|
||||
int total = 0;
|
||||
|
||||
for (TestAssertion assertion : result.getTestAssertions()) {
|
||||
if (assertion.getStatus() != TestAssertion.Status.FAILED) {
|
||||
continue;
|
||||
}
|
||||
total++;
|
||||
String clause =
|
||||
assertion.getRuleId() != null ? assertion.getRuleId().getClause() : "unknown";
|
||||
int test = assertion.getRuleId() != null ? assertion.getRuleId().getTestNumber() : 0;
|
||||
String key = clause + "-" + test;
|
||||
|
||||
AccessibilityIssue issue =
|
||||
grouped.computeIfAbsent(
|
||||
key,
|
||||
k -> {
|
||||
AccessibilityIssue created = new AccessibilityIssue();
|
||||
created.setClause(clause);
|
||||
created.setTestNumber(String.valueOf(test));
|
||||
created.setSeverity("error");
|
||||
ClauseInfo info = lookupClause(clause);
|
||||
created.setMessage(
|
||||
info != null ? info.message() : assertion.getMessage());
|
||||
created.setTechnicalMessage(assertion.getMessage());
|
||||
created.setAutoFixable(info != null && info.autoFixable());
|
||||
created.setSpecification(profile.displayName());
|
||||
return created;
|
||||
});
|
||||
issue.setOccurrences(issue.getOccurrences() + 1);
|
||||
if (issue.getLocation() == null && assertion.getLocation() != null) {
|
||||
issue.setLocation(assertion.getLocation().toString());
|
||||
}
|
||||
}
|
||||
|
||||
List<AccessibilityIssue> issues = new ArrayList<>(grouped.values());
|
||||
return new UaValidationResult(
|
||||
profile.displayName(), result.isCompliant() && total == 0, issues, total);
|
||||
}
|
||||
|
||||
/**
|
||||
* Finds the most specific entry covering a clause by walking up the dotted hierarchy. String
|
||||
* prefixes would be wrong: {@code 7.1} prefixes {@code 7.18.1} without being its ancestor.
|
||||
*/
|
||||
static ClauseInfo lookupClause(String clause) {
|
||||
if (clause == null) {
|
||||
return null;
|
||||
}
|
||||
String current = clause;
|
||||
while (!current.isEmpty()) {
|
||||
ClauseInfo info = CLAUSES.get(current);
|
||||
if (info != null) {
|
||||
return info;
|
||||
}
|
||||
int dot = current.lastIndexOf('.');
|
||||
if (dot < 0) {
|
||||
return null;
|
||||
}
|
||||
current = current.substring(0, dot);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private static Map<String, ClauseInfo> buildClauseTable() {
|
||||
Map<String, ClauseInfo> table = new LinkedHashMap<>();
|
||||
table.put(
|
||||
"7.1",
|
||||
new ClauseInfo(
|
||||
"Document is not tagged, or some content is neither tagged nor marked as an artifact.",
|
||||
true));
|
||||
table.put(
|
||||
"7.2",
|
||||
new ClauseInfo(
|
||||
"Text cannot be mapped to Unicode, or the document language is not declared.",
|
||||
true));
|
||||
table.put(
|
||||
"7.3",
|
||||
new ClauseInfo("An image or graphic has no alternative description.", false));
|
||||
table.put(
|
||||
"7.4",
|
||||
new ClauseInfo(
|
||||
"Heading levels skip a level, or headings are nested incorrectly.", true));
|
||||
table.put(
|
||||
"7.5",
|
||||
new ClauseInfo("A table is missing header cells or header associations.", false));
|
||||
table.put(
|
||||
"7.6", new ClauseInfo("A list is not structured as list items with bodies.", true));
|
||||
table.put(
|
||||
"7.7",
|
||||
new ClauseInfo("A mathematical expression has no alternative description.", false));
|
||||
table.put(
|
||||
"7.8",
|
||||
new ClauseInfo("Running heads or page numbers are not marked as artifacts.", true));
|
||||
table.put("7.9", new ClauseInfo("A note is missing a unique identifier.", true));
|
||||
// Tagging does not touch optional content groups, so this needs the authoring tool.
|
||||
table.put("7.10", new ClauseInfo("An optional content group has no name.", false));
|
||||
// The attachment's own /AFRelationship and /Desc are not something tagging can supply.
|
||||
table.put(
|
||||
"7.11",
|
||||
new ClauseInfo(
|
||||
"An embedded file is missing its relationship or description.", false));
|
||||
table.put(
|
||||
"7.15",
|
||||
new ClauseInfo(
|
||||
"The document uses a dynamic XFA form, which PDF/UA does not allow.",
|
||||
false));
|
||||
table.put(
|
||||
"7.16",
|
||||
new ClauseInfo(
|
||||
"Security settings prevent assistive technology from reading the content.",
|
||||
true));
|
||||
table.put("7.17", new ClauseInfo("Navigation aids such as page labels are missing.", true));
|
||||
table.put(
|
||||
"7.18",
|
||||
new ClauseInfo(
|
||||
"An annotation is missing a description, tab order, or structure entry.",
|
||||
true));
|
||||
table.put(
|
||||
"7.20",
|
||||
new ClauseInfo(
|
||||
"A form or group XObject is not marked as content or as an artifact.",
|
||||
false));
|
||||
// Most font defects (CIDFont, CMap, metrics, encoding) need the font itself repaired.
|
||||
table.put(
|
||||
"7.21",
|
||||
new ClauseInfo("A font in the document does not meet PDF/UA rules.", false));
|
||||
// The one font defect embedding does fix.
|
||||
table.put("7.21.4.1", new ClauseInfo("A font used in the document is not embedded.", true));
|
||||
// ToUnicode gaps need the font itself repaired, which embedding does not do.
|
||||
table.put(
|
||||
"7.21.7",
|
||||
new ClauseInfo(
|
||||
"A font does not map every character it uses to Unicode, so extracted text"
|
||||
+ " may be wrong.",
|
||||
false));
|
||||
table.put(
|
||||
"5",
|
||||
new ClauseInfo(
|
||||
"The document does not declare PDF/UA conformance in its XMP metadata.",
|
||||
true));
|
||||
return table;
|
||||
}
|
||||
}
|
||||
+297
@@ -0,0 +1,297 @@
|
||||
package stirling.software.proprietary.service.ua;
|
||||
|
||||
import java.io.ByteArrayInputStream;
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.pdfwriter.compress.CompressParameters;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.common.PDMetadata;
|
||||
import org.apache.xmpbox.XMPMetadata;
|
||||
import org.apache.xmpbox.schema.PDFAExtensionSchema;
|
||||
import org.apache.xmpbox.schema.PDFAIdentificationSchema;
|
||||
import org.apache.xmpbox.type.AbstractStructuredType;
|
||||
import org.apache.xmpbox.type.ArrayProperty;
|
||||
import org.apache.xmpbox.type.Cardinality;
|
||||
import org.apache.xmpbox.type.PDFAPropertyType;
|
||||
import org.apache.xmpbox.type.PDFASchemaType;
|
||||
import org.apache.xmpbox.xml.DomXmpParser;
|
||||
import org.apache.xmpbox.xml.XmpSerializer;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
|
||||
import stirling.software.common.service.PdfaLevelAServiceInterface;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaIdentificationSchema;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaTagger;
|
||||
import stirling.software.proprietary.pdf.ua.TaggingOptions;
|
||||
import stirling.software.proprietary.pdf.ua.TaggingResult;
|
||||
|
||||
/**
|
||||
* Raises a PDF/A file from conformance level B to level A, which adds the tagging the PDF/UA tagger
|
||||
* already does. Must run after Ghostscript, which discards any structure tree it is given.
|
||||
*/
|
||||
@Service
|
||||
@Slf4j
|
||||
@RequiredArgsConstructor
|
||||
public class PdfaAccessibilityService implements PdfaLevelAServiceInterface {
|
||||
|
||||
/**
|
||||
* Matches the PDF/UA converter's own cap; beyond this the structure model exhausts the heap.
|
||||
*/
|
||||
private static final int MAX_TAGGABLE_PAGES = 2000;
|
||||
|
||||
private final PdfUaValidationService validationService;
|
||||
|
||||
/**
|
||||
* Tags a converted PDF/A and marks it conformance A, or returns it unchanged rather than
|
||||
* claiming level A over untagged content. part is 1 to 3; part 1 keeps its PDF 1.4 version.
|
||||
*/
|
||||
public Result upgradeToLevelA(byte[] pdfBytes, int part, String language, String title) {
|
||||
return upgradeToLevelA(pdfBytes, part, language, title, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param alsoDeclareUa additionally claim PDF/UA, but only if it validates
|
||||
*/
|
||||
@Override
|
||||
public Result upgradeToLevelA(
|
||||
byte[] pdfBytes, int part, String language, String title, boolean alsoDeclareUa) {
|
||||
List<String> warnings = new ArrayList<>();
|
||||
try {
|
||||
byte[] tagged;
|
||||
TaggingResult taggingResult;
|
||||
|
||||
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
|
||||
// Tagging holds a model of the whole document; without a cap a large file exhausts
|
||||
// the heap, and OutOfMemoryError is an Error, so the catch below never sees it.
|
||||
if (document.getNumberOfPages() > MAX_TAGGABLE_PAGES) {
|
||||
warnings.add(
|
||||
"This document has "
|
||||
+ document.getNumberOfPages()
|
||||
+ " pages, more than the "
|
||||
+ MAX_TAGGABLE_PAGES
|
||||
+ " that can be tagged, so it was left at conformance level B.");
|
||||
return new Result(pdfBytes, false, warnings);
|
||||
}
|
||||
TaggingOptions options =
|
||||
TaggingOptions.builder()
|
||||
.profile(PdfUaProfile.UA1)
|
||||
.language(language)
|
||||
.title(title)
|
||||
.fallbackTitle(title)
|
||||
// Fonts were embedded on the PDF/A pass; a rewrite would undo it.
|
||||
.embedFonts(false)
|
||||
// PDF/A-1 is defined on PDF 1.4; raising it breaks conformance.
|
||||
.preservePdfVersion(part == 1)
|
||||
.existingTags(TaggingOptions.ExistingTags.AUTO)
|
||||
.build();
|
||||
|
||||
taggingResult = new PdfUaTagger().tag(document, options);
|
||||
warnings.addAll(taggingResult.getWarnings());
|
||||
tagged = save(document, part);
|
||||
}
|
||||
|
||||
if (taggingResult.getTaggedElements() == 0 && taggingResult.isRebuilt()) {
|
||||
warnings.add(
|
||||
"No taggable content was found, so the file cannot claim PDF/A level A."
|
||||
+ " It remains valid at level B.");
|
||||
return new Result(pdfBytes, false, warnings);
|
||||
}
|
||||
if (taggingResult.isContentSuppressed()) {
|
||||
warnings.add(
|
||||
"Some text could not be tagged reliably and was marked as an artifact, so"
|
||||
+ " no level A claim was written. The file remains valid at level B.");
|
||||
return new Result(tagged, false, warnings);
|
||||
}
|
||||
|
||||
byte[] declared = setConformance(tagged, part, "A");
|
||||
|
||||
// Tagging is necessary for level A but not sufficient: Unicode mappings are too.
|
||||
if (!validationService.validatesAsPdfaLevelA(declared, part)) {
|
||||
warnings.add(
|
||||
"The document was tagged but does not validate as PDF/A-"
|
||||
+ part
|
||||
+ "a, so it was left at conformance level B.");
|
||||
return new Result(setConformance(tagged, part, "B"), false, warnings);
|
||||
}
|
||||
|
||||
if (alsoDeclareUa) {
|
||||
byte[] withUa = declarePdfUaAlongsidePdfa(declared, part);
|
||||
var uaResult = validationService.validate(withUa, PdfUaProfile.UA1);
|
||||
if (uaResult.compliant()) {
|
||||
log.info("Upgraded PDF/A-{} to level A and declared PDF/UA", part);
|
||||
return new Result(withUa, true, warnings);
|
||||
}
|
||||
// The archival upgrade stands on its own; only the accessibility claim is dropped.
|
||||
warnings.add(
|
||||
"PDF/UA was requested alongside PDF/A but "
|
||||
+ uaResult.totalFailures()
|
||||
+ " accessibility check(s) still fail, so no PDF/UA claim was"
|
||||
+ " written. The file is valid PDF/A-"
|
||||
+ part
|
||||
+ "a.");
|
||||
}
|
||||
|
||||
log.info("Upgraded PDF/A-{} to conformance level A", part);
|
||||
return new Result(declared, true, warnings);
|
||||
|
||||
} catch (Exception e) {
|
||||
log.warn("Could not upgrade to PDF/A level A: {}", e.getMessage());
|
||||
warnings.add(
|
||||
"Level A upgrade failed ("
|
||||
+ e.getMessage()
|
||||
+ "), so the file was left at conformance level B.");
|
||||
return new Result(pdfBytes, false, warnings);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Declares PDF/UA alongside PDF/A in one file. The extension schema is required: PDF/A forbids
|
||||
* XMP properties no schema describes, and XMPBox has none for {@code pdfuaid}.
|
||||
*/
|
||||
static byte[] declarePdfUaAlongsidePdfa(byte[] pdfBytes, int part) throws Exception {
|
||||
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
|
||||
XMPMetadata xmp = parseOrCreate(document);
|
||||
|
||||
PdfUaIdentificationSchema identification = new PdfUaIdentificationSchema(xmp);
|
||||
identification.setPart(1);
|
||||
xmp.addSchema(identification);
|
||||
|
||||
addPdfUaExtensionSchema(xmp);
|
||||
writeMetadata(document, xmp);
|
||||
return save(document, part);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Describes the pdfuaid namespace so a PDF/A validator accepts it. Fields are set individually,
|
||||
* not by subclassing: XMPBox reads the namespace from an annotation, which is not inherited.
|
||||
*/
|
||||
private static void addPdfUaExtensionSchema(XMPMetadata xmp) {
|
||||
PDFAExtensionSchema extension =
|
||||
(PDFAExtensionSchema) xmp.getSchema(PDFAExtensionSchema.class);
|
||||
if (extension == null) {
|
||||
extension = xmp.createAndAddPDFAExtensionSchemaWithDefaultNS();
|
||||
}
|
||||
|
||||
PDFAPropertyType partProperty = new PDFAPropertyType(xmp);
|
||||
addField(xmp, partProperty, PDFAPropertyType.NAME, "part");
|
||||
addField(xmp, partProperty, PDFAPropertyType.VALUETYPE, "Integer");
|
||||
addField(xmp, partProperty, PDFAPropertyType.CATEGORY, "internal");
|
||||
addField(
|
||||
xmp,
|
||||
partProperty,
|
||||
PDFAPropertyType.DESCRIPTION,
|
||||
"Indicates which part of ISO 14289 the document conforms to");
|
||||
|
||||
PDFASchemaType schema = new PDFASchemaType(xmp);
|
||||
addField(xmp, schema, PDFASchemaType.SCHEMA, "PDF/UA Universal Accessibility Schema");
|
||||
addField(xmp, schema, PDFASchemaType.NAMESPACE_URI, PdfUaIdentificationSchema.NAMESPACE);
|
||||
addField(xmp, schema, PDFASchemaType.PREFIX, PdfUaIdentificationSchema.PREFERRED_PREFIX);
|
||||
|
||||
ArrayProperty properties =
|
||||
xmp.getTypeMapping()
|
||||
.createArrayProperty(
|
||||
schema.getNamespace(),
|
||||
schema.getPrefix(),
|
||||
PDFASchemaType.PROPERTY,
|
||||
Cardinality.Seq);
|
||||
properties.getContainer().addProperty(partProperty);
|
||||
schema.getContainer().addProperty(properties);
|
||||
|
||||
// A freshly created extension schema has no schemas bag yet, so make one.
|
||||
ArrayProperty schemas = extension.getSchemasProperty();
|
||||
if (schemas == null) {
|
||||
schemas =
|
||||
xmp.getTypeMapping()
|
||||
.createArrayProperty(
|
||||
extension.getNamespace(),
|
||||
extension.getPrefix(),
|
||||
PDFAExtensionSchema.SCHEMAS,
|
||||
Cardinality.Bag);
|
||||
extension.addProperty(schemas);
|
||||
}
|
||||
schemas.getContainer().addProperty(schema);
|
||||
}
|
||||
|
||||
/** Adds one text field to a structured type, in that type's own namespace. */
|
||||
private static void addField(
|
||||
XMPMetadata xmp, AbstractStructuredType target, String name, String value) {
|
||||
target.getContainer()
|
||||
.addProperty(
|
||||
xmp.getTypeMapping()
|
||||
.createText(
|
||||
target.getNamespace(), target.getPrefix(), name, value));
|
||||
}
|
||||
|
||||
private static XMPMetadata parseOrCreate(PDDocument document) throws Exception {
|
||||
PDMetadata existing = document.getDocumentCatalog().getMetadata();
|
||||
if (existing == null) {
|
||||
return XMPMetadata.createXMPMetadata();
|
||||
}
|
||||
try (InputStream in = new ByteArrayInputStream(existing.toByteArray())) {
|
||||
DomXmpParser parser = new DomXmpParser();
|
||||
parser.setStrictParsing(false);
|
||||
return parser.parse(in);
|
||||
}
|
||||
}
|
||||
|
||||
private static void writeMetadata(PDDocument document, XMPMetadata xmp) throws Exception {
|
||||
ByteArrayOutputStream serialised = new ByteArrayOutputStream();
|
||||
new XmpSerializer().serialize(xmp, serialised, true);
|
||||
PDMetadata metadata = new PDMetadata(document);
|
||||
metadata.importXMPMetadata(serialised.toByteArray());
|
||||
document.getDocumentCatalog().setMetadata(metadata);
|
||||
}
|
||||
|
||||
/** Rewrites {@code pdfaid:conformance} without disturbing the rest of the packet. */
|
||||
static byte[] setConformance(byte[] pdfBytes, int part, String conformance) throws Exception {
|
||||
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
|
||||
PDMetadata existing = document.getDocumentCatalog().getMetadata();
|
||||
XMPMetadata xmp;
|
||||
if (existing != null) {
|
||||
try (InputStream in = new ByteArrayInputStream(existing.toByteArray())) {
|
||||
DomXmpParser parser = new DomXmpParser();
|
||||
parser.setStrictParsing(false);
|
||||
xmp = parser.parse(in);
|
||||
}
|
||||
} else {
|
||||
xmp = XMPMetadata.createXMPMetadata();
|
||||
}
|
||||
|
||||
PDFAIdentificationSchema identification =
|
||||
(PDFAIdentificationSchema) xmp.getSchema(PDFAIdentificationSchema.class);
|
||||
if (identification == null) {
|
||||
identification = xmp.createAndAddPDFAIdentificationSchema();
|
||||
}
|
||||
identification.setPart(part);
|
||||
identification.setConformance(conformance);
|
||||
|
||||
ByteArrayOutputStream serialised = new ByteArrayOutputStream();
|
||||
new XmpSerializer().serialize(xmp, serialised, true);
|
||||
PDMetadata metadata = new PDMetadata(document);
|
||||
metadata.importXMPMetadata(serialised.toByteArray());
|
||||
document.getDocumentCatalog().setMetadata(metadata);
|
||||
|
||||
return save(document, part);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Part 1 is saved uncompressed: PDFBox's default object streams need PDF 1.5, which would push
|
||||
* a PDF/A-1 file off its required 1.4 version.
|
||||
*/
|
||||
private static byte[] save(PDDocument document, int part) throws IOException {
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
document.save(
|
||||
out, part == 1 ? CompressParameters.NO_COMPRESSION : new CompressParameters());
|
||||
return out.toByteArray();
|
||||
}
|
||||
}
|
||||
+286
@@ -0,0 +1,286 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Nested;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/** Unit tests for the heuristics that decide what a run of text means. */
|
||||
class LayoutAnalyzerTest {
|
||||
|
||||
private static final BBox A4 = new BBox(0, 0, 595, 842);
|
||||
|
||||
private static TextLineInfo line(String text, float size, float x, float y) {
|
||||
return line(text, size, x, y, false, 0, 0);
|
||||
}
|
||||
|
||||
private static TextLineInfo line(
|
||||
String text, float size, float x, float y, boolean bold, int start, int end) {
|
||||
List<WordInfo> words = new ArrayList<>();
|
||||
float cursor = x;
|
||||
for (String token : text.strip().split("\\s+")) {
|
||||
float width = token.length() * size * 0.5f;
|
||||
words.add(
|
||||
new WordInfo(
|
||||
token,
|
||||
new BBox(cursor, y, cursor + width, y + size),
|
||||
start,
|
||||
end,
|
||||
size,
|
||||
bold));
|
||||
cursor += width + size * 0.3f;
|
||||
}
|
||||
return new TextLineInfo(
|
||||
0, text, new BBox(x, y, cursor, y + size), size, bold, start, end, false, words);
|
||||
}
|
||||
|
||||
private static PageContent page(List<TextLineInfo> lines) {
|
||||
return new PageContent(0, lines, List.of(), lines.size(), false, false, false, A4);
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("body font size")
|
||||
class BodyFontSize {
|
||||
|
||||
@Test
|
||||
@DisplayName("weights by characters so one huge title does not skew the baseline")
|
||||
void weightsByCharacterCount() {
|
||||
List<TextLineInfo> lines =
|
||||
List.of(
|
||||
line("A Very Large Title", 32, 50, 700),
|
||||
line("Body text line one which is long", 11, 50, 650),
|
||||
line("Body text line two which is long", 11, 50, 630),
|
||||
line("Body text line three also long", 11, 50, 610));
|
||||
assertEquals(11f, LayoutAnalyzer.bodyFontSize(List.of(page(lines))));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("returns zero when there is no text")
|
||||
void handlesEmptyDocument() {
|
||||
assertEquals(0f, LayoutAnalyzer.bodyFontSize(List.of(page(List.of()))));
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("heading detection")
|
||||
class Headings {
|
||||
|
||||
@Test
|
||||
@DisplayName("assigns distinct sizes to descending levels")
|
||||
void assignsTiers() {
|
||||
List<TextLineInfo> lines =
|
||||
List.of(
|
||||
line("Title", 24, 50, 800),
|
||||
line("Chapter", 18, 50, 750),
|
||||
line("Section", 14, 50, 700),
|
||||
line("Body text that is long enough to set a baseline", 11, 50, 650));
|
||||
Map<Float, Integer> tiers = LayoutAnalyzer.headingTiers(List.of(page(lines)), 11f);
|
||||
assertEquals(1, tiers.get(24f));
|
||||
assertEquals(2, tiers.get(18f));
|
||||
assertEquals(3, tiers.get(14f));
|
||||
assertNull(tiers.get(11f), "body size must not be a heading tier");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("rejects long lines and full sentences whatever their size")
|
||||
void rejectsProse() {
|
||||
assertFalse(
|
||||
LayoutAnalyzer.isHeadingCandidate(
|
||||
line("This line ends like a sentence does.", 20, 50, 700)),
|
||||
"a line ending in a full stop reads as prose");
|
||||
assertFalse(
|
||||
LayoutAnalyzer.isHeadingCandidate(
|
||||
line(
|
||||
"one two three four five six seven eight nine ten eleven twelve"
|
||||
+ " thirteen",
|
||||
20,
|
||||
50,
|
||||
700)),
|
||||
"a long line is body text however large");
|
||||
assertTrue(LayoutAnalyzer.isHeadingCandidate(line("Financial Results", 20, 50, 700)));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("boldness alone never promotes a line to a heading")
|
||||
void boldIsNotAHeadingSignal() {
|
||||
List<TextLineInfo> lines =
|
||||
List.of(
|
||||
line("Bold Label", 11, 50, 700, true, 0, 0),
|
||||
line("Body text long enough to set the baseline here", 11, 50, 650));
|
||||
assertTrue(
|
||||
LayoutAnalyzer.headingTiers(List.of(page(lines)), 11f).isEmpty(),
|
||||
"a bold line at body size is emphasis, not a heading");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("rewrites skipped levels so H1 is never followed by H3")
|
||||
void normalisesSkippedLevels() {
|
||||
DocumentStructure structure = new DocumentStructure();
|
||||
structure.add(new StructBlock(StructType.H1, 0));
|
||||
structure.add(new StructBlock(StructType.H3, 0));
|
||||
structure.add(new StructBlock(StructType.H4, 0));
|
||||
LayoutAnalyzer.normaliseHeadingLevels(structure);
|
||||
|
||||
assertEquals(StructType.H1, structure.getBlocks().get(0).getType());
|
||||
assertEquals(StructType.H2, structure.getBlocks().get(1).getType());
|
||||
assertEquals(StructType.H3, structure.getBlocks().get(2).getType());
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("lists")
|
||||
class Lists {
|
||||
|
||||
@Test
|
||||
@DisplayName("recognises bullet and ordered markers")
|
||||
void recognisesMarkers() {
|
||||
assertTrue(LayoutAnalyzer.startsListItem(line("• First item", 11, 50, 700)));
|
||||
assertTrue(LayoutAnalyzer.startsListItem(line("- First item", 11, 50, 700)));
|
||||
assertTrue(LayoutAnalyzer.startsListItem(line("1. First item", 11, 50, 700)));
|
||||
assertTrue(LayoutAnalyzer.startsListItem(line("a) First item", 11, 50, 700)));
|
||||
assertFalse(LayoutAnalyzer.startsListItem(line("Ordinary prose here", 11, 50, 700)));
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("table cells")
|
||||
class Tables {
|
||||
|
||||
@Test
|
||||
@DisplayName("splits a row at wide gaps but not at ordinary word spacing")
|
||||
void splitsOnWideGaps() {
|
||||
List<WordInfo> words =
|
||||
List.of(
|
||||
new WordInfo("Region", new BBox(50, 700, 90, 711), 0, 0, 11, false),
|
||||
new WordInfo("name", new BBox(93, 700, 125, 711), 0, 0, 11, false),
|
||||
new WordInfo("Units", new BBox(250, 700, 285, 711), 1, 1, 11, false));
|
||||
TextLineInfo row =
|
||||
new TextLineInfo(
|
||||
0,
|
||||
"Region name Units",
|
||||
new BBox(50, 700, 285, 711),
|
||||
11,
|
||||
false,
|
||||
0,
|
||||
1,
|
||||
false,
|
||||
words);
|
||||
List<List<WordInfo>> cells = LayoutAnalyzer.splitCells(row);
|
||||
assertEquals(2, cells.size(), "the small gap is a word space, the large one is a cell");
|
||||
assertEquals(2, cells.get(0).size());
|
||||
assertEquals("Units", cells.get(1).get(0).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("words sharing an operator cannot become separate cells")
|
||||
void detectsInseparableWords() {
|
||||
List<WordInfo> shared =
|
||||
List.of(
|
||||
new WordInfo("A", new BBox(50, 700, 60, 711), 3, 3, 11, false),
|
||||
new WordInfo("B", new BBox(250, 700, 260, 711), 3, 3, 11, false));
|
||||
TextLineInfo row =
|
||||
new TextLineInfo(
|
||||
0, "A B", new BBox(50, 700, 260, 711), 11, false, 3, 3, false, shared);
|
||||
assertFalse(
|
||||
row.wordsAreSeparable(),
|
||||
"cells drawn by one operator cannot carry separate marked content ids");
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("running heads")
|
||||
class RunningHeads {
|
||||
|
||||
@Test
|
||||
@DisplayName("treats text repeating in the margin band across pages as an artifact")
|
||||
void findsRepeatedMarginText() {
|
||||
List<PageContent> pages = new ArrayList<>();
|
||||
for (int i = 0; i < 4; i++) {
|
||||
List<TextLineInfo> lines =
|
||||
List.of(
|
||||
line("Confidential Report", 9, 50, 800),
|
||||
line("Body content for the page", 11, 50, 400),
|
||||
line("Page " + (i + 1), 9, 300, 20));
|
||||
pages.add(new PageContent(i, lines, List.of(), 3, false, false, false, A4));
|
||||
}
|
||||
Map<Integer, List<TextLineInfo>> artifacts = LayoutAnalyzer.repeatedMarginLines(pages);
|
||||
assertEquals(
|
||||
2, artifacts.get(0).size(), "the running head and the folio are artifacts");
|
||||
assertTrue(
|
||||
artifacts.get(0).stream().noneMatch(l -> l.text().contains("Body content")),
|
||||
"body text must never be demoted to an artifact");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("does not treat a one-off margin line as a running head")
|
||||
void ignoresUniqueMarginText() {
|
||||
List<String> titles = List.of("Alpha", "Beta", "Gamma", "Delta");
|
||||
List<PageContent> pages = new ArrayList<>();
|
||||
for (int i = 0; i < 4; i++) {
|
||||
List<TextLineInfo> lines =
|
||||
List.of(
|
||||
line(titles.get(i) + " overview", 9, 50, 800),
|
||||
line("Body content", 11, 50, 400));
|
||||
pages.add(new PageContent(i, lines, List.of(), 2, false, false, false, A4));
|
||||
}
|
||||
assertTrue(LayoutAnalyzer.repeatedMarginLines(pages).get(0).isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a large heading high on the page stays a heading, not chrome")
|
||||
void doesNotDemoteHeadingsNearTheTop() {
|
||||
List<PageContent> pages = new ArrayList<>();
|
||||
for (int i = 0; i < 4; i++) {
|
||||
List<TextLineInfo> lines =
|
||||
List.of(
|
||||
// Masking digits makes these look identical across pages.
|
||||
line("Section " + (i + 1), 20, 50, 800),
|
||||
line("Body text long enough to set the baseline", 11, 50, 400));
|
||||
pages.add(new PageContent(i, lines, List.of(), 2, false, false, false, A4));
|
||||
}
|
||||
assertTrue(
|
||||
LayoutAnalyzer.repeatedMarginLines(pages, 11f).get(0).isEmpty(),
|
||||
"a heading larger than body text is content, wherever it sits");
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("columns")
|
||||
class Columns {
|
||||
|
||||
@Test
|
||||
@DisplayName("detects a gutter when text sits in two balanced blocks")
|
||||
void detectsTwoColumns() {
|
||||
List<TextLineInfo> lines = new ArrayList<>();
|
||||
for (int i = 0; i < 6; i++) {
|
||||
lines.add(line("Left column text", 10, 50, 700 - i * 14));
|
||||
lines.add(line("Right column text", 10, 320, 700 - i * 14));
|
||||
}
|
||||
assertNotNull(LayoutAnalyzer.detectGutter(page(lines)));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("does not split a page whose lines span the full width")
|
||||
void ignoresSingleColumn() {
|
||||
List<TextLineInfo> lines = new ArrayList<>();
|
||||
for (int i = 0; i < 10; i++) {
|
||||
lines.add(
|
||||
line(
|
||||
"A full width line of prose that crosses the centre of the page",
|
||||
10,
|
||||
50,
|
||||
700 - i * 14));
|
||||
}
|
||||
assertNull(LayoutAnalyzer.detectGutter(page(lines)));
|
||||
}
|
||||
}
|
||||
}
|
||||
+164
@@ -0,0 +1,164 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.PDPageContentStream;
|
||||
import org.apache.pdfbox.pdmodel.common.PDRectangle;
|
||||
import org.apache.pdfbox.pdmodel.font.PDFont;
|
||||
import org.apache.pdfbox.pdmodel.font.PDType1Font;
|
||||
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/** Tests the content-stream rewriting that makes tagging possible. */
|
||||
class MarkedContentInjectorTest {
|
||||
|
||||
private static byte[] threeLinePdf() throws IOException {
|
||||
try (PDDocument document = new PDDocument()) {
|
||||
PDPage page = new PDPage(PDRectangle.A4);
|
||||
document.addPage(page);
|
||||
PDFont font = new PDType1Font(Standard14Fonts.FontName.HELVETICA);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
|
||||
for (int i = 0; i < 3; i++) {
|
||||
cs.beginText();
|
||||
cs.setFont(font, 12);
|
||||
cs.newLineAtOffset(50, 700 - i * 20);
|
||||
cs.showText("Line " + i);
|
||||
cs.endText();
|
||||
}
|
||||
}
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
document.save(out);
|
||||
return out.toByteArray();
|
||||
}
|
||||
}
|
||||
|
||||
private static String contentOf(PDDocument document) throws IOException {
|
||||
try (InputStream in = document.getPage(0).getContents()) {
|
||||
return new String(in.readAllBytes(), StandardCharsets.ISO_8859_1);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("wraps claimed content in BDC/EMC with a marked content id")
|
||||
void wrapsClaimedContent() throws Exception {
|
||||
try (PDDocument document = Loader.loadPDF(threeLinePdf())) {
|
||||
StructBlock paragraph = new StructBlock(StructType.P, 0);
|
||||
paragraph.addRange(0, 1);
|
||||
|
||||
int next =
|
||||
new MarkedContentInjector()
|
||||
.inject(document, document.getPage(0), List.of(paragraph), 0, true);
|
||||
|
||||
String content = contentOf(document);
|
||||
assertTrue(content.contains("/P"), "the structure type was not written");
|
||||
assertTrue(content.contains("/MCID"), "no marked content id was written");
|
||||
assertTrue(content.contains("BDC"), "no marked content sequence was opened");
|
||||
assertTrue(content.contains("EMC"), "no marked content sequence was closed");
|
||||
assertFalse(paragraph.getMcids().isEmpty(), "the block was given no marked content id");
|
||||
assertTrue(next > 0, "the id counter did not advance");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("marks unclaimed content as an artifact so nothing is left untagged")
|
||||
void unclaimedContentBecomesArtifact() throws Exception {
|
||||
try (PDDocument document = Loader.loadPDF(threeLinePdf())) {
|
||||
StructBlock paragraph = new StructBlock(StructType.P, 0);
|
||||
paragraph.addRange(0, 0);
|
||||
|
||||
new MarkedContentInjector()
|
||||
.inject(document, document.getPage(0), List.of(paragraph), 0, true);
|
||||
|
||||
String content = contentOf(document);
|
||||
assertTrue(
|
||||
content.contains("/Artifact"),
|
||||
"content nobody claimed must be marked as an artifact, or PDF/UA clause 7.1"
|
||||
+ " fails");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("opens and closes sequences in balanced pairs")
|
||||
void sequencesAreBalanced() throws Exception {
|
||||
try (PDDocument document = Loader.loadPDF(threeLinePdf())) {
|
||||
StructBlock first = new StructBlock(StructType.P, 0);
|
||||
first.addRange(0, 0);
|
||||
StructBlock second = new StructBlock(StructType.H1, 0);
|
||||
second.addRange(2, 2);
|
||||
|
||||
new MarkedContentInjector()
|
||||
.inject(document, document.getPage(0), List.of(first, second), 0, true);
|
||||
|
||||
String content = contentOf(document);
|
||||
int opens = count(content, "BDC") + count(content, "BMC");
|
||||
int closes = count(content, "EMC");
|
||||
assertEquals(opens, closes, "every opened sequence must be closed");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("rewriting does not change what a reader extracts")
|
||||
void textIsUnchanged() throws Exception {
|
||||
byte[] original = threeLinePdf();
|
||||
String before = extract(original);
|
||||
|
||||
byte[] rewritten;
|
||||
try (PDDocument document = Loader.loadPDF(original)) {
|
||||
StructBlock paragraph = new StructBlock(StructType.P, 0);
|
||||
paragraph.addRange(0, 2);
|
||||
new MarkedContentInjector()
|
||||
.inject(document, document.getPage(0), List.of(paragraph), 0, true);
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
document.save(out);
|
||||
rewritten = out.toByteArray();
|
||||
}
|
||||
assertEquals(before, extract(rewritten), "marked content operators must not render");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("two blocks claiming the same content keep the first, not both")
|
||||
void overlappingClaimsAreResolved() {
|
||||
StructBlock first = new StructBlock(StructType.P, 0);
|
||||
first.addRange(0, 2);
|
||||
StructBlock second = new StructBlock(StructType.H1, 0);
|
||||
second.addRange(1, 1);
|
||||
|
||||
Map<Integer, StructBlock> owners =
|
||||
MarkedContentInjector.ownersByOrdinal(List.of(first, second));
|
||||
assertSame(first, owners.get(1), "the first claim wins so reading order stays unambiguous");
|
||||
assertEquals(3, owners.size());
|
||||
}
|
||||
|
||||
private static int count(String haystack, String needle) {
|
||||
int total = 0;
|
||||
int index = 0;
|
||||
while ((index = haystack.indexOf(needle, index)) >= 0) {
|
||||
total++;
|
||||
index += needle.length();
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
private static String extract(byte[] pdf) throws IOException {
|
||||
try (PDDocument document = Loader.loadPDF(pdf)) {
|
||||
return new org.apache.pdfbox.text.PDFTextStripper()
|
||||
.getText(document)
|
||||
.replaceAll("\\s+", " ")
|
||||
.strip();
|
||||
}
|
||||
}
|
||||
}
|
||||
+193
@@ -0,0 +1,193 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.PDPageContentStream;
|
||||
import org.apache.pdfbox.pdmodel.common.PDRectangle;
|
||||
import org.apache.pdfbox.pdmodel.common.PDStream;
|
||||
import org.apache.pdfbox.pdmodel.font.PDType1Font;
|
||||
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/** Regression tests for rewriter damage a validator cannot see, so it still passes validation. */
|
||||
class MarkedContentSafetyTest {
|
||||
|
||||
private static String contentOf(PDDocument document) throws IOException {
|
||||
try (InputStream in = document.getPage(0).getContents()) {
|
||||
return new String(in.readAllBytes(), StandardCharsets.ISO_8859_1);
|
||||
}
|
||||
}
|
||||
|
||||
private static void setContent(PDDocument document, String content) throws IOException {
|
||||
PDStream stream = new PDStream(document);
|
||||
try (var out = stream.createOutputStream()) {
|
||||
out.write(content.getBytes(StandardCharsets.ISO_8859_1));
|
||||
}
|
||||
document.getPage(0).setContents(stream);
|
||||
}
|
||||
|
||||
private static PDDocument onePage() throws IOException {
|
||||
PDDocument document = new PDDocument();
|
||||
PDPage page = new PDPage(PDRectangle.A4);
|
||||
document.addPage(page);
|
||||
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
|
||||
cs.beginText();
|
||||
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
|
||||
cs.newLineAtOffset(50, 700);
|
||||
cs.showText("visible");
|
||||
cs.endText();
|
||||
}
|
||||
return document;
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an optional-content layer survives the rebuild, so hidden content stays hidden")
|
||||
void optionalContentIsPreserved() throws Exception {
|
||||
try (PDDocument document = onePage()) {
|
||||
String original = contentOf(document);
|
||||
setContent(document, "/OC /MC0 BDC\n" + original + "\nEMC\n");
|
||||
|
||||
StructBlock paragraph = new StructBlock(StructType.P, 0);
|
||||
paragraph.addRange(0, 0);
|
||||
new MarkedContentInjector()
|
||||
.inject(document, document.getPage(0), List.of(paragraph), 0, true);
|
||||
|
||||
String rewritten = contentOf(document);
|
||||
assertTrue(
|
||||
rewritten.contains("/OC"),
|
||||
"the optional-content wrapper was stripped, which would make a hidden"
|
||||
+ " DRAFT/CONFIDENTIAL or redaction layer permanently visible:\n"
|
||||
+ rewritten);
|
||||
assertEquals(
|
||||
countOf(rewritten, "BDC") + countOf(rewritten, "BMC"),
|
||||
countOf(rewritten, "EMC"),
|
||||
"marked content is unbalanced after preserving the layer");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("replacement text survives the rebuild so ligatures still read correctly")
|
||||
void actualTextIsPreserved() throws Exception {
|
||||
try (PDDocument document = onePage()) {
|
||||
String original = contentOf(document);
|
||||
// A generator marks an ffi ligature with what it really spells.
|
||||
setContent(
|
||||
document, "/Span <</ActualText (ffi) /MCID 7>> BDC\n" + original + "\nEMC\n");
|
||||
|
||||
StructBlock paragraph = new StructBlock(StructType.P, 0);
|
||||
paragraph.addRange(0, 0);
|
||||
new MarkedContentInjector()
|
||||
.inject(document, document.getPage(0), List.of(paragraph), 0, true);
|
||||
|
||||
String rewritten = contentOf(document);
|
||||
assertTrue(
|
||||
rewritten.contains("ActualText"),
|
||||
"dropping ActualText leaves a screen reader announcing the raw glyph:\n"
|
||||
+ rewritten);
|
||||
assertFalse(
|
||||
rewritten.contains("/MCID 7"),
|
||||
"the source's own marked content id is meaningless after a rebuild");
|
||||
assertEquals(
|
||||
countOf(rewritten, "BDC") + countOf(rewritten, "BMC"),
|
||||
countOf(rewritten, "EMC"),
|
||||
"marked content is unbalanced after preserving replacement text");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a sequence wrapping a fill opens before the path, not inside it")
|
||||
void markedContentNeverOpensInsideAPathObject() throws Exception {
|
||||
try (PDDocument document = onePage()) {
|
||||
setContent(document, "0 0 0 rg\n10 10 50 5 re\nf\n");
|
||||
|
||||
new MarkedContentInjector().inject(document, document.getPage(0), List.of(), 0, true);
|
||||
|
||||
String rewritten = contentOf(document);
|
||||
int reAt = rewritten.indexOf(" re");
|
||||
int openAt = Math.max(rewritten.indexOf("BMC"), rewritten.indexOf("BDC"));
|
||||
assertTrue(openAt >= 0, "no sequence was opened at all: " + rewritten);
|
||||
assertTrue(
|
||||
openAt < reAt,
|
||||
"ISO 32000-1 does not permit a marked-content operator inside a path object;"
|
||||
+ " the sequence must open before the path construction:\n"
|
||||
+ rewritten);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("words drawn out of stream order are still claimed, not silently artifacted")
|
||||
void outOfOrderWordsAreClaimed() {
|
||||
// A line whose second word on the page was painted first: ordinals 1 then 0.
|
||||
WordInfo right = new WordInfo("label", new BBox(50, 700, 90, 712), 1, 1, 11, false);
|
||||
WordInfo left = new WordInfo("value", new BBox(200, 700, 240, 712), 0, 0, 11, false);
|
||||
TextLineInfo line =
|
||||
new TextLineInfo(
|
||||
0,
|
||||
"label value",
|
||||
new BBox(50, 700, 240, 712),
|
||||
11,
|
||||
false,
|
||||
0,
|
||||
1,
|
||||
false,
|
||||
List.of(right, left));
|
||||
|
||||
PageContent page =
|
||||
new PageContent(
|
||||
0,
|
||||
List.of(line),
|
||||
List.of(),
|
||||
2,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
new BBox(0, 0, 595, 842));
|
||||
DocumentStructure structure = new LayoutAnalyzer().analyse(List.of(page));
|
||||
|
||||
boolean[] claimed = new boolean[2];
|
||||
structure.visit(
|
||||
block -> {
|
||||
if (block.isArtifact()) {
|
||||
return;
|
||||
}
|
||||
block.getRanges()
|
||||
.forEach(
|
||||
r -> {
|
||||
for (int i = r.start(); i <= r.end() && i < 2; i++) {
|
||||
claimed[i] = true;
|
||||
}
|
||||
});
|
||||
});
|
||||
assertTrue(
|
||||
claimed[0] && claimed[1],
|
||||
"an out-of-order word was left unclaimed and would be hidden from assistive"
|
||||
+ " technology while the file still validated");
|
||||
}
|
||||
|
||||
private static int countOf(String haystack, String needle) {
|
||||
int total = 0;
|
||||
int index = 0;
|
||||
while ((index = haystack.indexOf(needle, index)) >= 0) {
|
||||
total++;
|
||||
index += needle.length();
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
private static byte[] bytes(PDDocument document) throws IOException {
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
document.save(out);
|
||||
return out.toByteArray();
|
||||
}
|
||||
}
|
||||
+171
@@ -0,0 +1,171 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.List;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.apache.pdfbox.pdmodel.common.PDRectangle;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureElement;
|
||||
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureNode;
|
||||
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationWidget;
|
||||
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
|
||||
import org.apache.pdfbox.pdmodel.interactive.form.PDTextField;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/** Covers form-field descriptions, widget nesting and withdrawing a conformance claim. */
|
||||
class PdfUaFormAndDeclarationTest {
|
||||
|
||||
/** A document with one named text field and one unnamed one. */
|
||||
private static PDDocument formDocument(boolean nameTheSecondField) throws IOException {
|
||||
PDDocument document = new PDDocument();
|
||||
PDPage page = new PDPage(PDRectangle.A4);
|
||||
document.addPage(page);
|
||||
|
||||
PDAcroForm form = new PDAcroForm(document);
|
||||
document.getDocumentCatalog().setAcroForm(form);
|
||||
|
||||
PDTextField named = new PDTextField(form);
|
||||
named.setPartialName("EmailAddress");
|
||||
addWidget(named, page, 700);
|
||||
form.getFields().add(named);
|
||||
|
||||
PDTextField second = new PDTextField(form);
|
||||
if (nameTheSecondField) {
|
||||
second.setPartialName("PostCode");
|
||||
}
|
||||
addWidget(second, page, 650);
|
||||
form.getFields().add(second);
|
||||
|
||||
return document;
|
||||
}
|
||||
|
||||
private static void addWidget(PDTextField field, PDPage page, float y) throws IOException {
|
||||
PDAnnotationWidget widget = field.getWidgets().get(0);
|
||||
PDRectangle rectangle = new PDRectangle();
|
||||
rectangle.setLowerLeftX(50);
|
||||
rectangle.setLowerLeftY(y);
|
||||
rectangle.setUpperRightX(250);
|
||||
rectangle.setUpperRightY(y + 18);
|
||||
widget.setRectangle(rectangle);
|
||||
widget.setPage(page);
|
||||
page.getAnnotations().add(widget);
|
||||
}
|
||||
|
||||
private static String xmpOf(PDDocument document) throws IOException {
|
||||
var metadata = document.getDocumentCatalog().getMetadata();
|
||||
assertNotNull(metadata, "no XMP packet");
|
||||
return new String(metadata.toByteArray(), StandardCharsets.UTF_8);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a form field gets its tooltip from its own name, not an invented one")
|
||||
void derivesTooltipFromFieldName() throws Exception {
|
||||
try (PDDocument document = formDocument(true)) {
|
||||
List<String> warnings =
|
||||
new PdfUaMetadataWriter()
|
||||
.applyDocumentRequirements(document, "Form", "en", PdfUaProfile.UA1);
|
||||
|
||||
PDAcroForm form = document.getDocumentCatalog().getAcroForm();
|
||||
assertEquals("EmailAddress", form.getField("EmailAddress").getAlternateFieldName());
|
||||
assertEquals("PostCode", form.getField("PostCode").getAlternateFieldName());
|
||||
assertTrue(warnings.isEmpty(), "nothing needed reporting: " + warnings);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a field with no name is reported rather than given a placeholder tooltip")
|
||||
void reportsUnnameableField() throws Exception {
|
||||
try (PDDocument document = formDocument(false)) {
|
||||
List<String> warnings =
|
||||
new PdfUaMetadataWriter()
|
||||
.applyDocumentRequirements(document, "Form", "en", PdfUaProfile.UA1);
|
||||
assertEquals(1, warnings.size());
|
||||
assertTrue(warnings.get(0).contains("form field"), warnings.get(0));
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an existing description is never overwritten")
|
||||
void keepsExistingDescription() throws Exception {
|
||||
try (PDDocument document = formDocument(true)) {
|
||||
PDAcroForm form = document.getDocumentCatalog().getAcroForm();
|
||||
form.getField("EmailAddress").setAlternateFieldName("Your email address");
|
||||
|
||||
new PdfUaMetadataWriter()
|
||||
.applyDocumentRequirements(document, "Form", "en", PdfUaProfile.UA1);
|
||||
assertEquals(
|
||||
"Your email address", form.getField("EmailAddress").getAlternateFieldName());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("widget annotations are nested inside a Form structure element")
|
||||
void widgetsAreNestedInFormElements() throws Exception {
|
||||
try (PDDocument document = formDocument(true)) {
|
||||
DocumentStructure structure = new DocumentStructure();
|
||||
StructBlock paragraph = new StructBlock(StructType.P, 0);
|
||||
paragraph.getMcids().add(0);
|
||||
structure.add(paragraph);
|
||||
|
||||
new StructTreeWriter().write(document, structure, PdfUaProfile.UA1);
|
||||
|
||||
var root = document.getDocumentCatalog().getStructureTreeRoot();
|
||||
assertTrue(
|
||||
typesUnder(root).contains("Form"),
|
||||
"clause 7.18.4 requires a widget to sit inside a Form element, found: "
|
||||
+ typesUnder(root));
|
||||
}
|
||||
}
|
||||
|
||||
private static List<String> typesUnder(PDStructureNode node) {
|
||||
List<String> types = new java.util.ArrayList<>();
|
||||
for (Object kid : node.getKids()) {
|
||||
if (kid instanceof PDStructureElement element) {
|
||||
types.add(element.getStructureType());
|
||||
types.addAll(typesUnder(element));
|
||||
}
|
||||
}
|
||||
return types;
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("withdrawing conformance removes the claim but keeps the other metadata")
|
||||
void withdrawingConformanceRemovesOnlyTheClaim() throws Exception {
|
||||
try (PDDocument document = new PDDocument()) {
|
||||
document.addPage(new PDPage());
|
||||
PdfUaTagger tagger = new PdfUaTagger();
|
||||
PdfUaMetadataWriter writer = new PdfUaMetadataWriter();
|
||||
|
||||
writer.applyDocumentRequirements(document, "Kept Title", "en-GB", PdfUaProfile.UA1);
|
||||
writer.declareConformance(document, PdfUaProfile.UA1);
|
||||
assertTrue(xmpOf(document).contains("pdfuaid"));
|
||||
|
||||
tagger.withdrawConformance(document);
|
||||
|
||||
String xmp = xmpOf(document);
|
||||
assertFalse(xmp.contains("pdfuaid"), "the conformance claim should be gone");
|
||||
assertTrue(xmp.contains("Kept Title"), "the title should survive");
|
||||
assertEquals("en-GB", document.getDocumentCatalog().getLanguage());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("withdrawing conformance on a document that never claimed it is harmless")
|
||||
void withdrawingIsIdempotent() throws Exception {
|
||||
try (PDDocument document = new PDDocument()) {
|
||||
document.addPage(new PDPage());
|
||||
new PdfUaMetadataWriter()
|
||||
.applyDocumentRequirements(document, "Title", "en", PdfUaProfile.UA1);
|
||||
new PdfUaTagger().withdrawConformance(document);
|
||||
assertFalse(xmpOf(document).contains("pdfuaid"));
|
||||
}
|
||||
}
|
||||
}
|
||||
+79
@@ -0,0 +1,79 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/**
|
||||
* A relabelled language is invisible to every validator, so the tagger must not guess over one the
|
||||
* document already declares.
|
||||
*/
|
||||
class PdfUaLanguageTest {
|
||||
|
||||
private static PDDocument documentWithLanguage(String language) {
|
||||
PDDocument document = new PDDocument();
|
||||
document.addPage(new PDPage());
|
||||
if (language != null) {
|
||||
document.getDocumentCatalog().setLanguage(language);
|
||||
}
|
||||
return document;
|
||||
}
|
||||
|
||||
private static TaggingOptions.TaggingOptionsBuilder options() {
|
||||
return TaggingOptions.builder()
|
||||
.profile(PdfUaProfile.UA1)
|
||||
.language("en-GB")
|
||||
.title("Rapport")
|
||||
.embedFonts(false);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("keeps the language the document already declares")
|
||||
void keepsExistingLanguage() throws Exception {
|
||||
try (PDDocument document = documentWithLanguage("fr-FR")) {
|
||||
TaggingResult result = new PdfUaTagger().tag(document, options().build());
|
||||
|
||||
assertEquals("fr-FR", document.getDocumentCatalog().getLanguage());
|
||||
assertTrue(
|
||||
result.getWarnings().stream().anyMatch(w -> w.contains("fr-FR")),
|
||||
"ignoring the requested language must be reported: " + result.getWarnings());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("applies the requested language when the document declares none")
|
||||
void fillsInMissingLanguage() throws Exception {
|
||||
try (PDDocument document = documentWithLanguage(null)) {
|
||||
new PdfUaTagger().tag(document, options().build());
|
||||
|
||||
assertEquals("en-GB", document.getDocumentCatalog().getLanguage());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("replaces the declared language only when the caller asks")
|
||||
void overridesOnRequest() throws Exception {
|
||||
try (PDDocument document = documentWithLanguage("fr-FR")) {
|
||||
new PdfUaTagger().tag(document, options().overrideLanguage(true).build());
|
||||
|
||||
assertEquals("en-GB", document.getDocumentCatalog().getLanguage());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("keeps the existing language when an existing structure tree is left alone")
|
||||
void keepsExistingLanguageWithoutRebuilding() throws Exception {
|
||||
try (PDDocument document = documentWithLanguage("de-DE")) {
|
||||
new PdfUaTagger()
|
||||
.tag(
|
||||
document,
|
||||
options().existingTags(TaggingOptions.ExistingTags.KEEP).build());
|
||||
|
||||
assertEquals("de-DE", document.getDocumentCatalog().getLanguage());
|
||||
}
|
||||
}
|
||||
}
|
||||
+137
@@ -0,0 +1,137 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.io.ByteArrayOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.apache.pdfbox.pdmodel.PDPage;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/** Tests the document-level requirements that have nothing to do with tagging. */
|
||||
class PdfUaMetadataWriterTest {
|
||||
|
||||
private static PDDocument twoPageDocument() {
|
||||
PDDocument document = new PDDocument();
|
||||
document.addPage(new PDPage());
|
||||
document.addPage(new PDPage());
|
||||
return document;
|
||||
}
|
||||
|
||||
private static String xmpOf(PDDocument document) throws IOException {
|
||||
var metadata = document.getDocumentCatalog().getMetadata();
|
||||
assertNotNull(metadata, "no XMP packet was written");
|
||||
return new String(metadata.toByteArray(), StandardCharsets.UTF_8);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("sets title, language, tab order and the display-title flag")
|
||||
void appliesDocumentRequirements() throws Exception {
|
||||
try (PDDocument document = twoPageDocument()) {
|
||||
new PdfUaMetadataWriter()
|
||||
.applyDocumentRequirements(
|
||||
document, "Annual Report", "en-GB", PdfUaProfile.UA1);
|
||||
|
||||
assertEquals("en-GB", document.getDocumentCatalog().getLanguage());
|
||||
assertEquals("Annual Report", document.getDocumentInformation().getTitle());
|
||||
assertTrue(
|
||||
document.getDocumentCatalog().getViewerPreferences().displayDocTitle(),
|
||||
"without DisplayDocTitle a viewer shows the filename instead of the title");
|
||||
|
||||
for (PDPage page : document.getPages()) {
|
||||
assertEquals(
|
||||
"S",
|
||||
page.getCOSObject().getNameAsString(COSName.getPDFName("Tabs")),
|
||||
"clause 7.18.1 requires an explicit tab order on every page");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("writes dc:title into the XMP packet, not just the info dictionary")
|
||||
void writesDublinCoreTitle() throws Exception {
|
||||
try (PDDocument document = twoPageDocument()) {
|
||||
new PdfUaMetadataWriter()
|
||||
.applyDocumentRequirements(
|
||||
document, "Annual Report", "en-GB", PdfUaProfile.UA1);
|
||||
assertTrue(xmpOf(document).contains("Annual Report"));
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("does not declare conformance as part of applying requirements")
|
||||
void doesNotDeclareEarly() throws Exception {
|
||||
try (PDDocument document = twoPageDocument()) {
|
||||
new PdfUaMetadataWriter()
|
||||
.applyDocumentRequirements(document, "Report", "en", PdfUaProfile.UA1);
|
||||
assertFalse(
|
||||
xmpOf(document).contains("pdfuaid"),
|
||||
"the conformance claim must wait until validation has passed");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("declaring conformance writes pdfuaid with the right part")
|
||||
void declaresConformance() throws Exception {
|
||||
try (PDDocument document = twoPageDocument()) {
|
||||
PdfUaMetadataWriter writer = new PdfUaMetadataWriter();
|
||||
writer.applyDocumentRequirements(document, "Report", "en", PdfUaProfile.UA1);
|
||||
writer.declareConformance(document, PdfUaProfile.UA1);
|
||||
|
||||
String xmp = xmpOf(document);
|
||||
assertTrue(xmp.contains("pdfuaid"), "no PDF/UA identifier was written");
|
||||
assertTrue(xmp.contains("part"), "no conformance part was written");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("UA-2 raises the PDF version to 2.0")
|
||||
void ua2RaisesVersion() throws Exception {
|
||||
try (PDDocument document = twoPageDocument()) {
|
||||
new PdfUaMetadataWriter()
|
||||
.applyDocumentRequirements(document, "Report", "en", PdfUaProfile.UA2);
|
||||
assertEquals(2.0f, document.getVersion());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("keeps an existing title when none is supplied")
|
||||
void keepsExistingTitle() throws Exception {
|
||||
try (PDDocument document = twoPageDocument()) {
|
||||
var info = document.getDocumentInformation();
|
||||
info.setTitle("Original Title");
|
||||
document.setDocumentInformation(info);
|
||||
|
||||
new PdfUaMetadataWriter()
|
||||
.applyDocumentRequirements(document, null, "en", PdfUaProfile.UA1);
|
||||
assertEquals("Original Title", document.getDocumentInformation().getTitle());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("survives a round trip through save and reload")
|
||||
void survivesRoundTrip() throws Exception {
|
||||
byte[] saved;
|
||||
try (PDDocument document = twoPageDocument()) {
|
||||
PdfUaMetadataWriter writer = new PdfUaMetadataWriter();
|
||||
writer.applyDocumentRequirements(document, "Round Trip", "fr-FR", PdfUaProfile.UA1);
|
||||
writer.declareConformance(document, PdfUaProfile.UA1);
|
||||
ByteArrayOutputStream out = new ByteArrayOutputStream();
|
||||
document.save(out);
|
||||
saved = out.toByteArray();
|
||||
}
|
||||
try (PDDocument reloaded = Loader.loadPDF(saved)) {
|
||||
assertEquals("fr-FR", reloaded.getDocumentCatalog().getLanguage());
|
||||
assertEquals("Round Trip", reloaded.getDocumentInformation().getTitle());
|
||||
assertTrue(xmpOf(reloaded).contains("pdfuaid"));
|
||||
}
|
||||
}
|
||||
}
|
||||
+160
@@ -0,0 +1,160 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Nested;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.params.ParameterizedTest;
|
||||
import org.junit.jupiter.params.provider.ValueSource;
|
||||
|
||||
/** Tests for the small types the tagger is built from. */
|
||||
class PdfUaModelTest {
|
||||
|
||||
@Nested
|
||||
@DisplayName("structure types")
|
||||
class Types {
|
||||
|
||||
@Test
|
||||
@DisplayName("maps levels to heading tags and back")
|
||||
void headingLevelsRoundTrip() {
|
||||
for (int level = 1; level <= 6; level++) {
|
||||
assertEquals(level, StructType.heading(level).headingLevel());
|
||||
assertEquals("H" + level, StructType.heading(level).tag());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("clamps out-of-range levels rather than throwing")
|
||||
void clampsLevels() {
|
||||
assertEquals(StructType.H1, StructType.heading(0));
|
||||
assertEquals(StructType.H1, StructType.heading(-3));
|
||||
assertEquals(StructType.H6, StructType.heading(9));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("reports zero for types that are not headings")
|
||||
void nonHeadingsHaveNoLevel() {
|
||||
assertEquals(0, StructType.P.headingLevel());
|
||||
assertFalse(StructType.TABLE.isHeading());
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("markable operators")
|
||||
class Markable {
|
||||
|
||||
@ParameterizedTest
|
||||
@ValueSource(strings = {"Tj", "TJ", "'", "\"", "Do", "BI", "S", "f", "f*", "B", "sh"})
|
||||
@DisplayName("counts text, XObjects and path painting")
|
||||
void counted(String operator) {
|
||||
assertTrue(MarkableOp.isMarkableOperator(operator), operator + " should be markable");
|
||||
}
|
||||
|
||||
@ParameterizedTest
|
||||
@ValueSource(strings = {"q", "Q", "cm", "BT", "ET", "Tf", "Td", "n", "W", "gs", "re"})
|
||||
@DisplayName("ignores operators that paint nothing")
|
||||
void notCounted(String operator) {
|
||||
assertFalse(
|
||||
MarkableOp.isMarkableOperator(operator), operator + " should not be markable");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("n ends a path without painting, so it is not content")
|
||||
void pathEndIsNotPainting() {
|
||||
assertFalse(MarkableOp.isPathPainting("n"));
|
||||
assertTrue(MarkableOp.isPathPainting("f"));
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("profiles")
|
||||
class Profiles {
|
||||
|
||||
@Test
|
||||
@DisplayName("parses the shapes a caller might send")
|
||||
void parsesRequestValues() {
|
||||
assertEquals(PdfUaProfile.UA1, PdfUaProfile.fromRequest("ua1"));
|
||||
assertEquals(PdfUaProfile.UA1, PdfUaProfile.fromRequest(null));
|
||||
assertEquals(PdfUaProfile.UA1, PdfUaProfile.fromRequest(""));
|
||||
assertEquals(PdfUaProfile.UA1, PdfUaProfile.fromRequest("nonsense"));
|
||||
assertEquals(PdfUaProfile.UA2, PdfUaProfile.fromRequest("ua2"));
|
||||
assertEquals(PdfUaProfile.UA2, PdfUaProfile.fromRequest("PDF/UA-2"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("UA-2 requires PDF 2.0")
|
||||
void ua2NeedsPdf2() {
|
||||
assertEquals(2.0f, PdfUaProfile.UA2.pdfVersion());
|
||||
assertEquals(1.7f, PdfUaProfile.UA1.pdfVersion());
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("bounding boxes")
|
||||
class Boxes {
|
||||
|
||||
@Test
|
||||
@DisplayName("union of an empty box is the other box")
|
||||
void unionWithEmpty() {
|
||||
BBox box = new BBox(10, 10, 20, 20);
|
||||
assertEquals(box, box.union(BBox.EMPTY));
|
||||
assertEquals(box, BBox.EMPTY.union(box));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("union covers both boxes")
|
||||
void unionCoversBoth() {
|
||||
BBox union = new BBox(0, 0, 10, 10).union(new BBox(20, 5, 30, 25));
|
||||
assertEquals(new BBox(0, 0, 30, 25), union);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("reports horizontal overlap as a fraction of the narrower box")
|
||||
void overlapIsRelative() {
|
||||
BBox wide = new BBox(0, 0, 100, 10);
|
||||
BBox narrow = new BBox(40, 0, 60, 10);
|
||||
assertEquals(1.0f, wide.horizontalOverlap(narrow));
|
||||
assertEquals(0f, wide.horizontalOverlap(new BBox(200, 0, 220, 10)));
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@DisplayName("structure blocks")
|
||||
class Blocks {
|
||||
|
||||
@Test
|
||||
@DisplayName("counts content across the whole subtree")
|
||||
void countsDescendantContent() {
|
||||
StructBlock table = new StructBlock(StructType.TABLE, 0);
|
||||
StructBlock row = new StructBlock(StructType.TR, 0);
|
||||
StructBlock cell = new StructBlock(StructType.TD, 0);
|
||||
cell.addRange(3, 5);
|
||||
row.addChild(cell);
|
||||
table.addChild(row);
|
||||
assertEquals(3, table.contentCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("collects text in tree order")
|
||||
void collectsText() {
|
||||
StructBlock list = new StructBlock(StructType.L, 0);
|
||||
StructBlock first = new StructBlock(StructType.LI, 0);
|
||||
first.setText("one");
|
||||
StructBlock second = new StructBlock(StructType.LI, 0);
|
||||
second.setText("two");
|
||||
list.addChild(first).addChild(second);
|
||||
assertEquals("one two", list.collectText());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an artifact is not a structure element")
|
||||
void artifactsAreDistinct() {
|
||||
StructBlock artifact = StructBlock.artifact(ArtifactType.PAGINATION, 0);
|
||||
assertTrue(artifact.isArtifact());
|
||||
assertEquals("Pagination", artifact.getArtifactType().subtype());
|
||||
}
|
||||
}
|
||||
}
|
||||
+155
@@ -0,0 +1,155 @@
|
||||
package stirling.software.proprietary.pdf.ua;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
/** Tests the heuristics that decide what counts as a drawing and what counts as a heading. */
|
||||
class VectorAndHeadingTest {
|
||||
|
||||
private static final BBox A4 = new BBox(0, 0, 595, 842);
|
||||
|
||||
private static TextLineInfo line(String text, float size, float x, float y) {
|
||||
List<WordInfo> words = new ArrayList<>();
|
||||
float cursor = x;
|
||||
for (String token : text.strip().split("\\s+")) {
|
||||
float width = token.length() * size * 0.5f;
|
||||
words.add(
|
||||
new WordInfo(
|
||||
token,
|
||||
new BBox(cursor, y, cursor + width, y + size),
|
||||
0,
|
||||
0,
|
||||
size,
|
||||
false));
|
||||
cursor += width + size * 0.3f;
|
||||
}
|
||||
return new TextLineInfo(
|
||||
0, text, new BBox(x, y, cursor, y + size), size, false, 0, 0, false, words);
|
||||
}
|
||||
|
||||
private static MarkableOp vector(int ordinal, BBox box) {
|
||||
return new MarkableOp(ordinal, MarkableOp.Kind.VECTOR, box, null);
|
||||
}
|
||||
|
||||
private static DocumentStructure analyse(List<TextLineInfo> lines, List<MarkableOp> ops) {
|
||||
PageContent page = new PageContent(0, lines, ops, ops.size(), false, false, false, A4);
|
||||
return new LayoutAnalyzer().analyse(List.of(page));
|
||||
}
|
||||
|
||||
private static long countOf(DocumentStructure structure, StructType type) {
|
||||
long[] total = {0};
|
||||
structure.visit(
|
||||
block -> {
|
||||
if (block.getType() == type) {
|
||||
total[0]++;
|
||||
}
|
||||
});
|
||||
return total[0];
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a cluster of substantial strokes becomes a figure, not silent decoration")
|
||||
void chartBecomesAFigure() {
|
||||
List<MarkableOp> bars = new ArrayList<>();
|
||||
for (int i = 0; i < 6; i++) {
|
||||
bars.add(vector(i, new BBox(100 + i * 20, 400, 115 + i * 20, 400 + 30 + i * 10)));
|
||||
}
|
||||
DocumentStructure structure = analyse(List.of(), bars);
|
||||
|
||||
assertTrue(
|
||||
countOf(structure, StructType.FIGURE) > 0,
|
||||
"a bar chart drawn with path operators must not vanish as decoration");
|
||||
assertTrue(
|
||||
structure.figuresWithoutAlt().size() > 0,
|
||||
"the report must say the chart needs a description");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("thin rules and table borders stay artifacts")
|
||||
void tableRulesStayDecoration() {
|
||||
List<MarkableOp> rules = new ArrayList<>();
|
||||
for (int i = 0; i < 8; i++) {
|
||||
rules.add(vector(i, new BBox(60, 700 - i * 20, 540, 701 - i * 20)));
|
||||
}
|
||||
DocumentStructure structure = analyse(List.of(), rules);
|
||||
|
||||
assertEquals(
|
||||
0,
|
||||
countOf(structure, StructType.FIGURE),
|
||||
"horizontal rules are page furniture and must not demand alt text");
|
||||
assertEquals(0, structure.figuresWithoutAlt().size());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a lone box is ornament, not a chart")
|
||||
void singleBoxIsNotAFigure() {
|
||||
DocumentStructure structure =
|
||||
analyse(List.of(), List.of(vector(0, new BBox(60, 400, 500, 700))));
|
||||
assertEquals(0, countOf(structure, StructType.FIGURE));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("shaded table rows behind text are not mistaken for a chart")
|
||||
void shadedTableRowsAreNotFigures() {
|
||||
List<MarkableOp> shading = new ArrayList<>();
|
||||
List<TextLineInfo> rows = new ArrayList<>();
|
||||
for (int i = 0; i < 6; i++) {
|
||||
float y = 600 - i * 20;
|
||||
// A filled row background, tall enough to pass the thinness test.
|
||||
shading.add(vector(i, new BBox(60, y, 540, y + 16)));
|
||||
rows.add(line("Expense line item " + i + " amount", 10, 64, y + 3));
|
||||
}
|
||||
DocumentStructure structure = analyse(rows, shading);
|
||||
|
||||
assertEquals(
|
||||
0,
|
||||
countOf(structure, StructType.FIGURE),
|
||||
"row shading sits behind the text it decorates and is not a drawing");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("small print dominating an invoice does not promote addresses to headings")
|
||||
void smallPrintDoesNotCreateHeadings() {
|
||||
List<TextLineInfo> lines = new ArrayList<>();
|
||||
// Address block at ordinary 11pt.
|
||||
lines.add(line("Acme Industries Limited", 11, 60, 780));
|
||||
lines.add(line("14 Example Street", 11, 60, 765));
|
||||
lines.add(line("Manchester M1 2AB", 11, 60, 750));
|
||||
// 40 lines of 9pt line-item small print, which dominates the character count.
|
||||
for (int i = 0; i < 40; i++) {
|
||||
lines.add(line("Item " + i + " widget assembly part number " + i, 9, 60, 700 - i * 12));
|
||||
}
|
||||
|
||||
Map<Float, Integer> tiers =
|
||||
LayoutAnalyzer.headingTiers(
|
||||
List.of(new PageContent(0, lines, List.of(), 0, false, false, false, A4)),
|
||||
9f);
|
||||
assertNull(
|
||||
tiers.get(11f),
|
||||
"11pt address lines are body text on an invoice, not headings: " + tiers);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a genuinely rare large size is still a heading")
|
||||
void realHeadingsSurvive() {
|
||||
List<TextLineInfo> lines = new ArrayList<>();
|
||||
lines.add(line("Annual Report", 24, 60, 780));
|
||||
for (int i = 0; i < 40; i++) {
|
||||
lines.add(
|
||||
line("Body prose line number " + i + " continues here", 11, 60, 700 - i * 12));
|
||||
}
|
||||
Map<Float, Integer> tiers =
|
||||
LayoutAnalyzer.headingTiers(
|
||||
List.of(new PageContent(0, lines, List.of(), 0, false, false, false, A4)),
|
||||
11f);
|
||||
assertEquals(1, tiers.get(24f), "a rare large size is exactly what a heading looks like");
|
||||
}
|
||||
}
|
||||
+12
@@ -39,6 +39,18 @@ class AdminPolicyManagementAuthorityTest {
|
||||
assertFalse(authority().canEditPolicies());
|
||||
}
|
||||
|
||||
@Test
|
||||
void adminMayTriggerPolicies() {
|
||||
when(userService.isCurrentUserAdmin()).thenReturn(true);
|
||||
assertTrue(authority().canTriggerPolicies());
|
||||
}
|
||||
|
||||
@Test
|
||||
void nonAdminMayNotTriggerPolicies() {
|
||||
when(userService.isCurrentUserAdmin()).thenReturn(false);
|
||||
assertFalse(authority().canTriggerPolicies());
|
||||
}
|
||||
|
||||
@Test
|
||||
void currentUserTeamIdResolvesFromTheCurrentUsersTeam() {
|
||||
Team team = new Team();
|
||||
|
||||
+116
-6
@@ -2,6 +2,7 @@ package stirling.software.proprietary.policy.controller;
|
||||
|
||||
import static org.assertj.core.api.Assertions.assertThat;
|
||||
import static org.assertj.core.api.Assertions.assertThatThrownBy;
|
||||
import static org.junit.jupiter.api.Assertions.assertDoesNotThrow;
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
@@ -82,6 +83,8 @@ class PolicyControllerTest {
|
||||
|
||||
@Mock private stirling.software.proprietary.policy.asset.PolicyAssetCleaner assetCleaner;
|
||||
|
||||
@Mock private stirling.software.proprietary.policy.asset.PolicyAssetResolver assetResolver;
|
||||
|
||||
@Mock private ProcessedLedger processedLedger;
|
||||
|
||||
@Mock private TempFileManager tempFileManager;
|
||||
@@ -114,6 +117,7 @@ class PolicyControllerTest {
|
||||
policyTriggerManager,
|
||||
policyOverviewService,
|
||||
assetCleaner,
|
||||
assetResolver,
|
||||
processedLedger,
|
||||
policyTriggers,
|
||||
applicationProperties,
|
||||
@@ -231,7 +235,7 @@ class PolicyControllerTest {
|
||||
.thenReturn(handle("run-1"));
|
||||
|
||||
ResponseEntity<JobResponse<Void>> response =
|
||||
controller.run(definitionWithStep(), new PolicyRunFiles());
|
||||
controller.run(definitionWithStep(), null, new PolicyRunFiles());
|
||||
|
||||
assertThat(response.getStatusCode()).isEqualTo(HttpStatus.ACCEPTED);
|
||||
assertThat(response.getBody().getJobId()).isEqualTo("run-1");
|
||||
@@ -244,7 +248,7 @@ class PolicyControllerTest {
|
||||
.thenReturn(handle("run-1"));
|
||||
when(sourceAccessGuard.currentTeamId()).thenReturn(3L);
|
||||
|
||||
controller.run(definitionWithStep(), new PolicyRunFiles());
|
||||
controller.run(definitionWithStep(), null, new PolicyRunFiles());
|
||||
|
||||
verify(docCounter).record(EditorSource.counterKey(3L), 0L);
|
||||
}
|
||||
@@ -254,7 +258,7 @@ class PolicyControllerTest {
|
||||
void rejectsEmptyPipeline() {
|
||||
PipelineDefinition empty = new PipelineDefinition("pipe", List.of(), List.of());
|
||||
|
||||
assertThatThrownBy(() -> controller.run(empty, new PolicyRunFiles()))
|
||||
assertThatThrownBy(() -> controller.run(empty, null, new PolicyRunFiles()))
|
||||
.isInstanceOf(ResponseStatusException.class)
|
||||
.satisfies(
|
||||
e ->
|
||||
@@ -276,7 +280,7 @@ class PolicyControllerTest {
|
||||
.when(policyValidator)
|
||||
.validateOutput(any());
|
||||
|
||||
assertThatThrownBy(() -> controller.run(definition, new PolicyRunFiles()))
|
||||
assertThatThrownBy(() -> controller.run(definition, null, new PolicyRunFiles()))
|
||||
.isInstanceOf(ResponseStatusException.class)
|
||||
.satisfies(
|
||||
e ->
|
||||
@@ -284,6 +288,38 @@ class PolicyControllerTest {
|
||||
.isEqualTo(HttpStatus.BAD_REQUEST));
|
||||
verify(policyRunner, never()).runAdHoc(any(), any(), any());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("resolves stored assets from the supplied policy when the caller may edit it")
|
||||
void resolvesStoredAssetsForEditor() throws Exception {
|
||||
applicationProperties.getSecurity().setEnableLogin(false); // editing allowed
|
||||
Policy p = policy("pol-1", 1L);
|
||||
when(policyStore.get("pol-1")).thenReturn(Optional.of(p));
|
||||
when(policyAccessGuard.canAccess(p)).thenReturn(true);
|
||||
when(assetResolver.resolve(eq(p), any())).thenAnswer(inv -> inv.getArgument(1));
|
||||
when(policyRunner.runAdHoc(any(), any(), eq(PolicyProgressListener.NOOP)))
|
||||
.thenReturn(handle("run-1"));
|
||||
|
||||
controller.run(definitionWithStep(), "pol-1", new PolicyRunFiles());
|
||||
|
||||
verify(assetResolver).resolve(eq(p), any());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("does not resolve a policy's stored assets for a caller who cannot edit it")
|
||||
void skipsStoredAssetsForNonEditor() throws Exception {
|
||||
// Gating asset resolution to editors keeps a member from rebinding a policy's stored
|
||||
// asset into an ad-hoc step to read it back.
|
||||
applicationProperties.getSecurity().setEnableLogin(true);
|
||||
when(policyManagementAuthority.canEditPolicies()).thenReturn(false);
|
||||
when(policyRunner.runAdHoc(any(), any(), eq(PolicyProgressListener.NOOP)))
|
||||
.thenReturn(handle("run-1"));
|
||||
|
||||
controller.run(definitionWithStep(), "pol-1", new PolicyRunFiles());
|
||||
|
||||
verify(assetResolver, never()).resolve(any(), any());
|
||||
verify(policyStore, never()).get(any());
|
||||
}
|
||||
}
|
||||
|
||||
@Nested
|
||||
@@ -295,7 +331,8 @@ class PolicyControllerTest {
|
||||
void returnsEmitter() throws Exception {
|
||||
when(policyRunner.runAdHoc(any(), any(), any())).thenReturn(handle("run-2"));
|
||||
|
||||
SseEmitter emitter = controller.runStream(definitionWithStep(), new PolicyRunFiles());
|
||||
SseEmitter emitter =
|
||||
controller.runStream(definitionWithStep(), null, new PolicyRunFiles());
|
||||
|
||||
assertThat(emitter).isNotNull();
|
||||
}
|
||||
@@ -305,7 +342,7 @@ class PolicyControllerTest {
|
||||
void rejectsEmpty() {
|
||||
PipelineDefinition empty = new PipelineDefinition("pipe", List.of(), List.of());
|
||||
|
||||
assertThatThrownBy(() -> controller.runStream(empty, new PolicyRunFiles()))
|
||||
assertThatThrownBy(() -> controller.runStream(empty, null, new PolicyRunFiles()))
|
||||
.isInstanceOf(ResponseStatusException.class);
|
||||
}
|
||||
}
|
||||
@@ -738,5 +775,78 @@ class PolicyControllerTest {
|
||||
assertThat(((ResponseStatusException) e).getStatusCode())
|
||||
.isEqualTo(HttpStatus.NOT_FOUND));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("trigger is forbidden for a team member who cannot manage policies")
|
||||
void triggerForbiddenForMember() {
|
||||
// Sweeping a policy's configured sources is a policy-management capability, so being
|
||||
// in the policy's team is not on its own enough to perform it.
|
||||
applicationProperties.getSecurity().setEnableLogin(true);
|
||||
when(policyManagementAuthority.canTriggerPolicies()).thenReturn(false);
|
||||
|
||||
assertThatThrownBy(() -> controller.trigger("a"))
|
||||
.isInstanceOf(ResponseStatusException.class)
|
||||
.satisfies(
|
||||
e ->
|
||||
assertThat(((ResponseStatusException) e).getStatusCode())
|
||||
.isEqualTo(HttpStatus.FORBIDDEN));
|
||||
// Rejected before the policy is looked up, so no run starts.
|
||||
verify(policyRunner, never()).run(any());
|
||||
verify(policyStore, never()).get(any());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("trigger runs for a caller who may manage policies")
|
||||
void triggerAllowedForLeader() {
|
||||
applicationProperties.getSecurity().setEnableLogin(true);
|
||||
when(policyManagementAuthority.canTriggerPolicies()).thenReturn(true);
|
||||
Policy p = policy("a", 1L);
|
||||
when(policyStore.get("a")).thenReturn(Optional.of(p));
|
||||
when(policyAccessGuard.canAccess(p)).thenReturn(true);
|
||||
SweepOutcome outcome = new SweepOutcome(List.of("run-a"), 1, 0, 0, 0);
|
||||
when(policyRunner.run(p)).thenReturn(outcome);
|
||||
|
||||
ResponseEntity<SweepOutcome> response = controller.trigger("a");
|
||||
|
||||
assertThat(response.getStatusCode()).isEqualTo(HttpStatus.ACCEPTED);
|
||||
assertThat(response.getBody()).isEqualTo(outcome);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("trigger skips the role check when login is disabled")
|
||||
void triggerTrustsTheLocalOperator() {
|
||||
// Single-user deployments have no roles at all; the gate must not lock them out of
|
||||
// their
|
||||
// own sweeps.
|
||||
applicationProperties.getSecurity().setEnableLogin(false);
|
||||
Policy p = policy("a", null);
|
||||
when(policyStore.get("a")).thenReturn(Optional.of(p));
|
||||
when(policyAccessGuard.canAccess(p)).thenReturn(true);
|
||||
SweepOutcome outcome = new SweepOutcome(List.of("run-a"), 1, 0, 0, 0);
|
||||
when(policyRunner.run(p)).thenReturn(outcome);
|
||||
|
||||
assertThat(controller.trigger("a").getStatusCode()).isEqualTo(HttpStatus.ACCEPTED);
|
||||
verify(policyManagementAuthority, never()).canTriggerPolicies();
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("running a policy over the caller's own files stays open to any member")
|
||||
void storedRunIsNotGatedByRole() {
|
||||
// Editor enforcement: every member's upload/export runs the team's stored policies on
|
||||
// their own documents. Gating this the way the sweep is gated would break the editor.
|
||||
applicationProperties.getSecurity().setEnableLogin(true);
|
||||
Policy p = policy("a", 1L);
|
||||
when(policyStore.get("a")).thenReturn(Optional.of(p));
|
||||
when(policyAccessGuard.canAccess(p)).thenReturn(true);
|
||||
when(policyRunner.runWith(eq(p), any(), eq(PolicyProgressListener.NOOP)))
|
||||
.thenReturn(handle("run-9"));
|
||||
|
||||
ResponseEntity<JobResponse<Void>> response =
|
||||
assertDoesNotThrow(() -> controller.runStoredPolicy("a", new PolicyRunFiles()));
|
||||
|
||||
assertThat(response.getStatusCode()).isEqualTo(HttpStatus.ACCEPTED);
|
||||
verify(policyManagementAuthority, never()).canTriggerPolicies();
|
||||
verify(policyManagementAuthority, never()).canEditPolicies();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+115
@@ -0,0 +1,115 @@
|
||||
package stirling.software.proprietary.service.ua;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
import org.junit.jupiter.api.BeforeAll;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import stirling.software.common.service.CustomPDFDocumentFactory;
|
||||
import stirling.software.common.service.PdfMetadataService;
|
||||
import stirling.software.proprietary.controller.api.converters.ConvertPdfToPdfUa;
|
||||
import stirling.software.proprietary.model.api.ua.AccessibilityReport;
|
||||
import stirling.software.proprietary.model.api.ua.FigureDescriptor;
|
||||
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
|
||||
import stirling.software.proprietary.pdf.ua.TaggingOptions;
|
||||
|
||||
/**
|
||||
* The alt-text loop end to end: the report hands out keys the conversion accepts. The converter
|
||||
* never invents descriptions, so a caller must be able to supply them.
|
||||
*/
|
||||
class AltTextRoundTripTest {
|
||||
|
||||
private static PdfUaConversionService conversion;
|
||||
private static AccessibilityAuditService audit;
|
||||
|
||||
@BeforeAll
|
||||
static void setUp() {
|
||||
PdfUaValidationService validation = new PdfUaValidationService();
|
||||
validation.initialise();
|
||||
conversion =
|
||||
new PdfUaConversionService(
|
||||
validation,
|
||||
new FontEmbeddingService(),
|
||||
new CustomPDFDocumentFactory(
|
||||
org.mockito.Mockito.mock(PdfMetadataService.class)));
|
||||
audit = new AccessibilityAuditService(validation);
|
||||
}
|
||||
|
||||
private static TaggingOptions.TaggingOptionsBuilder options() {
|
||||
return TaggingOptions.builder()
|
||||
.profile(PdfUaProfile.UA1)
|
||||
.language("en-GB")
|
||||
.title("Illustrated")
|
||||
.embedFonts(false);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("the report names the figures that need describing, with usable keys")
|
||||
void reportEnumeratesFigures() throws Exception {
|
||||
byte[] input = PdfUaTestDocuments.imageDocument();
|
||||
AccessibilityReport report = audit.audit(input, PdfUaProfile.UA1);
|
||||
|
||||
assertFalse(
|
||||
report.getFiguresNeedingDescription().isEmpty(),
|
||||
"a document with an undescribed image must say which figure needs text");
|
||||
|
||||
FigureDescriptor figure = report.getFiguresNeedingDescription().get(0);
|
||||
assertTrue(figure.key().matches("\\d+:\\d+"), "key should be pageIndex:ordinal: " + figure);
|
||||
assertEquals(1, figure.page(), "pages are reported 1-based for humans");
|
||||
assertTrue(figure.width() > 0 && figure.height() > 0, "figure should carry its box");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("feeding the report's key back makes the document conform")
|
||||
void suppliedDescriptionClosesTheLoop() throws Exception {
|
||||
byte[] input = PdfUaTestDocuments.imageDocument();
|
||||
|
||||
PdfUaConversionOutcome before = conversion.convert(input, options().build());
|
||||
assertFalse(before.declared(), "an undescribed image must block the claim");
|
||||
|
||||
String key =
|
||||
audit.audit(input, PdfUaProfile.UA1).getFiguresNeedingDescription().get(0).key();
|
||||
PdfUaConversionOutcome after =
|
||||
conversion.convert(
|
||||
input, options().altTextByFigure(Map.of(key, "A blue rectangle")).build());
|
||||
|
||||
assertEquals(
|
||||
0,
|
||||
after.tagging().figuresNeedingAltText(),
|
||||
"the description supplied against the report's own key was not applied");
|
||||
assertTrue(after.declared(), "with every figure described the document should conform");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("the request's key=text form parses the way the report emits keys")
|
||||
void parsesTheWireFormat() {
|
||||
Map<String, String> parsed =
|
||||
ConvertPdfToPdfUa.parseAltText(
|
||||
"0:12=Bar chart of quarterly revenue\r\n"
|
||||
+ "1:3=Company logo\n"
|
||||
+ " \n"
|
||||
+ "malformed-line\n"
|
||||
+ "2:7=Diagram showing the approval flow = end to end");
|
||||
|
||||
assertEquals(3, parsed.size(), "blank and malformed lines are skipped: " + parsed);
|
||||
assertEquals("Bar chart of quarterly revenue", parsed.get("0:12"));
|
||||
assertEquals("Company logo", parsed.get("1:3"));
|
||||
assertEquals(
|
||||
"Diagram showing the approval flow = end to end",
|
||||
parsed.get("2:7"),
|
||||
"only the first equals splits, so descriptions may contain one");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("no descriptions supplied means none invented")
|
||||
void emptyInputInventsNothing() {
|
||||
assertTrue(ConvertPdfToPdfUa.parseAltText(null).isEmpty());
|
||||
assertTrue(ConvertPdfToPdfUa.parseAltText(" ").isEmpty());
|
||||
}
|
||||
}
|
||||
+92
@@ -0,0 +1,92 @@
|
||||
package stirling.software.proprietary.service.ua;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
import org.apache.pdfbox.Loader;
|
||||
import org.apache.pdfbox.cos.COSName;
|
||||
import org.apache.pdfbox.pdmodel.PDDocument;
|
||||
import org.junit.jupiter.api.BeforeAll;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import stirling.software.common.service.CustomPDFDocumentFactory;
|
||||
import stirling.software.common.service.PdfMetadataService;
|
||||
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
|
||||
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
|
||||
import stirling.software.proprietary.pdf.ua.TaggingOptions;
|
||||
|
||||
/** PDF/UA-2 is not just a metadata number: it needs PDF 2.0 and namespaced structure types. */
|
||||
class PdfUa2ProfileTest {
|
||||
|
||||
private static PdfUaConversionService conversion;
|
||||
|
||||
@BeforeAll
|
||||
static void setUp() {
|
||||
PdfUaValidationService validation = new PdfUaValidationService();
|
||||
validation.initialise();
|
||||
conversion =
|
||||
new PdfUaConversionService(
|
||||
validation,
|
||||
new FontEmbeddingService(),
|
||||
new CustomPDFDocumentFactory(
|
||||
org.mockito.Mockito.mock(PdfMetadataService.class)));
|
||||
}
|
||||
|
||||
private static PdfUaConversionOutcome convertUa2(byte[] input) throws Exception {
|
||||
return conversion.convert(
|
||||
input,
|
||||
TaggingOptions.builder()
|
||||
.profile(PdfUaProfile.UA2)
|
||||
.language("en-GB")
|
||||
.title("UA-2 Document")
|
||||
.embedFonts(false)
|
||||
.build());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("raises the file to PDF 2.0 and namespaces the structure tree")
|
||||
void producesPdf2WithNamespaces() throws Exception {
|
||||
PdfUaConversionOutcome outcome = convertUa2(PdfUaTestDocuments.headingHierarchy());
|
||||
|
||||
try (PDDocument document = Loader.loadPDF(outcome.pdfBytes())) {
|
||||
assertEquals(2.0f, document.getVersion(), "UA-2 is defined on PDF 2.0");
|
||||
|
||||
var root = document.getDocumentCatalog().getStructureTreeRoot();
|
||||
assertNotNull(root, "no structure tree was written");
|
||||
assertNotNull(
|
||||
root.getCOSObject().getDictionaryObject(COSName.getPDFName("Namespaces")),
|
||||
"UA-2 requires the standard structure namespace to be declared");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("validates against the PDF/UA-2 profile, not the UA-1 one")
|
||||
void validatesAgainstUa2() throws Exception {
|
||||
PdfUaConversionOutcome outcome = convertUa2(PdfUaTestDocuments.simpleDocument());
|
||||
assertEquals("PDF/UA-2", outcome.validation().profile());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("reaches UA-2 conformance and declares it")
|
||||
void reachesUa2Conformance() throws Exception {
|
||||
PdfUaConversionOutcome outcome = convertUa2(PdfUaTestDocuments.simpleDocument());
|
||||
|
||||
String failures =
|
||||
outcome.validation().issues().stream()
|
||||
.map(issue -> issue.getClause() + ": " + issue.getTechnicalMessage())
|
||||
.collect(java.util.stream.Collectors.joining("; "));
|
||||
assertEquals(0, outcome.validation().totalFailures(), "UA-2 checks failed: " + failures);
|
||||
assertTrue(outcome.declared(), "a conforming UA-2 file must carry the declaration");
|
||||
assertTrue(outcome.pdfBytes().length > 0);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an illustrated document still cannot claim UA-2 without descriptions")
|
||||
void undescribedImageBlocksTheUa2Claim() throws Exception {
|
||||
PdfUaConversionOutcome outcome = convertUa2(PdfUaTestDocuments.imageDocument());
|
||||
assertFalse(outcome.declared(), "an undescribed image must block the claim");
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user