Merge branch 'main' into custom_task_20260812

This commit is contained in:
Ludy
2026-08-20 21:39:34 +02:00
committed by GitHub
166 changed files with 14197 additions and 1112 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
# Maintainer: Stirling PDF Inc <contact@stirlingpdf.com>
pkgname=stirling-pdf-desktop
pkgver=2.14.2
pkgver=2.14.3
pkgrel=1
pkgdesc="Locally hosted, web-based PDF manipulation tool (Tauri desktop app, official Stirling PDF Inc build)"
arch=('x86_64')
+1 -1
View File
@@ -1,6 +1,6 @@
# Maintainer: Stirling PDF Inc <contact@stirlingpdf.com>
pkgname=stirling-pdf-server-bin
pkgver=2.14.2
pkgver=2.14.3
pkgrel=1
pkgdesc="Locally hosted, web-based PDF manipulation tool (server JAR, prebuilt)"
arch=('any')
+17
View File
@@ -8,6 +8,7 @@ ci: &ci
build: &build
- *ci
- buildSrc/**
- build.gradle
- gradle/spotless.gradle
- app/(common|core|proprietary|saas)/build.gradle
@@ -15,6 +16,22 @@ build: &build
- .taskfiles/backend.yml
- .github/workflows/check-licence.yml
# Backend build inputs. This is intentionally broader than `build`: Java and
# backend resource changes must exercise the backend matrix even when Gradle
# build scripts themselves are unchanged.
backend: &backend
- *ci
- *build
- gradle/**
- gradle.properties
- gradlew
- gradlew.bat
- settings.gradle
- app/(common|core|proprietary|saas)/src/(main|test)/java/**
- "app/(common|core|proprietary|saas)/src/(main|test)/resources/**/!(messages_*.properties|*.md)*"
- scripts/db-migration/**
- .github/workflows/backend-build.yml
openapi: &openapi
- *ci
- *build
+8 -8
View File
@@ -66,6 +66,7 @@ updates:
directories:
- /devTools
- /frontend
- /testing/compose/mcp-client-check
schedule:
interval: "weekly"
cooldown:
@@ -94,6 +95,13 @@ updates:
- "react-dom"
- "@types/react"
- "@types/react-dom"
tanstack:
patterns:
- "@tanstack/*"
typescript:
patterns:
- "typescript"
- "@typescript/*"
vite:
patterns:
- "vite"
@@ -172,14 +180,6 @@ updates:
- "tokio"
- "tokio-*"
- package-ecosystem: pip
directory: /testing/cucumber
schedule:
interval: "weekly"
cooldown:
default-days: 7
rebase-strategy: "auto"
- package-ecosystem: "uv"
directory: "/engine"
schedule:
+5 -2
View File
@@ -462,7 +462,10 @@ jobs:
});
cleanup-v2-deployment:
environment: pr-preview
# Tearing a preview down is not a deployment - no deployment object.
environment:
name: pr-preview
deployment: false
if: github.event.action == 'closed'
runs-on: ubuntu-latest
permissions:
@@ -557,5 +560,5 @@ jobs:
- name: Cleanup temporary files
if: always()
run: |
rm -f ../private.key
rm -f ../private.key docker-compose.yml storybook.tgz
continue-on-error: true
+4 -1
View File
@@ -9,7 +9,10 @@ permissions:
jobs:
cleanup:
environment: pr-preview
# Tearing a preview down is not a deployment - no deployment object.
environment:
name: pr-preview
deployment: false
if: github.event.action == 'closed'
runs-on: ubuntu-latest
permissions:
+3 -1
View File
@@ -20,7 +20,9 @@ permissions:
jobs:
build:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
runs-on: ubuntu-latest
strategy:
fail-fast: false
+12 -2
View File
@@ -42,7 +42,9 @@ jobs:
uses: ./.github/workflows/_runner-pick.yml
playwright-e2e-enterprise:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
needs: pick
# Skip on fork PRs / untrusted authors: they have no PREMIUM_KEY_ENTERPRISE,
# so the suite can't boot premium and would fail. See the header comment.
@@ -322,10 +324,18 @@ jobs:
path: frontend/playwright-report/
retention-days: 7
- name: Cleanup temporary files
if: always()
run: |
rm -f /tmp/helpers.sh /tmp/backend.log /tmp/backend.pid
continue-on-error: true
# Multi-node regression: builds + seeds the clustered stack (testing/compose/docker-compose-multinode.yml)
# and runs behave features/multinode. Licence-gated, so it runs after the Playwright job (not in parallel).
multinode-e2e:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
needs: [pick, playwright-e2e-enterprise]
# Nightly cron + manual dispatch only (heavy build), fork-gated for the licence secret.
if: >-
+17 -3
View File
@@ -37,6 +37,7 @@ jobs:
timeout-minutes: 3
outputs:
build: ${{ steps.changes.outputs.build }}
backend: ${{ steps.changes.outputs.backend }}
project: ${{ steps.changes.outputs.project }}
openapi: ${{ steps.changes.outputs.openapi }}
frontend: ${{ steps.changes.outputs.frontend }}
@@ -61,8 +62,11 @@ jobs:
filters: .github/config/.files.yaml
gradle-cache-prime:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
name: Prime shared Gradle cache
if: needs.files-changed.outputs.project == 'true'
needs: [files-changed]
runs-on: ubuntu-latest
timeout-minutes: 15
@@ -116,6 +120,7 @@ jobs:
key: ${{ steps.gradle-cache-key.outputs.key }}
build:
if: needs.files-changed.outputs.backend == 'true'
needs: [files-changed, gradle-cache-prime]
permissions:
actions: read
@@ -194,7 +199,7 @@ jobs:
check-licence:
if: needs.files-changed.outputs.build == 'true'
needs: [files-changed, build, gradle-cache-prime]
needs: [files-changed, gradle-cache-prime]
permissions:
contents: read
uses: ./.github/workflows/check-licence.yml
@@ -213,7 +218,14 @@ jobs:
docker-base-changed: ${{ needs.files-changed.outputs.docker-base }}
test-build-docker-images:
if: github.event_name == 'pull_request' && needs.files-changed.outputs.project == 'true'
if: |
always() &&
github.event_name == 'pull_request' &&
needs.files-changed.outputs.project == 'true' &&
contains(fromJSON('["success", "skipped"]'), needs.gradle-cache-prime.result) &&
contains(fromJSON('["success", "skipped"]'), needs.build.result) &&
contains(fromJSON('["success", "skipped"]'), needs.check-generateOpenApiDocs.result) &&
contains(fromJSON('["success", "skipped"]'), needs.check-licence.result)
needs:
[
files-changed,
@@ -319,6 +331,7 @@ jobs:
if: always()
needs:
- files-changed
- gradle-cache-prime
- build
- db-migration-test
- check-generateOpenApiDocs
@@ -346,6 +359,7 @@ jobs:
env:
RESULTS: |
files-changed=${{ needs.files-changed.result }}
gradle-cache-prime=${{ needs.gradle-cache-prime.result }}
build=${{ needs.build.result }}
db-migration-test=${{ needs.db-migration-test.result }}
check-generateOpenApiDocs=${{ needs.check-generateOpenApiDocs.result }}
+3 -1
View File
@@ -10,7 +10,9 @@ permissions:
jobs:
check-licence:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
runs-on: ubuntu-latest
steps:
- name: Harden Runner
+3 -1
View File
@@ -11,7 +11,9 @@ permissions:
jobs:
check-generate-openapi-docs:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
runs-on: ubuntu-latest
steps:
- name: Harden Runner
+8 -1
View File
@@ -13,7 +13,9 @@ permissions:
jobs:
migration-test:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
@@ -79,3 +81,8 @@ jobs:
path: /tmp/stirling-migration-failed-*/app.log
retention-days: 7
if-no-files-found: warn
- name: Cleanup temporary files
if: always()
run: rm -rf /tmp/stirling-migration-failed-*
continue-on-error: true
+4 -2
View File
@@ -17,7 +17,9 @@ permissions:
jobs:
docker-compose-tests:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
runs-on: ubuntu-latest
permissions:
actions: write
@@ -64,7 +66,7 @@ jobs:
- name: Install Docker Compose
run: |
sudo curl -SL "https://github.com/docker/compose/releases/download/v2.39.4/docker-compose-$(uname -s)-$(uname -m)" -o /usr/local/bin/docker-compose
sudo curl -SL "https://github.com/docker/compose/releases/download/v5.4.0/docker-compose-$(uname -s)-$(uname -m)" -o /usr/local/bin/docker-compose
sudo chmod +x /usr/local/bin/docker-compose
- name: Install uv
+3 -1
View File
@@ -11,7 +11,9 @@ permissions:
jobs:
playwright-e2e-live:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
+15 -2
View File
@@ -14,6 +14,11 @@ jobs:
playwright-e2e:
name: playwright-e2e (${{ matrix.browser }})
runs-on: ubuntu-latest
# The image already contains the Playwright browsers and all Linux
# dependencies. This keeps the matrix for per-browser reporting while
# avoiding three concurrent `playwright install --with-deps` runs.
container:
image: mcr.microsoft.com/playwright:v1.58.2-noble@sha256:6446946a1d9fd62d9ae501312a2d76a43ee688542b21622056a372959b65d63d
strategy:
# One browser breaking must not mask a failure in another - report all.
fail-fast: false
@@ -40,15 +45,23 @@ jobs:
cache-dependency-path: frontend/package-lock.json
- name: Install Task
uses: ./.github/actions/setup-task
- name: Install Playwright (${{ matrix.browser }})
run: task e2e:install -- ${{ matrix.browser }}
- name: Build frontend (production bundle for vite preview)
env:
VITE_BUILD_FOR_PREVIEW: "1"
run: task frontend:build
- name: Run stubbed E2E tests (${{ matrix.browser }})
env:
# The official Playwright image expects its browser runtime under
# the root home directory. Keep this scoped to Playwright and use a
# neutral Docker config path so Docker does not read /root/.docker.
HOME: /root
DOCKER_CONFIG: /tmp/playwright-docker-config
PLAYWRIGHT_JSON_OUTPUT_FILE: ${{ github.workspace }}/frontend/playwright-report/results.json
NPM_CONFIG_PREFER_OFFLINE: "true"
NPM_CONFIG_FETCH_RETRIES: "5"
NPM_CONFIG_FETCH_RETRY_FACTOR: "2"
NPM_CONFIG_FETCH_RETRY_MINTIMEOUT: "1000"
NPM_CONFIG_FETCH_RETRY_MAXTIMEOUT: "120000"
run: task e2e:stubbed-project PROJECT=${{ matrix.project }} -- --workers=3
- name: Flag flaky tests
# Runs regardless of the test outcome: a flaky test (passed on retry)
@@ -43,7 +43,9 @@ jobs:
generate-frontend-license-report:
# ci-bot, not bot-identity: this job runs on PRs too, and bot-identity is main-only.
environment: ci-bot
environment:
name: ci-bot
deployment: false
if: needs.files-changed.outputs.licenses-frontend == 'true'
name: Generate Frontend License Report
needs: files-changed
@@ -319,7 +321,9 @@ jobs:
generate-backend-license-report:
# ci-bot, not bot-identity: this job runs on PRs too, and bot-identity is main-only.
environment: ci-bot
environment:
name: ci-bot
deployment: false
if: needs.files-changed.outputs.licenses-backend == 'true'
needs: files-changed
name: Generate Backend License Report
+17 -2
View File
@@ -38,7 +38,9 @@ permissions:
jobs:
determine-matrix:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
if: ${{ vars.CI_PROFILE != 'lite' }}
runs-on: ubuntu-latest
outputs:
@@ -116,7 +118,9 @@ jobs:
env:
INPUT_PLATFORM: ${{ github.event.inputs.platform }}
build-jars:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
needs: determine-matrix
runs-on: ubuntu-latest
strategy:
@@ -693,6 +697,17 @@ jobs:
path: ./dist/*
retention-days: 1
- name: Cleanup temporary files
if: always()
shell: bash
run: |
rm -f certificate.p12
rm -rf "$RUNNER_TEMP/msi-verify"
if [ "${{ matrix.platform }}" = "macos-15" ]; then
security delete-keychain "$RUNNER_TEMP/app-signing.keychain-db" 2>/dev/null || true
fi
continue-on-error: true
collect-and-release:
needs: [determine-matrix, build, build-jars]
runs-on: ubuntu-latest
+3 -1
View File
@@ -127,7 +127,9 @@ jobs:
# Runs the @nightly tag (conversion scenarios) plus a 10-shard concurrency run
# of every other feature.
cucumber-nightly:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
name: Cucumber (nightly scenarios + full concurrency)
runs-on: ubuntu-latest
# Fork pull requests get no MAVEN_* secrets, so the image build cannot work.
+14 -1
View File
@@ -63,7 +63,9 @@ jobs:
determine-matrix:
# Only probes APPLE_CERTIFICATE for presence, so it stays on the unrestricted
# signing environment - release-signing would block every PR run.
environment: ci-signing
environment:
name: ci-signing
deployment: false
if: ${{ vars.CI_PROFILE != 'lite' }}
runs-on: ubuntu-latest
outputs:
@@ -675,6 +677,17 @@ jobs:
fi
done
- name: Cleanup temporary files
if: always()
shell: bash
run: |
rm -f certificate.p12
rm -rf "$RUNNER_TEMP/msi-verify"
if [ "${{ matrix.platform }}" = "macos-15" ]; then
security delete-keychain "$RUNNER_TEMP/app-signing.keychain-db" 2>/dev/null || true
fi
continue-on-error: true
pr-comment:
needs: build
runs-on: ubuntu-latest
+3 -1
View File
@@ -37,7 +37,9 @@ jobs:
# spring-security=true matrix entry if `task backend:build` and
# `task backend:build:ci` produce equivalent JARs (verify before wiring).
test-build-docker-images:
environment: ci-unsigned
environment:
name: ci-unsigned
deployment: false
runs-on: ubuntu-latest
strategy:
fail-fast: false
+112
View File
@@ -0,0 +1,112 @@
name: Update Gradle
on:
workflow_dispatch:
schedule:
- cron: "0 3 * * 1"
concurrency:
group: update-gradle
cancel-in-progress: true
jobs:
update-gradle:
name: Update Gradle and Docker images
permissions:
contents: write
pull-requests: write
runs-on: ubuntu-latest
timeout-minutes: 20
steps:
- name: Harden runner
uses: step-security/harden-runner@bf7454d06d71f1098171f2acdf0cd4708d7b5920 # v2.20.0
with:
egress-policy: audit
- name: Check out repository
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Set up Java
uses: actions/setup-java@b6effb05e454b25005698d916606bdc6ffcbf961 # v5.7.0
with:
distribution: temurin
java-version: "25"
- name: Find latest Gradle release
id: gradle
shell: bash
run: |
set -euo pipefail
version=$(curl --fail --silent --show-error --retry 3 \
https://services.gradle.org/versions/current | jq -r '.version')
[[ "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]] || {
echo "Could not determine a stable Gradle version: $version" >&2
exit 1
}
echo "version=$version" >> "$GITHUB_OUTPUT"
- name: Find matching Docker image digest
id: docker
env:
GRADLE_VERSION: ${{ steps.gradle.outputs.version }}
shell: bash
run: |
set -euo pipefail
tag="${GRADLE_VERSION}-jdk25"
digest=$(curl --fail --silent --show-error --retry 3 \
"https://hub.docker.com/v2/repositories/library/gradle/tags/${tag}" \
| jq -r '.digest // empty')
[[ "$digest" =~ ^sha256:[0-9a-f]{64}$ ]] || {
echo "Docker image gradle:${tag} was not found" >&2
exit 1
}
echo "tag=$tag" >> "$GITHUB_OUTPUT"
echo "digest=$digest" >> "$GITHUB_OUTPUT"
- name: Update Gradle wrapper
env:
GRADLE_VERSION: ${{ steps.gradle.outputs.version }}
run: ./gradlew wrapper --gradle-version "$GRADLE_VERSION" --distribution-type bin
- name: Update Gradle Docker images
env:
DOCKER_TAG: ${{ steps.docker.outputs.tag }}
DOCKER_DIGEST: ${{ steps.docker.outputs.digest }}
shell: bash
run: |
set -euo pipefail
find docker -type f -name 'Dockerfile*' -print0 |
xargs -0 sed -E -i \
"s#gradle:[^@[:space:]]+-jdk25(@sha256:[^[:space:]]+)?#gradle:${DOCKER_TAG}@${DOCKER_DIGEST}#g"
- name: Verify Gradle update
env:
EXPECTED_VERSION: ${{ steps.gradle.outputs.version }}
shell: bash
run: |
set -euo pipefail
actual=$(./gradlew --version | sed -n 's/^Gradle \([0-9.]*\)$/\1/p')
[[ "$actual" == "$EXPECTED_VERSION" ]] || {
echo "Wrapper resolved Gradle $actual, expected $EXPECTED_VERSION" >&2
exit 1
}
if git diff --quiet; then
echo "Gradle is already up to date."
exit 0
fi
git diff --check
- name: Create pull request
uses: peter-evans/create-pull-request@5f6978faf089d4d20b00c7766989d076bb2fc7f1 # v8.1.1
with:
token: ${{ secrets.GITHUB_TOKEN }}
branch: automation/update-gradle
delete-branch: true
commit-message: "chore: update Gradle"
title: "chore: update Gradle to ${{ steps.gradle.outputs.version }}"
body: |
Automated update of the Gradle wrapper and Gradle Docker build images.
Gradle version: `${{ steps.gradle.outputs.version }}`
Docker image: `gradle:${{ steps.docker.outputs.tag }}`
labels: dependencies
+5
View File
@@ -27,3 +27,8 @@ app/core/src/main/java/stirling/software/SPDF/pdf/signature/CreateSignatureBase.
# Supabase publishable key (public by design, RLS-protected) used as a CI fallback
# default in the tauri-build workflow when the GitHub secret is unset - not a real secret.
.github/workflows/tauri-build.yml:generic-api-key:402
# Staging Supabase publishable key (public by design). Ignored here rather than with an
# inline gitleaks:allow because a trailing comment in a .properties file is part of the
# value, so the pragma would end up inside the key.
app/saas/src/main/resources/application-staging.properties:generic-api-key:16
+49 -6
View File
@@ -57,16 +57,57 @@ tasks:
- cmd: ./gradlew clean bootRun -PbuildWithFrontend=true
platforms: [linux, darwin]
# SaaS backend. dev:saas -> the PR's preview branch, staging:saas -> shared v3,
# PROFILES=none -> production against your own SAAS_DB_*. Production has no named
# task on purpose. Use `none`, not an empty value: Go template `default` treats ""
# as absent and would resolve back to dev.
dev:saas:
desc: "Start backend in SaaS flavor against Supabase"
# `dotenv:` reads from the root Taskfile's directory (".") because this
# subtaskfile is included with `dir: .`.
desc: "Start SaaS backend against the current PR's Supabase preview branch"
dotenv: ['app/.env.saas.local', 'app/.env.saas']
vars:
PROFILES: '{{.PROFILES | default "dev"}}'
cmds:
# Don't move this check into a `sh:` var: dotenv is visible in cmds but not
# during var evaluation, so the test would always see an empty value.
- cmd: |
if [ "{{.PROFILES}}" = "dev" ] && [ -z "${SAAS_DEV_PROJECT_REF:-}" ]; then
echo ">> SAAS_DEV_PROJECT_REF is not set."
echo ">> Testing a SaaS PR? Put its ref, DB password and publishable key in app/.env.saas.local."
echo ">> Wanted the shared v3 project? Use 'task backend:staging:saas' instead."
exit 1
fi
- task: _run:saas
vars:
PORT: '{{.PORT}}'
PROFILES: '{{.PROFILES}}'
AIENGINE_URL: '{{.AIENGINE_URL}}'
AIENGINE_ENABLED: '{{.AIENGINE_ENABLED}}'
AIENGINE_TIMEOUTSECONDS: '{{.AIENGINE_TIMEOUTSECONDS}}'
staging:saas:
desc: "Start SaaS backend against the shared v3 staging project"
cmds:
- task: _run:saas
vars:
PORT: '{{.PORT}}'
PROFILES: staging
AIENGINE_URL: '{{.AIENGINE_URL}}'
AIENGINE_ENABLED: '{{.AIENGINE_ENABLED}}'
AIENGINE_TIMEOUTSECONDS: '{{.AIENGINE_TIMEOUTSECONDS}}'
_run:saas:
internal: true
dotenv: ['app/.env.saas.local', 'app/.env.saas']
ignore_error: true
vars:
PORT: '{{.PORT | default "8080"}}'
# Override to "" to run the pure `saas` profile against your own SAAS_DB_*.
PROFILES: '{{.PROFILES | default "dev"}}'
# Built here rather than inline in the cmds below: the Windows line is an
# unquoted YAML scalar wrapping a cmd.exe string, so a nested {{if ne .X
# "none"}} needs escaped quotes that reach the Go template as literal
# backslashes and fail with `unexpected "\" in operand`.
PROFILE_ARGS: '{{if ne .PROFILES "none"}}--spring.profiles.include={{.PROFILES}}{{end}}'
AIENGINE_URL: '{{.AIENGINE_URL | default ""}}'
AIENGINE_ENABLED: '{{.AIENGINE_ENABLED | default "false"}}'
AIENGINE_TIMEOUTSECONDS: '{{.AIENGINE_TIMEOUTSECONDS | default "120"}}'
@@ -77,9 +118,11 @@ tasks:
AIENGINE_ENABLED: '{{.AIENGINE_ENABLED}}'
AIENGINE_TIMEOUTSECONDS: '{{.AIENGINE_TIMEOUTSECONDS}}'
cmds:
- cmd: cmd /c ".\gradlew.bat :stirling-pdf:bootRun {{if .PROFILES}}--args=\"--spring.profiles.include={{.PROFILES}}\"{{end}}"
# PROFILE_ARGS is empty when PROFILES=none, i.e. the bare `saas` profile
# against SAAS_DB_* (production).
- cmd: cmd /c ".\gradlew.bat :stirling-pdf:bootRun {{if .PROFILE_ARGS}}--args=\"{{.PROFILE_ARGS}}\"{{end}}"
platforms: [windows]
- cmd: ./gradlew :stirling-pdf:bootRun {{if .PROFILES}}--args='--spring.profiles.include={{.PROFILES}}'{{end}}
- cmd: ./gradlew :stirling-pdf:bootRun {{if .PROFILE_ARGS}}--args='{{.PROFILE_ARGS}}'{{end}}
platforms: [linux, darwin]
build:
+64 -10
View File
@@ -5,6 +5,14 @@ version: '3'
# mode flag) or use `--project editor/...` for tsc — so the editor lives
# under frontend/editor/ without each task needing a cd.
vars:
# Dev-only browser-tab label so concurrent worktrees are distinguishable. Only
# the worktree folder basename (e.g. "wt1") is exposed — never the full path,
# hostname, or user. Dropped from production builds.
DEV_LABEL:
sh: >-
{{if eq OS "windows"}}powershell -NoProfile -Command '$root = git rev-parse --show-toplevel 2>$null; if (-not $root) { $root = (Get-Location).Path }; Split-Path -Leaf $root'{{else}}basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)"{{end}}
tasks:
install:
desc: "Install dependencies"
@@ -80,16 +88,52 @@ tasks:
OPEN: '{{.OPEN | default ""}}'
env:
BACKEND_URL: '{{.BACKEND_URL}}'
# Dev-only browser-tab label so concurrent worktrees are distinguishable.
# Only the worktree folder basename (e.g. "wt1") is exposed — never the
# full path, hostname, or user. Consumed at dev-serve time by vite.config
# and dropped from production builds.
STIRLING_DEV_LABEL:
sh: >-
{{if eq OS "windows"}}powershell -NoProfile -Command '$root = git rev-parse --show-toplevel 2>$null; if (-not $root) { $root = (Get-Location).Path }; Split-Path -Leaf $root'{{else}}basename "$(git rev-parse --show-toplevel 2>/dev/null || pwd)"{{end}}
STIRLING_DEV_LABEL: '{{.DEV_LABEL}}'
cmds:
- npx vite editor --mode {{.MODE}} --port {{.PORT}}{{if .OPEN}} --open{{end}}
# Separate from dev:_run rather than a flag on it: Task sets an `env:` key even
# when its value resolves to empty, and Vite treats an empty process.env VITE_* as
# authoritative over the committed editor/.env, so folding these in blanks Supabase
# config for the core, proprietary and desktop dev servers.
dev:_run:saas:
internal: true
ignore_error: true
# The backend's own env files, so both halves target one project. Paths are
# relative to this taskfile's dir, `frontend`.
dotenv: ['../app/.env.saas.local', '../app/.env.saas']
vars:
PORT: '{{.PORT | default "5173"}}'
BACKEND_URL: '{{.BACKEND_URL | default "http://localhost:8080"}}'
OPEN: '{{.OPEN | default ""}}'
SAAS_ENV: '{{.SAAS_ENV | default "dev"}}'
env:
BACKEND_URL: '{{.BACKEND_URL}}'
STIRLING_DEV_LABEL: '{{.DEV_LABEL}}'
SAAS_ENV: '{{.SAAS_ENV}}'
# A real process.env VITE_* beats a committed .env in Vite (loadEnv applies
# process.env last), which is what lets this override editor/.env.
#
# These must stay `sh:`, not Go templates: dotenv values are visible to Task's
# embedded shell but not to templates, where {{.SAAS_DEV_PROJECT_REF}} is
# always empty.
VITE_SUPABASE_URL:
sh: |
case "${SAAS_ENV:-dev}" in
staging) ref="${SAAS_STAGING_PROJECT_REF:?set it in app/.env.saas.local}" ;;
*) ref="${SAAS_DEV_PROJECT_REF:?set it in app/.env.saas.local, or run task staging:saas}" ;;
esac
echo "https://${ref}.supabase.co"
VITE_SUPABASE_PUBLISHABLE_DEFAULT_KEY:
sh: |
case "${SAAS_ENV:-dev}" in
staging) echo "${SAAS_STAGING_PUBLISHABLE_KEY:?set it in app/.env.saas.local}" ;;
*) echo "${SAAS_DEV_PUBLISHABLE_KEY:?set it in app/.env.saas.local}" ;;
esac
cmds:
- 'echo ">> frontend Supabase target: $VITE_SUPABASE_URL"'
- npx vite editor --mode saas --port {{.PORT}}{{if .OPEN}} --open{{end}}
dev:
desc: "Start frontend dev server"
cmds:
@@ -111,13 +155,23 @@ tasks:
vars: { MODE: proprietary, PORT: '{{.PORT}}', BACKEND_URL: '{{.BACKEND_URL}}', OPEN: '{{.OPEN}}' }
dev:saas:
desc: "Start frontend dev server in SaaS mode"
desc: "Start frontend dev server in SaaS mode (SAAS_ENV=dev|staging|prod)"
deps:
- task: prepare
vars: { MODE: saas }
vars:
SAAS_ENV: '{{.SAAS_ENV | default "dev"}}'
# prod routes to the plain runner, which sets no VITE_SUPABASE_* and so leaves
# the committed editor/.env alone.
RUNNER: '{{if eq .SAAS_ENV "prod"}}dev:_run{{else}}dev:_run:saas{{end}}'
cmds:
- task: dev:_run
vars: { MODE: saas, PORT: '{{.PORT}}', BACKEND_URL: '{{.BACKEND_URL}}', OPEN: '{{.OPEN}}' }
- task: '{{.RUNNER}}'
vars:
MODE: saas
PORT: '{{.PORT}}'
BACKEND_URL: '{{.BACKEND_URL}}'
OPEN: '{{.OPEN}}'
SAAS_ENV: '{{.SAAS_ENV}}'
dev:desktop:
desc: "Start frontend dev server in desktop mode"
+2 -2
View File
@@ -46,8 +46,8 @@ This guide focuses on developing for Stirling 2.0, including both the React fron
- Docker
- Git
- Java JDK 25
- Node.js 18+ and npm (required for frontend development)
- Gradle 7.0 or later (Included within the repo)
- Node.js 22+ and npm (required for frontend development)
- Gradle 9.0 or later (Included within the repo)
- [uv](https://docs.astral.sh/uv/) — Python package manager (required for engine development)
- Rust and Cargo (required for Tauri desktop app development)
- Tauri CLI (install with `cargo install tauri-cli`)
+18 -3
View File
@@ -99,11 +99,22 @@ tasks:
BACKEND_URL: 'http://localhost:{{.BACKEND_PORT}}'
OPEN: "true"
# Set SAAS_DEV_PROJECT_REF in app/.env.saas.local to pick the PR.
dev:saas:
desc: "Start SaaS backend + frontend concurrently on free ports"
desc: "Start SaaS backend + frontend + engine against the current PR's preview branch"
cmds:
- task: dev:_all
vars: { FRONTEND: saas, BACKEND: saas }
vars: { FRONTEND: saas, BACKEND: saas, SAAS_ENV: dev }
staging:saas:
desc: "Start SaaS backend + frontend + engine against the shared v3 staging project"
cmds:
- task: dev:_all
vars:
FRONTEND: saas
BACKEND: saas
BACKEND_TASK: backend:staging:saas
SAAS_ENV: staging
dev:all:
desc: "Start backend + frontend + engine concurrently on free ports"
@@ -115,6 +126,9 @@ tasks:
vars:
FRONTEND: '{{.FRONTEND | default "proprietary"}}'
BACKEND: '{{.BACKEND | default "proprietary"}}'
BACKEND_TASK: '{{.BACKEND_TASK | default (printf "backend:dev:%s" .BACKEND)}}'
# Only meaningful to the saas frontend; every other flavor ignores it.
SAAS_ENV: '{{.SAAS_ENV | default ""}}'
PORTS:
sh: '{{if eq OS "windows"}}{{.FIND_FREE_PORT_PS}} 8080 5173 5001{{else}}{{.FIND_FREE_PORT_SH}} 8080 5173 5001{{end}}'
BACKEND_PORT: '{{index (splitList "\n" .PORTS) 0}}'
@@ -124,7 +138,7 @@ tasks:
- task: engine:dev
vars:
PORT: '{{.ENGINE_PORT}}'
- task: 'backend:dev:{{.BACKEND}}'
- task: '{{.BACKEND_TASK}}'
vars:
PORT: '{{.BACKEND_PORT}}'
AIENGINE_URL: 'http://localhost:{{.ENGINE_PORT}}'
@@ -134,6 +148,7 @@ tasks:
PORT: '{{.FRONTEND_PORT}}'
BACKEND_URL: 'http://localhost:{{.BACKEND_PORT}}'
OPEN: "true"
SAAS_ENV: '{{.SAAS_ENV}}'
# ============================================================
# Build
+35 -17
View File
@@ -1,15 +1,16 @@
###############################################################################
# Stirling-PDF SaaS environment defaults.
# Stirling-PDF SaaS environment defaults. Committed, non-secret. Real values for secrets go in
# .env.saas.local, which is loaded first and wins. Do not commit that file.
#
# This file is committed and provides non-secret defaults loaded by
# `task backend:dev:saas`. Put real values for secrets (passwords, project
# refs, edge function secrets) in `.env.saas.local` - any variable set there
# takes precedence over what's defined here.
# Three environments, each deriving its Supabase URLs, JWT issuer and JWKS from one project ref:
#
# DO NOT commit `.env.saas.local`. Only `.env.saas` is checked in.
###############################################################################
# prod PROFILES=none SAAS_DB_* the live project
# staging PROFILES=staging SAAS_STAGING_* pinned to v3, always there
# dev PROFILES=dev SAAS_DEV_* follows a SaaS PR's preview branch
#
# dev is the default for `task backend:dev:saas`. Use staging for somewhere stable; use dev when
# testing an open SaaS PR, since its preview branch is the only place those migrations are applied.
# ---------- Supabase project ----------
# ---------- Supabase project (prod / no-profile) ----------
# Project reference (the subdomain part of <ref>.supabase.co). Required.
# Set in .env.saas.local.
SAAS_DB_PROJECT_REF=
@@ -17,18 +18,35 @@ SAAS_DB_PROJECT_REF=
# Edge function secret used by billing/license rollup calls. Set in .env.saas.local.
SUPABASE_EDGE_FUNCTION_SECRET=
# ---------- Database (saas profile) ----------
# Direct JDBC URL to the Supabase Postgres. Required when running the plain
# `saas` profile (i.e. without `--spring.profiles.include=dev`).
# ---------- Database (no profile) ----------
# Direct JDBC URL to the Supabase Postgres. Required when running without
# `--spring.profiles.include=...`.
# Example: jdbc:postgresql://db.<project-ref>.supabase.co:5432/postgres
SAAS_DB_URL=
SAAS_DB_USERNAME=postgres
SAAS_DB_PASSWORD=
# ---------- Database (dev profile overrides) ----------
# Used when `--spring.profiles.include=dev` is active. The dev profile
# defaults the URL/username to the shared dev Supabase project, but the
# password must still be provided in .env.saas.local.
SAAS_DEV_DB_URL=
# ---------- staging profile ----------
# The shared long-lived v3 project. application-staging.properties defaults the ref,
# URL, database host and meter endpoint, so staging needs only the password, in
# .env.saas.local. Set SAAS_STAGING_PROJECT_REF to repoint it; everything derives.
#
# The ref and publishable key are duplicated here because the task derives the
# frontend's VITE_SUPABASE_* from them and a shell cannot read a Spring default.
# Neither is secret: the ref is a public subdomain, the key ships in the bundle.
SAAS_STAGING_PROJECT_REF=qacaivhsjtftfwtgjvva
SAAS_STAGING_PUBLISHABLE_KEY=sb_publishable_nIM8y-9ARPE7EzQwAQHKMg_40fCN6kY # gitleaks:allow
SAAS_STAGING_DB_USERNAME=postgres
SAAS_STAGING_DB_PASSWORD=
# ---------- dev profile ----------
# The SaaS PR's Supabase preview branch. Take the ref from that PR's "Supabase
# Preview" check; the profile derives URL, JWT issuer, JWKS, meter endpoint and
# database host from it, so this one value follows a different PR.
#
# A preview branch has its own password and keys; the parent project's will not
# authenticate. Both go in .env.saas.local, along with the ref.
SAAS_DEV_PROJECT_REF=
SAAS_DEV_PUBLISHABLE_KEY=
SAAS_DEV_DB_USERNAME=postgres
SAAS_DEV_DB_PASSWORD=
+2 -2
View File
@@ -21,8 +21,8 @@ dependencies {
api 'org.snakeyaml:snakeyaml-engine:3.0.1'
api "org.springdoc:springdoc-openapi-starter-webmvc-ui:3.0.3"
// Simple Java Mail for EML/MSG parsing (replaces direct Angus Mail usage)
api 'org.simplejavamail:simple-java-mail:9.2.0'
api 'org.simplejavamail:outlook-module:9.2.0' // MSG file support
api 'org.simplejavamail:simple-java-mail:9.3.1'
api 'org.simplejavamail:outlook-module:9.3.1' // MSG file support
api 'jakarta.mail:jakarta.mail-api:2.1.5'
runtimeOnly 'org.eclipse.angus:angus-mail:2.0.5'
@@ -6,6 +6,7 @@ import java.util.Map;
import java.util.Set;
import java.util.concurrent.ConcurrentHashMap;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.beans.factory.annotation.Qualifier;
import org.springframework.stereotype.Service;
@@ -13,6 +14,7 @@ import lombok.Getter;
import lombok.extern.slf4j.Slf4j;
import stirling.software.common.model.ApplicationProperties;
import stirling.software.common.service.PdfaLevelAServiceInterface;
@Service
@Slf4j
@@ -51,12 +53,16 @@ public class EndpointConfiguration {
private Map<String, DisableReason> groupDisableReasons = new ConcurrentHashMap<>();
private Map<String, Set<String>> endpointAlternatives = new ConcurrentHashMap<>();
private final boolean runningProOrHigher;
private final boolean pdfUaAvailable;
public EndpointConfiguration(
ApplicationProperties applicationProperties,
@Qualifier("runningProOrHigher") boolean runningProOrHigher) {
@Qualifier("runningProOrHigher") boolean runningProOrHigher,
@Autowired(required = false) PdfaLevelAServiceInterface pdfaLevelAService) {
this.applicationProperties = applicationProperties;
this.runningProOrHigher = runningProOrHigher;
// The PDF/UA tagger ships in the proprietary module, and so do its endpoints.
this.pdfUaAvailable = pdfaLevelAService != null;
init();
processEnvironmentConfigs();
}
@@ -356,6 +362,7 @@ public class EndpointConfiguration {
addEndpointToGroup("Convert", "pdf-to-img");
addEndpointToGroup("Convert", "img-to-pdf");
addEndpointToGroup("Convert", "pdf-to-pdfa");
addEndpointToGroup("Convert", "pdf-to-ua");
addEndpointToGroup("Convert", "file-to-pdf");
addEndpointToGroup("Convert", "pdf-to-word");
addEndpointToGroup("Convert", "pdf-to-presentation");
@@ -395,6 +402,7 @@ public class EndpointConfiguration {
// Backend-only endpoints (not in frontend tool registry endpoints)
addEndpointToGroup("Security", "redact");
addEndpointToGroup("Security", "verify-pdf");
addEndpointToGroup("Security", "accessibility-report");
addEndpointToGroup("Security", "sign");
// Adding endpoints to "Other" group
@@ -529,6 +537,8 @@ public class EndpointConfiguration {
addEndpointToGroup("Java", "json-to-pdf");
addEndpointToGroup("Java", "pdf-to-video");
addEndpointToGroup("Java", "verify-pdf");
addEndpointToGroup("Java", "pdf-to-ua");
addEndpointToGroup("Java", "accessibility-report");
addEndpointToGroup("Java", "flatten");
addEndpointToGroup("Java", "unlock-pdf-forms");
addEndpointToGroup("Java", "validate-signature");
@@ -600,6 +610,8 @@ public class EndpointConfiguration {
// veraPDF dependent endpoints
addEndpointToGroup("veraPDF", "verify-pdf");
addEndpointToGroup("veraPDF", "pdf-to-ua");
addEndpointToGroup("veraPDF", "accessibility-report");
// Pdftohtml dependent endpoints
addEndpointToGroup("Pdftohtml", "pdf-to-html");
@@ -630,6 +642,11 @@ public class EndpointConfiguration {
disableGroup("enterprise");
}
if (!pdfUaAvailable) {
disableEndpoint("pdf-to-ua");
disableEndpoint("accessibility-report");
}
if (!applicationProperties.getSystem().isEnableUrlToPDF()) {
disableEndpoint("url-to-pdf");
}
@@ -0,0 +1,22 @@
package stirling.software.common.service;
import java.util.List;
/**
* Raises a converted PDF/A file from conformance level B to level A, which needs the tagging the
* PDF/UA tagger does. Implemented only in the proprietary module; core builds convert at level B.
*/
public interface PdfaLevelAServiceInterface {
/**
* @param levelA true only when the file was tagged and validated, so the claim is never a guess
*/
record Result(byte[] pdfBytes, boolean levelA, List<String> warnings) {}
/**
* @param part PDF/A part, 1 to 3; part 1 keeps its PDF 1.4 version
* @param alsoDeclareUa additionally claim PDF/UA, but only if it validates
*/
Result upgradeToLevelA(
byte[] pdfBytes, int part, String language, String title, boolean alsoDeclareUa);
}
@@ -17,6 +17,7 @@ import org.junit.jupiter.api.Test;
import stirling.software.SPDF.config.EndpointConfiguration.DisableReason;
import stirling.software.SPDF.config.EndpointConfiguration.EndpointAvailability;
import stirling.software.common.model.ApplicationProperties;
import stirling.software.common.service.PdfaLevelAServiceInterface;
/**
* Unit tests for {@link EndpointConfiguration}. The class wires up its endpoint/group registry in
@@ -32,7 +33,14 @@ class EndpointConfigurationGapTest {
* Construct an EndpointConfiguration with the given pro flag and current applicationProperties.
*/
private EndpointConfiguration build(boolean runningProOrHigher) {
return new EndpointConfiguration(applicationProperties, runningProOrHigher);
return build(runningProOrHigher, null);
}
/** The PDF/UA service is only present in proprietary builds, so it is injected separately. */
private EndpointConfiguration build(
boolean runningProOrHigher, PdfaLevelAServiceInterface pdfaLevelAService) {
return new EndpointConfiguration(
applicationProperties, runningProOrHigher, pdfaLevelAService);
}
/** Default config: not pro, no removals, url-to-pdf disabled (default System flag is false). */
@@ -177,6 +185,28 @@ class EndpointConfigurationGapTest {
}
}
@Nested
@DisplayName("PDF/UA availability")
class PdfUaTests {
@Test
@DisplayName("the PDF/UA endpoints are off when the proprietary tagger is absent")
void disabledWithoutTagger() {
EndpointConfiguration config = build(false, null);
assertFalse(config.isEndpointEnabled("pdf-to-ua"));
assertFalse(config.isEndpointEnabled("accessibility-report"));
}
@Test
@DisplayName("they are on once the tagger is on the classpath")
void enabledWithTagger() {
EndpointConfiguration config =
build(false, (pdfBytes, part, language, title, alsoDeclareUa) -> null);
assertTrue(config.isEndpointEnabled("pdf-to-ua"));
assertTrue(config.isEndpointEnabled("accessibility-report"));
}
}
@Nested
@DisplayName("group enable / disable")
class GroupTests {
@@ -11,6 +11,7 @@ import java.time.Instant;
import java.time.ZoneId;
import java.time.ZonedDateTime;
import java.util.*;
import java.util.Locale;
import java.util.regex.Pattern;
import java.util.stream.Collectors;
import java.util.stream.Stream;
@@ -71,6 +72,7 @@ import org.apache.xmpbox.schema.PDFAIdentificationSchema;
import org.apache.xmpbox.schema.XMPBasicSchema;
import org.apache.xmpbox.xml.DomXmpParser;
import org.apache.xmpbox.xml.XmpSerializer;
import org.springframework.beans.factory.annotation.Autowired;
import org.springframework.core.io.Resource;
import org.springframework.http.HttpStatus;
import org.springframework.http.MediaType;
@@ -83,7 +85,6 @@ import io.github.pixee.security.Filenames;
import io.swagger.v3.oas.annotations.Operation;
import lombok.Getter;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import stirling.software.SPDF.model.api.converters.PdfToPdfARequest;
@@ -93,6 +94,7 @@ import stirling.software.common.configuration.RuntimePathConfig;
import stirling.software.common.enumeration.ResourceWeight;
import stirling.software.common.model.tool.ToolFormat;
import stirling.software.common.model.tool.ToolIO;
import stirling.software.common.service.PdfaLevelAServiceInterface;
import stirling.software.common.util.ExceptionUtils;
import stirling.software.common.util.ProcessExecutor;
import stirling.software.common.util.ProcessExecutor.ProcessExecutorResult;
@@ -102,14 +104,26 @@ import stirling.software.common.util.WebResponseUtils;
@ConvertApi
@Slf4j
@RequiredArgsConstructor
public class ConvertPDFToPDFA {
private static final Pattern NON_PRINTABLE_ASCII = Pattern.compile("[^\\x20-\\x7E]");
private final RuntimePathConfig runtimePathConfig;
private final stirling.software.SPDF.service.VeraPDFService veraPDFService;
// Level A needs the proprietary tagger; core builds convert at level B instead.
private final PdfaLevelAServiceInterface pdfaLevelAService;
private final TempFileManager tempFileManager;
public ConvertPDFToPDFA(
RuntimePathConfig runtimePathConfig,
stirling.software.SPDF.service.VeraPDFService veraPDFService,
@Autowired(required = false) PdfaLevelAServiceInterface pdfaLevelAService,
TempFileManager tempFileManager) {
this.runtimePathConfig = runtimePathConfig;
this.veraPDFService = veraPDFService;
this.pdfaLevelAService = pdfaLevelAService;
this.tempFileManager = tempFileManager;
}
private static final String ICC_RESOURCE_PATH = "/icc/sRGB2014.icc";
private static final int PDFA_COMPATIBILITY_POLICY = 1;
@@ -604,7 +618,10 @@ public class ConvertPDFToPDFA {
return handlePdfXConversion(inputFile, outputFormat);
} else {
return handlePdfAConversion(
inputFile, outputFormat, request.getStrict() != null && request.getStrict());
inputFile,
outputFormat,
request.getStrict() != null && request.getStrict(),
request.getPdfUa() != null && request.getPdfUa());
}
}
@@ -1815,8 +1832,64 @@ public class ConvertPDFToPDFA {
return Files.readAllBytes(outputPdf);
}
/** Tags a converted PDF/A for level A; must run after Ghostscript, which discards tags. */
private PdfaLevelAServiceInterface.Result applyLevelA(
byte[] converted,
Path original,
PdfaProfile profile,
String baseFileName,
boolean declarePdfUa) {
if (!profile.requiresTagging()) {
return new PdfaLevelAServiceInterface.Result(converted, true, List.of());
}
if (pdfaLevelAService == null) {
return new PdfaLevelAServiceInterface.Result(
converted,
false,
List.of(
"Level A tagging is not available in this build, so the file was left"
+ " at conformance level B."));
}
// Prefer the document's own title/language; hardcoding "en" mislabelled German reports.
// Read the original, not the converted bytes: Ghostscript discards /Lang, so probing its
// output always yields null and every document would be relabelled with the default.
String language = null;
String title = null;
try (PDDocument probe = Loader.loadPDF(original.toFile())) {
language = probe.getDocumentCatalog().getLanguage();
title = probe.getDocumentInformation().getTitle();
} catch (IOException e) {
log.debug("Could not read original title/language: {}", e.getMessage());
}
if (language == null || language.isBlank()) {
try (PDDocument probe = Loader.loadPDF(converted)) {
language = probe.getDocumentCatalog().getLanguage();
if (title == null || title.isBlank()) {
title = probe.getDocumentInformation().getTitle();
}
} catch (IOException e) {
log.debug("Could not read converted title/language: {}", e.getMessage());
}
}
PdfaLevelAServiceInterface.Result result =
pdfaLevelAService.upgradeToLevelA(
converted,
profile.getPart(),
language,
title != null && !title.isBlank() ? title : baseFileName,
declarePdfUa);
result.warnings().forEach(warning -> log.info("PDF/A level A: {}", warning));
if (!result.levelA()) {
log.warn(
"{} requested but the document could not be tagged; returning level B",
profile.getDisplayName());
}
return result;
}
private ResponseEntity<Resource> handlePdfAConversion(
MultipartFile inputFile, String outputFormat, boolean strict) throws Exception {
MultipartFile inputFile, String outputFormat, boolean strict, boolean declarePdfUa)
throws Exception {
PdfaProfile profile = PdfaProfile.fromRequest(outputFormat);
// Get the original filename without extension
@@ -1841,12 +1914,15 @@ public class ConvertPDFToPDFA {
log.info("Using Ghostscript for PDF/A conversion to {}", profile.getDisplayName());
try {
converted = convertWithGhostscript(inputPath, workingDir, profile);
String outputFilename = baseFileName + profile.outputSuffix();
var levelA =
applyLevelA(converted, inputPath, profile, baseFileName, declarePdfUa);
converted = levelA.pdfBytes();
String outputFilename = baseFileName + profile.outputSuffix(levelA.levelA());
validateAndWarnPdfA(converted, profile, "Ghostscript");
if (strict) {
verifyStrictCompliance(converted);
verifyStrictCompliance(converted, profile, levelA.levelA());
}
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
@@ -1867,13 +1943,15 @@ public class ConvertPDFToPDFA {
}
converted = convertWithPdfBoxMethod(inputPath, profile);
String outputFilename = baseFileName + profile.outputSuffix();
var levelA = applyLevelA(converted, inputPath, profile, baseFileName, declarePdfUa);
converted = levelA.pdfBytes();
String outputFilename = baseFileName + profile.outputSuffix(levelA.levelA());
// Validate with PDFBox preflight and warn if issues found
validateAndWarnPdfA(converted, profile, "PDFBox/LibreOffice");
if (strict) {
verifyStrictCompliance(converted);
verifyStrictCompliance(converted, profile, levelA.levelA());
}
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
@@ -1889,11 +1967,56 @@ public class ConvertPDFToPDFA {
}
}
private void verifyStrictCompliance(byte[] pdfBytes) throws IOException {
/** True for a PDF/UA or WCAG result, which says nothing about archival conformance. */
private static boolean isAccessibilityProfile(
stirling.software.SPDF.model.api.security.PDFVerificationResult result) {
String profile = result.getValidationProfile();
if (profile == null) {
return false;
}
String normalised = profile.toLowerCase(Locale.ROOT);
return normalised.contains("ua") || normalised.contains("wcag");
}
/**
* True when a result speaks for the requested profile. Only archival results count, and a level
* B pass must never satisfy a level A request.
*/
private static boolean answersRequest(
PdfaProfile profile,
stirling.software.SPDF.model.api.security.PDFVerificationResult result) {
if (isAccessibilityProfile(result)) {
return false;
}
String standard = result.getStandard();
if (standard == null || standard.length() < 2) {
return false;
}
if (standard.charAt(0) != Character.forDigit(profile.getPart(), 10)) {
return false;
}
return !profile.requiresTagging() || Character.toLowerCase(standard.charAt(1)) == 'a';
}
private void verifyStrictCompliance(byte[] pdfBytes, PdfaProfile profile, boolean levelAReached)
throws IOException {
// Tagging is the only route to level A, so an untagged file cannot answer a strict request.
if (!levelAReached) {
throw new ResponseStatusException(
HttpStatus.BAD_REQUEST,
"Strict PDF/A mode enabled: the document could not be tagged, so "
+ profile.getDisplayName()
+ " was not reached. It is valid at level B.");
}
try (InputStream is = new ByteArrayInputStream(pdfBytes)) {
List<stirling.software.SPDF.model.api.security.PDFVerificationResult> results =
veraPDFService.validatePDF(is);
boolean isCompliant = results.stream().anyMatch(result -> result.isCompliant());
boolean isCompliant =
results.stream()
.filter(result -> answersRequest(profile, result))
.anyMatch(
stirling.software.SPDF.model.api.security.PDFVerificationResult
::isCompliant);
if (!isCompliant) {
String details =
results.stream()
@@ -1901,7 +2024,9 @@ public class ConvertPDFToPDFA {
.collect(Collectors.joining("; "));
throw new ResponseStatusException(
HttpStatus.BAD_REQUEST,
"Strict PDF/A mode enabled: Conversion is not perfectly compliant. Details: "
"Strict PDF/A mode enabled: the output is not perfectly compliant with "
+ profile.getDisplayName()
+ ". Details: "
+ details);
}
} catch (Exception e) {
@@ -2466,11 +2591,16 @@ public class ConvertPDFToPDFA {
@Getter
private enum PdfaProfile {
PDF_A_1B(1, "PDF/A-1b", "_PDFA-1b.pdf", "1.4", Format.PDF_A1B, "pdfa-1"),
PDF_A_2B(2, "PDF/A-2b", "_PDFA-2b.pdf", "1.7", null, "pdfa", "pdfa-2", "pdfa-2b"),
PDF_A_3B(3, "PDF/A-3b", "_PDFA-3b.pdf", "1.7", null, "pdfa-3", "pdfa-3b");
PDF_A_1B(1, "B", "PDF/A-1b", "_PDFA-1b.pdf", "1.4", Format.PDF_A1B, "pdfa-1"),
PDF_A_2B(2, "B", "PDF/A-2b", "_PDFA-2b.pdf", "1.7", null, "pdfa", "pdfa-2", "pdfa-2b"),
PDF_A_3B(3, "B", "PDF/A-3b", "_PDFA-3b.pdf", "1.7", null, "pdfa-3", "pdfa-3b"),
// Level A = level B plus tagging, declared language and Unicode text; tagged post-convert.
PDF_A_1A(1, "A", "PDF/A-1a", "_PDFA-1a.pdf", "1.4", Format.PDF_A1B, "pdfa-1a"),
PDF_A_2A(2, "A", "PDF/A-2a", "_PDFA-2a.pdf", "1.7", null, "pdfa-2a"),
PDF_A_3A(3, "A", "PDF/A-3a", "_PDFA-3a.pdf", "1.7", null, "pdfa-3a");
private final int part;
private final String conformanceLevel;
private final String displayName;
private final String suffix;
private final String compatibilityLevel;
@@ -2479,12 +2609,14 @@ public class ConvertPDFToPDFA {
PdfaProfile(
int part,
String conformanceLevel,
String displayName,
String suffix,
String compatibilityLevel,
Format preflightFormat,
String... requestTokens) {
this.part = part;
this.conformanceLevel = conformanceLevel;
this.displayName = displayName;
this.suffix = suffix;
this.compatibilityLevel = compatibilityLevel;
@@ -2495,6 +2627,10 @@ public class ConvertPDFToPDFA {
.toList();
}
boolean requiresTagging() {
return "A".equals(conformanceLevel);
}
static PdfaProfile fromRequest(String requestToken) {
if (requestToken == null) {
return PDF_A_2B;
@@ -2508,8 +2644,11 @@ public class ConvertPDFToPDFA {
return match.orElse(PDF_A_2B);
}
String outputSuffix() {
return suffix;
/**
* Names the file at the level actually reached; a level A name over level B content lies.
*/
String outputSuffix(boolean levelAReached) {
return levelAReached ? suffix : "_PDFA-" + part + "b.pdf";
}
Optional<Format> preflightFormat() {
@@ -14,9 +14,19 @@ public class PdfToPdfARequest extends PDFFile {
@Schema(
description = "The output format type (PDF/A or PDF/X)",
requiredMode = Schema.RequiredMode.REQUIRED,
allowableValues = {"pdfa", "pdfa-1", "pdfa-2", "pdfa-2b", "pdfa-3", "pdfa-3b", "pdfx"})
allowableValues = {
"pdfa", "pdfa-1", "pdfa-2", "pdfa-2b", "pdfa-3", "pdfa-3b", "pdfa-1a", "pdfa-2a",
"pdfa-3a", "pdfx"
})
private String outputFormat;
@Schema(
description =
"Also declare PDF/UA accessibility alongside PDF/A. Only applies to the level A"
+ " formats, and the claim is written only if it validates.",
defaultValue = "false")
private Boolean pdfUa;
@Schema(
description =
"If true, the conversion will fail if the output is not perfectly compliant")
@@ -285,6 +285,8 @@ public class VeraPDFService {
}
}
// Never force PDF/UA here - it flags every ordinary document as non-compliant and doubles
// verify cost; /accessibility-report checks PDF/UA on demand.
if (!hasPdfaDeclaration) {
results.add(createNoPdfaDeclarationResult());
}
@@ -14,6 +14,13 @@
"moduleLicense": "GNU Lesser General Public License",
"moduleLicenseUrl": "https://www.gnu.org/licenses/old-licenses/lgpl-2.1.html"
},
{
"moduleName": "ch.qos.logback:logback-classic",
"moduleUrl": "http://www.qos.ch",
"moduleVersion": "1.6.1",
"moduleLicense": "LGPL-2.1-only",
"moduleLicenseUrl": "https://www.gnu.org/licenses/old-licenses/lgpl-2.1.html"
},
{
"moduleName": "ch.qos.logback:logback-core",
"moduleUrl": "http://www.qos.ch",
@@ -21,6 +28,13 @@
"moduleLicense": "GNU Lesser General Public License",
"moduleLicenseUrl": "https://www.gnu.org/licenses/old-licenses/lgpl-2.1.html"
},
{
"moduleName": "ch.qos.logback:logback-core",
"moduleUrl": "http://www.qos.ch",
"moduleVersion": "1.6.1",
"moduleLicense": "LGPL-2.1-only",
"moduleLicenseUrl": "https://www.gnu.org/licenses/old-licenses/lgpl-2.1.html"
},
{
"moduleName": "com.adobe.xmp:xmpcore",
"moduleUrl": "https://www.adobe.com/devnet/xmp/library/eula-xmp-library-java.html",
@@ -182,7 +196,7 @@
{
"moduleName": "com.github.mwiede:jsch",
"moduleUrl": "https://github.com/mwiede/jsch",
"moduleVersion": "0.2.23",
"moduleVersion": "2.28.6",
"moduleLicense": "Revised BSD",
"moduleLicenseUrl": "https://github.com/mwiede/jsch/blob/master/LICENSE.txt"
},
@@ -758,7 +772,7 @@
{
"moduleName": "commons-net:commons-net",
"moduleUrl": "https://commons.apache.org/proper/commons-net/",
"moduleVersion": "3.11.1",
"moduleVersion": "3.13.0",
"moduleLicense": "Apache-2.0",
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
},
@@ -1213,24 +1227,48 @@
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "net.shibboleth:shib-networking",
"moduleVersion": "9.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "net.shibboleth:shib-security",
"moduleVersion": "9.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "net.shibboleth:shib-security",
"moduleVersion": "9.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "net.shibboleth:shib-support",
"moduleVersion": "9.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "net.shibboleth:shib-support",
"moduleVersion": "9.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "net.shibboleth:shib-velocity",
"moduleVersion": "9.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "net.shibboleth:shib-velocity",
"moduleVersion": "9.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.antlr:antlr4-runtime",
"moduleUrl": "https://www.antlr.org/",
@@ -1325,13 +1363,6 @@
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.apache.httpcomponents:httpclient",
"moduleUrl": "http://hc.apache.org/httpcomponents-client",
"moduleVersion": "4.5.13",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.apache.httpcomponents:httpclient",
"moduleUrl": "http://hc.apache.org/httpcomponents-client-ga",
@@ -1456,6 +1487,13 @@
"moduleLicense": "Apache-2.0",
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.apache.santuario:xmlsec",
"moduleUrl": "https://www.apache.org/",
"moduleVersion": "3.0.6",
"moduleLicense": "Apache-2.0",
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.apache.tomcat.embed:tomcat-embed-el",
"moduleUrl": "https://tomcat.apache.org/",
@@ -1470,6 +1508,13 @@
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.apache.velocity:velocity-engine-core",
"moduleUrl": "https://www.apache.org/",
"moduleVersion": "2.4.1",
"moduleLicense": "Apache-2.0",
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.apache.xmlbeans:xmlbeans",
"moduleUrl": "https://xmlbeans.apache.org/",
@@ -1647,6 +1692,13 @@
"moduleLicense": "GNU Lesser General Public License",
"moduleLicenseUrl": "http://www.gnu.org/licenses/lgpl-3.0.txt"
},
{
"moduleName": "org.cryptacular:cryptacular",
"moduleUrl": "https://www.cryptacular.org",
"moduleVersion": "1.3.0",
"moduleLicense": "GNU Lesser General Public License",
"moduleLicenseUrl": "https://www.gnu.org/licenses/lgpl-3.0.txt"
},
{
"moduleName": "org.eclipse.angus:angus-activation",
"moduleUrl": "https://www.eclipse.org",
@@ -2016,78 +2068,156 @@
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-core-api",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-core-impl",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-core-impl",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-messaging-api",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-messaging-api",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-profile-api",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-profile-api",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-saml-api",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-saml-api",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-saml-impl",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-saml-impl",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-security-api",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-security-api",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-security-impl",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-security-impl",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-soap-api",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-soap-api",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-soap-impl",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-soap-impl",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-storage-api",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-storage-api",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-xmlsec-api",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-xmlsec-api",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-xmlsec-impl",
"moduleVersion": "5.1.6",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.opensaml:opensaml-xmlsec-impl",
"moduleVersion": "5.2.2",
"moduleLicense": "The Apache Software License, Version 2.0",
"moduleLicenseUrl": "http://www.apache.org/licenses/LICENSE-2.0.txt"
},
{
"moduleName": "org.ow2.asm:asm",
"moduleUrl": "http://asm.ow2.org",
@@ -2564,6 +2694,13 @@
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0"
},
{
"moduleName": "org.springframework.security:spring-security-core",
"moduleUrl": "https://spring.io/projects/spring-security",
"moduleVersion": "7.1.0",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0"
},
{
"moduleName": "org.springframework.security:spring-security-crypto",
"moduleUrl": "https://spring.io/projects/spring-security",
@@ -2606,6 +2743,13 @@
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0"
},
{
"moduleName": "org.springframework.security:spring-security-saml2-service-provider",
"moduleUrl": "https://spring.io/projects/spring-security",
"moduleVersion": "7.1.0",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://www.apache.org/licenses/LICENSE-2.0"
},
{
"moduleName": "org.springframework.security:spring-security-web",
"moduleUrl": "https://spring.io/projects/spring-security",
@@ -2805,207 +2949,207 @@
},
{
"moduleName": "software.amazon.awssdk:annotations",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:apache-client",
"moduleVersion": "2.44.12",
"moduleName": "software.amazon.awssdk:apache5-client",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:arns",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:auth",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:aws-core",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:aws-query-protocol",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:aws-xml-protocol",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:checksums",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:checksums-spi",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:crt-core",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:endpoints-spi",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:http-auth",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:http-auth-aws",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:http-auth-aws-eventstream",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:http-auth-spi",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:http-client-spi",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:identity-spi",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:json-utils",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:metrics-spi",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:netty-nio-client",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:profiles",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:protocol-core",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:regions",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:retries",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:retries-spi",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:s3",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:sdk-core",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:third-party-jackson-core",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:url-connection-client",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:utils",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
{
"moduleName": "software.amazon.awssdk:utils-lite",
"moduleUrl": "https://aws.amazon.com/sdkforjava",
"moduleVersion": "2.44.12",
"moduleVersion": "2.51.3",
"moduleLicense": "Apache License, Version 2.0",
"moduleLicenseUrl": "https://aws.amazon.com/apache2.0"
},
@@ -46,6 +46,7 @@ import stirling.software.SPDF.model.api.converters.PdfToPdfARequest;
import stirling.software.SPDF.model.api.security.PDFVerificationResult;
import stirling.software.SPDF.service.VeraPDFService;
import stirling.software.common.configuration.RuntimePathConfig;
import stirling.software.common.service.PdfaLevelAServiceInterface;
import stirling.software.common.util.TempFileManager;
/**
@@ -62,10 +63,12 @@ class ConvertPDFToPDFAGapTest {
@Mock private RuntimePathConfig runtimePathConfig;
@Mock private VeraPDFService veraPDFService;
@Mock private PdfaLevelAServiceInterface pdfaLevelAService;
@Mock private TempFileManager tempFileManager;
private ConvertPDFToPDFA newController() {
return new ConvertPDFToPDFA(runtimePathConfig, veraPDFService, tempFileManager);
return new ConvertPDFToPDFA(
runtimePathConfig, veraPDFService, pdfaLevelAService, tempFileManager);
}
// ---- reflection helpers ----------------------------------------------------------------
@@ -161,9 +164,21 @@ class ConvertPDFToPDFAGapTest {
}
private String suffixOf(Object profile) throws Exception {
Method m = profile.getClass().getDeclaredMethod("outputSuffix");
return suffixOf(profile, true);
}
private String suffixOf(Object profile, boolean levelAReached) throws Exception {
Method m = profile.getClass().getDeclaredMethod("outputSuffix", boolean.class);
m.setAccessible(true);
return (String) m.invoke(profile);
return (String) m.invoke(profile, levelAReached);
}
@Test
@DisplayName("a level A profile falls back to the level B name when tagging failed")
void levelANotReachedIsNamedLevelB() throws Exception {
assertThat(suffixOf(resolveProfile("pdfa-1a"), false)).isEqualTo("_PDFA-1b.pdf");
assertThat(suffixOf(resolveProfile("pdfa-2a"), false)).isEqualTo("_PDFA-2b.pdf");
assertThat(suffixOf(resolveProfile("pdfa-3a"), true)).isEqualTo("_PDFA-3a.pdf");
}
@Test
@@ -717,6 +732,30 @@ class ConvertPDFToPDFAGapTest {
@DisplayName("verifyStrictCompliance (VeraPDFService mocked)")
class StrictCompliance {
private Object profile(String token) throws Exception {
Class<?> enumClass = null;
for (Class<?> inner : ConvertPDFToPDFA.class.getDeclaredClasses()) {
if (inner.getSimpleName().equals("PdfaProfile")) {
enumClass = inner;
}
}
Method m = enumClass.getDeclaredMethod("fromRequest", String.class);
m.setAccessible(true);
return m.invoke(null, token);
}
private Throwable verify(String token, boolean levelAReached) throws Exception {
ConvertPDFToPDFA controller = newController();
return catchThrowable(
() ->
invokeInstance(
controller,
"verifyStrictCompliance",
(Object) "dummy".getBytes(),
profile(token),
levelAReached));
}
@Test
@DisplayName("compliant result passes without throwing")
void compliantPasses() throws Exception {
@@ -726,14 +765,7 @@ class ConvertPDFToPDFAGapTest {
ok.setComplianceSummary("PDF/A-1b compliant");
when(veraPDFService.validatePDF(any())).thenReturn(List.of(ok));
ConvertPDFToPDFA controller = newController();
assertThatCode(
() ->
invokeInstance(
controller,
"verifyStrictCompliance",
(Object) "dummy".getBytes()))
.doesNotThrowAnyException();
assertThat(verify("pdfa-1", true)).isNull();
}
@Test
@@ -745,34 +777,70 @@ class ConvertPDFToPDFAGapTest {
bad.setComplianceSummary("PDF/A-1b with errors");
when(veraPDFService.validatePDF(any())).thenReturn(List.of(bad));
ConvertPDFToPDFA controller = newController();
ResponseStatusException ex =
(ResponseStatusException)
catchThrowable(
() ->
invokeInstance(
controller,
"verifyStrictCompliance",
(Object) "dummy".getBytes()));
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-1", true);
assertThat(ex).isNotNull();
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
assertThat(ex.getReason()).contains("PDF/A-1b with errors");
}
@Test
@DisplayName("a level B pass does not satisfy a level A request")
void levelBDoesNotSatisfyLevelA() throws Exception {
PDFVerificationResult ok = new PDFVerificationResult();
ok.setCompliant(true);
ok.setStandard("1b");
ok.setComplianceSummary("PDF/A-1b compliant");
when(veraPDFService.validatePDF(any())).thenReturn(List.of(ok));
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-1a", true);
assertThat(ex).isNotNull();
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
assertThat(ex.getReason()).contains("PDF/A-1a");
}
@Test
@DisplayName("a level A result satisfies a level A request")
void levelASatisfiesLevelA() throws Exception {
PDFVerificationResult ok = new PDFVerificationResult();
ok.setCompliant(true);
ok.setStandard("2a");
ok.setComplianceSummary("PDF/A-2a compliant");
when(veraPDFService.validatePDF(any())).thenReturn(List.of(ok));
assertThat(verify("pdfa-2a", true)).isNull();
}
@Test
@DisplayName("untagged output fails a level A request before validation runs")
void untaggedLevelARequestFails() throws Exception {
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-2a", false);
assertThat(ex).isNotNull();
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
assertThat(ex.getReason()).contains("could not be tagged");
verifyNoInteractions(veraPDFService);
}
@Test
@DisplayName("a compliant PDF/UA result never satisfies a strict PDF/A request")
void accessibilityResultIsIgnored() throws Exception {
PDFVerificationResult ua = new PDFVerificationResult();
ua.setCompliant(true);
ua.setStandard("ua1");
ua.setValidationProfile("ua1");
ua.setComplianceSummary("PDF/UA-1 compliant");
when(veraPDFService.validatePDF(any())).thenReturn(List.of(ua));
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-2b", true);
assertThat(ex).isNotNull();
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
}
@Test
@DisplayName("empty result list is treated as non-compliant -> 400")
void emptyResultsTreatedNonCompliant() throws Exception {
when(veraPDFService.validatePDF(any())).thenReturn(Collections.emptyList());
ConvertPDFToPDFA controller = newController();
ResponseStatusException ex =
(ResponseStatusException)
catchThrowable(
() ->
invokeInstance(
controller,
"verifyStrictCompliance",
(Object) "dummy".getBytes()));
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-1", true);
assertThat(ex).isNotNull();
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.BAD_REQUEST);
}
@@ -782,15 +850,7 @@ class ConvertPDFToPDFAGapTest {
void serviceErrorWrappedAs500() throws Exception {
when(veraPDFService.validatePDF(any())).thenThrow(new IOException("boom"));
ConvertPDFToPDFA controller = newController();
ResponseStatusException ex =
(ResponseStatusException)
catchThrowable(
() ->
invokeInstance(
controller,
"verifyStrictCompliance",
(Object) "dummy".getBytes()));
ResponseStatusException ex = (ResponseStatusException) verify("pdfa-1", true);
assertThat(ex).isNotNull();
assertThat(ex.getStatusCode()).isEqualTo(HttpStatus.INTERNAL_SERVER_ERROR);
}
@@ -42,6 +42,7 @@ import org.springframework.mock.web.MockMultipartFile;
import stirling.software.SPDF.model.api.converters.PdfToPdfARequest;
import stirling.software.SPDF.service.VeraPDFService;
import stirling.software.common.configuration.RuntimePathConfig;
import stirling.software.common.service.PdfaLevelAServiceInterface;
import stirling.software.common.util.ProcessExecutor;
import stirling.software.common.util.ProcessExecutor.ProcessExecutorResult;
import stirling.software.common.util.TempFile;
@@ -63,10 +64,12 @@ class ConvertPDFToPDFAMoreTest {
@Mock private RuntimePathConfig runtimePathConfig;
@Mock private VeraPDFService veraPDFService;
@Mock private PdfaLevelAServiceInterface pdfaLevelAService;
@Mock private TempFileManager tempFileManager;
private ConvertPDFToPDFA newController() {
return new ConvertPDFToPDFA(runtimePathConfig, veraPDFService, tempFileManager);
return new ConvertPDFToPDFA(
runtimePathConfig, veraPDFService, pdfaLevelAService, tempFileManager);
}
private static ResponseEntity<Resource> streamingOk(byte[] bytes) {
@@ -90,7 +90,9 @@ class VeraPDFServicePdfaFixtureTest {
() -> service.validatePDF(new ByteArrayInputStream(pdfBytes)),
"Empty veraPDF flavour list must not surface as IndexOutOfBoundsException");
assertEquals(1, results.size());
// One result: PDF/UA is checked by the dedicated accessibility-report endpoint, not here.
assertEquals(1, results.size(), () -> "Expected a single PDF/A result, got: " + results);
PDFVerificationResult result = results.get(0);
assertEquals("not-pdfa", result.getStandard());
assertFalse(result.isDeclaredPdfa());
@@ -161,13 +163,22 @@ class VeraPDFServicePdfaFixtureTest {
}
}
/** The PDF/A result; every document is also checked against PDF/UA, so filter that one out. */
private PDFVerificationResult onlyResult(byte[] pdfBytes) throws Exception {
List<PDFVerificationResult> results =
service.validatePDF(new ByteArrayInputStream(pdfBytes));
assertNotNull(results);
assertEquals(1, results.size(), () -> "Expected a single result, got: " + results);
return results.get(0);
List<PDFVerificationResult> pdfaResults =
results.stream().filter(r -> !isUaResult(r)).toList();
assertEquals(
1, pdfaResults.size(), () -> "Expected a single PDF/A result, got: " + results);
return pdfaResults.get(0);
}
private static boolean isUaResult(PDFVerificationResult result) {
String profile = result.getValidationProfile();
return profile != null && profile.toLowerCase().contains("ua");
}
private static String messages(PDFVerificationResult result) {
+9
View File
@@ -37,6 +37,15 @@ dependencies {
// https://mvnrepository.com/artifact/com.bucket4j/bucket4j_jdk17
implementation "org.bouncycastle:bcprov-jdk18on:$bouncycastleVersion"
// PDF/UA tagging and its validation oracle.
implementation 'org.verapdf:validation-model:1.30.2'
// CVE-2025-66453: Explicit rhino 1.7.15 to override verapdf's 1.7.13
implementation "org.mozilla:rhino:${rhinoVersion}"
// veraPDF still uses javax.xml.bind, not the new jakarta namespace
implementation 'javax.xml.bind:jaxb-api:2.3.1'
runtimeOnly 'com.sun.xml.bind:jaxb-impl:2.3.9'
runtimeOnly 'com.sun.xml.bind:jaxb-core:4.0.9'
implementation "com.google.code.gson:gson:${gsonVersion}"
// jinjava/jjwt transitively request older Jackson 2 versions; declare the current
@@ -0,0 +1,176 @@
package stirling.software.proprietary.controller.api.converters;
import java.io.IOException;
import java.nio.file.Files;
import java.util.LinkedHashMap;
import java.util.Map;
import java.util.regex.Pattern;
import org.springframework.core.io.Resource;
import org.springframework.http.MediaType;
import org.springframework.http.ResponseEntity;
import org.springframework.web.bind.annotation.ModelAttribute;
import org.springframework.web.multipart.MultipartFile;
import io.github.pixee.security.Filenames;
import io.swagger.v3.oas.annotations.Operation;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import stirling.software.common.annotations.AutoJobPostMapping;
import stirling.software.common.annotations.api.ConvertApi;
import stirling.software.common.enumeration.ResourceWeight;
import stirling.software.common.model.tool.ToolFormat;
import stirling.software.common.model.tool.ToolIO;
import stirling.software.common.util.ExceptionUtils;
import stirling.software.common.util.TempFile;
import stirling.software.common.util.TempFileManager;
import stirling.software.common.util.WebResponseUtils;
import stirling.software.proprietary.model.api.converters.PdfToPdfUaRequest;
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
import stirling.software.proprietary.service.ua.PdfUaConversionService;
/** Converts a PDF to PDF/UA; response headers say whether the result actually conforms. */
@ConvertApi
@Slf4j
@RequiredArgsConstructor
public class ConvertPdfToPdfUa {
private static final String HEADER_DECLARED = "X-Stirling-UA-Declared";
private static final String HEADER_FAILURES = "X-Stirling-UA-Failures";
private static final String HEADER_ALT_NEEDED = "X-Stirling-UA-Figures-Needing-Alt";
private static final String HEADER_WARNINGS = "X-Stirling-UA-Warnings";
/** Any line ending, so descriptions pasted from any platform parse the same. */
private static final Pattern NEWLINE = Pattern.compile("\\R");
private final PdfUaConversionService conversionService;
private final TempFileManager tempFileManager;
@AutoJobPostMapping(
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
value = "/pdf/ua",
resourceWeight = ResourceWeight.LARGE_WEIGHT)
@ToolIO(produces = ToolFormat.PDF)
@Operation(
summary = "Convert a PDF to PDF/UA-1 or PDF/UA-2",
description =
"Tags the document, marks decorative content as artifacts, embeds fonts and"
+ " applies the document-level requirements of PDF/UA, then validates"
+ " the result. A conformance declaration is written only if validation"
+ " passes, so the returned file never claims more than it delivers.")
public ResponseEntity<Resource> pdfToPdfUa(@ModelAttribute PdfToPdfUaRequest request)
throws IOException {
MultipartFile input = request.getFileInput();
if (input == null || input.isEmpty()) {
throw ExceptionUtils.createPdfFileRequiredException();
}
String originalName = Filenames.toSimpleFileName(input.getOriginalFilename());
String stem = stripExtension(originalName == null ? "document" : originalName);
PdfUaProfile profile = PdfUaProfile.fromRequest(request.getProfile());
TaggingOptions options =
TaggingOptions.builder()
.profile(profile)
.title(request.getTitle())
.fallbackTitle(stem)
// Only used when the document declares no language of its own.
.language(
request.getLanguage() == null || request.getLanguage().isBlank()
? "en-GB"
: request.getLanguage())
.overrideLanguage(
request.getOverrideLanguage() != null
&& request.getOverrideLanguage())
.existingTags(existingTags(request.getExistingTags()))
.figurePolicy(figurePolicy(request.getFigurePolicy()))
.embedFonts(request.getEmbedFonts() == null || request.getEmbedFonts())
.altTextByFigure(parseAltText(request.getAltText()))
.build();
PdfUaConversionOutcome outcome = conversionService.convert(input.getBytes(), options);
log.info(
"Converted '{}' to {}: declared={}, {} remaining failure(s)",
originalName,
profile.displayName(),
outcome.declared(),
outcome.validation().totalFailures());
outcome.warnings().forEach(warning -> log.info("PDF/UA warning: {}", warning));
// Streamed from a temp file so a large conversion does not hold a second heap copy.
String suffix = outcome.declared() ? "_pdfua" + profile.part() : "_tagged";
TempFile tempOut = tempFileManager.createManagedTempFile(".pdf");
try {
Files.write(tempOut.getPath(), outcome.pdfBytes());
} catch (IOException e) {
tempOut.close();
throw e;
}
ResponseEntity<Resource> response =
WebResponseUtils.pdfFileToWebResponse(tempOut, stem + suffix + ".pdf");
return ResponseEntity.status(response.getStatusCode())
.headers(response.getHeaders())
.header(HEADER_DECLARED, String.valueOf(outcome.declared()))
.header(HEADER_FAILURES, String.valueOf(outcome.validation().totalFailures()))
.header(
HEADER_ALT_NEEDED,
String.valueOf(outcome.tagging().figuresNeedingAltText()))
// Count only: warning text is multi-line prose, which HTTP headers mangle.
.header(HEADER_WARNINGS, String.valueOf(outcome.warnings().size()))
.body(response.getBody());
}
/**
* Parses newline-separated {@code key=description} pairs, keyed as the report hands them out.
* Only the first "=" splits, since a description may contain one.
*/
public static Map<String, String> parseAltText(String raw) {
if (raw == null || raw.isBlank()) {
return Map.of();
}
Map<String, String> parsed = new LinkedHashMap<>();
for (String line : NEWLINE.split(raw)) {
int split = line.indexOf('=');
if (split <= 0) {
continue;
}
String key = line.substring(0, split).strip();
String description = line.substring(split + 1).strip();
if (!key.isEmpty() && !description.isEmpty()) {
parsed.put(key, description);
}
}
return parsed;
}
private static TaggingOptions.ExistingTags existingTags(String value) {
if (value == null) {
return TaggingOptions.ExistingTags.AUTO;
}
return switch (value.trim().toLowerCase()) {
case "keep" -> TaggingOptions.ExistingTags.KEEP;
case "rebuild" -> TaggingOptions.ExistingTags.REBUILD;
default -> TaggingOptions.ExistingTags.AUTO;
};
}
private static TaggingOptions.FigurePolicy figurePolicy(String value) {
if (value != null && value.trim().equalsIgnoreCase("mark-decorative")) {
return TaggingOptions.FigurePolicy.MARK_DECORATIVE;
}
return TaggingOptions.FigurePolicy.REQUIRE_ALT;
}
private static String stripExtension(String filename) {
int dot = filename.lastIndexOf('.');
return dot > 0 ? filename.substring(0, dot) : filename;
}
}
@@ -0,0 +1,67 @@
package stirling.software.proprietary.controller.api.security;
import java.io.IOException;
import org.springframework.http.MediaType;
import org.springframework.http.ResponseEntity;
import org.springframework.web.bind.annotation.ModelAttribute;
import org.springframework.web.multipart.MultipartFile;
import io.swagger.v3.oas.annotations.Operation;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import stirling.software.common.annotations.AutoJobPostMapping;
import stirling.software.common.annotations.api.SecurityApi;
import stirling.software.common.enumeration.ResourceWeight;
import stirling.software.common.model.tool.ToolFormat;
import stirling.software.common.model.tool.ToolIO;
import stirling.software.common.util.ExceptionUtils;
import stirling.software.proprietary.model.api.ua.AccessibilityReport;
import stirling.software.proprietary.model.api.ua.AccessibilityReportRequest;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.service.ua.AccessibilityAuditService;
/** Reports how accessible a document is, without modifying it. */
@SecurityApi
@RequiredArgsConstructor
@Slf4j
public class AccessibilityReportController {
private final AccessibilityAuditService auditService;
@ToolIO(produces = ToolFormat.JSON)
@Operation(
summary = "Report a document's accessibility standing",
description =
"Validates the document against PDF/UA and reports what fails, which failures"
+ " can be fixed automatically, and which checks still need a person."
+ " Does not modify the file.")
// Costs a full veraPDF pass plus the converter's own layout analysis over every page.
@AutoJobPostMapping(
value = "/accessibility-report",
consumes = MediaType.MULTIPART_FORM_DATA_VALUE,
resourceWeight = ResourceWeight.LARGE_WEIGHT)
public ResponseEntity<AccessibilityReport> report(
@ModelAttribute AccessibilityReportRequest request) {
MultipartFile file = request.getFileInput();
if (file == null || file.isEmpty()) {
throw ExceptionUtils.createPdfFileRequiredException();
}
PdfUaProfile profile = PdfUaProfile.fromRequest(request.getProfile());
try {
AccessibilityReport report = auditService.audit(file.getBytes(), profile);
log.info(
"Accessibility report for '{}': tagged={}, {} issue(s)",
file.getOriginalFilename(),
report.isTagged(),
report.getIssues().size());
return ResponseEntity.ok(report);
} catch (IOException e) {
throw ExceptionUtils.createRuntimeException(
"error.ioException", "Could not read the PDF: {0}", e, e.getMessage());
}
}
}
@@ -0,0 +1,73 @@
package stirling.software.proprietary.model.api.converters;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Data;
import lombok.EqualsAndHashCode;
import stirling.software.common.model.api.PDFFile;
@Data
@EqualsAndHashCode(callSuper = true)
public class PdfToPdfUaRequest extends PDFFile {
@Schema(
description = "PDF/UA conformance level to target",
defaultValue = "ua1",
allowableValues = {"ua1", "ua2"})
private String profile;
@Schema(
description =
"Document title, required by PDF/UA. Falls back to the first heading, then the"
+ " filename.")
private String title;
@Schema(
description =
"Document language as a BCP-47 tag, for example en-GB. Applied only when the"
+ " document does not already declare one, unless overrideLanguage is"
+ " set.",
defaultValue = "en-GB")
private String language;
@Schema(
description =
"Replace the language the document already declares. Off by default, so a"
+ " document is never relabelled into a language it is not written in.",
defaultValue = "false")
private Boolean overrideLanguage;
@Schema(
description =
"What to do with an existing structure tree: keep it, rebuild it, or decide"
+ " automatically",
defaultValue = "auto",
allowableValues = {"auto", "keep", "rebuild"})
private String existingTags;
@Schema(
description =
"How to treat images with no description. require-alt leaves them undescribed so"
+ " the report asks for input; mark-decorative treats every image as"
+ " decoration.",
defaultValue = "require-alt",
allowableValues = {"require-alt", "mark-decorative"})
private String figurePolicy;
@Schema(
description =
"Embed fonts the document references but does not carry. Required for"
+ " conformance and needs Ghostscript.",
defaultValue = "true")
private Boolean embedFonts;
@Schema(
description =
"Alternative descriptions for figures, as key=text pairs separated by newlines."
+ " Keys come from the accessibility-report endpoint's"
+ " figuresNeedingDescription list, for example \"0:12=Bar chart of"
+ " quarterly revenue\". Descriptions are never invented, so without"
+ " these an illustrated document cannot claim conformance.")
private String altText;
}
@@ -0,0 +1,38 @@
package stirling.software.proprietary.model.api.ua;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Data;
/** One accessibility problem, grouped across all of its occurrences. */
@Data
@Schema(description = "A single accessibility issue found in a document")
public class AccessibilityIssue {
@Schema(description = "ISO 14289 clause, e.g. 7.3")
private String clause;
@Schema(description = "Test number within the clause")
private String testNumber;
@Schema(description = "Plain-English description of the problem")
private String message;
@Schema(description = "The validator's own wording, for support and debugging")
private String technicalMessage;
@Schema(description = "error or warning")
private String severity = "error";
@Schema(description = "Standard the check came from, e.g. PDF/UA-1")
private String specification;
@Schema(description = "Where the problem was found, when the validator reports it")
private String location;
@Schema(description = "How many times this issue occurs")
private int occurrences;
@Schema(description = "True when the converter can fix this without human input")
private boolean autoFixable;
}
@@ -0,0 +1,63 @@
package stirling.software.proprietary.model.api.ua;
import java.util.List;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Data;
/**
* A document's accessibility standing. The machine/human split is load-bearing: veraPDF covers only
* about half of the Matterhorn Protocol, so a clean automated pass is not "accessible".
*/
@Data
@Schema(description = "Accessibility standing of a document")
public class AccessibilityReport {
@Schema(description = "Profile the document was checked against, e.g. PDF/UA-1")
private String profile;
@Schema(description = "Whether the document has a structure tree at all")
private boolean tagged;
@Schema(description = "Whether the document declares PDF/UA conformance in its metadata")
private boolean declaresConformance;
@Schema(description = "Whether every automated check passed")
private boolean passesAutomatedChecks;
@Schema(description = "Automated checks that failed, grouped by rule")
private List<AccessibilityIssue> issues = List.of();
@Schema(description = "Things a person still has to verify; automation cannot decide these")
private List<String> humanChecks = List.of();
@Schema(description = "How many of the failing checks the converter can fix on its own")
private int automaticallyFixable;
@Schema(description = "How many need information from the user, such as alternative text")
private int needsInput;
@Schema(
description =
"Figures that need an alternative description. Each carries the key to pass"
+ " back in the conversion request's altTextByFigure map, so a caller"
+ " can enumerate what is missing and then supply it.")
private List<FigureDescriptor> figuresNeedingDescription = List.of();
@Schema(description = "Document-level facts that drive most failures")
private Summary summary = new Summary();
@Data
@Schema(description = "Quick document-level facts")
public static class Summary {
private int pages;
private boolean hasTitle;
private boolean displaysDocTitle;
private boolean hasLanguage;
private boolean allFontsEmbedded;
private int unembeddedFonts;
private int figures;
private boolean encrypted;
}
}
@@ -0,0 +1,19 @@
package stirling.software.proprietary.model.api.ua;
import io.swagger.v3.oas.annotations.media.Schema;
import lombok.Data;
import lombok.EqualsAndHashCode;
import stirling.software.common.model.api.PDFFile;
@Data
@EqualsAndHashCode(callSuper = true)
public class AccessibilityReportRequest extends PDFFile {
@Schema(
description = "Profile to check against",
defaultValue = "ua1",
allowableValues = {"ua1", "ua2"})
private String profile;
}
@@ -0,0 +1,18 @@
package stirling.software.proprietary.model.api.ua;
import io.swagger.v3.oas.annotations.media.Schema;
/**
* One figure needing an alternative description, which is never invented. key is the
* altTextByFigure key "pageIndex:ordinal"; page is 1-based; kind is "figure" or "formula".
*/
@Schema(description = "A figure that needs an alternative description")
public record FigureDescriptor(
String key,
int page,
String kind,
float x,
float y,
float width,
float height,
String existingAlt) {}
@@ -0,0 +1,26 @@
package stirling.software.proprietary.model.api.ua;
import java.util.List;
import io.swagger.v3.oas.annotations.media.Schema;
/**
* Result of a PDF/UA conversion.
*
* @param declared whether a {@code pdfuaid} conformance claim was written into {@code pdfBytes}
*/
@Schema(description = "Result of converting a document to PDF/UA")
public record PdfUaConversionOutcome(
byte[] pdfBytes,
boolean declared,
UaValidationResult validation,
TaggingSummary tagging,
List<String> warnings) {
@Schema(description = "What the tagging pass produced")
public record TaggingSummary(
boolean rebuiltStructure,
int taggedElements,
int artifacts,
int figuresNeedingAltText) {}
}
@@ -0,0 +1,18 @@
package stirling.software.proprietary.model.api.ua;
import java.util.List;
import io.swagger.v3.oas.annotations.media.Schema;
/**
* Outcome of validating against one PDF/UA profile. compliant means every automated check passed,
* which is not the same as usable by assistive technology; totalFailures is ungrouped.
*/
@Schema(description = "Result of validating a document against a PDF/UA profile")
public record UaValidationResult(
String profile, boolean compliant, List<AccessibilityIssue> issues, int totalFailures) {
public boolean hasIssues() {
return !issues.isEmpty();
}
}
@@ -0,0 +1,23 @@
package stirling.software.proprietary.pdf.ua;
/** Artifact subtypes (ISO 32000-1 14.8.2.2). Artifacts are excluded from the structure tree. */
public enum ArtifactType {
/** Running heads, folios, page numbers. Required by PDF/UA-1 clause 7.8. */
PAGINATION("Pagination"),
/** Rules, boxes, and other layout ornamentation. */
LAYOUT("Layout"),
/** Cut marks and colour bars. */
PAGE("Page"),
/** Background graphics with no informational content. */
BACKGROUND("Background");
private final String subtype;
ArtifactType(String subtype) {
this.subtype = subtype;
}
public String subtype() {
return subtype;
}
}
@@ -0,0 +1,48 @@
package stirling.software.proprietary.pdf.ua;
/** An axis-aligned rectangle in PDF user space, with y increasing upwards. */
public record BBox(float x0, float y0, float x1, float y1) {
public static final BBox EMPTY = new BBox(0, 0, 0, 0);
public static BBox of(float x, float y, float width, float height) {
return new BBox(x, y, x + width, y + height);
}
public float width() {
return x1 - x0;
}
public float height() {
return y1 - y0;
}
public float centreX() {
return (x0 + x1) / 2f;
}
public BBox union(BBox other) {
if (other == null || other.isEmpty()) {
return this;
}
if (isEmpty()) {
return other;
}
return new BBox(
Math.min(x0, other.x0),
Math.min(y0, other.y0),
Math.max(x1, other.x1),
Math.max(y1, other.y1));
}
public boolean isEmpty() {
return x1 <= x0 || y1 <= y0;
}
/** Horizontal overlap with another box as a fraction of the narrower box's width. */
public float horizontalOverlap(BBox other) {
float overlap = Math.min(x1, other.x1) - Math.max(x0, other.x0);
float narrower = Math.min(width(), other.width());
return narrower <= 0 ? 0 : Math.max(0, overlap) / narrower;
}
}
@@ -0,0 +1,84 @@
package stirling.software.proprietary.pdf.ua;
import java.util.ArrayList;
import java.util.List;
import java.util.function.Consumer;
import lombok.Getter;
import lombok.Setter;
/** The derived logical structure of a document, ready for serialisation into a structure tree. */
@Getter
@Setter
public class DocumentStructure {
/** Top-level blocks in document reading order. */
private final List<StructBlock> blocks = new ArrayList<>();
/** Warnings raised during analysis, surfaced in the conversion report. */
private final List<String> warnings = new ArrayList<>();
private String title;
private String language;
/** True when real text was wrapped as artifacts, which blocks any conformance claim. */
private boolean textSuppressed;
/** Body text size used as the baseline for heading detection, in points. */
private float bodyFontSize;
public void add(StructBlock block) {
blocks.add(block);
}
public void warn(String message) {
if (!warnings.contains(message)) {
warnings.add(message);
}
}
public void visit(Consumer<StructBlock> visitor) {
blocks.forEach(block -> block.visit(visitor));
}
public int count(StructType type) {
int[] total = {0};
visit(
block -> {
if (block.getType() == type) {
total[0]++;
}
});
return total[0];
}
public int artifactCount() {
int[] total = {0};
visit(
block -> {
if (block.isArtifact()) {
total[0]++;
}
});
return total[0];
}
/** Figures with no alternative description, the most common PDF/UA failure. */
public List<StructBlock> figuresWithoutAlt() {
List<StructBlock> missing = new ArrayList<>();
visit(
block -> {
if ((block.getType() == StructType.FIGURE
|| block.getType() == StructType.FORMULA)
&& (block.getAlt() == null || block.getAlt().isBlank())
&& (block.getActualText() == null || block.getActualText().isBlank())) {
missing.add(block);
}
});
return missing;
}
public boolean isEmpty() {
return blocks.isEmpty();
}
}
@@ -0,0 +1,831 @@
package stirling.software.proprietary.pdf.ua;
import java.util.ArrayList;
import java.util.Collections;
import java.util.Comparator;
import java.util.HashMap;
import java.util.HashSet;
import java.util.IdentityHashMap;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.regex.Pattern;
import java.util.stream.Collectors;
import lombok.extern.slf4j.Slf4j;
/**
* Derives a logical structure from extracted lines and graphics, reusing {@code HeadingDetector}'s
* heuristics. Degrades to paragraphs rather than guessing, since a wrong tag misleads readers.
*/
@Slf4j
public class LayoutAnalyzer {
private static final Pattern BULLET = Pattern.compile("^[•‣◦⁃∙·▪●■o\\-\\*\\+]\\s+.*");
private static final Pattern ORDERED =
Pattern.compile("^(\\d{1,3}|[a-zA-Z]|[ivxlcIVXLC]{1,5})[\\.\\)]\\s+.*");
private static final Pattern PAGE_NUMBER =
Pattern.compile(
"^(page\\s+)?\\d{1,4}(\\s*(of|/)\\s*\\d{1,4})?$", Pattern.CASE_INSENSITIVE);
private static final Pattern DIGITS = Pattern.compile("\\d+");
/** Fraction of page height treated as the running head / foot band. */
private static final float MARGIN_BAND = 0.10f;
/** A line must exceed the body size by this ratio before it can be a heading. */
private static final float HEADING_RATIO = 1.10f;
/** Sizes within this many points are treated as the same heading tier. */
private static final float TIER_TOLERANCE = 0.4f;
private static final int MAX_HEADING_WORDS = 12;
/** Word gap beyond this multiple of the font size separates table cells. */
private static final float CELL_GAP_RATIO = 1.2f;
/** Images smaller than this in either dimension are decoration, not content. */
private static final float MIN_FIGURE_SIZE = 12f;
/** A size used by more than this share of lines is body text, however large the median says. */
private static final float MAX_HEADING_LINE_SHARE = 0.2f;
/** Consecutive lines sharing a size are a text block; headings appear alone. */
private static final int MAX_HEADING_RUN = 3;
/** A vector thinner than this in either dimension is a rule or border, not a drawing. */
private static final float MIN_VECTOR_THICKNESS = 3f;
/** Vector clusters smaller than this are ornament; larger ones are probably a chart. */
private static final float MIN_VECTOR_FIGURE_SIZE = 40f;
/** A drawing is built from several strokes; one big rectangle is a panel, not a chart. */
private static final int MIN_VECTOR_FIGURE_OPS = 4;
/** More text than this inside the region means shading behind content, not a drawing. */
private static final int MAX_LINES_INSIDE_FIGURE = 2;
public DocumentStructure analyse(List<PageContent> pages) {
DocumentStructure structure = new DocumentStructure();
float bodySize = bodyFontSize(pages);
structure.setBodyFontSize(bodySize);
Map<Float, Integer> tiers = headingTiers(pages, bodySize);
Map<Integer, List<TextLineInfo>> artifactLines = repeatedMarginLines(pages, bodySize);
for (PageContent page : pages) {
analysePage(
page,
structure,
bodySize,
tiers,
artifactLines.getOrDefault(page.pageIndex(), List.of()));
}
List<Integer> suppressedPages =
pages.stream()
.filter(PageContent::linesDropped)
.map(PageContent::pageIndex)
.toList();
if (!suppressedPages.isEmpty()) {
structure.setTextSuppressed(true);
structure.warn(
"Text on page(s) "
+ suppressedPages.stream()
.map(i -> String.valueOf(i + 1))
.collect(Collectors.joining(", "))
+ " could not be tagged reliably and was marked as artifacts. The"
+ " converter will not claim conformance while real text is hidden"
+ " from assistive technology.");
}
normaliseHeadingLevels(structure);
structure.setTitle(deriveTitle(structure));
return structure;
}
// --- Document-wide statistics -----------------------------------------
/** Character-weighted median line size, which is far more stable than a plain median. */
static float bodyFontSize(List<PageContent> pages) {
Map<Float, Integer> weights = new HashMap<>();
for (PageContent page : pages) {
for (TextLineInfo line : page.lines()) {
if (line.dominantFontSize() > 0 && !line.isBlank()) {
weights.merge(line.dominantFontSize(), line.charCount(), Integer::sum);
}
}
}
if (weights.isEmpty()) {
return 0f;
}
int total = weights.values().stream().mapToInt(Integer::intValue).sum();
List<Map.Entry<Float, Integer>> sorted =
weights.entrySet().stream().sorted(Map.Entry.comparingByKey()).toList();
int seen = 0;
for (Map.Entry<Float, Integer> entry : sorted) {
seen += entry.getValue();
if (seen >= total / 2) {
return entry.getKey();
}
}
return sorted.get(sorted.size() - 1).getKey();
}
/** Maps each distinct heading size to a 1-based level, largest size first. */
static Map<Float, Integer> headingTiers(List<PageContent> pages, float bodySize) {
if (bodySize <= 0) {
return Map.of();
}
// A size used by a large share of the lines is body text, whatever the median says.
Map<Float, Integer> lineCounts = new HashMap<>();
int totalLines = 0;
for (PageContent page : pages) {
for (TextLineInfo line : page.lines()) {
if (!line.isBlank()) {
lineCounts.merge(line.dominantFontSize(), 1, Integer::sum);
totalLines++;
}
}
}
int headingLineCeiling = Math.max(1, (int) (totalLines * MAX_HEADING_LINE_SHARE));
// Headings do not cluster; a run of same-size lines is a text block, not headings.
Map<Float, Integer> longestRun = new HashMap<>();
for (PageContent page : pages) {
Float runSize = null;
int runLength = 0;
for (TextLineInfo line : page.lines()) {
if (line.isBlank()) {
continue;
}
float size = line.dominantFontSize();
if (runSize != null && Float.compare(size, runSize) == 0) {
runLength++;
} else {
runSize = size;
runLength = 1;
}
int seen = longestRun.getOrDefault(size, 0);
if (runLength > seen) {
longestRun.put(size, runLength);
}
}
}
List<Float> sizes = new ArrayList<>();
for (PageContent page : pages) {
for (TextLineInfo line : page.lines()) {
if (isHeadingCandidate(line)
&& line.dominantFontSize() > bodySize * HEADING_RATIO
&& lineCounts.getOrDefault(line.dominantFontSize(), 0) <= headingLineCeiling
&& longestRun.getOrDefault(line.dominantFontSize(), 0) < MAX_HEADING_RUN) {
sizes.add(line.dominantFontSize());
}
}
}
List<Float> distinct = sizes.stream().distinct().sorted(Comparator.reverseOrder()).toList();
Map<Float, Integer> tiers = new LinkedHashMap<>();
int level = 0;
Float previous = null;
for (Float size : distinct) {
if (previous == null || previous - size > TIER_TOLERANCE) {
level = Math.min(level + 1, 6);
previous = size;
}
tiers.put(size, level);
}
return tiers;
}
/**
* Claims a line's operators word run by word run; claiming the whole ordinal interval would
* swallow anything drawn between them, an image included.
*/
private static void claimLine(StructBlock block, TextLineInfo line) {
// Sort by ordinal, not position: merging out-of-order runs silently drops them to
// /Artifact, hiding them from assistive technology while the file still validates.
List<WordInfo> words =
line.words().stream()
.filter(w -> !w.isBlank())
.sorted(Comparator.comparingInt(WordInfo::startOrdinal))
.toList();
if (words.isEmpty()) {
block.addRange(line.startOrdinal(), line.endOrdinal());
return;
}
int start = words.get(0).startOrdinal();
int end = words.get(0).endOrdinal();
for (int i = 1; i < words.size(); i++) {
WordInfo word = words.get(i);
if (word.startOrdinal() <= end + 1) {
end = Math.max(end, word.endOrdinal());
} else {
block.addRange(start, end);
start = word.startOrdinal();
end = word.endOrdinal();
}
}
block.addRange(start, end);
}
static boolean isHeadingCandidate(TextLineInfo line) {
String text = line.text().strip();
if (text.isEmpty() || line.wordCount() > MAX_HEADING_WORDS) {
return false;
}
char last = text.charAt(text.length() - 1);
return last != '.' && last != '!' && last != '?';
}
/**
* Finds lines in the head/foot bands whose text repeats across pages. Digits are masked first
* so that "Page 4" and "Page 5" count as the same running foot.
*/
static Map<Integer, List<TextLineInfo>> repeatedMarginLines(List<PageContent> pages) {
return repeatedMarginLines(pages, bodyFontSize(pages));
}
static Map<Integer, List<TextLineInfo>> repeatedMarginLines(
List<PageContent> pages, float bodySize) {
Map<Integer, List<TextLineInfo>> result = new HashMap<>();
if (pages.isEmpty()) {
return result;
}
Map<String, Integer> counts = new HashMap<>();
Map<Integer, List<TextLineInfo>> candidates = new HashMap<>();
for (PageContent page : pages) {
float height = page.mediaBox().height();
if (height <= 0) {
continue;
}
float topEdge = page.mediaBox().y1() - height * MARGIN_BAND;
float bottomEdge = page.mediaBox().y0() + height * MARGIN_BAND;
List<TextLineInfo> inBand = new ArrayList<>();
for (TextLineInfo line : page.lines()) {
if (line.bbox().y0() >= topEdge || line.bbox().y1() <= bottomEdge) {
inBand.add(line);
counts.merge(mask(line.text()), 1, Integer::sum);
}
}
candidates.put(page.pageIndex(), inBand);
}
int threshold = Math.max(2, pages.size() / 2);
for (Map.Entry<Integer, List<TextLineInfo>> entry : candidates.entrySet()) {
List<TextLineInfo> artifacts = new ArrayList<>();
for (TextLineInfo line : entry.getValue()) {
boolean repeats =
pages.size() >= 3 && counts.getOrDefault(mask(line.text()), 0) >= threshold;
boolean pageNumber = PAGE_NUMBER.matcher(line.text().strip()).matches();
// Masked digits merge "Section 1" and "Section 2"; size is the tie-break that stops
// a real heading being demoted, as running heads are never larger than body text.
boolean looksLikeChrome =
bodySize <= 0 || line.dominantFontSize() <= bodySize * 1.05f;
if (pageNumber || (repeats && looksLikeChrome)) {
artifacts.add(line);
}
}
result.put(entry.getKey(), artifacts);
}
return result;
}
private static String mask(String text) {
return DIGITS.matcher(text.strip().toLowerCase()).replaceAll("#").replaceAll("\\s+", " ");
}
// --- Per-page analysis -------------------------------------------------
private void analysePage(
PageContent page,
DocumentStructure structure,
float bodySize,
Map<Float, Integer> tiers,
List<TextLineInfo> marginArtifacts) {
for (TextLineInfo line : marginArtifacts) {
StructBlock artifact = StructBlock.artifact(ArtifactType.PAGINATION, page.pageIndex());
claimLine(artifact, line);
artifact.setBbox(line.bbox());
artifact.setText(line.text());
structure.add(artifact);
}
// Identity set, not List.contains: TextLineInfo is a record whose equals walks its word
// list, so a linear scan per line is quadratic with a deep comparison inside it.
java.util.Set<TextLineInfo> marginSet = Collections.newSetFromMap(new IdentityHashMap<>());
marginSet.addAll(marginArtifacts);
List<TextLineInfo> body =
page.lines().stream()
.filter(line -> !line.isBlank() && !marginSet.contains(line))
.sorted(readingOrder(page))
.toList();
List<StructBlock> blocks = new ArrayList<>();
int index = 0;
while (index < body.size()) {
TextLineInfo line = body.get(index);
int tableEnd = tableRunEnd(body, index);
if (tableEnd > index) {
StructBlock table = buildTable(body.subList(index, tableEnd + 1), page.pageIndex());
if (table != null) {
blocks.add(table);
index = tableEnd + 1;
continue;
}
}
int listEnd = listRunEnd(body, index);
if (listEnd > index) {
blocks.add(buildList(body.subList(index, listEnd + 1), page.pageIndex()));
index = listEnd + 1;
continue;
}
Integer level = headingLevel(line, tiers);
if (level != null) {
StructBlock heading = new StructBlock(StructType.heading(level), page.pageIndex());
claimLine(heading, line);
heading.setBbox(line.bbox());
heading.setText(line.text());
blocks.add(heading);
index++;
continue;
}
int paragraphEnd = paragraphRunEnd(body, index, tiers, bodySize);
blocks.add(buildParagraph(body.subList(index, paragraphEnd + 1), page.pageIndex()));
index = paragraphEnd + 1;
}
// Form XObject text is attributed to its Do, so a Figure too would double-claim it.
Set<Integer> claimed = new HashSet<>();
for (StructBlock block : blocks) {
block.visit(
node ->
node.getRanges()
.forEach(
range -> {
for (int i = range.start(); i <= range.end(); i++) {
claimed.add(i);
}
}));
}
blocks.addAll(buildGraphics(page, structure, claimed));
blocks.forEach(structure::add);
}
/**
* Orders lines top-to-bottom, splitting into columns first when the page is clearly
* multi-column. Without this, a two-column page reads as interleaved half-sentences.
*/
private Comparator<TextLineInfo> readingOrder(PageContent page) {
Float gutter = detectGutter(page);
if (gutter == null) {
return Comparator.comparingDouble((TextLineInfo l) -> -l.bbox().y1())
.thenComparingDouble(l -> l.bbox().x0());
}
return Comparator.comparingInt((TextLineInfo l) -> l.bbox().centreX() < gutter ? 0 : 1)
.thenComparingDouble(l -> -l.bbox().y1())
.thenComparingDouble(l -> l.bbox().x0());
}
/**
* Returns the x of a vertical gutter when the page is two-column, else null. A gutter must sit
* near the middle, be crossed by almost no line, and have substantial text on both sides.
*/
static Float detectGutter(PageContent page) {
List<TextLineInfo> lines = page.lines().stream().filter(line -> !line.isBlank()).toList();
if (lines.size() < 8) {
return null;
}
float pageWidth = page.mediaBox().width();
if (pageWidth <= 0) {
return null;
}
float centre = page.mediaBox().x0() + pageWidth / 2f;
long crossing =
lines.stream()
.filter(
line ->
line.bbox().x0() < centre - 5
&& line.bbox().x1() > centre + 5)
.count();
if (crossing > lines.size() * 0.1) {
return null;
}
long left = lines.stream().filter(line -> line.bbox().centreX() < centre).count();
long right = lines.size() - left;
boolean balanced = left > lines.size() * 0.25 && right > lines.size() * 0.25;
return balanced ? centre : null;
}
private static Integer headingLevel(TextLineInfo line, Map<Float, Integer> tiers) {
if (!isHeadingCandidate(line)) {
return null;
}
return tiers.get(line.dominantFontSize());
}
// --- Paragraphs --------------------------------------------------------
private static int paragraphRunEnd(
List<TextLineInfo> lines, int start, Map<Float, Integer> tiers, float bodySize) {
int end = start;
for (int i = start + 1; i < lines.size(); i++) {
TextLineInfo previous = lines.get(i - 1);
TextLineInfo current = lines.get(i);
if (headingLevel(current, tiers) != null || startsListItem(current)) {
break;
}
float gap = previous.bbox().y0() - current.bbox().y1();
float leading = Math.max(bodySize, current.bbox().height());
boolean sameBlock = gap < leading * 0.8f && gap > -leading;
boolean sentenceEnded = endsSentence(previous.text());
if (!sameBlock || (sentenceEnded && gap > leading * 0.4f)) {
break;
}
end = i;
}
return end;
}
private static boolean endsSentence(String text) {
String stripped = text.strip();
if (stripped.isEmpty()) {
return false;
}
char last = stripped.charAt(stripped.length() - 1);
return last == '.' || last == '!' || last == '?';
}
private static StructBlock buildParagraph(List<TextLineInfo> lines, int pageIndex) {
StructBlock paragraph = new StructBlock(StructType.P, pageIndex);
BBox box = BBox.EMPTY;
StringBuilder text = new StringBuilder();
for (TextLineInfo line : lines) {
claimLine(paragraph, line);
box = box.union(line.bbox());
if (text.length() > 0) {
text.append(' ');
}
text.append(line.text().strip());
}
paragraph.setBbox(box);
paragraph.setText(text.toString());
return paragraph;
}
// --- Lists -------------------------------------------------------------
static boolean startsListItem(TextLineInfo line) {
String text = line.text().strip();
return BULLET.matcher(text).matches() || ORDERED.matcher(text).matches();
}
private static int listRunEnd(List<TextLineInfo> lines, int start) {
if (!startsListItem(lines.get(start))) {
return start;
}
float indent = lines.get(start).bbox().x0();
int end = start;
for (int i = start + 1; i < lines.size(); i++) {
TextLineInfo line = lines.get(i);
boolean isItem = startsListItem(line) && Math.abs(line.bbox().x0() - indent) < 6f;
boolean isContinuation = !startsListItem(line) && line.bbox().x0() > indent + 2f;
if (!isItem && !isContinuation) {
break;
}
end = i;
}
// A single marker is a stray character, not a list.
long items =
lines.subList(start, end + 1).stream()
.filter(LayoutAnalyzer::startsListItem)
.count();
return items >= 2 ? end : start;
}
private static StructBlock buildList(List<TextLineInfo> lines, int pageIndex) {
StructBlock list = new StructBlock(StructType.L, pageIndex);
list.setListNumbering(listNumbering(lines.get(0)));
BBox box = BBox.EMPTY;
StructBlock currentBody = null;
for (TextLineInfo line : lines) {
box = box.union(line.bbox());
if (startsListItem(line) || currentBody == null) {
StructBlock item = new StructBlock(StructType.LI, pageIndex);
StructBlock body = new StructBlock(StructType.LBODY, pageIndex);
claimLine(body, line);
body.setBbox(line.bbox());
body.setText(line.text());
item.addChild(body);
item.setBbox(line.bbox());
list.addChild(item);
currentBody = body;
} else {
claimLine(currentBody, line);
currentBody.setBbox(currentBody.getBbox().union(line.bbox()));
currentBody.setText(currentBody.getText() + " " + line.text().strip());
}
}
list.setBbox(box);
return list;
}
private static String listNumbering(TextLineInfo first) {
String text = first.text().strip();
if (BULLET.matcher(text).matches()) {
return "Disc";
}
char c = text.charAt(0);
if (Character.isDigit(c)) {
return "Decimal";
}
if ("ivxlc".indexOf(Character.toLowerCase(c)) >= 0 && text.length() > 1) {
return Character.isUpperCase(c) ? "UpperRoman" : "LowerRoman";
}
return Character.isUpperCase(c) ? "UpperAlpha" : "LowerAlpha";
}
// --- Tables ------------------------------------------------------------
/** Splits a line into cells wherever the gap between words exceeds the cell threshold. */
static List<List<WordInfo>> splitCells(TextLineInfo line) {
List<WordInfo> words = line.words().stream().filter(w -> !w.isBlank()).toList();
List<List<WordInfo>> cells = new ArrayList<>();
if (words.isEmpty()) {
return cells;
}
float threshold = Math.max(line.dominantFontSize(), 1f) * CELL_GAP_RATIO;
List<WordInfo> current = new ArrayList<>();
current.add(words.get(0));
for (int i = 1; i < words.size(); i++) {
float gap = words.get(i).bbox().x0() - words.get(i - 1).bbox().x1();
if (gap > threshold) {
cells.add(List.copyOf(current));
current = new ArrayList<>();
}
current.add(words.get(i));
}
cells.add(List.copyOf(current));
return cells;
}
/**
* Index of the last line of a table run starting at {@code start}, or {@code start} if none.
*/
private static int tableRunEnd(List<TextLineInfo> lines, int start) {
int end = start;
for (int i = start; i < lines.size(); i++) {
if (splitCells(lines.get(i)).size() < 2) {
break;
}
end = i;
}
return end > start ? end : start;
}
/**
* Builds a Table when the run really looks tabular and each cell owns its own operators.
* Returns null when it does not, so the caller falls back to paragraphs.
*/
private static StructBlock buildTable(List<TextLineInfo> rows, int pageIndex) {
if (rows.size() < 2) {
return null;
}
List<List<List<WordInfo>>> grid = new ArrayList<>();
for (TextLineInfo row : rows) {
if (!row.wordsAreSeparable()) {
log.debug("Table row shares operators between cells; falling back to paragraphs");
return null;
}
grid.add(splitCells(row));
}
int columns = grid.get(0).size();
long consistent = grid.stream().filter(row -> row.size() == columns).count();
if (columns < 2 || consistent < Math.max(2, grid.size() * 0.6)) {
return null;
}
boolean headerRow = looksLikeHeader(rows, grid);
StructBlock table = new StructBlock(StructType.TABLE, pageIndex);
BBox box = BBox.EMPTY;
for (int r = 0; r < grid.size(); r++) {
List<List<WordInfo>> cells = grid.get(r);
if (cells.size() != columns) {
continue;
}
StructBlock tr = new StructBlock(StructType.TR, pageIndex);
boolean isHeader = headerRow && r == 0;
for (List<WordInfo> cell : cells) {
StructBlock td =
new StructBlock(isHeader ? StructType.TH : StructType.TD, pageIndex);
if (isHeader) {
td.setScope("Column");
}
BBox cellBox = BBox.EMPTY;
StringBuilder text = new StringBuilder();
int from = cell.get(0).startOrdinal();
int to = cell.get(cell.size() - 1).endOrdinal();
for (WordInfo word : cell) {
cellBox = cellBox.union(word.bbox());
if (text.length() > 0) {
text.append(' ');
}
text.append(word.text());
}
td.addRange(from, to);
td.setBbox(cellBox);
td.setText(text.toString());
tr.addChild(td);
box = box.union(cellBox);
}
tr.setBbox(box);
table.addChild(tr);
}
table.setBbox(box);
if (table.getChildren().size() < 2) {
return null;
}
// Clause 7.5 needs equal cell counts per row; a ragged table fails validation outright.
long distinctWidths =
table.getChildren().stream()
.map(row -> row.getChildren().size())
.distinct()
.count();
if (distinctWidths != 1) {
log.debug("Discarding a table whose rows have different cell counts");
return null;
}
return table;
}
/** The first row is a header when it is bold, or when only later rows carry numbers. */
private static boolean looksLikeHeader(
List<TextLineInfo> rows, List<List<List<WordInfo>>> grid) {
if (rows.get(0).bold()) {
return true;
}
boolean firstHasDigits = DIGITS.matcher(rows.get(0).text()).find();
boolean laterHasDigits =
rows.subList(1, rows.size()).stream()
.anyMatch(row -> DIGITS.matcher(row.text()).find());
return !firstHasDigits && laterHasDigits;
}
// --- Graphics ----------------------------------------------------------
private List<StructBlock> buildGraphics(
PageContent page, DocumentStructure structure, java.util.Set<Integer> claimed) {
List<StructBlock> blocks = new ArrayList<>();
boolean warnedForms = false;
// Vectors cluster: a chart is many strokes in one region, a rule is a single thin one.
java.util.Set<Integer> vectorFigureOrdinals = vectorFigureOrdinals(page, claimed);
for (MarkableOp op : page.ops()) {
if (op.kind() == MarkableOp.Kind.TEXT || claimed.contains(op.ordinal())) {
continue;
}
BBox box = op.bbox();
if (op.kind() == MarkableOp.Kind.VECTOR) {
StructBlock block;
if (vectorFigureOrdinals.contains(op.ordinal())) {
block = new StructBlock(StructType.FIGURE, page.pageIndex());
} else {
block = StructBlock.artifact(ArtifactType.LAYOUT, page.pageIndex());
}
block.addRange(op.ordinal(), op.ordinal());
block.setBbox(box);
blocks.add(block);
continue;
}
boolean decorative = box.width() < MIN_FIGURE_SIZE || box.height() < MIN_FIGURE_SIZE;
if (decorative) {
StructBlock artifact = StructBlock.artifact(ArtifactType.LAYOUT, page.pageIndex());
artifact.addRange(op.ordinal(), op.ordinal());
artifact.setBbox(box);
blocks.add(artifact);
continue;
}
if (op.kind() == MarkableOp.Kind.FORM && !warnedForms) {
structure.warn(
"Content inside form XObjects was tagged as a single region because its"
+ " text is not separately addressable; review those areas.");
warnedForms = true;
}
StructBlock figure = new StructBlock(StructType.FIGURE, page.pageIndex());
figure.addRange(op.ordinal(), op.ordinal());
figure.setBbox(box);
blocks.add(figure);
}
return blocks;
}
/**
* Finds vector operators belonging to a substantial drawing rather than page furniture; thin
* paths are rules and table borders, and a short run is ornament.
*/
private static Set<Integer> vectorFigureOrdinals(
PageContent page, java.util.Set<Integer> claimed) {
// A chart's plot area is mostly empty, while shading sits behind the text it decorates.
Set<Integer> result = new HashSet<>();
List<MarkableOp> run = new ArrayList<>();
BBox extent = BBox.EMPTY;
for (MarkableOp op : page.ops()) {
boolean substantial =
op.kind() == MarkableOp.Kind.VECTOR
&& !claimed.contains(op.ordinal())
&& !op.bbox().isEmpty()
&& op.bbox().width() >= MIN_VECTOR_THICKNESS
&& op.bbox().height() >= MIN_VECTOR_THICKNESS;
if (substantial) {
run.add(op);
extent = extent.isEmpty() ? op.bbox() : extent.union(op.bbox());
continue;
}
flushVectorRun(run, extent, page.lines(), result);
run = new ArrayList<>();
extent = BBox.EMPTY;
}
flushVectorRun(run, extent, page.lines(), result);
return result;
}
private static void flushVectorRun(
List<MarkableOp> run,
BBox extent,
List<TextLineInfo> lines,
java.util.Set<Integer> result) {
if (run.size() < MIN_VECTOR_FIGURE_OPS
|| extent.width() < MIN_VECTOR_FIGURE_SIZE
|| extent.height() < MIN_VECTOR_FIGURE_SIZE) {
return;
}
if (overlappingLines(extent, lines) > MAX_LINES_INSIDE_FIGURE) {
return;
}
run.forEach(op -> result.add(op.ordinal()));
}
/** How many text lines sit within the region a vector cluster covers. */
private static int overlappingLines(BBox extent, List<TextLineInfo> lines) {
int count = 0;
for (TextLineInfo line : lines) {
BBox box = line.bbox();
boolean inside =
box.x0() >= extent.x0() - 2
&& box.x1() <= extent.x1() + 2
&& box.y0() >= extent.y0() - 2
&& box.y1() <= extent.y1() + 2;
if (inside) {
count++;
}
}
return count;
}
// --- Post-processing ---------------------------------------------------
/**
* Rewrites heading levels so no level is skipped, which PDF/UA-1 clause 7.4 requires. A
* document that jumps H1 to H3 is remapped to H1, H2 while preserving relative depth.
*/
static void normaliseHeadingLevels(DocumentStructure structure) {
List<StructBlock> headings = new ArrayList<>();
structure.visit(
block -> {
if (block.getType().isHeading()) {
headings.add(block);
}
});
int previous = 0;
for (StructBlock heading : headings) {
int level = heading.getType().headingLevel();
int adjusted = level > previous + 1 ? previous + 1 : level;
heading.setType(StructType.heading(adjusted));
previous = adjusted;
}
}
/** Uses the first top-level heading as the title when the document has no metadata title. */
private static String deriveTitle(DocumentStructure structure) {
for (StructBlock block : structure.getBlocks()) {
if (block.getType().isHeading() && !block.getText().isBlank()) {
return block.getText().strip();
}
}
return null;
}
}
@@ -0,0 +1,44 @@
package stirling.software.proprietary.pdf.ua;
/**
* One operator in a page content stream that may be wrapped in a marked-content sequence. The
* ordinal counts only markable operators, joining text extraction to token rewriting.
*/
public record MarkableOp(int ordinal, Kind kind, BBox bbox, String resourceName) {
public enum Kind {
/** Tj, TJ, ' or " */
TEXT,
/** Do referencing an image XObject */
IMAGE,
/** Do referencing a form XObject */
FORM,
/** BI ... ID ... EI */
INLINE_IMAGE,
/** A path-painting or shading operator: rules, borders, fills, logos */
VECTOR;
public boolean isGraphic() {
return this == IMAGE || this == INLINE_IMAGE;
}
}
/**
* Operator names counted as markable; both passes must agree on this set. Path painting is
* included because clause 7.1 needs visible rules and borders tagged or artifacted.
*/
public static boolean isMarkableOperator(String name) {
return switch (name) {
case "Tj", "TJ", "'", "\"", "Do", "BI" -> true;
default -> isPathPainting(name);
};
}
/** Painting operators only: {@code n} ends a path without marking the page. */
public static boolean isPathPainting(String name) {
return switch (name) {
case "S", "s", "f", "F", "f*", "B", "B*", "b", "b*", "sh" -> true;
default -> false;
};
}
}
@@ -0,0 +1,284 @@
package stirling.software.proprietary.pdf.ua;
import java.io.IOException;
import java.io.OutputStream;
import java.util.ArrayDeque;
import java.util.ArrayList;
import java.util.Deque;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import org.apache.pdfbox.contentstream.operator.Operator;
import org.apache.pdfbox.cos.COSBase;
import org.apache.pdfbox.cos.COSDictionary;
import org.apache.pdfbox.cos.COSInteger;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdfparser.PDFStreamParser;
import org.apache.pdfbox.pdfwriter.ContentStreamWriter;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.common.PDStream;
import lombok.extern.slf4j.Slf4j;
/**
* Rewrites a page stream so every markable operator sits inside a marked-content sequence: claimed
* content gets an MCID, everything else /Artifact, satisfying PDF/UA-1 clause 7.1 by construction.
*/
@Slf4j
public class MarkedContentInjector {
private static final COSName ARTIFACT = COSName.getPDFName("Artifact");
private static final COSName MCID = COSName.getPDFName("MCID");
private static final COSName ACTUAL_TEXT = COSName.getPDFName("ActualText");
private static final COSName ALT = COSName.getPDFName("Alt");
/** Operators that force an open sequence to close so nesting stays legal. */
private static boolean isBoundary(String name) {
return "BT".equals(name) || "ET".equals(name) || "q".equals(name) || "Q".equals(name);
}
private static boolean isMarkedContentOperator(String name) {
return "BDC".equals(name) || "BMC".equals(name) || "EMC".equals(name);
}
/**
* Path-construction operators; ISO 32000-1 forbids marked content inside a path object, so a
* sequence wrapping a fill or stroke must open before the path starts.
*/
private static boolean isPathConstruction(String name) {
return switch (name) {
case "m", "l", "c", "v", "y", "h", "re" -> true;
default -> false;
};
}
private static boolean opensMarkedContent(String name) {
return "BDC".equals(name) || "BMC".equals(name);
}
/**
* True for an optional-content sequence; stripping an {@code /OC} wrapper would make hidden
* layers such as watermarks or redaction overlays visible.
*/
private static boolean isOptionalContent(String name, List<COSBase> operands) {
return opensMarkedContent(name)
&& !operands.isEmpty()
&& operands.get(0) instanceof COSName tag
&& "OC".equals(tag.getName());
}
/**
* True when a sequence supplies replacement text for its glyphs; dropping it leaves a screen
* reader with the font's own mapping, which for a ligature says nothing useful.
*/
private static boolean carriesReplacementText(String name, List<COSBase> operands) {
if (!opensMarkedContent(name)) {
return false;
}
for (COSBase operand : operands) {
if (operand instanceof COSDictionary properties
&& (properties.containsKey(ACTUAL_TEXT)
|| properties.containsKey(ALT)
|| properties.containsKey(COSName.E))) {
return true;
}
}
return false;
}
/** The source's own ids mean nothing once the tree is rebuilt, so they are dropped. */
private static void stripStaleMcid(List<COSBase> operands) {
for (COSBase operand : operands) {
if (operand instanceof COSDictionary properties) {
properties.removeItem(MCID);
}
}
}
/** Wraps every markable operator on the page; returns the next unused marked content id. */
public int inject(
PDDocument document,
PDPage page,
List<StructBlock> blocks,
int nextMcid,
boolean stripExisting)
throws IOException {
Map<Integer, StructBlock> owners = ownersByOrdinal(blocks);
List<Object> tokens = parse(page);
List<Object> output = new ArrayList<>(tokens.size() + owners.size() * 4);
List<COSBase> operands = new ArrayList<>();
// Tracks, for each surviving source sequence, whether its closer should be kept.
Deque<Boolean> keptSequences = new ArrayDeque<>();
StructBlock openBlock = null;
boolean open = false;
int ordinal = -1;
int mcid = nextMcid;
int pathStart = -1;
for (Object token : tokens) {
if (!(token instanceof Operator operator)) {
operands.add((COSBase) token);
continue;
}
String name = operator.getName();
if (stripExisting && isMarkedContentOperator(name)) {
boolean keep;
if (opensMarkedContent(name)) {
keep =
isOptionalContent(name, operands)
|| carriesReplacementText(name, operands);
if (keep) {
stripStaleMcid(operands);
}
keptSequences.push(keep);
} else {
// A closer is kept exactly when its matching opener was.
keep = !keptSequences.isEmpty() && keptSequences.pop();
}
if (!keep) {
operands.clear();
continue;
}
// Close our own sequence first so the two never interleave illegally.
if (open) {
output.add(Operator.getOperator("EMC"));
open = false;
openBlock = null;
}
output.addAll(operands);
output.add(operator);
operands.clear();
continue;
}
if (isBoundary(name) && open) {
output.add(Operator.getOperator("EMC"));
open = false;
openBlock = null;
}
// Remember where the current path object began so a sequence wrapping its painting
// operator can be opened before it rather than inside it.
if (isPathConstruction(name)) {
if (pathStart < 0) {
pathStart = output.size();
}
} else if (!MarkableOp.isPathPainting(name) && !"n".equals(name)) {
pathStart = -1;
}
if (MarkableOp.isMarkableOperator(name)) {
ordinal++;
StructBlock owner = owners.get(ordinal);
if (!open || owner != openBlock) {
boolean insidePath = MarkableOp.isPathPainting(name) && pathStart >= 0;
if (open) {
// Close before the path began, so the EMC also stays outside the path.
output.add(
insidePath ? pathStart : output.size(),
Operator.getOperator("EMC"));
if (insidePath) {
pathStart++;
}
}
int at = insidePath ? pathStart : output.size();
mcid = openSequenceAt(output, at, owner, mcid);
open = true;
openBlock = owner;
}
}
output.addAll(operands);
output.add(operator);
operands.clear();
if (MarkableOp.isPathPainting(name) || "n".equals(name)) {
pathStart = -1;
}
}
if (open) {
output.add(Operator.getOperator("EMC"));
}
write(document, page, output);
return mcid;
}
/** Emits the opening BDC/BMC at a given position and records the id on the owning block. */
private int openSequenceAt(List<Object> output, int at, StructBlock owner, int mcid) {
List<Object> opening = new ArrayList<>(3);
if (owner == null) {
opening.add(ARTIFACT);
opening.add(Operator.getOperator("BMC"));
} else if (owner.isArtifact()) {
COSDictionary properties = new COSDictionary();
if (owner.getArtifactType() != null) {
properties.setName(COSName.TYPE, owner.getArtifactType().subtype());
}
opening.add(ARTIFACT);
opening.add(properties);
opening.add(Operator.getOperator("BDC"));
} else {
COSDictionary properties = new COSDictionary();
properties.setItem(MCID, COSInteger.get(mcid));
opening.add(COSName.getPDFName(owner.getType().tag()));
opening.add(properties);
opening.add(Operator.getOperator("BDC"));
owner.getMcids().add(mcid);
mcid++;
}
output.addAll(at, opening);
return mcid;
}
/**
* Maps each claimed ordinal to its block. Overlapping claims are dropped rather than merged:
* two structure elements sharing content would make the reading order ambiguous.
*/
static Map<Integer, StructBlock> ownersByOrdinal(List<StructBlock> blocks) {
Map<Integer, StructBlock> owners = new HashMap<>();
for (StructBlock block : blocks) {
block.visit(
node -> {
for (StructBlock.OrdinalRange range : node.getRanges()) {
for (int i = range.start(); i <= range.end(); i++) {
StructBlock existing = owners.putIfAbsent(i, node);
if (existing != null && existing != node) {
log.debug(
"Ordinal {} claimed by both {} and {}; keeping the first",
i,
existing,
node);
}
}
}
});
}
return owners;
}
private static List<Object> parse(PDPage page) throws IOException {
PDFStreamParser parser = new PDFStreamParser(page);
List<Object> tokens = new ArrayList<>();
Object token;
while ((token = parser.parseNextToken()) != null) {
tokens.add(token);
}
return tokens;
}
private static void write(PDDocument document, PDPage page, List<Object> tokens)
throws IOException {
PDStream stream = new PDStream(document);
try (OutputStream out = stream.createOutputStream(COSName.FLATE_DECODE)) {
new ContentStreamWriter(out).writeTokens(tokens);
}
page.setContents(stream);
}
}
@@ -0,0 +1,33 @@
package stirling.software.proprietary.pdf.ua;
import java.util.List;
/**
* Everything the layout analyser needs about one page. carriesTextSemantics: existing marked
* content has ActualText/Alt/expansion a rebuild would discard. linesDropped: text became
* artifacts.
*/
public record PageContent(
int pageIndex,
List<TextLineInfo> lines,
List<MarkableOp> ops,
int markableCount,
boolean preExistingMarkedContent,
boolean carriesTextSemantics,
boolean linesDropped,
BBox mediaBox) {
public boolean hasText() {
return lines.stream().anyMatch(line -> !line.isBlank());
}
/** Markable operators that draw graphics rather than text. */
public List<MarkableOp> graphics() {
return ops.stream().filter(op -> op.kind().isGraphic()).toList();
}
/** Form XObject invocations, which are tagged as a unit because their text is opaque here. */
public List<MarkableOp> forms() {
return ops.stream().filter(op -> op.kind() == MarkableOp.Kind.FORM).toList();
}
}
@@ -0,0 +1,47 @@
package stirling.software.proprietary.pdf.ua;
import org.apache.xmpbox.XMPMetadata;
import org.apache.xmpbox.schema.XMPSchema;
import org.apache.xmpbox.type.IntegerType;
import org.apache.xmpbox.type.StructuredType;
/**
* The {@code pdfuaid} XMP conformance schema, which XMPBox does not ship. Only write it once
* validation has passed - it is a compliance claim.
*/
@StructuredType(
preferedPrefix = PdfUaIdentificationSchema.PREFERRED_PREFIX,
namespace = PdfUaIdentificationSchema.NAMESPACE)
public class PdfUaIdentificationSchema extends XMPSchema {
public static final String PREFERRED_PREFIX = "pdfuaid";
public static final String NAMESPACE = "http://www.aiim.org/pdfua/ns/id/";
public static final String PART = "part";
public static final String REV = "rev";
public PdfUaIdentificationSchema(XMPMetadata metadata) {
super(metadata);
}
public PdfUaIdentificationSchema(XMPMetadata metadata, String prefix) {
super(metadata, prefix);
}
/** Sets {@code pdfuaid:part}, the conformance level (1 or 2). */
public void setPart(int part) {
addProperty(new IntegerType(getMetadata(), getNamespace(), getPrefix(), PART, part));
}
/** Sets {@code pdfuaid:rev}, the four-digit revision year used by PDF/UA-2. */
public void setRevision(int year) {
addProperty(new IntegerType(getMetadata(), getNamespace(), getPrefix(), REV, year));
}
public Integer getPart() {
if (getProperty(PART) instanceof IntegerType part) {
return part.getValue();
}
return null;
}
}
@@ -0,0 +1,224 @@
package stirling.software.proprietary.pdf.ua;
import java.io.ByteArrayInputStream;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.util.ArrayList;
import java.util.List;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDDocumentCatalog;
import org.apache.pdfbox.pdmodel.PDDocumentInformation;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.common.PDMetadata;
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
import org.apache.pdfbox.pdmodel.interactive.form.PDField;
import org.apache.pdfbox.pdmodel.interactive.viewerpreferences.PDViewerPreferences;
import org.apache.xmpbox.XMPMetadata;
import org.apache.xmpbox.schema.DublinCoreSchema;
import org.apache.xmpbox.schema.XMPSchema;
import org.apache.xmpbox.xml.DomXmpParser;
import org.apache.xmpbox.xml.XmpSerializer;
import lombok.extern.slf4j.Slf4j;
/** Applies the document-level PDF/UA requirements: title, language, tab order, declaration. */
@Slf4j
public class PdfUaMetadataWriter {
private static final COSName TABS = COSName.getPDFName("Tabs");
private static final COSName SUSPECTS = COSName.getPDFName("Suspects");
/**
* Applies everything except the conformance declaration. Clause 7.1 requires a title, so a
* blank one falls back to the existing metadata title.
*/
public List<String> applyDocumentRequirements(
PDDocument document, String title, String language, PdfUaProfile profile)
throws IOException {
return applyDocumentRequirements(document, title, language, profile, false);
}
public List<String> applyDocumentRequirements(
PDDocument document,
String title,
String language,
PdfUaProfile profile,
boolean preserveVersion)
throws IOException {
List<String> warnings = new ArrayList<>();
PDDocumentCatalog catalog = document.getDocumentCatalog();
if (language != null && !language.isBlank()) {
catalog.setLanguage(language);
}
String effectiveTitle = resolveTitle(document, title);
if (effectiveTitle != null) {
PDDocumentInformation info = document.getDocumentInformation();
info.setTitle(effectiveTitle);
document.setDocumentInformation(info);
}
// Without this a viewer shows the filename instead of the title, which defeats the point.
PDViewerPreferences preferences = catalog.getViewerPreferences();
if (preferences == null) {
preferences = new PDViewerPreferences(catalog.getCOSObject());
}
preferences.setDisplayDocTitle(true);
catalog.setViewerPreferences(preferences);
// Clause 7.18.1: every page needs an explicit tab order.
for (PDPage page : document.getPages()) {
page.getCOSObject().setName(TABS, "S");
}
// A structure tree flagged as suspect is not conforming.
if (catalog.getMarkInfo() != null) {
catalog.getMarkInfo().getCOSObject().removeItem(SUSPECTS);
}
if (!preserveVersion && document.getVersion() < profile.pdfVersion()) {
document.setVersion(profile.pdfVersion());
}
warnings.addAll(describeFormFields(document));
writeXmp(document, effectiveTitle, language, null);
return warnings;
}
/**
* Gives every form field the {@code /TU} description clause 7.18.1 requires, reusing its
* authored partial name. Unnamed fields are reported, never given a useless placeholder.
*/
private static List<String> describeFormFields(PDDocument document) {
List<String> warnings = new ArrayList<>();
PDAcroForm form = document.getDocumentCatalog().getAcroForm();
if (form == null) {
return warnings;
}
int unnamed = 0;
for (PDField field : form.getFieldTree()) {
String existing = field.getAlternateFieldName();
if (existing != null && !existing.isBlank()) {
continue;
}
String partialName = field.getPartialName();
if (partialName == null || partialName.isBlank()) {
unnamed++;
continue;
}
field.setAlternateFieldName(partialName);
}
if (unnamed > 0) {
warnings.add(
unnamed
+ " form field(s) have neither a description nor a name, so no tooltip"
+ " could be derived. Add one for each before claiming conformance.");
}
return warnings;
}
/**
* Strips the {@code pdfuaid} declaration when validation fails after it was written, so the
* returned file does not assert conformance it lacks.
*/
public void removeConformanceDeclaration(PDDocument document) throws IOException {
PDDocumentCatalog catalog = document.getDocumentCatalog();
XMPMetadata metadata = loadOrCreate(catalog);
XMPSchema identification = metadata.getSchema(PdfUaIdentificationSchema.NAMESPACE);
if (identification == null) {
return;
}
metadata.removeSchema(identification);
serialiseInto(document, metadata);
}
/** Writes the {@code pdfuaid:part} declaration. Only call this after validation has passed. */
public void declareConformance(PDDocument document, PdfUaProfile profile) throws IOException {
writeXmp(document, resolveTitle(document, null), documentLanguage(document), profile);
}
private String resolveTitle(PDDocument document, String preferred) {
if (preferred != null && !preferred.isBlank()) {
return preferred.strip();
}
String existing = document.getDocumentInformation().getTitle();
return existing != null && !existing.isBlank() ? existing.strip() : null;
}
private static String documentLanguage(PDDocument document) {
return document.getDocumentCatalog().getLanguage();
}
/**
* Rewrites the XMP packet, preserving what was there. A malformed packet is replaced, since an
* unparseable one fails validation on its own.
*/
private void writeXmp(PDDocument document, String title, String language, PdfUaProfile profile)
throws IOException {
PDDocumentCatalog catalog = document.getDocumentCatalog();
XMPMetadata metadata = loadOrCreate(catalog);
if (title != null) {
DublinCoreSchema dublinCore = metadata.getDublinCoreSchema();
if (dublinCore == null) {
dublinCore = metadata.createAndAddDublinCoreSchema();
}
dublinCore.setTitle(title);
if (language != null
&& !language.isBlank()
&& (dublinCore.getLanguages() == null
|| !dublinCore.getLanguages().contains(language))) {
dublinCore.addLanguage(language);
}
}
if (profile != null) {
// Re-converting an already-declared file must not leave two pdfuaid schemas.
XMPSchema stale = metadata.getSchema(PdfUaIdentificationSchema.NAMESPACE);
if (stale != null) {
metadata.removeSchema(stale);
}
PdfUaIdentificationSchema identification = new PdfUaIdentificationSchema(metadata);
identification.setPart(profile.part());
if (profile.revision() > 0) {
identification.setRevision(profile.revision());
}
metadata.addSchema(identification);
}
serialiseInto(document, metadata);
}
private static void serialiseInto(PDDocument document, XMPMetadata metadata)
throws IOException {
ByteArrayOutputStream out = new ByteArrayOutputStream();
try {
new XmpSerializer().serialize(metadata, out, true);
} catch (javax.xml.transform.TransformerException e) {
throw new IOException("Could not serialise XMP metadata", e);
}
PDMetadata pdMetadata = new PDMetadata(document);
pdMetadata.importXMPMetadata(out.toByteArray());
document.getDocumentCatalog().setMetadata(pdMetadata);
}
private XMPMetadata loadOrCreate(PDDocumentCatalog catalog) {
PDMetadata existing = catalog.getMetadata();
if (existing != null) {
try {
DomXmpParser parser = new DomXmpParser();
// Strict parsing rejects pdfuaid, silently discarding a packet we just wrote.
parser.setStrictParsing(false);
return parser.parse(new ByteArrayInputStream(existing.toByteArray()));
} catch (Exception e) {
log.debug("Replacing unparseable XMP packet: {}", e.getMessage());
}
}
return XMPMetadata.createXMPMetadata();
}
}
@@ -0,0 +1,47 @@
package stirling.software.proprietary.pdf.ua;
/** The PDF/UA conformance level a conversion targets. */
public enum PdfUaProfile {
/** ISO 14289-1, layered on PDF 1.7. */
UA1(1, 1.7f, 0),
/** ISO 14289-2: needs PDF 2.0, namespaced structure types and a revision year. */
UA2(2, 2.0f, 2024);
private final int part;
private final float pdfVersion;
private final int revision;
PdfUaProfile(int part, float pdfVersion, int revision) {
this.part = part;
this.pdfVersion = pdfVersion;
this.revision = revision;
}
public int part() {
return part;
}
public float pdfVersion() {
return pdfVersion;
}
/** The {@code pdfuaid:rev} year, or 0 when the profile does not use one. */
public int revision() {
return revision;
}
public String displayName() {
return "PDF/UA-" + part;
}
public static PdfUaProfile fromRequest(String value) {
if (value == null || value.isBlank()) {
return UA1;
}
String normalised = value.trim().toLowerCase().replace("/", "").replace("-", "");
return switch (normalised) {
case "ua2", "pdfua2", "2" -> UA2;
default -> UA1;
};
}
}
@@ -0,0 +1,303 @@
package stirling.software.proprietary.pdf.ua;
import java.io.IOException;
import java.util.ArrayList;
import java.util.HashSet;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDDocumentCatalog;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureElement;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureTreeRoot;
import lombok.extern.slf4j.Slf4j;
/**
* Tags an untagged PDF and applies the document-level PDF/UA requirements. Content must be marked
* before the tree can reference it, and conformance is declared elsewhere, only after validation.
*/
@Slf4j
public class PdfUaTagger {
private final TaggedContentExtractor extractor = new TaggedContentExtractor();
private final LayoutAnalyzer analyzer = new LayoutAnalyzer();
private final MarkedContentInjector injector = new MarkedContentInjector();
private final PdfUaMetadataWriter metadataWriter = new PdfUaMetadataWriter();
public TaggingResult tag(PDDocument document, TaggingOptions options) throws IOException {
boolean alreadyTagged = hasUsableStructureTree(document);
boolean rebuild =
switch (options.getExistingTags()) {
case KEEP -> false;
case REBUILD -> true;
case AUTO -> !alreadyTagged;
};
List<String> languageWarnings = new ArrayList<>();
String language = resolveLanguage(document, options, languageWarnings);
if (!rebuild) {
log.info("Keeping existing structure tree; applying document requirements only");
DocumentStructure kept = new DocumentStructure();
languageWarnings.forEach(kept::warn);
metadataWriter
.applyDocumentRequirements(
document,
options.getTitle(),
language,
options.getProfile(),
options.isPreservePdfVersion())
.forEach(kept::warn);
return new TaggingResult(kept, false);
}
// Types the old tree carried, so a rebuild that cannot reproduce them can say so. Font
// embedding may already have deleted the tree, so fall back to what the source had.
Set<String> discardedTypes =
alreadyTagged
? structureTypes(document)
: options.getSourceFacts().structureTypes();
if (alreadyTagged) {
stripStructure(document);
}
List<PageContent> pages = extractor.extract(document);
DocumentStructure structure = analyzer.analyse(pages);
structure.setLanguage(language);
languageWarnings.forEach(structure::warn);
applyFigurePolicy(structure, options);
if (structure.isEmpty()) {
structure.warn(
"No taggable content was found; the document may be a scan with no text layer.");
}
injectMarkedContent(document, structure, pages);
new StructTreeWriter().write(document, structure, options.getProfile());
// Losing the tree to the embedder is a different problem from a requested rebuild, and
// the advice that helps differs too, so tell them apart.
boolean lostToEmbedder = !alreadyTagged && options.getSourceFacts().hasUsableTree();
warnAboutFlattenedStructure(
discardedTypes, structureTypes(document), structure, lostToEmbedder);
String title = resolveTitle(options, structure);
if (title == null) {
structure.warn(
"No document title could be derived. PDF/UA requires one, so supply a title.");
}
metadataWriter
.applyDocumentRequirements(
document,
title,
language,
options.getProfile(),
options.isPreservePdfVersion())
.forEach(structure::warn);
return new TaggingResult(structure, true);
}
/**
* Keeps the language the document already declares. Overwriting it relabels, say, a French file
* as English, and no validator can catch that.
*/
private static String resolveLanguage(
PDDocument document, TaggingOptions options, List<String> warnings) {
String existing = document.getDocumentCatalog().getLanguage();
if (existing == null || existing.isBlank()) {
// Font embedding discards /Lang, so without this a rewritten French document would
// silently take the caller's default language.
existing = options.getSourceFacts().language();
}
String requested = options.getLanguage();
if (existing == null || existing.isBlank() || options.isOverrideLanguage()) {
return requested;
}
if (requested != null && !requested.isBlank() && !requested.equalsIgnoreCase(existing)) {
warnings.add(
"The document already declares its language as '"
+ existing
+ "', so the requested '"
+ requested
+ "' was ignored. Ask to override the language to change it.");
}
return existing;
}
/** Explicit title first, then the first heading, then the caller's fallback. */
private static String resolveTitle(TaggingOptions options, DocumentStructure structure) {
for (String candidate :
new String[] {
options.getTitle(), structure.getTitle(), options.getFallbackTitle()
}) {
if (candidate != null && !candidate.isBlank()) {
return candidate.strip();
}
}
return null;
}
/** Writes the conformance declaration. Separate from tagging so validation can gate it. */
public void declareConformance(PDDocument document, PdfUaProfile profile) throws IOException {
metadataWriter.declareConformance(document, profile);
}
/** Withdraws the conformance claim, for a document that turned out not to validate. */
public void withdrawConformance(PDDocument document) throws IOException {
metadataWriter.removeConformanceDeclaration(document);
}
/** Wraps content page by page; marked content ids restart on each page. */
private void injectMarkedContent(
PDDocument document, DocumentStructure structure, List<PageContent> pages)
throws IOException {
Map<Integer, Integer> markableCounts = new LinkedHashMap<>();
pages.forEach(page -> markableCounts.put(page.pageIndex(), page.markableCount()));
Map<Integer, List<StructBlock>> byPage = new LinkedHashMap<>();
for (StructBlock block : structure.getBlocks()) {
byPage.computeIfAbsent(block.getPageIndex(), k -> new ArrayList<>()).add(block);
}
for (int pageIndex = 0; pageIndex < document.getNumberOfPages(); pageIndex++) {
List<StructBlock> blocks = byPage.getOrDefault(pageIndex, List.of());
// Nothing to wrap, and rewriting costs a parse and recompress for an identical stream.
if (blocks.isEmpty() && markableCounts.getOrDefault(pageIndex, 0) == 0) {
continue;
}
injector.inject(document, document.getPage(pageIndex), blocks, 0, true);
}
}
/** Applies alt text supplied by the caller, or demotes images to artifacts on request. */
private static void applyFigurePolicy(DocumentStructure structure, TaggingOptions options) {
int[] suppressed = {0};
structure.visit(
block -> {
if (block.getType() != StructType.FIGURE) {
return;
}
if (options.getFigurePolicy() == TaggingOptions.FigurePolicy.MARK_DECORATIVE) {
block.setType(StructType.ARTIFACT);
block.setArtifactType(ArtifactType.LAYOUT);
suppressed[0]++;
return;
}
int ordinal =
block.getRanges().isEmpty() ? -1 : block.getRanges().get(0).start();
String alt = options.altTextFor(block.getPageIndex(), ordinal);
if (alt != null && !alt.isBlank()) {
block.setAlt(alt);
}
});
// Marking images decorative validates by hiding content, so never report it as clean.
if (suppressed[0] > 0) {
structure.warn(
suppressed[0]
+ " image(s) were marked as decoration and are now hidden from"
+ " assistive technology. Confirm none of them carried meaning.");
}
int missing = structure.figuresWithoutAlt().size();
if (missing > 0) {
structure.warn(
missing
+ " figure(s) have no alternative description. PDF/UA requires one for"
+ " every image that carries meaning.");
}
}
/**
* A tree is only worth keeping when wired up: kids, a parent tree, and a marked catalog.
* Keeping one that fails any of those leaves the document permanently unfixable.
*/
public static boolean hasUsableStructureTree(PDDocument document) {
PDDocumentCatalog catalog = document.getDocumentCatalog();
PDStructureTreeRoot root = catalog.getStructureTreeRoot();
if (root == null) {
return false;
}
try {
boolean hasKids = root.getKids() != null && !root.getKids().isEmpty();
boolean hasParentTree = root.getParentTree() != null;
boolean marked = catalog.getMarkInfo() != null && catalog.getMarkInfo().isMarked();
return hasKids && hasParentTree && marked;
} catch (RuntimeException e) {
log.debug("Unreadable structure tree, treating as absent: {}", e.getMessage());
return false;
}
}
/**
* A rebuild derives structure from layout, so semantics the old tree carried can vanish - a
* table becomes loose paragraphs. Validators cannot see that loss, so it has to be reported.
*/
private static void warnAboutFlattenedStructure(
Set<String> before,
Set<String> after,
DocumentStructure structure,
boolean lostToEmbedder) {
List<String> lost =
MEANINGFUL_TYPES.stream()
.filter(type -> before.contains(type) && !after.contains(type))
.toList();
if (lost.isEmpty()) {
return;
}
// Keeping the tags cannot help once the embedder has deleted them, so do not suggest it.
String remedy =
lostToEmbedder
? " Embedding the missing fonts rewrote the document and deleted its"
+ " original tags. Turn off font embedding to keep them."
: " Keep the existing tags instead to preserve it.";
structure.warn(
"Rebuilding the tags could not reproduce "
+ String.join(", ", lost)
+ " structure, so that content is now plain paragraphs."
+ remedy);
}
/** Structure whose loss changes what a screen reader conveys, not just how it is nested. */
private static final List<String> MEANINGFUL_TYPES =
List.of("Table", "TH", "Formula", "L", "LI", "TOC", "Note");
private static Set<String> structureTypes(PDDocument document) {
Set<String> types = new HashSet<>();
try {
PDStructureTreeRoot root = document.getDocumentCatalog().getStructureTreeRoot();
if (root != null) {
collectTypes(root.getKids(), types, 0);
}
} catch (RuntimeException e) {
log.debug("Could not read structure types: {}", e.getMessage());
}
return types;
}
private static void collectTypes(Object node, Set<String> types, int depth) {
// Structure trees can be deep or, in damaged files, cyclic; cap rather than overflow.
if (node == null || depth > 64) {
return;
}
if (node instanceof List<?> list) {
list.forEach(child -> collectTypes(child, types, depth + 1));
} else if (node instanceof PDStructureElement element) {
types.add(element.getStructureType());
collectTypes(element.getKids(), types, depth + 1);
}
}
private static void stripStructure(PDDocument document) {
PDDocumentCatalog catalog = document.getDocumentCatalog();
catalog.getCOSObject().removeItem(COSName.getPDFName("StructTreeRoot"));
catalog.getCOSObject().removeItem(COSName.getPDFName("MarkInfo"));
document.getPages()
.forEach(
page ->
page.getCOSObject()
.removeItem(COSName.getPDFName("StructParents")));
log.info("Removed existing structure tree before rebuilding");
}
}
@@ -0,0 +1,59 @@
package stirling.software.proprietary.pdf.ua;
import java.util.HashSet;
import java.util.Set;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureElement;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureTreeRoot;
import lombok.extern.slf4j.Slf4j;
/**
* What the document said about itself before anything rewrote it. Font embedding shells out to
* Ghostscript, which returns a file with no structure tree, no {@code /Lang} and no XMP, so a
* tagger reading the rewritten document sees an untagged, language-less file and cannot tell that
* anything was lost. These facts are captured from the original and carried past that stage.
*
* @param language the catalog {@code /Lang} the author declared, or null
* @param structureTypes every structure element type the original tree contained
* @param hasUsableTree whether the original had a structure tree worth preserving
*/
@Slf4j
public record SourceFacts(String language, Set<String> structureTypes, boolean hasUsableTree) {
private static final int MAX_DEPTH = 64;
/** Facts for a document nothing has rewritten, used when font embedding did not run. */
public static final SourceFacts NONE = new SourceFacts(null, Set.of(), false);
public static SourceFacts of(PDDocument document) {
String language = null;
Set<String> types = new HashSet<>();
boolean usable = false;
try {
language = document.getDocumentCatalog().getLanguage();
usable = PdfUaTagger.hasUsableStructureTree(document);
PDStructureTreeRoot root = document.getDocumentCatalog().getStructureTreeRoot();
if (root != null) {
collect(root.getKids(), types, 0);
}
} catch (RuntimeException e) {
log.debug("Could not read source facts: {}", e.getMessage());
}
return new SourceFacts(language, Set.copyOf(types), usable);
}
private static void collect(Object node, Set<String> types, int depth) {
// Damaged files can present a cyclic tree; cap rather than overflow the stack.
if (node == null || depth > MAX_DEPTH) {
return;
}
if (node instanceof java.util.List<?> list) {
list.forEach(child -> collect(child, types, depth + 1));
} else if (node instanceof PDStructureElement element) {
types.add(element.getStructureType());
collect(element.getKids(), types, depth + 1);
}
}
}
@@ -0,0 +1,134 @@
package stirling.software.proprietary.pdf.ua;
import java.util.ArrayList;
import java.util.List;
import java.util.function.Consumer;
import lombok.Getter;
import lombok.Setter;
/**
* One node of the derived logical structure: either page content (ranges of markable operator
* ordinals) or child blocks. Containers with no content are pruned before serialisation.
*/
@Getter
@Setter
public class StructBlock {
/** A contiguous, inclusive run of markable operator ordinals within one page stream. */
public record OrdinalRange(int start, int end) {
public boolean contains(int ordinal) {
return ordinal >= start && ordinal <= end;
}
public int size() {
return end - start + 1;
}
}
private StructType type;
private ArtifactType artifactType;
private int pageIndex;
private BBox bbox = BBox.EMPTY;
private String text = "";
private final List<OrdinalRange> ranges = new ArrayList<>();
private final List<StructBlock> children = new ArrayList<>();
/** {@code /Alt} - required on Figure and Formula for PDF/UA. */
private String alt;
/** {@code /ActualText} - replacement text for content whose glyphs do not spell the word. */
private String actualText;
/** {@code /Lang} - set only where it differs from the document default. */
private String lang;
/** {@code /Scope} on a TH: Row, Column or Both. */
private String scope;
/** {@code /ListNumbering} on an L. */
private String listNumbering;
/** Unique {@code /ID}, required on Note and FENote elements. */
private String id;
/**
* Marked content ids assigned during injection; one block yields several when split, since a
* sequence must nest inside BT/ET and q/Q rather than straddle them.
*/
private final List<Integer> mcids = new ArrayList<>();
/** True when the source content was already inside a marked-content sequence. */
private boolean preMarked;
public StructBlock(StructType type, int pageIndex) {
this.type = type;
this.pageIndex = pageIndex;
}
public static StructBlock artifact(ArtifactType artifactType, int pageIndex) {
StructBlock block = new StructBlock(StructType.ARTIFACT, pageIndex);
block.artifactType = artifactType;
return block;
}
public StructBlock addChild(StructBlock child) {
children.add(child);
return this;
}
public StructBlock addRange(int start, int end) {
ranges.add(new OrdinalRange(start, end));
return this;
}
public boolean isArtifact() {
return type == StructType.ARTIFACT;
}
/** Depth-first walk over this block and all descendants. */
public void visit(Consumer<StructBlock> visitor) {
visitor.accept(this);
for (StructBlock child : children) {
child.visit(visitor);
}
}
/** Total number of ordinals owned by this block and its descendants. */
public int contentCount() {
int total = ranges.stream().mapToInt(OrdinalRange::size).sum();
for (StructBlock child : children) {
total += child.contentCount();
}
return total;
}
/** Concatenated text of this block and its descendants, in tree order. */
public String collectText() {
StringBuilder sb = new StringBuilder();
visit(
block -> {
if (!block.text.isBlank()) {
if (sb.length() > 0) {
sb.append(' ');
}
sb.append(block.text.strip());
}
});
return sb.toString();
}
@Override
public String toString() {
return type.tag()
+ (artifactType != null ? "[" + artifactType.subtype() + "]" : "")
+ "(p"
+ pageIndex
+ ", "
+ ranges.size()
+ " ranges, "
+ children.size()
+ " kids)";
}
}
@@ -0,0 +1,295 @@
package stirling.software.proprietary.pdf.ua;
import java.io.IOException;
import java.util.ArrayList;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import org.apache.pdfbox.cos.COSArray;
import org.apache.pdfbox.cos.COSBase;
import org.apache.pdfbox.cos.COSDictionary;
import org.apache.pdfbox.cos.COSInteger;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.common.PDNumberTreeNode;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDMarkInfo;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDObjectReference;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureElement;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureTreeRoot;
import org.apache.pdfbox.pdmodel.documentinterchange.taggedpdf.PDListAttributeObject;
import org.apache.pdfbox.pdmodel.documentinterchange.taggedpdf.PDTableAttributeObject;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationLink;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationWidget;
import lombok.extern.slf4j.Slf4j;
/**
* Serialises a {@link DocumentStructure} into a PDF structure tree. Must run after {@link
* MarkedContentInjector}, which assigns the marked content ids this writer references.
*/
@Slf4j
public class StructTreeWriter {
private static final COSName STRUCT_PARENT = COSName.getPDFName("StructParent");
private static final COSName NUMS = COSName.getPDFName("Nums");
private static final String PDF2_STANDARD_NAMESPACE = "http://iso.org/pdf2/ssn";
/** Per-page marked content id to owning element, built while walking the tree. */
private final Map<Integer, Map<Integer, PDStructureElement>> mcidOwners = new LinkedHashMap<>();
private COSDictionary standardNamespace;
private int nextParentKey;
public void write(PDDocument document, DocumentStructure structure, PdfUaProfile profile)
throws IOException {
PDStructureTreeRoot root = new PDStructureTreeRoot();
PDStructureElement documentElement =
new PDStructureElement(StructType.DOCUMENT.tag(), root);
if (structure.getLanguage() != null) {
documentElement.setLanguage(structure.getLanguage());
}
if (profile == PdfUaProfile.UA2) {
applyNamespace(documentElement, document);
}
for (StructBlock block : structure.getBlocks()) {
if (block.isArtifact()) {
continue;
}
PDStructureElement child = buildElement(document, block, documentElement, profile);
if (child != null) {
documentElement.appendKid(child);
}
}
root.appendKid(documentElement);
buildParentTree(document, root);
registerNamespaces(root);
PDMarkInfo markInfo = new PDMarkInfo();
markInfo.setMarked(true);
document.getDocumentCatalog().setMarkInfo(markInfo);
document.getDocumentCatalog().setStructureTreeRoot(root);
}
/** Recursively builds an element, returning null when the block carries no content at all. */
private PDStructureElement buildElement(
PDDocument document,
StructBlock block,
PDStructureElement parent,
PdfUaProfile profile) {
// Prune on assigned MCIDs, not claimed ranges: form-XObject lines all resolve to one Do,
// and emitting the losers would announce empty paragraphs to a screen reader.
if (!carriesContent(block)) {
return null;
}
StructType type = effectiveType(block, profile);
PDStructureElement element = new PDStructureElement(type.tag(), parent);
PDPage page = document.getPage(block.getPageIndex());
element.setPage(page);
if (profile == PdfUaProfile.UA2) {
applyNamespace(element, document);
}
applyAttributes(block, element);
for (int mcid : block.getMcids()) {
element.appendKid(mcid);
mcidOwners
.computeIfAbsent(block.getPageIndex(), k -> new LinkedHashMap<>())
.put(mcid, element);
}
for (StructBlock child : block.getChildren()) {
PDStructureElement childElement = buildElement(document, child, element, profile);
if (childElement != null) {
element.appendKid(childElement);
}
}
return element;
}
/** True when this block, or something beneath it, was actually given marked content. */
private static boolean carriesContent(StructBlock block) {
if (!block.getMcids().isEmpty()) {
return true;
}
return block.getChildren().stream().anyMatch(StructTreeWriter::carriesContent);
}
/** PDF/UA-2 replaces Note with FENote for footnotes. */
private static StructType effectiveType(StructBlock block, PdfUaProfile profile) {
if (profile == PdfUaProfile.UA2 && block.getType() == StructType.NOTE) {
return StructType.FENOTE;
}
return block.getType();
}
private static void applyAttributes(StructBlock block, PDStructureElement element) {
if (block.getAlt() != null && !block.getAlt().isBlank()) {
element.setAlternateDescription(block.getAlt());
}
if (block.getActualText() != null && !block.getActualText().isBlank()) {
element.setActualText(block.getActualText());
}
if (block.getLang() != null && !block.getLang().isBlank()) {
element.setLanguage(block.getLang());
}
if (block.getId() != null && !block.getId().isBlank()) {
element.setElementIdentifier(block.getId());
}
if (block.getScope() != null) {
PDTableAttributeObject table = new PDTableAttributeObject();
table.setScope(block.getScope());
element.addAttribute(table);
}
if (block.getListNumbering() != null) {
PDListAttributeObject list = new PDListAttributeObject();
list.setListNumbering(block.getListNumbering());
element.addAttribute(list);
}
}
/** PDF/UA-2 requires every element to declare the standard structure namespace. */
private void applyNamespace(PDStructureElement element, PDDocument document) {
element.getCOSObject().setItem(COSName.getPDFName("NS"), standardNamespace());
}
/** The PDF 2.0 standard structure namespace, created once per document. */
private COSDictionary standardNamespace() {
if (standardNamespace == null) {
standardNamespace = new COSDictionary();
standardNamespace.setName(COSName.TYPE, "Namespace");
standardNamespace.setString(COSName.getPDFName("NS"), PDF2_STANDARD_NAMESPACE);
}
return standardNamespace;
}
private void registerNamespaces(PDStructureTreeRoot root) {
if (standardNamespace == null) {
return;
}
COSArray namespaces = new COSArray();
namespaces.add(standardNamespace);
root.getCOSObject().setItem(COSName.getPDFName("Namespaces"), namespaces);
}
/**
* Builds {@code /ParentTree}: per page, an array indexed by marked content id keyed on {@code
* /StructParents}, plus one entry per annotation keyed on {@code /StructParent}.
*/
private void buildParentTree(PDDocument document, PDStructureTreeRoot root) {
COSArray nums = new COSArray();
nextParentKey = 0;
for (int pageIndex = 0; pageIndex < document.getNumberOfPages(); pageIndex++) {
Map<Integer, PDStructureElement> owners = mcidOwners.get(pageIndex);
if (owners == null || owners.isEmpty()) {
continue;
}
PDPage page = document.getPage(pageIndex);
int key = nextParentKey++;
page.setStructParents(key);
int maxMcid = owners.keySet().stream().mapToInt(Integer::intValue).max().orElse(-1);
COSArray entries = new COSArray();
for (int mcid = 0; mcid <= maxMcid; mcid++) {
PDStructureElement owner = owners.get(mcid);
entries.add(
owner != null ? owner.getCOSObject() : org.apache.pdfbox.cos.COSNull.NULL);
}
nums.add(COSInteger.get(key));
nums.add(entries);
}
List<COSBase> annotationEntries = tagAnnotations(document, root);
for (int i = 0; i + 1 < annotationEntries.size(); i += 2) {
nums.add(annotationEntries.get(i));
nums.add(annotationEntries.get(i + 1));
}
COSDictionary parentTreeDict = new COSDictionary();
parentTreeDict.setItem(NUMS, nums);
root.setParentTree(new PDNumberTreeNode(parentTreeDict, PDStructureElement.class));
root.setParentTreeNextKey(nextParentKey);
}
/**
* Clause 7.18: every visible annotation needs a structure element so it is reachable from the
* tree. Links become Link elements, anything else an Annot.
*/
private List<COSBase> tagAnnotations(PDDocument document, PDStructureTreeRoot root) {
List<COSBase> entries = new ArrayList<>();
PDStructureElement documentElement = firstDocumentElement(root);
if (documentElement == null) {
return entries;
}
for (int pageIndex = 0; pageIndex < document.getNumberOfPages(); pageIndex++) {
PDPage page = document.getPage(pageIndex);
List<PDAnnotation> annotations;
try {
annotations = page.getAnnotations();
} catch (IOException e) {
log.debug("Could not read annotations on page {}: {}", pageIndex, e.getMessage());
continue;
}
for (PDAnnotation annotation : annotations) {
if (annotation == null
|| annotation.isHidden()
|| annotation.isNoView()
|| "Popup".equals(annotation.getSubtype())) {
continue;
}
PDStructureElement element =
new PDStructureElement(annotationType(annotation), documentElement);
element.setPage(page);
PDObjectReference reference = new PDObjectReference();
reference.setReferencedObject(annotation);
element.appendKid(reference);
documentElement.appendKid(element);
int key = nextParentKey++;
annotation.getCOSObject().setInt(STRUCT_PARENT, key);
entries.add(COSInteger.get(key));
entries.add(element.getCOSObject());
if (annotation.getContents() == null || annotation.getContents().isBlank()) {
annotation.setContents(defaultContents(annotation));
}
}
}
return entries;
}
/** Clause 7.18.4: widgets need a Form element, links a Link element, everything else Annot. */
private static String annotationType(PDAnnotation annotation) {
if (annotation instanceof PDAnnotationWidget) {
return StructType.FORM.tag();
}
if (annotation instanceof PDAnnotationLink) {
return StructType.LINK.tag();
}
return "Annot";
}
private static String defaultContents(PDAnnotation annotation) {
if (annotation instanceof PDAnnotationLink link && link.getAction() != null) {
return "Link";
}
return annotation.getSubtype() == null ? "Annotation" : annotation.getSubtype();
}
private static PDStructureElement firstDocumentElement(PDStructureTreeRoot root) {
for (Object kid : root.getKids()) {
if (kid instanceof PDStructureElement element) {
return element;
}
}
return null;
}
}
@@ -0,0 +1,62 @@
package stirling.software.proprietary.pdf.ua;
/**
* PDF standard structure types emitted by the tagger (ISO 32000-1 14.8.4), limited to the PDF/UA
* subset. {@link #ARTIFACT} is not one: it marks content in the stream and stays out of the tree.
*/
public enum StructType {
DOCUMENT("Document"),
PART("Part"),
SECT("Sect"),
H1("H1"),
H2("H2"),
H3("H3"),
H4("H4"),
H5("H5"),
H6("H6"),
P("P"),
L("L"),
LI("LI"),
LBL("Lbl"),
LBODY("LBody"),
TABLE("Table"),
TR("TR"),
TH("TH"),
TD("TD"),
FIGURE("Figure"),
CAPTION("Caption"),
FORMULA("Formula"),
NOTE("Note"),
FENOTE("FENote"),
LINK("Link"),
/** Wraps a widget annotation; PDF/UA-1 clause 7.18.4 requires widgets to sit inside one. */
FORM("Form"),
SPAN("Span"),
ARTIFACT("Artifact");
private final String tag;
StructType(String tag) {
this.tag = tag;
}
/** The name written into the PDF {@code /S} entry. */
public String tag() {
return tag;
}
public boolean isHeading() {
return this == H1 || this == H2 || this == H3 || this == H4 || this == H5 || this == H6;
}
/** Heading level 1-6, or 0 when this is not a heading. */
public int headingLevel() {
return isHeading() ? ordinal() - H1.ordinal() + 1 : 0;
}
/** The heading type for a 1-based level, clamped to the H1-H6 range. */
public static StructType heading(int level) {
int clamped = Math.max(1, Math.min(6, level));
return values()[H1.ordinal() + clamped - 1];
}
}
@@ -0,0 +1,630 @@
package stirling.software.proprietary.pdf.ua;
import java.io.IOException;
import java.io.Writer;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.IdentityHashMap;
import java.util.List;
import java.util.Map;
import org.apache.pdfbox.contentstream.operator.Operator;
import org.apache.pdfbox.cos.COSBase;
import org.apache.pdfbox.cos.COSDictionary;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdfparser.PDFStreamParser;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDResources;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDFontDescriptor;
import org.apache.pdfbox.pdmodel.font.PDType3Font;
import org.apache.pdfbox.pdmodel.graphics.PDXObject;
import org.apache.pdfbox.pdmodel.graphics.form.PDFormXObject;
import org.apache.pdfbox.pdmodel.graphics.form.PDTransparencyGroup;
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
import org.apache.pdfbox.text.PDFTextStripper;
import org.apache.pdfbox.text.TextPosition;
import org.apache.pdfbox.util.Matrix;
import org.apache.pdfbox.util.Vector;
import lombok.extern.slf4j.Slf4j;
/**
* Extracts text lines and graphic ops from page streams, tagging each with its operator ordinal.
* Both passes count the same operators in the same order, so ordinals cross-reference.
*/
@Slf4j
public class TaggedContentExtractor {
/** Glyph size below which a run is treated as noise rather than a line. */
private static final float MIN_FONT_SIZE = 0.5f;
public List<PageContent> extract(PDDocument document) throws IOException {
LineCollector collector = new LineCollector();
collector.setSortByPosition(true);
collector.setStartPage(1);
collector.setEndPage(document.getNumberOfPages());
collector.writeText(document, Writer.nullWriter());
List<PageContent> pages = new ArrayList<>(document.getNumberOfPages());
for (int i = 0; i < document.getNumberOfPages(); i++) {
PDPage page = document.getPage(i);
List<MarkableOp> ops = collector.opsFor(i);
List<TextLineInfo> lines = collector.linesFor(i);
boolean dropped = false;
if (ops.size() < maxOrdinal(lines) + 1) {
// Untrusted ordinals: drop the lines so the page is untaggable rather than
// mis-tagged, and flag it so the caller refuses to declare conformance.
log.warn(
"Ordinal mismatch on page {} (ops={}, text={}); skipping page",
i,
ops.size(),
maxOrdinal(lines) + 1);
dropped = !lines.isEmpty();
lines = List.of();
}
pages.add(
new PageContent(
i,
lines,
ops,
ops.size(),
collector.preMarkedOn(i),
collector.textSemanticsOn(i),
dropped,
normalisedBox(page)));
}
return pages;
}
/**
* The page box in the space of extracted line coordinates: origin-zero, width and height
* swapped for 90/270 rotations, because the text engine reports in the rotated frame.
*/
static BBox normalisedBox(PDPage page) {
PDRectangle mediaBox = page.getMediaBox();
boolean sideways = page.getRotation() % 180 != 0;
float width = sideways ? mediaBox.getHeight() : mediaBox.getWidth();
float height = sideways ? mediaBox.getWidth() : mediaBox.getHeight();
return new BBox(0, 0, width, height);
}
/** Counts images with the token scan alone, skipping the expensive text pass. */
public int countGraphics(PDDocument document) {
int total = 0;
for (int i = 0; i < document.getNumberOfPages(); i++) {
try {
PDResources resources = document.getPage(i).getResources();
PDFStreamParser parser = new PDFStreamParser(document.getPage(i));
List<COSBase> operands = new ArrayList<>();
Object token;
while ((token = parser.parseNextToken()) != null) {
if (!(token instanceof Operator operator)) {
operands.add((COSBase) token);
continue;
}
if (isGraphicOperator(operator.getName(), operands, resources)) {
total++;
}
operands.clear();
}
} catch (IOException e) {
log.debug("Could not scan page {} for graphics: {}", i, e.getMessage());
}
}
return total;
}
/** True for an inline image, or a Do that resolves to an image XObject. */
private static boolean isGraphicOperator(
String name, List<COSBase> operands, PDResources resources) {
if ("BI".equals(name)) {
return true;
}
if (!"Do".equals(name) || resources == null || operands.size() != 1) {
return false;
}
if (!(operands.get(0) instanceof COSName resourceName)) {
return false;
}
try {
return resources.getXObject(resourceName) instanceof PDImageXObject;
} catch (IOException e) {
return false;
}
}
private static int maxOrdinal(List<TextLineInfo> lines) {
return lines.stream().mapToInt(TextLineInfo::endOrdinal).max().orElse(-1);
}
static BBox toBBox(PDRectangle rect) {
return new BBox(
rect.getLowerLeftX(),
rect.getLowerLeftY(),
rect.getUpperRightX(),
rect.getUpperRightY());
}
// --- Operator classification -------------------------------------------
private static boolean isPathConstruction(String name) {
return switch (name) {
case "m", "l", "c", "v", "y", "re" -> true;
default -> false;
};
}
/** True when a sequence carries replacement or alternative text, which a rebuild would drop. */
private static boolean carriesTextSemantics(List<COSBase> operands) {
for (COSBase operand : operands) {
if (operand instanceof COSDictionary dictionary
&& (dictionary.containsKey(COSName.getPDFName("ActualText"))
|| dictionary.containsKey(COSName.getPDFName("Alt"))
|| dictionary.containsKey(COSName.E))) {
return true;
}
}
return false;
}
/**
* Describes one markable operator, placed with the engine's own matrix rather than a
* hand-rolled q/Q/cm stack that would get nesting and form matrices wrong.
*/
private static MarkableOp classify(
String name,
List<COSBase> operands,
PDResources resources,
Matrix ctm,
BBox pathBox,
int ordinal) {
if ("BI".equals(name)) {
return new MarkableOp(ordinal, MarkableOp.Kind.INLINE_IMAGE, unitSquare(ctm), null);
}
if (MarkableOp.isPathPainting(name)) {
return new MarkableOp(ordinal, MarkableOp.Kind.VECTOR, pathBox, null);
}
if (!"Do".equals(name)) {
return new MarkableOp(ordinal, MarkableOp.Kind.TEXT, BBox.EMPTY, null);
}
COSName resourceName =
operands.size() == 1 && operands.get(0) instanceof COSName n ? n : null;
if (resourceName == null || resources == null) {
return new MarkableOp(ordinal, MarkableOp.Kind.FORM, unitSquare(ctm), null);
}
try {
PDXObject xobject = resources.getXObject(resourceName);
if (xobject instanceof PDImageXObject) {
return new MarkableOp(
ordinal, MarkableOp.Kind.IMAGE, unitSquare(ctm), resourceName.getName());
}
if (xobject instanceof PDFormXObject form) {
return new MarkableOp(
ordinal, MarkableOp.Kind.FORM, formBox(form, ctm), resourceName.getName());
}
} catch (IOException e) {
log.debug("Could not resolve XObject {}: {}", resourceName.getName(), e.getMessage());
}
return new MarkableOp(
ordinal, MarkableOp.Kind.FORM, unitSquare(ctm), resourceName.getName());
}
/**
* Extends the running path box with one path-construction operator's points; without it every
* vector had an empty box and charts and vector logos vanished from the structure tree.
*/
private static BBox extendPath(BBox current, String name, List<COSBase> operands, Matrix ctm) {
int pairs =
switch (name) {
case "m", "l" -> 1;
case "re" -> 2;
case "v", "y" -> 2;
case "c" -> 3;
default -> 0;
};
if (pairs == 0 || operands.size() < pairs * 2) {
return current;
}
// Deliberately allocation-free; the obvious version cost a third of the extraction budget.
float minX = current.isEmpty() ? Float.MAX_VALUE : current.x0();
float minY = current.isEmpty() ? Float.MAX_VALUE : current.y0();
float maxX = current.isEmpty() ? -Float.MAX_VALUE : current.x1();
float maxY = current.isEmpty() ? -Float.MAX_VALUE : current.y1();
for (int pair = 0; pair < pairs; pair++) {
Float x = numberAt(operands, pair * 2);
Float y = numberAt(operands, pair * 2 + 1);
if (x == null || y == null) {
continue;
}
float px = x;
float py = y;
// "re" gives origin plus size, so the second pair is a corner offset from the first.
if ("re".equals(name) && pair == 1) {
Float ox = numberAt(operands, 0);
Float oy = numberAt(operands, 1);
if (ox == null || oy == null) {
continue;
}
px = ox + x;
py = oy + y;
}
float tx = ctm.getScaleX() * px + ctm.getShearX() * py + ctm.getTranslateX();
float ty = ctm.getShearY() * px + ctm.getScaleY() * py + ctm.getTranslateY();
minX = Math.min(minX, tx);
minY = Math.min(minY, ty);
maxX = Math.max(maxX, tx);
maxY = Math.max(maxY, ty);
}
return maxX < minX ? current : new BBox(minX, minY, maxX, maxY);
}
private static Float numberAt(List<COSBase> operands, int index) {
return index < operands.size()
&& operands.get(index) instanceof org.apache.pdfbox.cos.COSNumber number
? number.floatValue()
: null;
}
/** The unit square mapped through the CTM, which is how images are placed. */
private static BBox unitSquare(Matrix ctm) {
return transformBox(new BBox(0, 0, 1, 1), ctm);
}
private static BBox formBox(PDFormXObject form, Matrix ctm) {
PDRectangle box = form.getBBox();
if (box == null) {
return unitSquare(ctm);
}
Matrix combined = form.getMatrix() != null ? form.getMatrix().multiply(ctm) : ctm;
return transformBox(toBBox(box), combined);
}
private static BBox transformBox(BBox box, Matrix m) {
float[] xs = new float[4];
float[] ys = new float[4];
float[][] corners = {
{box.x0(), box.y0()}, {box.x1(), box.y0()},
{box.x0(), box.y1()}, {box.x1(), box.y1()}
};
for (int i = 0; i < 4; i++) {
Vector v = m.transform(new Vector(corners[i][0], corners[i][1]));
xs[i] = v.getX();
ys[i] = v.getY();
}
float minX = Math.min(Math.min(xs[0], xs[1]), Math.min(xs[2], xs[3]));
float maxX = Math.max(Math.max(xs[0], xs[1]), Math.max(xs[2], xs[3]));
float minY = Math.min(Math.min(ys[0], ys[1]), Math.min(ys[2], ys[3]));
float maxY = Math.max(Math.max(ys[0], ys[1]), Math.max(ys[2], ys[3]));
return new BBox(minX, minY, maxX, maxY);
}
// --- Text pass ---------------------------------------------------------
/** Marker recorded for each glyph so a finished line knows where it came from. */
private record GlyphOrigin(int ordinal, boolean marked) {}
private static final class LineCollector extends PDFTextStripper {
private final Map<Integer, List<TextLineInfo>> byPage = new HashMap<>();
private final Map<Integer, List<MarkableOp>> opsByPage = new HashMap<>();
private final Map<Integer, Boolean> preMarkedByPage = new HashMap<>();
private final Map<Integer, Boolean> textSemanticsByPage = new HashMap<>();
private final Map<TextPosition, GlyphOrigin> origins = new IdentityHashMap<>();
private final List<TextPosition> lineBuffer = new ArrayList<>();
private final List<WordInfo> lineWords = new ArrayList<>();
private final StringBuilder lineText = new StringBuilder();
private int ordinal = -1;
private int markedDepth;
private int nestedDepth;
private BBox pathBox = BBox.EMPTY;
private int syntheticDepth;
private float pageHeight;
private int pageIndex;
LineCollector() throws IOException {
super();
}
List<TextLineInfo> linesFor(int index) {
return byPage.getOrDefault(index, List.of());
}
List<MarkableOp> opsFor(int index) {
return opsByPage.getOrDefault(index, List.of());
}
boolean preMarkedOn(int index) {
return preMarkedByPage.getOrDefault(index, false);
}
boolean textSemanticsOn(int index) {
return textSemanticsByPage.getOrDefault(index, false);
}
@Override
protected void startPage(PDPage page) throws IOException {
ordinal = -1;
markedDepth = 0;
nestedDepth = 0;
syntheticDepth = 0;
pathBox = BBox.EMPTY;
origins.clear();
lineBuffer.clear();
lineWords.clear();
lineText.setLength(0);
// Dir-adjusted glyph coordinates live in the rotated frame, so the flip must too.
pageHeight = normalisedBox(page).height();
pageIndex = getCurrentPageNo() - 1;
super.startPage(page);
}
@Override
protected void endPage(PDPage page) throws IOException {
flushLine();
super.endPage(page);
}
/**
* Counts only operators physically present in the page's own stream: PDFBox re-enters here
* with synthetic calls for {@code '} and {@code "}, and descends into form XObjects.
*/
@Override
protected void processOperator(Operator operator, List<COSBase> operands)
throws IOException {
String name = operator.getName();
if (nestedDepth == 0 && syntheticDepth == 0) {
if (isPathConstruction(name)) {
pathBox =
extendPath(
pathBox,
name,
operands,
getGraphicsState().getCurrentTransformationMatrix());
}
if (MarkableOp.isMarkableOperator(name)) {
ordinal++;
// Classified here rather than in a second parse of the same stream: the engine
// already has the operands and the live transformation matrix.
opsByPage
.computeIfAbsent(pageIndex, k -> new ArrayList<>())
.add(
classify(
name,
operands,
getResources(),
getGraphicsState().getCurrentTransformationMatrix(),
pathBox,
ordinal));
if (MarkableOp.isPathPainting(name)) {
pathBox = BBox.EMPTY;
}
} else if ("n".equals(name)) {
pathBox = BBox.EMPTY;
} else if ("BDC".equals(name) || "BMC".equals(name)) {
markedDepth++;
preMarkedByPage.put(pageIndex, true);
if (carriesTextSemantics(operands)) {
textSemanticsByPage.put(pageIndex, true);
}
} else if ("EMC".equals(name) && markedDepth > 0) {
markedDepth--;
}
}
boolean synthesises = "'".equals(name) || "\"".equals(name);
if (synthesises) {
syntheticDepth++;
}
try {
super.processOperator(operator, operands);
} finally {
if (synthesises) {
syntheticDepth--;
}
}
}
@Override
public void showForm(PDFormXObject form) throws IOException {
nestedDepth++;
try {
super.showForm(form);
} finally {
nestedDepth--;
}
}
@Override
public void showTransparencyGroup(PDTransparencyGroup group) throws IOException {
nestedDepth++;
try {
super.showTransparencyGroup(group);
} finally {
nestedDepth--;
}
}
@Override
protected void showType3Glyph(
Matrix textRenderingMatrix,
PDType3Font font,
int code,
org.apache.pdfbox.util.Vector displacement)
throws IOException {
nestedDepth++;
try {
super.showType3Glyph(textRenderingMatrix, font, code, displacement);
} finally {
nestedDepth--;
}
}
@Override
protected void processChildStream(
org.apache.pdfbox.contentstream.PDContentStream contentStream, PDPage page)
throws IOException {
nestedDepth++;
try {
super.processChildStream(contentStream, page);
} finally {
nestedDepth--;
}
}
@Override
protected void processTextPosition(TextPosition text) {
origins.put(text, new GlyphOrigin(ordinal, markedDepth > 0));
super.processTextPosition(text);
}
@Override
protected void writeString(String text, List<TextPosition> positions) {
lineText.append(text);
lineBuffer.addAll(positions);
WordInfo word = buildWord(text, positions);
if (word != null) {
lineWords.add(word);
}
}
private WordInfo buildWord(String text, List<TextPosition> positions) {
if (text == null || text.isBlank() || positions.isEmpty()) {
return null;
}
Bounds bounds = new Bounds();
for (TextPosition tp : positions) {
bounds.accept(tp, pageHeight, origins.get(tp));
}
if (bounds.end < 0) {
return null;
}
return new WordInfo(
text,
bounds.box(),
bounds.start,
bounds.end,
bounds.dominantSize(),
bounds.bold);
}
@Override
protected void writeWordSeparator() {
lineText.append(' ');
}
@Override
protected void writeLineSeparator() {
flushLine();
}
@Override
protected void writeParagraphSeparator() {
flushLine();
}
private void flushLine() {
if (lineBuffer.isEmpty()) {
lineText.setLength(0);
lineWords.clear();
return;
}
TextLineInfo line = buildLine();
lineBuffer.clear();
lineWords.clear();
lineText.setLength(0);
if (line != null) {
byPage.computeIfAbsent(pageIndex, k -> new ArrayList<>()).add(line);
}
}
private TextLineInfo buildLine() {
String text = lineText.toString();
if (text.isBlank()) {
return null;
}
Bounds bounds = new Bounds();
for (TextPosition tp : lineBuffer) {
bounds.accept(tp, pageHeight, origins.get(tp));
}
if (bounds.end < 0) {
return null;
}
return new TextLineInfo(
pageIndex,
text,
bounds.box(),
bounds.dominantSize(),
bounds.bold,
bounds.start,
bounds.end,
bounds.marked,
List.copyOf(lineWords));
}
}
/** Accumulates glyph geometry, ordinals and font signals for a word or a line. */
private static final class Bounds {
private float minX = Float.MAX_VALUE;
private float maxX = -Float.MAX_VALUE;
private float minY = Float.MAX_VALUE;
private float maxY = -Float.MAX_VALUE;
private int start = Integer.MAX_VALUE;
private int end = -1;
private boolean marked;
private boolean bold;
private final Map<Float, Integer> sizeCounts = new HashMap<>();
void accept(TextPosition tp, float pageHeight, GlyphOrigin origin) {
float top = pageHeight - tp.getYDirAdj();
float bottom = top - Math.max(tp.getHeightDir(), 0);
minX = Math.min(minX, tp.getXDirAdj());
maxX = Math.max(maxX, tp.getXDirAdj() + tp.getWidthDirAdj());
minY = Math.min(minY, bottom);
maxY = Math.max(maxY, top);
if (origin != null) {
start = Math.min(start, origin.ordinal());
end = Math.max(end, origin.ordinal());
marked |= origin.marked();
}
float size = tp.getFontSizeInPt();
if (size > MIN_FONT_SIZE) {
sizeCounts.merge(round(size), 1, Integer::sum);
}
bold |= isBold(tp);
}
BBox box() {
return new BBox(minX, minY, maxX, maxY);
}
float dominantSize() {
return sizeCounts.entrySet().stream()
.max(Map.Entry.comparingByValue())
.map(Map.Entry::getKey)
.orElse(0f);
}
private static float round(float value) {
return Math.round(value * 10f) / 10f;
}
private static boolean isBold(TextPosition tp) {
if (tp.getFont() == null) {
return false;
}
String name = tp.getFont().getName();
if (name != null && name.toLowerCase().contains("bold")) {
return true;
}
PDFontDescriptor descriptor = tp.getFont().getFontDescriptor();
return descriptor != null
&& (descriptor.getFontWeight() >= 600 || descriptor.isForceBold());
}
}
}
@@ -0,0 +1,63 @@
package stirling.software.proprietary.pdf.ua;
import java.util.Map;
import lombok.Builder;
import lombok.Getter;
/** Inputs that change how a document is tagged. */
@Getter
@Builder(toBuilder = true)
public class TaggingOptions {
/** What to do when the source already has a structure tree. */
public enum ExistingTags {
/** Leave the tree alone and fix only document-level requirements. */
KEEP,
/** Discard the tree and derive a new one. */
REBUILD,
/** Keep a usable tree, rebuild an empty or trivially broken one. */
AUTO
}
/** How images with no alternative description are handled. */
public enum FigurePolicy {
/** Leave undescribed so validation fails honestly; a faked {@code /Alt} helps nobody. */
REQUIRE_ALT,
/** Treat every image as decoration and mark it as an artifact. */
MARK_DECORATIVE
}
@Builder.Default private PdfUaProfile profile = PdfUaProfile.UA1;
/** BCP-47 language tag for the document, for example {@code en-GB}. */
private String language;
/** Replace a language the document already declares. Off, so a French file stays French. */
@Builder.Default private boolean overrideLanguage = false;
private String title;
/** Last resort when no title is given and none can be derived; pass the uploaded filename. */
private String fallbackTitle;
/** Embed any font the document references but does not carry, which clause 7.21 requires. */
@Builder.Default private boolean embedFonts = true;
/** Leave the PDF version alone; raising it would break PDF/A-1, defined on PDF 1.4. */
@Builder.Default private boolean preservePdfVersion = false;
@Builder.Default private ExistingTags existingTags = ExistingTags.AUTO;
@Builder.Default private FigurePolicy figurePolicy = FigurePolicy.REQUIRE_ALT;
/** Alternative descriptions supplied by the caller, keyed by "pageIndex:ordinal". */
@Builder.Default private Map<String, String> altTextByFigure = Map.of();
/** What the document said before font embedding rewrote it; see {@link SourceFacts}. */
@Builder.Default private SourceFacts sourceFacts = SourceFacts.NONE;
public String altTextFor(int pageIndex, int ordinal) {
return altTextByFigure.get(pageIndex + ":" + ordinal);
}
}
@@ -0,0 +1,42 @@
package stirling.software.proprietary.pdf.ua;
import java.util.ArrayList;
import java.util.List;
import lombok.Getter;
/** What a tagging run produced, for the conversion report. */
@Getter
public class TaggingResult {
private final List<String> warnings = new ArrayList<>();
private final DocumentStructure structure;
private final boolean rebuilt;
private final int taggedElements;
private final int artifacts;
private final int figuresNeedingAlt;
/** True when text was hidden as artifacts; the caller must not declare conformance. */
private final boolean contentSuppressed;
public TaggingResult(DocumentStructure structure, boolean rebuilt) {
this.structure = structure;
this.rebuilt = rebuilt;
this.warnings.addAll(structure.getWarnings());
int[] elements = {0};
structure.visit(
block -> {
if (!block.isArtifact()) {
elements[0]++;
}
});
this.taggedElements = elements[0];
this.artifacts = structure.artifactCount();
this.figuresNeedingAlt = structure.figuresWithoutAlt().size();
this.contentSuppressed = structure.isTextSuppressed();
}
public boolean needsHumanReview() {
return figuresNeedingAlt > 0 || !warnings.isEmpty();
}
}
@@ -0,0 +1,42 @@
package stirling.software.proprietary.pdf.ua;
import java.util.List;
/**
* A run of text on one baseline, with the operator ordinals that produced it. {@code preMarked}
* means the source stream already wrapped this text in BDC/EMC.
*/
public record TextLineInfo(
int pageIndex,
String text,
BBox bbox,
float dominantFontSize,
boolean bold,
int startOrdinal,
int endOrdinal,
boolean preMarked,
List<WordInfo> words) {
public boolean isBlank() {
return text == null || text.isBlank();
}
public int charCount() {
return text == null ? 0 : text.strip().length();
}
public int wordCount() {
return (int) words.stream().filter(w -> !w.isBlank()).count();
}
/** True when every word occupies its own operator run, so cells can be tagged separately. */
public boolean wordsAreSeparable() {
List<WordInfo> real = words.stream().filter(w -> !w.isBlank()).toList();
for (int i = 1; i < real.size(); i++) {
if (!real.get(i - 1).isSeparableFrom(real.get(i))) {
return false;
}
}
return true;
}
}
@@ -0,0 +1,18 @@
package stirling.software.proprietary.pdf.ua;
/**
* A whitespace-delimited run of glyphs, with the operator ordinals that produced it. Cell detection
* needs both: geometry to find cells, ordinals to tell whether they can be tagged separately.
*/
public record WordInfo(
String text, BBox bbox, int startOrdinal, int endOrdinal, float fontSize, boolean bold) {
public boolean isBlank() {
return text == null || text.isBlank();
}
/** True when this word shares no operator with the other, so both can carry their own MCID. */
public boolean isSeparableFrom(WordInfo other) {
return endOrdinal < other.startOrdinal || other.endOrdinal < startOrdinal;
}
}
@@ -26,6 +26,11 @@ public class AdminPolicyManagementAuthority implements PolicyManagementAuthority
return userService.isCurrentUserAdmin();
}
@Override
public boolean canTriggerPolicies() {
return userService.isCurrentUserAdmin();
}
@Override
public Long currentUserTeamId() {
String username = userService.getCurrentUsername();
@@ -12,6 +12,17 @@ public interface PolicyManagementAuthority {
/** Whether the current user may create, edit, or delete policies (for their own team). */
boolean canEditPolicies();
/**
* Whether the current user may run a policy against its <em>configured sources</em> (the manual
* "run now" sweep). Kept separate from {@link #canEditPolicies()} because the two are distinct
* capabilities, even where a deployment grants both to the same people: a sweep operates on the
* team's configured sources using the server's stored connection credentials, which makes it a
* policy-management capability rather than ordinary use. Running a policy over the caller's
* <em>own</em> uploaded files is not covered by this and stays open to every team member — that
* is ordinary editor enforcement.
*/
boolean canTriggerPolicies();
/**
* The team that scopes the current user's policies — the team a new policy is stamped with and
* the only team whose policies the user may see/run/edit. {@code null} when it can't be
@@ -21,6 +21,7 @@ import org.springframework.web.bind.annotation.PostMapping;
import org.springframework.web.bind.annotation.PutMapping;
import org.springframework.web.bind.annotation.RequestBody;
import org.springframework.web.bind.annotation.RequestMapping;
import org.springframework.web.bind.annotation.RequestParam;
import org.springframework.web.bind.annotation.RequestPart;
import org.springframework.web.bind.annotation.RestController;
import org.springframework.web.context.request.RequestContextHolder;
@@ -51,6 +52,7 @@ import stirling.software.common.util.TempFile;
import stirling.software.common.util.TempFileManager;
import stirling.software.proprietary.audit.AuditContext;
import stirling.software.proprietary.policy.asset.PolicyAssetCleaner;
import stirling.software.proprietary.policy.asset.PolicyAssetResolver;
import stirling.software.proprietary.policy.config.PolicyAccessGuard;
import stirling.software.proprietary.policy.config.PolicyManagementAuthority;
import stirling.software.proprietary.policy.engine.PolicyRunHandle;
@@ -106,6 +108,7 @@ public class PolicyController {
private final PolicyTriggerManager policyTriggerManager;
private final PolicyOverviewService policyOverviewService;
private final PolicyAssetCleaner assetCleaner;
private final PolicyAssetResolver assetResolver;
private final ProcessedLedger processedLedger;
private final List<PolicyTrigger> policyTriggers;
private final ApplicationProperties applicationProperties;
@@ -125,12 +128,13 @@ public class PolicyController {
+ " endpoint and download outputs via /api/v1/general/files/{id}.")
public ResponseEntity<JobResponse<Void>> run(
@RequestPart("json") PipelineDefinition definition,
@RequestParam(value = "policyId", required = false) String policyId,
@Valid @ModelAttribute PolicyRunFiles files)
throws IOException {
stampPolicyAudit(definition);
requireRunnable(definition);
validateAdHocRun(definition);
PolicyInputs inputs = toInputs(files);
PolicyInputs inputs = resolveStoredAssets(policyId, toInputs(files));
PolicyRunHandle handle =
policyRunner.runAdHoc(definition, inputs, PolicyProgressListener.NOOP);
recordEditorDocs(inputs);
@@ -146,12 +150,13 @@ public class PolicyController {
+ " 'cancelled', or 'waiting' event carrying the final run view.")
public SseEmitter runStream(
@RequestPart("json") PipelineDefinition definition,
@RequestParam(value = "policyId", required = false) String policyId,
@Valid @ModelAttribute PolicyRunFiles files)
throws IOException {
stampPolicyAudit(definition);
requireRunnable(definition);
validateAdHocRun(definition);
PolicyInputs inputs = toInputs(files);
PolicyInputs inputs = resolveStoredAssets(policyId, toInputs(files));
SseEmitter emitter =
new SseEmitter(applicationProperties.getPolicies().getStreamTimeoutMs());
@@ -432,21 +437,47 @@ public class PolicyController {
* admin gets no say on SaaS. Team scoping (which team's policies) is enforced separately by
* {@link PolicyAccessGuard}. Every mutation routes through {@link #savePolicy} (pause/resume
* re-save with a flipped {@code enabled} flag) or {@link #deletePolicy}, so gating those two
* covers them all; runs ({@code /run}) stay open to the team. Single-user deployments (login
* disabled) have no such role, so they trust the local operator. The path allowlist for folder
* sources/outputs is enforced separately by {@link PolicyValidator} at validation time.
* covers them all; runs over the caller's own files ({@code /{id}/run}) stay open to the team,
* while source sweeps are gated by {@link #requirePolicySweepAllowed}. Single-user deployments
* (login disabled) have no such role, so they trust the local operator. The path allowlist for
* folder sources/outputs is enforced separately by {@link PolicyValidator} at validation time.
*/
private void requirePolicyEditingAllowed() {
if (!applicationProperties.getSecurity().isEnableLogin()) {
return;
}
if (!policyManagementAuthority.canEditPolicies()) {
if (!policyEditingAllowed()) {
throw new ResponseStatusException(
HttpStatus.FORBIDDEN,
"Policies may only be created or modified by a team leader");
}
}
/**
* Sweeping a policy's configured sources requires the same role as managing policies: the sweep
* operates on the team's configured sources using the server's stored connection credentials,
* which makes it a policy-management capability rather than ordinary use, and team scoping on
* its own does not express that. Deliberately narrower than it looks: it gates only the sweep,
* not {@link #runStoredPolicy}, because running a policy over documents the caller supplied is
* ordinary editor enforcement that every member performs on upload and export.
*/
private void requirePolicySweepAllowed() {
if (!applicationProperties.getSecurity().isEnableLogin()) {
return;
}
if (!policyManagementAuthority.canTriggerPolicies()) {
throw new ResponseStatusException(
HttpStatus.FORBIDDEN,
"Not permitted to run this policy against its configured sources");
}
}
/**
* Whether the caller may create/modify policies (a team leader, or any operator when login is
* off).
*/
private boolean policyEditingAllowed() {
return !applicationProperties.getSecurity().isEnableLogin()
|| policyManagementAuthority.canEditPolicies();
}
@GetMapping
@Operation(
summary = "List policies",
@@ -571,8 +602,10 @@ public class PolicyController {
+ " the enabled flag (which only gates automatic triggering). Returns"
+ " the ids of the runs started (poll the run-status endpoint for each)"
+ " plus what the sweep skipped - already-processed, parked-by-failure,"
+ " and in-flight counts - so an empty result explains itself.")
+ " and in-flight counts - so an empty result explains itself. Requires"
+ " the policy-management role.")
public ResponseEntity<SweepOutcome> trigger(@PathVariable String policyId) {
requirePolicySweepAllowed();
Policy policy =
policyStore
.get(policyId)
@@ -650,6 +683,25 @@ public class PolicyController {
inputs.primary().size());
}
/**
* Resolve a test run's stored {@code asset:<id>} bindings from the saved policy the builder is
* editing, so their bytes need not be re-uploaded. Scoped to that policy (the resolver loads
* only the assets it references, in its own team) and gated to policy editors - the same
* authority that can read asset bytes - so a member can't rebind a policy's stored asset into
* an ad-hoc step to read it back. A blank id (an unsaved pipeline has no stored bindings) or an
* inaccessible policy leaves the run-supplied inputs untouched.
*/
private PolicyInputs resolveStoredAssets(String policyId, PolicyInputs inputs) {
if (policyId == null || policyId.isBlank() || !policyEditingAllowed()) {
return inputs;
}
return policyStore
.get(policyId)
.filter(policyAccessGuard::canAccess)
.map(policy -> assetResolver.resolve(policy, inputs))
.orElse(inputs);
}
/**
* Turn the typed run files into engine {@link PolicyInputs}: the primary documents plus the
* named supporting-file store, where each asset's {@code key} is the name a step references
@@ -0,0 +1,187 @@
package stirling.software.proprietary.service.ua;
import java.io.IOException;
import java.util.ArrayList;
import java.util.List;
import java.util.Set;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDDocumentCatalog;
import org.apache.pdfbox.pdmodel.interactive.viewerpreferences.PDViewerPreferences;
import org.springframework.stereotype.Service;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import stirling.software.common.util.ExceptionUtils;
import stirling.software.proprietary.model.api.ua.AccessibilityIssue;
import stirling.software.proprietary.model.api.ua.AccessibilityReport;
import stirling.software.proprietary.model.api.ua.FigureDescriptor;
import stirling.software.proprietary.model.api.ua.UaValidationResult;
import stirling.software.proprietary.pdf.ua.BBox;
import stirling.software.proprietary.pdf.ua.DocumentStructure;
import stirling.software.proprietary.pdf.ua.LayoutAnalyzer;
import stirling.software.proprietary.pdf.ua.PageContent;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.StructBlock;
import stirling.software.proprietary.pdf.ua.StructType;
import stirling.software.proprietary.pdf.ua.TaggedContentExtractor;
/** Produces an accessibility report without changing the document. */
@Service
@Slf4j
@RequiredArgsConstructor
public class AccessibilityAuditService {
/** Checks no validator can make; omitting them implies the work does not exist. */
private static final List<String> HUMAN_CHECKS =
List.of(
"Is the reading order correct for someone who cannot see the layout?",
"Does each alternative description convey what the image is for, not just what"
+ " it looks like?",
"Are headings used for structure rather than for visual emphasis?",
"Is any information conveyed by colour alone also available another way?",
"Do tables have headers that identify the right rows and columns?",
"Is the document language correct, including for quoted passages?",
"Do links describe their destination rather than saying 'click here'?");
/** The report walks every page and validates, so it carries the conversion's own caps. */
private static final long MAX_INPUT_BYTES = 100L * 1024 * 1024;
private static final int MAX_PAGES = 2000;
private final PdfUaValidationService validationService;
public AccessibilityReport audit(byte[] pdfBytes, PdfUaProfile profile) throws IOException {
enforceLimits(pdfBytes);
AccessibilityReport report = new AccessibilityReport();
report.setProfile(profile.displayName());
UaValidationResult validation = validationService.validate(pdfBytes, profile);
report.setIssues(validation.issues());
report.setPassesAutomatedChecks(validation.compliant());
report.setHumanChecks(HUMAN_CHECKS);
int fixable = 0;
int needsInput = 0;
for (AccessibilityIssue issue : validation.issues()) {
if (issue.isAutoFixable()) {
fixable++;
} else {
needsInput++;
}
}
report.setAutomaticallyFixable(fixable);
report.setNeedsInput(needsInput);
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
populateSummary(document, report);
report.setFiguresNeedingDescription(figuresNeedingDescription(document));
} catch (IOException e) {
log.debug("Could not inspect document for the summary: {}", e.getMessage());
}
return report;
}
/**
* Rejects before the expensive pass. An unreadable file is left to the report itself to say.
*/
private static void enforceLimits(byte[] pdfBytes) {
if (pdfBytes.length > MAX_INPUT_BYTES) {
throw ExceptionUtils.createIllegalArgumentException(
"error.fileTooLarge",
"This PDF is {0} MB. The accessibility report is limited to {1} MB.",
pdfBytes.length / (1024 * 1024),
MAX_INPUT_BYTES / (1024 * 1024));
}
int pages;
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
pages = document.getNumberOfPages();
} catch (IOException e) {
return;
}
if (pages > MAX_PAGES) {
throw ExceptionUtils.createIllegalArgumentException(
"error.tooManyPages",
"This PDF has {0} pages. The accessibility report is limited to {1} pages;"
+ " split it first.",
pages,
MAX_PAGES);
}
}
private void populateSummary(PDDocument document, AccessibilityReport report)
throws IOException {
PDDocumentCatalog catalog = document.getDocumentCatalog();
AccessibilityReport.Summary summary = report.getSummary();
report.setTagged(catalog.getStructureTreeRoot() != null);
report.setDeclaresConformance(declaresUa(document));
summary.setPages(document.getNumberOfPages());
summary.setEncrypted(document.isEncrypted());
summary.setHasLanguage(catalog.getLanguage() != null && !catalog.getLanguage().isBlank());
String title = document.getDocumentInformation().getTitle();
summary.setHasTitle(title != null && !title.isBlank());
PDViewerPreferences preferences = catalog.getViewerPreferences();
summary.setDisplaysDocTitle(preferences != null && preferences.displayDocTitle());
Set<String> unembedded = FontEmbeddingService.findUnembeddedFonts(document);
summary.setUnembeddedFonts(unembedded.size());
summary.setAllFontsEmbedded(unembedded.isEmpty());
try {
summary.setFigures(new TaggedContentExtractor().countGraphics(document));
} catch (Exception e) {
log.debug("Could not count figures: {}", e.getMessage());
}
}
/**
* Lists the figures a conversion would leave undescribed, running the converter's own analysis
* because counting raster images would miss vector charts and existing descriptions.
*/
private List<FigureDescriptor> figuresNeedingDescription(PDDocument document) {
try {
List<PageContent> pages = new TaggedContentExtractor().extract(document);
DocumentStructure structure = new LayoutAnalyzer().analyse(pages);
List<FigureDescriptor> figures = new ArrayList<>();
for (StructBlock block : structure.figuresWithoutAlt()) {
int ordinal = block.getRanges().isEmpty() ? -1 : block.getRanges().get(0).start();
BBox box = block.getBbox();
figures.add(
new FigureDescriptor(
block.getPageIndex() + ":" + ordinal,
block.getPageIndex() + 1,
block.getType() == StructType.FORMULA ? "formula" : "figure",
box.x0(),
box.y0(),
box.width(),
box.height(),
block.getAlt()));
}
return figures;
} catch (Exception e) {
log.debug("Could not enumerate figures: {}", e.getMessage());
return List.of();
}
}
/** True when the XMP packet carries a pdfuaid identifier. */
private static boolean declaresUa(PDDocument document) {
try {
var metadata = document.getDocumentCatalog().getMetadata();
if (metadata == null) {
return false;
}
String xmp =
new String(metadata.toByteArray(), java.nio.charset.StandardCharsets.UTF_8);
return xmp.contains("pdfuaid");
} catch (IOException e) {
return false;
}
}
}
@@ -0,0 +1,254 @@
package stirling.software.proprietary.service.ua;
import java.io.IOException;
import java.io.InputStream;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.HashSet;
import java.util.List;
import java.util.Set;
import java.util.stream.Stream;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDResources;
import org.apache.pdfbox.pdmodel.font.PDFont;
import org.springframework.stereotype.Service;
import lombok.extern.slf4j.Slf4j;
import stirling.software.common.util.ProcessExecutor;
import stirling.software.common.util.ProcessExecutor.ProcessExecutorResult;
/**
* Embeds any font the document references but does not carry, as PDF/UA-1 clause 7.21 requires.
* Ghostscript does the embedding and discards the structure tree, so this must run before tagging.
*/
@Service
@Slf4j
public class FontEmbeddingService {
public boolean hasUnembeddedFonts(byte[] pdfBytes) {
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
return !findUnembeddedFonts(document).isEmpty();
} catch (IOException e) {
log.debug("Could not inspect fonts: {}", e.getMessage());
return false;
}
}
public static Set<String> findUnembeddedFonts(PDDocument document) {
Set<String> missing = new HashSet<>();
for (PDPage page : document.getPages()) {
PDResources resources = page.getResources();
if (resources == null) {
continue;
}
for (COSName name : resources.getFontNames()) {
try {
PDFont font = resources.getFont(name);
if (font != null && !font.isEmbedded()) {
missing.add(font.getName());
}
} catch (IOException e) {
log.debug("Could not read font {}: {}", name.getName(), e.getMessage());
}
}
}
return missing;
}
/**
* Returns the document with all fonts embedded, or the input unchanged. Never throws: failing
* to embed is a reportable shortfall, not a reason to abandon the conversion.
*/
public Result embedFonts(byte[] pdfBytes) {
Set<String> missing;
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
missing = findUnembeddedFonts(document);
} catch (IOException e) {
return new Result(
pdfBytes, false, Set.of(), "Could not inspect fonts: " + e.getMessage());
}
if (missing.isEmpty()) {
return new Result(pdfBytes, false, Set.of(), null);
}
if (!isGhostscriptAvailable()) {
return new Result(
pdfBytes,
false,
missing,
"Ghostscript is not installed, so "
+ missing.size()
+ " unembedded font(s) could not be embedded. PDF/UA requires every font"
+ " to be embedded.");
}
Path workingDir = null;
try {
workingDir = Files.createTempDirectory("pdfua_fonts_");
Path input = workingDir.resolve("input.pdf");
Path output = workingDir.resolve("output.pdf");
Files.write(input, pdfBytes);
ProcessExecutorResult result =
ProcessExecutor.getInstance(ProcessExecutor.Processes.GHOSTSCRIPT)
.runCommandWithOutputHandling(command(input, output, workingDir));
if (result.getRc() != 0 || !Files.exists(output)) {
return new Result(
pdfBytes,
false,
missing,
"Font embedding failed with code " + result.getRc());
}
byte[] embedded = Files.readAllBytes(output);
// Ghostscript can exit 0 having written a blank page, so keep the original rather than
// return an empty document.
if (!survived(pdfBytes, embedded)) {
log.warn("Ghostscript produced a degenerate document; keeping the original");
return new Result(
pdfBytes,
false,
missing,
"Font embedding was skipped because the embedder returned a document that"
+ " had lost content. "
+ missing.size()
+ " font(s) remain unembedded.");
}
// It can also exit 0 while simply leaving fonts unembedded.
Set<String> remaining;
try (PDDocument check = Loader.loadPDF(embedded)) {
remaining = findUnembeddedFonts(check);
}
if (!remaining.isEmpty()) {
return new Result(
embedded,
true,
remaining,
remaining.size()
+ " font(s) could not be embedded ("
+ String.join(", ", remaining)
+ "). PDF/UA requires every font to be embedded.");
}
log.info("Embedded {} previously unembedded font(s)", missing.size());
return new Result(embedded, true, missing, null);
} catch (Exception e) {
log.warn("Font embedding failed: {}", e.getMessage());
return new Result(pdfBytes, false, missing, "Font embedding failed: " + e.getMessage());
} finally {
deleteQuietly(workingDir);
}
}
/**
* True when the rewritten document still holds the original's content. A collapse in page count
* or content-stream size is the only signature of a failed rewrite the exit code hides.
*/
private static boolean survived(byte[] original, byte[] rewritten) {
try (PDDocument before = Loader.loadPDF(original);
PDDocument after = Loader.loadPDF(rewritten)) {
if (after.getNumberOfPages() != before.getNumberOfPages()) {
return false;
}
long beforeBytes = contentBytes(before);
long afterBytes = contentBytes(after);
if (beforeBytes == 0) {
return true;
}
return afterBytes * 20L >= beforeBytes;
} catch (IOException e) {
log.debug("Could not compare documents after embedding: {}", e.getMessage());
return false;
}
}
private static long contentBytes(PDDocument document) {
long total = 0;
for (PDPage page : document.getPages()) {
try (InputStream in = page.getContents()) {
if (in != null) {
byte[] buffer = new byte[8192];
int read;
while ((read = in.read(buffer)) > 0) {
total += read;
}
}
} catch (IOException e) {
log.debug("Could not measure page content: {}", e.getMessage());
}
}
return total;
}
private static List<String> command(Path input, Path output, Path workingDir) {
List<String> command = new ArrayList<>();
command.add("gs");
command.add("--permit-file-read=" + workingDir.toAbsolutePath());
command.add("--permit-file-write=" + workingDir.toAbsolutePath());
command.add("-sDEVICE=pdfwrite");
command.add("-dEmbedAllFonts=true");
command.add("-dSubsetFonts=true");
command.add("-dCompressFonts=true");
command.add("-dNOSUBSTFONTS=false");
command.add("-dPDFSETTINGS=/prepress");
command.add("-dNOPAUSE");
command.add("-dBATCH");
command.add("-sOutputFile=" + output.toAbsolutePath());
command.add(input.toAbsolutePath().toString());
return command;
}
/** Cached after the first probe: availability does not change mid-process. */
private volatile Boolean ghostscriptAvailable;
private boolean isGhostscriptAvailable() {
Boolean cached = ghostscriptAvailable;
if (cached != null) {
return cached;
}
boolean available;
try {
ProcessExecutorResult result =
ProcessExecutor.getInstance(ProcessExecutor.Processes.GHOSTSCRIPT)
.runCommandWithOutputHandling(List.of("gs", "--version"));
available = result.getRc() == 0;
} catch (Exception e) {
log.debug("Ghostscript availability check failed: {}", e.getMessage());
available = false;
}
ghostscriptAvailable = available;
return available;
}
private static void deleteQuietly(Path directory) {
if (directory == null) {
return;
}
try (Stream<Path> stream = Files.walk(directory)) {
stream.sorted(Comparator.reverseOrder())
.forEach(
path -> {
try {
Files.deleteIfExists(path);
} catch (IOException e) {
log.debug("Could not delete {}", path);
}
});
} catch (IOException e) {
log.debug("Could not clean {}", directory);
}
}
/**
* @param warning non-null when fonts remain unembedded, for the conversion report
*/
public record Result(byte[] pdfBytes, boolean changed, Set<String> fonts, String warning) {}
}
@@ -0,0 +1,251 @@
package stirling.software.proprietary.service.ua;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.util.ArrayList;
import java.util.List;
import java.util.Locale;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.encryption.InvalidPasswordException;
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
import org.springframework.stereotype.Service;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
import stirling.software.proprietary.model.api.ua.UaValidationResult;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.PdfUaTagger;
import stirling.software.proprietary.pdf.ua.SourceFacts;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
import stirling.software.proprietary.pdf.ua.TaggingResult;
/**
* Converts a PDF to PDF/UA. The declaration is written first and withdrawn unless validation
* passes, so a returned file either conforms or does not claim to.
*/
@Service
@Slf4j
@RequiredArgsConstructor
public class PdfUaConversionService {
private final PdfUaValidationService validationService;
private final FontEmbeddingService fontEmbeddingService;
private final stirling.software.common.service.CustomPDFDocumentFactory pdfDocumentFactory;
/** Matches the cap GetInfoOnPDF already applies to comparable whole-document work. */
private static final long MAX_INPUT_BYTES = 100L * 1024 * 1024;
/** Beyond this the structure model alone runs to hundreds of megabytes. */
private static final int MAX_PAGES = 2000;
public PdfUaConversionOutcome convert(byte[] input, TaggingOptions options) throws IOException {
if (input.length > MAX_INPUT_BYTES) {
throw new IOException(
"This PDF is "
+ (input.length / (1024 * 1024))
+ " MB. PDF/UA conversion is limited to "
+ (MAX_INPUT_BYTES / (1024 * 1024))
+ " MB.");
}
PdfUaProfile profile = options.getProfile();
List<String> warnings = new ArrayList<>();
// Read the document's own facts before anything rewrites it. Font embedding runs
// Ghostscript over the whole file, which discards the structure tree, /Lang and XFA, so
// every guard and every "what did the source say" question must be answered from here.
SourceFacts facts;
try (PDDocument original = load(input)) {
rejectUnsupportedSource(original);
warnSignatures(original, warnings);
facts = SourceFacts.of(original);
}
byte[] source = input;
if (options.isEmbedFonts()) {
// Must precede tagging: the embedder rewrites the file and drops any structure tree.
FontEmbeddingService.Result fonts = fontEmbeddingService.embedFonts(input);
source = fonts.pdfBytes();
if (fonts.warning() != null) {
warnings.add(fonts.warning());
}
source = keepTagsOverFonts(input, source, facts, options, warnings);
}
TaggingOptions effective = options.toBuilder().sourceFacts(facts).build();
// Tag and declare in one pass; the claim is withdrawn below if validation disagrees.
byte[] declared;
TaggingResult taggingResult;
PdfUaTagger tagger = new PdfUaTagger();
try (PDDocument document = load(source)) {
rejectEncrypted(document);
taggingResult = tagger.tag(document, effective);
warnings.addAll(taggingResult.getWarnings());
tagger.declareConformance(document, profile);
declared = save(document);
}
UaValidationResult validation = validationService.validate(declared, profile);
// A validator cannot see text hidden behind artifact markers, so a clean verdict over
// suppressed content would be a false claim.
boolean honest = !taggingResult.isContentSuppressed();
if (validation.compliant() && honest) {
log.info("{} conversion passed validation", profile.displayName());
return new PdfUaConversionOutcome(
declared, true, validation, summary(taggingResult), warnings);
}
byte[] undeclared;
try (PDDocument document = load(declared)) {
tagger.withdrawConformance(document);
undeclared = save(document);
}
if (!validation.compliant()) {
warnings.add(
"The document could not be made "
+ profile.displayName()
+ " conformant, so no conformance claim was written. "
+ validation.totalFailures()
+ " automated check(s) still fail.");
}
log.info(
"{} conversion left undeclared: {} failures, suppressedText={}",
profile.displayName(),
validation.totalFailures(),
!honest);
return new PdfUaConversionOutcome(
undeclared, false, validation, summary(taggingResult), warnings);
}
private static PdfUaConversionOutcome.TaggingSummary summary(TaggingResult result) {
return new PdfUaConversionOutcome.TaggingSummary(
result.isRebuilt(),
result.getTaggedElements(),
result.getArtifacts(),
result.getFiguresNeedingAlt());
}
/**
* Tagging rewrites the content streams a signature covers, so the conversion still runs but the
* caller has to know the signature will no longer verify.
*/
private static void warnSignatures(PDDocument document, List<String> warnings) {
int signatures = document.getSignatureDictionaries().size();
if (signatures > 0) {
warnings.add(
signatures
+ " digital signature(s) will stop verifying: tagging rewrites the"
+ " content streams they cover. Convert first, then re-sign.");
}
}
/** Replaces PDFBox's "incorrect password" wording, baffling when the caller supplied none. */
private PDDocument load(byte[] bytes) throws IOException {
try {
// The factory spills large documents to a temp-file cache instead of the heap.
return pdfDocumentFactory.load(bytes);
} catch (IOException | RuntimeException e) {
// The factory wraps the parse failure, so check the cause chain rather than the type.
if (mentionsPassword(e)) {
throw new IOException(
"This PDF is encrypted. Remove the password before converting it to"
+ " PDF/UA.",
e);
}
throw e;
}
}
private static boolean mentionsPassword(Throwable error) {
for (Throwable cause = error; cause != null; cause = cause.getCause()) {
if (cause instanceof InvalidPasswordException) {
return true;
}
String message = cause.getMessage();
if (message != null) {
String lower = message.toLowerCase(Locale.ROOT);
if (lower.contains("password") || lower.contains("decrypt")) {
return true;
}
}
}
return false;
}
/** XFA is forbidden by PDF/UA-1 clause 7.15; encrypted or huge files cannot be restructured. */
/**
* Under KEEP nothing rebuilds a tree, so if the embedder deleted one we would hand back an
* untagged document. Fonts are not worth the whole structure; give the tags back instead.
*/
private byte[] keepTagsOverFonts(
byte[] input,
byte[] embedded,
SourceFacts facts,
TaggingOptions options,
List<String> warnings)
throws IOException {
if (options.getExistingTags() != TaggingOptions.ExistingTags.KEEP
|| !facts.hasUsableTree()
|| embedded == input) {
return embedded;
}
boolean survived;
try (PDDocument rewritten = load(embedded)) {
survived = PdfUaTagger.hasUsableStructureTree(rewritten);
}
if (survived) {
return embedded;
}
warnings.add(
"Embedding the missing fonts would have deleted the document's existing tags, so"
+ " the tags were kept and the fonts left unembedded. Turn off font"
+ " embedding to silence this, or rebuild the tags to embed them.");
return input;
}
/**
* Checks that must see the document as the author wrote it. Font embedding strips XFA, so
* running this afterwards would let a dynamic form through unnoticed, and it would push a
* document we are about to reject through the whole embedder first.
*/
private static void rejectUnsupportedSource(PDDocument document) throws IOException {
if (document.getNumberOfPages() > MAX_PAGES) {
throw new IOException(
"This PDF has "
+ document.getNumberOfPages()
+ " pages. PDF/UA conversion is limited to "
+ MAX_PAGES
+ " pages; split it first.");
}
PDAcroForm form = document.getDocumentCatalog().getAcroForm();
if (form != null && form.xfaIsDynamic()) {
throw new IOException(
"Dynamic XFA forms are not permitted by PDF/UA. Flatten the form first.");
}
}
/**
* Deliberately checked on the working document rather than the source. Permissions-only
* encryption with an empty user password is common in published documents, the embedder
* resolves it, and those files convert usefully; rejecting them up front would fail a document
* for a password its author never set.
*/
private static void rejectEncrypted(PDDocument document) throws IOException {
if (document.isEncrypted()) {
throw new IOException(
"Encrypted PDFs cannot be converted to PDF/UA. Remove the password first.");
}
}
private static byte[] save(PDDocument document) throws IOException {
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
return out.toByteArray();
}
}
@@ -0,0 +1,245 @@
package stirling.software.proprietary.service.ua;
import java.io.ByteArrayInputStream;
import java.util.ArrayList;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import org.springframework.stereotype.Service;
import org.verapdf.gf.foundry.VeraGreenfieldFoundryProvider;
import org.verapdf.pdfa.Foundries;
import org.verapdf.pdfa.PDFAParser;
import org.verapdf.pdfa.PDFAValidator;
import org.verapdf.pdfa.flavours.PDFAFlavour;
import org.verapdf.pdfa.results.TestAssertion;
import org.verapdf.pdfa.results.ValidationResult;
import jakarta.annotation.PostConstruct;
import lombok.extern.slf4j.Slf4j;
import stirling.software.proprietary.model.api.ua.AccessibilityIssue;
import stirling.software.proprietary.model.api.ua.UaValidationResult;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
/**
* Validates a document against a PDF/UA profile using veraPDF, the oracle a conversion is declared
* against. It checks only the machine-verifiable subset, so a clean result is not "accessible".
*/
@Service
@Slf4j
public class PdfUaValidationService {
/** Plain-English text and remediability for the clauses users actually hit. */
private static final Map<String, ClauseInfo> CLAUSES = buildClauseTable();
record ClauseInfo(String message, boolean autoFixable) {}
@PostConstruct
public void initialise() {
try {
VeraGreenfieldFoundryProvider.initialise();
} catch (Exception e) {
log.error("Failed to initialise veraPDF for PDF/UA validation", e);
}
}
public UaValidationResult validate(byte[] pdfBytes, PdfUaProfile profile) {
PDFAFlavour flavour = flavourFor(profile);
try (PDFAParser parser =
Foundries.defaultInstance()
.createParser(new ByteArrayInputStream(pdfBytes), flavour)) {
PDFAValidator validator = Foundries.defaultInstance().createValidator(flavour, false);
ValidationResult result = validator.validate(parser);
return toResult(profile, result);
} catch (Exception e) {
log.warn("PDF/UA validation failed for {}: {}", profile.displayName(), e.getMessage());
AccessibilityIssue issue = new AccessibilityIssue();
issue.setMessage("Validation could not run: " + e.getMessage());
issue.setSeverity("error");
issue.setClause("n/a");
return new UaValidationResult(profile.displayName(), false, List.of(issue), 0);
}
}
public static PDFAFlavour flavourFor(PdfUaProfile profile) {
return profile == PdfUaProfile.UA2 ? PDFAFlavour.PDFUA_2 : PDFAFlavour.PDFUA_1;
}
/**
* Whether the bytes really validate as PDF/A level A for the given part. Tagging is necessary
* for level A but not sufficient, so the claim is only written once veraPDF agrees.
*/
public boolean validatesAsPdfaLevelA(byte[] pdfBytes, int part) {
PDFAFlavour flavour =
switch (part) {
case 1 -> PDFAFlavour.PDFA_1_A;
case 2 -> PDFAFlavour.PDFA_2_A;
case 3 -> PDFAFlavour.PDFA_3_A;
default -> null;
};
if (flavour == null) {
log.warn("No PDF/A level A flavour for part {}", part);
return false;
}
try (PDFAParser parser =
Foundries.defaultInstance()
.createParser(new ByteArrayInputStream(pdfBytes), flavour)) {
PDFAValidator validator = Foundries.defaultInstance().createValidator(flavour, false);
return validator.validate(parser).isCompliant();
} catch (Exception e) {
log.warn("Level A validation could not run: {}", e.getMessage());
return false;
}
}
/**
* Groups repeated failures of the same rule so a report lists issues, not thousands of lines.
*/
private static UaValidationResult toResult(PdfUaProfile profile, ValidationResult result) {
Map<String, AccessibilityIssue> grouped = new LinkedHashMap<>();
int total = 0;
for (TestAssertion assertion : result.getTestAssertions()) {
if (assertion.getStatus() != TestAssertion.Status.FAILED) {
continue;
}
total++;
String clause =
assertion.getRuleId() != null ? assertion.getRuleId().getClause() : "unknown";
int test = assertion.getRuleId() != null ? assertion.getRuleId().getTestNumber() : 0;
String key = clause + "-" + test;
AccessibilityIssue issue =
grouped.computeIfAbsent(
key,
k -> {
AccessibilityIssue created = new AccessibilityIssue();
created.setClause(clause);
created.setTestNumber(String.valueOf(test));
created.setSeverity("error");
ClauseInfo info = lookupClause(clause);
created.setMessage(
info != null ? info.message() : assertion.getMessage());
created.setTechnicalMessage(assertion.getMessage());
created.setAutoFixable(info != null && info.autoFixable());
created.setSpecification(profile.displayName());
return created;
});
issue.setOccurrences(issue.getOccurrences() + 1);
if (issue.getLocation() == null && assertion.getLocation() != null) {
issue.setLocation(assertion.getLocation().toString());
}
}
List<AccessibilityIssue> issues = new ArrayList<>(grouped.values());
return new UaValidationResult(
profile.displayName(), result.isCompliant() && total == 0, issues, total);
}
/**
* Finds the most specific entry covering a clause by walking up the dotted hierarchy. String
* prefixes would be wrong: {@code 7.1} prefixes {@code 7.18.1} without being its ancestor.
*/
static ClauseInfo lookupClause(String clause) {
if (clause == null) {
return null;
}
String current = clause;
while (!current.isEmpty()) {
ClauseInfo info = CLAUSES.get(current);
if (info != null) {
return info;
}
int dot = current.lastIndexOf('.');
if (dot < 0) {
return null;
}
current = current.substring(0, dot);
}
return null;
}
private static Map<String, ClauseInfo> buildClauseTable() {
Map<String, ClauseInfo> table = new LinkedHashMap<>();
table.put(
"7.1",
new ClauseInfo(
"Document is not tagged, or some content is neither tagged nor marked as an artifact.",
true));
table.put(
"7.2",
new ClauseInfo(
"Text cannot be mapped to Unicode, or the document language is not declared.",
true));
table.put(
"7.3",
new ClauseInfo("An image or graphic has no alternative description.", false));
table.put(
"7.4",
new ClauseInfo(
"Heading levels skip a level, or headings are nested incorrectly.", true));
table.put(
"7.5",
new ClauseInfo("A table is missing header cells or header associations.", false));
table.put(
"7.6", new ClauseInfo("A list is not structured as list items with bodies.", true));
table.put(
"7.7",
new ClauseInfo("A mathematical expression has no alternative description.", false));
table.put(
"7.8",
new ClauseInfo("Running heads or page numbers are not marked as artifacts.", true));
table.put("7.9", new ClauseInfo("A note is missing a unique identifier.", true));
// Tagging does not touch optional content groups, so this needs the authoring tool.
table.put("7.10", new ClauseInfo("An optional content group has no name.", false));
// The attachment's own /AFRelationship and /Desc are not something tagging can supply.
table.put(
"7.11",
new ClauseInfo(
"An embedded file is missing its relationship or description.", false));
table.put(
"7.15",
new ClauseInfo(
"The document uses a dynamic XFA form, which PDF/UA does not allow.",
false));
table.put(
"7.16",
new ClauseInfo(
"Security settings prevent assistive technology from reading the content.",
true));
table.put("7.17", new ClauseInfo("Navigation aids such as page labels are missing.", true));
table.put(
"7.18",
new ClauseInfo(
"An annotation is missing a description, tab order, or structure entry.",
true));
table.put(
"7.20",
new ClauseInfo(
"A form or group XObject is not marked as content or as an artifact.",
false));
// Most font defects (CIDFont, CMap, metrics, encoding) need the font itself repaired.
table.put(
"7.21",
new ClauseInfo("A font in the document does not meet PDF/UA rules.", false));
// The one font defect embedding does fix.
table.put("7.21.4.1", new ClauseInfo("A font used in the document is not embedded.", true));
// ToUnicode gaps need the font itself repaired, which embedding does not do.
table.put(
"7.21.7",
new ClauseInfo(
"A font does not map every character it uses to Unicode, so extracted text"
+ " may be wrong.",
false));
table.put(
"5",
new ClauseInfo(
"The document does not declare PDF/UA conformance in its XMP metadata.",
true));
return table;
}
}
@@ -0,0 +1,297 @@
package stirling.software.proprietary.service.ua;
import java.io.ByteArrayInputStream;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.util.ArrayList;
import java.util.List;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdfwriter.compress.CompressParameters;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.common.PDMetadata;
import org.apache.xmpbox.XMPMetadata;
import org.apache.xmpbox.schema.PDFAExtensionSchema;
import org.apache.xmpbox.schema.PDFAIdentificationSchema;
import org.apache.xmpbox.type.AbstractStructuredType;
import org.apache.xmpbox.type.ArrayProperty;
import org.apache.xmpbox.type.Cardinality;
import org.apache.xmpbox.type.PDFAPropertyType;
import org.apache.xmpbox.type.PDFASchemaType;
import org.apache.xmpbox.xml.DomXmpParser;
import org.apache.xmpbox.xml.XmpSerializer;
import org.springframework.stereotype.Service;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import stirling.software.common.service.PdfaLevelAServiceInterface;
import stirling.software.proprietary.pdf.ua.PdfUaIdentificationSchema;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.PdfUaTagger;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
import stirling.software.proprietary.pdf.ua.TaggingResult;
/**
* Raises a PDF/A file from conformance level B to level A, which adds the tagging the PDF/UA tagger
* already does. Must run after Ghostscript, which discards any structure tree it is given.
*/
@Service
@Slf4j
@RequiredArgsConstructor
public class PdfaAccessibilityService implements PdfaLevelAServiceInterface {
/**
* Matches the PDF/UA converter's own cap; beyond this the structure model exhausts the heap.
*/
private static final int MAX_TAGGABLE_PAGES = 2000;
private final PdfUaValidationService validationService;
/**
* Tags a converted PDF/A and marks it conformance A, or returns it unchanged rather than
* claiming level A over untagged content. part is 1 to 3; part 1 keeps its PDF 1.4 version.
*/
public Result upgradeToLevelA(byte[] pdfBytes, int part, String language, String title) {
return upgradeToLevelA(pdfBytes, part, language, title, false);
}
/**
* @param alsoDeclareUa additionally claim PDF/UA, but only if it validates
*/
@Override
public Result upgradeToLevelA(
byte[] pdfBytes, int part, String language, String title, boolean alsoDeclareUa) {
List<String> warnings = new ArrayList<>();
try {
byte[] tagged;
TaggingResult taggingResult;
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
// Tagging holds a model of the whole document; without a cap a large file exhausts
// the heap, and OutOfMemoryError is an Error, so the catch below never sees it.
if (document.getNumberOfPages() > MAX_TAGGABLE_PAGES) {
warnings.add(
"This document has "
+ document.getNumberOfPages()
+ " pages, more than the "
+ MAX_TAGGABLE_PAGES
+ " that can be tagged, so it was left at conformance level B.");
return new Result(pdfBytes, false, warnings);
}
TaggingOptions options =
TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language(language)
.title(title)
.fallbackTitle(title)
// Fonts were embedded on the PDF/A pass; a rewrite would undo it.
.embedFonts(false)
// PDF/A-1 is defined on PDF 1.4; raising it breaks conformance.
.preservePdfVersion(part == 1)
.existingTags(TaggingOptions.ExistingTags.AUTO)
.build();
taggingResult = new PdfUaTagger().tag(document, options);
warnings.addAll(taggingResult.getWarnings());
tagged = save(document, part);
}
if (taggingResult.getTaggedElements() == 0 && taggingResult.isRebuilt()) {
warnings.add(
"No taggable content was found, so the file cannot claim PDF/A level A."
+ " It remains valid at level B.");
return new Result(pdfBytes, false, warnings);
}
if (taggingResult.isContentSuppressed()) {
warnings.add(
"Some text could not be tagged reliably and was marked as an artifact, so"
+ " no level A claim was written. The file remains valid at level B.");
return new Result(tagged, false, warnings);
}
byte[] declared = setConformance(tagged, part, "A");
// Tagging is necessary for level A but not sufficient: Unicode mappings are too.
if (!validationService.validatesAsPdfaLevelA(declared, part)) {
warnings.add(
"The document was tagged but does not validate as PDF/A-"
+ part
+ "a, so it was left at conformance level B.");
return new Result(setConformance(tagged, part, "B"), false, warnings);
}
if (alsoDeclareUa) {
byte[] withUa = declarePdfUaAlongsidePdfa(declared, part);
var uaResult = validationService.validate(withUa, PdfUaProfile.UA1);
if (uaResult.compliant()) {
log.info("Upgraded PDF/A-{} to level A and declared PDF/UA", part);
return new Result(withUa, true, warnings);
}
// The archival upgrade stands on its own; only the accessibility claim is dropped.
warnings.add(
"PDF/UA was requested alongside PDF/A but "
+ uaResult.totalFailures()
+ " accessibility check(s) still fail, so no PDF/UA claim was"
+ " written. The file is valid PDF/A-"
+ part
+ "a.");
}
log.info("Upgraded PDF/A-{} to conformance level A", part);
return new Result(declared, true, warnings);
} catch (Exception e) {
log.warn("Could not upgrade to PDF/A level A: {}", e.getMessage());
warnings.add(
"Level A upgrade failed ("
+ e.getMessage()
+ "), so the file was left at conformance level B.");
return new Result(pdfBytes, false, warnings);
}
}
/**
* Declares PDF/UA alongside PDF/A in one file. The extension schema is required: PDF/A forbids
* XMP properties no schema describes, and XMPBox has none for {@code pdfuaid}.
*/
static byte[] declarePdfUaAlongsidePdfa(byte[] pdfBytes, int part) throws Exception {
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
XMPMetadata xmp = parseOrCreate(document);
PdfUaIdentificationSchema identification = new PdfUaIdentificationSchema(xmp);
identification.setPart(1);
xmp.addSchema(identification);
addPdfUaExtensionSchema(xmp);
writeMetadata(document, xmp);
return save(document, part);
}
}
/**
* Describes the pdfuaid namespace so a PDF/A validator accepts it. Fields are set individually,
* not by subclassing: XMPBox reads the namespace from an annotation, which is not inherited.
*/
private static void addPdfUaExtensionSchema(XMPMetadata xmp) {
PDFAExtensionSchema extension =
(PDFAExtensionSchema) xmp.getSchema(PDFAExtensionSchema.class);
if (extension == null) {
extension = xmp.createAndAddPDFAExtensionSchemaWithDefaultNS();
}
PDFAPropertyType partProperty = new PDFAPropertyType(xmp);
addField(xmp, partProperty, PDFAPropertyType.NAME, "part");
addField(xmp, partProperty, PDFAPropertyType.VALUETYPE, "Integer");
addField(xmp, partProperty, PDFAPropertyType.CATEGORY, "internal");
addField(
xmp,
partProperty,
PDFAPropertyType.DESCRIPTION,
"Indicates which part of ISO 14289 the document conforms to");
PDFASchemaType schema = new PDFASchemaType(xmp);
addField(xmp, schema, PDFASchemaType.SCHEMA, "PDF/UA Universal Accessibility Schema");
addField(xmp, schema, PDFASchemaType.NAMESPACE_URI, PdfUaIdentificationSchema.NAMESPACE);
addField(xmp, schema, PDFASchemaType.PREFIX, PdfUaIdentificationSchema.PREFERRED_PREFIX);
ArrayProperty properties =
xmp.getTypeMapping()
.createArrayProperty(
schema.getNamespace(),
schema.getPrefix(),
PDFASchemaType.PROPERTY,
Cardinality.Seq);
properties.getContainer().addProperty(partProperty);
schema.getContainer().addProperty(properties);
// A freshly created extension schema has no schemas bag yet, so make one.
ArrayProperty schemas = extension.getSchemasProperty();
if (schemas == null) {
schemas =
xmp.getTypeMapping()
.createArrayProperty(
extension.getNamespace(),
extension.getPrefix(),
PDFAExtensionSchema.SCHEMAS,
Cardinality.Bag);
extension.addProperty(schemas);
}
schemas.getContainer().addProperty(schema);
}
/** Adds one text field to a structured type, in that type's own namespace. */
private static void addField(
XMPMetadata xmp, AbstractStructuredType target, String name, String value) {
target.getContainer()
.addProperty(
xmp.getTypeMapping()
.createText(
target.getNamespace(), target.getPrefix(), name, value));
}
private static XMPMetadata parseOrCreate(PDDocument document) throws Exception {
PDMetadata existing = document.getDocumentCatalog().getMetadata();
if (existing == null) {
return XMPMetadata.createXMPMetadata();
}
try (InputStream in = new ByteArrayInputStream(existing.toByteArray())) {
DomXmpParser parser = new DomXmpParser();
parser.setStrictParsing(false);
return parser.parse(in);
}
}
private static void writeMetadata(PDDocument document, XMPMetadata xmp) throws Exception {
ByteArrayOutputStream serialised = new ByteArrayOutputStream();
new XmpSerializer().serialize(xmp, serialised, true);
PDMetadata metadata = new PDMetadata(document);
metadata.importXMPMetadata(serialised.toByteArray());
document.getDocumentCatalog().setMetadata(metadata);
}
/** Rewrites {@code pdfaid:conformance} without disturbing the rest of the packet. */
static byte[] setConformance(byte[] pdfBytes, int part, String conformance) throws Exception {
try (PDDocument document = Loader.loadPDF(pdfBytes)) {
PDMetadata existing = document.getDocumentCatalog().getMetadata();
XMPMetadata xmp;
if (existing != null) {
try (InputStream in = new ByteArrayInputStream(existing.toByteArray())) {
DomXmpParser parser = new DomXmpParser();
parser.setStrictParsing(false);
xmp = parser.parse(in);
}
} else {
xmp = XMPMetadata.createXMPMetadata();
}
PDFAIdentificationSchema identification =
(PDFAIdentificationSchema) xmp.getSchema(PDFAIdentificationSchema.class);
if (identification == null) {
identification = xmp.createAndAddPDFAIdentificationSchema();
}
identification.setPart(part);
identification.setConformance(conformance);
ByteArrayOutputStream serialised = new ByteArrayOutputStream();
new XmpSerializer().serialize(xmp, serialised, true);
PDMetadata metadata = new PDMetadata(document);
metadata.importXMPMetadata(serialised.toByteArray());
document.getDocumentCatalog().setMetadata(metadata);
return save(document, part);
}
}
/**
* Part 1 is saved uncompressed: PDFBox's default object streams need PDF 1.5, which would push
* a PDF/A-1 file off its required 1.4 version.
*/
private static byte[] save(PDDocument document, int part) throws IOException {
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(
out, part == 1 ? CompressParameters.NO_COMPRESSION : new CompressParameters());
return out.toByteArray();
}
}
@@ -0,0 +1,286 @@
package stirling.software.proprietary.pdf.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Nested;
import org.junit.jupiter.api.Test;
/** Unit tests for the heuristics that decide what a run of text means. */
class LayoutAnalyzerTest {
private static final BBox A4 = new BBox(0, 0, 595, 842);
private static TextLineInfo line(String text, float size, float x, float y) {
return line(text, size, x, y, false, 0, 0);
}
private static TextLineInfo line(
String text, float size, float x, float y, boolean bold, int start, int end) {
List<WordInfo> words = new ArrayList<>();
float cursor = x;
for (String token : text.strip().split("\\s+")) {
float width = token.length() * size * 0.5f;
words.add(
new WordInfo(
token,
new BBox(cursor, y, cursor + width, y + size),
start,
end,
size,
bold));
cursor += width + size * 0.3f;
}
return new TextLineInfo(
0, text, new BBox(x, y, cursor, y + size), size, bold, start, end, false, words);
}
private static PageContent page(List<TextLineInfo> lines) {
return new PageContent(0, lines, List.of(), lines.size(), false, false, false, A4);
}
@Nested
@DisplayName("body font size")
class BodyFontSize {
@Test
@DisplayName("weights by characters so one huge title does not skew the baseline")
void weightsByCharacterCount() {
List<TextLineInfo> lines =
List.of(
line("A Very Large Title", 32, 50, 700),
line("Body text line one which is long", 11, 50, 650),
line("Body text line two which is long", 11, 50, 630),
line("Body text line three also long", 11, 50, 610));
assertEquals(11f, LayoutAnalyzer.bodyFontSize(List.of(page(lines))));
}
@Test
@DisplayName("returns zero when there is no text")
void handlesEmptyDocument() {
assertEquals(0f, LayoutAnalyzer.bodyFontSize(List.of(page(List.of()))));
}
}
@Nested
@DisplayName("heading detection")
class Headings {
@Test
@DisplayName("assigns distinct sizes to descending levels")
void assignsTiers() {
List<TextLineInfo> lines =
List.of(
line("Title", 24, 50, 800),
line("Chapter", 18, 50, 750),
line("Section", 14, 50, 700),
line("Body text that is long enough to set a baseline", 11, 50, 650));
Map<Float, Integer> tiers = LayoutAnalyzer.headingTiers(List.of(page(lines)), 11f);
assertEquals(1, tiers.get(24f));
assertEquals(2, tiers.get(18f));
assertEquals(3, tiers.get(14f));
assertNull(tiers.get(11f), "body size must not be a heading tier");
}
@Test
@DisplayName("rejects long lines and full sentences whatever their size")
void rejectsProse() {
assertFalse(
LayoutAnalyzer.isHeadingCandidate(
line("This line ends like a sentence does.", 20, 50, 700)),
"a line ending in a full stop reads as prose");
assertFalse(
LayoutAnalyzer.isHeadingCandidate(
line(
"one two three four five six seven eight nine ten eleven twelve"
+ " thirteen",
20,
50,
700)),
"a long line is body text however large");
assertTrue(LayoutAnalyzer.isHeadingCandidate(line("Financial Results", 20, 50, 700)));
}
@Test
@DisplayName("boldness alone never promotes a line to a heading")
void boldIsNotAHeadingSignal() {
List<TextLineInfo> lines =
List.of(
line("Bold Label", 11, 50, 700, true, 0, 0),
line("Body text long enough to set the baseline here", 11, 50, 650));
assertTrue(
LayoutAnalyzer.headingTiers(List.of(page(lines)), 11f).isEmpty(),
"a bold line at body size is emphasis, not a heading");
}
@Test
@DisplayName("rewrites skipped levels so H1 is never followed by H3")
void normalisesSkippedLevels() {
DocumentStructure structure = new DocumentStructure();
structure.add(new StructBlock(StructType.H1, 0));
structure.add(new StructBlock(StructType.H3, 0));
structure.add(new StructBlock(StructType.H4, 0));
LayoutAnalyzer.normaliseHeadingLevels(structure);
assertEquals(StructType.H1, structure.getBlocks().get(0).getType());
assertEquals(StructType.H2, structure.getBlocks().get(1).getType());
assertEquals(StructType.H3, structure.getBlocks().get(2).getType());
}
}
@Nested
@DisplayName("lists")
class Lists {
@Test
@DisplayName("recognises bullet and ordered markers")
void recognisesMarkers() {
assertTrue(LayoutAnalyzer.startsListItem(line("• First item", 11, 50, 700)));
assertTrue(LayoutAnalyzer.startsListItem(line("- First item", 11, 50, 700)));
assertTrue(LayoutAnalyzer.startsListItem(line("1. First item", 11, 50, 700)));
assertTrue(LayoutAnalyzer.startsListItem(line("a) First item", 11, 50, 700)));
assertFalse(LayoutAnalyzer.startsListItem(line("Ordinary prose here", 11, 50, 700)));
}
}
@Nested
@DisplayName("table cells")
class Tables {
@Test
@DisplayName("splits a row at wide gaps but not at ordinary word spacing")
void splitsOnWideGaps() {
List<WordInfo> words =
List.of(
new WordInfo("Region", new BBox(50, 700, 90, 711), 0, 0, 11, false),
new WordInfo("name", new BBox(93, 700, 125, 711), 0, 0, 11, false),
new WordInfo("Units", new BBox(250, 700, 285, 711), 1, 1, 11, false));
TextLineInfo row =
new TextLineInfo(
0,
"Region name Units",
new BBox(50, 700, 285, 711),
11,
false,
0,
1,
false,
words);
List<List<WordInfo>> cells = LayoutAnalyzer.splitCells(row);
assertEquals(2, cells.size(), "the small gap is a word space, the large one is a cell");
assertEquals(2, cells.get(0).size());
assertEquals("Units", cells.get(1).get(0).text());
}
@Test
@DisplayName("words sharing an operator cannot become separate cells")
void detectsInseparableWords() {
List<WordInfo> shared =
List.of(
new WordInfo("A", new BBox(50, 700, 60, 711), 3, 3, 11, false),
new WordInfo("B", new BBox(250, 700, 260, 711), 3, 3, 11, false));
TextLineInfo row =
new TextLineInfo(
0, "A B", new BBox(50, 700, 260, 711), 11, false, 3, 3, false, shared);
assertFalse(
row.wordsAreSeparable(),
"cells drawn by one operator cannot carry separate marked content ids");
}
}
@Nested
@DisplayName("running heads")
class RunningHeads {
@Test
@DisplayName("treats text repeating in the margin band across pages as an artifact")
void findsRepeatedMarginText() {
List<PageContent> pages = new ArrayList<>();
for (int i = 0; i < 4; i++) {
List<TextLineInfo> lines =
List.of(
line("Confidential Report", 9, 50, 800),
line("Body content for the page", 11, 50, 400),
line("Page " + (i + 1), 9, 300, 20));
pages.add(new PageContent(i, lines, List.of(), 3, false, false, false, A4));
}
Map<Integer, List<TextLineInfo>> artifacts = LayoutAnalyzer.repeatedMarginLines(pages);
assertEquals(
2, artifacts.get(0).size(), "the running head and the folio are artifacts");
assertTrue(
artifacts.get(0).stream().noneMatch(l -> l.text().contains("Body content")),
"body text must never be demoted to an artifact");
}
@Test
@DisplayName("does not treat a one-off margin line as a running head")
void ignoresUniqueMarginText() {
List<String> titles = List.of("Alpha", "Beta", "Gamma", "Delta");
List<PageContent> pages = new ArrayList<>();
for (int i = 0; i < 4; i++) {
List<TextLineInfo> lines =
List.of(
line(titles.get(i) + " overview", 9, 50, 800),
line("Body content", 11, 50, 400));
pages.add(new PageContent(i, lines, List.of(), 2, false, false, false, A4));
}
assertTrue(LayoutAnalyzer.repeatedMarginLines(pages).get(0).isEmpty());
}
@Test
@DisplayName("a large heading high on the page stays a heading, not chrome")
void doesNotDemoteHeadingsNearTheTop() {
List<PageContent> pages = new ArrayList<>();
for (int i = 0; i < 4; i++) {
List<TextLineInfo> lines =
List.of(
// Masking digits makes these look identical across pages.
line("Section " + (i + 1), 20, 50, 800),
line("Body text long enough to set the baseline", 11, 50, 400));
pages.add(new PageContent(i, lines, List.of(), 2, false, false, false, A4));
}
assertTrue(
LayoutAnalyzer.repeatedMarginLines(pages, 11f).get(0).isEmpty(),
"a heading larger than body text is content, wherever it sits");
}
}
@Nested
@DisplayName("columns")
class Columns {
@Test
@DisplayName("detects a gutter when text sits in two balanced blocks")
void detectsTwoColumns() {
List<TextLineInfo> lines = new ArrayList<>();
for (int i = 0; i < 6; i++) {
lines.add(line("Left column text", 10, 50, 700 - i * 14));
lines.add(line("Right column text", 10, 320, 700 - i * 14));
}
assertNotNull(LayoutAnalyzer.detectGutter(page(lines)));
}
@Test
@DisplayName("does not split a page whose lines span the full width")
void ignoresSingleColumn() {
List<TextLineInfo> lines = new ArrayList<>();
for (int i = 0; i < 10; i++) {
lines.add(
line(
"A full width line of prose that crosses the centre of the page",
10,
50,
700 - i * 14));
}
assertNull(LayoutAnalyzer.detectGutter(page(lines)));
}
}
}
@@ -0,0 +1,164 @@
package stirling.software.proprietary.pdf.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertSame;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import java.util.List;
import java.util.Map;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDFont;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
/** Tests the content-stream rewriting that makes tagging possible. */
class MarkedContentInjectorTest {
private static byte[] threeLinePdf() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDFont font = new PDType1Font(Standard14Fonts.FontName.HELVETICA);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
for (int i = 0; i < 3; i++) {
cs.beginText();
cs.setFont(font, 12);
cs.newLineAtOffset(50, 700 - i * 20);
cs.showText("Line " + i);
cs.endText();
}
}
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
return out.toByteArray();
}
}
private static String contentOf(PDDocument document) throws IOException {
try (InputStream in = document.getPage(0).getContents()) {
return new String(in.readAllBytes(), StandardCharsets.ISO_8859_1);
}
}
@Test
@DisplayName("wraps claimed content in BDC/EMC with a marked content id")
void wrapsClaimedContent() throws Exception {
try (PDDocument document = Loader.loadPDF(threeLinePdf())) {
StructBlock paragraph = new StructBlock(StructType.P, 0);
paragraph.addRange(0, 1);
int next =
new MarkedContentInjector()
.inject(document, document.getPage(0), List.of(paragraph), 0, true);
String content = contentOf(document);
assertTrue(content.contains("/P"), "the structure type was not written");
assertTrue(content.contains("/MCID"), "no marked content id was written");
assertTrue(content.contains("BDC"), "no marked content sequence was opened");
assertTrue(content.contains("EMC"), "no marked content sequence was closed");
assertFalse(paragraph.getMcids().isEmpty(), "the block was given no marked content id");
assertTrue(next > 0, "the id counter did not advance");
}
}
@Test
@DisplayName("marks unclaimed content as an artifact so nothing is left untagged")
void unclaimedContentBecomesArtifact() throws Exception {
try (PDDocument document = Loader.loadPDF(threeLinePdf())) {
StructBlock paragraph = new StructBlock(StructType.P, 0);
paragraph.addRange(0, 0);
new MarkedContentInjector()
.inject(document, document.getPage(0), List.of(paragraph), 0, true);
String content = contentOf(document);
assertTrue(
content.contains("/Artifact"),
"content nobody claimed must be marked as an artifact, or PDF/UA clause 7.1"
+ " fails");
}
}
@Test
@DisplayName("opens and closes sequences in balanced pairs")
void sequencesAreBalanced() throws Exception {
try (PDDocument document = Loader.loadPDF(threeLinePdf())) {
StructBlock first = new StructBlock(StructType.P, 0);
first.addRange(0, 0);
StructBlock second = new StructBlock(StructType.H1, 0);
second.addRange(2, 2);
new MarkedContentInjector()
.inject(document, document.getPage(0), List.of(first, second), 0, true);
String content = contentOf(document);
int opens = count(content, "BDC") + count(content, "BMC");
int closes = count(content, "EMC");
assertEquals(opens, closes, "every opened sequence must be closed");
}
}
@Test
@DisplayName("rewriting does not change what a reader extracts")
void textIsUnchanged() throws Exception {
byte[] original = threeLinePdf();
String before = extract(original);
byte[] rewritten;
try (PDDocument document = Loader.loadPDF(original)) {
StructBlock paragraph = new StructBlock(StructType.P, 0);
paragraph.addRange(0, 2);
new MarkedContentInjector()
.inject(document, document.getPage(0), List.of(paragraph), 0, true);
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
rewritten = out.toByteArray();
}
assertEquals(before, extract(rewritten), "marked content operators must not render");
}
@Test
@DisplayName("two blocks claiming the same content keep the first, not both")
void overlappingClaimsAreResolved() {
StructBlock first = new StructBlock(StructType.P, 0);
first.addRange(0, 2);
StructBlock second = new StructBlock(StructType.H1, 0);
second.addRange(1, 1);
Map<Integer, StructBlock> owners =
MarkedContentInjector.ownersByOrdinal(List.of(first, second));
assertSame(first, owners.get(1), "the first claim wins so reading order stays unambiguous");
assertEquals(3, owners.size());
}
private static int count(String haystack, String needle) {
int total = 0;
int index = 0;
while ((index = haystack.indexOf(needle, index)) >= 0) {
total++;
index += needle.length();
}
return total;
}
private static String extract(byte[] pdf) throws IOException {
try (PDDocument document = Loader.loadPDF(pdf)) {
return new org.apache.pdfbox.text.PDFTextStripper()
.getText(document)
.replaceAll("\\s+", " ")
.strip();
}
}
}
@@ -0,0 +1,193 @@
package stirling.software.proprietary.pdf.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import java.util.List;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.common.PDStream;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
/** Regression tests for rewriter damage a validator cannot see, so it still passes validation. */
class MarkedContentSafetyTest {
private static String contentOf(PDDocument document) throws IOException {
try (InputStream in = document.getPage(0).getContents()) {
return new String(in.readAllBytes(), StandardCharsets.ISO_8859_1);
}
}
private static void setContent(PDDocument document, String content) throws IOException {
PDStream stream = new PDStream(document);
try (var out = stream.createOutputStream()) {
out.write(content.getBytes(StandardCharsets.ISO_8859_1));
}
document.getPage(0).setContents(stream);
}
private static PDDocument onePage() throws IOException {
PDDocument document = new PDDocument();
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(50, 700);
cs.showText("visible");
cs.endText();
}
return document;
}
@Test
@DisplayName("an optional-content layer survives the rebuild, so hidden content stays hidden")
void optionalContentIsPreserved() throws Exception {
try (PDDocument document = onePage()) {
String original = contentOf(document);
setContent(document, "/OC /MC0 BDC\n" + original + "\nEMC\n");
StructBlock paragraph = new StructBlock(StructType.P, 0);
paragraph.addRange(0, 0);
new MarkedContentInjector()
.inject(document, document.getPage(0), List.of(paragraph), 0, true);
String rewritten = contentOf(document);
assertTrue(
rewritten.contains("/OC"),
"the optional-content wrapper was stripped, which would make a hidden"
+ " DRAFT/CONFIDENTIAL or redaction layer permanently visible:\n"
+ rewritten);
assertEquals(
countOf(rewritten, "BDC") + countOf(rewritten, "BMC"),
countOf(rewritten, "EMC"),
"marked content is unbalanced after preserving the layer");
}
}
@Test
@DisplayName("replacement text survives the rebuild so ligatures still read correctly")
void actualTextIsPreserved() throws Exception {
try (PDDocument document = onePage()) {
String original = contentOf(document);
// A generator marks an ffi ligature with what it really spells.
setContent(
document, "/Span <</ActualText (ffi) /MCID 7>> BDC\n" + original + "\nEMC\n");
StructBlock paragraph = new StructBlock(StructType.P, 0);
paragraph.addRange(0, 0);
new MarkedContentInjector()
.inject(document, document.getPage(0), List.of(paragraph), 0, true);
String rewritten = contentOf(document);
assertTrue(
rewritten.contains("ActualText"),
"dropping ActualText leaves a screen reader announcing the raw glyph:\n"
+ rewritten);
assertFalse(
rewritten.contains("/MCID 7"),
"the source's own marked content id is meaningless after a rebuild");
assertEquals(
countOf(rewritten, "BDC") + countOf(rewritten, "BMC"),
countOf(rewritten, "EMC"),
"marked content is unbalanced after preserving replacement text");
}
}
@Test
@DisplayName("a sequence wrapping a fill opens before the path, not inside it")
void markedContentNeverOpensInsideAPathObject() throws Exception {
try (PDDocument document = onePage()) {
setContent(document, "0 0 0 rg\n10 10 50 5 re\nf\n");
new MarkedContentInjector().inject(document, document.getPage(0), List.of(), 0, true);
String rewritten = contentOf(document);
int reAt = rewritten.indexOf(" re");
int openAt = Math.max(rewritten.indexOf("BMC"), rewritten.indexOf("BDC"));
assertTrue(openAt >= 0, "no sequence was opened at all: " + rewritten);
assertTrue(
openAt < reAt,
"ISO 32000-1 does not permit a marked-content operator inside a path object;"
+ " the sequence must open before the path construction:\n"
+ rewritten);
}
}
@Test
@DisplayName("words drawn out of stream order are still claimed, not silently artifacted")
void outOfOrderWordsAreClaimed() {
// A line whose second word on the page was painted first: ordinals 1 then 0.
WordInfo right = new WordInfo("label", new BBox(50, 700, 90, 712), 1, 1, 11, false);
WordInfo left = new WordInfo("value", new BBox(200, 700, 240, 712), 0, 0, 11, false);
TextLineInfo line =
new TextLineInfo(
0,
"label value",
new BBox(50, 700, 240, 712),
11,
false,
0,
1,
false,
List.of(right, left));
PageContent page =
new PageContent(
0,
List.of(line),
List.of(),
2,
false,
false,
false,
new BBox(0, 0, 595, 842));
DocumentStructure structure = new LayoutAnalyzer().analyse(List.of(page));
boolean[] claimed = new boolean[2];
structure.visit(
block -> {
if (block.isArtifact()) {
return;
}
block.getRanges()
.forEach(
r -> {
for (int i = r.start(); i <= r.end() && i < 2; i++) {
claimed[i] = true;
}
});
});
assertTrue(
claimed[0] && claimed[1],
"an out-of-order word was left unclaimed and would be hidden from assistive"
+ " technology while the file still validated");
}
private static int countOf(String haystack, String needle) {
int total = 0;
int index = 0;
while ((index = haystack.indexOf(needle, index)) >= 0) {
total++;
index += needle.length();
}
return total;
}
private static byte[] bytes(PDDocument document) throws IOException {
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
return out.toByteArray();
}
}
@@ -0,0 +1,171 @@
package stirling.software.proprietary.pdf.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.IOException;
import java.nio.charset.StandardCharsets;
import java.util.List;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureElement;
import org.apache.pdfbox.pdmodel.documentinterchange.logicalstructure.PDStructureNode;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationWidget;
import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
import org.apache.pdfbox.pdmodel.interactive.form.PDTextField;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
/** Covers form-field descriptions, widget nesting and withdrawing a conformance claim. */
class PdfUaFormAndDeclarationTest {
/** A document with one named text field and one unnamed one. */
private static PDDocument formDocument(boolean nameTheSecondField) throws IOException {
PDDocument document = new PDDocument();
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDAcroForm form = new PDAcroForm(document);
document.getDocumentCatalog().setAcroForm(form);
PDTextField named = new PDTextField(form);
named.setPartialName("EmailAddress");
addWidget(named, page, 700);
form.getFields().add(named);
PDTextField second = new PDTextField(form);
if (nameTheSecondField) {
second.setPartialName("PostCode");
}
addWidget(second, page, 650);
form.getFields().add(second);
return document;
}
private static void addWidget(PDTextField field, PDPage page, float y) throws IOException {
PDAnnotationWidget widget = field.getWidgets().get(0);
PDRectangle rectangle = new PDRectangle();
rectangle.setLowerLeftX(50);
rectangle.setLowerLeftY(y);
rectangle.setUpperRightX(250);
rectangle.setUpperRightY(y + 18);
widget.setRectangle(rectangle);
widget.setPage(page);
page.getAnnotations().add(widget);
}
private static String xmpOf(PDDocument document) throws IOException {
var metadata = document.getDocumentCatalog().getMetadata();
assertNotNull(metadata, "no XMP packet");
return new String(metadata.toByteArray(), StandardCharsets.UTF_8);
}
@Test
@DisplayName("a form field gets its tooltip from its own name, not an invented one")
void derivesTooltipFromFieldName() throws Exception {
try (PDDocument document = formDocument(true)) {
List<String> warnings =
new PdfUaMetadataWriter()
.applyDocumentRequirements(document, "Form", "en", PdfUaProfile.UA1);
PDAcroForm form = document.getDocumentCatalog().getAcroForm();
assertEquals("EmailAddress", form.getField("EmailAddress").getAlternateFieldName());
assertEquals("PostCode", form.getField("PostCode").getAlternateFieldName());
assertTrue(warnings.isEmpty(), "nothing needed reporting: " + warnings);
}
}
@Test
@DisplayName("a field with no name is reported rather than given a placeholder tooltip")
void reportsUnnameableField() throws Exception {
try (PDDocument document = formDocument(false)) {
List<String> warnings =
new PdfUaMetadataWriter()
.applyDocumentRequirements(document, "Form", "en", PdfUaProfile.UA1);
assertEquals(1, warnings.size());
assertTrue(warnings.get(0).contains("form field"), warnings.get(0));
}
}
@Test
@DisplayName("an existing description is never overwritten")
void keepsExistingDescription() throws Exception {
try (PDDocument document = formDocument(true)) {
PDAcroForm form = document.getDocumentCatalog().getAcroForm();
form.getField("EmailAddress").setAlternateFieldName("Your email address");
new PdfUaMetadataWriter()
.applyDocumentRequirements(document, "Form", "en", PdfUaProfile.UA1);
assertEquals(
"Your email address", form.getField("EmailAddress").getAlternateFieldName());
}
}
@Test
@DisplayName("widget annotations are nested inside a Form structure element")
void widgetsAreNestedInFormElements() throws Exception {
try (PDDocument document = formDocument(true)) {
DocumentStructure structure = new DocumentStructure();
StructBlock paragraph = new StructBlock(StructType.P, 0);
paragraph.getMcids().add(0);
structure.add(paragraph);
new StructTreeWriter().write(document, structure, PdfUaProfile.UA1);
var root = document.getDocumentCatalog().getStructureTreeRoot();
assertTrue(
typesUnder(root).contains("Form"),
"clause 7.18.4 requires a widget to sit inside a Form element, found: "
+ typesUnder(root));
}
}
private static List<String> typesUnder(PDStructureNode node) {
List<String> types = new java.util.ArrayList<>();
for (Object kid : node.getKids()) {
if (kid instanceof PDStructureElement element) {
types.add(element.getStructureType());
types.addAll(typesUnder(element));
}
}
return types;
}
@Test
@DisplayName("withdrawing conformance removes the claim but keeps the other metadata")
void withdrawingConformanceRemovesOnlyTheClaim() throws Exception {
try (PDDocument document = new PDDocument()) {
document.addPage(new PDPage());
PdfUaTagger tagger = new PdfUaTagger();
PdfUaMetadataWriter writer = new PdfUaMetadataWriter();
writer.applyDocumentRequirements(document, "Kept Title", "en-GB", PdfUaProfile.UA1);
writer.declareConformance(document, PdfUaProfile.UA1);
assertTrue(xmpOf(document).contains("pdfuaid"));
tagger.withdrawConformance(document);
String xmp = xmpOf(document);
assertFalse(xmp.contains("pdfuaid"), "the conformance claim should be gone");
assertTrue(xmp.contains("Kept Title"), "the title should survive");
assertEquals("en-GB", document.getDocumentCatalog().getLanguage());
}
}
@Test
@DisplayName("withdrawing conformance on a document that never claimed it is harmless")
void withdrawingIsIdempotent() throws Exception {
try (PDDocument document = new PDDocument()) {
document.addPage(new PDPage());
new PdfUaMetadataWriter()
.applyDocumentRequirements(document, "Title", "en", PdfUaProfile.UA1);
new PdfUaTagger().withdrawConformance(document);
assertFalse(xmpOf(document).contains("pdfuaid"));
}
}
}
@@ -0,0 +1,79 @@
package stirling.software.proprietary.pdf.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertTrue;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
/**
* A relabelled language is invisible to every validator, so the tagger must not guess over one the
* document already declares.
*/
class PdfUaLanguageTest {
private static PDDocument documentWithLanguage(String language) {
PDDocument document = new PDDocument();
document.addPage(new PDPage());
if (language != null) {
document.getDocumentCatalog().setLanguage(language);
}
return document;
}
private static TaggingOptions.TaggingOptionsBuilder options() {
return TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.title("Rapport")
.embedFonts(false);
}
@Test
@DisplayName("keeps the language the document already declares")
void keepsExistingLanguage() throws Exception {
try (PDDocument document = documentWithLanguage("fr-FR")) {
TaggingResult result = new PdfUaTagger().tag(document, options().build());
assertEquals("fr-FR", document.getDocumentCatalog().getLanguage());
assertTrue(
result.getWarnings().stream().anyMatch(w -> w.contains("fr-FR")),
"ignoring the requested language must be reported: " + result.getWarnings());
}
}
@Test
@DisplayName("applies the requested language when the document declares none")
void fillsInMissingLanguage() throws Exception {
try (PDDocument document = documentWithLanguage(null)) {
new PdfUaTagger().tag(document, options().build());
assertEquals("en-GB", document.getDocumentCatalog().getLanguage());
}
}
@Test
@DisplayName("replaces the declared language only when the caller asks")
void overridesOnRequest() throws Exception {
try (PDDocument document = documentWithLanguage("fr-FR")) {
new PdfUaTagger().tag(document, options().overrideLanguage(true).build());
assertEquals("en-GB", document.getDocumentCatalog().getLanguage());
}
}
@Test
@DisplayName("keeps the existing language when an existing structure tree is left alone")
void keepsExistingLanguageWithoutRebuilding() throws Exception {
try (PDDocument document = documentWithLanguage("de-DE")) {
new PdfUaTagger()
.tag(
document,
options().existingTags(TaggingOptions.ExistingTags.KEEP).build());
assertEquals("de-DE", document.getDocumentCatalog().getLanguage());
}
}
}
@@ -0,0 +1,137 @@
package stirling.software.proprietary.pdf.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.nio.charset.StandardCharsets;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
/** Tests the document-level requirements that have nothing to do with tagging. */
class PdfUaMetadataWriterTest {
private static PDDocument twoPageDocument() {
PDDocument document = new PDDocument();
document.addPage(new PDPage());
document.addPage(new PDPage());
return document;
}
private static String xmpOf(PDDocument document) throws IOException {
var metadata = document.getDocumentCatalog().getMetadata();
assertNotNull(metadata, "no XMP packet was written");
return new String(metadata.toByteArray(), StandardCharsets.UTF_8);
}
@Test
@DisplayName("sets title, language, tab order and the display-title flag")
void appliesDocumentRequirements() throws Exception {
try (PDDocument document = twoPageDocument()) {
new PdfUaMetadataWriter()
.applyDocumentRequirements(
document, "Annual Report", "en-GB", PdfUaProfile.UA1);
assertEquals("en-GB", document.getDocumentCatalog().getLanguage());
assertEquals("Annual Report", document.getDocumentInformation().getTitle());
assertTrue(
document.getDocumentCatalog().getViewerPreferences().displayDocTitle(),
"without DisplayDocTitle a viewer shows the filename instead of the title");
for (PDPage page : document.getPages()) {
assertEquals(
"S",
page.getCOSObject().getNameAsString(COSName.getPDFName("Tabs")),
"clause 7.18.1 requires an explicit tab order on every page");
}
}
}
@Test
@DisplayName("writes dc:title into the XMP packet, not just the info dictionary")
void writesDublinCoreTitle() throws Exception {
try (PDDocument document = twoPageDocument()) {
new PdfUaMetadataWriter()
.applyDocumentRequirements(
document, "Annual Report", "en-GB", PdfUaProfile.UA1);
assertTrue(xmpOf(document).contains("Annual Report"));
}
}
@Test
@DisplayName("does not declare conformance as part of applying requirements")
void doesNotDeclareEarly() throws Exception {
try (PDDocument document = twoPageDocument()) {
new PdfUaMetadataWriter()
.applyDocumentRequirements(document, "Report", "en", PdfUaProfile.UA1);
assertFalse(
xmpOf(document).contains("pdfuaid"),
"the conformance claim must wait until validation has passed");
}
}
@Test
@DisplayName("declaring conformance writes pdfuaid with the right part")
void declaresConformance() throws Exception {
try (PDDocument document = twoPageDocument()) {
PdfUaMetadataWriter writer = new PdfUaMetadataWriter();
writer.applyDocumentRequirements(document, "Report", "en", PdfUaProfile.UA1);
writer.declareConformance(document, PdfUaProfile.UA1);
String xmp = xmpOf(document);
assertTrue(xmp.contains("pdfuaid"), "no PDF/UA identifier was written");
assertTrue(xmp.contains("part"), "no conformance part was written");
}
}
@Test
@DisplayName("UA-2 raises the PDF version to 2.0")
void ua2RaisesVersion() throws Exception {
try (PDDocument document = twoPageDocument()) {
new PdfUaMetadataWriter()
.applyDocumentRequirements(document, "Report", "en", PdfUaProfile.UA2);
assertEquals(2.0f, document.getVersion());
}
}
@Test
@DisplayName("keeps an existing title when none is supplied")
void keepsExistingTitle() throws Exception {
try (PDDocument document = twoPageDocument()) {
var info = document.getDocumentInformation();
info.setTitle("Original Title");
document.setDocumentInformation(info);
new PdfUaMetadataWriter()
.applyDocumentRequirements(document, null, "en", PdfUaProfile.UA1);
assertEquals("Original Title", document.getDocumentInformation().getTitle());
}
}
@Test
@DisplayName("survives a round trip through save and reload")
void survivesRoundTrip() throws Exception {
byte[] saved;
try (PDDocument document = twoPageDocument()) {
PdfUaMetadataWriter writer = new PdfUaMetadataWriter();
writer.applyDocumentRequirements(document, "Round Trip", "fr-FR", PdfUaProfile.UA1);
writer.declareConformance(document, PdfUaProfile.UA1);
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
saved = out.toByteArray();
}
try (PDDocument reloaded = Loader.loadPDF(saved)) {
assertEquals("fr-FR", reloaded.getDocumentCatalog().getLanguage());
assertEquals("Round Trip", reloaded.getDocumentInformation().getTitle());
assertTrue(xmpOf(reloaded).contains("pdfuaid"));
}
}
}
@@ -0,0 +1,160 @@
package stirling.software.proprietary.pdf.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertTrue;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Nested;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.params.ParameterizedTest;
import org.junit.jupiter.params.provider.ValueSource;
/** Tests for the small types the tagger is built from. */
class PdfUaModelTest {
@Nested
@DisplayName("structure types")
class Types {
@Test
@DisplayName("maps levels to heading tags and back")
void headingLevelsRoundTrip() {
for (int level = 1; level <= 6; level++) {
assertEquals(level, StructType.heading(level).headingLevel());
assertEquals("H" + level, StructType.heading(level).tag());
}
}
@Test
@DisplayName("clamps out-of-range levels rather than throwing")
void clampsLevels() {
assertEquals(StructType.H1, StructType.heading(0));
assertEquals(StructType.H1, StructType.heading(-3));
assertEquals(StructType.H6, StructType.heading(9));
}
@Test
@DisplayName("reports zero for types that are not headings")
void nonHeadingsHaveNoLevel() {
assertEquals(0, StructType.P.headingLevel());
assertFalse(StructType.TABLE.isHeading());
}
}
@Nested
@DisplayName("markable operators")
class Markable {
@ParameterizedTest
@ValueSource(strings = {"Tj", "TJ", "'", "\"", "Do", "BI", "S", "f", "f*", "B", "sh"})
@DisplayName("counts text, XObjects and path painting")
void counted(String operator) {
assertTrue(MarkableOp.isMarkableOperator(operator), operator + " should be markable");
}
@ParameterizedTest
@ValueSource(strings = {"q", "Q", "cm", "BT", "ET", "Tf", "Td", "n", "W", "gs", "re"})
@DisplayName("ignores operators that paint nothing")
void notCounted(String operator) {
assertFalse(
MarkableOp.isMarkableOperator(operator), operator + " should not be markable");
}
@Test
@DisplayName("n ends a path without painting, so it is not content")
void pathEndIsNotPainting() {
assertFalse(MarkableOp.isPathPainting("n"));
assertTrue(MarkableOp.isPathPainting("f"));
}
}
@Nested
@DisplayName("profiles")
class Profiles {
@Test
@DisplayName("parses the shapes a caller might send")
void parsesRequestValues() {
assertEquals(PdfUaProfile.UA1, PdfUaProfile.fromRequest("ua1"));
assertEquals(PdfUaProfile.UA1, PdfUaProfile.fromRequest(null));
assertEquals(PdfUaProfile.UA1, PdfUaProfile.fromRequest(""));
assertEquals(PdfUaProfile.UA1, PdfUaProfile.fromRequest("nonsense"));
assertEquals(PdfUaProfile.UA2, PdfUaProfile.fromRequest("ua2"));
assertEquals(PdfUaProfile.UA2, PdfUaProfile.fromRequest("PDF/UA-2"));
}
@Test
@DisplayName("UA-2 requires PDF 2.0")
void ua2NeedsPdf2() {
assertEquals(2.0f, PdfUaProfile.UA2.pdfVersion());
assertEquals(1.7f, PdfUaProfile.UA1.pdfVersion());
}
}
@Nested
@DisplayName("bounding boxes")
class Boxes {
@Test
@DisplayName("union of an empty box is the other box")
void unionWithEmpty() {
BBox box = new BBox(10, 10, 20, 20);
assertEquals(box, box.union(BBox.EMPTY));
assertEquals(box, BBox.EMPTY.union(box));
}
@Test
@DisplayName("union covers both boxes")
void unionCoversBoth() {
BBox union = new BBox(0, 0, 10, 10).union(new BBox(20, 5, 30, 25));
assertEquals(new BBox(0, 0, 30, 25), union);
}
@Test
@DisplayName("reports horizontal overlap as a fraction of the narrower box")
void overlapIsRelative() {
BBox wide = new BBox(0, 0, 100, 10);
BBox narrow = new BBox(40, 0, 60, 10);
assertEquals(1.0f, wide.horizontalOverlap(narrow));
assertEquals(0f, wide.horizontalOverlap(new BBox(200, 0, 220, 10)));
}
}
@Nested
@DisplayName("structure blocks")
class Blocks {
@Test
@DisplayName("counts content across the whole subtree")
void countsDescendantContent() {
StructBlock table = new StructBlock(StructType.TABLE, 0);
StructBlock row = new StructBlock(StructType.TR, 0);
StructBlock cell = new StructBlock(StructType.TD, 0);
cell.addRange(3, 5);
row.addChild(cell);
table.addChild(row);
assertEquals(3, table.contentCount());
}
@Test
@DisplayName("collects text in tree order")
void collectsText() {
StructBlock list = new StructBlock(StructType.L, 0);
StructBlock first = new StructBlock(StructType.LI, 0);
first.setText("one");
StructBlock second = new StructBlock(StructType.LI, 0);
second.setText("two");
list.addChild(first).addChild(second);
assertEquals("one two", list.collectText());
}
@Test
@DisplayName("an artifact is not a structure element")
void artifactsAreDistinct() {
StructBlock artifact = StructBlock.artifact(ArtifactType.PAGINATION, 0);
assertTrue(artifact.isArtifact());
assertEquals("Pagination", artifact.getArtifactType().subtype());
}
}
}
@@ -0,0 +1,155 @@
package stirling.software.proprietary.pdf.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.util.ArrayList;
import java.util.List;
import java.util.Map;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
/** Tests the heuristics that decide what counts as a drawing and what counts as a heading. */
class VectorAndHeadingTest {
private static final BBox A4 = new BBox(0, 0, 595, 842);
private static TextLineInfo line(String text, float size, float x, float y) {
List<WordInfo> words = new ArrayList<>();
float cursor = x;
for (String token : text.strip().split("\\s+")) {
float width = token.length() * size * 0.5f;
words.add(
new WordInfo(
token,
new BBox(cursor, y, cursor + width, y + size),
0,
0,
size,
false));
cursor += width + size * 0.3f;
}
return new TextLineInfo(
0, text, new BBox(x, y, cursor, y + size), size, false, 0, 0, false, words);
}
private static MarkableOp vector(int ordinal, BBox box) {
return new MarkableOp(ordinal, MarkableOp.Kind.VECTOR, box, null);
}
private static DocumentStructure analyse(List<TextLineInfo> lines, List<MarkableOp> ops) {
PageContent page = new PageContent(0, lines, ops, ops.size(), false, false, false, A4);
return new LayoutAnalyzer().analyse(List.of(page));
}
private static long countOf(DocumentStructure structure, StructType type) {
long[] total = {0};
structure.visit(
block -> {
if (block.getType() == type) {
total[0]++;
}
});
return total[0];
}
@Test
@DisplayName("a cluster of substantial strokes becomes a figure, not silent decoration")
void chartBecomesAFigure() {
List<MarkableOp> bars = new ArrayList<>();
for (int i = 0; i < 6; i++) {
bars.add(vector(i, new BBox(100 + i * 20, 400, 115 + i * 20, 400 + 30 + i * 10)));
}
DocumentStructure structure = analyse(List.of(), bars);
assertTrue(
countOf(structure, StructType.FIGURE) > 0,
"a bar chart drawn with path operators must not vanish as decoration");
assertTrue(
structure.figuresWithoutAlt().size() > 0,
"the report must say the chart needs a description");
}
@Test
@DisplayName("thin rules and table borders stay artifacts")
void tableRulesStayDecoration() {
List<MarkableOp> rules = new ArrayList<>();
for (int i = 0; i < 8; i++) {
rules.add(vector(i, new BBox(60, 700 - i * 20, 540, 701 - i * 20)));
}
DocumentStructure structure = analyse(List.of(), rules);
assertEquals(
0,
countOf(structure, StructType.FIGURE),
"horizontal rules are page furniture and must not demand alt text");
assertEquals(0, structure.figuresWithoutAlt().size());
}
@Test
@DisplayName("a lone box is ornament, not a chart")
void singleBoxIsNotAFigure() {
DocumentStructure structure =
analyse(List.of(), List.of(vector(0, new BBox(60, 400, 500, 700))));
assertEquals(0, countOf(structure, StructType.FIGURE));
}
@Test
@DisplayName("shaded table rows behind text are not mistaken for a chart")
void shadedTableRowsAreNotFigures() {
List<MarkableOp> shading = new ArrayList<>();
List<TextLineInfo> rows = new ArrayList<>();
for (int i = 0; i < 6; i++) {
float y = 600 - i * 20;
// A filled row background, tall enough to pass the thinness test.
shading.add(vector(i, new BBox(60, y, 540, y + 16)));
rows.add(line("Expense line item " + i + " amount", 10, 64, y + 3));
}
DocumentStructure structure = analyse(rows, shading);
assertEquals(
0,
countOf(structure, StructType.FIGURE),
"row shading sits behind the text it decorates and is not a drawing");
}
@Test
@DisplayName("small print dominating an invoice does not promote addresses to headings")
void smallPrintDoesNotCreateHeadings() {
List<TextLineInfo> lines = new ArrayList<>();
// Address block at ordinary 11pt.
lines.add(line("Acme Industries Limited", 11, 60, 780));
lines.add(line("14 Example Street", 11, 60, 765));
lines.add(line("Manchester M1 2AB", 11, 60, 750));
// 40 lines of 9pt line-item small print, which dominates the character count.
for (int i = 0; i < 40; i++) {
lines.add(line("Item " + i + " widget assembly part number " + i, 9, 60, 700 - i * 12));
}
Map<Float, Integer> tiers =
LayoutAnalyzer.headingTiers(
List.of(new PageContent(0, lines, List.of(), 0, false, false, false, A4)),
9f);
assertNull(
tiers.get(11f),
"11pt address lines are body text on an invoice, not headings: " + tiers);
}
@Test
@DisplayName("a genuinely rare large size is still a heading")
void realHeadingsSurvive() {
List<TextLineInfo> lines = new ArrayList<>();
lines.add(line("Annual Report", 24, 60, 780));
for (int i = 0; i < 40; i++) {
lines.add(
line("Body prose line number " + i + " continues here", 11, 60, 700 - i * 12));
}
Map<Float, Integer> tiers =
LayoutAnalyzer.headingTiers(
List.of(new PageContent(0, lines, List.of(), 0, false, false, false, A4)),
11f);
assertEquals(1, tiers.get(24f), "a rare large size is exactly what a heading looks like");
}
}
@@ -39,6 +39,18 @@ class AdminPolicyManagementAuthorityTest {
assertFalse(authority().canEditPolicies());
}
@Test
void adminMayTriggerPolicies() {
when(userService.isCurrentUserAdmin()).thenReturn(true);
assertTrue(authority().canTriggerPolicies());
}
@Test
void nonAdminMayNotTriggerPolicies() {
when(userService.isCurrentUserAdmin()).thenReturn(false);
assertFalse(authority().canTriggerPolicies());
}
@Test
void currentUserTeamIdResolvesFromTheCurrentUsersTeam() {
Team team = new Team();
@@ -2,6 +2,7 @@ package stirling.software.proprietary.policy.controller;
import static org.assertj.core.api.Assertions.assertThat;
import static org.assertj.core.api.Assertions.assertThatThrownBy;
import static org.junit.jupiter.api.Assertions.assertDoesNotThrow;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertTrue;
@@ -82,6 +83,8 @@ class PolicyControllerTest {
@Mock private stirling.software.proprietary.policy.asset.PolicyAssetCleaner assetCleaner;
@Mock private stirling.software.proprietary.policy.asset.PolicyAssetResolver assetResolver;
@Mock private ProcessedLedger processedLedger;
@Mock private TempFileManager tempFileManager;
@@ -114,6 +117,7 @@ class PolicyControllerTest {
policyTriggerManager,
policyOverviewService,
assetCleaner,
assetResolver,
processedLedger,
policyTriggers,
applicationProperties,
@@ -231,7 +235,7 @@ class PolicyControllerTest {
.thenReturn(handle("run-1"));
ResponseEntity<JobResponse<Void>> response =
controller.run(definitionWithStep(), new PolicyRunFiles());
controller.run(definitionWithStep(), null, new PolicyRunFiles());
assertThat(response.getStatusCode()).isEqualTo(HttpStatus.ACCEPTED);
assertThat(response.getBody().getJobId()).isEqualTo("run-1");
@@ -244,7 +248,7 @@ class PolicyControllerTest {
.thenReturn(handle("run-1"));
when(sourceAccessGuard.currentTeamId()).thenReturn(3L);
controller.run(definitionWithStep(), new PolicyRunFiles());
controller.run(definitionWithStep(), null, new PolicyRunFiles());
verify(docCounter).record(EditorSource.counterKey(3L), 0L);
}
@@ -254,7 +258,7 @@ class PolicyControllerTest {
void rejectsEmptyPipeline() {
PipelineDefinition empty = new PipelineDefinition("pipe", List.of(), List.of());
assertThatThrownBy(() -> controller.run(empty, new PolicyRunFiles()))
assertThatThrownBy(() -> controller.run(empty, null, new PolicyRunFiles()))
.isInstanceOf(ResponseStatusException.class)
.satisfies(
e ->
@@ -276,7 +280,7 @@ class PolicyControllerTest {
.when(policyValidator)
.validateOutput(any());
assertThatThrownBy(() -> controller.run(definition, new PolicyRunFiles()))
assertThatThrownBy(() -> controller.run(definition, null, new PolicyRunFiles()))
.isInstanceOf(ResponseStatusException.class)
.satisfies(
e ->
@@ -284,6 +288,38 @@ class PolicyControllerTest {
.isEqualTo(HttpStatus.BAD_REQUEST));
verify(policyRunner, never()).runAdHoc(any(), any(), any());
}
@Test
@DisplayName("resolves stored assets from the supplied policy when the caller may edit it")
void resolvesStoredAssetsForEditor() throws Exception {
applicationProperties.getSecurity().setEnableLogin(false); // editing allowed
Policy p = policy("pol-1", 1L);
when(policyStore.get("pol-1")).thenReturn(Optional.of(p));
when(policyAccessGuard.canAccess(p)).thenReturn(true);
when(assetResolver.resolve(eq(p), any())).thenAnswer(inv -> inv.getArgument(1));
when(policyRunner.runAdHoc(any(), any(), eq(PolicyProgressListener.NOOP)))
.thenReturn(handle("run-1"));
controller.run(definitionWithStep(), "pol-1", new PolicyRunFiles());
verify(assetResolver).resolve(eq(p), any());
}
@Test
@DisplayName("does not resolve a policy's stored assets for a caller who cannot edit it")
void skipsStoredAssetsForNonEditor() throws Exception {
// Gating asset resolution to editors keeps a member from rebinding a policy's stored
// asset into an ad-hoc step to read it back.
applicationProperties.getSecurity().setEnableLogin(true);
when(policyManagementAuthority.canEditPolicies()).thenReturn(false);
when(policyRunner.runAdHoc(any(), any(), eq(PolicyProgressListener.NOOP)))
.thenReturn(handle("run-1"));
controller.run(definitionWithStep(), "pol-1", new PolicyRunFiles());
verify(assetResolver, never()).resolve(any(), any());
verify(policyStore, never()).get(any());
}
}
@Nested
@@ -295,7 +331,8 @@ class PolicyControllerTest {
void returnsEmitter() throws Exception {
when(policyRunner.runAdHoc(any(), any(), any())).thenReturn(handle("run-2"));
SseEmitter emitter = controller.runStream(definitionWithStep(), new PolicyRunFiles());
SseEmitter emitter =
controller.runStream(definitionWithStep(), null, new PolicyRunFiles());
assertThat(emitter).isNotNull();
}
@@ -305,7 +342,7 @@ class PolicyControllerTest {
void rejectsEmpty() {
PipelineDefinition empty = new PipelineDefinition("pipe", List.of(), List.of());
assertThatThrownBy(() -> controller.runStream(empty, new PolicyRunFiles()))
assertThatThrownBy(() -> controller.runStream(empty, null, new PolicyRunFiles()))
.isInstanceOf(ResponseStatusException.class);
}
}
@@ -738,5 +775,78 @@ class PolicyControllerTest {
assertThat(((ResponseStatusException) e).getStatusCode())
.isEqualTo(HttpStatus.NOT_FOUND));
}
@Test
@DisplayName("trigger is forbidden for a team member who cannot manage policies")
void triggerForbiddenForMember() {
// Sweeping a policy's configured sources is a policy-management capability, so being
// in the policy's team is not on its own enough to perform it.
applicationProperties.getSecurity().setEnableLogin(true);
when(policyManagementAuthority.canTriggerPolicies()).thenReturn(false);
assertThatThrownBy(() -> controller.trigger("a"))
.isInstanceOf(ResponseStatusException.class)
.satisfies(
e ->
assertThat(((ResponseStatusException) e).getStatusCode())
.isEqualTo(HttpStatus.FORBIDDEN));
// Rejected before the policy is looked up, so no run starts.
verify(policyRunner, never()).run(any());
verify(policyStore, never()).get(any());
}
@Test
@DisplayName("trigger runs for a caller who may manage policies")
void triggerAllowedForLeader() {
applicationProperties.getSecurity().setEnableLogin(true);
when(policyManagementAuthority.canTriggerPolicies()).thenReturn(true);
Policy p = policy("a", 1L);
when(policyStore.get("a")).thenReturn(Optional.of(p));
when(policyAccessGuard.canAccess(p)).thenReturn(true);
SweepOutcome outcome = new SweepOutcome(List.of("run-a"), 1, 0, 0, 0);
when(policyRunner.run(p)).thenReturn(outcome);
ResponseEntity<SweepOutcome> response = controller.trigger("a");
assertThat(response.getStatusCode()).isEqualTo(HttpStatus.ACCEPTED);
assertThat(response.getBody()).isEqualTo(outcome);
}
@Test
@DisplayName("trigger skips the role check when login is disabled")
void triggerTrustsTheLocalOperator() {
// Single-user deployments have no roles at all; the gate must not lock them out of
// their
// own sweeps.
applicationProperties.getSecurity().setEnableLogin(false);
Policy p = policy("a", null);
when(policyStore.get("a")).thenReturn(Optional.of(p));
when(policyAccessGuard.canAccess(p)).thenReturn(true);
SweepOutcome outcome = new SweepOutcome(List.of("run-a"), 1, 0, 0, 0);
when(policyRunner.run(p)).thenReturn(outcome);
assertThat(controller.trigger("a").getStatusCode()).isEqualTo(HttpStatus.ACCEPTED);
verify(policyManagementAuthority, never()).canTriggerPolicies();
}
@Test
@DisplayName("running a policy over the caller's own files stays open to any member")
void storedRunIsNotGatedByRole() {
// Editor enforcement: every member's upload/export runs the team's stored policies on
// their own documents. Gating this the way the sweep is gated would break the editor.
applicationProperties.getSecurity().setEnableLogin(true);
Policy p = policy("a", 1L);
when(policyStore.get("a")).thenReturn(Optional.of(p));
when(policyAccessGuard.canAccess(p)).thenReturn(true);
when(policyRunner.runWith(eq(p), any(), eq(PolicyProgressListener.NOOP)))
.thenReturn(handle("run-9"));
ResponseEntity<JobResponse<Void>> response =
assertDoesNotThrow(() -> controller.runStoredPolicy("a", new PolicyRunFiles()));
assertThat(response.getStatusCode()).isEqualTo(HttpStatus.ACCEPTED);
verify(policyManagementAuthority, never()).canTriggerPolicies();
verify(policyManagementAuthority, never()).canEditPolicies();
}
}
}
@@ -0,0 +1,115 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.util.Map;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.service.PdfMetadataService;
import stirling.software.proprietary.controller.api.converters.ConvertPdfToPdfUa;
import stirling.software.proprietary.model.api.ua.AccessibilityReport;
import stirling.software.proprietary.model.api.ua.FigureDescriptor;
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
/**
* The alt-text loop end to end: the report hands out keys the conversion accepts. The converter
* never invents descriptions, so a caller must be able to supply them.
*/
class AltTextRoundTripTest {
private static PdfUaConversionService conversion;
private static AccessibilityAuditService audit;
@BeforeAll
static void setUp() {
PdfUaValidationService validation = new PdfUaValidationService();
validation.initialise();
conversion =
new PdfUaConversionService(
validation,
new FontEmbeddingService(),
new CustomPDFDocumentFactory(
org.mockito.Mockito.mock(PdfMetadataService.class)));
audit = new AccessibilityAuditService(validation);
}
private static TaggingOptions.TaggingOptionsBuilder options() {
return TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.title("Illustrated")
.embedFonts(false);
}
@Test
@DisplayName("the report names the figures that need describing, with usable keys")
void reportEnumeratesFigures() throws Exception {
byte[] input = PdfUaTestDocuments.imageDocument();
AccessibilityReport report = audit.audit(input, PdfUaProfile.UA1);
assertFalse(
report.getFiguresNeedingDescription().isEmpty(),
"a document with an undescribed image must say which figure needs text");
FigureDescriptor figure = report.getFiguresNeedingDescription().get(0);
assertTrue(figure.key().matches("\\d+:\\d+"), "key should be pageIndex:ordinal: " + figure);
assertEquals(1, figure.page(), "pages are reported 1-based for humans");
assertTrue(figure.width() > 0 && figure.height() > 0, "figure should carry its box");
}
@Test
@DisplayName("feeding the report's key back makes the document conform")
void suppliedDescriptionClosesTheLoop() throws Exception {
byte[] input = PdfUaTestDocuments.imageDocument();
PdfUaConversionOutcome before = conversion.convert(input, options().build());
assertFalse(before.declared(), "an undescribed image must block the claim");
String key =
audit.audit(input, PdfUaProfile.UA1).getFiguresNeedingDescription().get(0).key();
PdfUaConversionOutcome after =
conversion.convert(
input, options().altTextByFigure(Map.of(key, "A blue rectangle")).build());
assertEquals(
0,
after.tagging().figuresNeedingAltText(),
"the description supplied against the report's own key was not applied");
assertTrue(after.declared(), "with every figure described the document should conform");
}
@Test
@DisplayName("the request's key=text form parses the way the report emits keys")
void parsesTheWireFormat() {
Map<String, String> parsed =
ConvertPdfToPdfUa.parseAltText(
"0:12=Bar chart of quarterly revenue\r\n"
+ "1:3=Company logo\n"
+ " \n"
+ "malformed-line\n"
+ "2:7=Diagram showing the approval flow = end to end");
assertEquals(3, parsed.size(), "blank and malformed lines are skipped: " + parsed);
assertEquals("Bar chart of quarterly revenue", parsed.get("0:12"));
assertEquals("Company logo", parsed.get("1:3"));
assertEquals(
"Diagram showing the approval flow = end to end",
parsed.get("2:7"),
"only the first equals splits, so descriptions may contain one");
}
@Test
@DisplayName("no descriptions supplied means none invented")
void emptyInputInventsNothing() {
assertTrue(ConvertPdfToPdfUa.parseAltText(null).isEmpty());
assertTrue(ConvertPdfToPdfUa.parseAltText(" ").isEmpty());
}
}
@@ -0,0 +1,92 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.cos.COSName;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.service.PdfMetadataService;
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
/** PDF/UA-2 is not just a metadata number: it needs PDF 2.0 and namespaced structure types. */
class PdfUa2ProfileTest {
private static PdfUaConversionService conversion;
@BeforeAll
static void setUp() {
PdfUaValidationService validation = new PdfUaValidationService();
validation.initialise();
conversion =
new PdfUaConversionService(
validation,
new FontEmbeddingService(),
new CustomPDFDocumentFactory(
org.mockito.Mockito.mock(PdfMetadataService.class)));
}
private static PdfUaConversionOutcome convertUa2(byte[] input) throws Exception {
return conversion.convert(
input,
TaggingOptions.builder()
.profile(PdfUaProfile.UA2)
.language("en-GB")
.title("UA-2 Document")
.embedFonts(false)
.build());
}
@Test
@DisplayName("raises the file to PDF 2.0 and namespaces the structure tree")
void producesPdf2WithNamespaces() throws Exception {
PdfUaConversionOutcome outcome = convertUa2(PdfUaTestDocuments.headingHierarchy());
try (PDDocument document = Loader.loadPDF(outcome.pdfBytes())) {
assertEquals(2.0f, document.getVersion(), "UA-2 is defined on PDF 2.0");
var root = document.getDocumentCatalog().getStructureTreeRoot();
assertNotNull(root, "no structure tree was written");
assertNotNull(
root.getCOSObject().getDictionaryObject(COSName.getPDFName("Namespaces")),
"UA-2 requires the standard structure namespace to be declared");
}
}
@Test
@DisplayName("validates against the PDF/UA-2 profile, not the UA-1 one")
void validatesAgainstUa2() throws Exception {
PdfUaConversionOutcome outcome = convertUa2(PdfUaTestDocuments.simpleDocument());
assertEquals("PDF/UA-2", outcome.validation().profile());
}
@Test
@DisplayName("reaches UA-2 conformance and declares it")
void reachesUa2Conformance() throws Exception {
PdfUaConversionOutcome outcome = convertUa2(PdfUaTestDocuments.simpleDocument());
String failures =
outcome.validation().issues().stream()
.map(issue -> issue.getClause() + ": " + issue.getTechnicalMessage())
.collect(java.util.stream.Collectors.joining("; "));
assertEquals(0, outcome.validation().totalFailures(), "UA-2 checks failed: " + failures);
assertTrue(outcome.declared(), "a conforming UA-2 file must carry the declaration");
assertTrue(outcome.pdfBytes().length > 0);
}
@Test
@DisplayName("an illustrated document still cannot claim UA-2 without descriptions")
void undescribedImageBlocksTheUa2Claim() throws Exception {
PdfUaConversionOutcome outcome = convertUa2(PdfUaTestDocuments.imageDocument());
assertFalse(outcome.declared(), "an undescribed image must block the claim");
}
}
@@ -0,0 +1,341 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.util.ArrayList;
import java.util.List;
import java.util.Locale;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDFont;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.service.PdfMetadataService;
import stirling.software.proprietary.pdf.ua.DocumentStructure;
import stirling.software.proprietary.pdf.ua.LayoutAnalyzer;
import stirling.software.proprietary.pdf.ua.PageContent;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.PdfUaTagger;
import stirling.software.proprietary.pdf.ua.TaggedContentExtractor;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
/**
* Measures where conversion time and memory go; meant to be read, not to gate CI. Assertions catch
* only order-of-magnitude regressions - wall-clock numbers are no contract.
*/
class PdfUaBenchmarkTest {
private static PdfUaConversionService service;
private static PdfUaValidationService validation;
@BeforeAll
static void setUp() {
validation = new PdfUaValidationService();
validation.initialise();
service =
new PdfUaConversionService(
validation,
new FontEmbeddingService(),
new CustomPDFDocumentFactory(
org.mockito.Mockito.mock(PdfMetadataService.class)));
}
/** A realistic page: heading, prose, a small table, a bullet list. */
private static byte[] document(int pages) throws IOException {
try (PDDocument document = new PDDocument()) {
PDFont font = null;
for (int p = 0; p < pages; p++) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
if (font == null) {
font = PdfUaTestDocuments.font(document);
}
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
float y = 790;
write(cs, font, 9, 60, 810, "Benchmark Corpus Running Head");
write(cs, font, 18, 60, y, "Section " + (p + 1));
y -= 30;
for (int line = 0; line < 22; line++) {
write(
cs,
font,
11,
60,
y,
"Body line " + line + " of section " + (p + 1) + " with prose.");
y -= 15;
}
for (int row = 0; row < 4; row++) {
cs.beginText();
cs.setFont(font, 11);
cs.newLineAtOffset(60, y);
cs.showText("Row " + row);
cs.newLineAtOffset(160, 0);
cs.showText(String.valueOf(row * 120));
cs.newLineAtOffset(140, 0);
cs.showText(String.valueOf(row * 480));
cs.endText();
y -= 16;
}
write(cs, font, 11, 60, y - 10, "• First bullet point");
write(cs, font, 11, 60, y - 25, "• Second bullet point");
write(cs, font, 9, 300, 30, "Page " + (p + 1));
}
}
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
return out.toByteArray();
}
}
private static void write(
PDPageContentStream cs, PDFont font, float size, float x, float y, String text)
throws IOException {
cs.beginText();
cs.setFont(font, size);
cs.newLineAtOffset(x, y);
cs.showText(text);
cs.endText();
}
private static TaggingOptions options() {
return TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.title("Benchmark")
.embedFonts(false)
.existingTags(TaggingOptions.ExistingTags.REBUILD)
.build();
}
private static long usedHeap() {
Runtime runtime = Runtime.getRuntime();
System.gc();
return runtime.totalMemory() - runtime.freeMemory();
}
@Test
@DisplayName("reports throughput and memory across document sizes")
void throughputAcrossSizes() throws Exception {
int[] sizes = {1, 10, 50, 150};
StringBuilder report =
new StringBuilder("\nPDF/UA conversion throughput\n")
.append(
String.format(
" %-7s %-10s %-12s %-12s %-10s %s%n",
"pages",
"input",
"convert ms",
"ms/page",
"pages/s",
"heap MB"));
// Warm up so the first timed run is not measuring class loading and JIT.
service.convert(document(5), options());
for (int pages : sizes) {
byte[] input = document(pages);
long heapBefore = usedHeap();
long start = System.nanoTime();
var outcome = service.convert(input, options());
long elapsedMs = (System.nanoTime() - start) / 1_000_000;
long heapDelta = (usedHeap() - heapBefore) / (1024 * 1024);
assertTrue(outcome.pdfBytes().length > 0);
report.append(
String.format(
Locale.ROOT,
" %-7d %-10s %-12d %-12.2f %-10.1f %d%n",
pages,
humanBytes(input.length),
elapsedMs,
elapsedMs / (double) pages,
pages * 1000.0 / Math.max(elapsedMs, 1),
Math.max(heapDelta, 0)));
}
System.out.println(report);
}
@Test
@DisplayName("breaks conversion down by phase so optimisation has a target")
void phaseBreakdown() throws Exception {
byte[] input = document(60);
// Warm up.
try (PDDocument warm = Loader.loadPDF(input)) {
new TaggedContentExtractor().extract(warm);
}
long parseMs;
long extractMs;
long analyseMs;
long tagMs;
List<PageContent> pages;
DocumentStructure structure;
long t0 = System.nanoTime();
try (PDDocument document = Loader.loadPDF(input)) {
parseMs = ms(t0);
long t1 = System.nanoTime();
pages = new TaggedContentExtractor().extract(document);
extractMs = ms(t1);
long t2 = System.nanoTime();
structure = new LayoutAnalyzer().analyse(pages);
analyseMs = ms(t2);
}
long t3 = System.nanoTime();
try (PDDocument document = Loader.loadPDF(input)) {
new PdfUaTagger().tag(document, options());
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
}
tagMs = ms(t3);
long t4 = System.nanoTime();
var outcome = service.convert(input, options());
long totalMs = ms(t4);
long t5 = System.nanoTime();
validation.validate(outcome.pdfBytes(), PdfUaProfile.UA1);
long validateMs = ms(t5);
System.out.printf(
Locale.ROOT,
"%nPhase breakdown over %d pages (%d blocks)%n"
+ " parse %5d ms%n"
+ " extract %5d ms (text pass + token scan)%n"
+ " analyse %5d ms%n"
+ " tag end-to-end %5d ms (includes parse, extract, analyse, inject, write)%n"
+ " validate %5d ms (veraPDF)%n"
+ " full convert %5d ms (tag + declare + validate)%n",
60,
structure.getBlocks().size(),
parseMs,
extractMs,
analyseMs,
tagMs,
validateMs,
totalMs);
assertTrue(pages.size() == 60, "extractor lost pages");
}
@Test
@DisplayName("splits the tagging pass into its own sub-phases")
void taggingSubPhases() throws Exception {
byte[] input = document(60);
try (PDDocument warm = Loader.loadPDF(input)) {
new TaggedContentExtractor().extract(warm);
}
long extractMs;
long analyseMs;
long injectMs;
long treeMs;
long saveMs;
try (PDDocument document = Loader.loadPDF(input)) {
long t = System.nanoTime();
List<PageContent> pages = new TaggedContentExtractor().extract(document);
extractMs = ms(t);
t = System.nanoTime();
DocumentStructure structure = new LayoutAnalyzer().analyse(pages);
analyseMs = ms(t);
t = System.nanoTime();
var injector = new stirling.software.proprietary.pdf.ua.MarkedContentInjector();
var byPage =
new java.util.LinkedHashMap<
Integer, List<stirling.software.proprietary.pdf.ua.StructBlock>>();
structure
.getBlocks()
.forEach(
b ->
byPage.computeIfAbsent(b.getPageIndex(), k -> new ArrayList<>())
.add(b));
for (int p = 0; p < document.getNumberOfPages(); p++) {
injector.inject(
document, document.getPage(p), byPage.getOrDefault(p, List.of()), 0, true);
}
injectMs = ms(t);
t = System.nanoTime();
new stirling.software.proprietary.pdf.ua.StructTreeWriter()
.write(document, structure, PdfUaProfile.UA1);
treeMs = ms(t);
t = System.nanoTime();
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
saveMs = ms(t);
}
System.out.printf(
Locale.ROOT,
"%nTagging sub-phases over 60 pages%n"
+ " extract %5d ms%n"
+ " analyse %5d ms%n"
+ " inject %5d ms%n"
+ " struct tree %5d ms%n"
+ " save %5d ms%n",
extractMs,
analyseMs,
injectMs,
treeMs,
saveMs);
}
@Test
@DisplayName("memory stays proportional to document size, not quadratic")
void memoryScales() throws Exception {
List<String> rows = new ArrayList<>();
long previousPerPage = 0;
boolean blewUp = false;
for (int pages : new int[] {20, 80, 200}) {
byte[] input = document(pages);
long before = usedHeap();
var outcome = service.convert(input, options());
long after = usedHeap();
long perPageKb = Math.max(after - before, 0) / 1024 / pages;
rows.add(
String.format(
Locale.ROOT,
" %-6d pages in %-9s out %-9s ~%d KB/page retained",
pages,
humanBytes(input.length),
humanBytes(outcome.pdfBytes().length),
perPageKb));
// Per-page cost should stay roughly flat; a big jump means something accumulates.
if (previousPerPage > 0 && perPageKb > previousPerPage * 4 && perPageKb > 200) {
blewUp = true;
}
previousPerPage = Math.max(perPageKb, 1);
}
System.out.println("\nMemory scaling\n" + String.join("\n", rows));
assertTrue(!blewUp, "per-page memory grew superlinearly: " + rows);
}
private static long ms(long startNanos) {
return (System.nanoTime() - startNanos) / 1_000_000;
}
private static String humanBytes(int bytes) {
return bytes < 1024 * 1024
? (bytes / 1024) + " KB"
: String.format(Locale.ROOT, "%.1f MB", bytes / 1024.0 / 1024.0);
}
}
@@ -0,0 +1,231 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.IOException;
import java.util.LinkedHashMap;
import java.util.Map;
import java.util.concurrent.Callable;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.text.PDFTextStripper;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.service.PdfMetadataService;
import stirling.software.proprietary.model.api.ua.AccessibilityIssue;
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
import stirling.software.proprietary.model.api.ua.UaValidationResult;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
/**
* End-to-end conversion over the fixture corpus, validated with veraPDF. The corpus is deliberately
* varied: what breaks a tagger is rarely the simple case.
*/
class PdfUaConversionIntegrationTest {
private static PdfUaConversionService service;
@BeforeAll
static void setUp() {
PdfUaValidationService validation = new PdfUaValidationService();
validation.initialise();
service =
new PdfUaConversionService(
validation,
new FontEmbeddingService(),
new CustomPDFDocumentFactory(
org.mockito.Mockito.mock(PdfMetadataService.class)));
}
/** Fixtures already embed fonts, so the Ghostscript pass is off to keep tests hermetic. */
private static TaggingOptions options() {
return TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.title("Test Document")
.embedFonts(false)
.build();
}
private static PdfUaConversionOutcome convert(byte[] input) throws IOException {
return service.convert(input, options());
}
private static String extractText(byte[] pdf) throws IOException {
try (PDDocument document = Loader.loadPDF(pdf)) {
PDFTextStripper stripper = new PDFTextStripper();
stripper.setSortByPosition(true);
return PdfUaRealCorpusTest.normalise(stripper.getText(document));
}
}
@Test
@DisplayName("every fixture converts without error and gains a structure tree")
void corpusConverts() throws Exception {
Map<String, Callable<byte[]>> corpus = corpus();
StringBuilder report = new StringBuilder("\nPDF/UA conversion over the fixture corpus\n");
for (Map.Entry<String, Callable<byte[]>> entry : corpus.entrySet()) {
byte[] input =
entry.getKey().equals("empty")
? entry.getValue().call()
: entry.getValue().call();
PdfUaConversionOutcome outcome = convert(input);
assertNotNull(outcome.pdfBytes(), entry.getKey() + " produced no output");
report.append(
String.format(
" %-18s declared=%-5s failures=%-3d elements=%-3d artifacts=%-3d altNeeded=%d%n",
entry.getKey(),
outcome.declared(),
outcome.validation().totalFailures(),
outcome.tagging().taggedElements(),
outcome.tagging().artifacts(),
outcome.tagging().figuresNeedingAltText()));
for (AccessibilityIssue issue : outcome.validation().issues()) {
report.append(
String.format(
" clause %-6s x%-4d %s%n",
issue.getClause(),
issue.getOccurrences(),
issue.getTechnicalMessage()));
}
}
System.out.println(report);
}
@Test
@DisplayName("tagging never changes the text content of a page")
void textIsPreserved() throws Exception {
for (Map.Entry<String, Callable<byte[]>> entry : corpus().entrySet()) {
byte[] input = entry.getValue().call();
String before = extractText(input);
String after = extractText(convert(input).pdfBytes());
assertEquals(before, after, "text changed for fixture " + entry.getKey());
}
}
@Test
@DisplayName("a simple document gains a structure tree with headings and paragraphs")
void simpleDocumentIsTagged() throws Exception {
PdfUaConversionOutcome outcome = convert(PdfUaTestDocuments.simpleDocument());
try (PDDocument document = Loader.loadPDF(outcome.pdfBytes())) {
assertNotNull(
document.getDocumentCatalog().getStructureTreeRoot(),
"no structure tree was written");
assertTrue(
document.getDocumentCatalog().getMarkInfo() != null
&& document.getDocumentCatalog().getMarkInfo().isMarked(),
"MarkInfo/Marked was not set");
assertEquals("en-GB", document.getDocumentCatalog().getLanguage());
assertEquals("Test Document", document.getDocumentInformation().getTitle());
}
assertTrue(outcome.tagging().taggedElements() > 0, "nothing was tagged");
}
@Test
@DisplayName("running heads and page numbers become artifacts, not content")
void runningHeadersBecomeArtifacts() throws Exception {
PdfUaConversionOutcome outcome = convert(PdfUaTestDocuments.runningHeadersDocument());
assertTrue(
outcome.tagging().artifacts() >= 4,
"expected the repeated header on each page to become an artifact, got "
+ outcome.tagging().artifacts());
}
@Test
@DisplayName("an image is tagged as a figure and reported as needing alt text")
void imagesNeedAltText() throws Exception {
PdfUaConversionOutcome outcome = convert(PdfUaTestDocuments.imageDocument());
assertEquals(1, outcome.tagging().figuresNeedingAltText());
assertFalse(
outcome.declared(),
"a document with an undescribed image must not claim conformance");
}
@Test
@DisplayName("supplying alt text lets an illustrated document conform")
void suppliedAltTextIsApplied() throws Exception {
byte[] input = PdfUaTestDocuments.imageDocument();
PdfUaConversionOutcome probe = convert(input);
assertEquals(1, probe.tagging().figuresNeedingAltText());
TaggingOptions withAlt =
options().toBuilder()
.altTextByFigure(Map.of(figureKey(input), "A blue rectangle"))
.build();
PdfUaConversionOutcome outcome = service.convert(input, withAlt);
assertEquals(
0,
outcome.tagging().figuresNeedingAltText(),
"alt text supplied by the caller was not applied");
}
@Test
@DisplayName("an empty document does not crash the converter")
void emptyDocumentIsHandled() throws Exception {
PdfUaConversionOutcome outcome = convert(PdfUaTestDocuments.emptyDocument());
assertNotNull(outcome.pdfBytes());
assertFalse(outcome.warnings().isEmpty(), "an empty document should warn");
}
@Test
@DisplayName("an un-OCRed scan is reported rather than silently declared conformant")
void scannedDocumentIsNotDeclared() throws Exception {
PdfUaConversionOutcome outcome = convert(PdfUaTestDocuments.scannedDocument());
assertFalse(outcome.declared(), "a scan with no text layer must not claim conformance");
}
@Test
@DisplayName("validation of an untagged document reports the missing structure")
void untaggedDocumentFailsValidation() throws Exception {
PdfUaValidationService validation = new PdfUaValidationService();
validation.initialise();
UaValidationResult result =
validation.validate(PdfUaTestDocuments.simpleDocument(), PdfUaProfile.UA1);
assertFalse(result.compliant(), "an untagged document cannot be PDF/UA compliant");
assertTrue(result.hasIssues());
}
/** The key the tagger uses for figure alt text is "pageIndex:firstOrdinal". */
private static String figureKey(byte[] input) throws IOException {
try (PDDocument document = Loader.loadPDF(input)) {
var pages =
new stirling.software.proprietary.pdf.ua.TaggedContentExtractor()
.extract(document);
for (var page : pages) {
for (var op : page.graphics()) {
return page.pageIndex() + ":" + op.ordinal();
}
}
}
return "0:0";
}
private static Map<String, Callable<byte[]>> corpus() {
Map<String, Callable<byte[]>> corpus = new LinkedHashMap<>();
corpus.put("simple", PdfUaTestDocuments::simpleDocument);
corpus.put("headings", PdfUaTestDocuments::headingHierarchy);
corpus.put("lists", PdfUaTestDocuments::listDocument);
corpus.put("table", PdfUaTestDocuments::tableDocument);
corpus.put("image", PdfUaTestDocuments::imageDocument);
corpus.put("runningHeads", PdfUaTestDocuments::runningHeadersDocument);
corpus.put("twoColumn", PdfUaTestDocuments::twoColumnDocument);
corpus.put("link", PdfUaTestDocuments::linkDocument);
corpus.put("empty", PdfUaTestDocuments::emptyDocument);
corpus.put("scanned", PdfUaTestDocuments::scannedDocument);
corpus.put("formXObject", PdfUaTestDocuments::formXObjectDocument);
corpus.put("multiStream", PdfUaTestDocuments::multiStreamDocument);
corpus.put("rotated", PdfUaTestDocuments::rotatedDocument);
corpus.put("offsetMediaBox", PdfUaTestDocuments::offsetMediaBoxDocument);
return corpus;
}
}
@@ -0,0 +1,322 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.nio.charset.StandardCharsets;
import java.util.List;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.contentstream.operator.Operator;
import org.apache.pdfbox.pdfparser.PDFStreamParser;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.text.PDFTextStripper;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Nested;
import org.junit.jupiter.api.Test;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.service.PdfMetadataService;
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
import stirling.software.proprietary.pdf.ua.PageContent;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.StructBlock;
import stirling.software.proprietary.pdf.ua.StructType;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
/** Covers the document shapes and failure modes the first round of tests missed. */
class PdfUaHardeningTest {
private static PdfUaConversionService service;
@BeforeAll
static void setUp() {
PdfUaValidationService validation = new PdfUaValidationService();
validation.initialise();
service =
new PdfUaConversionService(
validation,
new FontEmbeddingService(),
new CustomPDFDocumentFactory(
org.mockito.Mockito.mock(PdfMetadataService.class)));
}
private static PdfUaConversionOutcome convert(byte[] input) throws IOException {
return service.convert(
input,
TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.title("Hardening")
.embedFonts(false)
.build());
}
private static String extract(byte[] pdf) throws IOException {
try (PDDocument document = Loader.loadPDF(pdf)) {
PDFTextStripper stripper = new PDFTextStripper();
stripper.setSortByPosition(true);
return PdfUaRealCorpusTest.normalise(stripper.getText(document));
}
}
@Nested
@DisplayName("document shapes")
class Shapes {
@Test
@DisplayName("text inside a form XObject is attributed to the Do and tagged as prose")
void formXObjectTextIsTagged() throws Exception {
byte[] input = PdfUaTestDocuments.formXObjectDocument();
assertTrue(
extract(input).contains("inside the form XObject"),
"fixture must actually draw text inside a form");
PdfUaConversionOutcome outcome = convert(input);
assertTrue(outcome.declared(), warnings(outcome));
assertEquals(extract(input), extract(outcome.pdfBytes()));
assertEquals(
0,
outcome.tagging().figuresNeedingAltText(),
"a form whose text is reachable must not degrade to an undescribed figure");
}
@Test
@DisplayName("a page built from multiple content streams converts as one sequence")
void multiStreamPageConverts() throws Exception {
byte[] input = PdfUaTestDocuments.multiStreamDocument();
PdfUaConversionOutcome outcome = convert(input);
assertTrue(outcome.declared(), warnings(outcome));
assertEquals(extract(input), extract(outcome.pdfBytes()));
}
@Test
@DisplayName("a rotated page keeps its text and converts")
void rotatedPageConverts() throws Exception {
byte[] input = PdfUaTestDocuments.rotatedDocument();
PdfUaConversionOutcome outcome = convert(input);
assertEquals(extract(input), extract(outcome.pdfBytes()));
assertTrue(outcome.tagging().taggedElements() > 0, "rotated text was not tagged");
}
@Test
@DisplayName("a MediaBox that does not start at the origin does not break analysis")
void offsetMediaBoxConverts() throws Exception {
byte[] input = PdfUaTestDocuments.offsetMediaBoxDocument();
PdfUaConversionOutcome outcome = convert(input);
assertTrue(outcome.declared(), warnings(outcome));
assertEquals(extract(input), extract(outcome.pdfBytes()));
}
private static String warnings(PdfUaConversionOutcome outcome) {
return "warnings: "
+ String.join(" | ", outcome.warnings())
+ " issues: "
+ outcome.validation().issues();
}
}
@Nested
@DisplayName("pre-marked content")
class PreMarked {
@Test
@DisplayName("existing BDC/EMC operators are stripped before new ones are written")
void stripsExistingMarkedContent() throws Exception {
byte[] premarked = premarkedDocument();
PdfUaConversionOutcome outcome = convert(premarked);
try (PDDocument document = Loader.loadPDF(outcome.pdfBytes())) {
Counts counts = countMarkedContent(document.getPage(0));
assertEquals(
counts.opens(),
counts.closes(),
"unbalanced marked content after stripping and re-injection");
assertFalse(
contentOf(document).contains("/OldTag"),
"the source's own marked content survived the rebuild");
}
assertEquals(extract(premarked), extract(outcome.pdfBytes()));
}
/** A document whose stream already contains a BDC sequence under a custom tag. */
private static byte[] premarkedDocument() throws Exception {
byte[] plain = PdfUaTestDocuments.simpleDocument();
try (PDDocument document = Loader.loadPDF(plain)) {
PDPage page = document.getPage(0);
String content;
try (InputStream in = page.getContents()) {
content = new String(in.readAllBytes(), StandardCharsets.ISO_8859_1);
}
String wrapped = "/OldTag <</MCID 0>> BDC\n" + content + "\nEMC\n";
var stream = new org.apache.pdfbox.pdmodel.common.PDStream(document);
try (var out = stream.createOutputStream()) {
out.write(wrapped.getBytes(StandardCharsets.ISO_8859_1));
}
page.setContents(stream);
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
return out.toByteArray();
}
}
private record Counts(int opens, int closes) {}
private static Counts countMarkedContent(PDPage page) throws IOException {
int opens = 0;
int closes = 0;
PDFStreamParser parser = new PDFStreamParser(page);
Object token;
while ((token = parser.parseNextToken()) != null) {
if (token instanceof Operator operator) {
switch (operator.getName()) {
case "BDC", "BMC" -> opens++;
case "EMC" -> closes++;
default -> {}
}
}
}
return new Counts(opens, closes);
}
private static String contentOf(PDDocument document) throws IOException {
try (InputStream in = document.getPage(0).getContents()) {
return new String(in.readAllBytes(), StandardCharsets.ISO_8859_1);
}
}
}
@Nested
@DisplayName("honesty rules")
class Honesty {
@Test
@DisplayName("suppressed text blocks the conformance claim even when validation passes")
void suppressedTextBlocksDeclaration() {
var structure = new stirling.software.proprietary.pdf.ua.DocumentStructure();
structure.setTextSuppressed(true);
var result = new stirling.software.proprietary.pdf.ua.TaggingResult(structure, true);
assertTrue(
result.isContentSuppressed(),
"the suppression flag must survive into the tagging result");
}
@Test
@DisplayName("dropped lines are reported per page by the analyser")
void analyserWarnsOnDroppedLines() {
PageContent dropped =
new PageContent(
0,
List.of(),
List.of(),
5,
false,
false,
true,
new stirling.software.proprietary.pdf.ua.BBox(0, 0, 595, 842));
var structure =
new stirling.software.proprietary.pdf.ua.LayoutAnalyzer()
.analyse(List.of(dropped));
assertTrue(structure.isTextSuppressed());
assertTrue(
structure.getWarnings().stream().anyMatch(w -> w.contains("page(s) 1")),
"warning should name the affected page: " + structure.getWarnings());
}
}
@Nested
@DisplayName("clause table")
class Clauses {
@Test
@DisplayName("subclauses resolve to their parent entry, not to a string prefix")
void subclauseLookupWalksSegments() {
assertNotNull(PdfUaValidationService.lookupClause("7.21.4.1"), "7.21.4.1 -> 7.21");
assertNotNull(PdfUaValidationService.lookupClause("7.18.1"), "7.18.1 -> 7.18");
var toUnicode = PdfUaValidationService.lookupClause("7.21.7");
assertNotNull(toUnicode);
assertFalse(
toUnicode.autoFixable(),
"a missing ToUnicode map is not fixable by embedding fonts");
assertNull(PdfUaValidationService.lookupClause("9.9.9"));
assertNull(PdfUaValidationService.lookupClause(null));
}
}
@Nested
@DisplayName("range claiming")
class Claiming {
@Test
@DisplayName("a figure drawn between two text runs on one line stays a figure")
void interleavedFigureIsNotSwallowed() throws Exception {
// Words at ordinals 0 and 2 with an image at ordinal 1 between them.
var line =
new stirling.software.proprietary.pdf.ua.TextLineInfo(
0,
"left right",
new stirling.software.proprietary.pdf.ua.BBox(50, 700, 400, 712),
11,
false,
0,
2,
false,
List.of(
new stirling.software.proprietary.pdf.ua.WordInfo(
"left",
new stirling.software.proprietary.pdf.ua.BBox(
50, 700, 90, 712),
0,
0,
11,
false),
new stirling.software.proprietary.pdf.ua.WordInfo(
"right",
new stirling.software.proprietary.pdf.ua.BBox(
360, 700, 400, 712),
2,
2,
11,
false)));
var image =
new stirling.software.proprietary.pdf.ua.MarkableOp(
1,
stirling.software.proprietary.pdf.ua.MarkableOp.Kind.IMAGE,
new stirling.software.proprietary.pdf.ua.BBox(150, 650, 350, 760),
"Im0");
PageContent page =
new PageContent(
0,
List.of(line),
List.of(image),
3,
false,
false,
false,
new stirling.software.proprietary.pdf.ua.BBox(0, 0, 595, 842));
var structure =
new stirling.software.proprietary.pdf.ua.LayoutAnalyzer()
.analyse(List.of(page));
List<StructBlock> figures = new java.util.ArrayList<>();
structure.visit(
block -> {
if (block.getType() == StructType.FIGURE) {
figures.add(block);
}
});
assertEquals(
1,
figures.size(),
"the image between the words must survive as its own figure");
}
}
}
@@ -0,0 +1,183 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import static org.springframework.test.web.servlet.request.MockMvcRequestBuilders.multipart;
import static org.springframework.test.web.servlet.result.MockMvcResultMatchers.header;
import static org.springframework.test.web.servlet.result.MockMvcResultMatchers.status;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import org.springframework.mock.web.MockMultipartFile;
import org.springframework.test.web.servlet.MockMvc;
import org.springframework.test.web.servlet.MvcResult;
import org.springframework.test.web.servlet.setup.MockMvcBuilders;
import com.fasterxml.jackson.databind.JsonNode;
import com.fasterxml.jackson.databind.ObjectMapper;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.service.PdfMetadataService;
import stirling.software.common.util.TempFileManager;
import stirling.software.proprietary.controller.api.converters.ConvertPdfToPdfUa;
import stirling.software.proprietary.controller.api.security.AccessibilityReportController;
/**
* Exercises the endpoints over HTTP, not through the service layer. Covers route mapping, multipart
* binding, and the headers and JSON a client depends on.
*/
class PdfUaHttpEndpointTest {
private static MockMvc convertMvc;
private static MockMvc reportMvc;
private static final ObjectMapper JSON = new ObjectMapper();
@BeforeAll
static void setUp() throws Exception {
PdfUaValidationService validation = new PdfUaValidationService();
validation.initialise();
PdfUaConversionService conversion =
new PdfUaConversionService(
validation,
new FontEmbeddingService(),
new CustomPDFDocumentFactory(
org.mockito.Mockito.mock(PdfMetadataService.class)));
// A real TempFileManager, so the streamed response path is exercised rather than mocked.
TempFileManager tempFiles =
new TempFileManager(
new stirling.software.common.util.TempFileRegistry(),
new stirling.software.common.model.ApplicationProperties());
// Stand in for the app's advice, which lives in core; without one every rejection is a 500.
var advice = new BadRequestAdvice();
convertMvc =
MockMvcBuilders.standaloneSetup(new ConvertPdfToPdfUa(conversion, tempFiles))
.setControllerAdvice(advice)
.build();
reportMvc =
MockMvcBuilders.standaloneSetup(
new AccessibilityReportController(
new AccessibilityAuditService(validation)))
.setControllerAdvice(advice)
.build();
}
private static MockMultipartFile upload(byte[] pdf, String name) {
return new MockMultipartFile("fileInput", name, "application/pdf", pdf);
}
/** Mirrors the one rule these endpoints rely on: a rejected input is a 400, not a 500. */
@org.springframework.web.bind.annotation.RestControllerAdvice
static class BadRequestAdvice {
@org.springframework.web.bind.annotation.ExceptionHandler(IllegalArgumentException.class)
org.springframework.http.ResponseEntity<String> badRequest(IllegalArgumentException ex) {
return org.springframework.http.ResponseEntity.badRequest().body(ex.getMessage());
}
}
@Test
@DisplayName("POST /api/v1/convert/pdf/ua returns a PDF and reports what it did in headers")
void conversionEndpointResponds() throws Exception {
MvcResult result =
convertMvc
.perform(
multipart("/api/v1/convert/pdf/ua")
.file(upload(PdfUaTestDocuments.simpleDocument(), "in.pdf"))
.param("language", "en-GB")
.param("title", "Over The Wire")
.param("embedFonts", "false"))
.andExpect(status().isOk())
.andExpect(header().exists("X-Stirling-UA-Declared"))
.andExpect(header().exists("X-Stirling-UA-Failures"))
.andReturn();
byte[] body = result.getResponse().getContentAsByteArray();
assertTrue(body.length > 0, "no document came back");
assertEquals(
"%PDF",
new String(body, 0, 4, java.nio.charset.StandardCharsets.ISO_8859_1),
"the response body is not a PDF");
assertEquals(
"true",
result.getResponse().getHeader("X-Stirling-UA-Declared"),
"a simple embedded-font document should convert and be declared conformant");
}
@Test
@DisplayName("POST /api/v1/security/accessibility-report returns the figure inventory as JSON")
void reportEndpointResponds() throws Exception {
MvcResult result =
reportMvc
.perform(
multipart("/api/v1/security/accessibility-report")
.file(upload(PdfUaTestDocuments.imageDocument(), "in.pdf"))
.param("profile", "ua1"))
.andExpect(status().isOk())
.andReturn();
JsonNode json = JSON.readTree(result.getResponse().getContentAsString());
assertEquals("PDF/UA-1", json.get("profile").asText());
assertNotNull(json.get("summary"), "the report should carry a summary");
JsonNode figures = json.get("figuresNeedingDescription");
assertNotNull(figures, "the field a caller needs to supply alt text is missing");
assertTrue(figures.isArray() && figures.size() > 0, "the image should be listed: " + json);
assertTrue(
figures.get(0).get("key").asText().matches("\\d+:\\d+"),
"the key must be usable in a follow-up conversion request");
}
@Test
@DisplayName("alt text supplied as form data reaches the converter over HTTP")
void altTextBindsFromFormData() throws Exception {
byte[] input = PdfUaTestDocuments.imageDocument();
// Discover the key the way a client would, through the report endpoint.
MvcResult reported =
reportMvc
.perform(
multipart("/api/v1/security/accessibility-report")
.file(upload(input, "in.pdf")))
.andExpect(status().isOk())
.andReturn();
String key =
JSON.readTree(reported.getResponse().getContentAsString())
.get("figuresNeedingDescription")
.get(0)
.get("key")
.asText();
MvcResult converted =
convertMvc
.perform(
multipart("/api/v1/convert/pdf/ua")
.file(upload(input, "in.pdf"))
.param("embedFonts", "false")
.param("altText", key + "=A blue rectangle"))
.andExpect(status().isOk())
.andReturn();
assertEquals(
"0",
converted.getResponse().getHeader("X-Stirling-UA-Figures-Needing-Alt"),
"the description posted as form data was not applied");
}
@Test
@DisplayName("a request with no file is rejected rather than processed")
void missingFileIsRejected() throws Exception {
convertMvc
.perform(
multipart("/api/v1/convert/pdf/ua")
.file(
new MockMultipartFile(
"fileInput",
"e.pdf",
"application/pdf",
new byte[0])))
.andExpect(status().is4xxClientError());
}
}
@@ -0,0 +1,249 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.List;
import java.util.stream.Stream;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.text.PDFTextStripper;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.service.PdfMetadataService;
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
/**
* Runs the converter over every PDF in the repository, where real-tool output breaks assumptions. A
* clean refusal counts as a pass; nothing may crash or make a false conformance claim.
*/
class PdfUaRealCorpusTest {
private static PdfUaConversionService service;
private static Path repoRoot;
/** Files the converter is expected to refuse rather than process. */
private static final List<String> EXPECTED_REJECTS = List.of("encrypted.pdf", "corrupted.pdf");
@BeforeAll
static void setUp() {
PdfUaValidationService validation = new PdfUaValidationService();
validation.initialise();
service =
new PdfUaConversionService(
validation,
new FontEmbeddingService(),
new CustomPDFDocumentFactory(
org.mockito.Mockito.mock(PdfMetadataService.class)));
repoRoot = findRepoRoot();
}
private static Path findRepoRoot() {
Path current = Path.of("").toAbsolutePath();
while (current != null && !Files.exists(current.resolve("settings.gradle"))) {
current = current.getParent();
}
return current;
}
private record Outcome(String name, String status, int failures, int elements, int artifacts) {}
private static TaggingOptions.TaggingOptionsBuilder options(String fallbackTitle) {
return TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.fallbackTitle(fallbackTitle)
.existingTags(TaggingOptions.ExistingTags.REBUILD);
}
@Test
@DisplayName("converts every PDF in the repository without crashing or lying about conformance")
void realCorpusConverts() throws Exception {
assertNotNull(repoRoot, "could not locate the repository root");
List<Path> pdfs = findPdfs();
assertTrue(pdfs.size() >= 20, "expected a substantial corpus, found " + pdfs.size());
List<Outcome> outcomes = new ArrayList<>();
List<String> crashes = new ArrayList<>();
java.util.Map<String, Integer> clauseFiles = new java.util.TreeMap<>();
java.util.Map<String, String> clauseText = new java.util.HashMap<>();
java.util.Map<String, String> clauseExamples = new java.util.HashMap<>();
for (Path pdf : pdfs) {
String name = repoRoot.relativize(pdf).toString().replace('\\', '/');
byte[] input;
try {
input = Files.readAllBytes(pdf);
} catch (IOException e) {
continue;
}
try {
String stem = pdf.getFileName().toString().replaceFirst("\\.pdf$", "");
// Fidelity is a tagging property, so measure it with font embedding off.
PdfUaConversionOutcome taggedOnly =
service.convert(input, options(stem).embedFonts(false).build());
assertTextPreserved(name, input, taggedOnly.pdfBytes());
PdfUaConversionOutcome outcome = service.convert(input, options(stem).build());
// Full pipeline too: Ghostscript can exit 0 having blanked the document.
assertTextPreserved(name + " (with font embedding)", input, outcome.pdfBytes());
outcomes.add(
new Outcome(
name,
outcome.declared() ? "CONFORMS" : "improved",
outcome.validation().totalFailures(),
outcome.tagging().taggedElements(),
outcome.tagging().artifacts()));
outcome.validation()
.issues()
.forEach(
issue -> {
clauseFiles.merge(issue.getClause(), 1, Integer::sum);
clauseText.putIfAbsent(
issue.getClause(), issue.getTechnicalMessage());
clauseExamples.putIfAbsent(issue.getClause(), name);
});
} catch (IOException e) {
// A refusal with an explanation is an acceptable outcome.
outcomes.add(new Outcome(name, "refused: " + e.getMessage(), 0, 0, 0));
} catch (RuntimeException e) {
crashes.add(name + " -> " + e);
outcomes.add(new Outcome(name, "CRASH: " + e, 0, 0, 0));
}
}
System.out.println(render(outcomes));
StringBuilder clauses =
new StringBuilder("\nBlocking clauses, by number of files affected\n");
clauseFiles.entrySet().stream()
.sorted(java.util.Map.Entry.<String, Integer>comparingByValue().reversed())
.forEach(
e ->
clauses.append(
String.format(
" clause %-9s %-3d files e.g. %s%n %s%n",
e.getKey(),
e.getValue(),
clauseExamples.get(e.getKey()),
abbreviate(clauseText.get(e.getKey())))));
System.out.println(clauses);
List<String> unexpectedCrashes =
crashes.stream()
.filter(c -> EXPECTED_REJECTS.stream().noneMatch(c::contains))
.toList();
assertTrue(
unexpectedCrashes.isEmpty(),
"converter crashed on: " + String.join("; ", unexpectedCrashes));
}
/** Tagging must not change extracted text; a diff means the rewrite corrupted the page. */
private static void assertTextPreserved(String name, byte[] before, byte[] after) {
String textBefore = safeExtract(before);
if (textBefore == null) {
// The source itself is unreadable, so there is nothing to compare against.
return;
}
String textAfter = safeExtract(after);
assertTrue(
textAfter != null, "the converted file could not be read back at all for " + name);
assertTrue(
textBefore.equals(textAfter),
"tagging changed extracted text for "
+ name
+ "\n before: "
+ preview(textBefore)
+ "\n after: "
+ preview(textAfter));
}
private static String abbreviate(String text) {
if (text == null) {
return "";
}
return text.length() <= 110 ? text : text.substring(0, 110) + "...";
}
private static String preview(String text) {
return text.length() <= 160 ? text : text.substring(0, 160) + "...";
}
private static String safeExtract(byte[] pdf) {
try (PDDocument document = Loader.loadPDF(pdf)) {
PDFTextStripper stripper = new PDFTextStripper();
stripper.setSortByPosition(true);
return normalise(stripper.getText(document));
} catch (Exception e) {
return null;
}
}
/** Drops invisible formatting characters: a rebuild legitimately loses soft hyphens. */
static String normalise(String text) {
StringBuilder sb = new StringBuilder(text.length());
text.codePoints()
.forEach(
cp -> {
if (Character.getType(cp) != Character.FORMAT && cp != 0x00AD) {
sb.appendCodePoint(cp);
}
});
return sb.toString().replaceAll("\\s+", " ").strip();
}
private List<Path> findPdfs() throws IOException {
try (Stream<Path> stream = Files.walk(repoRoot)) {
return stream.filter(Files::isRegularFile)
.filter(p -> p.toString().toLowerCase().endsWith(".pdf"))
.filter(p -> !p.toString().contains("node_modules"))
.filter(p -> !p.toString().contains(File_BUILD))
.filter(p -> !p.toString().contains(".git"))
.sorted(Comparator.comparing(Path::toString))
.toList();
}
}
private static final String File_BUILD = "build" + java.io.File.separator;
private static String render(List<Outcome> outcomes) {
StringBuilder sb = new StringBuilder("\nPDF/UA conversion over the repository corpus\n");
long conforming = outcomes.stream().filter(o -> "CONFORMS".equals(o.status())).count();
long refused = outcomes.stream().filter(o -> o.status().startsWith("refused")).count();
sb.append(
String.format(
" %d files: %d conform, %d improved but not conformant, %d refused%n%n",
outcomes.size(),
conforming,
outcomes.size() - conforming - refused,
refused));
for (Outcome outcome : outcomes) {
sb.append(
String.format(
" %-62s %-10s fail=%-4d el=%-5d art=%d%n",
outcome.name().length() > 60
? "..." + outcome.name().substring(outcome.name().length() - 57)
: outcome.name(),
outcome.status().length() > 10
? outcome.status().substring(0, 10)
: outcome.status(),
outcome.failures(),
outcome.elements(),
outcome.artifacts()));
}
return sb.toString();
}
}
@@ -0,0 +1,103 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.List;
import java.util.stream.Stream;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.condition.EnabledIfEnvironmentVariable;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.service.PdfMetadataService;
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
/**
* Dumps converted output for independent checkers; validating PDFBox with PDFBox is circular. Off
* by default as it writes outside the build directory: run with {@code DUMP_UA_SAMPLES=<dir>}.
*/
@EnabledIfEnvironmentVariable(named = "DUMP_UA_SAMPLES", matches = ".+")
class PdfUaSampleDumpTest {
private static PdfUaConversionService service;
private static Path repoRoot;
@BeforeAll
static void setUp() {
PdfUaValidationService validation = new PdfUaValidationService();
validation.initialise();
service =
new PdfUaConversionService(
validation,
new FontEmbeddingService(),
new CustomPDFDocumentFactory(
org.mockito.Mockito.mock(PdfMetadataService.class)));
repoRoot = Path.of("").toAbsolutePath();
while (repoRoot != null && !Files.exists(repoRoot.resolve("settings.gradle"))) {
repoRoot = repoRoot.getParent();
}
}
@Test
@DisplayName("writes original and converted pairs for external validation")
void dumpSamples() throws Exception {
Path out = Path.of(System.getenv("DUMP_UA_SAMPLES"));
Files.createDirectories(out);
List<Path> pdfs;
try (Stream<Path> stream = Files.walk(repoRoot)) {
pdfs =
stream.filter(Files::isRegularFile)
.filter(p -> p.toString().toLowerCase().endsWith(".pdf"))
.filter(p -> !p.toString().contains("node_modules"))
.filter(p -> !p.toString().contains(java.io.File.separator + "build"))
.filter(p -> !p.toString().contains(".git"))
.sorted()
.toList();
}
List<String> manifest = new ArrayList<>();
int written = 0;
for (Path pdf : pdfs) {
String stem = pdf.getFileName().toString().replaceFirst("\\.pdf$", "");
byte[] input;
try {
input = Files.readAllBytes(pdf);
} catch (Exception e) {
continue;
}
try {
PdfUaConversionOutcome outcome =
service.convert(
input,
TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.fallbackTitle(stem)
.existingTags(TaggingOptions.ExistingTags.REBUILD)
.build());
Files.write(out.resolve(stem + "__before.pdf"), input);
Files.write(out.resolve(stem + "__after.pdf"), outcome.pdfBytes());
manifest.add(
stem
+ "\tdeclared="
+ outcome.declared()
+ "\tfailures="
+ outcome.validation().totalFailures());
written++;
} catch (Exception e) {
manifest.add(stem + "\tREFUSED\t" + e.getMessage());
}
}
Files.write(out.resolve("manifest.tsv"), manifest);
System.out.println("Wrote " + written + " before/after pairs to " + out);
assertTrue(written > 10, "expected a usable sample set, wrote " + written);
}
}
@@ -0,0 +1,319 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertThrows;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.encryption.AccessPermission;
import org.apache.pdfbox.pdmodel.encryption.StandardProtectionPolicy;
import org.apache.pdfbox.pdmodel.font.PDType1Font;
import org.apache.pdfbox.pdmodel.font.Standard14Fonts;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Nested;
import org.junit.jupiter.api.Test;
import stirling.software.common.service.CustomPDFDocumentFactory;
import stirling.software.common.service.PdfMetadataService;
import stirling.software.proprietary.model.api.ua.AccessibilityReport;
import stirling.software.proprietary.model.api.ua.PdfUaConversionOutcome;
import stirling.software.proprietary.model.api.ua.UaValidationResult;
import stirling.software.proprietary.pdf.ua.PdfUaProfile;
import stirling.software.proprietary.pdf.ua.TaggingOptions;
/** Tests for the services that validate, audit and convert. */
class PdfUaServicesTest {
private static PdfUaValidationService validation;
private static PdfUaConversionService conversion;
private static AccessibilityAuditService audit;
private static FontEmbeddingService fonts;
@BeforeAll
static void setUp() {
validation = new PdfUaValidationService();
validation.initialise();
fonts = new FontEmbeddingService();
conversion =
new PdfUaConversionService(
validation,
fonts,
new CustomPDFDocumentFactory(
org.mockito.Mockito.mock(PdfMetadataService.class)));
audit = new AccessibilityAuditService(validation);
}
/** Uses a standard 14 font deliberately: never embedded, which clause 7.21 forbids. */
private static byte[] unembeddedFontPdf() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
cs.beginText();
cs.setFont(new PDType1Font(Standard14Fonts.FontName.HELVETICA), 12);
cs.newLineAtOffset(50, 700);
cs.showText("Hello accessibility");
cs.endText();
}
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
return out.toByteArray();
}
}
private static byte[] manyPages(int pages) throws IOException {
try (PDDocument document = new PDDocument()) {
for (int i = 0; i < pages; i++) {
document.addPage(new PDPage());
}
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
return out.toByteArray();
}
}
private static byte[] encryptedPdf() throws IOException {
try (PDDocument document = new PDDocument()) {
document.addPage(new PDPage());
AccessPermission permissions = new AccessPermission();
document.protect(new StandardProtectionPolicy("owner", "user", permissions));
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
return out.toByteArray();
}
}
@Nested
@DisplayName("validation")
class Validation {
@Test
@DisplayName("an untagged document fails and the failures are grouped by rule")
void untaggedFails() throws Exception {
UaValidationResult result = validation.validate(unembeddedFontPdf(), PdfUaProfile.UA1);
assertFalse(result.compliant());
assertTrue(result.hasIssues());
assertTrue(
result.totalFailures() >= result.issues().size(),
"grouping must not invent failures");
assertTrue(
result.issues().stream().allMatch(i -> i.getOccurrences() > 0),
"every grouped issue should count its occurrences");
}
@Test
@DisplayName("malformed input reports a failure instead of throwing")
void malformedInputIsReported() {
UaValidationResult result =
validation.validate("not a pdf".getBytes(), PdfUaProfile.UA1);
assertFalse(result.compliant());
assertFalse(result.issues().isEmpty());
}
@Test
@DisplayName("issues carry plain-English text as well as the validator's own wording")
void issuesAreReadable() throws Exception {
UaValidationResult result = validation.validate(unembeddedFontPdf(), PdfUaProfile.UA1);
assertTrue(
result.issues().stream()
.allMatch(i -> i.getMessage() != null && !i.getMessage().isBlank()));
}
}
@Nested
@DisplayName("auditing")
class Auditing {
@Test
@DisplayName("reports the document facts that drive most failures")
void reportsSummary() throws Exception {
AccessibilityReport report = audit.audit(unembeddedFontPdf(), PdfUaProfile.UA1);
assertFalse(report.isTagged(), "the fixture has no structure tree");
assertFalse(report.isDeclaresConformance());
assertFalse(report.isPassesAutomatedChecks());
assertEquals(1, report.getSummary().getPages());
assertFalse(report.getSummary().isAllFontsEmbedded());
assertTrue(report.getSummary().getUnembeddedFonts() > 0);
assertFalse(report.getSummary().isHasLanguage());
assertFalse(report.getSummary().isHasTitle());
}
@Test
@DisplayName("always lists the checks a person still has to make")
void listsHumanChecks() throws Exception {
AccessibilityReport report = audit.audit(unembeddedFontPdf(), PdfUaProfile.UA1);
assertFalse(
report.getHumanChecks().isEmpty(),
"a report showing only automated results implies the rest does not exist");
}
@Test
@DisplayName("splits failures into automatically fixable and needs-input")
void splitsRemediability() throws Exception {
AccessibilityReport report = audit.audit(unembeddedFontPdf(), PdfUaProfile.UA1);
assertEquals(
report.getIssues().size(),
report.getAutomaticallyFixable() + report.getNeedsInput());
}
@Test
@DisplayName("refuses a document past the page cap the conversion also applies")
void refusesTooManyPages() throws Exception {
byte[] oversized = manyPages(2001);
assertThrows(
IllegalArgumentException.class,
() -> audit.audit(oversized, PdfUaProfile.UA1),
"an uncapped report walks every page of any document a caller uploads");
}
@Test
@DisplayName("a converted document reports as tagged and conformant")
void reportsAfterConversion() throws Exception {
PdfUaConversionOutcome outcome =
conversion.convert(
unembeddedFontPdf(),
TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.title("Converted")
.build());
AccessibilityReport report = audit.audit(outcome.pdfBytes(), PdfUaProfile.UA1);
assertTrue(report.isTagged());
assertTrue(report.getSummary().isHasTitle());
assertTrue(report.getSummary().isHasLanguage());
assertTrue(report.getSummary().isDisplaysDocTitle());
}
}
@Nested
@DisplayName("font embedding")
class Fonts {
@Test
@DisplayName("detects a standard 14 font as unembedded")
void detectsUnembedded() throws Exception {
assertTrue(fonts.hasUnembeddedFonts(unembeddedFontPdf()));
}
@Test
@DisplayName("embeds fonts, or explains why it could not")
void embedsOrExplains() throws Exception {
FontEmbeddingService.Result result = fonts.embedFonts(unembeddedFontPdf());
assertNotNull(result.pdfBytes());
if (result.changed()) {
assertFalse(
fonts.hasUnembeddedFonts(result.pdfBytes()),
"embedding reported success but fonts are still missing");
} else {
assertNotNull(
result.warning(), "failing to embed must be explained, not passed over");
}
}
@Test
@DisplayName("leaves a document alone when every font is already embedded")
void skipsWhenNothingToDo() throws Exception {
byte[] embedded = PdfUaTestDocuments.simpleDocument();
FontEmbeddingService.Result result = fonts.embedFonts(embedded);
assertFalse(result.changed());
assertEquals(embedded.length, result.pdfBytes().length);
}
}
@Nested
@DisplayName("conversion")
class Conversion {
@Test
@DisplayName("refuses an encrypted document with an explanation")
void refusesEncrypted() throws Exception {
byte[] encrypted = encryptedPdf();
IOException error =
assertThrows(
IOException.class,
() ->
conversion.convert(
encrypted,
TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.build()));
assertTrue(error.getMessage().toLowerCase().contains("encrypted"));
}
@Test
@DisplayName("keeping existing tags does not rebuild the tree")
void keepRespectsExistingTags() throws Exception {
byte[] tagged =
conversion
.convert(
PdfUaTestDocuments.simpleDocument(),
TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.title("First pass")
.embedFonts(false)
.build())
.pdfBytes();
PdfUaConversionOutcome second =
conversion.convert(
tagged,
TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.title("Second pass")
.embedFonts(false)
.existingTags(TaggingOptions.ExistingTags.KEEP)
.build());
assertFalse(second.tagging().rebuiltStructure(), "KEEP must not rebuild");
try (PDDocument document = Loader.loadPDF(second.pdfBytes())) {
assertEquals("Second pass", document.getDocumentInformation().getTitle());
}
}
@Test
@DisplayName("marking images decorative removes the alt-text blocker")
void decorativePolicyClearsFigures() throws Exception {
PdfUaConversionOutcome outcome =
conversion.convert(
PdfUaTestDocuments.imageDocument(),
TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.title("Decorative")
.embedFonts(false)
.figurePolicy(TaggingOptions.FigurePolicy.MARK_DECORATIVE)
.build());
assertEquals(0, outcome.tagging().figuresNeedingAltText());
assertTrue(outcome.declared(), "with no undescribed figures the file should conform");
}
@Test
@DisplayName("converting twice produces the same conformance verdict")
void conversionIsStable() throws Exception {
byte[] input = PdfUaTestDocuments.headingHierarchy();
TaggingOptions options =
TaggingOptions.builder()
.profile(PdfUaProfile.UA1)
.language("en-GB")
.title("Stable")
.embedFonts(false)
.build();
PdfUaConversionOutcome first = conversion.convert(input, options);
PdfUaConversionOutcome second = conversion.convert(first.pdfBytes(), options);
assertEquals(first.declared(), second.declared());
}
}
}
@@ -0,0 +1,390 @@
package stirling.software.proprietary.service.ua;
import java.awt.Color;
import java.awt.image.BufferedImage;
import java.io.ByteArrayOutputStream;
import java.io.IOException;
import java.io.InputStream;
import java.nio.file.Files;
import java.nio.file.Path;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.pdmodel.PDFormContentStream;
import org.apache.pdfbox.pdmodel.PDPage;
import org.apache.pdfbox.pdmodel.PDPageContentStream;
import org.apache.pdfbox.pdmodel.PDResources;
import org.apache.pdfbox.pdmodel.common.PDRectangle;
import org.apache.pdfbox.pdmodel.font.PDFont;
import org.apache.pdfbox.pdmodel.font.PDType0Font;
import org.apache.pdfbox.pdmodel.graphics.form.PDFormXObject;
import org.apache.pdfbox.pdmodel.graphics.image.LosslessFactory;
import org.apache.pdfbox.pdmodel.graphics.image.PDImageXObject;
import org.apache.pdfbox.pdmodel.interactive.action.PDActionURI;
import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationLink;
import org.apache.pdfbox.util.Matrix;
/**
* Builds the fixture corpus used by the PDF/UA tests. Fonts are embedded deliberately: the standard
* 14 fail clause 7.21 and would mask every result.
*/
final class PdfUaTestDocuments {
private static final String FONT_RESOURCE = "/static/fonts/DejaVuSans.ttf";
// The font ships with core's resources, which are not on this module's classpath.
private static final String FONT_REPO_PATH =
"app/core/src/main/resources/static/fonts/DejaVuSans.ttf";
private static final float MARGIN = 60f;
private PdfUaTestDocuments() {}
static PDFont font(PDDocument document) throws IOException {
try (InputStream in = PdfUaTestDocuments.class.getResourceAsStream(FONT_RESOURCE)) {
if (in != null) {
return PDType0Font.load(document, in, true);
}
}
Path repoRoot = Path.of("").toAbsolutePath();
while (repoRoot != null && !Files.exists(repoRoot.resolve("settings.gradle"))) {
repoRoot = repoRoot.getParent();
}
Path font = repoRoot == null ? null : repoRoot.resolve(FONT_REPO_PATH);
if (font == null || !Files.exists(font)) {
throw new IOException(
"Test font not found: " + FONT_RESOURCE + " or " + FONT_REPO_PATH);
}
try (InputStream in = Files.newInputStream(font)) {
return PDType0Font.load(document, in, true);
}
}
/** A heading followed by two paragraphs: the simplest thing that should convert cleanly. */
static byte[] simpleDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDFont font = font(document);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
float y = 760;
y = text(cs, font, 20, MARGIN, y, "Quarterly Report");
y -= 14;
y = text(cs, font, 11, MARGIN, y, "This document summarises the results for the");
y = text(cs, font, 11, MARGIN, y, "period and outlines the outlook for next year.");
y -= 14;
text(cs, font, 11, MARGIN, y, "A second paragraph follows the first one here.");
}
return bytes(document);
}
}
/** Three heading tiers, to exercise level assignment and the no-skipped-levels rule. */
static byte[] headingHierarchy() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDFont font = font(document);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
float y = 760;
y = text(cs, font, 24, MARGIN, y, "Annual Review");
y -= 12;
y = text(cs, font, 11, MARGIN, y, "Introductory prose sits under the title here.");
y -= 16;
y = text(cs, font, 17, MARGIN, y, "Financial Results");
y -= 10;
y = text(cs, font, 11, MARGIN, y, "Revenue grew steadily across every region.");
y -= 16;
y = text(cs, font, 13, MARGIN, y, "Europe");
y -= 10;
text(cs, font, 11, MARGIN, y, "European revenue rose by eleven per cent.");
}
return bytes(document);
}
}
/** A bulleted and a numbered list. */
static byte[] listDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDFont font = font(document);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
float y = 760;
y = text(cs, font, 18, MARGIN, y, "Checklist");
y -= 14;
y = text(cs, font, 11, MARGIN, y, "• Review the source document");
y = text(cs, font, 11, MARGIN, y, "• Check every heading level");
y = text(cs, font, 11, MARGIN, y, "• Describe each image");
y -= 16;
y = text(cs, font, 11, MARGIN, y, "1. Open the file");
y = text(cs, font, 11, MARGIN, y, "2. Run the converter");
text(cs, font, 11, MARGIN, y, "3. Validate the result");
}
return bytes(document);
}
}
/** A three-column table whose cells each occupy their own text-showing operator. */
static byte[] tableDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDFont font = font(document);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
float y = 760;
y = text(cs, font, 18, MARGIN, y, "Regional Totals");
y -= 20;
String[][] rows = {
{"Region", "Units", "Revenue"},
{"North", "1200", "48000"},
{"South", "980", "39200"},
{"East", "1430", "57200"}
};
float[] columns = {MARGIN, MARGIN + 160, MARGIN + 300};
for (String[] row : rows) {
tableRow(cs, font, 11, columns, y, row);
y -= 20;
}
}
return bytes(document);
}
}
/** A page with a real image, which must end up as a Figure needing alternative text. */
static byte[] imageDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDFont font = font(document);
BufferedImage bitmap = new BufferedImage(120, 90, BufferedImage.TYPE_INT_RGB);
java.awt.Graphics2D graphics = bitmap.createGraphics();
graphics.setColor(Color.BLUE);
graphics.fillRect(0, 0, 120, 90);
graphics.dispose();
PDImageXObject image = LosslessFactory.createFromImage(document, bitmap);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
float y = 760;
y = text(cs, font, 18, MARGIN, y, "Illustrated Page");
y -= 20;
y = text(cs, font, 11, MARGIN, y, "The chart below shows the trend.");
cs.drawImage(image, MARGIN, y - 120, 180, 100);
}
return bytes(document);
}
}
/** Four pages sharing a running head and a page number, which must become artifacts. */
static byte[] runningHeadersDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDFont font = null;
for (int i = 1; i <= 4; i++) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
if (font == null) {
font = font(document);
}
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
text(cs, font, 9, MARGIN, 810, "Confidential Internal Report");
float y = 750;
y = text(cs, font, 16, MARGIN, y, "Section " + i);
y -= 12;
text(cs, font, 11, MARGIN, y, "Body text for section number " + i + " here.");
text(cs, font, 9, 300, 30, "Page " + i);
}
}
return bytes(document);
}
}
/** Two columns of prose, to exercise reading order. */
static byte[] twoColumnDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDFont font = font(document);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
float left = MARGIN;
float right = 320;
float y = 740;
for (int i = 1; i <= 8; i++) {
text(cs, font, 10, left, y - i * 16, "Left column line number " + i);
text(cs, font, 10, right, y - i * 16, "Right column line number " + i);
}
}
return bytes(document);
}
}
/** A page carrying a link annotation, which must be reachable from the structure tree. */
static byte[] linkDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDFont font = font(document);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
float y = 760;
y = text(cs, font, 18, MARGIN, y, "Useful Links");
text(cs, font, 11, MARGIN, y - 20, "Visit the project home page for details.");
}
PDAnnotationLink link = new PDAnnotationLink();
PDRectangle rectangle = new PDRectangle();
rectangle.setLowerLeftX(MARGIN);
rectangle.setLowerLeftY(725);
rectangle.setUpperRightX(MARGIN + 200);
rectangle.setUpperRightY(740);
link.setRectangle(rectangle);
PDActionURI action = new PDActionURI();
action.setURI("https://example.org");
link.setAction(action);
page.getAnnotations().add(link);
return bytes(document);
}
}
/** A page with no content at all. */
static byte[] emptyDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
document.addPage(new PDPage(PDRectangle.A4));
return bytes(document);
}
}
/** A page whose only content is a full-page image, standing in for an un-OCRed scan. */
static byte[] scannedDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
BufferedImage bitmap = new BufferedImage(600, 850, BufferedImage.TYPE_INT_RGB);
java.awt.Graphics2D graphics = bitmap.createGraphics();
graphics.setColor(Color.WHITE);
graphics.fillRect(0, 0, 600, 850);
graphics.setColor(Color.BLACK);
graphics.drawString("scanned page", 40, 60);
graphics.dispose();
PDImageXObject image = LosslessFactory.createFromImage(document, bitmap);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
cs.drawImage(image, 0, 0, PDRectangle.A4.getWidth(), PDRectangle.A4.getHeight());
}
return bytes(document);
}
}
/** Text drawn inside a form XObject, attributed to the Do operator that invoked it. */
static byte[] formXObjectDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDFont font = font(document);
PDFormXObject form = new PDFormXObject(document);
form.setBBox(new PDRectangle(220, 40));
form.setResources(new PDResources());
try (PDFormContentStream fcs = new PDFormContentStream(form)) {
fcs.beginText();
fcs.setFont(font, 11);
fcs.newLineAtOffset(4, 14);
fcs.showText("Text living inside the form XObject");
fcs.endText();
}
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
text(cs, font, 18, MARGIN, 760, "Page With Embedded Form");
text(cs, font, 11, MARGIN, 730, "Ordinary page text sits above the form.");
cs.saveGraphicsState();
cs.transform(Matrix.getTranslateInstance(MARGIN, 650));
cs.drawForm(form);
cs.restoreGraphicsState();
}
return bytes(document);
}
}
/** Content split across two streams (a PDF array) - the parser must see one sequence. */
static byte[] multiStreamDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
document.addPage(page);
PDFont font = font(document);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
text(cs, font, 18, MARGIN, 760, "First Stream Heading");
}
try (PDPageContentStream cs =
new PDPageContentStream(
document, page, PDPageContentStream.AppendMode.APPEND, true)) {
text(cs, font, 11, MARGIN, 720, "Second stream paragraph appended later.");
}
return bytes(document);
}
}
/** A landscape page via /Rotate 90, which flips the frame the text engine reports in. */
static byte[] rotatedDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(PDRectangle.A4);
page.setRotation(90);
document.addPage(page);
PDFont font = font(document);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
// Drawn rotated so the text reads upright on the rotated page.
cs.transform(Matrix.getRotateInstance(Math.toRadians(90), 595, 0));
text(cs, font, 18, MARGIN, 500, "Rotated Page Title");
text(cs, font, 11, MARGIN, 470, "Body text on a landscape page.");
}
return bytes(document);
}
}
/** A MediaBox whose origin is not (0,0), which some scanners produce. */
static byte[] offsetMediaBoxDocument() throws IOException {
try (PDDocument document = new PDDocument()) {
PDPage page = new PDPage(new PDRectangle(100, 200, 595, 842));
document.addPage(page);
PDFont font = font(document);
try (PDPageContentStream cs = new PDPageContentStream(document, page)) {
text(cs, font, 18, 160, 960, "Offset Origin Title");
text(cs, font, 11, 160, 930, "Text on a page whose MediaBox starts at 100,200.");
}
return bytes(document);
}
}
// --- helpers -----------------------------------------------------------
private static float text(
PDPageContentStream cs, PDFont font, float size, float x, float y, String value)
throws IOException {
cs.beginText();
cs.setFont(font, size);
cs.newLineAtOffset(x, y);
cs.showText(value);
cs.endText();
return y - size * 1.35f;
}
/**
* Emits one row with a separate show-text operator per cell, so each cell gets its own MCID.
*/
private static void tableRow(
PDPageContentStream cs,
PDFont font,
float size,
float[] columns,
float y,
String[] values)
throws IOException {
cs.beginText();
cs.setFont(font, size);
cs.newLineAtOffset(columns[0], y);
cs.showText(values[0]);
for (int i = 1; i < values.length; i++) {
cs.newLineAtOffset(columns[i] - columns[i - 1], 0);
cs.showText(values[i]);
}
cs.endText();
}
private static byte[] bytes(PDDocument document) throws IOException {
ByteArrayOutputStream out = new ByteArrayOutputStream();
document.save(out);
return out.toByteArray();
}
}
@@ -0,0 +1,202 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.ByteArrayInputStream;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.List;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import org.verapdf.pdfa.Foundries;
import org.verapdf.pdfa.PDFAParser;
import org.verapdf.pdfa.flavours.PDFAFlavour;
import org.verapdf.pdfa.results.TestAssertion;
import org.verapdf.pdfa.results.ValidationResult;
/**
* Proves tagging raises a PDF/A file from level B to the accessible level A. veraPDF is the
* arbiter: the claim only counts if the validator agrees.
*/
class PdfaLevelATest {
private static PdfaAccessibilityService service;
private static Path repoRoot;
@BeforeAll
static void setUp() {
PdfUaValidationService uaValidation = new PdfUaValidationService();
uaValidation.initialise();
service = new PdfaAccessibilityService(uaValidation);
repoRoot = Path.of("").toAbsolutePath();
while (repoRoot != null && !Files.exists(repoRoot.resolve("settings.gradle"))) {
repoRoot = repoRoot.getParent();
}
}
private static byte[] fixture(String name) throws Exception {
return Files.readAllBytes(
repoRoot.resolve("app/core/src/test/resources/pdfa").resolve(name));
}
private static String xmpOf(byte[] pdf) throws Exception {
try (PDDocument document = Loader.loadPDF(pdf)) {
var metadata = document.getDocumentCatalog().getMetadata();
assertNotNull(metadata, "no XMP packet");
return new String(metadata.toByteArray(), StandardCharsets.UTF_8);
}
}
/** The flavour the file declares in its XMP, which is what a validator picks up by itself. */
private static String declaredStandard(byte[] pdf) throws Exception {
try (PDFAParser parser =
Foundries.defaultInstance().createParser(new ByteArrayInputStream(pdf))) {
List<PDFAFlavour> flavours = parser.getFlavours();
return flavours == null || flavours.isEmpty() ? null : flavours.get(0).getId();
}
}
private static ValidationResult validate(byte[] pdf, PDFAFlavour flavour) throws Exception {
try (PDFAParser parser =
Foundries.defaultInstance().createParser(new ByteArrayInputStream(pdf), flavour)) {
return Foundries.defaultInstance().createValidator(flavour, false).validate(parser);
}
}
private static List<String> failures(ValidationResult result) {
return result.getTestAssertions().stream()
.filter(assertion -> assertion.getStatus() == TestAssertion.Status.FAILED)
.map(TestAssertion::getMessage)
.toList();
}
@Test
@DisplayName("a level B file gains a structure tree and a conformance A claim")
void upgradesLevelBToLevelA() throws Exception {
byte[] levelB = fixture("valid-pdfa-2b.pdf");
try (PDDocument before = Loader.loadPDF(levelB)) {
assertEquals(
null,
before.getDocumentCatalog().getStructureTreeRoot(),
"the fixture should start untagged, or the test proves nothing");
}
PdfaAccessibilityService.Result result =
service.upgradeToLevelA(levelB, 2, "en-GB", "Archived Report");
assertTrue(result.levelA(), "upgrade failed: " + result.warnings());
try (PDDocument after = Loader.loadPDF(result.pdfBytes())) {
assertNotNull(
after.getDocumentCatalog().getStructureTreeRoot(), "no structure tree written");
assertTrue(after.getDocumentCatalog().getMarkInfo().isMarked());
assertEquals("en-GB", after.getDocumentCatalog().getLanguage());
}
String xmp = xmpOf(result.pdfBytes());
assertTrue(xmp.contains("part"), "pdfaid:part missing");
assertTrue(
xmp.contains(">A<") || xmp.contains("conformance=\"A\""),
"conformance was not raised to A: " + xmp);
}
@Test
@DisplayName("the upgraded file still validates as PDF/A, now at level A")
void upgradedFileStillValidates() throws Exception {
byte[] levelB = fixture("valid-pdfa-2b.pdf");
PdfaAccessibilityService.Result result =
service.upgradeToLevelA(levelB, 2, "en-GB", "Archived Report");
assertTrue(result.levelA(), "upgrade failed: " + result.warnings());
assertEquals(
"2a", declaredStandard(result.pdfBytes()), "the file should now declare PDF/A-2a");
ValidationResult pdfa = validate(result.pdfBytes(), PDFAFlavour.PDFA_2_A);
assertTrue(pdfa.isCompliant(), () -> "PDF/A-2a validation failed: " + failures(pdfa));
}
@Test
@DisplayName("PDF/A-1 keeps its 1.4 version, since level A must not change the part")
void partOneKeepsItsVersion() throws Exception {
byte[] levelB = fixture("valid-pdfa-1b.pdf");
float versionBefore;
try (PDDocument document = Loader.loadPDF(levelB)) {
versionBefore = document.getVersion();
}
PdfaAccessibilityService.Result result =
service.upgradeToLevelA(levelB, 1, "en", "Archived");
try (PDDocument document = Loader.loadPDF(result.pdfBytes())) {
assertEquals(
versionBefore,
document.getVersion(),
"raising the PDF version would break PDF/A-1 conformance");
}
}
@Test
@DisplayName("a document with nothing to tag is left at level B rather than mislabelled")
void refusesToClaimLevelAWithoutTags() throws Exception {
byte[] blank;
try (PDDocument document = new PDDocument()) {
document.addPage(new org.apache.pdfbox.pdmodel.PDPage());
var out = new java.io.ByteArrayOutputStream();
document.save(out);
blank = out.toByteArray();
}
PdfaAccessibilityService.Result result = service.upgradeToLevelA(blank, 2, "en", "Empty");
assertFalse(result.levelA(), "an untaggable document must not claim level A");
assertFalse(result.warnings().isEmpty(), "the refusal should be explained");
}
@Test
@DisplayName("setting conformance leaves the rest of the XMP packet intact")
void conformanceRewritePreservesPacket() throws Exception {
byte[] levelB = fixture("valid-pdfa-2b.pdf");
byte[] rewritten = PdfaAccessibilityService.setConformance(levelB, 2, "A");
assertEquals("2a", declaredStandard(rewritten), "the rewritten packet should declare 2a");
}
@Test
@DisplayName("a file can declare PDF/A and PDF/UA at once without breaking either")
void combinedPdfaAndPdfUa() throws Exception {
byte[] levelB = fixture("valid-pdfa-2b.pdf");
PdfaAccessibilityService.Result upgraded =
service.upgradeToLevelA(levelB, 2, "en-GB", "Archived and Accessible");
assertTrue(upgraded.levelA(), "upgrade failed: " + upgraded.warnings());
byte[] both = PdfaAccessibilityService.declarePdfUaAlongsidePdfa(upgraded.pdfBytes(), 2);
String xmp = xmpOf(both);
assertTrue(xmp.contains("pdfuaid"), "no PDF/UA identifier");
assertTrue(
xmp.contains("pdfaSchema") || xmp.contains("schemas"),
"PDF/A requires an extension schema describing pdfuaid, none found: " + xmp);
assertEquals(
"2a", declaredStandard(both), "the combined file should still declare PDF/A-2a");
ValidationResult pdfa = validate(both, PDFAFlavour.PDFA_2_A);
assertTrue(
pdfa.isCompliant(),
() -> "adding the PDF/UA identifier broke PDF/A: " + failures(pdfa));
}
@Test
@DisplayName("PDFAFlavour exposes the level A profiles the converter now targets")
void flavoursExistForLevelA() {
assertNotNull(PDFAFlavour.PDFA_1_A);
assertNotNull(PDFAFlavour.PDFA_2_A);
assertNotNull(PDFAFlavour.PDFA_3_A);
}
}
@@ -0,0 +1,99 @@
package stirling.software.proprietary.service.ua;
import static org.junit.jupiter.api.Assertions.assertNotNull;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.List;
import java.util.stream.Stream;
import org.apache.pdfbox.Loader;
import org.apache.pdfbox.pdmodel.PDDocument;
import org.apache.pdfbox.text.PDFTextStripper;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Test;
import stirling.software.proprietary.pdf.ua.PageContent;
import stirling.software.proprietary.pdf.ua.TaggedContentExtractor;
/**
* Guards the invariant the tagger rests on: the token pass and text pass must agree on ordinals.
* They can silently disagree, and the extractor then drops the page rather than mis-tag it.
*/
class TaggedContentExtractorRealFilesTest {
private static Path repoRoot;
@BeforeAll
static void setUp() {
repoRoot = Path.of("").toAbsolutePath();
while (repoRoot != null && !Files.exists(repoRoot.resolve("settings.gradle"))) {
repoRoot = repoRoot.getParent();
}
}
@Test
@DisplayName("pages with extractable text always yield lines across the repository corpus")
void ordinalsAgreeOnRealFiles() throws Exception {
assertNotNull(repoRoot, "could not locate the repository root");
List<String> dropped = new ArrayList<>();
int inspected = 0;
for (Path pdf : findPdfs()) {
byte[] bytes;
try {
bytes = Files.readAllBytes(pdf);
} catch (Exception e) {
continue;
}
try (PDDocument document = Loader.loadPDF(bytes)) {
if (document.getNumberOfPages() > 30) {
continue;
}
inspected++;
List<PageContent> pages = new TaggedContentExtractor().extract(document);
for (PageContent page : pages) {
if (page.markableCount() == 0 || !hasText(document, page.pageIndex())) {
continue;
}
if (page.lines().isEmpty() && page.forms().isEmpty()) {
dropped.add(repoRoot.relativize(pdf) + " page " + page.pageIndex());
}
}
} catch (Exception e) {
// Unreadable files are covered by the conversion tests.
}
}
assertTrue(inspected > 15, "expected to inspect a real corpus, saw " + inspected);
assertTrue(
dropped.isEmpty(),
"the two extraction passes disagreed, so these pages were skipped: " + dropped);
}
private static boolean hasText(PDDocument document, int pageIndex) {
try {
PDFTextStripper stripper = new PDFTextStripper();
stripper.setStartPage(pageIndex + 1);
stripper.setEndPage(pageIndex + 1);
return !stripper.getText(document).isBlank();
} catch (Exception e) {
return false;
}
}
private List<Path> findPdfs() throws Exception {
try (Stream<Path> stream = Files.walk(repoRoot)) {
return stream.filter(Files::isRegularFile)
.filter(p -> p.toString().toLowerCase().endsWith(".pdf"))
.filter(p -> !p.toString().contains("node_modules"))
.filter(p -> !p.toString().contains(java.io.File.separator + "build"))
.filter(p -> !p.toString().contains(".git"))
.toList();
}
}
}
@@ -0,0 +1,53 @@
package stirling.software.saas.config;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.boot.context.event.ApplicationReadyEvent;
import org.springframework.context.annotation.Profile;
import org.springframework.context.event.EventListener;
import org.springframework.core.env.Environment;
import org.springframework.stereotype.Component;
import lombok.extern.slf4j.Slf4j;
/** Logs which Supabase project this backend is talking to, and its schema policy. */
@Slf4j
@Component
@Profile({"dev", "staging"})
public class SaasProjectNotice {
private final Environment environment;
private final String projectRef;
private final String ddlAuto;
public SaasProjectNotice(
Environment environment,
@Value("${app.supabase.project-ref:unknown}") String projectRef,
@Value("${spring.jpa.hibernate.ddl-auto:none}") String ddlAuto) {
this.environment = environment;
this.projectRef = projectRef;
this.ddlAuto = ddlAuto;
}
@EventListener(ApplicationReadyEvent.class)
public void announceProject() {
boolean staging = environment.matchesProfiles("staging");
if (staging) {
log.info(
"""
SaaS staging profile: Supabase project {}, ddl-auto={}. This is the SHARED \
long-lived environment, so its data and schema are not yours alone. Testing an \
open SaaS PR? Use that PR's preview branch instead \
(SAAS_DEV_PROJECT_REF in app/.env.saas.local); staging will not have its \
migrations.\
""",
projectRef,
ddlAuto);
return;
}
log.info(
"SaaS dev profile: Supabase preview branch {}, ddl-auto={}. Disposable, so Hibernate"
+ " is allowed to add the inherited tables the migrations do not create.",
projectRef,
ddlAuto);
}
}
@@ -25,6 +25,11 @@ public class TeamLeaderPolicyManagementAuthority implements PolicyManagementAuth
return teamSecurity.isCurrentUserTeamLeader();
}
@Override
public boolean canTriggerPolicies() {
return teamSecurity.isCurrentUserTeamLeader();
}
@Override
public Long currentUserTeamId() {
return teamSecurity.currentUserTeamId();

Some files were not shown because too many files have changed in this diff Show More