From 813e4cf46c075ae4613a88e305f7ffe94224d095 Mon Sep 17 00:00:00 2001 From: Zuri Klaschka Date: Tue, 1 Sep 2026 11:40:06 +0200 Subject: [PATCH] Remove `--output-type pdf` from PDF OCR processing This ran into a seeming bug in `ocrmypdf` where for some PowerPoint PDF exports, the OCR didn't succeed and instead ran into a seemingly infinite output of ```txt Recursion depth exceeded in _find_image_xrefs_page optimize.py:259 ``` This seems to be tracked in https://github.com/ocrmypdf/OCRmyPDF/issues/1321, but for now has to be fixed by a workaround. This previously resulted in a RAM usage overload of the DMS, presumably because it tried to collect and log the (infinite) output logs of `ocrmypdf`. For internal discussions on this, cf. messages from "SysAdmin Int" from August 31st and September 1st 2026. --- server/document-processing/ocrPDF.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/server/document-processing/ocrPDF.ts b/server/document-processing/ocrPDF.ts index 8e55fb5..6743892 100644 --- a/server/document-processing/ocrPDF.ts +++ b/server/document-processing/ocrPDF.ts @@ -17,8 +17,8 @@ export async function ocrPDF( "-l", "deu+eng", "--deskew", - "--output-type", - "pdf", + // "--output-type", + // "pdf", // Runs into infinite https://github.com/ocrmypdf/OCRmyPDF/issues/1321 "--rotate-pages", "--skip-text", // "--invalidate-digital-signatures", // TODO: Add back in once the Debian version supports it