dorfteich/apps/api/src/import-export/pandoc.converter.ts
Claude Fable 5 74a9e495e4
Some checks failed
CI / Lint, typecheck, test (pull_request) Failing after 6m34s
CI / Import/export fidelity gate (pull_request) Has been skipped
CI / Build container images (pull_request) Has been skipped
CI / Auth e2e pack (pull_request) Has been skipped
#209: pandoc reference documents carry the VS-NfD marking for DOCX/ODT
reference-vs-nfd.docx/.odt ship as derived binaries: the pinned pandoc's
default reference documents plus a header and footer with the marking —
part of the document's page setup, so it repeats on every page in Word
and LibreOffice and is not deletable body text. Source of truth is
scripts/gen-classified-reference-docs.mjs (wording from shared
classificationMarking(); maintenance documented in assets/README.md).
The converter passes reference docs to pandoc-server via in-request
files + reference-doc; the worker attaches them for marked docx/odt jobs
(job option {marking}, as in #208). Unclassified exports pass nothing
and are unchanged (pinned by fake-converter test). Fidelity suite
asserts against real pandoc 3.6 that marked outputs carry the
header/footer parts and unmarked ones do not; per-page repetition
verified via LibreOffice 25.8 headless PDF (5/5 pages, 2 markings each,
both formats). Word: quick manual look pending (sample files in the
workspace), procedure documented in assets/README.md.

Co-Authored-By: Claude Fable 5 (1M context) <noreply@anthropic.com>
2026-07-31 07:09:47 +02:00

190 lines
7.3 KiB
TypeScript

import { Injectable } from '@nestjs/common';
import { AppConfig } from '../config/app-config.service';
/** A single pandoc conversion request (ADR 0009). `input` is the raw source
* bytes; the converter base64-encodes it for pandoc when `from` is a binary
* format (docx/odt/…) and passes it as text otherwise. */
export interface ConversionRequest {
from: string;
to: string;
input: Buffer;
standalone?: boolean;
/** Inline referenced/embedded resources (images) as `data:` URIs in the
* output. Used by the import pipeline (#63): pandoc-server is stateless and
* will not hand back a document's media bytes any other way. */
embedResources?: boolean;
/** Line-wrapping of the writer's output. Import uses `none` so a paragraph
* stays on one line (no soft breaks inside image alt text or links). */
wrap?: 'none' | 'auto' | 'preserve';
/** Reference document for the docx/odt writers (issue #209, ADR 0022):
* pandoc copies its page setup — including the header/footer that carry
* the VS-NfD marking — into the output. Sent to pandoc-server as an
* in-request file plus the `reference-doc` option. */
referenceDoc?: { name: string; bytes: Buffer };
}
export interface ConversionResult {
output: Buffer;
mimeType: string;
}
export type ConversionErrorCode =
| 'converter_unavailable'
| 'converter_timeout'
| 'conversion_failed'
// The pond ran out of storage while an import stored the document's media
// (#63) — final, and surfaced the same way through the worker.
| 'quota_exceeded'
// Vault import (#117): rejected archive / unpacked ceiling exceeded.
| 'import_vault_invalid_zip'
| 'import_vault_too_large';
/** A conversion failure with a stable, localizable code. `retryable` marks
* the transient causes (sidecar down / timed out) the worker retries before
* giving up; a genuine `conversion_failed` (bad/unsupported content) is not
* retried. */
export class ConversionError extends Error {
constructor(
readonly code: ConversionErrorCode,
readonly retryable: boolean,
message?: string,
) {
super(message ?? code);
this.name = 'ConversionError';
}
}
/** The job's input bytes. Since #233 the column is nullable — the retention
* job prunes finished jobs' payloads. It never touches PENDING/RUNNING rows
* (incl. stale-lock recovery), so a claimed job without input was re-queued
* by hand; fail it finally instead of crashing the worker. */
export function conversionInputOf(job: { input: Uint8Array | null }): Uint8Array {
if (!job.input) {
throw new ConversionError('conversion_failed', false, 'input payload was pruned');
}
return job.input;
}
/** Server-side conversion limits (ADR 0009). Input is checked before the
* sidecar call; output is capped while reading the response so a runaway
* conversion can't exhaust memory. */
export const MAX_CONVERSION_INPUT_BYTES = 25 * 1024 * 1024;
export const MAX_CONVERSION_OUTPUT_BYTES = 50 * 1024 * 1024;
/** Per ADR 0009 / issue #62: a single conversion may run for at most 60 s. */
export const CONVERSION_TIMEOUT_MS = 60_000;
/** Formats pandoc reads as binary — their bytes are base64-encoded in the
* request `text` field; text formats are sent verbatim. */
const BINARY_INPUT_FORMATS = new Set(['docx', 'odt', 'epub', 'pptx']);
/** Served content type per pandoc output format. */
const OUTPUT_MIME_TYPES: Readonly<Record<string, string>> = {
html: 'text/html',
markdown: 'text/markdown',
gfm: 'text/markdown',
commonmark: 'text/markdown',
plain: 'text/plain',
docx: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
odt: 'application/vnd.oasis.opendocument.text',
};
/**
* Typed client for the pandoc-server sidecar (ADR 0009). Abstract so the job
* worker and its tests depend on the contract, not the HTTP transport — the
* queue-mechanics tests inject a fake, this real implementation is exercised
* against a running container in the integration test.
*/
export abstract class PandocConverter {
abstract convert(request: ConversionRequest): Promise<ConversionResult>;
abstract reachable(): Promise<boolean>;
}
@Injectable()
export class PandocServerConverter extends PandocConverter {
/** Overridable so the timeout path is testable without a 60 s wait. */
protected timeoutMs = CONVERSION_TIMEOUT_MS;
constructor(private readonly config: AppConfig) {
super();
}
private get baseUrl(): string {
return this.config.env.PANDOC_URL;
}
async reachable(): Promise<boolean> {
try {
const response = await fetch(`${this.baseUrl}/version`);
return response.ok;
} catch {
return false;
}
}
async convert(request: ConversionRequest): Promise<ConversionResult> {
if (request.input.byteLength > MAX_CONVERSION_INPUT_BYTES) {
throw new ConversionError('conversion_failed', false, 'input exceeds size limit');
}
const text = BINARY_INPUT_FORMATS.has(request.from)
? request.input.toString('base64')
: request.input.toString('utf8');
const controller = new AbortController();
const timer = setTimeout(() => controller.abort(), this.timeoutMs);
let response: Response;
try {
response = await fetch(`${this.baseUrl}/`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
text,
from: request.from,
to: request.to,
standalone: request.standalone ?? true,
// pandoc-server uses hyphenated option keys; unknown keys are ignored,
// so these are only present when the import pipeline sets them.
...(request.embedResources ? { 'embed-resources': true } : {}),
...(request.wrap ? { wrap: request.wrap } : {}),
...(request.referenceDoc
? {
'reference-doc': request.referenceDoc.name,
files: {
[request.referenceDoc.name]: request.referenceDoc.bytes.toString('base64'),
},
}
: {}),
}),
signal: controller.signal,
});
} catch (error) {
// AbortController.abort() surfaces as an AbortError → the run timed out;
// anything else means the sidecar could not be reached.
if (error instanceof Error && error.name === 'AbortError') {
throw new ConversionError('converter_timeout', true, 'pandoc timed out');
}
throw new ConversionError('converter_unavailable', true, shortMessage(error));
} finally {
clearTimeout(timer);
}
if (!response.ok) {
// Non-200 = pandoc rejected the content (unknown format, malformed
// document). Not retryable — the same input fails the same way.
const detail = (await response.text().catch(() => '')).slice(0, 300);
throw new ConversionError('conversion_failed', false, detail || `pandoc ${response.status}`);
}
const output = Buffer.from(await response.arrayBuffer());
if (output.byteLength > MAX_CONVERSION_OUTPUT_BYTES) {
throw new ConversionError('conversion_failed', false, 'output exceeds size limit');
}
return { output, mimeType: OUTPUT_MIME_TYPES[request.to] ?? 'application/octet-stream' };
}
}
function shortMessage(error: unknown): string {
const message = error instanceof Error ? error.message : String(error);
return message.split('\n')[0]?.slice(0, 200) ?? 'unknown error';
}