From ce0b41d881e8d41ffe97a177308311f05ef7e0b1 Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Mon, 21 Sep 2026 21:55:23 +0100
Subject: [PATCH 01/10] feat(api): offer a semantic backend the layout instead
of imposing it
A semantic export is defined over the authored tree, and some documents
export from it that the fixed-layout pipeline refuses: a list item made
of inline runs without marker geometry exports as an ordinary Word list
item and cannot be laid out at all. Before a backend could ask for the
layout that difference never came up. Letting the request turn such a
document from "exports" into "throws" would take away something that
worked, in exchange for a number the backend wanted as an improvement.
So a layout that cannot be compiled is reported once and the export
continues without it. SemanticExportContext.layoutGraph() is nullable
for exactly this, and a backend that cannot proceed without one says so
through requireLayoutGraph().
Tests: a context whose layout compilation throws still reaches export,
and hands the backend nothing.
---
.../document/api/DocumentRenderingFacade.java | 32 ++++++++++++++++++-
.../SemanticExportLayoutResolutionTest.java | 23 +++++++++++++
2 files changed, 54 insertions(+), 1 deletion(-)
diff --git a/core/src/main/java/com/demcha/compose/document/api/DocumentRenderingFacade.java b/core/src/main/java/com/demcha/compose/document/api/DocumentRenderingFacade.java
index 32d500b0f..1c4f40f43 100644
--- a/core/src/main/java/com/demcha/compose/document/api/DocumentRenderingFacade.java
+++ b/core/src/main/java/com/demcha/compose/document/api/DocumentRenderingFacade.java
@@ -36,6 +36,7 @@
*/
final class DocumentRenderingFacade {
private static final Logger LIFECYCLE_LOG = LoggerFactory.getLogger("com.demcha.compose.document.lifecycle");
+ private static final Logger LOG = LoggerFactory.getLogger(DocumentRenderingFacade.class);
/** Provider format keys for the fixed-layout convenience paths. */
private static final String PDF = "pdf";
@@ -96,7 +97,7 @@ R export(SemanticBackend backend, Path outputFile) throws Exception {
// needed a FontMetricsProvider to be constructed at all.
// The graph, the canvas and the layout all come off the same session state, so a
// backend given both is given a layout compiled from the graph beside it.
- LayoutGraph resolvedLayout = backend.requiresResolvedLayout() ? context.layoutGraph() : null;
+ LayoutGraph resolvedLayout = backend.requiresResolvedLayout() ? compiledLayout(backend) : null;
return backend.export(context.documentGraph(),
new SemanticExportContext(
context.canvas(),
@@ -106,6 +107,35 @@ R export(SemanticBackend backend, Path outputFile) throws Exception {
resolvedLayout));
}
+ /**
+ * Compiles the layout for a backend that asked for it, or hands it nothing.
+ *
+ * A semantic export is defined over the authored tree, and some documents export
+ * from it that the fixed-layout pipeline refuses — a list item made of inline runs
+ * without the marker geometry that resolving one needs, for instance, exports as an
+ * ordinary Word list item and cannot be laid out at all. Before a backend could ask
+ * for the layout that difference never came up; letting the request turn such a
+ * document from "exports" into "throws" would take something away that worked, in
+ * exchange for a number the backend only wanted as an improvement.
+ *
+ * So the layout is offered rather than imposed: what compiles is handed over, what
+ * does not is reported once and the export continues without it.
+ * {@code SemanticExportContext.layoutGraph()} is nullable for exactly this, and a
+ * backend that cannot proceed without it says so through
+ * {@code requireLayoutGraph()}.
+ */
+ private LayoutGraph compiledLayout(SemanticBackend> backend) {
+ try {
+ return context.layoutGraph();
+ } catch (RuntimeException failure) {
+ LOG.warn("Backend '{}' asked for the resolved layout and this document cannot be "
+ + "laid out ({}); exporting without it — measured geometry is unavailable, "
+ + "so the export falls back to what the document itself states",
+ backend.name(), failure.toString());
+ return null;
+ }
+ }
+
byte[] toPdfBytes() throws Exception {
return renderBytes(PDF);
}
diff --git a/core/src/test/java/com/demcha/compose/document/api/SemanticExportLayoutResolutionTest.java b/core/src/test/java/com/demcha/compose/document/api/SemanticExportLayoutResolutionTest.java
index 1bddb3831..7c9ff1cbb 100644
--- a/core/src/test/java/com/demcha/compose/document/api/SemanticExportLayoutResolutionTest.java
+++ b/core/src/test/java/com/demcha/compose/document/api/SemanticExportLayoutResolutionTest.java
@@ -53,12 +53,32 @@ void aBackendThatAsksShouldCauseExactlyOneCompilation() throws Exception {
.isEqualTo(1);
}
+ @Test
+ void aDocumentThatCannotBeLaidOutShouldStillExport() throws Exception {
+ // A semantic export is defined over the authored tree, and some documents export
+ // from it that the fixed-layout pipeline refuses — a list item made of inline runs
+ // without marker geometry is one. Asking for the layout must not turn those from
+ // "exports" into "throws": the request is for an improvement, not a precondition.
+ CountingContext context = new CountingContext();
+ context.failure = new IllegalStateException("this document cannot be laid out");
+ ProbeBackend backend = new ProbeBackend(true);
+
+ LayoutGraph handedOver = new DocumentRenderingFacade(context).export(backend, null);
+
+ assertThat(context.layoutRequests).as("it did try").isEqualTo(1);
+ assertThat(handedOver)
+ .as("and handed the backend nothing rather than failing the export")
+ .isNull();
+ }
+
/** Answers whatever it is asked, and counts how often the layout is wanted. */
private static final class CountingContext implements DocumentRenderingFacade.Context {
private final LayoutCanvas canvas = LayoutCanvas.from(595, 842, Margin.of(36));
private final DocumentGraph graph = new DocumentGraph(List.of());
private int layoutRequests;
+ /** Set to make compiling a layout fail the way an unlayoutable document does. */
+ private RuntimeException failure;
@Override
public void ensureOpen() {
@@ -96,6 +116,9 @@ public List customFontFamilies() {
@Override
public LayoutGraph layoutGraph() {
layoutRequests++;
+ if (failure != null) {
+ throw failure;
+ }
return new LayoutGraph(canvas, 1, List.of(), List.of());
}
From 3815a4268860478e9a8db03843b1911ad693d652 Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Mon, 21 Sep 2026 21:56:15 +0100
Subject: [PATCH 02/10] feat(render-docx): take line height and column widths
from the layout
Two of the things that decide how an exported document looks are
measurements over the font: how tall a line of text is, and how wide a
column came out. This backend has no font runtime, so both were left to
Word, and Word answered with its own -- measured against the reference
render, a body line came out at 13.9pt where the document says 9.7 and
LibreOffice says 12.1, so everything below the first paragraph sat lower
than it should and the gap grew with every line. The export now asks for
the resolved layout and reads the answers the engine already worked out:
line height as w:lineRule="exact", and a table's and a row's columns from
the resolved cells and the placed children. Only the starts of a row's
children are read; a placed child is as wide as its own content, so its
width says nothing about where its column ends.
The space around a block is not a measurement and was missing too. A
paragraph's margin and padding now become w:spacing before and after,
and a container -- which is not a Word object, its children written
where it stood -- hands its top edge to the first paragraph inside and
its bottom edge to the last. Both add to what a paragraph asks for
itself, so nesting sums the way the page does. A container that begins
with a table leaves that edge unwritten rather than parking it on
whatever paragraph comes next.
Measured through Word 16.0, first page: the body now starts at 65.5pt
against the reference's 65.2 and sets lines at 9.8 against 9.7, the
worst grid cell falls from 72% to 52%, and the largest landmark drift
down the page falls from 44pt to 20pt. The editing protocol still passes
7 of 7.
Tests: line height on a body paragraph, a row's cells and a list item,
and none invented without a layout; vertical spacing on a paragraph, a
container's two edges, nesting, and a container of tables; a table and a
row taking every column from the layout, beside the cases that stay
Word's when there is none. Asking for the layout also reads an image a
second time, which is now its own test rather than a surprise.
---
.../semantic/docx/DocxLayoutMetrics.java | 239 ++++++++++++
.../semantic/docx/DocxSemanticBackend.java | 348 +++++++++++++++---
.../backend/semantic/docx/DocxExports.java | 85 +++++
.../docx/DocxImageResolutionTest.java | 54 ++-
.../semantic/docx/DocxLetterSpacingTest.java | 5 +-
.../semantic/docx/DocxLineHeightTest.java | 140 +++++++
.../semantic/docx/DocxRowLayoutTest.java | 50 ++-
.../semantic/docx/DocxTableWidthTest.java | 74 +++-
.../docx/DocxVerticalSpacingTest.java | 145 ++++++++
9 files changed, 1050 insertions(+), 90 deletions(-)
create mode 100644 render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
create mode 100644 render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxExports.java
create mode 100644 render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxLineHeightTest.java
create mode 100644 render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxVerticalSpacingTest.java
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
new file mode 100644
index 000000000..770a44813
--- /dev/null
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
@@ -0,0 +1,239 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.document.layout.DocumentGraph;
+import com.demcha.compose.document.layout.LayoutGraph;
+import com.demcha.compose.document.layout.PlacedFragment;
+import com.demcha.compose.document.layout.PlacedNode;
+import com.demcha.compose.document.layout.payloads.ParagraphFragmentPayload;
+import com.demcha.compose.document.layout.payloads.TableRowFragmentPayload;
+import com.demcha.compose.document.node.DocumentNode;
+import com.demcha.compose.engine.components.content.table.TableResolvedCell;
+
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.IdentityHashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.OptionalDouble;
+import java.util.TreeSet;
+
+/**
+ * The numbers the compiled layout already worked out, addressed by the node that owns them.
+ *
+ * The semantic export writes what the document says and leaves what it does not know to
+ * Word. Two of the things it does not know matter most to how the file looks: how tall a
+ * line of text is, and how wide a table's columns came out. Both are measurements over the
+ * font, which this backend has no runtime for — but the engine made them already, and a
+ * backend that asks for {@code requiresResolvedLayout()} is handed them.
+ *
+ * A {@link LayoutGraph} addresses everything by a path built from each node's name (or
+ * its kind, when it has none) and its index among its siblings. That path is rebuilt here
+ * by walking the same tree the compiler walked, so a node can be handed over and its
+ * geometry looked up — no signature in the writers has to carry a path around, and nothing
+ * depends on the writers visiting nodes in the compiler's order.
+ *
+ * Every accessor answers "I do not know" rather than guessing: an empty index — which is
+ * what an export without a compiled layout gets — makes the whole class inert and the
+ * writers fall back to what they can work out themselves.
+ *
+ * @author Artem Demchyshyn
+ */
+final class DocxLayoutMetrics {
+
+ /** What an export with no compiled layout uses: every question answers "unknown". */
+ static final DocxLayoutMetrics EMPTY =
+ new DocxLayoutMetrics(new IdentityHashMap<>(), Map.of(), Map.of());
+
+ private final Map paths;
+ private final Map> fragments;
+ private final Map placed;
+
+ private DocxLayoutMetrics(Map paths,
+ Map> fragments,
+ Map placed) {
+ this.paths = paths;
+ this.fragments = fragments;
+ this.placed = placed;
+ }
+
+ /**
+ * Indexes a compiled layout against the tree it was compiled from.
+ *
+ * @param graph the document being exported
+ * @param layout the layout compiled from it, or {@code null}
+ * @return an index, or {@link #EMPTY} when there is no layout to index
+ */
+ static DocxLayoutMetrics of(DocumentGraph graph, LayoutGraph layout) {
+ if (graph == null || layout == null) {
+ return EMPTY;
+ }
+ Map paths = new IdentityHashMap<>();
+ for (int index = 0; index < graph.roots().size(); index++) {
+ indexPaths(graph.roots().get(index), null, index, paths);
+ }
+ Map> fragments = new HashMap<>();
+ for (PlacedFragment fragment : layout.fragments()) {
+ fragments.computeIfAbsent(fragment.path(), key -> new ArrayList<>()).add(fragment);
+ }
+ Map placed = new HashMap<>();
+ for (PlacedNode node : layout.nodes()) {
+ placed.putIfAbsent(node.path(), node);
+ }
+ return new DocxLayoutMetrics(paths, fragments, placed);
+ }
+
+ /**
+ * Rebuilds {@code LayoutCompiler.pathFor} over the authored tree.
+ *
+ * The rule is the compiler's: a node is named by its own name when it has one and by
+ * its kind when it does not, suffixed with its index among its siblings, and joined to
+ * its parent's path with a slash. Separators in a name are replaced the same way, so a
+ * name carrying one cannot collide with the structure.
+ */
+ private static void indexPaths(DocumentNode node,
+ String parentPath,
+ int childIndex,
+ Map into) {
+ if (node == null) {
+ return;
+ }
+ String base = node.name() == null || node.name().isBlank() ? node.nodeKind() : node.name().trim();
+ String segment = base.replace('\\', '_').replace('/', '_') + "[" + childIndex + "]";
+ String path = parentPath == null ? segment : parentPath + "/" + segment;
+ into.put(node, path);
+ List children = node.children();
+ for (int index = 0; index < children.size(); index++) {
+ indexPaths(children.get(index), path, index, into);
+ }
+ }
+
+ /** @return true when there is no layout behind this index */
+ boolean isEmpty() {
+ return fragments.isEmpty();
+ }
+
+ /**
+ * The height of one line of the paragraph's text, as the engine measured it.
+ *
+ * Read from the first fragment the node emitted: a paragraph split across a page
+ * boundary emits one fragment per page and every one of them carries the same line
+ * height, which is a property of the text style rather than of the split.
+ *
+ * @param node any node that lays its text out as paragraph lines
+ * @return the line height in points, or empty when the node laid out nothing
+ */
+ OptionalDouble lineHeight(DocumentNode node) {
+ for (PlacedFragment fragment : fragmentsOf(node)) {
+ if (fragment.payload() instanceof ParagraphFragmentPayload paragraph
+ && paragraph.lineHeight() > 0) {
+ return OptionalDouble.of(paragraph.lineHeight());
+ }
+ }
+ return OptionalDouble.empty();
+ }
+
+ /**
+ * The resolved width of every column of a table.
+ *
+ * Derived from the cells rather than from the column specs, because that is where
+ * the answer is: an {@code auto} column's width is its content's, and the resolved
+ * cells carry the width it came out at. The boundaries are collected across every row
+ * so a row whose cells span columns contributes its edges without hiding the ones a
+ * plainer row shows.
+ *
+ * @param node the table being written
+ * @return one width per column in order, or {@code null} when the table laid out nothing
+ */
+ double[] tableColumns(DocumentNode node) {
+ List cells = new ArrayList<>();
+ double rightEdge = 0;
+ for (PlacedFragment fragment : fragmentsOf(node)) {
+ if (fragment.payload() instanceof TableRowFragmentPayload row) {
+ cells.addAll(row.cells());
+ }
+ }
+ if (cells.isEmpty()) {
+ return null;
+ }
+ TreeSet boundaries = new TreeSet<>();
+ for (TableResolvedCell cell : cells) {
+ boundaries.add(round(cell.x()));
+ rightEdge = Math.max(rightEdge, cell.x() + cell.width());
+ }
+ boundaries.add(round(rightEdge));
+ return widthsBetween(new ArrayList<>(boundaries));
+ }
+
+ /**
+ * Where each of a row's children starts, and how wide the row is.
+ *
+ * A row divides its width by weights, by an even split, by explicit columns or by
+ * the flex path, and two of those measure their children. The layout resolved whichever
+ * applied, so this reads the answer instead of reproducing the rule.
+ *
+ * Only the starts are read, because only they are the slots'. A placed
+ * child is as wide as its own content — a paragraph reading "Left" measures a few
+ * points whatever slot it was given — so its width says nothing about where its column
+ * ends. Where the next one begins does, and the gap between them is a number the row
+ * itself states.
+ *
+ * @param row the row being carried as a one-row table
+ * @return {@code {rowWidth, x0, x1, ...}}, the starts relative to the row's left edge,
+ * or {@code null} when the row or one of its children laid out nothing
+ */
+ double[] rowChildStarts(DocumentNode row) {
+ PlacedNode rowNode = placedFor(row);
+ if (rowNode == null || rowNode.placementWidth() <= 0) {
+ return null;
+ }
+ List children = row.children();
+ if (children.isEmpty()) {
+ return null;
+ }
+ double[] starts = new double[1 + children.size()];
+ starts[0] = rowNode.placementWidth();
+ for (int index = 0; index < children.size(); index++) {
+ PlacedNode child = placedFor(children.get(index));
+ if (child == null) {
+ return null;
+ }
+ starts[1 + index] = child.placementX() - rowNode.placementX();
+ }
+ return starts;
+ }
+
+ private List fragmentsOf(DocumentNode node) {
+ String path = paths.get(node);
+ return path == null ? List.of() : fragments.getOrDefault(path, List.of());
+ }
+
+ /** The placed node for a semantic node, or null when this index knows neither. */
+ private PlacedNode placedFor(DocumentNode node) {
+ String path = paths.get(node);
+ return path == null ? null : placed.get(path);
+ }
+
+ /** Consecutive differences of an ascending boundary list. */
+ private static double[] widthsBetween(List boundaries) {
+ if (boundaries.size() < 2) {
+ return null;
+ }
+ double[] widths = new double[boundaries.size() - 1];
+ for (int index = 0; index < widths.length; index++) {
+ widths[index] = boundaries.get(index + 1) - boundaries.get(index);
+ }
+ return widths;
+ }
+
+ /**
+ * Rounds to a hundredth of a point before two edges are compared.
+ *
+ * Column boundaries are arrived at by addition, so the same edge reached along two
+ * rows can differ in the last bits and split one column into two of nearly zero width.
+ * A hundredth of a point is a twentieth of a twip: far below anything Word can write,
+ * and far above the noise.
+ */
+ private static double round(double value) {
+ return Math.round(value * 100.0) / 100.0;
+ }
+}
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
index 012dfbac3..0f3ba98b6 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
@@ -74,6 +74,8 @@
import org.openxmlformats.schemas.wordprocessingml.x2006.main.STStyleType;
import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTFonts;
import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTShd;
+import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTSpacing;
+import org.openxmlformats.schemas.wordprocessingml.x2006.main.STLineSpacingRule;
import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTTblBorders;
import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTTblGrid;
import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTTblLayoutType;
@@ -155,6 +157,16 @@ public final class DocxSemanticBackend implements SemanticBackend {
// two lists reading the same are still two lists.
private final java.util.Map
listNumbering = new java.util.IdentityHashMap<>();
+ // What the engine already measured for the nodes being written: line heights and
+ // resolved column widths. Empty when the export was handed no layout.
+ private DocxLayoutMetrics layout = DocxLayoutMetrics.EMPTY;
+ // The vertical edge of a container whose children have not been written yet, waiting
+ // for the first paragraph inside it — a container is not a Word object, so the space it
+ // holds above itself has to be carried by something that is.
+ private double carriedSpacingBefore;
+ // The last paragraph written into the body, so a container can hand it the space it
+ // holds below itself once its children are done.
+ private XWPFParagraph lastBodyParagraph;
/**
* A container's paint, reduced to what a Word paragraph can carry.
@@ -181,6 +193,26 @@ public String name() {
return "docx-semantic";
}
+ /**
+ * {@inheritDoc}
+ *
+ * This backend asks for the layout, and pays the measurement and pagination pass for
+ * it. Two of the things that decide how the file looks are measurements over the font —
+ * how tall a line of text is, and how wide an {@code auto} column came out — and this
+ * backend has no font runtime of its own. The engine made both already; without them
+ * the export hands the questions to Word, whose answers are its own: measured against
+ * the reference render, Word set each body line at 13.9pt where the document says 9.7,
+ * and sized a table to its text rather than to the width the layout gave it.
+ *
+ * Nothing here depends on the layout being present. An export handed none still
+ * writes a complete document: every place that reads a measured number falls back to
+ * what the document itself states, and states what that costs.
+ */
+ @Override
+ public boolean requiresResolvedLayout() {
+ return true;
+ }
+
@Override
public byte[] export(DocumentGraph graph, SemanticExportContext context) throws Exception {
shapeContainerWarned.set(false);
@@ -190,6 +222,9 @@ public byte[] export(DocumentGraph graph, SemanticExportContext context) throws
containerPaint.clear();
listNumbering.clear();
documentDefaultStyle = dominantTextStyle(graph);
+ layout = DocxLayoutMetrics.of(graph, context.layoutGraph());
+ carriedSpacingBefore = 0;
+ lastBodyParagraph = null;
contentWidth = context.canvas() == null ? Double.MAX_VALUE : context.canvas().innerWidth();
try (XWPFDocument document = new XWPFDocument()) {
applyPageGeometry(document, context.canvas());
@@ -590,17 +625,18 @@ private void writeList(XWPFDocument document,
if (normalized.isBlank()) {
continue;
}
+ java.util.OptionalDouble lineHeight = layout.lineHeight(list);
if (list.marker().isRich()) {
// A drawn marker's pieces are runs, so the row is written the way
// any row with runs in it is; its item is still just a label.
writeRichListLine(document, list.textStyle(), list.marker(),
- com.demcha.compose.document.node.ListItem.of(normalized), 0);
+ com.demcha.compose.document.node.ListItem.of(normalized), 0, lineHeight);
} else if (numId != null) {
// Word draws the marker, so the text is the item and nothing else.
- writeListLine(document, list.textStyle(), normalized, 0, numId);
+ writeListLine(document, list.textStyle(), normalized, 0, numId, lineHeight);
} else {
writeListLine(document, list.textStyle(),
- list.marker().prefix() + normalized, 0, null);
+ list.marker().prefix() + normalized, 0, null, lineHeight);
}
}
for (com.demcha.compose.document.node.ListItem item : list.nestedItems()) {
@@ -621,12 +657,14 @@ private void writeNestedItem(XWPFDocument document,
item.marker() != null
? item.marker()
: com.demcha.compose.document.node.ListMarker.defaultForDepth(depth);
+ java.util.OptionalDouble lineHeight = layout.lineHeight(list);
if (item.isRich() || marker.isRich()) {
- writeRichListLine(document, list.textStyle(), marker, item, depth);
+ writeRichListLine(document, list.textStyle(), marker, item, depth, lineHeight);
} else if (numId != null) {
- writeListLine(document, list.textStyle(), item.label(), depth, numId);
+ writeListLine(document, list.textStyle(), item.label(), depth, numId, lineHeight);
} else {
- writeListLine(document, list.textStyle(), marker.prefix() + item.label(), depth, null);
+ writeListLine(document, list.textStyle(), marker.prefix() + item.label(), depth, null,
+ lineHeight);
}
for (com.demcha.compose.document.node.ListItem child : item.children()) {
writeNestedItem(document, list, child, depth + 1, numId);
@@ -641,8 +679,10 @@ private void writeNestedItem(XWPFDocument document,
* and the nesting indent as characters
*/
private void writeListLine(XWPFDocument document, DocumentTextStyle style,
- String text, int depth, BigInteger numId) {
+ String text, int depth, BigInteger numId,
+ java.util.OptionalDouble lineHeight) {
XWPFParagraph para = newBodyParagraph(document);
+ applyLineHeight(para, lineHeight);
if (numId != null) {
para.setNumID(numId);
para.setNumILvl(BigInteger.valueOf(depth));
@@ -674,10 +714,12 @@ private void writeListLine(XWPFDocument document, DocumentTextStyle style,
private void writeRichListLine(XWPFDocument document, DocumentTextStyle style,
com.demcha.compose.document.node.ListMarker marker,
com.demcha.compose.document.node.ListItem item,
- int depth) {
+ int depth,
+ java.util.OptionalDouble lineHeight) {
warnDroppedInlineRuns(marker.runs());
warnDroppedInlineRuns(item.runs());
XWPFParagraph para = newBodyParagraph(document);
+ applyLineHeight(para, lineHeight);
XWPFRun leading = para.createRun();
applyStyle(leading, style);
leading.setText(" ".repeat(depth) + (marker.isRich() ? "" : marker.prefix()));
@@ -768,22 +810,50 @@ private void writeChartFallback(XWPFDocument document, ChartNode node) throws Ex
private void writeContainerChildren(XWPFDocument document, DocumentNode node) throws Exception {
ContainerPaint paint = paintOf(node);
if (paint.isEmpty()) {
- for (DocumentNode child : node.children()) {
- writeNode(document, child);
- }
+ writeContainerBody(document, node);
return;
}
warnContainerRadiusDropped(node);
containerPaint.push(paint);
try {
- for (DocumentNode child : node.children()) {
- writeNode(document, child);
- }
+ writeContainerBody(document, node);
} finally {
containerPaint.pop();
}
}
+ /**
+ * Writes a container's children, carrying the container's own vertical box with them.
+ *
+ * A container is not a Word object — its children are written where it stood — so
+ * the space it holds above and below itself had nowhere to go and was dropped. Word has
+ * that space on a paragraph and only on a paragraph, so the top goes to the first
+ * paragraph written inside and the bottom to the last, which is where a reader sees it
+ * either way.
+ *
+ * Both are added to whatever that paragraph asks for itself, and both survive
+ * nesting: a card inside a section hands its top to the same first paragraph, which
+ * ends up carrying the sum — the same sum the page shows.
+ *
+ * A container that begins or ends with a table keeps that edge unwritten. Word has
+ * no space-before on a table, and the alternatives — an empty paragraph, a floating
+ * table's {@code w:tblpPr} — either add a line the document never asked for or move the
+ * table out of the flow it is in.
+ */
+ private void writeContainerBody(XWPFDocument document, DocumentNode node) throws Exception {
+ carriedSpacingBefore += node.margin().top() + node.padding().top();
+ for (DocumentNode child : node.children()) {
+ writeNode(document, child);
+ }
+ double after = node.margin().bottom() + node.padding().bottom();
+ if (after > 0 && lastBodyParagraph != null) {
+ addSpacing(lastBodyParagraph, 0, after);
+ }
+ // Nothing inside took the top edge — a container of tables, or an empty one — so it
+ // is not left waiting to land on whatever paragraph comes next.
+ carriedSpacingBefore = 0;
+ }
+
private static boolean hasRadius(com.demcha.compose.document.style.DocumentCornerRadius radius) {
return radius != null && !radius.isZero();
}
@@ -975,9 +1045,44 @@ private XWPFParagraph newBodyParagraph(XWPFDocument document) {
if (paint != null) {
applyContainerPaint(para, paint);
}
+ if (carriedSpacingBefore > 0) {
+ addSpacing(para, carriedSpacingBefore, 0);
+ carriedSpacingBefore = 0;
+ }
+ lastBodyParagraph = para;
return para;
}
+ /**
+ * Adds to the space above and below a paragraph, rather than replacing it.
+ *
+ * Two things state it — the paragraph's own box, and the vertical edge of every
+ * container it sits at the start or the end of — and on the page a reader sees their
+ * sum. Setting it would mean whichever ran last silently won.
+ *
+ * @param para the Word paragraph
+ * @param before points to add above, zero to leave it alone
+ * @param after points to add below, zero to leave it alone
+ */
+ private static void addSpacing(XWPFParagraph para, double before, double after) {
+ CTPPr properties = para.getCTP().isSetPPr()
+ ? para.getCTP().getPPr() : para.getCTP().addNewPPr();
+ CTSpacing spacing = properties.isSetSpacing() ? properties.getSpacing() : properties.addNewSpacing();
+ if (before > 0) {
+ spacing.setBefore(BigInteger.valueOf(
+ twipsOf(spacing.isSetBefore() ? spacing.getBefore() : null) + toTwips(before)));
+ }
+ if (after > 0) {
+ spacing.setAfter(BigInteger.valueOf(
+ twipsOf(spacing.isSetAfter() ? spacing.getAfter() : null) + toTwips(after)));
+ }
+ }
+
+ /** Reads a twip measure back, treating an unset one as zero. */
+ private static long twipsOf(Object measure) {
+ return measure == null ? 0 : Long.parseLong(String.valueOf(measure));
+ }
+
private static void applyContainerPaint(XWPFParagraph para, ContainerPaint paint) {
CTPPr properties = para.getCTP().isSetPPr()
? para.getCTP().getPPr()
@@ -1045,13 +1150,43 @@ private void writeParagraph(XWPFDocument document, ParagraphNode node) {
writeParagraphRuns(para, node, rightToLeft);
}
+ /**
+ * Sets the paragraph's lines to the height the engine measured them at.
+ *
+ * A line's height is the font's, and Word uses its own — measured against the
+ * reference render, Word set a body line at 13.9pt and LibreOffice at 12.1 where the
+ * document says 9.7. Over a page that difference is the largest single reason an
+ * exported document stops matching: everything below the first paragraph sits lower
+ * than it should, and the gap grows with every line.
+ *
+ * Written as {@code w:lineRule="exact"} rather than as a multiple: the number is a
+ * measurement in points, and a multiple would be measured again by Word against
+ * whichever font it substituted. Exact is also the only rule that can make a line
+ * shorter than the font would like — which is the direction this always moves, since
+ * the engine's line box is the face's ascent plus descent with no leading.
+ *
+ * @param para the Word paragraph
+ * @param height the measured line height in points, empty when nothing measured it
+ */
+ private static void applyLineHeight(XWPFParagraph para, java.util.OptionalDouble height) {
+ if (height.isEmpty()) {
+ return;
+ }
+ CTPPr properties = para.getCTP().isSetPPr() ? para.getCTP().getPPr() : para.getCTP().addNewPPr();
+ CTSpacing spacing = properties.isSetSpacing() ? properties.getSpacing() : properties.addNewSpacing();
+ spacing.setLineRule(STLineSpacingRule.EXACT);
+ spacing.setLine(BigInteger.valueOf(Math.round(height.getAsDouble() * POINT_TO_TWIP)));
+ }
+
/**
* Writes the properties a paragraph carries whatever it sits in.
*
* One place on purpose. These were written where a paragraph is a document child
* and not where it is a table cell's, which is how every right-to-left invoice line
* came out undeclared; splitting them again would set the next property up for the
- * same fate.
+ * same fate. The measured line height is the next property, and it went the same way
+ * once before landing here: written beside the call rather than inside it, it reached
+ * the body and not the two columns of a row, whose paragraphs are a cell's.
*
* {@link TextDirection#AUTO} is resolved here rather than passed on. Leaving it
* unwritten was leaving Word to guess, and Word guessing is the thing {@code w:bidi}
@@ -1065,13 +1200,40 @@ private void writeParagraph(XWPFDocument document, ParagraphNode node) {
* @param source the node it was written from
* @return whether the paragraph runs right to left, for its runs to declare too
*/
- private static boolean applyParagraphProperties(XWPFParagraph target, ParagraphNode source) {
+ private boolean applyParagraphProperties(XWPFParagraph target, ParagraphNode source) {
boolean rightToLeft = ParagraphDirection.resolve(source) == TextDirection.RTL;
target.setAlignment(toAlignment(source.align(), rightToLeft));
applyDirection(target, rightToLeft);
+ applyLineHeight(target, layout.lineHeight(source));
+ applyVerticalSpacing(target, source);
return rightToLeft;
}
+ /**
+ * Carries the space a paragraph holds above and below itself.
+ *
+ *
A paragraph's own {@code margin} and {@code padding} are what separate one block
+ * from the next, and none of it was written: every exported document ran its blocks
+ * together and leaned on whatever Word puts between paragraphs instead. That was
+ * invisible while the line height was Word's too — the lines were tall enough to stand
+ * in for the gaps — and became the largest remaining difference the moment the lines
+ * were right.
+ *
+ * Vertical space is one of the few pieces of a node's box Word holds natively, which
+ * is why this is written and the horizontal half is not: {@code w:spacing} is the gap
+ * above and below a paragraph, while the left and right insets of a shaded block have
+ * no paragraph-level equivalent at all.
+ *
+ * Margin and padding are added together. They are different things to the engine —
+ * one outside the box, one inside it — but Word has one gap, and a reader looking at
+ * the page sees their sum.
+ */
+ private static void applyVerticalSpacing(XWPFParagraph target, DocumentNode source) {
+ addSpacing(target,
+ source.margin().top() + source.padding().top(),
+ source.margin().bottom() + source.padding().bottom());
+ }
+
/**
* Marks a right-to-left paragraph so Word lays it out in that direction.
*
@@ -1573,44 +1735,137 @@ private void writeRowCellChild(XWPFTableCell cell, DocumentNode child) throws Ex
* it was given.
*/
private void applyRowGeometry(XWPFTable table, RowNode node) {
+ double[] starts = layout.rowChildStarts(node);
+ if (starts != null) {
+ // The layout placed each child, so every way a row can divide — the two that
+ // measure their children included — is already answered.
+ setTableWidth(table, starts[0]);
+ writeRowColumns(table, placedColumns(node, starts));
+ return;
+ }
+
if (!Double.isFinite(contentWidth) || contentWidth <= 0) {
return;
}
setTableWidth(table, contentWidth);
-
double[] slots = resolveRowSlots(node, contentWidth);
if (slots == null) {
return;
}
+ writeRowColumns(table, statedColumns(node, slots));
+ }
+
+ /**
+ * One column of a row's carrier: the text box, and what sits either side of it inside
+ * the same cell.
+ *
+ * @param width the grid column's full width in points
+ * @param leading space before the text box, written as the cell's left margin
+ * @param trailing space after it, written as the cell's right margin
+ */
+ private record CellColumn(double width, double leading, double trailing) {
+ }
+ /**
+ * Columns from where the layout started each child.
+ *
+ * A slot runs from its child's start to the next child's, less the gap the row puts
+ * between them; the last runs to the row's edge, less its padding. That is the same
+ * shape {@link #statedColumns} builds — the difference is only where the starts come
+ * from, and these were measured rather than worked out.
+ */
+ private static List placedColumns(RowNode node, double[] starts) {
+ int count = starts.length - 1;
+ double rowWidth = starts[0];
+ List columns = new ArrayList<>(count);
+ for (int index = 0; index < count; index++) {
+ double x = starts[1 + index];
+ double columnStart = index == 0 ? 0.0 : x;
+ double columnEnd = index == count - 1 ? rowWidth : starts[2 + index];
+ double trailing = index == count - 1 ? node.padding().right() : node.gap();
+ columns.add(new CellColumn(columnEnd - columnStart, x - columnStart,
+ Math.max(0.0, Math.min(trailing, columnEnd - x))));
+ }
+ return columns;
+ }
+
+ /** Columns from the row's own arithmetic: the slot, plus the gap and padding beside it. */
+ private static List statedColumns(RowNode node, double[] slots) {
+ List columns = new ArrayList<>(slots.length);
+ for (int index = 0; index < slots.length; index++) {
+ double leading = index == 0 ? node.padding().left() : 0.0;
+ double trailing = index == slots.length - 1 ? node.padding().right() : node.gap();
+ columns.add(new CellColumn(slots[index] + leading + trailing, leading, trailing));
+ }
+ return columns;
+ }
+
+ /**
+ * Writes a row carrier's grid, and the cell margins that hold its text box in place.
+ *
+ * Word has no inter-column gap and no row padding, so both ride in the neighbouring
+ * column's width and are taken back out as that cell's margin. The margins are written
+ * even when they are zero: Word's own default is not.
+ */
+ private static void writeRowColumns(XWPFTable table, List columns) {
CTTblGrid grid = table.getCTTbl().getTblGrid() != null
? table.getCTTbl().getTblGrid()
: table.getCTTbl().addNewTblGrid();
while (grid.sizeOfGridColArray() > 0) {
grid.removeGridCol(0);
}
- for (int index = 0; index < slots.length; index++) {
- double leading = index == 0 ? node.padding().left() : 0.0;
- double trailing = index == slots.length - 1 ? node.padding().right() : node.gap();
- double column = slots[index] + leading + trailing;
- grid.addNewGridCol().setW(BigInteger.valueOf(Math.round(column * POINT_TO_TWIP)));
+ for (int index = 0; index < columns.size(); index++) {
+ CellColumn column = columns.get(index);
+ grid.addNewGridCol().setW(BigInteger.valueOf(Math.round(column.width() * POINT_TO_TWIP)));
CTTcPr properties = cellProperties(table.getRow(0).getCell(index));
CTTblWidth cellWidth = properties.isSetTcW() ? properties.getTcW() : properties.addNewTcW();
cellWidth.setType(STTblWidth.DXA);
- cellWidth.setW(BigInteger.valueOf(Math.round(column * POINT_TO_TWIP)));
+ cellWidth.setW(BigInteger.valueOf(Math.round(column.width() * POINT_TO_TWIP)));
CTTcMar margins = properties.isSetTcMar() ? properties.getTcMar() : properties.addNewTcMar();
- setCellMargin(margins.isSetLeft() ? margins.getLeft() : margins.addNewLeft(), leading);
- setCellMargin(margins.isSetRight() ? margins.getRight() : margins.addNewRight(), trailing);
+ setCellMargin(margins.isSetLeft() ? margins.getLeft() : margins.addNewLeft(),
+ column.leading());
+ setCellMargin(margins.isSetRight() ? margins.getRight() : margins.addNewRight(),
+ column.trailing());
+ }
+ setFixedLayout(table);
+ }
+
+ /** Replaces a table's grid with the given column widths. */
+ private static void writeGrid(XWPFTable table, double[] widths) {
+ CTTblGrid grid = table.getCTTbl().getTblGrid() != null
+ ? table.getCTTbl().getTblGrid()
+ : table.getCTTbl().addNewTblGrid();
+ while (grid.sizeOfGridColArray() > 0) {
+ grid.removeGridCol(0);
}
+ for (double width : widths) {
+ grid.addNewGridCol().setW(BigInteger.valueOf(Math.round(width * POINT_TO_TWIP)));
+ }
+ }
- // Without this Word treats the grid as a starting suggestion and re-fits the
- // columns to their content, which is the behaviour being replaced.
- CTTblPr properties = table.getCTTbl().getTblPr();
- CTTblLayoutType layout = properties.isSetTblLayout()
+ /**
+ * Makes the grid the answer rather than a suggestion.
+ *
+ * Left to its default Word re-fits a table's columns to their content, which is the
+ * behaviour every written grid exists to replace.
+ */
+ private static void setFixedLayout(XWPFTable table) {
+ CTTblPr properties = table.getCTTbl().getTblPr() != null
+ ? table.getCTTbl().getTblPr()
+ : table.getCTTbl().addNewTblPr();
+ CTTblLayoutType type = properties.isSetTblLayout()
? properties.getTblLayout()
: properties.addNewTblLayout();
- layout.setType(STTblLayoutType.FIXED);
+ type.setType(STTblLayoutType.FIXED);
+ }
+
+ private static double sum(double[] values) {
+ double total = 0;
+ for (double value : values) {
+ total += value;
+ }
+ return total;
}
/**
@@ -1723,6 +1978,16 @@ private static void setCellMargin(CTTblWidth margin, double points) {
* widths.
*/
private void applyTableWidth(XWPFTable table, TableNode node, int columnCount) {
+ double[] measured = layout.tableColumns(node);
+ if (measured != null && measured.length > 0) {
+ // The layout resolved every column, an auto one included, so there is nothing
+ // left to decide: write the widths it arrived at and stop Word re-fitting them.
+ writeGrid(table, measured);
+ setTableWidth(table, sum(measured));
+ setFixedLayout(table);
+ return;
+ }
+
Double authored = node.width() != null && node.width() > 0 ? node.width() : null;
List fixedColumns = fixedColumnWidths(node, columnCount);
@@ -1739,22 +2004,15 @@ private void applyTableWidth(XWPFTable table, TableNode node, int columnCount) {
double width = authored != null ? Math.max(authored, natural) : natural;
setTableWidth(table, width);
- CTTblGrid grid = table.getCTTbl().getTblGrid() != null
- ? table.getCTTbl().getTblGrid()
- : table.getCTTbl().addNewTblGrid();
- while (grid.sizeOfGridColArray() > 0) {
- grid.removeGridCol(0);
- }
- for (int index = 0; index < fixedColumns.size(); index++) {
- // With no auto column to absorb it, the engine hands a stated width's surplus
- // to the last column. Splitting it evenly instead would put every column edge
- // but the first in a different place than the PDF draws it.
- double column = fixedColumns.get(index);
- if (index == fixedColumns.size() - 1) {
- column += width - natural;
- }
- grid.addNewGridCol().setW(BigInteger.valueOf(Math.round(column * POINT_TO_TWIP)));
+ double[] columns = new double[fixedColumns.size()];
+ for (int index = 0; index < columns.length; index++) {
+ columns[index] = fixedColumns.get(index);
}
+ // With no auto column to absorb it, the engine hands a stated width's surplus to
+ // the last column. Splitting it evenly instead would put every column edge but the
+ // first in a different place than the PDF draws it.
+ columns[columns.length - 1] += width - natural;
+ writeGrid(table, columns);
}
/**
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxExports.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxExports.java
new file mode 100644
index 000000000..a41134e8d
--- /dev/null
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxExports.java
@@ -0,0 +1,85 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.GraphCompose;
+import com.demcha.compose.document.api.DocumentSession;
+import com.demcha.compose.document.backend.semantic.SemanticBackend;
+import com.demcha.compose.document.backend.semantic.SemanticExportContext;
+import com.demcha.compose.document.dsl.PageFlowBuilder;
+import com.demcha.compose.document.layout.DocumentGraph;
+import com.demcha.compose.document.layout.LayoutCanvas;
+import com.demcha.compose.document.style.DocumentInsets;
+import org.apache.poi.xwpf.usermodel.XWPFDocument;
+
+import java.io.ByteArrayInputStream;
+import java.util.List;
+import java.util.function.Consumer;
+
+/**
+ * Exports a small document both ways: with the compiled layout behind it, and without.
+ *
+ * The backend asks for the resolved layout and uses it where it has one — measured line
+ * heights, resolved column widths — so a session export and a bare export are two different
+ * outputs of the same writer. Both are real: a session is how the export is reached, and a
+ * caller holding a graph and a canvas can call {@code export} directly, which is also what
+ * happens when a document cannot be laid out at all. Tests that pin the fallback have to
+ * ask for it.
+ *
+ * @author Artem Demchyshyn
+ */
+final class DocxExports {
+
+ private DocxExports() {
+ }
+
+ /** Exports through a session, so the backend is handed the compiled layout. */
+ static XWPFDocument withLayout(double pageWidth, double pageHeight, double margin,
+ Consumer content) throws Exception {
+ byte[] docx;
+ try (DocumentSession session = session(pageWidth, pageHeight, margin, content)) {
+ docx = session.export(new DocxSemanticBackend());
+ }
+ return new XWPFDocument(new ByteArrayInputStream(docx));
+ }
+
+ /** Exports the same document with no layout behind it, the way a bare caller does. */
+ static XWPFDocument withoutLayout(double pageWidth, double pageHeight, double margin,
+ Consumer content) throws Exception {
+ Captured captured = new Captured();
+ byte[] docx;
+ try (DocumentSession session = session(pageWidth, pageHeight, margin, content)) {
+ session.export(captured);
+ docx = new DocxSemanticBackend().export(captured.graph,
+ new SemanticExportContext(captured.canvas, List.of(), null, null));
+ }
+ return new XWPFDocument(new ByteArrayInputStream(docx));
+ }
+
+ private static DocumentSession session(double pageWidth, double pageHeight, double margin,
+ Consumer content) {
+ DocumentSession session = GraphCompose.document()
+ .pageSize(pageWidth, pageHeight)
+ .margin(DocumentInsets.of(margin))
+ .create();
+ session.pageFlow(content::accept);
+ return session;
+ }
+
+ /** Takes the graph and the canvas the session would hand any backend. */
+ private static final class Captured implements SemanticBackend {
+
+ private DocumentGraph graph;
+ private LayoutCanvas canvas;
+
+ @Override
+ public String name() {
+ return "capture";
+ }
+
+ @Override
+ public byte[] export(DocumentGraph documentGraph, SemanticExportContext context) {
+ this.graph = documentGraph;
+ this.canvas = context.canvas();
+ return new byte[0];
+ }
+ }
+}
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxImageResolutionTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxImageResolutionTest.java
index bb7a41144..e8313468a 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxImageResolutionTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxImageResolutionTest.java
@@ -43,37 +43,57 @@
class DocxImageResolutionTest {
@Test
- void anImageIsResolvedOncePerExport(@TempDir Path dir) throws Exception {
+ void theWriterResolvesAnImageOnce(@TempDir Path dir) throws Exception {
Path png = dir.resolve("probe.png");
ImageIO.write(new BufferedImage(40, 20, BufferedImage.TYPE_INT_RGB), "png", png.toFile());
+ assertThat(resolutionsOf(png, false))
+ .describedAs("one image, one resolution — the sizing pass takes the data the "
+ + "writer already holds instead of resolving it again")
+ .hasSize(1);
+ }
+
+ @Test
+ void askingForTheLayoutResolvesItAgain(@TempDir Path dir) throws Exception {
+ // Worth stating rather than discovering: the export asks for the compiled layout,
+ // and compiling one measures every image for itself. The writer's own share is
+ // still the single resolution above — this is what the request costs on top, and
+ // it is the same work a PDF render of the document does.
+ Path png = dir.resolve("probe.png");
+ ImageIO.write(new BufferedImage(40, 20, BufferedImage.TYPE_INT_RGB), "png", png.toFile());
+
+ assertThat(resolutionsOf(png, true))
+ .describedAs("the writer's one, plus the layout's own")
+ .hasSizeGreaterThan(1);
+ }
+
+ /**
+ * How many times exporting that image reads it.
+ *
+ * @param png the image the document carries
+ * @param withSession true to export through a session, which compiles a layout first
+ */
+ private static List resolutionsOf(Path png, boolean withSession) throws Exception {
Logger logger = (Logger) LoggerFactory.getLogger(
"com.demcha.compose.engine.components.content.ImageData");
ListAppender appender = new ListAppender<>();
appender.start();
logger.addAppender(appender);
try {
- try (DocumentSession document = GraphCompose.document()
- .pageSize(400, 200)
- .margin(DocumentInsets.of(20))
- .create()) {
- document.pageFlow(page -> page.addImage(image -> image
- .source(DocumentImageData.fromPath(png))
- .width(100)));
- document.export(new DocxSemanticBackend());
+ java.util.function.Consumer content =
+ page -> page.addImage(image -> image
+ .source(DocumentImageData.fromPath(png))
+ .width(100));
+ if (withSession) {
+ DocxExports.withLayout(400, 200, 20, content).close();
+ } else {
+ DocxExports.withoutLayout(400, 200, 20, content).close();
}
-
- List arrivals = appender.list.stream()
+ return appender.list.stream()
.map(ILoggingEvent::getFormattedMessage)
.filter(message -> message.contains("Create an image from path")
&& message.contains(png.getFileName().toString()))
.toList();
-
- assertThat(arrivals)
- .describedAs("one image, one resolution — the sizing pass takes the data "
- + "the writer already holds instead of resolving it again; "
- + "messages were %s", arrivals)
- .hasSize(1);
} finally {
logger.detachAppender(appender);
appender.stop();
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxLetterSpacingTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxLetterSpacingTest.java
index 7fae512c9..4b4047e66 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxLetterSpacingTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxLetterSpacingTest.java
@@ -83,7 +83,10 @@ void noTrackingWritesNoElementAtAll() throws Exception {
byte[] docx = render(NAME, style(DocumentLetterSpacing.NONE, 20));
assertThat(spacingOf(docx)).isEmpty();
- assertThat(xmlOf(docx)).doesNotContain("Line height is a property of the face, and Word uses its own: measured through Word
+ * 16.0 against the reference render, a body line came out at 13.9pt where the document
+ * says 9.7, and LibreOffice at 12.1. Nothing about that is visible in one paragraph — it
+ * is visible by the bottom of the page, where everything sits lower than it should and the
+ * gap has grown with every line since the first.
+ *
+ * The number is written as {@code w:lineRule="exact"} rather than as a multiple. It is a
+ * measurement in points; a multiple would be measured again by the editor, against whatever
+ * font it substituted, which is the thing being replaced.
+ *
+ * @author Artem Demchyshyn
+ */
+class DocxLineHeightTest {
+
+ private static final double PAGE_WIDTH = 400;
+ private static final double MARGIN = 20;
+ private static final double TWIPS_PER_POINT = 20.0;
+
+ @Test
+ void aBodyParagraphCarriesTheMeasuredHeightExactly() throws Exception {
+ try (XWPFDocument document = withLayout(page -> page.addParagraph(p -> p
+ .text("A sentence long enough to wrap onto a second line in this column.")
+ .textStyle(DocumentTextStyle.DEFAULT.withSize(10.5))))) {
+
+ long line = lineTwips(document.getParagraphs().get(0));
+ assertThat(lineRule(document.getParagraphs().get(0))).isEqualTo("exact");
+ // 10.5pt of Helvetica measures 9.71pt from ascent to descent, so the line is
+ // shorter than the type size — which is the direction this always moves, and
+ // the reason a multiple could not express it.
+ assertThat(line).isBetween(190L, 196L);
+ assertThat(line).isLessThan(Math.round(10.5 * TWIPS_PER_POINT));
+ }
+ }
+
+ @Test
+ void aParagraphInsideARowCarriesItToo() throws Exception {
+ // A row's columns are table cells, and a cell's paragraph is written down its own
+ // path. The direction mark reached the body and not the cells once already; this
+ // pins that the line height did not repeat it.
+ try (XWPFDocument document = withLayout(page -> page.addRow(r -> r
+ .addParagraph(p -> p.text("Left column"))
+ .addParagraph(p -> p.text("Right column"))))) {
+
+ List cells = document.getTables().get(0).getRow(0).getTableCells()
+ .stream()
+ .flatMap(cell -> cell.getParagraphs().stream())
+ .filter(paragraph -> !paragraph.getText().isBlank())
+ .toList();
+
+ assertThat(cells).hasSize(2);
+ for (XWPFParagraph paragraph : cells) {
+ assertThat(lineRule(paragraph))
+ .as("the cell paragraph '%s'", paragraph.getText())
+ .isEqualTo("exact");
+ assertThat(lineTwips(paragraph)).isGreaterThan(0);
+ }
+ }
+ }
+
+ @Test
+ void aListItemCarriesItAsWell() throws Exception {
+ try (XWPFDocument document = withLayout(page -> page.addList(l -> l
+ .bullet()
+ .items("First item", "Second item")))) {
+
+ List items = document.getParagraphs().stream()
+ .filter(paragraph -> !paragraph.getText().isBlank())
+ .toList();
+
+ assertThat(items).hasSize(2);
+ items.forEach(item -> assertThat(lineRule(item)).isEqualTo("exact"));
+ }
+ }
+
+ @Test
+ void differentTextStylesGetDifferentHeights() throws Exception {
+ try (XWPFDocument document = withLayout(page -> page
+ .addParagraph(p -> p.text("Heading")
+ .textStyle(DocumentTextStyle.DEFAULT.withSize(21)))
+ .addParagraph(p -> p.text("Body")
+ .textStyle(DocumentTextStyle.DEFAULT.withSize(10.5))))) {
+
+ long heading = lineTwips(document.getParagraphs().get(0));
+ long body = lineTwips(document.getParagraphs().get(1));
+ assertThat(heading)
+ .as("twice the type size is about twice the line")
+ .isBetween(body * 2 - 20, body * 2 + 20);
+ }
+ }
+
+ @Test
+ void withNothingMeasuredNoHeightIsInvented() throws Exception {
+ // The export still writes a complete document with no layout behind it — it just
+ // has no measurement to state, and Word's own line height applies as before.
+ try (XWPFDocument document = DocxExports.withoutLayout(PAGE_WIDTH, 600, MARGIN,
+ page -> page.addParagraph(p -> p.text("Body")))) {
+
+ // Narrowly: no line height. The same w:spacing element also carries the space
+ // above and below a paragraph, which is the document's own number and is
+ // written either way — looking for the element could not tell them apart.
+ CTPPr properties = document.getParagraphs().get(0).getCTP().getPPr();
+ assertThat(properties == null || !properties.isSetSpacing()
+ || !properties.getSpacing().isSetLine())
+ .as("no w:line, rather than a guessed one")
+ .isTrue();
+ }
+ }
+
+ private static XWPFDocument withLayout(Consumer content) throws Exception {
+ return DocxExports.withLayout(PAGE_WIDTH, 600, MARGIN, content);
+ }
+
+ private static long lineTwips(XWPFParagraph paragraph) {
+ return Long.parseLong(String.valueOf(paragraph.getCTP().getPPr().getSpacing().getLine()));
+ }
+
+ private static String lineRule(XWPFParagraph paragraph) {
+ CTPPr properties = paragraph.getCTP().getPPr();
+ return properties == null || !properties.isSetSpacing()
+ ? null
+ : properties.getSpacing().getLineRule().toString();
+ }
+}
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxRowLayoutTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxRowLayoutTest.java
index 0994f47a7..0b7da5ec3 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxRowLayoutTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxRowLayoutTest.java
@@ -149,6 +149,39 @@ void anArrangementThatJustifiesTheChildrenLeavesTheSplitToWord() throws Exceptio
assertThat(layoutType(table)).isNull();
}
+ @Test
+ void theMeasuredSplitAgreesWithTheStatedOneWhereBothCanAnswer() throws Exception {
+ // Weights are arithmetic either way, so the two paths have to arrive at the same
+ // columns — if they ever part, one of them is reading the row wrong.
+ Consumer weighted = page -> page.addRow(r -> r
+ .gap(20)
+ .weights(3, 2)
+ .addParagraph(p -> p.text("Scope"))
+ .addParagraph(p -> p.text("Period")));
+
+ assertThat(gridTwips(measuredTable(weighted)))
+ .isEqualTo(gridTwips(onlyTable(weighted)))
+ .containsExactly(4480L, 2720L);
+ }
+
+ @Test
+ void theLayoutAnswersTheSplitTheRowCannotStateItself() throws Exception {
+ // The case the stated path declines: an auto column is as wide as its content, and
+ // the layout is where that width was worked out.
+ Consumer mixed = page -> page.addRow(r -> r
+ .columns(DocumentRowColumn.fixed(100), DocumentRowColumn.auto())
+ .addParagraph(p -> p.text("Label"))
+ .addParagraph(p -> p.text("Value")));
+
+ assertThat(gridTwips(onlyTable(mixed))).as("nothing to write").isEmpty();
+ List measured = gridTwips(measuredTable(mixed));
+ assertThat(measured).hasSize(2);
+ assertThat(measured.get(0)).as("the fixed column, exactly").isEqualTo(2000L);
+ assertThat(sum(measured))
+ .as("and the pair still tiles the row")
+ .isEqualTo(Math.round(CONTENT_WIDTH * TWIPS_PER_POINT));
+ }
+
/** Grid column widths in twips, empty when the export wrote no grid. */
private static List gridTwips(XWPFTable table) {
CTTblGrid grid = table.getCTTbl().getTblGrid();
@@ -188,16 +221,17 @@ private static int occurrences(String haystack, String needle) {
return count;
}
+ /** The carrier as written with nothing measured behind it. */
private static XWPFTable onlyTable(Consumer content) throws Exception {
- byte[] docx;
- try (DocumentSession session = GraphCompose.document()
- .pageSize(400, 600)
- .margin(DocumentInsets.of(20))
- .create()) {
- session.pageFlow(content::accept);
- docx = session.export(new DocxSemanticBackend());
+ try (XWPFDocument document = DocxExports.withoutLayout(400, 600, 20, content)) {
+ assertThat(document.getTables()).hasSize(1);
+ return document.getTables().get(0);
}
- try (XWPFDocument document = new XWPFDocument(new ByteArrayInputStream(docx))) {
+ }
+
+ /** The carrier as written when the compiled layout is there to read. */
+ private static XWPFTable measuredTable(Consumer content) throws Exception {
+ try (XWPFDocument document = DocxExports.withLayout(400, 600, 20, content)) {
assertThat(document.getTables()).hasSize(1);
return document.getTables().get(0);
}
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxTableWidthTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxTableWidthTest.java
index 7d7582805..8bd8a5d8f 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxTableWidthTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxTableWidthTest.java
@@ -14,19 +14,20 @@
import static org.assertj.core.api.Assertions.assertThat;
/**
- * A table is as wide as the document says — in the cases where the document says.
+ * A table is as wide as the document says — and, when a layout was compiled, as wide as it
+ * came out.
*
- * Nothing used to write a width at all, so Word sized every table to its own content
- * and a table of short values came out far narrower than the reference draws it. But the
- * fix is not "as wide as the page": with no stated width the engine gives a table the sum
- * of its natural column widths, and an {@code auto} column's natural width is its widest
- * unwrapped cell. That is a measurement, and this backend has no font runtime to make it,
- * so the content width would be a guess that happens to be right for a table whose text
- * fills the line and wrong for one with three short values in it.
+ * These pin the export with nothing measured behind it, which is what a caller holding
+ * a graph and a canvas gets, and what a document the engine cannot lay out falls back to.
+ * There the rule is: write what is knowable and nothing else. A width the author stated and
+ * a grid of fixed columns are the document's own numbers; an {@code auto} column's width is
+ * its widest unwrapped cell, and measuring is what this backend has no font runtime for. So
+ * "as wide as the page" is not a fallback — it would be right for a table whose text fills
+ * the line and wrong for one holding three short values, which is how the old
+ * shrink-to-content was wrong in the other direction.
*
- * What is knowable without measuring is written: a width the author stated, and the
- * column widths when every column is fixed. What is not stays Word's until resolved
- * layout can supply the measured widths.
+ * {@link #aMeasuredTableTakesEveryColumnFromTheLayout()} is the other half: given the
+ * layout, the same table stops guessing entirely.
*
* @author Artem Demchyshyn
*/
@@ -124,6 +125,39 @@ void oneAutoColumnLeavesTheWholeTableToWord() throws Exception {
.isTrue();
}
+ @Test
+ void aMeasuredTableTakesEveryColumnFromTheLayout() throws Exception {
+ // The same table the first case leaves to Word. Given the layout there is nothing
+ // to decide: the resolved cells carry the width each column came out at, auto
+ // included, and the grid is written from them.
+ XWPFTable table = measuredTable(page -> page.addTable(t -> t
+ .autoColumns(3)
+ .headerRow("Item", "Qty", "Amount")
+ .row("Platform subscription", "12", "1 440.00")));
+
+ var grid = table.getCTTbl().getTblGrid();
+ assertThat(grid).isNotNull();
+ assertThat(grid.sizeOfGridColArray()).isEqualTo(3);
+ assertThat(twips(grid.getGridColArray(0).getW()))
+ .as("the widest column is the one holding the longest text")
+ .isGreaterThan(twips(grid.getGridColArray(1).getW()));
+ assertThat(sumOfColumns(table))
+ .as("the columns tile the table exactly")
+ .isEqualTo(widthTwips(table));
+ assertThat(widthTwips(table))
+ .as("and the table is narrower than the page, because its content is")
+ .isLessThan(Math.round(CONTENT_WIDTH * TWIPS_PER_POINT));
+ assertThat(table.getCTTbl().getTblPr().getTblLayout().getType().toString())
+ .as("fixed, so Word does not re-fit what was measured")
+ .isEqualTo("fixed");
+ }
+
+ private static long sumOfColumns(XWPFTable table) {
+ return table.getCTTbl().getTblGrid().getGridColList().stream()
+ .mapToLong(column -> twips(column.getW()))
+ .sum();
+ }
+
@Test
void aRowCarriedAsATableSpansTheContentWidth() throws Exception {
// A row takes the whole width it is offered whatever its children measure —
@@ -154,17 +188,19 @@ private static long twips(Object measure) {
return Long.parseLong(String.valueOf(measure));
}
+ /** The table as written with nothing measured behind it. */
private static XWPFTable onlyTable(
Consumer content) throws Exception {
- byte[] docx;
- try (DocumentSession session = GraphCompose.document()
- .pageSize(PAGE_WIDTH, 600)
- .margin(DocumentInsets.of(MARGIN))
- .create()) {
- session.pageFlow(content::accept);
- docx = session.export(new DocxSemanticBackend());
+ try (XWPFDocument document = DocxExports.withoutLayout(PAGE_WIDTH, 600, MARGIN, content)) {
+ assertThat(document.getTables()).hasSize(1);
+ return document.getTables().get(0);
}
- try (XWPFDocument document = new XWPFDocument(new ByteArrayInputStream(docx))) {
+ }
+
+ /** The table as written when the compiled layout is there to read. */
+ private static XWPFTable measuredTable(
+ Consumer content) throws Exception {
+ try (XWPFDocument document = DocxExports.withLayout(PAGE_WIDTH, 600, MARGIN, content)) {
assertThat(document.getTables()).hasSize(1);
return document.getTables().get(0);
}
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxVerticalSpacingTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxVerticalSpacingTest.java
new file mode 100644
index 000000000..95b6a7b03
--- /dev/null
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxVerticalSpacingTest.java
@@ -0,0 +1,145 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.document.dsl.PageFlowBuilder;
+import com.demcha.compose.document.style.DocumentColor;
+import com.demcha.compose.document.style.DocumentInsets;
+import org.apache.poi.xwpf.usermodel.XWPFDocument;
+import org.apache.poi.xwpf.usermodel.XWPFParagraph;
+import org.junit.jupiter.api.Test;
+import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTPPr;
+
+import java.util.List;
+import java.util.function.Consumer;
+
+import static org.assertj.core.api.Assertions.assertThat;
+
+/**
+ * The space a block holds above and below itself reaches Word.
+ *
+ * None of it used to: a paragraph's {@code margin} and {@code padding} were dropped, and
+ * a container's were dropped twice over, since a container is not a Word object and its
+ * children are written where it stood. Every exported document ran its blocks together and
+ * leaned on whatever Word puts between paragraphs — which was invisible only while the line
+ * height was Word's too, tall enough to stand in for the gaps.
+ *
+ * Vertical space is one of the few parts of a node's box Word holds natively, which is
+ * why this is written where the horizontal half is not. A container hands its top edge to
+ * the first paragraph written inside it and its bottom edge to the last, because that is
+ * where a reader sees it either way.
+ *
+ * @author Artem Demchyshyn
+ */
+class DocxVerticalSpacingTest {
+
+ private static final double TWIPS_PER_POINT = 20.0;
+ private static final DocumentColor SURFACE = DocumentColor.rgb(238, 243, 249);
+
+ @Test
+ void aParagraphCarriesItsOwnMarginAndPadding() throws Exception {
+ List paragraphs = bodyOf(page -> page
+ .addParagraph(p -> p.text("Above"))
+ .addParagraph(p -> p.text("Spaced")
+ .padding(DocumentInsets.top(16))
+ .margin(DocumentInsets.bottom(6))));
+
+ assertThat(before(paragraphs.get(1))).isEqualTo(Math.round(16 * TWIPS_PER_POINT));
+ assertThat(after(paragraphs.get(1))).isEqualTo(Math.round(6 * TWIPS_PER_POINT));
+ }
+
+ @Test
+ void marginAndPaddingOnTheSameEdgeAddUp() throws Exception {
+ // Different things to the engine — outside the box and inside it — but Word has one
+ // gap above a paragraph, and the page shows their sum.
+ List paragraphs = bodyOf(page -> page.addParagraph(p -> p.text("Both")
+ .padding(DocumentInsets.top(10))
+ .margin(DocumentInsets.top(4))));
+
+ assertThat(before(paragraphs.get(0))).isEqualTo(Math.round(14 * TWIPS_PER_POINT));
+ }
+
+ @Test
+ void aContainersEdgesGoToItsFirstAndLastParagraph() throws Exception {
+ List paragraphs = bodyOf(page -> page.addSection("Card", card -> card
+ .fillColor(SURFACE)
+ .padding(DocumentInsets.symmetric(14, 0))
+ .margin(DocumentInsets.symmetric(6, 0))
+ .addParagraph(p -> p.text("First"))
+ .addParagraph(p -> p.text("Middle"))
+ .addParagraph(p -> p.text("Last"))));
+
+ assertThat(paragraphs).hasSize(3);
+ assertThat(before(paragraphs.get(0)))
+ .as("14pt of padding and 6pt of margin, on the paragraph that starts the card")
+ .isEqualTo(Math.round(20 * TWIPS_PER_POINT));
+ assertThat(before(paragraphs.get(1))).as("nothing in the middle").isZero();
+ assertThat(after(paragraphs.get(1))).isZero();
+ assertThat(after(paragraphs.get(2)))
+ .as("and the same below, on the one that ends it")
+ .isEqualTo(Math.round(20 * TWIPS_PER_POINT));
+ }
+
+ @Test
+ void nestingAddsUpOnTheSameParagraph() throws Exception {
+ // An outer section and an inner card both start at the same paragraph, so it
+ // carries both edges — which is what the page shows.
+ List paragraphs = bodyOf(page -> page.addSection("Outer", outer -> outer
+ .padding(DocumentInsets.top(8))
+ .addSection("Inner", inner -> inner
+ .padding(DocumentInsets.top(12))
+ .addParagraph(p -> p.text("Deep").padding(DocumentInsets.top(3))))));
+
+ assertThat(before(paragraphs.get(0))).isEqualTo(Math.round(23 * TWIPS_PER_POINT));
+ }
+
+ @Test
+ void aContainerWithNoParagraphLeavesNothingBehindForTheNextOne() throws Exception {
+ // A container of tables has nowhere to put its top edge — Word has no space-before
+ // on a table. What must not happen is the edge waiting around and landing on
+ // whatever paragraph comes next, which would put it somewhere the document never
+ // asked for.
+ List paragraphs = bodyOf(page -> page
+ .addSection("TablesOnly", section -> section
+ .padding(DocumentInsets.symmetric(30, 0))
+ .addTable(t -> t.autoColumns(2).row("A", "B")))
+ .addParagraph(p -> p.text("After")));
+
+ assertThat(before(paragraphs.get(paragraphs.size() - 1)))
+ .as("the section's 30pt did not follow the table out")
+ .isZero();
+ }
+
+ @Test
+ void aBlockThatAsksForNothingCarriesNoSpacingAtAll() throws Exception {
+ List paragraphs = bodyOf(page -> page.addParagraph(p -> p.text("Plain")));
+
+ CTPPr properties = paragraphs.get(0).getCTP().getPPr();
+ assertThat(properties == null || !properties.isSetSpacing()
+ || (!properties.getSpacing().isSetBefore() && !properties.getSpacing().isSetAfter()))
+ .as("no w:before and no w:after, rather than zeros")
+ .isTrue();
+ }
+
+ private static long before(XWPFParagraph paragraph) {
+ CTPPr properties = paragraph.getCTP().getPPr();
+ if (properties == null || !properties.isSetSpacing() || !properties.getSpacing().isSetBefore()) {
+ return 0;
+ }
+ return Long.parseLong(String.valueOf(properties.getSpacing().getBefore()));
+ }
+
+ private static long after(XWPFParagraph paragraph) {
+ CTPPr properties = paragraph.getCTP().getPPr();
+ if (properties == null || !properties.isSetSpacing() || !properties.getSpacing().isSetAfter()) {
+ return 0;
+ }
+ return Long.parseLong(String.valueOf(properties.getSpacing().getAfter()));
+ }
+
+ private static List bodyOf(Consumer content) throws Exception {
+ try (XWPFDocument document = DocxExports.withLayout(400, 600, 20, content)) {
+ return document.getParagraphs().stream()
+ .filter(paragraph -> !paragraph.getText().isBlank())
+ .toList();
+ }
+ }
+}
From f7fd3a85b33cdb0858c665a24283797c1476719c Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Mon, 21 Sep 2026 21:57:17 +0100
Subject: [PATCH 03/10] docs(render-docx): record the geometry the export now
measures
The recipe said what maps to Word and not which numbers the export gets
to know. Names the three the resolved layout supplies, what the space
around a block becomes, what asking for the layout costs, and what a
document that cannot be laid out falls back to.
---
CHANGELOG.md | 41 +++++++++++++++++++++++++++++++++----
docs/recipes/docx-export.md | 39 +++++++++++++++++++++++++++--------
2 files changed, 68 insertions(+), 12 deletions(-)
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 5dce8be37..b3e53fa38 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -5,7 +5,6 @@ follow semantic versioning; release dates are ISO 8601.
## v2.5.0 — Planned
-
### Public API
- **A container's fill and borders now reach the DOCX export.** A `SectionNode` or
@@ -229,6 +228,43 @@ follow semantic versioning; release dates are ISO 8601.
size and an alt text. A document's own page keeps its own preview: there the document is the
subject.
+- **The DOCX export now takes line height and column widths from the resolved layout, and
+ carries the space around a block.** _(The layout opt-in is experimental — see
+ [API stability](docs/api-stability.md).)_ Two of the things that decide how an exported
+ document looks are measurements over the font — how tall a line of text is, and how wide
+ a column came out — and a semantic backend has no font runtime, so both were Word's to
+ decide. Measured against the reference render, Word set a body line at 13.9pt where the
+ document says 9.7 (LibreOffice: 12.1), so everything below the first paragraph sat lower
+ than it should and the gap grew with every line. The export now asks for the layout the
+ engine already compiled: the line height is written as `w:spacing w:lineRule="exact"` on
+ every paragraph, table cells and list items included, and a table's columns come from
+ the resolved cells with `w:tblLayout` fixed so Word does not re-fit them. A row's
+ columns come from where the layout placed its children — only their starts, since a
+ placed child is as wide as its own content and its width says nothing about where its
+ column ends.
+
+ The space a block holds above and below itself is not a measurement and was missing too.
+ A paragraph's `margin` and `padding` now become `w:spacing` before and after, and a
+ container — which is not a Word object, its children written where it stood — hands its
+ top edge to the first paragraph inside it and its bottom edge to the last. Both add to
+ what a paragraph asks for itself, so a card inside a section sums the way the page does.
+ A container that begins or ends with a table leaves that edge unwritten rather than
+ parking it on whatever paragraph comes next: Word has no space-before on a table, and an
+ empty paragraph would add a line the document never asked for. The horizontal half of
+ that box still has no paragraph-level equivalent and is still dropped.
+
+ Measured through Word 16.0 on the two-page probe: the body now starts at 65.5pt against
+ the reference's 65.2 and sets lines at 9.8 against 9.7; the worst grid cell falls from
+ 78.9% to 52.1%; and the largest landmark drift down the first page falls from 44pt to
+ 20pt. Editing is unchanged at 7 of 7 scenarios.
+
+ Asking for the layout costs a measurement and pagination pass over the document — the
+ same work a PDF render does — and reads each image a second time. A document the
+ fixed-layout pipeline refuses still exports: the failure is logged once and the writer
+ falls back to what the document itself states, which writes a table width only when the
+ author stated one or every column is fixed, a row's columns only when they are weights,
+ an even split or fixed, and no line height at all.
+
## v2.4.1 — 2026-09-21
### Performance
@@ -2717,7 +2753,6 @@ follow semantic versioning; release dates are ISO 8601.
build until it is marked. Annotation and documentation only; no signature or behaviour
changed.
-
## v2.2.2 — 2026-08-27
### Public API
@@ -2806,7 +2841,6 @@ follow semantic versioning; release dates are ISO 8601.
No rendered output changed, and no layout, pagination or render behaviour changed.
-
## v2.2.1 — 2026-08-25
### Public API
@@ -3890,7 +3924,6 @@ follow semantic versioning; release dates are ISO 8601.
catch the site going back. Three rendered documents change — the engine
deck, the feature catalogue and the chart showcase. ([#451](https://github.com/DemchaAV/GraphCompose/issues/451))
-
- **A heading no longer strands above a block that was asked to stay whole.**
`keepWithNext()` decides by asking whether the heading plus the *first line* of
the next block fits, but a `keepTogether()` block has no first line to break
diff --git a/docs/recipes/docx-export.md b/docs/recipes/docx-export.md
index 6c330fb78..951123805 100644
--- a/docs/recipes/docx-export.md
+++ b/docs/recipes/docx-export.md
@@ -53,6 +53,27 @@ PDF never pull POI.
Page geometry (size and margins) and session metadata (title, author,
subject, keywords) carry into the Word document as well.
+## Measured geometry
+
+The export asks the session for the resolved layout and writes three things from it that
+it cannot work out for itself:
+
+| What | Where it lands |
+|---|---|
+| Line height | `w:spacing w:lineRule="exact"` on every paragraph, cells and list items included — the height the engine measured, not a multiple Word would measure again against a substituted font |
+| Table columns | the resolved cell widths as `w:gridCol`, with `w:tblLayout` fixed so Word does not re-fit them |
+| Row columns | where the layout placed each child, with the row's gap and padding folded into the neighbouring column and taken back out as that cell's margin |
+
+The space a block holds above and below itself needs no measuring and is written from the
+document: a paragraph's `margin` and `padding` become `w:spacing` before and after, and a
+container hands its top edge to the first paragraph inside it and its bottom edge to the
+last, since a container is not a Word object. Both add to what a paragraph asks for
+itself, so nesting sums the way the page does. The horizontal half of that box has no
+paragraph-level equivalent and is still dropped — see "What a panel keeps and loses".
+
+Asking for the layout costs a measurement and pagination pass over the document, the same
+work a PDF render does, and it reads each image a second time.
+
## Named styles, so the document can be restyled
The export writes a styles part whose `Normal` carries the document's own body text —
@@ -123,14 +144,16 @@ Not representable, and left undone rather than approximated:
## What falls back
-- **An `auto` column's width → Word's own sizing.** A table with no stated width is as
- wide as its columns naturally need, and an `auto` column's natural width is its widest
- unwrapped cell. That is a measurement, and this backend has no font runtime to make it,
- so such a table is left to Word's autofit rather than given a guessed width — writing
- the content width instead would be right for a table whose text fills the line and
- wrong for one holding three short values. A row divides the same way: an `auto` column,
- a non-`START` arrangement or a grow spacer all ask what a child's content measures, so
- those rows keep Word's split too. State a width, or fixed columns, to pin either.
+- **A document the engine cannot lay out → the same export, without measured geometry.**
+ The export asks the session for the resolved layout (see "Measured geometry" above). A
+ document that the fixed-layout pipeline refuses — a list item made of inline runs
+ without marker geometry, for instance — still exports: the failure is logged once and
+ the writer falls back to what the document itself states. In that fallback a table's
+ width is written only when the author stated one or every column is fixed, a row's
+ columns only when they are weights, an even split or fixed, and no line height is
+ written at all. An `auto` column and the flex path are measurements, and a guess in
+ their place would be right for a table whose text fills the line and wrong for one
+ holding three short values.
- **Charts → data table.** The semantic export has no layout pass, so a
chart's compiled vector geometry does not exist here. Its *semantic*
From 24a9e398366708f0740b3939e4b09c3f9572ec89 Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Tue, 22 Sep 2026 01:00:05 +0100
Subject: [PATCH 04/10] feat(render-docx): ship the faces the document is set
in
A face was named and never shipped. The export declared Lato on its
runs and embedded nothing, so on a machine without Lato installed --
which is most of them, this one included -- Word substituted another
face, and a substituted face has different glyph widths, so every line
breaks somewhere else and the geometry above it stops meaning anything.
The package now carries a font table and one obfuscated part per face,
for every family the document names that has a file behind it: the
bundled families and whatever the session registered. The standard PDF
faces are never written, and not because of their terms -- they are
names, not files, and a reader gets Word's substitution for them, the
same one a PDF viewer applies.
Only the faces the document uses travel. A family's four faces are
about two and a half megabytes, which the probe paid in full for one
line of Lato before the slot was read from the style; it now pays 366
kilobytes for the one face that line needs.
A face states its own terms in OS/2, and the format distinguishes
"embeddable to read and print" from "embeddable in a document someone
will type into". Only the second is written; the first is refused with
a line in the log naming the family, rather than shipped and hoped for.
Measured through Word 16.0: the exported document now reports Lato in
the render, not a substitution, and the Lato paragraph breaks at the
same word as the reference, ending within 2.3pt over a 489pt line. The
grid metric is unchanged, and honestly so -- one line of this corpus is
set in an embeddable family.
Tests: the key derivation against the one worked example the format
publishes, that only the first 32 bytes are scrambled and come back,
each fsType value, a bundled family travelling with the document, only
the used faces travelling, a standard face shipping nothing, and the
shipped part being the real font behind its scrambled header.
---
.../semantic/docx/DocxFontEmbedding.java | 163 ++++++++
.../backend/semantic/docx/DocxFontTable.java | 353 ++++++++++++++++++
.../semantic/docx/DocxSemanticBackend.java | 1 +
.../semantic/docx/DocxFontEmbeddingTest.java | 125 +++++++
.../semantic/docx/DocxFontTableTest.java | 146 ++++++++
5 files changed, 788 insertions(+)
create mode 100644 render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontEmbedding.java
create mode 100644 render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
create mode 100644 render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontEmbeddingTest.java
create mode 100644 render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTableTest.java
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontEmbedding.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontEmbedding.java
new file mode 100644
index 000000000..108fe2bdf
--- /dev/null
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontEmbedding.java
@@ -0,0 +1,163 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import java.nio.ByteOrder;
+import java.nio.ByteBuffer;
+import java.util.UUID;
+
+/**
+ * What a font file says about being embedded, and how Word wants it written.
+ *
+ * Two things stand between a font on the classpath and a font inside a {@code .docx}.
+ * The first is permission: an OpenType face states in its {@code OS/2} table what an
+ * embedder may do with it, and a face that may only be previewed and printed is not a
+ * face a document can be edited in. The second is the format: Word does not store the
+ * font as it came, but with its first bytes scrambled against a key the package carries,
+ * so a font part is not a font file anyone can lift out and install.
+ *
+ * @author Artem Demchyshyn
+ */
+final class DocxFontEmbedding {
+
+ /** {@code fsType} bit 1: the face may not be embedded at all. */
+ private static final int RESTRICTED_LICENSE = 0x0002;
+ /** {@code fsType} bit 2: it may be embedded to read and print, but not to edit. */
+ private static final int PREVIEW_AND_PRINT = 0x0004;
+ /** {@code fsType} bit 3: it may be embedded in a document that is edited. */
+ private static final int EDITABLE = 0x0008;
+
+ /** Bytes of a font part that are scrambled — always the first 32, or the whole file. */
+ private static final int OBFUSCATED_PREFIX = 32;
+
+ private DocxFontEmbedding() {
+ }
+
+ /** What a face's own licence bits allow. */
+ enum Permission {
+ /** Installable or editable: it may go into a document people will edit. */
+ ALLOWED,
+ /** Preview-and-print only: readable and printable, but not for an editable file. */
+ PREVIEW_ONLY,
+ /** Restricted: the face may not be embedded at all. */
+ RESTRICTED,
+ /** No {@code OS/2} table, or one too short to carry {@code fsType}. */
+ UNKNOWN
+ }
+
+ /**
+ * Reads what {@code font} permits, from its own {@code OS/2} table.
+ *
+ * {@code fsType} is a bit field with one exception to that: zero means installable
+ * embedding, the most permissive value there is. Bits 1 and 2 are mutually exclusive
+ * with bit 3 by the specification, and a font that sets none of the three is treated
+ * the same as zero — it restricts nothing.
+ *
+ * @param font the whole font file
+ * @return what its licence bits allow
+ * @see
+ * OpenType OS/2 fsType
+ */
+ static Permission permissionOf(byte[] font) {
+ int fsType = readFsType(font);
+ if (fsType < 0) {
+ return Permission.UNKNOWN;
+ }
+ if ((fsType & RESTRICTED_LICENSE) != 0) {
+ return Permission.RESTRICTED;
+ }
+ if ((fsType & PREVIEW_AND_PRINT) != 0 && (fsType & EDITABLE) == 0) {
+ return Permission.PREVIEW_ONLY;
+ }
+ return Permission.ALLOWED;
+ }
+
+ /**
+ * Finds {@code fsType} in the font's table directory.
+ *
+ * @return the raw {@code fsType}, or -1 when the font carries no readable {@code OS/2}
+ */
+ private static int readFsType(byte[] font) {
+ if (font == null || font.length < 12) {
+ return -1;
+ }
+ ByteBuffer buffer = ByteBuffer.wrap(font).order(ByteOrder.BIG_ENDIAN);
+ int numTables = Short.toUnsignedInt(buffer.getShort(4));
+ for (int index = 0; index < numTables; index++) {
+ int entry = 12 + index * 16;
+ if (entry + 16 > font.length) {
+ return -1;
+ }
+ String tag = new String(font, entry, 4, java.nio.charset.StandardCharsets.US_ASCII);
+ if (!"OS/2".equals(tag)) {
+ continue;
+ }
+ int offset = buffer.getInt(entry + 8);
+ // version(2) avgCharWidth(2) weightClass(2) widthClass(2) then fsType.
+ if (offset < 0 || offset + 10 > font.length) {
+ return -1;
+ }
+ return Short.toUnsignedInt(buffer.getShort(offset + 8));
+ }
+ return -1;
+ }
+
+ /**
+ * A font part: the bytes Word stores, and the key it needs to read them back.
+ *
+ * @param bytes the obfuscated font
+ * @param fontKey the key, formatted the way {@code w:fontKey} carries it
+ */
+ record Obfuscated(byte[] bytes, String fontKey) {
+ }
+
+ /**
+ * Scrambles a font the way a Word package stores one.
+ *
+ * A font part is the font file with its first 32 bytes — the sfnt header and the
+ * start of the table directory — exclusive-ored against a key derived from a GUID the
+ * package states beside it. It is not encryption and is not meant to be: it stops a
+ * font being lifted out of a document and installed, and nothing more.
+ *
+ * The key is the GUID's sixteen bytes, read from its hexadecimal form in reverse
+ * order, pair by pair, and applied twice over the thirty-two bytes.
+ *
+ * @param font the font file as it came
+ * @return the part's bytes and the key that unscrambles them
+ */
+ static Obfuscated obfuscate(byte[] font) {
+ return obfuscate(font, UUID.randomUUID());
+ }
+
+ /**
+ * The same, against a stated key — so the derivation can be checked against the one
+ * worked example the format publishes rather than only against itself.
+ *
+ * @param font the font file as it came
+ * @param uuid the key to scramble it with
+ * @return the part's bytes and the key that unscrambles them
+ */
+ static Obfuscated obfuscate(byte[] font, UUID uuid) {
+ byte[] key = keyOf(uuid);
+ byte[] scrambled = font.clone();
+ for (int index = 0; index < Math.min(OBFUSCATED_PREFIX, scrambled.length); index++) {
+ scrambled[index] ^= key[index % key.length];
+ }
+ return new Obfuscated(scrambled,
+ "{" + uuid.toString().toUpperCase(java.util.Locale.ROOT) + "}");
+ }
+
+ /**
+ * The sixteen key bytes of a font key: its hexadecimal digits, read in reverse by pair.
+ *
+ * @param uuid the GUID {@code w:fontKey} states
+ * @return the key the first bytes of the font are exclusive-ored against
+ */
+ static byte[] keyOf(UUID uuid) {
+ String hex = uuid.toString().replace("-", "");
+ byte[] key = new byte[16];
+ for (int index = 0; index < key.length; index++) {
+ int at = hex.length() - 2 - index * 2;
+ key[index] = (byte) Integer.parseInt(hex.substring(at, at + 2), 16);
+ }
+ return key;
+ }
+}
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
new file mode 100644
index 000000000..944b2d9e0
--- /dev/null
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
@@ -0,0 +1,353 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.document.layout.DocumentGraph;
+import com.demcha.compose.document.node.DocumentNode;
+import com.demcha.compose.document.node.InlineHighlightRun;
+import com.demcha.compose.document.node.InlineRun;
+import com.demcha.compose.document.node.InlineTextRun;
+import com.demcha.compose.document.node.ListNode;
+import com.demcha.compose.document.node.ParagraphNode;
+import com.demcha.compose.document.node.TableNode;
+import com.demcha.compose.document.style.DocumentTextStyle;
+import com.demcha.compose.document.table.DocumentTableCell;
+import com.demcha.compose.document.table.DocumentTableStyle;
+import com.demcha.compose.font.DefaultFonts;
+import com.demcha.compose.font.FontFamilyDefinition;
+import com.demcha.compose.font.FontName;
+import org.apache.poi.openxml4j.opc.OPCPackage;
+import org.apache.poi.openxml4j.opc.PackagePart;
+import org.apache.poi.openxml4j.opc.PackagePartName;
+import org.apache.poi.openxml4j.opc.PackagingURIHelper;
+import org.apache.poi.openxml4j.opc.TargetMode;
+import org.apache.poi.xwpf.usermodel.XWPFDocument;
+import org.slf4j.Logger;
+import org.slf4j.LoggerFactory;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.io.OutputStream;
+import java.nio.charset.StandardCharsets;
+import java.util.ArrayList;
+import java.util.Collection;
+import java.util.LinkedHashMap;
+import java.util.LinkedHashSet;
+import java.util.List;
+import java.util.Map;
+import java.util.Set;
+
+/**
+ * Puts the document's own faces inside the package, so it reads in the font it was
+ * written in.
+ *
+ * A face was named and never shipped: the export declared {@code Lato} on its runs and
+ * embedded nothing, so on a machine without Lato installed — which is most of them — Word
+ * substituted another face. A substituted face has different glyph widths, so every line
+ * breaks somewhere else and no amount of correct geometry above it survives. Measured on
+ * this machine, where Lato is not installed, that is exactly what happened.
+ *
+ * Only a face the document may be edited in is written. An OpenType face states its
+ * terms in {@code OS/2}, and the format takes them seriously enough to distinguish
+ * "embeddable for reading and printing" from "embeddable in a document someone will type
+ * into" — this writes the second and refuses the first, with a line in the log naming the
+ * family rather than a silent omission.
+ *
+ * The standard PDF faces are never written, and not because of their terms: they are
+ * names, not files. Nothing ships Helvetica, and a reader opening the document has Word's
+ * own substitution for it, which is what a PDF viewer does with the same document.
+ *
+ * @author Artem Demchyshyn
+ */
+final class DocxFontTable {
+
+ private static final Logger LOG = LoggerFactory.getLogger(DocxFontTable.class);
+
+ private static final String FONT_TABLE_TYPE =
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.fontTable+xml";
+ private static final String FONT_TABLE_RELATION =
+ "http://schemas.openxmlformats.org/officeDocument/2006/relationships/fontTable";
+ private static final String FONT_PART_TYPE =
+ "application/vnd.openxmlformats-officedocument.obfuscatedFont";
+ private static final String FONT_RELATION =
+ "http://schemas.openxmlformats.org/officeDocument/2006/relationships/font";
+
+ private static final String W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main";
+ private static final String R_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships";
+
+ private DocxFontTable() {
+ }
+
+ /**
+ * Writes a font table for every family the document uses that can be embedded.
+ *
+ * @param document the document being written
+ * @param graph the document's own tree, read for the faces it names
+ * @param custom families the session registered, which win over the bundled ones
+ * @throws IOException if a part cannot be written
+ */
+ static void write(XWPFDocument document,
+ DocumentGraph graph,
+ Collection custom) throws IOException {
+ List embedded = resolve(graph, custom);
+ if (embedded.isEmpty()) {
+ return;
+ }
+
+ OPCPackage pkg = document.getPackage();
+ PackagePartName tableName = partName("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/word/fontTable.xml");
+ PackagePart table = pkg.createPart(tableName, FONT_TABLE_TYPE);
+ document.getPackagePart().addRelationship(tableName, TargetMode.INTERNAL, FONT_TABLE_RELATION);
+
+ int index = 0;
+ for (Embedded family : embedded) {
+ for (Face face : family.faces()) {
+ PackagePartName facePart = partName("/word/fonts/font" + (++index) + ".odttf");
+ PackagePart part = pkg.createPart(facePart, FONT_PART_TYPE);
+ try (OutputStream out = part.getOutputStream()) {
+ out.write(face.bytes());
+ }
+ face.relationshipId(table.addRelationship(facePart, TargetMode.INTERNAL, FONT_RELATION)
+ .getId());
+ }
+ }
+
+ try (OutputStream out = table.getOutputStream()) {
+ out.write(toXml(embedded).getBytes(StandardCharsets.UTF_8));
+ }
+ }
+
+ /** One family that will be written, and the faces of it that may be. */
+ private record Embedded(String wordFamily, List faces) {
+ }
+
+ /** One face of a family: which slot it fills, its bytes, and how the part refers to it. */
+ private static final class Face {
+ private final String element;
+ private final byte[] bytes;
+ private final String fontKey;
+ private String relationshipId;
+
+ Face(String element, byte[] bytes, String fontKey) {
+ this.element = element;
+ this.bytes = bytes;
+ this.fontKey = fontKey;
+ }
+
+ String element() {
+ return element;
+ }
+
+ byte[] bytes() {
+ return bytes;
+ }
+
+ void relationshipId(String id) {
+ this.relationshipId = id;
+ }
+ }
+
+ /**
+ * Reads every face the document could be set in, and keeps the ones it may ship.
+ *
+ * A family the session registered wins over a bundled one of the same name: it was
+ * registered to be used, and the document was laid out with it.
+ */
+ private static List resolve(DocumentGraph graph, Collection custom) {
+ Map> used = new LinkedHashMap<>();
+ for (DocumentNode root : graph.roots()) {
+ collectFonts(root, used);
+ }
+ if (used.isEmpty()) {
+ return List.of();
+ }
+
+ Map families = new LinkedHashMap<>();
+ for (FontFamilyDefinition family : DefaultFonts.bundledFamilies()) {
+ families.put(family.name(), family);
+ }
+ if (custom != null) {
+ for (FontFamilyDefinition family : custom) {
+ families.put(family.name(), family);
+ }
+ }
+
+ List embedded = new ArrayList<>();
+ for (Map.Entry> entry : used.entrySet()) {
+ FontFamilyDefinition family = families.get(entry.getKey());
+ if (family == null || family.fontSourceSet().isEmpty()) {
+ // A standard-14 name, or one nothing registered: there is no file to ship.
+ continue;
+ }
+ List faces = facesOf(family, entry.getValue());
+ if (!faces.isEmpty()) {
+ embedded.add(new Embedded(family.wordFamily(), faces));
+ }
+ }
+ return embedded;
+ }
+
+ /** Which of a family's four faces a document asked for. */
+ private enum Slot {
+ REGULAR("embedRegular"),
+ BOLD("embedBold"),
+ ITALIC("embedItalic"),
+ BOLD_ITALIC("embedBoldItalic");
+
+ private final String element;
+
+ Slot(String element) {
+ this.element = element;
+ }
+
+ String element() {
+ return element;
+ }
+
+ /**
+ * The face a style is set in.
+ *
+ * An underline or a strikethrough is drawn over the regular face rather than
+ * being a face of its own, so they answer the same as no decoration at all.
+ */
+ static Slot of(DocumentTextStyle style) {
+ if (style.decoration() == null) {
+ return REGULAR;
+ }
+ return switch (style.decoration()) {
+ case BOLD -> BOLD;
+ case ITALIC -> ITALIC;
+ case BOLD_ITALIC -> BOLD_ITALIC;
+ default -> REGULAR;
+ };
+ }
+ }
+
+ /**
+ * The faces the document uses, minus the ones that may not travel in an edited file.
+ *
+ * Only what is used: the four faces of one family are two and a half megabytes, and
+ * a document setting one line in a family has no use for the other three. A reader who
+ * later bolds a word gets whatever their machine does for a missing bold face, which is
+ * what happens in any document that does not carry one.
+ */
+ private static List facesOf(FontFamilyDefinition family, Set slots) {
+ FontFamilyDefinition.FontSourceSet sources = family.fontSourceSet().orElseThrow();
+ List faces = new ArrayList<>(slots.size());
+ for (Slot slot : Slot.values()) {
+ if (!slots.contains(slot)) {
+ continue;
+ }
+ addFace(faces, family, slot, switch (slot) {
+ case REGULAR -> sources.regular();
+ case BOLD -> sources.bold();
+ case ITALIC -> sources.italic();
+ case BOLD_ITALIC -> sources.boldItalic();
+ });
+ }
+ return faces;
+ }
+
+ private static void addFace(List faces,
+ FontFamilyDefinition family,
+ Slot slot,
+ FontFamilyDefinition.FontBinarySource source) {
+ if (source == null) {
+ return;
+ }
+ byte[] bytes;
+ try (InputStream in = source.openStream()) {
+ bytes = in.readAllBytes();
+ } catch (IOException failure) {
+ LOG.warn("DocxSemanticBackend: '{}' face of '{}' could not be read ({}); the "
+ + "document declares the family and ships this face without it",
+ slot.element(), family.wordFamily(), failure.toString());
+ return;
+ }
+ switch (DocxFontEmbedding.permissionOf(bytes)) {
+ case RESTRICTED -> {
+ LOG.warn("DocxSemanticBackend: '{}' does not permit embedding (OS/2 fsType), so "
+ + "it is named but not shipped; a reader without it installed sees a "
+ + "substituted face", family.wordFamily());
+ return;
+ }
+ case PREVIEW_ONLY -> {
+ LOG.warn("DocxSemanticBackend: '{}' permits embedding for reading and printing "
+ + "only, which a document meant to be edited cannot rely on, so it is "
+ + "named but not shipped", family.wordFamily());
+ return;
+ }
+ default -> {
+ }
+ }
+ DocxFontEmbedding.Obfuscated obfuscated = DocxFontEmbedding.obfuscate(bytes);
+ faces.add(new Face(slot.element(), obfuscated.bytes(), obfuscated.fontKey()));
+ }
+
+ /** Collects every face the tree names, wherever a style can sit. */
+ private static void collectFonts(DocumentNode node, Map> into) {
+ if (node instanceof ParagraphNode paragraph) {
+ add(paragraph.textStyle(), into);
+ for (InlineRun run : paragraph.inlineRuns()) {
+ if (run instanceof InlineTextRun text) {
+ add(text.textStyle(), into);
+ } else if (run instanceof InlineHighlightRun highlight) {
+ add(highlight.textStyle(), into);
+ }
+ }
+ } else if (node instanceof ListNode list) {
+ add(list.textStyle(), into);
+ } else if (node instanceof TableNode table) {
+ add(styleOf(table.defaultCellStyle()), into);
+ table.rowStyles().values().forEach(style -> add(styleOf(style), into));
+ table.columnStyles().values().forEach(style -> add(styleOf(style), into));
+ table.rows().forEach(row -> row.forEach(cell -> add(styleOf(cell), into)));
+ }
+ for (DocumentNode child : node.children()) {
+ collectFonts(child, into);
+ }
+ }
+
+ private static DocumentTextStyle styleOf(DocumentTableCell cell) {
+ return cell == null ? null : styleOf(cell.style());
+ }
+
+ private static DocumentTextStyle styleOf(DocumentTableStyle style) {
+ return style == null ? null : style.textStyle();
+ }
+
+ private static void add(DocumentTextStyle style, Map> into) {
+ if (style != null && style.fontName() != null) {
+ into.computeIfAbsent(style.fontName(), key -> java.util.EnumSet.noneOf(Slot.class))
+ .add(Slot.of(style));
+ }
+ }
+
+ /** The font table part, naming each family and pointing at the faces shipped for it. */
+ private static String toXml(List embedded) {
+ StringBuilder xml = new StringBuilder(512);
+ xml.append("")
+ .append("");
+ for (Embedded family : embedded) {
+ xml.append("");
+ for (Face face : family.faces()) {
+ xml.append("");
+ }
+ xml.append("");
+ }
+ return xml.append("").toString();
+ }
+
+ private static String escape(String value) {
+ return value.replace("&", "&").replace("<", "<").replace("\"", """);
+ }
+
+ private static PackagePartName partName(String path) throws IOException {
+ try {
+ return PackagingURIHelper.createPartName(path);
+ } catch (org.apache.poi.openxml4j.exceptions.InvalidFormatException failure) {
+ throw new IOException("cannot name the part " + path, failure);
+ }
+ }
+}
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
index 0f3ba98b6..eb8a587ac 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
@@ -229,6 +229,7 @@ public byte[] export(DocumentGraph graph, SemanticExportContext context) throws
try (XWPFDocument document = new XWPFDocument()) {
applyPageGeometry(document, context.canvas());
writeStylesPart(document);
+ DocxFontTable.write(document, graph, context.customFontFamilies());
applyOutputOptions(document, context.outputOptions());
for (DocumentNode root : graph.roots()) {
writeNode(document, root);
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontEmbeddingTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontEmbeddingTest.java
new file mode 100644
index 000000000..4bb54072f
--- /dev/null
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontEmbeddingTest.java
@@ -0,0 +1,125 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import org.junit.jupiter.api.Test;
+
+import java.nio.ByteBuffer;
+import java.nio.ByteOrder;
+import java.nio.charset.StandardCharsets;
+import java.util.UUID;
+
+import static org.assertj.core.api.Assertions.assertThat;
+
+/**
+ * The two things standing between a font file and a font part: its licence and its format.
+ *
+ * Both are read from the specification rather than from our own output, so both are
+ * checked against it: the key derivation against the one worked example the format
+ * publishes, and the licence bits against fonts built here to carry each value.
+ *
+ * @author Artem Demchyshyn
+ */
+class DocxFontEmbeddingTest {
+
+ /**
+ * The example {@code fontKey} published with the obfuscation algorithm, and the key
+ * bytes it must produce — its hexadecimal digits read backwards, pair by pair.
+ */
+ private static final UUID PUBLISHED_EXAMPLE = UUID.fromString("F9168C5E-CEB2-4FAA-B6BF-329BF39FA1E4");
+ private static final int[] PUBLISHED_KEY = {
+ 0xE4, 0xA1, 0x9F, 0xF3, 0x9B, 0x32, 0xBF, 0xB6,
+ 0xAA, 0x4F, 0xB2, 0xCE, 0x5E, 0x8C, 0x16, 0xF9};
+
+ @Test
+ void theKeyIsTheOneTheFormatPublishes() {
+ byte[] key = DocxFontEmbedding.keyOf(PUBLISHED_EXAMPLE);
+
+ assertThat(key).hasSize(16);
+ for (int index = 0; index < key.length; index++) {
+ assertThat(Byte.toUnsignedInt(key[index]))
+ .as("key byte %d", index)
+ .isEqualTo(PUBLISHED_KEY[index]);
+ }
+ }
+
+ @Test
+ void onlyTheFirstThirtyTwoBytesAreScrambledAndTheyComeBack() {
+ byte[] font = new byte[200];
+ for (int index = 0; index < font.length; index++) {
+ font[index] = (byte) index;
+ }
+
+ DocxFontEmbedding.Obfuscated part = DocxFontEmbedding.obfuscate(font, PUBLISHED_EXAMPLE);
+
+ assertThat(part.bytes()).hasSameSizeAs(font);
+ assertThat(part.fontKey())
+ .as("as w:fontKey carries it — braced and upper case")
+ .isEqualTo("{F9168C5E-CEB2-4FAA-B6BF-329BF39FA1E4}");
+ // Past the header the font is untouched: a part is a font with a scrambled front,
+ // not an encrypted file.
+ for (int index = 32; index < font.length; index++) {
+ assertThat(part.bytes()[index]).as("byte %d", index).isEqualTo(font[index]);
+ }
+ // And the front comes back with the same key, which is what Word does to read it.
+ byte[] restored = DocxFontEmbedding.obfuscate(part.bytes(), PUBLISHED_EXAMPLE).bytes();
+ assertThat(restored).isEqualTo(font);
+ }
+
+ @Test
+ void aShortFontIsScrambledAsFarAsItGoes() {
+ byte[] font = new byte[8];
+
+ byte[] scrambled = DocxFontEmbedding.obfuscate(font, PUBLISHED_EXAMPLE).bytes();
+
+ assertThat(scrambled).hasSize(8).isNotEqualTo(font);
+ }
+
+ @Test
+ void aFaceSaysWhatMayBeDoneWithIt() {
+ // 0 is installable, and bit 3 is editable: both may travel in a document people
+ // type into.
+ assertThat(DocxFontEmbedding.permissionOf(fontWithFsType(0x0000)))
+ .isEqualTo(DocxFontEmbedding.Permission.ALLOWED);
+ assertThat(DocxFontEmbedding.permissionOf(fontWithFsType(0x0008)))
+ .isEqualTo(DocxFontEmbedding.Permission.ALLOWED);
+ // Bit 1 forbids embedding outright.
+ assertThat(DocxFontEmbedding.permissionOf(fontWithFsType(0x0002)))
+ .isEqualTo(DocxFontEmbedding.Permission.RESTRICTED);
+ // Bit 2 alone permits reading and printing, which an edited document cannot rely
+ // on — the distinction this whole check exists for.
+ assertThat(DocxFontEmbedding.permissionOf(fontWithFsType(0x0004)))
+ .isEqualTo(DocxFontEmbedding.Permission.PREVIEW_ONLY);
+ // Bits that restrict nothing about embedding — no subsetting, bitmap only — leave
+ // the answer alone.
+ assertThat(DocxFontEmbedding.permissionOf(fontWithFsType(0x0300)))
+ .isEqualTo(DocxFontEmbedding.Permission.ALLOWED);
+ }
+
+ @Test
+ void aFontWithNoReadableTableSaysSo() {
+ assertThat(DocxFontEmbedding.permissionOf(null))
+ .isEqualTo(DocxFontEmbedding.Permission.UNKNOWN);
+ assertThat(DocxFontEmbedding.permissionOf(new byte[4]))
+ .isEqualTo(DocxFontEmbedding.Permission.UNKNOWN);
+ // A well-formed directory that simply has no OS/2 table in it.
+ assertThat(DocxFontEmbedding.permissionOf(fontWithTable("cmap", 0)))
+ .isEqualTo(DocxFontEmbedding.Permission.UNKNOWN);
+ }
+
+ /** A font file carrying one {@code OS/2} table, long enough to reach {@code fsType}. */
+ private static byte[] fontWithFsType(int fsType) {
+ return fontWithTable("OS/2", fsType);
+ }
+
+ private static byte[] fontWithTable(String tag, int fsType) {
+ int tableOffset = 12 + 16;
+ byte[] font = new byte[tableOffset + 96];
+ ByteBuffer buffer = ByteBuffer.wrap(font).order(ByteOrder.BIG_ENDIAN);
+ buffer.putInt(0, 0x00010000);
+ buffer.putShort(4, (short) 1);
+ System.arraycopy(tag.getBytes(StandardCharsets.US_ASCII), 0, font, 12, 4);
+ buffer.putInt(20, tableOffset);
+ buffer.putInt(24, 96);
+ buffer.putShort(tableOffset + 8, (short) fsType);
+ return font;
+ }
+}
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTableTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTableTest.java
new file mode 100644
index 000000000..d3f8626b7
--- /dev/null
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTableTest.java
@@ -0,0 +1,146 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.document.dsl.PageFlowBuilder;
+import com.demcha.compose.document.style.DocumentTextDecoration;
+import com.demcha.compose.document.style.DocumentTextStyle;
+import com.demcha.compose.font.FontName;
+import org.apache.poi.openxml4j.opc.PackagePart;
+import org.apache.poi.xwpf.usermodel.XWPFDocument;
+import org.junit.jupiter.api.Test;
+
+import java.io.InputStream;
+import java.nio.charset.StandardCharsets;
+import java.util.List;
+import java.util.function.Consumer;
+
+import static org.assertj.core.api.Assertions.assertThat;
+
+/**
+ * The document ships the faces it is set in.
+ *
+ * A face was named and never shipped, so on a machine without it installed the reader
+ * saw a substituted one — measured here, where Lato is not installed, Word swapped it. A
+ * substituted face has different glyph widths, so every line breaks somewhere else and the
+ * geometry above it stops meaning anything.
+ *
+ * What is shipped is narrow on purpose: the faces the document uses, and only those. One
+ * family's four faces are about two and a half megabytes, which is not a price a document
+ * setting a single line in it should pay.
+ *
+ * @author Artem Demchyshyn
+ */
+class DocxFontTableTest {
+
+ @Test
+ void aBundledFamilyTravelsWithTheDocument() throws Exception {
+ try (XWPFDocument document = export(page -> page.addParagraph(p -> p
+ .text("Set in Lato")
+ .textStyle(DocumentTextStyle.builder().fontName(FontName.LATO).build())))) {
+
+ String table = fontTable(document);
+ assertThat(table).contains("w:name=\"Lato\"").contains(" page
+ .addParagraph(p -> p.text("Regular")
+ .textStyle(DocumentTextStyle.builder().fontName(FontName.LATO).build()))
+ .addParagraph(p -> p.text("Bold")
+ .textStyle(DocumentTextStyle.builder().fontName(FontName.LATO)
+ .decoration(DocumentTextDecoration.BOLD).build())))) {
+
+ String table = fontTable(document);
+ assertThat(table).contains(" page.addParagraph(p -> p
+ .text("Set in Helvetica")
+ .textStyle(DocumentTextStyle.builder().fontName(FontName.HELVETICA).build())))) {
+
+ assertThat(fontParts(document)).isEmpty();
+ assertThat(partNames(document))
+ .as("and no font table either, rather than an empty one")
+ .noneMatch(name -> name.contains("fontTable"));
+ }
+ }
+
+ @Test
+ void theShippedFaceIsTheRealFontBehindItsScrambledHeader() throws Exception {
+ try (XWPFDocument document = export(page -> page.addParagraph(p -> p
+ .text("Set in Lato")
+ .textStyle(DocumentTextStyle.builder().fontName(FontName.LATO).build())))) {
+
+ byte[] part = fontParts(document).get(0);
+ byte[] source = latoRegular();
+
+ assertThat(part).hasSameSizeAs(source);
+ assertThat(java.util.Arrays.copyOfRange(part, 0, 32))
+ .as("the header is not the font's own")
+ .isNotEqualTo(java.util.Arrays.copyOfRange(source, 0, 32));
+ assertThat(java.util.Arrays.copyOfRange(part, 32, part.length))
+ .as("and everything past it is, byte for byte")
+ .isEqualTo(java.util.Arrays.copyOfRange(source, 32, source.length));
+
+ // Unscrambling with the key the table states gives the font back, which is the
+ // only thing that makes the part usable.
+ java.util.UUID key = java.util.UUID.fromString(
+ fontTable(document).replaceAll("(?s).*w:fontKey=\"\\{([^}]+)\\}\".*", "$1"));
+ assertThat(DocxFontEmbedding.obfuscate(part, key).bytes()).isEqualTo(source);
+ }
+ }
+
+ private static byte[] latoRegular() throws Exception {
+ try (InputStream in = DocxFontTableTest.class.getClassLoader()
+ .getResourceAsStream("fonts/google/lato/Lato-Regular.ttf")) {
+ assertThat(in).as("the bundled font artifact is on the test classpath").isNotNull();
+ return in.readAllBytes();
+ }
+ }
+
+ private static String fontTable(XWPFDocument document) throws Exception {
+ PackagePart part = document.getPackage()
+ .getPart(org.apache.poi.openxml4j.opc.PackagingURIHelper
+ .createPartName("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/word/fontTable.xml"));
+ assertThat(part).as("the document carries a font table").isNotNull();
+ try (InputStream in = part.getInputStream()) {
+ return new String(in.readAllBytes(), StandardCharsets.UTF_8);
+ }
+ }
+
+ private static List fontParts(XWPFDocument document) throws Exception {
+ return document.getPackage().getParts().stream()
+ .filter(part -> part.getPartName().getName().startsWith("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/word/fonts/"))
+ .map(part -> {
+ try (InputStream in = part.getInputStream()) {
+ return in.readAllBytes();
+ } catch (Exception failure) {
+ throw new IllegalStateException(failure);
+ }
+ })
+ .toList();
+ }
+
+ private static List partNames(XWPFDocument document) throws Exception {
+ return document.getPackage().getParts().stream()
+ .map(part -> part.getPartName().getName())
+ .toList();
+ }
+
+ private static XWPFDocument export(Consumer content) throws Exception {
+ return DocxExports.withLayout(400, 600, 20, content);
+ }
+}
From 1bf2d4403dc33111f6d3989feb2ba014097e80d2 Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Tue, 22 Sep 2026 01:00:49 +0100
Subject: [PATCH 05/10] docs(render-docx): record which faces travel with a
document
The recipe listed what the export writes and said nothing about the
fonts it writes it in, which is the difference between a reader seeing
the document and seeing a substitution. Says which families ship, which
faces of them, what a face's own terms can veto, and how each one is
stored.
---
CHANGELOG.md | 137 ++++++++++++++++++++++++++----------
docs/recipes/docx-export.md | 20 ++++++
2 files changed, 120 insertions(+), 37 deletions(-)
diff --git a/CHANGELOG.md b/CHANGELOG.md
index b3e53fa38..21edb3476 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -5,6 +5,7 @@ follow semantic versioning; release dates are ISO 8601.
## v2.5.0 — Planned
+
### Public API
- **A container's fill and borders now reach the DOCX export.** A `SectionNode` or
@@ -108,6 +109,102 @@ follow semantic versioning; release dates are ISO 8601.
short had been pulling the page back up. Line height is measured from the font, so
closing that needs resolved layout rather than arithmetic.
+- **The DOCX export now takes line height and column widths from the resolved layout, and
+ carries the space around a block.** _(The layout opt-in is experimental — see
+ [API stability](docs/api-stability.md).)_ Two of the things that decide how an exported
+ document looks are measurements over the font — how tall a line of text is, and how wide
+ a column came out — and a semantic backend has no font runtime, so both were Word's to
+ decide. Measured against the reference render, Word set a body line at 13.9pt where the
+ document says 9.7 (LibreOffice: 12.1), so everything below the first paragraph sat lower
+ than it should and the gap grew with every line. The export now asks for the layout the
+ engine already compiled: the line height is written as `w:spacing w:lineRule="exact"` on
+ every paragraph, table cells and list items included, and a table's columns come from
+ the resolved cells with `w:tblLayout` fixed so Word does not re-fit them. A row's
+ columns come from where the layout placed its children — only their starts, since a
+ placed child is as wide as its own content and its width says nothing about where its
+ column ends.
+
+ The space a block holds above and below itself is not a measurement and was missing too.
+ A paragraph's `margin` and `padding` now become `w:spacing` before and after, and a
+ container — which is not a Word object, its children written where it stood — hands its
+ top edge to the first paragraph inside it and its bottom edge to the last. Both add to
+ what a paragraph asks for itself, so a card inside a section sums the way the page does.
+ A container that begins or ends with a table leaves that edge unwritten rather than
+ parking it on whatever paragraph comes next: Word has no space-before on a table, and an
+ empty paragraph would add a line the document never asked for. The horizontal half of
+ that box still has no paragraph-level equivalent and is still dropped.
+
+ Measured through Word 16.0 on the two-page probe: the body now starts at 65.5pt against
+ the reference's 65.2 and sets lines at 9.8 against 9.7; the worst grid cell falls from
+ 78.9% to 52.1%; and the largest landmark drift down the first page falls from 44pt to
+ 20pt. Editing is unchanged at 7 of 7 scenarios.
+
+ Asking for the layout costs a measurement and pagination pass over the document — the
+ same work a PDF render does — and reads each image a second time. A document the
+ fixed-layout pipeline refuses still exports: the failure is logged once and the writer
+ falls back to what the document itself states, which writes a table width only when the
+ author stated one or every column is fixed, a row's columns only when they are weights,
+ an even split or fixed, and no line height at all.
+
+- **A session can export Word without naming the backend, and can be told what the export
+ could not carry.** _(Experimental — see [API stability](docs/api-stability.md).)_ Reaching
+ the Word export meant constructing `DocxSemanticBackend`, which means importing the render
+ artifact in the code that builds the document and carrying that dependency wherever
+ documents are built. A render backend has not needed that since 2.0, and now neither does
+ this one: `session.buildDocx(path)`, `session.writeDocx(stream)` and
+ `session.toDocxBytes()` find the backend through a new `SemanticBackendProvider` /
+ `SemanticBackendProviders` pair — the semantic half of the `ServiceLoader` path, with the
+ fixed-layout locator's rules, since a classpath behaves the same way whichever backend is
+ on it. One rule is deliberately not copied: there is no default-format lookup, because
+ "render this document" has an obvious answer worth defaulting to and "export it
+ semantically" does not. The stream stays the caller's and is not closed; the bytes are
+ produced in full before any are written, since a `.docx` is a ZIP whose directory comes
+ last; a file is written through the same atomic path as `buildPdf`, so a failed export
+ leaves the previous file rather than a damaged one.
+
+ What the export cannot carry, it has always said — to the log, one line per kind, which a
+ service generating documents for other people has no way to read. A backend built with
+ `new DocxSemanticBackend(report -> ...)` now hands over a `DocxExportReport` once the
+ bytes are complete: every dropped node and every approximation, each with the path of the
+ authored node it came from, and each marked `DROPPED` (the page draws it, the document
+ does not carry it) or `APPROXIMATED` (it is there as the nearest thing Word owns). Errors
+ do not travel this way — an export that cannot proceed throws. The log keeps saying it
+ once per kind; the report records every one, because a caller asking what the document
+ lost wants the three charts it lost and which three.
+
+- **A DOCX export now ships the faces the document is set in.** A face was named and never
+ shipped: the export declared `Lato` on its runs and embedded nothing, so on a machine
+ without Lato installed Word substituted another face — and a substituted face has
+ different glyph widths, so every line breaks somewhere else and the geometry above it
+ stops meaning anything. The package now carries a font table and one obfuscated font
+ part per face, for every family the document names that has a file behind it: the
+ bundled families and whatever the session registered. The standard PDF faces are never
+ written, and not because of their terms — they are names rather than files, and a reader
+ gets the editor's substitution for them, the same one a PDF viewer applies.
+
+ Only the faces the document uses travel: a family's four faces are about 2.5 MB, so the
+ face is chosen from each style's decoration, and a reader who later bolds a word gets
+ whatever their machine does for a missing bold face. And only what the face permits: an
+ OpenType face states its terms in `OS/2`, and the format distinguishes embedding for
+ reading and printing from embedding in a document someone will edit — a face that allows
+ only the first is named but not shipped, with one warning naming the family.
+
+ Measured through Word 16.0 on a machine where Lato is not installed: the exported
+ document reports `Lato` in the render rather than a substitution, and its Lato paragraph
+ breaks at the same word as the reference, ending within 2.3pt over a 489pt line. The
+ two-page probe's file grows from 366 KB with one face to 2.9 MB if all four are written,
+ which is why the slot is read from the style.
+
+ A run also names the family rather than the face. `FontName.HELVETICA_BOLD` is a face,
+ and it was written where Word expects a family: Word resolves a family and takes the
+ weight from `w:b`, so asked for a family by that name it found none and substituted —
+ which is how a document naming its headings by face came out set in something else. The
+ face is now resolved to its family exactly as the layout resolves it, through
+ `FontLibrary.resolveFamily`, and the name written is that family's `wordFamily()`. The
+ weight is deliberately not read from the face name, because the engine does not read it
+ either: a style naming `HELVETICA_BOLD` with no `decoration` lays out regular, and
+ writing `w:b` would make Word bolder than the page it matches. Two names for one family
+ now also weigh as one style when `Normal` is chosen, since they are written identically.
- **A semantic export backend can ask for the compiled layout.** _(Experimental — see
[API stability](docs/api-stability.md).)_ A semantic backend walks the authored tree and
gets no geometry, which is right for most of them and wrong for the ones that need a
@@ -228,43 +325,6 @@ follow semantic versioning; release dates are ISO 8601.
size and an alt text. A document's own page keeps its own preview: there the document is the
subject.
-- **The DOCX export now takes line height and column widths from the resolved layout, and
- carries the space around a block.** _(The layout opt-in is experimental — see
- [API stability](docs/api-stability.md).)_ Two of the things that decide how an exported
- document looks are measurements over the font — how tall a line of text is, and how wide
- a column came out — and a semantic backend has no font runtime, so both were Word's to
- decide. Measured against the reference render, Word set a body line at 13.9pt where the
- document says 9.7 (LibreOffice: 12.1), so everything below the first paragraph sat lower
- than it should and the gap grew with every line. The export now asks for the layout the
- engine already compiled: the line height is written as `w:spacing w:lineRule="exact"` on
- every paragraph, table cells and list items included, and a table's columns come from
- the resolved cells with `w:tblLayout` fixed so Word does not re-fit them. A row's
- columns come from where the layout placed its children — only their starts, since a
- placed child is as wide as its own content and its width says nothing about where its
- column ends.
-
- The space a block holds above and below itself is not a measurement and was missing too.
- A paragraph's `margin` and `padding` now become `w:spacing` before and after, and a
- container — which is not a Word object, its children written where it stood — hands its
- top edge to the first paragraph inside it and its bottom edge to the last. Both add to
- what a paragraph asks for itself, so a card inside a section sums the way the page does.
- A container that begins or ends with a table leaves that edge unwritten rather than
- parking it on whatever paragraph comes next: Word has no space-before on a table, and an
- empty paragraph would add a line the document never asked for. The horizontal half of
- that box still has no paragraph-level equivalent and is still dropped.
-
- Measured through Word 16.0 on the two-page probe: the body now starts at 65.5pt against
- the reference's 65.2 and sets lines at 9.8 against 9.7; the worst grid cell falls from
- 78.9% to 52.1%; and the largest landmark drift down the first page falls from 44pt to
- 20pt. Editing is unchanged at 7 of 7 scenarios.
-
- Asking for the layout costs a measurement and pagination pass over the document — the
- same work a PDF render does — and reads each image a second time. A document the
- fixed-layout pipeline refuses still exports: the failure is logged once and the writer
- falls back to what the document itself states, which writes a table width only when the
- author stated one or every column is fixed, a row's columns only when they are weights,
- an even split or fixed, and no line height at all.
-
## v2.4.1 — 2026-09-21
### Performance
@@ -2753,6 +2813,7 @@ follow semantic versioning; release dates are ISO 8601.
build until it is marked. Annotation and documentation only; no signature or behaviour
changed.
+
## v2.2.2 — 2026-08-27
### Public API
@@ -2841,6 +2902,7 @@ follow semantic versioning; release dates are ISO 8601.
No rendered output changed, and no layout, pagination or render behaviour changed.
+
## v2.2.1 — 2026-08-25
### Public API
@@ -3924,6 +3986,7 @@ follow semantic versioning; release dates are ISO 8601.
catch the site going back. Three rendered documents change — the engine
deck, the feature catalogue and the chart showcase. ([#451](https://github.com/DemchaAV/GraphCompose/issues/451))
+
- **A heading no longer strands above a block that was asked to stay whole.**
`keepWithNext()` decides by asking whether the heading plus the *first line* of
the next block fits, but a `keepTogether()` block has no first line to break
diff --git a/docs/recipes/docx-export.md b/docs/recipes/docx-export.md
index 951123805..bd8f73755 100644
--- a/docs/recipes/docx-export.md
+++ b/docs/recipes/docx-export.md
@@ -74,6 +74,26 @@ paragraph-level equivalent and is still dropped — see "What a panel keeps and
Asking for the layout costs a measurement and pagination pass over the document, the same
work a PDF render does, and it reads each image a second time.
+## Fonts travel with the document
+
+The package carries the faces the document is set in, so a reader without them installed
+sees the document rather than a substitution. What is shipped is narrow on purpose:
+
+- **Only families with a file behind them** — the bundled ones and whatever the session
+ registered. The standard PDF faces are names rather than files: nothing bundles
+ Helvetica, and a reader gets the editor's substitution for it, the same one a PDF viewer
+ applies.
+- **Only the faces the document uses.** One family's four faces are about 2.5 MB, so the
+ face is chosen from each style's decoration. A reader who later bolds a word gets
+ whatever their machine does for a missing bold face.
+- **Only what the face permits.** An OpenType face states its terms in `OS/2`, and the
+ format distinguishes embedding for reading and printing from embedding in a document
+ someone will edit. A face that allows only the first is named but not shipped, with one
+ warning naming the family.
+
+Each face is stored the way Word stores one: the font with its first 32 bytes scrambled
+against a key the font table states beside it.
+
## Named styles, so the document can be restyled
The export writes a styles part whose `Normal` carries the document's own body text —
From 3a568421e073bfb2066ce445cce73227238fcdfb Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Tue, 22 Sep 2026 09:03:22 +0100
Subject: [PATCH 06/10] fix(render-docx): name the font family Word has, not
the face it does not
FontName.HELVETICA_BOLD is a face, and it was written where Word expects
a family. Word resolves a family and takes the weight from w:b, so asked
for a family by that name it finds none and substitutes -- which is how
a document naming its headings by face came out set in something else.
It matters more now that faces are shipped: a family named wrong is a
font that sits in the package and never gets used.
The face resolves to its family through FontLibrary.resolveFamily, the
same call the layout makes, and the name written is that family's
wordFamily() -- which is what that field is for, so a registered family
can carry a Word name that differs from its logical one.
The weight is deliberately not taken from the face name. The engine does
not take it either: a style naming HELVETICA_BOLD with no decoration
lays out regular, measured in the reference render, where the headings
of the probe are Helvetica at 21 and 13 points and not bold at all.
Writing w:b here would make Word bolder than the page it matches.
Two names for one family now weigh as one style when Normal is chosen.
They are written identically, so weighing them apart could split one
body style in two and elect the lighter half.
Measured through Word 16.0: no substituted face remains under visible
text -- ArialMT is left only under spaces, and Courier maps to Courier
New, which is Word's own pairing. Cells over 25% fall from 33 to 32.
Tests: a face name written as its family, that it does not make the run
bold, that a decoration still does, and that two names for one family
leave the Normal style saying it once. The test pinning the old spelling
now pins the new behaviour: a heading states its size and stays silent
about a family it shares with the body.
---
.../backend/semantic/docx/DocxFontTable.java | 40 ++++--
.../semantic/docx/DocxSemanticBackend.java | 55 +++++++-
.../semantic/docx/DocxDocumentStyleTest.java | 10 +-
.../semantic/docx/DocxFontNamingTest.java | 118 ++++++++++++++++++
4 files changed, 204 insertions(+), 19 deletions(-)
create mode 100644 render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontNamingTest.java
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
index 944b2d9e0..77f9d9387 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
@@ -13,6 +13,7 @@
import com.demcha.compose.document.table.DocumentTableStyle;
import com.demcha.compose.font.DefaultFonts;
import com.demcha.compose.font.FontFamilyDefinition;
+import com.demcha.compose.font.FontLibrary;
import com.demcha.compose.font.FontName;
import org.apache.poi.openxml4j.opc.OPCPackage;
import org.apache.poi.openxml4j.opc.PackagePart;
@@ -115,6 +116,28 @@ static void write(XWPFDocument document,
}
}
+ /**
+ * Every family an export can name, by the logical name a style asks for.
+ *
+ * A family the session registered wins over a bundled one of the same name: it was
+ * registered to be used, and the document was laid out with it.
+ *
+ * @param custom families the session registered, possibly null
+ * @return the families, bundled ones first
+ */
+ static Map familiesByName(Collection custom) {
+ Map families = new LinkedHashMap<>();
+ for (FontFamilyDefinition family : DefaultFonts.bundledFamilies()) {
+ families.put(family.name(), family);
+ }
+ if (custom != null) {
+ for (FontFamilyDefinition family : custom) {
+ families.put(family.name(), family);
+ }
+ }
+ return families;
+ }
+
/** One family that will be written, and the faces of it that may be. */
private record Embedded(String wordFamily, List faces) {
}
@@ -148,8 +171,9 @@ void relationshipId(String id) {
/**
* Reads every face the document could be set in, and keeps the ones it may ship.
*
- * A family the session registered wins over a bundled one of the same name: it was
- * registered to be used, and the document was laid out with it.
+ * A face name resolves to its family first: a style naming {@code Helvetica-Bold}
+ * asks for the Helvetica family, which is a name rather than a file and ships
+ * nothing.
*/
private static List resolve(DocumentGraph graph, Collection custom) {
Map> used = new LinkedHashMap<>();
@@ -160,19 +184,11 @@ private static List resolve(DocumentGraph graph, Collection families = new LinkedHashMap<>();
- for (FontFamilyDefinition family : DefaultFonts.bundledFamilies()) {
- families.put(family.name(), family);
- }
- if (custom != null) {
- for (FontFamilyDefinition family : custom) {
- families.put(family.name(), family);
- }
- }
+ Map families = familiesByName(custom);
List embedded = new ArrayList<>();
for (Map.Entry> entry : used.entrySet()) {
- FontFamilyDefinition family = families.get(entry.getKey());
+ FontFamilyDefinition family = families.get(FontLibrary.resolveFamily(entry.getKey()));
if (family == null || family.fontSourceSet().isEmpty()) {
// A standard-14 name, or one nothing registered: there is no file to ship.
continue;
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
index eb8a587ac..562eda0f1 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
@@ -39,6 +39,8 @@
import com.demcha.compose.document.style.DocumentTextStyle;
import com.demcha.compose.document.table.DocumentTableCell;
import com.demcha.compose.document.table.DocumentTableStyle;
+import com.demcha.compose.font.FontFamilyDefinition;
+import com.demcha.compose.font.FontLibrary;
import com.demcha.compose.font.FontName;
import org.apache.poi.util.Units;
import org.apache.poi.xwpf.usermodel.BreakType;
@@ -167,6 +169,9 @@ public final class DocxSemanticBackend implements SemanticBackend {
// The last paragraph written into the body, so a container can hand it the space it
// holds below itself once its children are done.
private XWPFParagraph lastBodyParagraph;
+ // Every family this export can name, by the logical name a style asks for. The
+ // session's own registrations win over the bundled ones, the way they do everywhere.
+ private java.util.Map wordFamilies = java.util.Map.of();
/**
* A container's paint, reduced to what a Word paragraph can carry.
@@ -221,6 +226,7 @@ public byte[] export(DocumentGraph graph, SemanticExportContext context) throws
warnedNodeKinds.clear();
containerPaint.clear();
listNumbering.clear();
+ wordFamilies = DocxFontTable.familiesByName(context.customFontFamilies());
documentDefaultStyle = dominantTextStyle(graph);
layout = DocxLayoutMetrics.of(graph, context.layoutGraph());
carriedSpacingBefore = 0;
@@ -887,14 +893,14 @@ private void writeStylesPart(XWPFDocument document) {
document.createStyles().setStyles(styles);
}
- private static void applyDefaultRunProperties(CTRPr properties, DocumentTextStyle defaults) {
+ private void applyDefaultRunProperties(CTRPr properties, DocumentTextStyle defaults) {
if (defaults.fontName() != null) {
// All four slots, exactly as XWPFRun.setFontFamily writes them on a run.
// w:ascii alone covers only ASCII: High-ANSI characters read w:hAnsi, Hebrew
// and Arabic read w:cs, CJK reads w:eastAsia. Naming one and suppressing the
// run's own rFonts would send every accented letter and every complex script
// to Word's theme font while the rest of the line kept the asked-for family.
- String family = defaults.fontName().name();
+ String family = wordFamilyOf(defaults.fontName());
CTFonts fonts = properties.addNewRFonts();
fonts.setAscii(family);
fonts.setHAnsi(family);
@@ -912,6 +918,38 @@ private static void applyDefaultRunProperties(CTRPr properties, DocumentTextStyl
}
}
+ /**
+ * The family name to write for a style's font, as Word understands families.
+ *
+ * A {@link FontName} can name a face rather than a family — {@code Helvetica-Bold}
+ * is one — and the two are not interchangeable here. Word resolves a family and takes
+ * the weight from {@code w:b}; asked for a family called "Helvetica-Bold" it finds
+ * none and substitutes, which is how a document that named its headings by face came
+ * out set in something else entirely.
+ *
+ * The face is resolved to its family exactly as the layout resolves it, through
+ * {@link FontLibrary#resolveFamily(FontName)}, so both renders are set in the same
+ * family. The weight is deliberately not taken from the face name: the engine
+ * does not take it either — a style naming {@code Helvetica-Bold} with no decoration
+ * lays out regular — and writing {@code w:b} here would make Word bolder than the page
+ * it is meant to match.
+ *
+ * The name itself comes from the family's own {@code wordFamily()}, which is what
+ * that field is for, so a registered family can carry a Word name that differs from
+ * its logical one.
+ *
+ * @param fontName the style's font, possibly null or a face alias
+ * @return the family name to write, or null when the style named no font
+ */
+ private String wordFamilyOf(FontName fontName) {
+ if (fontName == null) {
+ return null;
+ }
+ FontName family = FontLibrary.resolveFamily(fontName);
+ FontFamilyDefinition definition = wordFamilies.get(family);
+ return definition == null ? family.name() : definition.wordFamily();
+ }
+
/**
* The text style the document is mostly written in.
*
@@ -979,7 +1017,11 @@ private static void weigh(DocumentTextStyle style,
private record StyleKey(FontName fontName, long halfPoints, Integer colour) {
static StyleKey of(DocumentTextStyle style) {
- return new StyleKey(style.fontName(),
+ // By family, not by the name the style used: Helvetica and Helvetica-Bold are
+ // written identically — the second resolves to the first and takes its weight
+ // from the decoration — so weighing them apart would split one body style in
+ // two and could elect the lighter half as Normal.
+ return new StyleKey(FontLibrary.resolveFamily(style.fontName()),
Math.round(style.size() * HALF_POINTS_PER_POINT),
style.color() == null ? null : style.color().color().getRGB());
}
@@ -2262,9 +2304,10 @@ private void applyStyle(XWPFRun run, DocumentTextStyle style) {
// the style and a global restyle silently does nothing — which is what this
// exporter used to produce for every run in every document.
DocumentTextStyle defaults = documentDefaultStyle;
- if (style.fontName() != null
- && (defaults == null || !style.fontName().equals(defaults.fontName()))) {
- run.setFontFamily(style.fontName().name());
+ String family = wordFamilyOf(style.fontName());
+ if (family != null
+ && (defaults == null || !family.equals(wordFamilyOf(defaults.fontName())))) {
+ run.setFontFamily(family);
}
// Complex-script size rides along with the ordinary one, so it is skipped for the
// same reason when the style already carries it.
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxDocumentStyleTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxDocumentStyleTest.java
index 2d9127388..93b141d8e 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxDocumentStyleTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxDocumentStyleTest.java
@@ -82,7 +82,15 @@ void aRunThatDiffersShouldKeepSayingSo() throws Exception {
assertThat(heading).isNotNull();
// 18pt in half-points. A heading must not be swallowed by the body style.
assertThat(heading.getSzArray(0).getVal().toString()).isEqualTo("36");
- assertThat(heading.getRFontsArray(0).getAscii()).isEqualTo("Helvetica-Bold");
+ // The font is where it stops differing. This heading names the face
+ // Helvetica-Bold and the body names Helvetica, but a face resolves to its
+ // family and takes its weight from the decoration — neither of these carries
+ // one, so the page draws both in Helvetica regular at different sizes. Writing
+ // the face name made the run look different in the file while being identical
+ // on the page, and sent Word looking for a family it does not have.
+ assertThat(heading.sizeOfRFontsArray())
+ .as("the same family as the body, so the Normal style already says it")
+ .isZero();
}
}
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontNamingTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontNamingTest.java
new file mode 100644
index 000000000..c26dc0894
--- /dev/null
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxFontNamingTest.java
@@ -0,0 +1,118 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.document.dsl.PageFlowBuilder;
+import com.demcha.compose.document.style.DocumentTextDecoration;
+import com.demcha.compose.document.style.DocumentTextStyle;
+import com.demcha.compose.font.FontName;
+import org.apache.poi.xwpf.usermodel.XWPFDocument;
+import org.apache.poi.xwpf.usermodel.XWPFRun;
+import org.junit.jupiter.api.Test;
+
+import java.util.function.Consumer;
+
+import static org.assertj.core.api.Assertions.assertThat;
+
+/**
+ * A run names a family Word has, not a face it does not.
+ *
+ * {@code Helvetica-Bold} is a face, and the export wrote it where Word expects a family.
+ * Word resolves a family and takes the weight from {@code w:b}; asked for a family by that
+ * name it finds none and substitutes, which is how a document that named its headings by
+ * face came out set in something else.
+ *
+ * The weight is deliberately not read from the face name. The engine does not read it
+ * either — {@code FontLibrary.resolveFamily} rewrites the face to its family and the face
+ * is chosen from the style's decoration — so a style naming {@code Helvetica-Bold} and
+ * setting no decoration lays the page out in Helvetica regular. Writing {@code w:b} here
+ * would make Word bolder than the page it is meant to match.
+ *
+ * @author Artem Demchyshyn
+ */
+class DocxFontNamingTest {
+
+ /** Long enough to be the body the Normal style is chosen from, by character weight. */
+ private static final String LONG_BODY =
+ "A body paragraph long enough that the style it is set in is the one the "
+ + "document is mostly written in, so a heading beside it has to say what "
+ + "differs about itself.";
+
+ @Test
+ void aFaceNameIsWrittenAsItsFamily() throws Exception {
+ try (XWPFDocument document = exported(page -> page
+ .addParagraph(p -> p.text(LONG_BODY).textStyle(style(FontName.LATO)))
+ .addParagraph(p -> p.text("Heading").textStyle(style(FontName.HELVETICA_BOLD))))) {
+
+ XWPFRun heading = runOf(document, "Heading");
+ assertThat(heading.getFontFamily())
+ .as("the family, not the face")
+ .isEqualTo("Helvetica");
+ }
+ }
+
+ @Test
+ void theFaceNameDoesNotMakeTheRunBold() throws Exception {
+ try (XWPFDocument document = exported(page -> page
+ .addParagraph(p -> p.text(LONG_BODY).textStyle(style(FontName.LATO)))
+ .addParagraph(p -> p.text("Heading").textStyle(style(FontName.HELVETICA_BOLD))))) {
+
+ assertThat(runOf(document, "Heading").isBold())
+ .as("the page draws this regular, so the file must not say bold")
+ .isFalse();
+ }
+ }
+
+ @Test
+ void theDecorationStillDecidesTheWeight() throws Exception {
+ try (XWPFDocument document = exported(page -> page
+ .addParagraph(p -> p.text(LONG_BODY).textStyle(style(FontName.LATO)))
+ .addParagraph(p -> p.text("Strong").textStyle(DocumentTextStyle.builder()
+ .fontName(FontName.HELVETICA)
+ .decoration(DocumentTextDecoration.BOLD)
+ .size(13)
+ .build())))) {
+
+ assertThat(runOf(document, "Strong").isBold())
+ .as("this one asked for bold, and the page draws it bold")
+ .isTrue();
+ }
+ }
+
+ @Test
+ void twoNamesForOneFamilyCountAsOneStyle() throws Exception {
+ // Helvetica and Helvetica-Bold are written identically, so weighing them apart
+ // would split one body style in two and could elect the lighter half as Normal.
+ try (XWPFDocument document = exported(page -> {
+ page.addParagraph(p -> p.text("A long stretch of body text set one way")
+ .textStyle(style(FontName.HELVETICA)));
+ page.addParagraph(p -> p.text("A long stretch of body text set the other")
+ .textStyle(style(FontName.HELVETICA_BOLD)));
+ })) {
+ assertThat(document.getStyles().getCtStyles().getDocDefaults()
+ .getRPrDefault().getRPr().getRFontsArray(0).getAscii())
+ .isEqualTo("Helvetica");
+ for (String text : new String[] {"A long stretch of body text set one way",
+ "A long stretch of body text set the other"}) {
+ assertThat(runOf(document, text).getCTR().getRPr() == null
+ || runOf(document, text).getCTR().getRPr().sizeOfRFontsArray() == 0)
+ .as("neither run repeats a family the Normal style already carries: %s", text)
+ .isTrue();
+ }
+ }
+ }
+
+ private static DocumentTextStyle style(FontName fontName) {
+ return DocumentTextStyle.builder().fontName(fontName).size(13).build();
+ }
+
+ private static XWPFRun runOf(XWPFDocument document, String text) {
+ return document.getParagraphs().stream()
+ .filter(paragraph -> text.equals(paragraph.getText()))
+ .findFirst()
+ .orElseThrow(() -> new AssertionError("no paragraph reading " + text))
+ .getRuns().get(0);
+ }
+
+ private static XWPFDocument exported(Consumer content) throws Exception {
+ return DocxExports.withLayout(400, 600, 20, content);
+ }
+}
From 190ca865ef7f718462f448531ffa72c08d18ace8 Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Tue, 22 Sep 2026 09:03:28 +0100
Subject: [PATCH 07/10] docs(render-docx): record that a run names a family,
not a face
---
docs/recipes/docx-export.md | 9 +++++++++
1 file changed, 9 insertions(+)
diff --git a/docs/recipes/docx-export.md b/docs/recipes/docx-export.md
index bd8f73755..f49f14e1c 100644
--- a/docs/recipes/docx-export.md
+++ b/docs/recipes/docx-export.md
@@ -94,6 +94,15 @@ sees the document rather than a substitution. What is shipped is narrow on purpo
Each face is stored the way Word stores one: the font with its first 32 bytes scrambled
against a key the font table states beside it.
+A run names the **family**, not the face. `FontName.HELVETICA_BOLD` is a face, and Word
+resolves families and takes the weight from `w:b`; asked for a family by that name it finds
+none and substitutes. The face is resolved to its family exactly as the layout resolves it,
+through `FontLibrary.resolveFamily`, and the name written is that family's `wordFamily()`.
+The weight is not read from the face name, because the engine does not read it either — a
+style naming `HELVETICA_BOLD` and setting no `decoration` lays out regular, so writing
+`w:b` would make Word bolder than the page it is matching. Set `decoration(BOLD)` to get
+bold in both.
+
## Named styles, so the document can be restyled
The export writes a styles part whose `Normal` carries the document's own body text —
From deedf9274bfbdc034ca72e91e247b7bf759dd020 Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Tue, 22 Sep 2026 09:20:23 +0100
Subject: [PATCH 08/10] feat(api): let a session export Word without naming the
backend
Reaching the Word export meant constructing DocxSemanticBackend, which
means importing the render artifact in the code that builds the
document and carrying that dependency wherever documents are built. A
render backend has not needed that since 2.0 -- the artifact on the
classpath is enough -- and now neither does this one:
session.buildDocx(path);
session.writeDocx(outputStream);
byte[] bytes = session.toDocxBytes();
SemanticBackendProvider and SemanticBackendProviders are the semantic
half of that ServiceLoader path, with the fixed-layout locator's rules,
because a classpath behaves the same way whichever backend is on it: the
format is a case-insensitive key, a missing provider names the artifact
to add, and two providers for one format is reported rather than
resolved -- leaving enumeration order to decide would export one
document differently on two machines.
One rule is deliberately not copied. There is no default-format lookup:
"render this document" has an obvious answer worth defaulting to, and
"export this document semantically" does not, so choosing between two
installed formats would be picking a file type for the caller.
The stream is the caller's and is not closed. The bytes are produced in
full before any are written, because a .docx is a ZIP whose directory
comes last, so a half-written one is unreadable rather than short. A
file is written through the same atomic path as buildPdf, so a failed
export leaves the previous file rather than a damaged one.
Tests: the provider found by either spelling, a fresh backend per
export, bytes/stream/file producing the same document, the stream left
open -- verified by sabotage -- a failed export leaving the old file,
and the locator's own choices: case, missing, duplicate, and a provider
declaring no format at all.
---
.../document/api/DocumentRenderingFacade.java | 60 +++++++
.../compose/document/api/DocumentSession.java | 78 ++++++++-
.../semantic/SemanticBackendProvider.java | 47 ++++++
.../semantic/SemanticBackendProviders.java | 126 +++++++++++++++
.../SemanticBackendProvidersTest.java | 91 +++++++++++
docs/api-stability.md | 11 ++
knowledge/api/authoring.json | 42 ++++-
knowledge/api/authoring.md | 5 +-
knowledge/api/backends.json | 110 ++++++++++++-
knowledge/api/backends.md | 15 +-
.../semantic/docx/DocxBackendProvider.java | 39 +++++
...t.backend.semantic.SemanticBackendProvider | 1 +
.../semantic/docx/DocxSessionExportTest.java | 149 ++++++++++++++++++
13 files changed, 767 insertions(+), 7 deletions(-)
create mode 100644 core/src/main/java/com/demcha/compose/document/backend/semantic/SemanticBackendProvider.java
create mode 100644 core/src/main/java/com/demcha/compose/document/backend/semantic/SemanticBackendProviders.java
create mode 100644 core/src/test/java/com/demcha/compose/document/backend/semantic/SemanticBackendProvidersTest.java
create mode 100644 render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxBackendProvider.java
create mode 100644 render-docx/src/main/resources/META-INF/services/com.demcha.compose.document.backend.semantic.SemanticBackendProvider
create mode 100644 render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionExportTest.java
diff --git a/core/src/main/java/com/demcha/compose/document/api/DocumentRenderingFacade.java b/core/src/main/java/com/demcha/compose/document/api/DocumentRenderingFacade.java
index 1c4f40f43..4313eb03d 100644
--- a/core/src/main/java/com/demcha/compose/document/api/DocumentRenderingFacade.java
+++ b/core/src/main/java/com/demcha/compose/document/api/DocumentRenderingFacade.java
@@ -4,6 +4,7 @@
import com.demcha.compose.document.backend.fixed.FixedLayoutRenderContext;
import com.demcha.compose.document.backend.fixed.FixedLayoutRenderer;
import com.demcha.compose.document.backend.semantic.SemanticBackend;
+import com.demcha.compose.document.backend.semantic.SemanticBackendProviders;
import com.demcha.compose.document.backend.semantic.SemanticExportContext;
import com.demcha.compose.document.layout.DocumentGraph;
import com.demcha.compose.document.layout.LayoutCanvas;
@@ -148,6 +149,65 @@ void buildPdf(Path outputFile) throws Exception {
buildFixedLayout(PDF, outputFile);
}
+ /**
+ * Exports through the semantic backend registered for {@code format}.
+ *
+ * {@code ensureRenderable()} is deliberately not called: a semantic export is
+ * defined over the authored tree, and a document the fixed-layout pipeline refuses can
+ * still export from it. Refusing here would take that away for no gain.
+ */
+ byte[] toSemanticBytes(String format) throws Exception {
+ context.ensureOpen();
+ long startNanos = System.nanoTime();
+ LIFECYCLE_LOG.debug("document.{}.bytes.start sessionId={} revision={} roots={}",
+ format, context.sessionId(), context.revision(), context.rootCount());
+ try {
+ byte[] bytes = export(SemanticBackendProviders.forFormat(format).create(), null);
+ LIFECYCLE_LOG.debug(
+ "document.{}.bytes.end sessionId={} revision={} byteCount={} durationMs={}",
+ format, context.sessionId(), context.revision(), bytes.length,
+ elapsedMillis(startNanos));
+ return bytes;
+ } catch (Exception ex) {
+ LIFECYCLE_LOG.error("document.{}.bytes.failed sessionId={} revision={} errorType={}",
+ format, context.sessionId(), context.revision(),
+ ex.getClass().getSimpleName(), ex);
+ throw ex;
+ }
+ }
+
+ /**
+ * Writes the export to a stream the caller owns and keeps open.
+ *
+ * The bytes are produced in full before any of them are written. The format is a ZIP
+ * package whose directory is written last, so an export that fails part-way would
+ * otherwise leave a truncated archive on a stream nobody can rewind.
+ */
+ void writeSemantic(String format, OutputStream output) throws Exception {
+ Objects.requireNonNull(output, "output");
+ output.write(toSemanticBytes(format));
+ output.flush();
+ }
+
+ /** Writes the export to a file, replacing it only once the whole export succeeded. */
+ void buildSemantic(String format, Path outputFile) throws Exception {
+ Path target = Objects.requireNonNull(outputFile, "outputFile");
+ long startNanos = System.nanoTime();
+ LIFECYCLE_LOG.debug("document.{}.build.start sessionId={} revision={} roots={}",
+ format, context.sessionId(), context.revision(), context.rootCount());
+ try {
+ byte[] bytes = toSemanticBytes(format);
+ AtomicFileOutput.write(target, output -> output.write(bytes));
+ LIFECYCLE_LOG.debug("document.{}.build.end sessionId={} revision={} durationMs={}",
+ format, context.sessionId(), context.revision(), elapsedMillis(startNanos));
+ } catch (Exception ex) {
+ LIFECYCLE_LOG.error("document.{}.build.failed sessionId={} revision={} errorType={}",
+ format, context.sessionId(), context.revision(),
+ ex.getClass().getSimpleName(), ex);
+ throw ex;
+ }
+ }
+
byte[] toPptxBytes() throws Exception {
return renderBytes(PPTX);
}
diff --git a/core/src/main/java/com/demcha/compose/document/api/DocumentSession.java b/core/src/main/java/com/demcha/compose/document/api/DocumentSession.java
index af8586f26..cb847115a 100644
--- a/core/src/main/java/com/demcha/compose/document/api/DocumentSession.java
+++ b/core/src/main/java/com/demcha/compose/document/api/DocumentSession.java
@@ -56,7 +56,10 @@
* inspect {@link #layoutGraph()} / {@link #layoutSnapshot()} as needed
* render with {@link #writePdf(OutputStream)}, {@link #toPdfBytes()}, {@link #buildPdf()},
* their PPTX counterparts ({@link #writePptx(OutputStream)}, {@link #toPptxBytes()},
- * {@link #buildPptx(Path)}), or a custom backend
+ * {@link #buildPptx(Path)}), or a custom backend — or export an editable Word
+ * document with {@link #buildDocx(Path)}, {@link #writeDocx(OutputStream)} and
+ * {@link #toDocxBytes()}, which write the document's structure rather than its
+ * pixels and let Word lay it out
*
*
* Thread-safety: this type is mutable and not thread-safe.
@@ -65,6 +68,9 @@
* @since 1.0.0
*/
public final class DocumentSession implements AutoCloseable {
+
+ /** Provider format key for the Word export path. */
+ private static final String DOCX = "docx";
private static final Logger LIFECYCLE_LOG = LoggerFactory.getLogger("com.demcha.compose.document.lifecycle");
private final String sessionId = Integer.toHexString(System.identityHashCode(this));
@@ -1054,6 +1060,76 @@ public void buildPdf(Path outputFile) throws DocumentRenderingException {
});
}
+ /**
+ * Exports the current session as an editable Word document and returns the bytes.
+ *
+ * Unlike a PDF render, this is an export: the document's structure is written as
+ * Word's own paragraphs, tables and lists, and Word lays the result out itself. A
+ * reader can edit it the way they edit any document — lengthen a sentence and the
+ * paragraph reflows, insert a table row and the table grows. What the export cannot
+ * carry, it says so about rather than approximating; the
+ * DOCX
+ * recipe lists what maps, what falls back and what is deliberately left out.
+ *
+ * Requires {@code io.github.demchaav:graph-compose-render-docx} on the classpath;
+ * without it the export fails with a {@link com.demcha.compose.document.exceptions.MissingBackendException} naming the
+ * artifact. The returned array is not cached by the session, so code that can stream
+ * should prefer {@link #writeDocx(OutputStream)}.
+ *
+ * Experimental ({@code @Beta}) — see {@code docs/api-stability.md}.
+ *
+ * @return the exported .docx bytes
+ * @throws DocumentRenderingException if the export fails
+ * @since 2.5.0
+ */
+ @Beta
+ public byte[] toDocxBytes() throws DocumentRenderingException {
+ return wrapRendering("export DOCX bytes", () -> renderingFacade.toSemanticBytes(DOCX));
+ }
+
+ /**
+ * Streams the current session's Word export to a stream the caller owns.
+ *
+ * GraphCompose writes the bytes and does not close the stream, which makes this
+ * suitable for an HTTP response or an upload. The export is produced in full before
+ * anything is written: a {@code .docx} is a ZIP whose directory comes last, so a
+ * half-written one is not a shorter document but an unreadable file.
+ *
+ * Experimental ({@code @Beta}) — see {@code docs/api-stability.md}.
+ *
+ * @param output destination stream that receives the exported bytes
+ * @throws DocumentRenderingException if the export fails
+ * @since 2.5.0
+ */
+ @Beta
+ public void writeDocx(OutputStream output) throws DocumentRenderingException {
+ wrapRendering("write DOCX to stream", () -> {
+ renderingFacade.writeSemantic(DOCX, output);
+ return null;
+ });
+ }
+
+ /**
+ * Exports the current session into the supplied file.
+ *
+ * Written through the same atomic path as {@link #buildPdf(Path)}: an export that
+ * fails leaves whatever was there before, rather than a partial file that opens as a
+ * damaged document.
+ *
+ * Experimental ({@code @Beta}) — see {@code docs/api-stability.md}.
+ *
+ * @param outputFile destination .docx path
+ * @throws DocumentRenderingException if the export fails
+ * @since 2.5.0
+ */
+ @Beta
+ public void buildDocx(Path outputFile) throws DocumentRenderingException {
+ wrapRendering("build DOCX at '" + outputFile + "'", () -> {
+ renderingFacade.buildSemantic(DOCX, outputFile);
+ return null;
+ });
+ }
+
/**
* Renders the current session through the fixed-layout PPTX backend and
* returns the .pptx bytes. One resolved page becomes one identically-sized
diff --git a/core/src/main/java/com/demcha/compose/document/backend/semantic/SemanticBackendProvider.java b/core/src/main/java/com/demcha/compose/document/backend/semantic/SemanticBackendProvider.java
new file mode 100644
index 000000000..846380e95
--- /dev/null
+++ b/core/src/main/java/com/demcha/compose/document/backend/semantic/SemanticBackendProvider.java
@@ -0,0 +1,47 @@
+package com.demcha.compose.document.backend.semantic;
+
+import com.demcha.compose.document.api.Beta;
+
+/**
+ * Service-provider that supplies a semantic export backend to the document API's
+ * convenience output methods, without the caller naming a concrete backend type.
+ *
+ * The fixed-layout half of this has existed since 2.0: a render backend artifact
+ * registers a {@code FixedLayoutBackendProvider} and the session's {@code buildPdf} finds
+ * it. A semantic backend had no such path — the only way to reach one was to construct it,
+ * which means naming its artifact in code and carrying the dependency everywhere the
+ * document is built. This is the missing half, and deliberately no more than that: the
+ * fixed-layout locator keeps its own contracts and is not reorganised around this one.
+ *
+ * Implementations are discovered through {@link java.util.ServiceLoader} and resolved by
+ * {@link SemanticBackendProviders#forFormat(String)}, which matches {@link #format()}
+ * case-insensitively and refuses a classpath carrying two providers for one format.
+ *
+ * Experimental ({@code @Beta}) — see {@code docs/api-stability.md}.
+ *
+ * @author Artem Demchyshyn
+ * @since 2.5.0
+ */
+@Beta
+public interface SemanticBackendProvider {
+
+ /**
+ * The output format this provider exports, as the selection key.
+ *
+ * Matched case-insensitively, so a provider may spell it however reads best.
+ *
+ * @return a stable format identifier such as {@code "docx"}
+ */
+ String format();
+
+ /**
+ * Creates a backend for one export.
+ *
+ * Called once per export rather than cached, because a semantic backend holds the
+ * state of the export it is running. A provider that returned a shared instance would
+ * make two concurrent exports write into each other.
+ *
+ * @return a configured backend producing the format's bytes
+ */
+ SemanticBackend create();
+}
diff --git a/core/src/main/java/com/demcha/compose/document/backend/semantic/SemanticBackendProviders.java b/core/src/main/java/com/demcha/compose/document/backend/semantic/SemanticBackendProviders.java
new file mode 100644
index 000000000..7a89f89c6
--- /dev/null
+++ b/core/src/main/java/com/demcha/compose/document/backend/semantic/SemanticBackendProviders.java
@@ -0,0 +1,126 @@
+package com.demcha.compose.document.backend.semantic;
+
+import com.demcha.compose.document.api.Beta;
+import com.demcha.compose.document.exceptions.MissingBackendException;
+
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Locale;
+import java.util.Map;
+import java.util.Objects;
+import java.util.ServiceLoader;
+import java.util.concurrent.ConcurrentHashMap;
+import java.util.stream.Collectors;
+
+/**
+ * Locates the semantic export backend registered for one output format.
+ *
+ * The rules are the fixed-layout locator's, because a classpath behaves the same way
+ * whichever kind of backend is on it: the format is matched case-insensitively, a missing
+ * provider names the artifact to add rather than failing as a null, and two providers
+ * claiming one format is a mistake to report rather than a preference to resolve — leaving
+ * {@link ServiceLoader} enumeration order to pick the backend would make the same document
+ * export differently on two machines.
+ *
+ * There is no default-format lookup here, and that is the difference from the
+ * fixed-layout side. "Render this document" has an obvious answer worth defaulting to;
+ * "export this document semantically" does not, and guessing between two installed formats
+ * would be picking a file type on the caller's behalf.
+ *
+ * Experimental ({@code @Beta}) — see {@code docs/api-stability.md}.
+ *
+ * @author Artem Demchyshyn
+ * @since 2.5.0
+ */
+@Beta
+public final class SemanticBackendProviders {
+
+ /** The artifact that provides each format the project itself publishes. */
+ private static final Map KNOWN_ARTIFACTS =
+ Map.of("docx", "io.github.demchaav:graph-compose-render-docx");
+
+ private static final Map byFormat = new ConcurrentHashMap<>();
+
+ private SemanticBackendProviders() {
+ }
+
+ /**
+ * Returns the provider exporting {@code format}, resolving and caching it on first use.
+ *
+ * @param format output format identifier such as {@code "docx"}
+ * @return the provider exporting that format
+ * @throws MissingBackendException if no provider for the format is on the classpath
+ * @throws IllegalStateException if more than one provider declares the format
+ */
+ public static SemanticBackendProvider forFormat(String format) {
+ Objects.requireNonNull(format, "format");
+ String key = format.toLowerCase(Locale.ROOT);
+ SemanticBackendProvider cached = byFormat.get(key);
+ if (cached != null) {
+ return cached;
+ }
+ SemanticBackendProvider resolved = select(key, load());
+ byFormat.put(key, resolved);
+ return resolved;
+ }
+
+ /**
+ * Chooses the one provider declaring {@code format}, or says why it cannot.
+ *
+ * Separated from the classpath lookup and left package-private so the choice can be
+ * tested against a stated list of providers. Registering a second provider through
+ * {@link ServiceLoader} inside a test would mean writing a services file that every
+ * other test in the module then shares.
+ *
+ * @param format the lower-cased format being looked for
+ * @param candidates every provider on the classpath
+ * @return the single provider declaring that format
+ */
+ static SemanticBackendProvider select(String format, List candidates) {
+ List matches = candidates.stream()
+ .filter(provider -> format.equals(normalized(provider)))
+ .toList();
+ if (matches.isEmpty()) {
+ throw new MissingBackendException(missingMessage(format));
+ }
+ if (matches.size() > 1) {
+ String names = matches.stream()
+ .map(provider -> provider.getClass().getName())
+ .collect(Collectors.joining(", "));
+ throw new IllegalStateException(
+ "Multiple semantic export backends are registered for format \"" + format
+ + "\": " + names + ". Remove the duplicate artifact from the classpath, or "
+ + "pass a backend to export(...) instead of resolving it by format.");
+ }
+ return matches.get(0);
+ }
+
+ private static List load() {
+ List providers = new ArrayList<>();
+ ServiceLoader.load(SemanticBackendProvider.class).forEach(providers::add);
+ if (providers.isEmpty()) {
+ // The thread context loader is what a container hands an application, and the
+ // service file lives with the backend artifact rather than with the core.
+ ClassLoader context = Thread.currentThread().getContextClassLoader();
+ if (context != null) {
+ ServiceLoader.load(SemanticBackendProvider.class, context).forEach(providers::add);
+ }
+ }
+ return providers;
+ }
+
+ private static String normalized(SemanticBackendProvider provider) {
+ String format = provider.format();
+ return format == null ? "" : format.toLowerCase(Locale.ROOT);
+ }
+
+ private static String missingMessage(String format) {
+ String artifact = KNOWN_ARTIFACTS.get(format);
+ return "No semantic export backend on the classpath for format \"" + format + "\": add "
+ + (artifact == null
+ ? "the artifact providing it"
+ : "the " + artifact + " artifact (or the "
+ + "io.github.demchaav:graph-compose-bundle aggregate)")
+ + ", or pass a backend to export(...) directly.";
+ }
+}
diff --git a/core/src/test/java/com/demcha/compose/document/backend/semantic/SemanticBackendProvidersTest.java b/core/src/test/java/com/demcha/compose/document/backend/semantic/SemanticBackendProvidersTest.java
new file mode 100644
index 000000000..adeb3b71c
--- /dev/null
+++ b/core/src/test/java/com/demcha/compose/document/backend/semantic/SemanticBackendProvidersTest.java
@@ -0,0 +1,91 @@
+package com.demcha.compose.document.backend.semantic;
+
+import com.demcha.compose.document.exceptions.MissingBackendException;
+import com.demcha.compose.document.layout.DocumentGraph;
+import org.junit.jupiter.api.Test;
+
+import java.util.List;
+
+import static org.assertj.core.api.Assertions.assertThat;
+import static org.assertj.core.api.Assertions.assertThatThrownBy;
+
+/**
+ * Choosing one semantic export backend out of what a classpath happens to carry.
+ *
+ * The rules are the fixed-layout locator's, for the reason they were chosen there: a
+ * classpath is assembled by a build rather than by the person exporting, so a format
+ * nobody provides has to name the artifact to add, and two providers for one format has to
+ * fail rather than let {@link java.util.ServiceLoader} order decide — otherwise one
+ * document exports differently on two machines and nothing says why.
+ *
+ * @author Artem Demchyshyn
+ */
+class SemanticBackendProvidersTest {
+
+ @Test
+ void theFormatIsAKeyRatherThanASpelling() {
+ SemanticBackendProvider docx = provider("DocX");
+
+ assertThat(SemanticBackendProviders.select("docx", List.of(docx))).isSameAs(docx);
+ }
+
+ @Test
+ void aFormatNobodyProvidesNamesTheArtifactToAdd() {
+ assertThatThrownBy(() -> SemanticBackendProviders.select("docx", List.of(provider("rtf"))))
+ .isInstanceOf(MissingBackendException.class)
+ .hasMessageContaining("graph-compose-render-docx")
+ .as("and says what to do instead of naming an artifact")
+ .hasMessageContaining("export(...)");
+ }
+
+ @Test
+ void anUnknownFormatStillSaysWhatIsMissing() {
+ // Nothing here knows which artifact would provide "odt", and saying nothing at all
+ // would leave a caller with a null and no idea why.
+ assertThatThrownBy(() -> SemanticBackendProviders.select("odt", List.of()))
+ .isInstanceOf(MissingBackendException.class)
+ .hasMessageContaining("odt")
+ .hasMessageContaining("classpath");
+ }
+
+ @Test
+ void twoProvidersForOneFormatIsReportedRatherThanResolved() {
+ assertThatThrownBy(() -> SemanticBackendProviders.select(
+ "docx", List.of(provider("docx"), provider("docx"))))
+ .isInstanceOf(IllegalStateException.class)
+ .hasMessageContaining("Multiple semantic export backends")
+ .hasMessageContaining("docx");
+ }
+
+ @Test
+ void aProviderDeclaringNoFormatIsSkippedRatherThanMatchingEverything() {
+ SemanticBackendProvider docx = provider("docx");
+
+ assertThat(SemanticBackendProviders.select("docx", List.of(provider(null), docx)))
+ .isSameAs(docx);
+ }
+
+ private static SemanticBackendProvider provider(String format) {
+ return new SemanticBackendProvider() {
+ @Override
+ public String format() {
+ return format;
+ }
+
+ @Override
+ public SemanticBackend create() {
+ return new SemanticBackend<>() {
+ @Override
+ public String name() {
+ return "probe";
+ }
+
+ @Override
+ public byte[] export(DocumentGraph graph, SemanticExportContext context) {
+ return new byte[0];
+ }
+ };
+ }
+ };
+ }
+}
diff --git a/docs/api-stability.md b/docs/api-stability.md
index 46a92b7a7..78aa5cdb6 100644
--- a/docs/api-stability.md
+++ b/docs/api-stability.md
@@ -66,6 +66,17 @@ matrix.
> a backend actually needs instead of the whole graph. Backends that do not ask are
> unaffected, and the published `SemanticExportContext` constructors keep working.
>
+> **Semantic backend discovery** is Experimental in 2.5.0: the `SemanticBackendProvider`
+> and `SemanticBackendProviders` types, and the DOCX convenience methods on
+> `DocumentSession` (`toDocxBytes`, `writeDocx`, `buildDocx`). They are the semantic
+> half of the `ServiceLoader` path the fixed-layout backends have had since 2.0, so an
+> artifact on the classpath is enough to export — no naming the backend in code. The
+> shape is marked Experimental because it is deliberately narrower than the
+> fixed-layout locator (no default-format lookup, no per-backend configuration through
+> the provider) and a later minor may widen it once a second semantic format exists to
+> generalise against. Constructing `DocxSemanticBackend` and calling
+> `session.export(backend, path)` stays the Stable path and is unaffected.
+>
> Seven members of the otherwise-Stable **PDF backend** also carry `@Beta`. The
> package is not Experimental — these are:
> `PdfFixedLayoutBackend.renderSections` / `writeSections`, the low-level seam
diff --git a/knowledge/api/authoring.json b/knowledge/api/authoring.json
index a8b5cb3cb..eea247433 100644
--- a/knowledge/api/authoring.json
+++ b/knowledge/api/authoring.json
@@ -23,7 +23,7 @@
],
"counts": {
"types": 237,
- "methods": 2125,
+ "methods": 2128,
"constants": 236,
"generated": 1129
},
@@ -1195,6 +1195,46 @@
}
]
},
+ {
+ "kind": "method",
+ "name": "toDocxBytes",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": "byte[]",
+ "params": [],
+ "stability": "beta"
+ },
+ {
+ "kind": "method",
+ "name": "writeDocx",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": "void",
+ "params": [
+ {
+ "type": "OutputStream",
+ "name": "output"
+ }
+ ],
+ "stability": "beta"
+ },
+ {
+ "kind": "method",
+ "name": "buildDocx",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": "void",
+ "params": [
+ {
+ "type": "Path",
+ "name": "outputFile"
+ }
+ ],
+ "stability": "beta"
+ },
{
"kind": "method",
"name": "toPptxBytes",
diff --git a/knowledge/api/authoring.md b/knowledge/api/authoring.md
index b8d600c4f..dcb917c12 100644
--- a/knowledge/api/authoring.md
+++ b/knowledge/api/authoring.md
@@ -28,7 +28,7 @@ note: "Generated from the pinned artifact's class files. Authoritative closed se
**GraphCompose version:** 2.5.0-SNAPSHOT
-Types: 237 · methods: 2125 · constants: 236 · compiler-generated members: 1129
+Types: 237 · methods: 2128 · constants: 236 · compiler-generated members: 1129
## com.demcha.compose
@@ -120,6 +120,9 @@ Types: 237 · methods: 2125 · constants: 236 · compiler-generated members: 112
- `void writePdf(OutputStream output)`
- `void buildPdf()`
- `void buildPdf(Path outputFile)`
+- `byte[] toDocxBytes() [beta]`
+- `void writeDocx(OutputStream output) [beta]`
+- `void buildDocx(Path outputFile) [beta]`
- `byte[] toPptxBytes() [beta]`
- `void writePptx(OutputStream output) [beta]`
- `void buildPptx(Path outputFile) [beta]`
diff --git a/knowledge/api/backends.json b/knowledge/api/backends.json
index 66c573cd4..02e3242fa 100644
--- a/knowledge/api/backends.json
+++ b/knowledge/api/backends.json
@@ -22,10 +22,10 @@
"graph-compose-testing:sources"
],
"counts": {
- "types": 70,
- "methods": 379,
+ "types": 73,
+ "methods": 386,
"constants": 18,
- "generated": 193
+ "generated": 195
},
"packages": [
{
@@ -5957,6 +5957,63 @@
{
"name": "com.demcha.compose.document.backend.semantic",
"types": [
+ {
+ "name": "SemanticBackendProvider",
+ "binaryName": "com.demcha.compose.document.backend.semantic.SemanticBackendProvider",
+ "kind": "interface",
+ "modifiers": [],
+ "artifact": "graph-compose-core",
+ "stability": "beta",
+ "members": [
+ {
+ "kind": "method",
+ "name": "format",
+ "static": false,
+ "origin": "generated",
+ "typeParameters": null,
+ "returns": "String",
+ "params": [],
+ "stability": "beta"
+ },
+ {
+ "kind": "method",
+ "name": "create",
+ "static": false,
+ "origin": "generated",
+ "typeParameters": null,
+ "returns": "SemanticBackend",
+ "params": [],
+ "stability": "beta"
+ }
+ ]
+ },
+ {
+ "name": "SemanticBackendProviders",
+ "binaryName": "com.demcha.compose.document.backend.semantic.SemanticBackendProviders",
+ "kind": "class",
+ "modifiers": [
+ "final"
+ ],
+ "artifact": "graph-compose-core",
+ "stability": "beta",
+ "members": [
+ {
+ "kind": "method",
+ "name": "forFormat",
+ "static": true,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": "SemanticBackendProvider",
+ "params": [
+ {
+ "type": "String",
+ "name": "format"
+ }
+ ],
+ "stability": "beta"
+ }
+ ]
+ },
{
"name": "SemanticExportContext",
"binaryName": "com.demcha.compose.document.backend.semantic.SemanticExportContext",
@@ -6180,6 +6237,44 @@
{
"name": "com.demcha.compose.document.backend.semantic.docx",
"types": [
+ {
+ "name": "DocxBackendProvider",
+ "binaryName": "com.demcha.compose.document.backend.semantic.docx.DocxBackendProvider",
+ "kind": "class",
+ "modifiers": [
+ "final"
+ ],
+ "artifact": "graph-compose-render-docx",
+ "members": [
+ {
+ "kind": "constructor",
+ "name": "DocxBackendProvider",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": null,
+ "params": []
+ },
+ {
+ "kind": "method",
+ "name": "format",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": "String",
+ "params": []
+ },
+ {
+ "kind": "method",
+ "name": "create",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": "SemanticBackend",
+ "params": []
+ }
+ ]
+ },
{
"name": "DocxSemanticBackend",
"binaryName": "com.demcha.compose.document.backend.semantic.docx.DocxSemanticBackend",
@@ -6207,6 +6302,15 @@
"returns": "String",
"params": []
},
+ {
+ "kind": "method",
+ "name": "requiresResolvedLayout",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": "boolean",
+ "params": []
+ },
{
"kind": "method",
"name": "export",
diff --git a/knowledge/api/backends.md b/knowledge/api/backends.md
index cac216805..d03a5e11a 100644
--- a/knowledge/api/backends.md
+++ b/knowledge/api/backends.md
@@ -28,7 +28,7 @@ note: "Generated from the pinned artifact's class files. Authoritative closed se
**GraphCompose version:** 2.5.0-SNAPSHOT
-Types: 70 · methods: 379 · constants: 18 · compiler-generated members: 193
+Types: 73 · methods: 386 · constants: 18 · compiler-generated members: 195
## com.demcha.compose.document.backend.fixed
@@ -537,6 +537,13 @@ Types: 70 · methods: 379 · constants: 18 · compiler-generated members: 193
## com.demcha.compose.document.backend.semantic
+### SemanticBackendProvider (interface) [beta]
+- `String format() [beta]`
+- `SemanticBackend create() [beta]`
+
+### SemanticBackendProviders (class) [beta]
+- `SemanticBackendProvider forFormat(String format) [beta]`
+
### SemanticExportContext (record)
- `new SemanticExportContext(LayoutCanvas, Collection, Path, DocumentOutputOptions, LayoutGraph)`
- `new SemanticExportContext(LayoutCanvas canvas, Collection customFontFamilies, Path outputFile, DocumentOutputOptions outputOptions)`
@@ -557,9 +564,15 @@ Types: 70 · methods: 379 · constants: 18 · compiler-generated members: 193
## com.demcha.compose.document.backend.semantic.docx
+### DocxBackendProvider (class)
+- `new DocxBackendProvider()`
+- `String format()`
+- `SemanticBackend create()`
+
### DocxSemanticBackend (class)
- `new DocxSemanticBackend()`
- `String name()`
+- `boolean requiresResolvedLayout()`
- `byte[] export(DocumentGraph graph, SemanticExportContext context)`
- `Object export(DocumentGraph graph, SemanticExportContext context)`
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxBackendProvider.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxBackendProvider.java
new file mode 100644
index 000000000..c7044a466
--- /dev/null
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxBackendProvider.java
@@ -0,0 +1,39 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.document.backend.semantic.SemanticBackend;
+import com.demcha.compose.document.backend.semantic.SemanticBackendProvider;
+
+/**
+ * Registers the Word export with the document API, so a session can reach it without
+ * naming it.
+ *
+ * Before this, using the export meant constructing {@code DocxSemanticBackend} — which
+ * means importing this artifact in the code that builds the document, and carrying that
+ * dependency wherever the document is built. A render backend has not needed that since
+ * 2.0; now neither does this one: put the artifact on the classpath and
+ * {@code session.buildDocx(path)} finds it.
+ *
+ * A backend is created per export rather than shared, because it holds the state of the
+ * export it is running.
+ *
+ * @author Artem Demchyshyn
+ * @since 2.5.0
+ */
+public final class DocxBackendProvider implements SemanticBackendProvider {
+
+ /**
+ * Creates the provider. Invoked by {@link java.util.ServiceLoader}.
+ */
+ public DocxBackendProvider() {
+ }
+
+ @Override
+ public String format() {
+ return "docx";
+ }
+
+ @Override
+ public SemanticBackend create() {
+ return new DocxSemanticBackend();
+ }
+}
diff --git a/render-docx/src/main/resources/META-INF/services/com.demcha.compose.document.backend.semantic.SemanticBackendProvider b/render-docx/src/main/resources/META-INF/services/com.demcha.compose.document.backend.semantic.SemanticBackendProvider
new file mode 100644
index 000000000..cc0501b97
--- /dev/null
+++ b/render-docx/src/main/resources/META-INF/services/com.demcha.compose.document.backend.semantic.SemanticBackendProvider
@@ -0,0 +1 @@
+com.demcha.compose.document.backend.semantic.docx.DocxBackendProvider
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionExportTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionExportTest.java
new file mode 100644
index 000000000..4862eef66
--- /dev/null
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionExportTest.java
@@ -0,0 +1,149 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.GraphCompose;
+import com.demcha.compose.document.api.DocumentSession;
+import com.demcha.compose.document.backend.semantic.SemanticBackendProviders;
+import com.demcha.compose.document.style.DocumentInsets;
+import org.apache.poi.xwpf.usermodel.XWPFDocument;
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.io.TempDir;
+
+import java.io.ByteArrayInputStream;
+import java.io.ByteArrayOutputStream;
+import java.io.IOException;
+import java.io.OutputStream;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.function.Consumer;
+
+import static org.assertj.core.api.Assertions.assertThat;
+import static org.assertj.core.api.Assertions.assertThatThrownBy;
+
+/**
+ * Reaching the Word export the way a caller reaches a PDF render.
+ *
+ * Until now the only way in was to construct {@code DocxSemanticBackend}, which means
+ * importing this artifact in the code that builds the document and carrying it wherever the
+ * document is built. A render backend has not needed that since 2.0 — the artifact on the
+ * classpath is enough — and these pin that the same now holds here.
+ *
+ * @author Artem Demchyshyn
+ */
+class DocxSessionExportTest {
+
+ @Test
+ void theProviderIsFoundOnTheClasspath() {
+ assertThat(SemanticBackendProviders.forFormat("docx"))
+ .isInstanceOf(DocxBackendProvider.class);
+ assertThat(SemanticBackendProviders.forFormat("DOCX"))
+ .as("the format is a key, not a spelling")
+ .isInstanceOf(DocxBackendProvider.class);
+ }
+
+ @Test
+ void aProviderCreatesAFreshBackendPerExport() {
+ // A semantic backend holds the state of the export it is running, so two exports
+ // sharing one instance would write into each other.
+ DocxBackendProvider provider = new DocxBackendProvider();
+
+ assertThat(provider.create()).isNotSameAs(provider.create());
+ }
+
+ @Test
+ void bytesFileAndStreamAllProduceTheSameDocument() throws Exception {
+ byte[] fromBytes;
+ byte[] fromStream;
+ byte[] fromFile;
+ Path file = Files.createTempFile("session-export", ".docx");
+ try (DocumentSession session = session(page -> page.addParagraph(p -> p.text("Exported")))) {
+ fromBytes = session.toDocxBytes();
+ ByteArrayOutputStream stream = new ByteArrayOutputStream();
+ session.writeDocx(stream);
+ fromStream = stream.toByteArray();
+ session.buildDocx(file);
+ fromFile = Files.readAllBytes(file);
+ } finally {
+ Files.deleteIfExists(file);
+ }
+
+ // Not compared byte for byte: a .docx carries creation timestamps, so three
+ // exports of one document differ in the package while holding the same document.
+ for (byte[] export : new byte[][] {fromBytes, fromStream, fromFile}) {
+ try (XWPFDocument document = new XWPFDocument(new ByteArrayInputStream(export))) {
+ assertThat(document.getParagraphs())
+ .extracting(p -> p.getText())
+ .contains("Exported");
+ }
+ }
+ }
+
+ @Test
+ void theStreamIsLeftOpenForItsOwner() throws Exception {
+ // The caller owns the stream: an HTTP response or an upload is not ours to close.
+ CloseCountingStream stream = new CloseCountingStream();
+ try (DocumentSession session = session(page -> page.addParagraph(p -> p.text("Body")))) {
+ session.writeDocx(stream);
+ }
+
+ assertThat(stream.closes).isZero();
+ assertThat(stream.size()).isPositive();
+ }
+
+ @Test
+ void aFailedExportLeavesTheOldFileAlone() throws Exception {
+ // Written atomically, like buildPdf: a document that fails half-way must not
+ // replace a good file with an unreadable one.
+ Path file = Files.createTempFile("kept", ".docx");
+ Files.writeString(file, "the previous export");
+ try (DocumentSession session = session(page -> page.addParagraph(p -> p.text("Body")))) {
+ session.close();
+ assertThatThrownBy(() -> session.buildDocx(file)).isInstanceOf(IllegalStateException.class);
+ assertThat(Files.readString(file)).isEqualTo("the previous export");
+ } finally {
+ Files.deleteIfExists(file);
+ }
+ }
+
+ @Test
+ void anUnknownFormatNamesWhatToAdd() {
+ assertThatThrownBy(() -> SemanticBackendProviders.forFormat("odt"))
+ .hasMessageContaining("odt")
+ .hasMessageContaining("classpath");
+ }
+
+ private static DocumentSession session(Consumer content) {
+ DocumentSession session = GraphCompose.document()
+ .pageSize(400, 600)
+ .margin(DocumentInsets.of(20))
+ .create();
+ session.pageFlow(content::accept);
+ return session;
+ }
+
+ /** Counts closes so the contract can be asserted rather than assumed. */
+ private static final class CloseCountingStream extends OutputStream {
+
+ private final ByteArrayOutputStream delegate = new ByteArrayOutputStream();
+ private int closes;
+
+ @Override
+ public void write(int b) {
+ delegate.write(b);
+ }
+
+ @Override
+ public void write(byte[] b, int off, int len) {
+ delegate.write(b, off, len);
+ }
+
+ @Override
+ public void close() throws IOException {
+ closes++;
+ super.close();
+ }
+
+ int size() {
+ return delegate.size();
+ }
+ }
+}
From 4a1edad891cb604b8a0400735b87105cec4fade6 Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Tue, 22 Sep 2026 09:37:18 +0100
Subject: [PATCH 09/10] feat(render-docx): tell the caller what the export
could not carry
What a Word document cannot hold, this export has always said -- to the
log, one line per kind. A service generating documents for other people
has no log to read: it needs to know whether the file it is about to
send lost a chart, and which one.
A backend built with a sink hands over a DocxExportReport once the bytes
are complete:
session.export(new DocxSemanticBackend(report -> log(report)));
Errors do not travel this way. An export that cannot proceed throws, and
a report is not how a caller finds that out; the sink is called after
the bytes exist, so nobody is told what an export lost by an export that
did not finish.
Two severities, and the difference is the point. DROPPED means the page
draws it and the document does not carry it. APPROXIMATED means it is
there as the nearest thing Word owns -- a panel that keeps its fill and
loses its rounded corners. A caller deciding whether to send the file
needs to tell those apart.
Every occurrence is recorded even though the log still says it once per
kind: asking what the document lost means wanting the three charts it
lost and which three. Each note carries the authored node's path, the
same one the layout graph addresses it by, so a note traces back to the
code that wrote it instead of being guessed at from its text. That
needed one change: the path index was built only alongside a layout, and
is now built always -- a path is not a measurement, and a document the
engine cannot lay out is exactly the one whose report matters.
Recorded: dropped nodes, dropped inline runs, a dropped corner radius, a
clipped shape container, a chart becoming its data table, a face that
may not be embedded with the reason, and an export with no resolved
layout behind it.
Tests: a clean document reporting nothing, a dropped node named with its
path, three drops of one kind recorded three times, an approximation
counted apart from a drop, a chart saying what it became, a note reading
as one line, and the report not changing the document it reports on.
---
docs/api-stability.md | 9 +
docs/recipes/docx-export.md | 21 ++
knowledge/api/backends.json | 200 +++++++++++++++++-
knowledge/api/backends.md | 21 +-
.../semantic/docx/DocxExportReport.java | 120 +++++++++++
.../backend/semantic/docx/DocxFontTable.java | 28 ++-
.../semantic/docx/DocxLayoutMetrics.java | 18 +-
.../semantic/docx/DocxSemanticBackend.java | 84 +++++++-
.../semantic/docx/DocxExportReportTest.java | 143 +++++++++++++
9 files changed, 623 insertions(+), 21 deletions(-)
create mode 100644 render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxExportReport.java
create mode 100644 render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxExportReportTest.java
diff --git a/docs/api-stability.md b/docs/api-stability.md
index 78aa5cdb6..a147ae9f5 100644
--- a/docs/api-stability.md
+++ b/docs/api-stability.md
@@ -77,6 +77,15 @@ matrix.
> generalise against. Constructing `DocxSemanticBackend` and calling
> `session.export(backend, path)` stays the Stable path and is unaffected.
>
+> The **DOCX export report** is Experimental in 2.5.0: the `DocxExportReport` type and
+> the `DocxSemanticBackend(Consumer)` constructor, which hand the
+> calling program what the export dropped or approximated and the path of the node each
+> came from — the same information the export has always written to the log, where a
+> service generating documents for other people cannot read it. It is marked
+> Experimental because what a report should carry is still growing: coverage of what
+> *was* written is not in it yet, and adding that will move the shape. The no-argument
+> constructor and the log are unaffected.
+>
> Seven members of the otherwise-Stable **PDF backend** also carry `@Beta`. The
> package is not Experimental — these are:
> `PdfFixedLayoutBackend.renderSections` / `writeSections`, the low-level seam
diff --git a/docs/recipes/docx-export.md b/docs/recipes/docx-export.md
index f49f14e1c..913e088cf 100644
--- a/docs/recipes/docx-export.md
+++ b/docs/recipes/docx-export.md
@@ -53,6 +53,27 @@ PDF never pull POI.
Page geometry (size and margins) and session metadata (title, author,
subject, keywords) carry into the Word document as well.
+## Finding out what the export could not carry
+
+The export says what it drops — but it says it to the log, which a service generating
+documents for other people cannot read. Pass a sink and the same information arrives as a
+value:
+
+```java
+var notes = new ArrayList();
+session.export(new DocxSemanticBackend(report -> notes.addAll(report.notes())));
+```
+
+Each note carries a severity, what it was about, the authored node's path, and what it
+means for the document. `DROPPED` means the page draws it and the document does not carry
+it; `APPROXIMATED` means it is in the document as the nearest thing Word owns — a panel
+that keeps its fill and loses its rounded corners. Neither is an error: an export that
+cannot proceed throws, and the report is not how you find that out.
+
+The sink is called once, after the bytes are complete. The convenience methods
+(`buildDocx`, `writeDocx`, `toDocxBytes`) build their own backend and so have no sink —
+use `session.export(...)` when you need the report.
+
## Measured geometry
The export asks the session for the resolved layout and writes three things from it that
diff --git a/knowledge/api/backends.json b/knowledge/api/backends.json
index 02e3242fa..0c97d2ae3 100644
--- a/knowledge/api/backends.json
+++ b/knowledge/api/backends.json
@@ -22,10 +22,10 @@
"graph-compose-testing:sources"
],
"counts": {
- "types": 73,
- "methods": 386,
- "constants": 18,
- "generated": 195
+ "types": 76,
+ "methods": 397,
+ "constants": 21,
+ "generated": 205
},
"packages": [
{
@@ -6275,6 +6275,184 @@
}
]
},
+ {
+ "name": "DocxExportReport",
+ "binaryName": "com.demcha.compose.document.backend.semantic.docx.DocxExportReport",
+ "kind": "record",
+ "modifiers": [
+ "final"
+ ],
+ "artifact": "graph-compose-render-docx",
+ "stability": "beta",
+ "members": [
+ {
+ "kind": "constant",
+ "name": "EMPTY",
+ "static": true,
+ "origin": "generated",
+ "type": "DocxExportReport",
+ "stability": "beta"
+ },
+ {
+ "kind": "constructor",
+ "name": "DocxExportReport",
+ "static": false,
+ "origin": "generated",
+ "typeParameters": null,
+ "returns": null,
+ "params": [
+ {
+ "type": "List",
+ "name": null
+ }
+ ],
+ "stability": "beta"
+ },
+ {
+ "kind": "method",
+ "name": "isEmpty",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": "boolean",
+ "params": [],
+ "stability": "beta"
+ },
+ {
+ "kind": "method",
+ "name": "count",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": "long",
+ "params": [
+ {
+ "type": "DocxExportReport.Severity",
+ "name": "severity"
+ }
+ ],
+ "stability": "beta"
+ },
+ {
+ "kind": "method",
+ "name": "bySubject",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": "Map>",
+ "params": [],
+ "stability": "beta"
+ },
+ {
+ "kind": "method",
+ "name": "notes",
+ "static": false,
+ "origin": "generated",
+ "typeParameters": null,
+ "returns": "List",
+ "params": [],
+ "stability": "beta"
+ }
+ ]
+ },
+ {
+ "name": "DocxExportReport.Note",
+ "binaryName": "com.demcha.compose.document.backend.semantic.docx.DocxExportReport$Note",
+ "kind": "record",
+ "modifiers": [
+ "final"
+ ],
+ "artifact": "graph-compose-render-docx",
+ "members": [
+ {
+ "kind": "constructor",
+ "name": "Note",
+ "static": false,
+ "origin": "generated",
+ "typeParameters": null,
+ "returns": null,
+ "params": [
+ {
+ "type": "DocxExportReport.Severity",
+ "name": null
+ },
+ {
+ "type": "String",
+ "name": null
+ },
+ {
+ "type": "String",
+ "name": null
+ },
+ {
+ "type": "String",
+ "name": null
+ }
+ ]
+ },
+ {
+ "kind": "method",
+ "name": "severity",
+ "static": false,
+ "origin": "generated",
+ "typeParameters": null,
+ "returns": "DocxExportReport.Severity",
+ "params": []
+ },
+ {
+ "kind": "method",
+ "name": "subject",
+ "static": false,
+ "origin": "generated",
+ "typeParameters": null,
+ "returns": "String",
+ "params": []
+ },
+ {
+ "kind": "method",
+ "name": "path",
+ "static": false,
+ "origin": "generated",
+ "typeParameters": null,
+ "returns": "String",
+ "params": []
+ },
+ {
+ "kind": "method",
+ "name": "detail",
+ "static": false,
+ "origin": "generated",
+ "typeParameters": null,
+ "returns": "String",
+ "params": []
+ }
+ ]
+ },
+ {
+ "name": "DocxExportReport.Severity",
+ "binaryName": "com.demcha.compose.document.backend.semantic.docx.DocxExportReport$Severity",
+ "kind": "enum",
+ "modifiers": [
+ "final"
+ ],
+ "artifact": "graph-compose-render-docx",
+ "members": [
+ {
+ "kind": "constant",
+ "name": "DROPPED",
+ "static": true,
+ "origin": "generated",
+ "type": "DocxExportReport.Severity"
+ },
+ {
+ "kind": "constant",
+ "name": "APPROXIMATED",
+ "static": true,
+ "origin": "generated",
+ "type": "DocxExportReport.Severity"
+ }
+ ]
+ },
{
"name": "DocxSemanticBackend",
"binaryName": "com.demcha.compose.document.backend.semantic.docx.DocxSemanticBackend",
@@ -6293,6 +6471,20 @@
"returns": null,
"params": []
},
+ {
+ "kind": "constructor",
+ "name": "DocxSemanticBackend",
+ "static": false,
+ "origin": "source",
+ "typeParameters": null,
+ "returns": null,
+ "params": [
+ {
+ "type": "Consumer",
+ "name": "reportSink"
+ }
+ ]
+ },
{
"kind": "method",
"name": "name",
diff --git a/knowledge/api/backends.md b/knowledge/api/backends.md
index d03a5e11a..5bdfe2570 100644
--- a/knowledge/api/backends.md
+++ b/knowledge/api/backends.md
@@ -28,7 +28,7 @@ note: "Generated from the pinned artifact's class files. Authoritative closed se
**GraphCompose version:** 2.5.0-SNAPSHOT
-Types: 73 · methods: 386 · constants: 18 · compiler-generated members: 195
+Types: 76 · methods: 397 · constants: 21 · compiler-generated members: 205
## com.demcha.compose.document.backend.fixed
@@ -569,8 +569,27 @@ Types: 73 · methods: 386 · constants: 18 · compiler-generated members: 195
- `String format()`
- `SemanticBackend create()`
+### DocxExportReport (record) [beta]
+- `new DocxExportReport(List) [beta]`
+- `boolean isEmpty() [beta]`
+- `long count(DocxExportReport.Severity severity) [beta]`
+- `Map> bySubject() [beta]`
+- `List notes() [beta]`
+- constants: `EMPTY`
+
+### DocxExportReport.Note (record)
+- `new Note(DocxExportReport.Severity, String, String, String)`
+- `DocxExportReport.Severity severity()`
+- `String subject()`
+- `String path()`
+- `String detail()`
+
+### DocxExportReport.Severity (enum)
+- constants: `DROPPED`, `APPROXIMATED`
+
### DocxSemanticBackend (class)
- `new DocxSemanticBackend()`
+- `new DocxSemanticBackend(Consumer reportSink)`
- `String name()`
- `boolean requiresResolvedLayout()`
- `byte[] export(DocumentGraph graph, SemanticExportContext context)`
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxExportReport.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxExportReport.java
new file mode 100644
index 000000000..da746ef6d
--- /dev/null
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxExportReport.java
@@ -0,0 +1,120 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.document.api.Beta;
+
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Map;
+import java.util.stream.Collectors;
+
+/**
+ * What an export could not carry, and where in the document it was.
+ *
+ * A Word document cannot hold everything a page can draw, and this export says so
+ * rather than approximating in silence — but until now it said so to the log, one line per
+ * kind, with no way for the calling program to find out. A service generating documents
+ * for other people has no log to read: it needs to know whether the file it is about to
+ * send dropped a chart, and which one.
+ *
+ * Two kinds of note, and the difference matters. Something {@link Severity#DROPPED} is
+ * not in the file: the page draws it and the document does not. Something
+ * {@link Severity#APPROXIMATED} is in the file as the nearest thing Word owns — a panel
+ * with square corners where the page rounds them. Neither is an error; an export that
+ * cannot proceed throws instead, and a report is never a substitute for that.
+ *
+ * Every note carries the path of the node it came from, the same path the layout graph
+ * addresses that node by, so a note can be traced back to the authoring code rather than
+ * guessed at from its text.
+ *
+ * Experimental ({@code @Beta}) — see {@code docs/api-stability.md}.
+ *
+ * @param notes everything worth telling the caller, in the order it was found
+ * @author Artem Demchyshyn
+ * @since 2.5.0
+ */
+@Beta
+public record DocxExportReport(List notes) {
+
+ /** An empty report: the whole document was written as it was authored. */
+ public static final DocxExportReport EMPTY = new DocxExportReport(List.of());
+
+ /**
+ * Freezes the notes.
+ */
+ public DocxExportReport {
+ notes = List.copyOf(notes);
+ }
+
+ /** How much of the thing survived. */
+ public enum Severity {
+ /** The page draws it and the document does not carry it at all. */
+ DROPPED,
+ /** It is in the document as the nearest thing Word owns, which is not the same. */
+ APPROXIMATED
+ }
+
+ /**
+ * One thing the export could not carry as authored.
+ *
+ * @param severity whether it is missing or merely different
+ * @param subject what it was, in a few words — {@code "chart"}, {@code "corner radius"}
+ * @param path the authored node's path, or null when it belongs to the whole export
+ * @param detail what happened and what it means for the document
+ */
+ public record Note(Severity severity, String subject, String path, String detail) {
+
+ /**
+ * Validates the parts a reader needs.
+ */
+ public Note {
+ if (severity == null || subject == null || subject.isBlank()) {
+ throw new IllegalArgumentException("a note states its severity and its subject");
+ }
+ detail = detail == null ? "" : detail;
+ }
+
+ @Override
+ public String toString() {
+ return severity + " " + subject + (path == null ? "" : " at " + path)
+ + (detail.isEmpty() ? "" : ": " + detail);
+ }
+ }
+
+ /** @return true when nothing was dropped or approximated */
+ public boolean isEmpty() {
+ return notes.isEmpty();
+ }
+
+ /**
+ * @param severity the severity to count
+ * @return how many notes carry it
+ */
+ public long count(Severity severity) {
+ return notes.stream().filter(note -> note.severity() == severity).count();
+ }
+
+ /**
+ * The notes grouped by what they are about, for a caller summarising rather than
+ * listing.
+ *
+ * @return subject to its notes, in the order each subject was first seen
+ */
+ public Map> bySubject() {
+ return notes.stream().collect(Collectors.groupingBy(Note::subject,
+ java.util.LinkedHashMap::new, Collectors.toList()));
+ }
+
+ /** Builds a report while an export runs. Not thread-safe; one export owns one. */
+ static final class Builder {
+
+ private final List notes = new ArrayList<>();
+
+ void add(Severity severity, String subject, String path, String detail) {
+ notes.add(new Note(severity, subject, path, detail));
+ }
+
+ DocxExportReport build() {
+ return notes.isEmpty() ? EMPTY : new DocxExportReport(notes);
+ }
+ }
+}
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
index 77f9d9387..1f5785798 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxFontTable.java
@@ -83,12 +83,14 @@ private DocxFontTable() {
* @param document the document being written
* @param graph the document's own tree, read for the faces it names
* @param custom families the session registered, which win over the bundled ones
+ * @param report where a face that may not travel is recorded
* @throws IOException if a part cannot be written
*/
static void write(XWPFDocument document,
DocumentGraph graph,
- Collection custom) throws IOException {
- List embedded = resolve(graph, custom);
+ Collection custom,
+ DocxExportReport.Builder report) throws IOException {
+ List embedded = resolve(graph, custom, report);
if (embedded.isEmpty()) {
return;
}
@@ -175,7 +177,9 @@ void relationshipId(String id) {
* asks for the Helvetica family, which is a name rather than a file and ships
* nothing.
*/
- private static List resolve(DocumentGraph graph, Collection custom) {
+ private static List resolve(DocumentGraph graph,
+ Collection custom,
+ DocxExportReport.Builder report) {
Map> used = new LinkedHashMap<>();
for (DocumentNode root : graph.roots()) {
collectFonts(root, used);
@@ -193,7 +197,7 @@ private static List resolve(DocumentGraph graph, Collection faces = facesOf(family, entry.getValue());
+ List faces = facesOf(family, entry.getValue(), report);
if (!faces.isEmpty()) {
embedded.add(new Embedded(family.wordFamily(), faces));
}
@@ -245,14 +249,15 @@ static Slot of(DocumentTextStyle style) {
* later bolds a word gets whatever their machine does for a missing bold face, which is
* what happens in any document that does not carry one.
*/
- private static List facesOf(FontFamilyDefinition family, Set slots) {
+ private static List facesOf(FontFamilyDefinition family, Set slots,
+ DocxExportReport.Builder report) {
FontFamilyDefinition.FontSourceSet sources = family.fontSourceSet().orElseThrow();
List faces = new ArrayList<>(slots.size());
for (Slot slot : Slot.values()) {
if (!slots.contains(slot)) {
continue;
}
- addFace(faces, family, slot, switch (slot) {
+ addFace(faces, family, slot, report, switch (slot) {
case REGULAR -> sources.regular();
case BOLD -> sources.bold();
case ITALIC -> sources.italic();
@@ -265,6 +270,7 @@ private static List facesOf(FontFamilyDefinition family, Set slots)
private static void addFace(List faces,
FontFamilyDefinition family,
Slot slot,
+ DocxExportReport.Builder report,
FontFamilyDefinition.FontBinarySource source) {
if (source == null) {
return;
@@ -276,6 +282,10 @@ private static void addFace(List faces,
LOG.warn("DocxSemanticBackend: '{}' face of '{}' could not be read ({}); the "
+ "document declares the family and ships this face without it",
slot.element(), family.wordFamily(), failure.toString());
+ report.add(DocxExportReport.Severity.DROPPED, "embedded font", null,
+ "the " + slot.element() + " face of '" + family.wordFamily()
+ + "' could not be read (" + failure + "), so a reader without it "
+ + "installed sees a substituted face");
return;
}
switch (DocxFontEmbedding.permissionOf(bytes)) {
@@ -283,12 +293,18 @@ private static void addFace(List faces,
LOG.warn("DocxSemanticBackend: '{}' does not permit embedding (OS/2 fsType), so "
+ "it is named but not shipped; a reader without it installed sees a "
+ "substituted face", family.wordFamily());
+ report.add(DocxExportReport.Severity.DROPPED, "embedded font", null,
+ "'" + family.wordFamily() + "' does not permit embedding (OS/2 fsType), "
+ + "so a reader without it installed sees a substituted face");
return;
}
case PREVIEW_ONLY -> {
LOG.warn("DocxSemanticBackend: '{}' permits embedding for reading and printing "
+ "only, which a document meant to be edited cannot rely on, so it is "
+ "named but not shipped", family.wordFamily());
+ report.add(DocxExportReport.Severity.DROPPED, "embedded font", null,
+ "'" + family.wordFamily() + "' permits embedding for reading and "
+ + "printing only, which a document meant to be edited cannot rely on");
return;
}
default -> {
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
index 770a44813..02428853e 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
@@ -64,13 +64,18 @@ private DocxLayoutMetrics(Map paths,
* @return an index, or {@link #EMPTY} when there is no layout to index
*/
static DocxLayoutMetrics of(DocumentGraph graph, LayoutGraph layout) {
- if (graph == null || layout == null) {
+ if (graph == null) {
return EMPTY;
}
Map paths = new IdentityHashMap<>();
for (int index = 0; index < graph.roots().size(); index++) {
indexPaths(graph.roots().get(index), null, index, paths);
}
+ if (layout == null) {
+ // No measurements, but the paths still name the nodes — which is what a
+ // diagnostic note needs to say where in the document it came from.
+ return new DocxLayoutMetrics(paths, Map.of(), Map.of());
+ }
Map> fragments = new HashMap<>();
for (PlacedFragment fragment : layout.fragments()) {
fragments.computeIfAbsent(fragment.path(), key -> new ArrayList<>()).add(fragment);
@@ -112,6 +117,17 @@ boolean isEmpty() {
return fragments.isEmpty();
}
+ /**
+ * The path the layout graph addresses a node by, for a note that has to say where in
+ * the document it came from.
+ *
+ * @param node any authored node
+ * @return its path, or null when this index was built from no graph at all
+ */
+ String pathOf(DocumentNode node) {
+ return node == null ? null : paths.get(node);
+ }
+
/**
* The height of one line of the paragraph's text, as the engine measured it.
*
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
index 562eda0f1..ca466a133 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
@@ -172,6 +172,12 @@ public final class DocxSemanticBackend implements SemanticBackend {
// Every family this export can name, by the logical name a style asks for. The
// session's own registrations win over the bundled ones, the way they do everywhere.
private java.util.Map wordFamilies = java.util.Map.of();
+ // What this export could not carry as authored. Collected whether or not anyone asked
+ // for it: building it costs a list, and deciding later that nobody wanted it is not
+ // something the writers can do halfway through.
+ private DocxExportReport.Builder report = new DocxExportReport.Builder();
+ // Where the finished report goes, when the caller configured somewhere for it to go.
+ private final java.util.function.Consumer reportSink;
/**
* A container's paint, reduced to what a Word paragraph can carry.
@@ -191,6 +197,27 @@ boolean isEmpty() {
* Creates a DOCX semantic backend.
*/
public DocxSemanticBackend() {
+ this(null);
+ }
+
+ /**
+ * Creates a backend that hands its report to {@code reportSink} when an export ends.
+ *
+ * A Word document cannot hold everything a page can draw, and this export says so
+ * rather than approximating in silence — but it said so to the log, which a service
+ * generating documents for other people has no way to read. Configure a sink and the
+ * same information arrives as a {@link DocxExportReport}: what was dropped, what was
+ * approximated, and the path of the node each came from.
+ *
+ * The sink is called once per export, after the bytes are complete, and only for an
+ * export that finished — an export that fails throws, and a report is not a way to
+ * discover that it did.
+ *
+ * @param reportSink where the report goes, or null to keep the log as the only channel
+ * @since 2.5.0
+ */
+ public DocxSemanticBackend(java.util.function.Consumer reportSink) {
+ this.reportSink = reportSink;
}
@Override
@@ -226,16 +253,25 @@ public byte[] export(DocumentGraph graph, SemanticExportContext context) throws
warnedNodeKinds.clear();
containerPaint.clear();
listNumbering.clear();
+ report = new DocxExportReport.Builder();
wordFamilies = DocxFontTable.familiesByName(context.customFontFamilies());
documentDefaultStyle = dominantTextStyle(graph);
layout = DocxLayoutMetrics.of(graph, context.layoutGraph());
+ if (layout.isEmpty()) {
+ // Said once, for the whole export: without measurements the line height is
+ // Word's and so is every auto column, and a caller comparing this file against
+ // the rendered page deserves to know that before they look.
+ report.add(DocxExportReport.Severity.APPROXIMATED, "measured geometry", null,
+ "this document could not be laid out, so line heights and auto column "
+ + "widths are the editor's rather than the engine's");
+ }
carriedSpacingBefore = 0;
lastBodyParagraph = null;
contentWidth = context.canvas() == null ? Double.MAX_VALUE : context.canvas().innerWidth();
try (XWPFDocument document = new XWPFDocument()) {
applyPageGeometry(document, context.canvas());
writeStylesPart(document);
- DocxFontTable.write(document, graph, context.customFontFamilies());
+ DocxFontTable.write(document, graph, context.customFontFamilies(), report);
applyOutputOptions(document, context.outputOptions());
for (DocumentNode root : graph.roots()) {
writeNode(document, root);
@@ -246,6 +282,11 @@ public byte[] export(DocumentGraph graph, SemanticExportContext context) throws
if (context.outputFile() != null) {
Files.write(context.outputFile(), bytes);
}
+ // Handed over once the bytes exist, so a caller is never told what an
+ // export lost by an export that did not finish.
+ if (reportSink != null) {
+ reportSink.accept(report.build());
+ }
return bytes;
}
}
@@ -427,12 +468,20 @@ private void writeNode(XWPFDocument document, DocumentNode node) throws Exceptio
}
}
- /** One warning per dropped node kind, deduplicated across the export. */
+ /**
+ * One warning per dropped node kind, deduplicated across the export.
+ *
+ * The report is told about every one of them, not one per kind: a caller asking
+ * what the document lost wants the three charts it lost, and which three. The log is
+ * the summary and the report is the record.
+ */
private void warnUnsupported(DocumentNode node) {
if (warnedNodeKinds.add(node.nodeKind())) {
LOG.warn("DocxSemanticBackend: dropping '{}' node(s) — geometry has no semantic "
+ "Word analogue; use the PDF backend for pixel-perfect output", node.nodeKind());
}
+ report.add(DocxExportReport.Severity.DROPPED, node.nodeKind(), layout.pathOf(node),
+ "geometry has no semantic Word analogue, so it is not in the document at all");
}
/**
@@ -444,10 +493,10 @@ private void warnUnsupported(DocumentNode node) {
* no signal at all, weaker than the block-level drop path.
*/
private void warnDroppedInlineRuns(ParagraphNode node) {
- warnDroppedInlineRuns(node.inlineRuns());
+ warnDroppedInlineRuns(node.inlineRuns(), layout.pathOf(node));
}
- private void warnDroppedInlineRuns(List runs) {
+ private void warnDroppedInlineRuns(List runs, String path) {
for (InlineRun run : runs) {
if (run instanceof InlineTextRun || run instanceof InlineHighlightRun) {
continue;
@@ -458,6 +507,8 @@ private void warnDroppedInlineRuns(List runs) {
+ "analogue; the paragraph text renders without them, use the PDF "
+ "backend for full fidelity", kind);
}
+ report.add(DocxExportReport.Severity.DROPPED, "inline " + kind, path,
+ "no semantic Word analogue; the paragraph's text is written without it");
}
}
@@ -637,7 +688,8 @@ private void writeList(XWPFDocument document,
// A drawn marker's pieces are runs, so the row is written the way
// any row with runs in it is; its item is still just a label.
writeRichListLine(document, list.textStyle(), list.marker(),
- com.demcha.compose.document.node.ListItem.of(normalized), 0, lineHeight);
+ com.demcha.compose.document.node.ListItem.of(normalized), 0, lineHeight,
+ layout.pathOf(list));
} else if (numId != null) {
// Word draws the marker, so the text is the item and nothing else.
writeListLine(document, list.textStyle(), normalized, 0, numId, lineHeight);
@@ -666,7 +718,8 @@ private void writeNestedItem(XWPFDocument document,
: com.demcha.compose.document.node.ListMarker.defaultForDepth(depth);
java.util.OptionalDouble lineHeight = layout.lineHeight(list);
if (item.isRich() || marker.isRich()) {
- writeRichListLine(document, list.textStyle(), marker, item, depth, lineHeight);
+ writeRichListLine(document, list.textStyle(), marker, item, depth, lineHeight,
+ layout.pathOf(list));
} else if (numId != null) {
writeListLine(document, list.textStyle(), item.label(), depth, numId, lineHeight);
} else {
@@ -722,9 +775,10 @@ private void writeRichListLine(XWPFDocument document, DocumentTextStyle style,
com.demcha.compose.document.node.ListMarker marker,
com.demcha.compose.document.node.ListItem item,
int depth,
- java.util.OptionalDouble lineHeight) {
- warnDroppedInlineRuns(marker.runs());
- warnDroppedInlineRuns(item.runs());
+ java.util.OptionalDouble lineHeight,
+ String path) {
+ warnDroppedInlineRuns(marker.runs(), path);
+ warnDroppedInlineRuns(item.runs(), path);
XWPFParagraph para = newBodyParagraph(document);
applyLineHeight(para, lineHeight);
XWPFRun leading = para.createRun();
@@ -772,6 +826,9 @@ private void writeInlineTextRuns(XWPFParagraph para, DocumentTextStyle style,
* fixed-layout backend, where charts compile into ordinary primitives.
*/
private void writeChartFallback(XWPFDocument document, ChartNode node) throws Exception {
+ report.add(DocxExportReport.Severity.APPROXIMATED, "chart", layout.pathOf(node),
+ "exported as its data table — a categories-by-series table in the chart's own "
+ + "value format — because the drawn chart is layout geometry");
if (chartWarned.compareAndSet(false, true)) {
LOG.warn("docx.export.chart-fallback kind={} — the semantic DOCX export has no "
+ "layout pass, so charts are exported as their data table. "
@@ -1064,6 +1121,11 @@ private void warnContainerRadiusDropped(DocumentNode node) {
boolean rounded = node instanceof SectionNode section
? hasRadius(section.cornerRadius())
: node instanceof ContainerNode container && hasRadius(container.cornerRadius());
+ if (rounded) {
+ report.add(DocxExportReport.Severity.APPROXIMATED, "corner radius", layout.pathOf(node),
+ "Word paragraph shading is rectangular, so the panel keeps its fill and "
+ + "loses its rounded corners");
+ }
if (rounded && containerRadiusWarned.compareAndSet(false, true)) {
LOG.warn("docx.export.container-radius-dropped node='{}' — Word paragraph shading "
+ "is rectangular, so the panel renders with square corners. "
@@ -1174,6 +1236,10 @@ private void writeShapeContainer(XWPFDocument document, ShapeContainerNode node)
// the outline frame and without clipping. The resulting Word document
// shows the layer content but not the shape boundary — authors who
// need the boundary must export to PDF.
+ report.add(DocxExportReport.Severity.APPROXIMATED, "clipped shape container",
+ layout.pathOf(node),
+ "DOCX has no graphics-state clip, so the layers are written inline, in source "
+ + "order, without the outline and without being clipped to it");
if (shapeContainerWarned.compareAndSet(false, true)) {
LOG.warn("docx.export.shape-container-fallback "
+ "outline='{}' clipPolicy={} — DOCX has no graphics-state clip; "
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxExportReportTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxExportReportTest.java
new file mode 100644
index 000000000..2cb7dbf6b
--- /dev/null
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxExportReportTest.java
@@ -0,0 +1,143 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.GraphCompose;
+import com.demcha.compose.document.api.DocumentSession;
+import com.demcha.compose.document.chart.ChartData;
+import com.demcha.compose.document.chart.ChartSpec;
+import com.demcha.compose.document.dsl.PageFlowBuilder;
+import com.demcha.compose.document.style.DocumentColor;
+import com.demcha.compose.document.style.DocumentCornerRadius;
+import com.demcha.compose.document.style.DocumentInsets;
+import org.junit.jupiter.api.Test;
+
+import java.util.List;
+import java.util.concurrent.atomic.AtomicReference;
+import java.util.function.Consumer;
+
+import static org.assertj.core.api.Assertions.assertThat;
+
+/**
+ * What the export could not carry, told to the program rather than to the log.
+ *
+ * The export has always said what it drops — one line per kind, to a logger. A service
+ * generating documents for other people has no log to read: it needs to know whether the
+ * file it is about to send lost a chart, and which one. That is what the report is, and it
+ * is deliberately not an error channel — an export that cannot proceed throws.
+ *
+ * @author Artem Demchyshyn
+ */
+class DocxExportReportTest {
+
+ private static final DocumentColor SURFACE = DocumentColor.rgb(238, 243, 249);
+
+ @Test
+ void aDocumentThatLosesNothingReportsNothing() throws Exception {
+ assertThat(reportOf(page -> page.addParagraph(p -> p.text("Ordinary text")))
+ .isEmpty())
+ .as("an export with nothing to say says nothing")
+ .isTrue();
+ }
+
+ @Test
+ void aDroppedNodeIsNamedWithItsPath() throws Exception {
+ DocxExportReport report = reportOf(page -> page
+ .addParagraph(p -> p.text("Before"))
+ .addLine(line -> line.name("Divider").thickness(1)));
+
+ List notes = report.notes();
+ assertThat(notes).hasSize(1);
+ assertThat(notes.get(0).severity()).isEqualTo(DocxExportReport.Severity.DROPPED);
+ assertThat(notes.get(0).path())
+ .as("the path the layout graph addresses that node by, so it can be traced back")
+ .contains("Divider");
+ assertThat(notes.get(0).detail()).isNotEmpty();
+ }
+
+ @Test
+ void everyDroppedNodeIsRecordedEvenThoughTheLogSaysItOnce() throws Exception {
+ // The log is the summary and the report is the record: a caller asking what the
+ // document lost wants the three it lost, not the fact that it lost a kind.
+ DocxExportReport report = reportOf(page -> page
+ .addLine(line -> line.name("First").thickness(1))
+ .addLine(line -> line.name("Second").thickness(1))
+ .addLine(line -> line.name("Third").thickness(1)));
+
+ assertThat(report.count(DocxExportReport.Severity.DROPPED)).isEqualTo(3);
+ assertThat(report.notes()).extracting(DocxExportReport.Note::path)
+ .anyMatch(path -> path.contains("First"))
+ .anyMatch(path -> path.contains("Third"));
+ }
+
+ @Test
+ void whatSurvivesInAnotherFormIsApproximatedRatherThanDropped() throws Exception {
+ // A rounded card keeps its fill and loses its corners. That is not the same as
+ // losing the card, and a caller deciding whether to send the file needs the
+ // difference.
+ DocxExportReport report = reportOf(page -> page.addSection("Card", card -> card
+ .fillColor(SURFACE)
+ .cornerRadius(DocumentCornerRadius.of(14))
+ .addParagraph(p -> p.text("Inside the card"))));
+
+ assertThat(report.count(DocxExportReport.Severity.APPROXIMATED)).isEqualTo(1);
+ assertThat(report.count(DocxExportReport.Severity.DROPPED)).isZero();
+ assertThat(report.bySubject()).containsKey("corner radius");
+ assertThat(report.notes().get(0).path()).contains("Card");
+ }
+
+ @Test
+ void aChartSaysWhatItBecame() throws Exception {
+ ChartData data = ChartData.builder()
+ .categories("Q1", "Q2")
+ .series("2025", 12.4, 15.1)
+ .build();
+ DocxExportReport report = reportOf(page -> page
+ .chart(ChartSpec.bar().data(data).build()));
+
+ assertThat(report.bySubject()).containsKey("chart");
+ DocxExportReport.Note note = report.bySubject().get("chart").get(0);
+ assertThat(note.severity()).isEqualTo(DocxExportReport.Severity.APPROXIMATED);
+ assertThat(note.detail()).contains("data table");
+ assertThat(note.path())
+ .as("an unnamed node is addressed by its kind, which still locates it")
+ .isEqualTo("ContainerNode[0]/Chart[0]");
+ }
+
+ @Test
+ void withNoSinkTheExportIsUnchanged() throws Exception {
+ // The report is opt-in, and asking for it must not change the document.
+ byte[] withSink;
+ byte[] without;
+ try (DocumentSession session = session(page -> page.addParagraph(p -> p.text("Body")))) {
+ withSink = session.export(new DocxSemanticBackend(report -> { }));
+ without = session.export(new DocxSemanticBackend());
+ }
+ // Not byte-for-byte: a .docx carries a creation timestamp. The document is.
+ assertThat(withSink.length).isCloseTo(without.length, org.assertj.core.data.Offset.offset(64));
+ }
+
+ @Test
+ void aNoteReadsAsOneLine() {
+ DocxExportReport.Note note = new DocxExportReport.Note(
+ DocxExportReport.Severity.DROPPED, "chart", "Body[0]/Revenue[2]", "no analogue");
+
+ assertThat(note.toString()).isEqualTo("DROPPED chart at Body[0]/Revenue[2]: no analogue");
+ }
+
+ private static DocxExportReport reportOf(Consumer content) throws Exception {
+ AtomicReference captured = new AtomicReference<>();
+ try (DocumentSession session = session(content)) {
+ session.export(new DocxSemanticBackend(captured::set));
+ }
+ assertThat(captured.get()).as("the sink is called once the bytes exist").isNotNull();
+ return captured.get();
+ }
+
+ private static DocumentSession session(Consumer content) {
+ DocumentSession session = GraphCompose.document()
+ .pageSize(400, 600)
+ .margin(DocumentInsets.of(20))
+ .create();
+ session.pageFlow(content::accept);
+ return session;
+ }
+}
From c9ab6d8079ec6255b69d863862353df5888b5d7b Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Tue, 22 Sep 2026 13:00:30 +0100
Subject: [PATCH 10/10] chore(examples): re-render the committed Word preview
for the layout work
This branch states a measured line height on every paragraph, carries the space a block holds above and below itself, and writes a grid the layout resolved, so the same example produces a different document than the one the base carries.
Verification: ./mvnw -B -ntp test -f examples/pom.xml, 93 tests, BUILD SUCCESS.
---
.../examples/word-export-companion.docx | Bin 9137 -> 9215 bytes
1 file changed, 0 insertions(+), 0 deletions(-)
diff --git a/assets/readme/examples/word-export-companion.docx b/assets/readme/examples/word-export-companion.docx
index 409d1c01b26973a178b08eb23869f6b66de28db3..2facc29b103d4257db1d55df28b19ec2e176f7b8 100644
GIT binary patch
delta 2543
zcmYk8c{mjM8pmg`?^~8I&5-6;Ms_7c3}cBwVl+n9Io2*FgbasdpJs-TV;@^mSrZN-
zWQ<+XSWc*skTeP*_vk*)z4!Oe_xZff`#kR-zdyd8ddnkA0kjR6Lju6V!vhG!*9*i0
zL4o+jETGJBQfok?#}N%yBEGO&)(}bd@O>Zd1UD^FvhFSI#z`W?kjNPA?_+)iGxzb(
zjalz1rS5MtNA>O=8~Z{F8tAq4OA>4>z-_}eP@jLRwQV?#2YPk7dIW<_0aIE7
z0yb1PKnvU}n*U1Y827H3h=a8-Iw_QG$rvOXHFGXDwA4k;qt}8MT@e|O*uY@(zg1K%
zcO5*kNE92q5z)tCwrs~}AYWK8bMpo9FMzmAXD1ikqh2t}Aa
z86`S|hl`Ky0;Xj+7MAdzf9!naEn2=+qC2K)uN~#!Zn>=guLv%6z=oT>cfGK?WB8DK
z+?vo@>*pFiS|z}(QgNnY6f~8C8_>KveA?5kI@RV_
zExppdZ=xo=c{@rL`0+z+sPsJ;H7&h5*x6o1(vu92g?yWxJ{6feaCz-@AL(L;TyyNA
z5-eROL6jVj$(%M>YxSL_%Mzd7dnhj2$is}jlzgSPe0^3CcIe7SQs(67?T%2%NtSnn
zV9w`9i5qm%%4I-opLTz5JBN@<=1x_Eom|~!yKo7*Szllfl6(=F3oYqYfyq}=I1^q*s5Rz(dg;N
z`Xje_no5wrUCGKP66MBHQ9Ak5SkggoyQBX6&}bjst(qq@b{EcAdC(R=^9E$S^K7Kp
zL+FHF)}1Ma{V-Bk+Ce8tc-vgT0Olg9VobK8y{hUu+Rv!oLQd`SP0=e7as*o=oA75Y
zG_%=mcttM8Io4)DVn7|P#^ZAt_A!sinvw8a@sBX;14yUjkXS4`L51eL7`%z2=~5v1c2j8b2DRe*@fx2K0xZ
z&iySRR%q^zQVZkGvjlE)`hVIwSniguEP%FEihDWQ-}C_`qzV8ya^K19Qtp+^L-*;g
zp1-ingk6dq#>m_cTSt)`+)tc<
z89L74yVP#{_|iHU_p(wJTlGZcxgqP&m-$TOXWe^a$^9?$qNaSvX177y5UaM(ubI|{
zF)9=rSzRT7&K@!<*;f^;(1FaWV15f{sPlS6;QSjKO`LX_#G$S?ecJbfNQ|kv?r$9_
zL@5oGP>~LYp5+2ofqpLM6`3=kp4LB>c9ME(%`nhsRcHucUT*A~o{vM1Q2b~9F;5)1&^qP0^lDLhBcMy?Q%$
zWY%6!S`Q|B)nT64T0L5DI5>BAK>@RIX#GI|n{K4a{-UOLVfYVH_E;GvsC1sVy4r2W
zJZutA43wDnHocyexz(0@@8OIZ+yus}=o}H(F?E=eP;{0_H6?Q0`}^cLAHzu0-|%UV
zIKMMC50kLVj;L^fZD@DH2wT!lcuPHFHb|w^V(V^${Ontk?D_ABUILHl*y^q46Oq-1
zm4s{AIH`(2wR~7m(EVQ7atY!v7p@(;z%i;nlD4L;iNkl0llGHM%+0!W+fk@{3Qmxh
z_E`<@08B0T_}q}X%!GUh;z1xuo7Q&4iGu0`*W|?*kOituYfeZ7nXV2_l`RVDOQs4h
zv@a@H>-%qNWq%c2jY?a`@^9kEC<2$xr^aGuhT(imlkG%`_Qjq!qE46Wmt6`s(6LvG
zj3{>vw!!T+6MVDp+s1&^mW+20jCdOUlD&~d^Lb<
zBoDEv_**Mw*cHHv=Wd)LMAd19d{vjZpbU;^iK`sQz%h)=Q{N|FXH+~T`EtFd_iXjQ
zoyaEKRu|&FViQ0_MoS#j+x!(%hWtzb41VbJ3*W09{t@x@#?AAaOF~QAmv<;ANl%vV
z(svOZp6)GlixlezSO%?gn&R%5S>tGFp1P{^61BUB03u7yDnVX9I(r&Vf8QctO$c~QA}@9PVpr>jfBO+kFva?#;5)u
zQCV6N2Dfv6+}`y`koGMF-=MGRe)D;@D_MH-SzHz$Mb_$Z{sYHD
zUV7f4&s(B?l*@2+SFX(~rHffF)~0Pk$`Gif17M(PNB+{7p2`X{m606@R)e~Xy&mo_!V2&nw>rYb>2kz;>i$ou{BCv{FKs?Djq
zGA(RZv?c
zc59c~MQDu}t<~J?z8~KAJzxIM@0{~I=fnB%|1(UhO~JMn%q#)`5C{ZlvaJS30hyX?
zYmJl#{KS6utSNF2@m0xU8I&vpj^D+w?zFv3!nQwewMTU
zy(%|zOD4o@SvTM2!409t${37rX<&%dyE_*>6#2*o4EKSxX?k|y#t8}BfRs*)lhaECZ*s-
zPE^^x}Ecfk@TqQder3d}-c{Y%67
zQ%8kQ#wK<*MNM2KMhV=d!d%8i)Nts#rQ7t-nkE<5ga%5&Fv4Ik-ZLZr#X-}n!F>3E
zOg=UiCA5}PaE!S})7b>^Yjc0P78OPQ)q^$@6Dh
zf@8buH>%GRzv=nlqsao<8rnSSA~Kk5E!aVh2Qdf1Kt2E&~CAk%p(L;RFbIfezz=csC=yG+R6idz72@-UxJvrrf7YQ!=!NIQW-@
z*n54Djq48j(Q&Dj^_V^?dxjck)@3q{?m@}Lk|ux`-i*MA2biC>->FgN9>orSR#Iw4*6yDJ-*wKa)kz${g4RL&FL7{Ww^1
zZlYxkSFYiwZSkijH4;-2T?+1%1^QKF{A7
zJW=yYygz}yJT{y(6-Hb3-S_NWa~P{aw!1kE|K1zi42tPf>>V{<(Zl#|)tK{^2-ddM
z_4KYyrZUkkekZP%_?PH;2FOR=LS_xiF~xkKq*SE?YdSmt;zd>&KQ2
zbc+C7tWa8mdoGz2h4EPfdfKq24=8`1g+c_i%&;Q}l6(PsB22W6RsUJsAmy%j5Q$Se
zQSjh+#lW3MvpG@JJsMf9p;9aL!$vZT8nUL8v;U^hGhHU1*8w##pKN5T(8PB4!wV(GL75Yx590eb?+;q`-dNJGjx${*b
z{~`cq$}(TiVzCH!Dco=0%}m>APu?n1MG8*uX*?wX!!96XhK@xw0CVa#($tA||CcF~
z)^S#L`klg^Ed?g@Uj+f6WI@P;O?l3xY5yRhN%RVaP|kKB8HycB(CY!}=A%8lgnn4Y
zcmW2j(c-zIt2lNRmYWZ?``mt<2krO2u@fHRSkVs~P4dBBuq~@FqTP?|TE3yLICd{R
z??n>K_g3UzxMdXH@SwHi)j;?lbI6*NjcTJwu`ql=)!i8Euw1hBYWZhK-?FRA$O>Or
z8`h7q(zR80l&xCs`s&2_q#Zn1_JSVT%Y}hvicN=CJf6?^a&hpEN+YedncV6x2O&+w
zhTp{ge!N4kMZfSUY(N?)_g}jslc2xX&AOXO^=6AG7^|l(V>+;Iue~IkWuj
zdgSpfu6Xx+vBt#v`A;9D#){q6`4R`pg4%Ov#$yzFq6T#Lcc%pe_yhQtDOXxEnl*3o
zJ2g-(s&+mJX7=%Dsq!50r@waVzy&VYCGV7d}6Go-biP>O;hSk$F~CPEzKT_3uKa^*?4dWAtcr6=QEF
z`I=h-o
z(-CEpv)0U-O+YZOxEpiAS2n_7=Na+}e%7dNvjVhB=q$z*<
zBc#Pkbx!p{3GEM|rWImb$NB9y^`UboKY3jrx0=o+6l<$R4@_c;jG&bWtc~yb!uB-2
z&YBUkH%odQuZeh9aU{Q%q!2G#FTRFKlO|U$WfMh4p#_QM&(WHkZno;8--5pJL2zp1
z{tCZ{fEJ&~@P6;cl!vU8EsQ(N_T(paYDQjwugLe_J;H0APV!RE(t>>5U+&KL
z;<5mvXeIAAC0e*k2Qp=bzdRO_BWjNCZa%7IJPklG7#|OXPQCN!t%MynI{=`e_W#yd
zO9dJKCzu