Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,15 @@ follow semantic versioning; release dates are ISO 8601.

### Public API

- **A header or footer sits as far from its page edge as the page puts it.** Nothing was
written, so Word used its own distance — 36pt — and the probe corpus's footer sat 14.5pt
higher than the page draws it, on every page. The engine does not state the distance
either: a zone is a band of a given height against the edge, with its content laid out
inside it from the top, so the distance is read from where the zone's content actually
landed in the resolved layout and written as `w:pgMar/@w:header` or `@w:footer`. Measured
through LibreOffice, the footer now lands within 0.6pt of the page; without a layout the
zone's own padding on that edge stands in.

- **A picture and a list keep the space they hold around themselves.** Neither wrote its own
box, so on the probe corpus an image holding 12pt at each edge ran straight into the
heading under it, and a four-item checklist came out 13pt short of the page —
Expand Down
10 changes: 8 additions & 2 deletions docs/recipes/docx-export.md
Original file line number Diff line number Diff line change
Expand Up @@ -296,8 +296,14 @@ tint it was flattened to. Recorded, like the other two.

Lines, ellipses, standalone shapes, and barcodes are **silently skipped**
— they are pure fixed-layout geometry with no semantic equivalent.
Headers/footers, watermarks, and protection options are also ignored by
the current exporter.
The text header and footer slots, watermarks, and protection options are
also ignored by the current exporter.

A page zone (`session.chrome().zone(...)`) is not: it exports as a real
Word header or footer part, with the page number as a live field, and it
sits as far from its page edge as the page puts it — the distance is read
from where the zone's content landed in the resolved layout and written as
`w:pgMar/@w:header` or `@w:footer`, rather than left to Word's 36pt.

The rule of thumb: if the document leans on geometry — shapes, layered
designs, precise placement — export PDF for the reader and DOCX only as
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -117,6 +117,44 @@ boolean isEmpty() {
return fragments.isEmpty();
}

/**
* How far a page zone's content sits from the page edge it belongs to, as laid out on
* the first page.
*
* <p>Word places a footer by the distance from the page's bottom edge to the bottom of
* the footer, and a header by the distance from the top edge to the top of the header.
* The engine does not state either: a zone is a band of a given height against the
* edge, and its content is laid out inside it from the top. So the distance is read
* from where the content actually landed — the lowest edge of a footer's fragments, the
* highest edge of a header's — rather than rebuilt from the band's parts.</p>
*
* <p>Zone fragments are spliced into the graph under {@code @page-zone[page][index]},
* outside the node paths this index is built from, so they are found by that prefix.</p>
*
* @param zoneIndex the zone's position in the session's zone list
* @param header whether it is a header, measured from the top edge
* @param pageHeight the page's height in points
* @return the distance in points, or empty when the layout carries no such zone
*/
OptionalDouble zoneDistanceFromEdge(int zoneIndex, boolean header, double pageHeight) {
String prefix = "@page-zone[0][" + zoneIndex + "]";
double lowest = Double.POSITIVE_INFINITY;
double highest = Double.NEGATIVE_INFINITY;
for (Map.Entry<String, List<PlacedFragment>> entry : fragments.entrySet()) {
if (!entry.getKey().startsWith(prefix)) {
continue;
}
for (PlacedFragment fragment : entry.getValue()) {
lowest = Math.min(lowest, fragment.y());
highest = Math.max(highest, fragment.y() + fragment.height());
}
}
if (lowest == Double.POSITIVE_INFINITY) {
return OptionalDouble.empty();
}
return OptionalDouble.of(header ? pageHeight - highest : lowest);
}

/**
* The path the layout graph addresses a node by, for a note that has to say where in
* the document it came from.
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -117,6 +117,7 @@
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.List;
import java.util.OptionalDouble;
import java.util.function.Function;
import java.util.concurrent.atomic.AtomicBoolean;

Expand Down Expand Up @@ -179,6 +180,9 @@ public final class DocxSemanticBackend implements SemanticBackend<byte[]> {
/** Space the last body paragraph holds below itself, not yet written — see {@link #owePendingSpacingAfter}. */
private double pendingSpacingAfter;

/** The page's height in points, or {@code NaN} when the export has no canvas. */
private double canvasHeight = Double.NaN;

/** The gap the list being written puts between its items. */
private double pendingItemSpacing;

Expand Down Expand Up @@ -307,6 +311,7 @@ public byte[] export(DocumentGraph graph, SemanticExportContext context) throws
currentCell = null;
currentCellWidth = Double.NaN;
contentWidth = context.canvas() == null ? Double.MAX_VALUE : context.canvas().innerWidth();
canvasHeight = context.canvas() == null ? Double.NaN : context.canvas().height();
try (XWPFDocument document = new XWPFDocument()) {
applyPageGeometry(document, context.canvas());
writeStylesPart(document);
Expand Down Expand Up @@ -377,14 +382,17 @@ private void applyOutputOptions(XWPFDocument document, DocumentOutputOptions opt
* absence, and the export says on the log what it could not honor.</p>
*/
private void applyPageZones(XWPFDocument document, List<DocumentPageZone> zones) {
// The page height the zones are measured against is the canvas's, which is what the
// page geometry was written from — not a value parsed back out of the XML.
if (zones == null || zones.isEmpty()) {
return;
}
XWPFHeaderFooterPolicy policy = document.getHeaderFooterPolicy();
if (policy == null) {
policy = document.createHeaderFooterPolicy();
}
for (DocumentPageZone zone : zones) {
for (int index = 0; index < zones.size(); index++) {
DocumentPageZone zone = zones.get(index);
if (zone.getAppliesTo() != null) {
LOG.warn("docx.zone.pagePredicate zone={} — appliesTo cannot be evaluated in a"
+ " semantic export: Word paginates the document, so there is no page to"
Expand All @@ -397,10 +405,43 @@ private void applyPageZones(XWPFDocument document, List<DocumentPageZone> zones)
if (content == null) {
continue;
}
XWPFHeaderFooter target = zone.getZone() == DocumentHeaderFooterZone.HEADER
boolean header = zone.getZone() == DocumentHeaderFooterZone.HEADER;
XWPFHeaderFooter target = header
? policy.createHeader(XWPFHeaderFooterPolicy.DEFAULT)
: policy.createFooter(XWPFHeaderFooterPolicy.DEFAULT);
writeZoneLine(target, content);
placeZone(document, zone, index, header);
}
}

/**
* Puts a header or footer as far from its page edge as the page puts it.
*
* <p>Nothing was written, so Word used its own distance — 36pt — and the probe's footer
* sat 14.5pt higher than the page draws it, on every page. Word holds the distance as
* {@code w:pgMar/@w:header} and {@code @w:footer}, so this is a mapping.</p>
*
* <p>The distance is where the zone's content landed in the resolved layout, which is
* the number the page was drawn with. Without a layout it falls back to the zone's own
* padding on that edge — a band's content is laid from its top, so for a footer that is
* the nearer estimate rather than the exact one, and it is only reached when the
* document could not be laid out at all.</p>
*/
private void placeZone(XWPFDocument document, DocumentPageZone zone, int index, boolean header) {
CTSectPr sectPr = document.getDocument().getBody().isSetSectPr()
? document.getDocument().getBody().getSectPr()
: document.getDocument().getBody().addNewSectPr();
CTPageMar margin = sectPr.isSetPgMar() ? sectPr.getPgMar() : sectPr.addNewPgMar();
double pageHeight = canvasHeight;
OptionalDouble measured = Double.isNaN(pageHeight)
? OptionalDouble.empty()
: layout.zoneDistanceFromEdge(index, header, pageHeight);
DocumentInsets padding = zone.getPadding() == null ? DocumentInsets.zero() : zone.getPadding();
double distance = measured.orElse(header ? padding.top() : padding.bottom());
if (header) {
margin.setHeader(BigInteger.valueOf(toTwips(distance)));
} else {
margin.setFooter(BigInteger.valueOf(toTwips(distance)));
}
}

Expand Down Expand Up @@ -1398,7 +1439,20 @@ private static void addSpacing(XWPFParagraph para, double before, double after)

/** Reads a twip measure back, treating an unset one as zero. */
private static long twipsOf(Object measure) {
return measure == null ? 0 : Long.parseLong(String.valueOf(measure));
Long twips = writtenTwips(measure);
return twips == null ? 0 : twips;
}

/**
* A twip value this export wrote, read back — or null when it is not a plain number.
*
* <p>XmlBeans hands a measure back as the schema's union, and a measure may legally be a
* string with a unit ({@code 1in}) or a percentage. This export only ever writes plain
* twips, which come back as a number; anything else is not one of its own values and is
* reported as unknown rather than parsed and thrown on.</p>
*/
private static Long writtenTwips(Object measure) {
return measure instanceof Number number ? number.longValue() : null;
}

private static void applyContainerPaint(XWPFParagraph para, ContainerPaint paint) {
Expand Down Expand Up @@ -2240,9 +2294,8 @@ private static double horizontalMarginsOf(XWPFTableCell cell) {
}

private static double marginPoints(CTTblWidth margin) {
return margin == null || margin.getW() == null
? WORD_DEFAULT_CELL_MARGIN_POINTS
: Long.parseLong(String.valueOf(margin.getW())) / POINT_TO_TWIP;
Long twips = margin == null ? null : writtenTwips(margin.getW());
return twips == null ? WORD_DEFAULT_CELL_MARGIN_POINTS : twips / POINT_TO_TWIP;
}

/** Word keeps this much clear inside every cell edge unless a table says otherwise. */
Expand Down Expand Up @@ -2841,7 +2894,12 @@ private static double usableWidthOf(XWPFTableCell cell, TableGrid.Placement plac
double twips = 0;
int last = Math.min(placement.column() + placement.colSpan(), grid.sizeOfGridColArray());
for (int index = placement.column(); index < last; index++) {
twips += Long.parseLong(String.valueOf(grid.getGridColArray(index).getW()));
Long column = writtenTwips(grid.getGridColArray(index).getW());
if (column == null) {
// A column this export did not write as plain twips has no width to add up.
return Double.NaN;
}
twips += column;
}
double points = twips / POINT_TO_TWIP - horizontalMarginsOf(cell);
return points > 0 ? points : Double.NaN;
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,146 @@
package com.demcha.compose.document.backend.semantic.docx;

import com.demcha.compose.GraphCompose;
import com.demcha.compose.document.api.DocumentSession;
import com.demcha.compose.document.backend.semantic.SemanticBackend;
import com.demcha.compose.document.backend.semantic.SemanticExportContext;
import com.demcha.compose.document.dsl.ParagraphBuilder;
import com.demcha.compose.document.layout.DocumentGraph;
import com.demcha.compose.document.layout.LayoutCanvas;
import com.demcha.compose.document.output.DocumentHeaderFooterZone;
import com.demcha.compose.document.output.DocumentOutputOptions;
import com.demcha.compose.document.output.DocumentPageZone;
import com.demcha.compose.document.style.DocumentInsets;
import org.apache.poi.xwpf.usermodel.XWPFDocument;
import org.junit.jupiter.api.Test;
import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTPageMar;

import java.io.ByteArrayInputStream;

import static org.assertj.core.api.Assertions.assertThat;

/**
* A header or footer sits as far from its page edge as the page puts it.
*
* <p>Nothing was written, so Word used its own distance — 36pt — and the probe's footer sat
* 14.5pt higher than the page draws it, on every page. The engine does not state the
* distance either: a zone is a band of a given height against the edge, with its content
* laid out inside it from the top. So the distance is read from where the content landed
* in the resolved layout.</p>
*
* @author Artem Demchyshyn
*/
class DocxPageZonePositionTest {

private static final double TWIPS_PER_POINT = 20.0;

@Test
void aFootersDistanceFollowsItsBandRatherThanWordsDefault() throws Exception {
// The one relation that holds whatever the font measures: content is laid from the
// band's top, so a band 30pt taller lifts its content 30pt further from the edge.
long shallow = footerDistance(zone(DocumentHeaderFooterZone.FOOTER, 30, DocumentInsets.zero()));
long deep = footerDistance(zone(DocumentHeaderFooterZone.FOOTER, 60, DocumentInsets.zero()));

assertThat(deep - shallow).isEqualTo(Math.round(30 * TWIPS_PER_POINT));
assertThat(shallow)
.as("inside the band, and not Word's 720-twip default")
.isBetween(0L, Math.round(30 * TWIPS_PER_POINT))
.isNotEqualTo(720L);
}

@Test
void aHeadersDistanceIsItsContentsTopFromThePageTop() throws Exception {
// A header's content starts at the band's top, so its padding is exactly the gap
// between the page's top edge and the content.
long distance = headerDistance(zone(DocumentHeaderFooterZone.HEADER, 40, new DocumentInsets(9, 0, 0, 0)));

assertThat(distance).isEqualTo(Math.round(9 * TWIPS_PER_POINT));
}

@Test
void withoutALayoutAFooterFallsBackToItsOwnPadding() throws Exception {
// A document that cannot be laid out still exports, and a zone's own padding on
// that edge is the nearest thing to the distance it states.
long distance = footerDistanceWithoutLayout(
zone(DocumentHeaderFooterZone.FOOTER, 30, new DocumentInsets(0, 0, 7, 0)));

assertThat(distance).isEqualTo(Math.round(7 * TWIPS_PER_POINT));
}

private static DocumentPageZone zone(DocumentHeaderFooterZone kind, double height, DocumentInsets padding) {
return DocumentPageZone.builder()
.zone(kind)
.height(height)
.padding(padding)
.content(page -> new ParagraphBuilder().name("ZoneLine").text("Chrome").build())
.build();
}

private static long footerDistance(DocumentPageZone zone) throws Exception {
CTPageMar margin = marginOf(exportWithLayout(zone));
return margin.getFooter() == null ? -1 : Long.parseLong(String.valueOf(margin.getFooter()));
}

private static long headerDistance(DocumentPageZone zone) throws Exception {
CTPageMar margin = marginOf(exportWithLayout(zone));
return margin.getHeader() == null ? -1 : Long.parseLong(String.valueOf(margin.getHeader()));
}

private static long footerDistanceWithoutLayout(DocumentPageZone zone) throws Exception {
CTPageMar margin = marginOf(exportWithoutLayout(zone));
return margin.getFooter() == null ? -1 : Long.parseLong(String.valueOf(margin.getFooter()));
}

private static CTPageMar marginOf(byte[] docx) throws Exception {
try (XWPFDocument document = new XWPFDocument(new ByteArrayInputStream(docx))) {
return (CTPageMar) document.getDocument().getBody().getSectPr().getPgMar().copy();
}
}

private static byte[] exportWithLayout(DocumentPageZone zone) throws Exception {
try (DocumentSession session = session(zone)) {
return session.export(new DocxSemanticBackend());
}
}

/** The same document handed to the backend with no layout, the way a bare caller does. */
private static byte[] exportWithoutLayout(DocumentPageZone zone) throws Exception {
Captured captured = new Captured();
try (DocumentSession session = session(zone)) {
session.export(captured);
return new DocxSemanticBackend().export(captured.graph,
new SemanticExportContext(captured.canvas, java.util.List.of(), null,
captured.options));
}
}

private static DocumentSession session(DocumentPageZone zone) {
DocumentSession session = GraphCompose.document()
.pageSize(400, 600)
.margin(DocumentInsets.of(40))
.create();
session.chrome().zone(zone);
session.pageFlow(page -> page.addParagraph(p -> p.text("Body")));
return session;
}

private static final class Captured implements SemanticBackend<byte[]> {

private DocumentGraph graph;
private LayoutCanvas canvas;
private DocumentOutputOptions options;

@Override
public String name() {
return "capture";
}

@Override
public byte[] export(DocumentGraph documentGraph, SemanticExportContext context) {
this.graph = documentGraph;
this.canvas = context.canvas();
this.options = context.outputOptions();
return new byte[0];
}
}
}
Loading