badgePieces(ParagraphNode paragraph) {
+ return DocxMarkdown.mayRead(paragraph) ? markdownPieces(paragraph, layout.lines(paragraph)) : null;
+ }
+
/** A paragraph held in no body, with no space above or below it: a shape's text. */
private static XWPFParagraph detachedParagraph(XWPFDocument document) {
org.openxmlformats.schemas.wordprocessingml.x2006.main.CTP markup =
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
index 9de5f2cd3..67eb7f2a4 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
@@ -19,9 +19,11 @@
import static org.assertj.core.api.Assertions.assertThat;
/**
- * A paragraph or a list item the page reads as markdown is named in the report: the page sets the
- * text its marks style and drops the marks, and the Word file holds the text as authored, marks and
- * all — on the paragraph's note, a zone paragraph's on the zone's, a list's items' on the list's.
+ * A paragraph the page reads as markdown is written as the page sets it, and not named
+ * ({@link DocxSessionMarkdownTest}); where the page's lines do not tell how it sets it — with no
+ * layout — it is written as authored and named. A list item the page reads as markdown is named in
+ * the report: the page sets the text its marks style and drops the marks, and the Word file holds
+ * the text as authored, marks and all — on the list's note.
*
* A session reads markdown unless it is told not to ({@code markdown(false)}), in a paragraph
* or a list item of plain text holding a mark of emphasis or code. Text the page sets as authored —
@@ -30,31 +32,29 @@
*/
class DocxMarkdownReportTest {
- private static final String MARKS = "its markdown marks are written as letters, where the page sets the text "
- + "they mark and drops them";
private static final String UNMEASURED = "markdown marks are written as letters — whether the page reads them is "
+ "not measured";
private static final String ITEMS = "its items' markdown marks are written as letters, where the page sets the "
+ "text they mark and drops them";
@Test
- void aParagraphThePageReadsAsMarkdownIsNamed() throws Exception {
- assertThat(paragraphNotes(true, page -> page.addParagraph("Some **bold** and `code` text")))
- .containsExactly("written as a paragraph; " + MARKS);
- assertThat(paragraphNotes(true, page -> page.addParagraph("# Title *now*"))).as("a heading")
- .containsExactly("written as a paragraph; " + MARKS);
- assertThat(paragraphNotes(true, page -> page.addParagraph("`x`"))).as("a code span alone")
- .containsExactly("written as a paragraph; " + MARKS);
- // A prefix with a mark of its own is laid out with it; the paragraph's marks still go.
+ void aParagraphThePageReadsAsMarkdownIsWrittenSoAndNotNamed() throws Exception {
+ assertThat(paragraphNotes(true, page -> page.addParagraph("Some **bold** and `code` text"))).isEmpty();
+ assertThat(paragraphNotes(true, page -> page.addParagraph("# Title *now*"))).as("a heading, past its line")
+ .containsExactly("written as a paragraph; its markdown heading is written at the size the page sets it, "
+ + "28pt, in a line only as tall as the paragraph's own: the page draws its letters past "
+ + "the line, and Word cuts their tops on screen");
+ assertThat(paragraphNotes(true, page -> page.addParagraph("`x`"))).as("a code span alone").isEmpty();
+ // A prefix with a mark of its own is laid out with it, and leads the letters the page sets.
assertThat(paragraphNotes(true, page -> page.addParagraph(p -> p.text("Some *emphasis* here")
.bulletOffset("* ").indentStrategy(DocumentTextIndent.FIRST_LINE))))
- .containsExactly("written as a paragraph; " + MARKS + "; its bulletOffset's letters, \"*\", are not "
- + "written before its first line");
+ .containsExactly("written as a paragraph; its bulletOffset's letters, \"*\", are not written before "
+ + "its first line");
assertThat(paragraphNotes(true, page -> page.addParagraph(p -> p.text("*x*")
.bulletOffset("** ").indentStrategy(DocumentTextIndent.FIRST_LINE))))
.as("as many marks in the prefix as the text drops")
- .containsExactly("written as a paragraph; " + MARKS + "; its bulletOffset's letters, \"**\", are not "
- + "written before its first line");
+ .containsExactly("written as a paragraph; its bulletOffset's letters, \"**\", are not written before "
+ + "its first line");
}
@Test
@@ -106,11 +106,13 @@ void whereTheLinesAreNotReadWhetherThePageReadsTheMarksIsNotMeasured() throws Ex
.containsExactly("written as a paragraph; its " + UNMEASURED);
assertThat(DocxExports.reportWithoutLayout(300, 400, 30, page -> page.addParagraph("Plain text"))
.bySubject()).as("no mark").doesNotContainKey("ParagraphNode");
- // Composed in a table cell, a paragraph is matched to its lines by its text, which the page
- // set otherwise than authored; a list's lines are not matched at all.
+ // Read line by line, as the page does: a list marker opening a line is kept.
+ assertThat(DocxExports.reportWithoutLayout(300, 400, 30, page -> page.addParagraph("* a_b"))
+ .bySubject()).as("a marker the page keeps, a mark it keeps").doesNotContainKey("ParagraphNode");
+ // Composed in a table cell, a paragraph is matched to its lines by its text as the page reads
+ // it, and written so; a list's lines are not matched at all.
assertThat(paragraphNotes(true, page -> page.add(cell(new ParagraphBuilder().name("Note")
- .text("Some **bold** text").build()))))
- .containsExactly("written as a paragraph; its " + UNMEASURED);
+ .text("Some **bold** text").build())))).isEmpty();
assertThat(paragraphNotes(true, page -> page.add(cell(new ParagraphBuilder().name("Note")
.text("Install node_js first").build())))).as("a mark the parser keeps").isEmpty();
assertThat(listNotes(true, page -> page.add(cell(new com.demcha.compose.document.dsl.ListBuilder()
@@ -122,10 +124,8 @@ void whereTheLinesAreNotReadWhetherThePageReadsTheMarksIsNotMeasured() throws Ex
}
@Test
- void aZoneParagraphThePageReadsAsMarkdownIsNamedOnTheZone() throws Exception {
- assertThat(zoneNotes(true)).containsExactly("a footer written as one line of Word's footer; a paragraph's "
- + "markdown marks are written as letters, where the page sets the "
- + "text they mark and drops them");
+ void aZoneParagraphThePageReadsAsMarkdownIsWrittenSoAndNotNamed() throws Exception {
+ assertThat(zoneNotes(true)).isEmpty();
assertThat(zoneNotes(false)).as("markdown off").isEmpty();
}
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
new file mode 100644
index 000000000..b59d94212
--- /dev/null
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
@@ -0,0 +1,159 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.document.layout.payloads.ParagraphLine;
+import com.demcha.compose.document.layout.payloads.ParagraphShapeSpan;
+import com.demcha.compose.document.layout.payloads.ParagraphSpan;
+import com.demcha.compose.document.layout.payloads.ParagraphTextSpan;
+import com.demcha.compose.document.style.DocumentLetterSpacing;
+import com.demcha.compose.document.style.DocumentTextDecoration;
+import com.demcha.compose.document.style.DocumentTextStyle;
+import com.demcha.compose.engine.components.content.text.TextDecoration;
+import com.demcha.compose.engine.components.content.text.TextStyle;
+import com.demcha.compose.font.FontName;
+import org.junit.jupiter.api.Test;
+
+import java.awt.Color;
+import java.util.ArrayList;
+import java.util.List;
+
+import static org.assertj.core.api.Assertions.assertThat;
+
+/**
+ * A paragraph's text is read as the page reads its markdown, and its pieces are the page's only
+ * where the lines it laid out hold their letters in their faces and sizes.
+ */
+class DocxMarkdownTest {
+
+ private static final DocumentTextStyle BODY = DocumentTextStyle.builder().size(10).build();
+
+ @Test
+ void textIsReadIntoThePiecesItsMarksStyle() {
+ assertThat(DocxMarkdown.read("Some **bold** and `code` text", BODY)).containsExactly(
+ piece("Some ", DocumentTextDecoration.DEFAULT, 10),
+ piece("bold", DocumentTextDecoration.BOLD, 10),
+ piece(" and code text", DocumentTextDecoration.DEFAULT, 10));
+ assertThat(DocxMarkdown.read("*a* ***b*** _c_", BODY)).containsExactly(
+ piece("a", DocumentTextDecoration.ITALIC, 10),
+ piece(" ", DocumentTextDecoration.DEFAULT, 10),
+ piece("b", DocumentTextDecoration.BOLD_ITALIC, 10),
+ piece(" ", DocumentTextDecoration.DEFAULT, 10),
+ piece("c", DocumentTextDecoration.ITALIC, 10));
+ assertThat(DocxMarkdown.read("Read the file_name [docs](https://x.org)", BODY)).as("a link keeps its text")
+ .containsExactly(piece("Read the file_name docs", DocumentTextDecoration.DEFAULT, 10));
+ }
+
+ @Test
+ void eachLineIsReadOnItsOwnAListMarkerKept() {
+ assertThat(DocxMarkdown.read("# Title *now*\nnext **b**\n- dash *i*\n\n", BODY)).containsExactly(
+ // The parser takes a heading's text as it stands, marks and all, bold at twice the size.
+ piece("Title *now*", DocumentTextDecoration.BOLD, 20),
+ piece("\nnext ", DocumentTextDecoration.DEFAULT, 10),
+ piece("b", DocumentTextDecoration.BOLD, 10),
+ piece("\n- dash ", DocumentTextDecoration.DEFAULT, 10),
+ piece("i", DocumentTextDecoration.ITALIC, 10),
+ piece("\n\n", DocumentTextDecoration.DEFAULT, 10));
+ }
+
+ @Test
+ void thePiecesTakeTheParsersFaceNotTheParagraphs() {
+ DocumentTextStyle bold = DocumentTextStyle.builder().size(10).decoration(DocumentTextDecoration.BOLD).build();
+ assertThat(DocxMarkdown.read("file_name *x*", bold)).containsExactly(
+ piece("file_name ", DocumentTextDecoration.DEFAULT, 10),
+ piece("x", DocumentTextDecoration.ITALIC, 10));
+ // A list marker is the paragraph's, face and all.
+ assertThat(DocxMarkdown.read("* *x*", bold)).containsExactly(
+ piece("* ", DocumentTextDecoration.BOLD, 10),
+ piece("x", DocumentTextDecoration.ITALIC, 10));
+ }
+
+ @Test
+ void aHeadingKeepsTheParagraphsTrackingInPoints() {
+ DocumentTextStyle tracked = DocumentTextStyle.builder().size(10)
+ .letterSpacing(DocumentLetterSpacing.ofFontSize(0.1)).build();
+ List pieces = DocxMarkdown.read("# Head\nbody *x*", tracked);
+ assertThat(pieces.get(0).style().size()).isEqualTo(20);
+ assertThat(pieces.get(0).style().letterSpacing()).isEqualTo(DocumentLetterSpacing.points(1.0));
+ assertThat(pieces.get(1).style().letterSpacing()).as("the body's own").isEqualTo(tracked.letterSpacing());
+ }
+
+ @Test
+ void thePiecesAreThePagesWhereItsLinesHoldThemSoAndNotOtherwise() {
+ List pieces = DocxMarkdown.read("Some **bold** text", BODY);
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some", TextDecoration.DEFAULT, 10),
+ span(" ", TextDecoration.DEFAULT, 10), span("bold", TextDecoration.BOLD, 10),
+ span(" text", TextDecoration.DEFAULT, 10))), "", false)).isTrue();
+ // Broken over two lines, the space at the break dropped.
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some ", TextDecoration.DEFAULT, 10),
+ span("bold", TextDecoration.BOLD, 10)), line(span("text", TextDecoration.DEFAULT, 10))), "", false)).isTrue();
+ // An auto-sized paragraph's text, at a size its style does not hold, in proportion.
+ List fitted = List.of(line(span("A *b*", TextDecoration.BOLD, 14)),
+ line(span("c", TextDecoration.DEFAULT, 7)));
+ assertThat(DocxMarkdown.laidOutIn(DocxMarkdown.read("# A *b*\nc", BODY), fitted, "", true)).isTrue();
+ assertThat(DocxMarkdown.laidOutIn(DocxMarkdown.read("# A *b*\nc", BODY), fitted, "", false))
+ .as("in proportion, where the page fits no size of its own").isFalse();
+ // A prefix the page sets before the first line leads its letters.
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("• ", TextDecoration.DEFAULT, 10),
+ span("Some ", TextDecoration.DEFAULT, 10), span("bold", TextDecoration.BOLD, 10),
+ span(" text", TextDecoration.DEFAULT, 10))), "• ", false)).isTrue();
+ // Marks alone the page sets as nothing, which are not taken for the page's.
+ assertThat(DocxMarkdown.read("***", BODY)).isEmpty();
+ assertThat(DocxMarkdown.laidOutIn(List.of(), List.of(line()), "", false)).isFalse();
+
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some **bold** text", TextDecoration.DEFAULT, 10))),
+ "", false)).as("the marks laid out: the session reads no markdown").isFalse();
+ assertThat(DocxMarkdown.laidOutIn(List.of(), List.of(line(span("***", TextDecoration.DEFAULT, 10))), "", false))
+ .as("marks alone laid out").isFalse();
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some bold text", TextDecoration.DEFAULT, 10))),
+ "", false)).as("another face").isFalse();
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some ", TextDecoration.DEFAULT, 10),
+ span("bold", TextDecoration.BOLD, 12), span(" text", TextDecoration.DEFAULT, 10))), "", true))
+ .as("sizes out of proportion").isFalse();
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some ", TextDecoration.DEFAULT, 10),
+ span("bold", TextDecoration.BOLD, 10), span(" text", TextDecoration.DEFAULT, 10, FontName.COURIER,
+ Color.BLACK))), "", false)).as("another family").isFalse();
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some ", TextDecoration.DEFAULT, 10),
+ span("bold", TextDecoration.BOLD, 10), span(" text", TextDecoration.DEFAULT, 10, BODY.fontName(),
+ Color.RED))), "", false)).as("another colour").isFalse();
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some ", TextDecoration.DEFAULT, 10),
+ span("bolt", TextDecoration.BOLD, 10), span(" text", TextDecoration.DEFAULT, 10))), "", false))
+ .as("another letter").isFalse();
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some ", TextDecoration.DEFAULT, 10),
+ span("bold", TextDecoration.BOLD, 10))), "", false)).as("a letter short").isFalse();
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("- Some ", TextDecoration.DEFAULT, 10),
+ span("bold", TextDecoration.BOLD, 10), span(" text", TextDecoration.DEFAULT, 10))), "• ", false))
+ .as("a prefix of other letters").isFalse();
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(), "", false)).as("no lines").isFalse();
+ List withAPicture = new ArrayList<>(line(span("Some ", TextDecoration.DEFAULT, 10),
+ span("bold", TextDecoration.BOLD, 10), span(" text", TextDecoration.DEFAULT, 10)).spans());
+ withAPicture.add(new ParagraphShapeSpan(List.of(), 4, 4, null, 0, null));
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(new ParagraphLine("", 0, 10, 10, 8, 2, withAPicture)), "", false))
+ .as("anything but text").isFalse();
+ }
+
+ @Test
+ void whatThePageMayReadAsMarkdownHoldsAMarkOfEmphasisOrCode() {
+ assertThat(DocxMarkdown.holdsAMark("a *b*")).isTrue();
+ assertThat(DocxMarkdown.holdsAMark("snake_case")).isTrue();
+ assertThat(DocxMarkdown.holdsAMark("`x`")).isTrue();
+ assertThat(DocxMarkdown.holdsAMark("# Title [link](u)")).as("a heading or a link alone").isFalse();
+ assertThat(DocxMarkdown.holdsAMark(null)).isFalse();
+ }
+
+ private static DocxMarkdown.Piece piece(String text, DocumentTextDecoration face, double size) {
+ return new DocxMarkdown.Piece(text, new DocumentTextStyle(BODY.fontName(), size, face, BODY.color(),
+ BODY.letterSpacing()));
+ }
+
+ private static ParagraphTextSpan span(String text, TextDecoration face, double size) {
+ return span(text, face, size, BODY.fontName(), BODY.color().color());
+ }
+
+ private static ParagraphTextSpan span(String text, TextDecoration face, double size, FontName family, Color color) {
+ return new ParagraphTextSpan(text, new TextStyle(family, size, face, color), text.length() * 5.0,
+ size, null, null, false);
+ }
+
+ private static ParagraphLine line(ParagraphSpan... spans) {
+ return new ParagraphLine("", 100, 12, 12, 9, 3, List.of(spans));
+ }
+}
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxNodeFieldLedgerTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxNodeFieldLedgerTest.java
index a5dafdbde..72f4572c9 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxNodeFieldLedgerTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxNodeFieldLedgerTest.java
@@ -195,8 +195,10 @@ private record Entry(Fate fate, String note) {
"padding:REPORTED:the sides its alignment sets it from",
"margin:REPORTED:the sides its alignment sets it from");
node(ParagraphNode.class, "name:INERT",
- "text:REPORTED:where the page reads it as markdown, its marks written as letters; where its "
- + "lines are not read and the page's parser drops a mark, not measured; any other is written",
+ "text:REPORTED:where the page reads it as markdown, written as the page sets it where its lines "
+ + "hold the pieces read so, a heading larger than its line named; its marks written as letters "
+ + "where the lines hold other letters or none, and not measured where they are not read and the "
+ + "page's parser drops a mark; any other is written",
"inlineRuns:REPORTED:a chip's translucent fill, flattened against the colour under it, with what "
+ "else of its shape Word cannot hold; a run's translucent colour is written as Word's text fill",
"textStyle:WRITTEN", "align:WRITTEN", "lineSpacing:WRITTEN",
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionMarkdownTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionMarkdownTest.java
new file mode 100644
index 000000000..b3acdd0ef
--- /dev/null
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionMarkdownTest.java
@@ -0,0 +1,361 @@
+package com.demcha.compose.document.backend.semantic.docx;
+
+import com.demcha.compose.GraphCompose;
+import com.demcha.compose.document.api.DocumentSession;
+import com.demcha.compose.document.dsl.PageFlowBuilder;
+import com.demcha.compose.document.dsl.ParagraphBuilder;
+import com.demcha.compose.document.dsl.ShapeBuilder;
+import com.demcha.compose.document.dsl.ShapeContainerBuilder;
+import com.demcha.compose.document.layout.PlacedFragment;
+import com.demcha.compose.document.layout.payloads.ParagraphFragmentPayload;
+import com.demcha.compose.document.layout.payloads.ParagraphTextSpan;
+import com.demcha.compose.document.node.DocumentBookmarkOptions;
+import com.demcha.compose.document.node.DocumentLinkOptions;
+import com.demcha.compose.document.node.LayerAlign;
+import com.demcha.compose.document.node.TextAlign;
+import com.demcha.compose.document.node.TextDirection;
+import com.demcha.compose.document.output.DocumentPageZone;
+import com.demcha.compose.document.style.ClipPolicy;
+import com.demcha.compose.document.style.DocumentColor;
+import com.demcha.compose.document.style.DocumentInsets;
+import com.demcha.compose.document.style.DocumentTextDecoration;
+import com.demcha.compose.document.style.DocumentTextStyle;
+import com.demcha.compose.document.table.DocumentTableCell;
+import com.demcha.compose.document.table.DocumentTableColumn;
+import com.demcha.compose.engine.components.content.text.TextDecoration;
+import com.demcha.compose.font.FontName;
+import org.apache.poi.openxml4j.opc.PackagePart;
+import org.apache.poi.xwpf.usermodel.XWPFDocument;
+import org.apache.poi.xwpf.usermodel.XWPFParagraph;
+import org.apache.poi.xwpf.usermodel.XWPFRun;
+import org.junit.jupiter.api.Test;
+
+import java.io.ByteArrayInputStream;
+import java.io.InputStream;
+import java.nio.charset.StandardCharsets;
+import java.util.List;
+import java.util.Map;
+import java.util.concurrent.atomic.AtomicReference;
+import java.util.function.Consumer;
+
+import static org.assertj.core.api.Assertions.assertThat;
+
+/**
+ * A paragraph the page reads as markdown is written as the page sets it: its marks dropped, the
+ * text they mark in the face — and a heading at the size — the page sets it in, and nothing named
+ * but a heading the page draws past its line. The page's own lines say whether it reads the
+ * paragraph so; where they hold the marks, or other letters than the pieces, the paragraph is
+ * written as authored and named.
+ */
+class DocxSessionMarkdownTest {
+
+ private static final String MARKS = "its markdown marks are written as letters, where the page sets the text "
+ + "they mark and drops them";
+ private static final String HEADING_CUT = "its markdown heading is written at the size the page sets it, 20pt, "
+ + "in a line only as tall as the paragraph's own: the page draws its "
+ + "letters past the line, and Word cuts their tops on screen";
+
+ @Test
+ void aParagraphIsWrittenInThePiecesThePageSetsIt() throws Exception {
+ Export export = export(true, page -> page.addParagraph(p -> p.name("Intro").text("Some **bold** and `code` text")));
+ XWPFParagraph paragraph = export.paragraphWith("bold");
+ assertThat(paragraph.getText()).isEqualTo("Some bold and code text");
+ assertThat(paragraph.getRuns()).extracting(XWPFRun::text).containsExactly("Some ", "bold", " and code text");
+ assertThat(paragraph.getRuns()).extracting(XWPFRun::isBold).containsExactly(false, true, false);
+ assertThat(export.notes("ParagraphNode")).isEmpty();
+ }
+
+ @Test
+ void aHeadingLineIsWrittenBoldAtTheSizeThePageSetsItAndEachLineBreaks() throws Exception {
+ Export export = export(true, page -> page.addParagraph(p -> p.text("# Title\nnext *word*\n- dash")
+ .textStyle(DocumentTextStyle.builder().size(10).build())));
+ XWPFParagraph paragraph = export.paragraphWith("Title");
+ List runs = paragraph.getRuns();
+ assertThat(runs.get(0).text()).isEqualTo("Title");
+ assertThat(runs.get(0).isBold()).isTrue();
+ assertThat(runs.get(0).getFontSizeAsDouble()).isEqualTo(20.0);
+ assertThat(runs).filteredOn(run -> "word".equals(run.text())).singleElement()
+ .satisfies(run -> assertThat(run.isItalic()).isTrue());
+ assertThat(paragraph.getCTP().xmlText()).as("a line break where the page starts a line").contains("");
+ assertThat(paragraph.getText()).contains("- dash").doesNotContain("#").doesNotContain("*");
+ // The page sets the heading in a line as tall as the paragraph's own, and draws it past.
+ assertThat(export.notes("ParagraphNode")).containsExactly("written as a paragraph; " + HEADING_CUT);
+
+ // The paragraph's mark closes its last line, sized as the piece that ends it.
+ XWPFParagraph heading = export(true, page -> page.addParagraph(p -> p.text("# Title *x*")
+ .textStyle(DocumentTextStyle.builder().size(10).build()))).paragraphWith("Title");
+ assertThat(heading.getCTP().getPPr().getRPr().getSzArray(0).getVal()).hasToString("40");
+ }
+
+ @Test
+ void aMarkOfDirectionIsWrittenAsThePageKeepsIt() throws Exception {
+ Export export = export(true, page -> page.addParagraph("Price **now**"));
+ XWPFParagraph paragraph = export.paragraphWith("now");
+ assertThat(paragraph.getRuns()).extracting(XWPFRun::isBold).containsExactly(false, true);
+ assertThat(paragraph.getText()).isEqualTo("Price now");
+ assertThat(export.notes("ParagraphNode")).isEmpty();
+ }
+
+ @Test
+ void aRightToLeftParagraphIsWrittenAsThePageSetsIt() throws Exception {
+ Export export = export(true, page -> page.addParagraph(p -> p.text("שלום **עולם**").direction(TextDirection.RTL)));
+ XWPFParagraph paragraph = export.paragraphWith("עולם");
+ assertThat(paragraph.getRuns()).extracting(XWPFRun::text).containsExactly("שלום ", "עולם");
+ assertThat(paragraph.getRuns()).extracting(XWPFRun::isBold).containsExactly(false, true);
+ assertThat(paragraph.getRuns()).allSatisfy(run -> assertThat(run.getCTR().getRPr().xmlText()).contains(" page.addParagraph(p -> p.name("Long").text(sentence.repeat(40))));
+ assertThat(export.fragmentsOf("Long")).as("the page breaks it across pages").isGreaterThan(1);
+ XWPFParagraph paragraph = export.paragraphWith("runs on");
+ assertThat(paragraph.getText()).doesNotContain("*");
+ assertThat(paragraph.getRuns()).filteredOn(XWPFRun::isBold).hasSize(40);
+ assertThat(export.notes("ParagraphNode")).isEmpty();
+ }
+
+ @Test
+ void marksAloneThePageSetsAsNothingAreWrittenAsTheyStandAndNamed() throws Exception {
+ // The page reads `***` as a rule and a lone `*` as an empty list item, and sets nothing.
+ for (String marks : List.of("***", "*")) {
+ Export export = export(true, page -> page.addParagraph("Above").addParagraph(marks).addParagraph("Below"));
+ assertThat(export.document().getDocument().xmlText()).contains(">" + marks + "<");
+ assertThat(export.notes("ParagraphNode")).containsExactly("written as a paragraph; its markdown marks are "
+ + "written as letters, where the page reads them as marks alone and sets nothing");
+ }
+ Export off = export(false, page -> page.addParagraph("***"));
+ assertThat(off.document().getDocument().xmlText()).as("markdown off, the page sets the marks").contains(">***<");
+ assertThat(off.notes("ParagraphNode")).isEmpty();
+ }
+
+ @Test
+ void aSessionThatReadsNoMarkdownHasItsMarksWrittenAsTheyStand() throws Exception {
+ Export export = export(false, page -> page.addParagraph("Some **bold** text"));
+ XWPFParagraph paragraph = export.paragraphWith("bold");
+ assertThat(paragraph.getText()).isEqualTo("Some **bold** text");
+ assertThat(paragraph.getRuns()).singleElement().satisfies(run -> assertThat(run.isBold()).isFalse());
+ assertThat(export.notes("ParagraphNode")).isEmpty();
+ }
+
+ @Test
+ void thePiecesStandInTheFacesThePageSetsThemIn() throws Exception {
+ // The page reads the text of a bold paragraph holding a mark into faces of the parser's
+ // own, the paragraph's left aside: it sets Senior_Engineer regular, and the file follows.
+ // Should the page come to keep the paragraph's face, this fails, and the docs that say so
+ // are to change with it.
+ DocumentTextStyle bold = DocumentTextStyle.builder().decoration(DocumentTextDecoration.BOLD).build();
+ Export export = export(true, page -> page.addParagraph(p -> p.name("Role").text("Senior_Engineer").textStyle(bold)));
+ assertThat(export.firstSpanFace("Role")).isEqualTo(TextDecoration.DEFAULT);
+ assertThat(export.paragraphWith("Senior_Engineer").getRuns()).singleElement()
+ .satisfies(run -> assertThat(run.isBold()).isFalse());
+ assertThat(export.notes("ParagraphNode")).isEmpty();
+ }
+
+ @Test
+ void aLinkedParagraphsPiecesShareOneLink() throws Exception {
+ Export export = export(true, page -> page.addParagraph(p -> p.text("**Java** docs")
+ .link(new DocumentLinkOptions("https://example.org"))));
+ XWPFParagraph paragraph = export.paragraphWith("docs");
+ assertThat(paragraph.getCTP().sizeOfHyperlinkArray()).isEqualTo(1);
+ assertThat(paragraph.getCTP().getHyperlinkArray(0).sizeOfRArray()).isEqualTo(2);
+ assertThat(paragraph.getText()).isEqualTo("Java docs");
+ }
+
+ @Test
+ void anOutlineEntryListsTheTextAsWritten() throws Exception {
+ Export export = export(true, page -> page.addParagraph(p -> p.text("**Results**")
+ .bookmark(new DocumentBookmarkOptions("Results", 1))));
+ assertThat(export.paragraphWith("Results").getText()).isEqualTo("Results");
+ assertThat(export.notes("ParagraphNode")).isEmpty();
+ }
+
+ @Test
+ void aZoneParagraphIsWrittenAsThePageSetsIt() throws Exception {
+ AtomicReference report = new AtomicReference<>();
+ String footer = zoneFooter("**Confidential**", report);
+ assertThat(footer).contains(">Confidential<").doesNotContain("**").contains("");
+ assertThat(report.get().bySubject().getOrDefault("page zone", List.of())).isEmpty();
+ // A heading in a zone stands taller than the zone's line, as in the body.
+ zoneFooter("# Confidential *now*", report);
+ assertThat(report.get().bySubject().get("page zone")).extracting(DocxExportReport.Note::detail)
+ .containsExactly("a footer written as one line of Word's footer; a paragraph's markdown heading is "
+ + "written at the size the page sets it, 16pt, in a line only as tall as the "
+ + "paragraph's own: the page draws its letters past the line, and Word cuts their "
+ + "tops on screen");
+ }
+
+ private static String zoneFooter(String text, AtomicReference report) throws Exception {
+ byte[] docx;
+ try (DocumentSession session = GraphCompose.document().pageSize(300, 400).margin(DocumentInsets.of(36)).create()) {
+ session.chrome().zone(DocumentPageZone.footer(30, page -> new ParagraphBuilder().name("ZoneLine")
+ .text(text).textStyle(DocumentTextStyle.DEFAULT.withSize(8)).build()));
+ session.pageFlow(page -> page.addParagraph("Body"));
+ docx = session.export(new DocxSemanticBackend(report::set));
+ }
+ try (XWPFDocument document = new XWPFDocument(new ByteArrayInputStream(docx))) {
+ return partXml(document, "/word/footer");
+ }
+ }
+
+ @Test
+ void aParagraphComposedInATableCellIsWrittenAsThePageSetsIt() throws Exception {
+ Export export = export(true, page -> page.add(new com.demcha.compose.document.dsl.TableBuilder().name("Rota")
+ .columns(DocumentTableColumn.fixed(200))
+ .rowCells(DocumentTableCell.node(new ParagraphBuilder().name("Note").text("Some **bold** text").build()))
+ .build()));
+ XWPFRun bold = export.document().getTables().get(0).getRow(0).getCell(0).getParagraphs().stream()
+ .flatMap(paragraph -> paragraph.getRuns().stream()).filter(run -> "bold".equals(run.text()))
+ .findFirst().orElseThrow();
+ assertThat(bold.isBold()).isTrue();
+ assertThat(export.document().getDocument().xmlText()).doesNotContain("**bold**");
+ assertThat(export.notes("ParagraphNode")).isEmpty();
+ }
+
+ @Test
+ void aCellsParagraphKeepsItsOwnLinesBesideOneReadAsMarkdown() throws Exception {
+ // Laid out with its prefix, the first cell's line reads "• Total", its text neither as
+ // authored nor as the page reads it; the cell below keeps its own line, "Total", rather than
+ // lend it to the first, and is written at the page's line height.
+ Export export = export(true, page -> page.add(new com.demcha.compose.document.dsl.TableBuilder().name("Sums")
+ .columns(DocumentTableColumn.fixed(200))
+ .rowCells(DocumentTableCell.node(new ParagraphBuilder().name("First").text("**Total**").bulletOffset("•")
+ .indentStrategy(com.demcha.compose.document.style.DocumentTextIndent.FIRST_LINE).build()))
+ .rowCells(DocumentTableCell.node(new ParagraphBuilder().name("Second").text("Total").build()))
+ .build()));
+ XWPFParagraph second = export.document().getTables().get(0).getRow(1).getCell(0).getParagraphs().get(0);
+ assertThat(second.getCTP().getPPr().getSpacing().getLineRule())
+ .isEqualTo(org.openxmlformats.schemas.wordprocessingml.x2006.main.STLineSpacingRule.EXACT);
+ }
+
+ @Test
+ void aBadgesInitialsAreWrittenAsThePageSetsThem() throws Exception {
+ Export export = export(true, page -> page.add(badge("*JR*")));
+ String body = export.document().getDocument().xmlText();
+ assertThat(body).contains("").contains(">JR<").doesNotContain("*JR*").contains("");
+ assertThat(export.notes("ParagraphNode")).isEmpty();
+ // Four letters and more with their marks, two as the page sets them: a badge's initials.
+ assertThat(export(true, page -> page.add(badge("**JR**"))).document().getDocument().xmlText())
+ .contains("").contains(">JR<").contains("");
+ // Initials in two faces are written in the flow, as initials of two runs' faces are.
+ Export twoFaces = export(true, page -> page.add(badge("*J*R")));
+ assertThat(twoFaces.document().getDocument().xmlText()).doesNotContain("");
+ assertThat(twoFaces.paragraphWith("JR").getRuns()).filteredOn(run -> !run.text().isEmpty())
+ .extracting(XWPFRun::text, XWPFRun::isItalic)
+ .containsExactly(org.assertj.core.groups.Tuple.tuple("J", true), org.assertj.core.groups.Tuple.tuple("R", false));
+ }
+
+ @Test
+ void aLinePairsSidesAreWrittenAsThePageSetsThem() throws Exception {
+ Export export = export(true, page -> page.add(new ShapeContainerBuilder().name("EntryHead").rectangle(240, 20)
+ .clipPolicy(ClipPolicy.OVERFLOW_VISIBLE)
+ .position(new ParagraphBuilder().name("Title").text("**Lead** Engineer")
+ .bookmark(new DocumentBookmarkOptions("Lead Engineer 2022", 1)).build(), 0, 0, LayerAlign.CENTER_LEFT)
+ .position(new ParagraphBuilder().name("Dates").text("*2022*").align(TextAlign.RIGHT).build(),
+ 0, 0, LayerAlign.CENTER_RIGHT)
+ .build()));
+ XWPFParagraph line = export.paragraphWith("Engineer");
+ assertThat(line.getText()).isEqualTo("Lead Engineer\t2022");
+ assertThat(line.getRuns()).filteredOn(run -> "Lead".equals(run.text())).singleElement()
+ .satisfies(run -> assertThat(run.isBold()).isTrue());
+ assertThat(line.getRuns()).filteredOn(run -> "2022".equals(run.text())).singleElement()
+ .satisfies(run -> assertThat(run.isItalic()).isTrue());
+ // Word's outline lists the line's text as written.
+ assertThat(export.notes("ParagraphNode")).isEmpty();
+ }
+
+ @Test
+ void textOverTheFlowIsWrittenAsThePageSetsIt() throws Exception {
+ Export export = export(true, page -> page.add(new ShapeContainerBuilder().name("Sidebar")
+ .rectangle(100, 120).clipPolicy(ClipPolicy.OVERFLOW_VISIBLE)
+ .margin(new DocumentInsets(-30, 0, -90, -30))
+ .position(new ShapeBuilder().name("Block").size(100, 120)
+ .fillColor(DocumentColor.rgb(160, 80, 50)).build(), 0, 0, LayerAlign.TOP_LEFT, 0)
+ .position(new ParagraphBuilder().name("Monogram").text("**LM**").build(), 10, 10, LayerAlign.TOP_LEFT, 1)
+ .build()).addParagraph("Masthead"));
+ String body = export.document().getDocument().xmlText();
+ assertThat(body).contains("").contains(">LM<").doesNotContain("**LM**");
+ assertThat(export.notes("ParagraphNode")).allSatisfy(note -> assertThat(note).doesNotContain("markdown"));
+ }
+
+ private static com.demcha.compose.document.node.DocumentNode badge(String initials) {
+ return new ShapeContainerBuilder().name("Badge").circle(40).fillColor(DocumentColor.rgb(30, 50, 90))
+ .center(new ParagraphBuilder().text(initials).build()).build();
+ }
+
+ @Test
+ void theFacesThePiecesAreSetInTravelWithTheDocument() throws Exception {
+ Export export = export(true, page -> page.addParagraph(p -> p.text("Set **in** Lato *here*")
+ .textStyle(DocumentTextStyle.builder().fontName(FontName.LATO).build())));
+ String table = partXml(export.document(), "/word/fontTable.xml");
+ assertThat(table).contains(" page.addParagraph(p -> p.text("مرحبا **بالعالم**")
+ .textStyle(DocumentTextStyle.builder().fontName(FontName.AMIRI).build())));
+ assertThat(export.document().getDocument().xmlText()).contains("**");
+ assertThat(export.notes("ParagraphNode")).containsExactly("written as a paragraph; " + MARKS);
+ }
+
+ private record Export(XWPFDocument document, DocxExportReport report,
+ Map> fragments) {
+
+ XWPFParagraph paragraphWith(String text) {
+ return document.getParagraphs().stream().filter(paragraph -> paragraph.getText().contains(text))
+ .findFirst().orElseThrow(() -> new AssertionError("no paragraph holds " + text));
+ }
+
+ long fragmentsOf(String name) {
+ return fragments.entrySet().stream().filter(entry -> entry.getKey().contains(name))
+ .mapToLong(entry -> entry.getValue().size()).sum();
+ }
+
+ List notes(String subject) {
+ return report.bySubject().getOrDefault(subject, List.of()).stream().map(DocxExportReport.Note::detail).toList();
+ }
+
+ TextDecoration firstSpanFace(String name) {
+ return fragments.entrySet().stream().filter(entry -> entry.getKey().contains(name))
+ .flatMap(entry -> entry.getValue().stream())
+ .map(PlacedFragment::payload).filter(ParagraphFragmentPayload.class::isInstance)
+ .map(ParagraphFragmentPayload.class::cast)
+ .flatMap(payload -> payload.lines().stream()).flatMap(line -> line.spans().stream())
+ .filter(ParagraphTextSpan.class::isInstance).map(ParagraphTextSpan.class::cast)
+ .findFirst().orElseThrow().textStyle().decoration();
+ }
+ }
+
+ private static Export export(boolean markdown, Consumer content) throws Exception {
+ AtomicReference report = new AtomicReference<>();
+ byte[] docx;
+ Map> fragments;
+ try (DocumentSession session = GraphCompose.document().pageSize(300, 400).margin(DocumentInsets.of(30))
+ .markdown(markdown).create()) {
+ session.pageFlow(content::accept);
+ fragments = session.layoutGraph().fragments().stream()
+ .collect(java.util.stream.Collectors.groupingBy(PlacedFragment::path));
+ docx = session.export(new DocxSemanticBackend(report::set));
+ }
+ return new Export(new XWPFDocument(new ByteArrayInputStream(docx)), report.get(), fragments);
+ }
+
+ private static String partXml(XWPFDocument document, String prefix) throws Exception {
+ StringBuilder xml = new StringBuilder();
+ for (PackagePart part : document.getPackage().getParts()) {
+ if (part.getPartName().getName().startsWith(prefix)) {
+ try (InputStream in = part.getInputStream()) {
+ xml.append(new String(in.readAllBytes(), StandardCharsets.UTF_8));
+ }
+ }
+ }
+ assertThat(xml).as("the document holds " + prefix).isNotEmpty();
+ return xml.toString();
+ }
+}
From 3799efad65a674b4cc36521b0ef9a8118860df27 Mon Sep 17 00:00:00 2001
From: DemchaAV
Date: Wed, 7 Oct 2026 16:10:14 +0100
Subject: [PATCH 2/2] fix(docx): pair a markdown cell only to lines that set
it, and name a heading only where it is cut
A paragraph composed in a table cell is matched by its text as the
page reads it only to a fragment whose lines set its pieces. A plain
paragraph of the same text after it may have taken its own; the one
left is in another face or size, and held to it, the paragraph's
letters were cut in Word with no note but its marks.
A markdown heading is named where it is written taller than the line
the page sets it in, by its own line's height: an auto-sized
paragraph's heading, written at a multiple of its style's size, may fit
the line the page fits the text to. Initials that are a heading are
written in the flow, as initials in two faces are.
The pieces are compared with the page's lines in tracking too, where
the page fits no size of its own. Pieces that change nothing - every
mark kept, every piece in the paragraph's style - leave the text written
as it stands, its white space and tabs with it. Text the parser reads
into nothing, a lone `*`, `***` or a line four spaces in, is named for
what it is.
---
CHANGELOG.md | 54 +++++----
.../architecture/backend-capability-matrix.md | 2 +-
docs/recipes/docx-export.md | 2 +-
render-docx/README.md | 8 +-
.../semantic/docx/DocxLayoutMetrics.java | 32 ++++--
.../backend/semantic/docx/DocxMarkdown.java | 43 +++----
.../semantic/docx/DocxSemanticBackend.java | 67 +++++++----
.../semantic/docx/DocxMarkdownReportTest.java | 7 +-
.../semantic/docx/DocxMarkdownTest.java | 4 +
.../docx/DocxNodeFieldLedgerTest.java | 2 +-
.../docx/DocxSessionMarkdownTest.java | 106 ++++++++++++++----
11 files changed, 222 insertions(+), 105 deletions(-)
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 37301722d..cfac3121d 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -15,40 +15,46 @@ follow semantic versioning; release dates are ISO 8601.
export wrote the text as authored, so Word showed `**bold**` with its asterisks and none of the
bold, and the report named it.
- **The text is read as the page reads it** — line by line, through the page's own parser, a
- list marker opening a line kept — and written as one run a piece, in the face, family, colour
- and size the page sets it in: `**Java**` is a bold run reading `Java`, a heading line bold at
- its larger size, a line break where the page starts a line. A linked paragraph's pieces stay
- in one link, and Word's outline lists a heading by the text written. An auto-sized paragraph's pieces are written at
- its style's size, a heading's at its multiple of it, as its text was.
+ list marker opening a line kept — and written as one run a piece, in the face, family, colour,
+ tracking and size the page sets it in: `**Java**` is a bold run reading `Java`, a heading line
+ bold at its larger size, a line break where the page starts a line. A linked paragraph's
+ pieces stay in one link, and Word's outline lists a heading by the text written. An
+ auto-sized paragraph's pieces are written at its style's size, a heading's at its multiple of
+ it, as its text was.
- **The page's own lines decide it.** The pieces are written only where the lines the page laid
- the paragraph out in hold the pieces' letters in their faces, families and colours, at their
- sizes — or, where the page fits the text to a size of its own, at sizes in the same
- proportion. A session that reads no markdown lays the marks out, and the text is written as
- it stands, as before.
+ the paragraph out in hold the pieces' letters in their faces, families, colours and tracking,
+ at their sizes to a hundredth of a point — or, where the page fits the text to a size of its
+ own, at sizes in the same proportion. A session that reads no markdown lays the marks out, and
+ the text is written as it stands, as before; so is text the parser changes nothing of, an
+ underscore inside a word.
- **The faces are the page's.** Where the session reads markdown, its parser sets every piece in
a face of its own and leaves the paragraph's aside: a bold paragraph's `Senior_Engineer`
stands regular on the page, and is written so.
- **Every path that writes a paragraph does it:** the body, a table cell, text over the flow,
- the line an overlay's two sides share, a badge's initials and a page zone's line. A paragraph
- composed in a table cell is matched to its lines by its text as the page reads it, where no
- line carries it as authored. A badge's initials are counted as the page sets them: `**JR**` is
- now a badge's two bold letters; initials in two faces, `*J*R`, are written in the flow, as
- initials in two runs' faces are.
- - **A heading the page sets larger than its line is named.** The page sets a markdown heading in
- a line as tall as the paragraph's own and draws its letters past it; written in that exact
- line, Word cuts their tops on screen.
+ the line an overlay's two sides share, a badge's initials and a page zone's line.
+ - A paragraph composed in a table cell is matched to its lines by its text as the page reads
+ it, where no line carries it as authored, and only to lines that set its pieces so.
+ - A badge's initials are counted as the page sets them: `**JR**` is now a badge's two bold
+ letters. Initials in two faces, `*J*R`, or a heading, are written in the flow, as initials
+ in two runs' faces are.
+ - **A heading written taller than its line is named.** The page sets a markdown heading in a
+ line as tall as the paragraph's own and draws its letters past it; written in that exact line,
+ Word cuts their tops on screen. An auto-sized paragraph's heading, written at a multiple of its
+ style's size, may fit the line the page fits the text to, and is named only where it does not.
- **The font table ships the faces the pieces of a paragraph outside table cells and page zones
are set in** — the paragraphs it reads. It is written before any paragraph, so it reads them
off the text: a session that reads no markdown ships a face it does not use.
- **Still written as authored, and named:**
- a paragraph whose lines are not read — with no layout, composed in a table cell whose text
- no line of its table carries, or a page zone's the layout shows none of — where the note
+ no line of its table carries so, or a page zone's the layout shows none of — where the note
says whether the page reads its marks is not measured;
- one the page sets in other letters than its text, as Arabic, which the page shapes before it
reads the marks;
- - one of marks alone, a lone `*` or a rule of `***`, which the page reads as an empty list item
- or a rule and sets as nothing, and which went unnamed;
+ - text the parser reads into nothing, which the page sets as nothing and which went unnamed:
+ a lone `*`, an empty list item; `***`, a rule; a line set four spaces in, a block of code;
- a list's items, as before.
+ - **With no layout**, a paragraph is read line by line for the note too: a list marker opening
+ a line, `* a_b`, is no longer counted as a mark the page drops.
Across the DOCX fidelity corpus one document's bytes change: `TimelineMinimal`'s open-source
project line is written in regular Lato, where Word drew it bold, with `(Open source)` in italic
@@ -208,13 +214,13 @@ follow semantic versioning; release dates are ISO 8601.
- a marker typed before a list item;
- runs, which the page never reads.
- Where the lines are not read — with no layout, or composed in a table cell, whose paragraphs are
- matched to their lines by text and whose lists not at all — the note says whether the page reads
+ Where the lines are not read — with no layout, or composed in a table cell whose paragraph no
+ line its table laid out carries, and a list there at all — the note says whether the page reads
the marks is not measured, where the page's own parser drops a mark from the text.
Across the DOCX fidelity corpus the report named one paragraph, `TimelineMinimal`'s open-source
project line, whose `*(Open source)*` the page sets in italic and Word showed with its
- asterisks — now written as the page sets it. None of this changed what is written: the 62 documents of the corpus export to the
- same bytes. In `DocxNodeFieldLedgerTest` a paragraph's `text` and a list's `nestedItems` move
+ asterisks — now written as the page sets it. None of this changed what is written: the 62
+ documents of the corpus export to the same bytes. In `DocxNodeFieldLedgerTest` a paragraph's `text` and a list's `nestedItems` move
from `WRITTEN` to `REPORTED`, and a list's `items` name it too.
- **A DOCX export's report names where a page zone's parts stand, and what its paragraphs
lose.** A page zone is written as one Word line. Word sets its parts one after another from
diff --git a/docs/architecture/backend-capability-matrix.md b/docs/architecture/backend-capability-matrix.md
index 138e77f09..b3813f9a9 100644
--- a/docs/architecture/backend-capability-matrix.md
+++ b/docs/architecture/backend-capability-matrix.md
@@ -63,7 +63,7 @@ Payload records live in `core` under
| Capability (payload) | PDF (fixed) | PPTX (fixed) | DOCX (semantic) |
|---|---|---|---|
-| Paragraph — pre-wrapped lines, runs, alignment (`ParagraphFragmentPayload`) | ✅ `PdfParagraphFragmentRenderHandler` | ✅ `PptxParagraphFragmentRenderHandler` (one absolute, wrap-disabled frame per measured line) | ⚠️ semantic paragraphs (`DocxSemanticBackend`) — each run keeps its own style, falling back to the paragraph's when it has none; a centred or right-aligned left-to-right line of its own, of text alone and untracked, that Word sets a point or more wider or narrower at its half-point size has its letters spaced by the difference (`w:spacing`) and its room reckoned from the page's width; a `linkTarget` becomes a `w:hyperlink`, with a relationship for an address or `w:anchor` for one of the document's own anchors, and a run's own link wins over the paragraph's; a paragraph seated off its baseline (`TextVerticalAlign`) has its runs raised or lowered in the line (`w:position`) by the PDF backend's own correction (`ParagraphSeating`), one shift for the paragraph where the page seats each line by its own; Word and LibreOffice stand an exact line's baseline four fifths of the way down it whatever the face, where the page sets it the face's ascent down, so a paragraph whose face puts the two half a point or more apart — Spectral's, not Lato's — has its text moved to the page's baseline in the same position, matched at its middle line (not yet a list item's or a table text cell's; a picture among it moves with it in Word and stays on its own baseline in LibreOffice); lines a container stacks over one another tighter than their face each end halfway between their letters and the next line's (Word draws an exact line's text on screen only inside the line; its PDF export does not cut it), and the last layer of a shape container on one page, where its line runs past the foot, ends at the foot or below its letters; letters two lines share are split halfway so the page does not move, and a stack that holds a picture keeps its lines' own heights; a `bulletOffset` of spaces becomes the paragraph's indent (`w:ind` left, hanging or first line, by `indentStrategy`) in the flow and in cells, not yet over the flow, in an overlay's left-and-right pair, as a badge's initials or in a header or footer; one with letters in it is not written, its wrapped lines still set after the spaces that cover it; an auto-sized paragraph's text is written at its style's size, not the one the page fits it to; a paragraph a session reads as markdown — the default, unless `markdown(false)` — is written as the page sets it, read through the page's own parser, one run a piece in the face, family, colour and size the page's laid-out lines hold (an auto-sized one's at its style's size), its marks dropped, wherever the lines hold the pieces' letters so; where they are not read, or hold other letters or none (marks alone, which the page sets as nothing), as authored, its marks as letters; a `bookmark(...)` is Word's `HeadingN`, which Word's outline lists by the text of its Word paragraph — an overlay's pair's whole line, one level for both sides — at no level past the ninth. Outside a header or footer, the paragraph's report note (`ParagraphNode`) names each of these where it moves or renames something: the prefix's letters, and the room a path that writes no prefix leaves out where it moves a line; the size written and the size the page fits the text to, to Word's half point; the marks of a paragraph the page read as markdown and the file holds as letters, where its laid-out lines hold fewer of them than its text, not measured where its lines are not read; a markdown heading the page sets larger than the paragraph's line, which Word cuts on screen; an outline title that is not the text Word lists, a level past the ninth that shares it with another, and the right side's entry where the left holds the line's level |
+| Paragraph — pre-wrapped lines, runs, alignment (`ParagraphFragmentPayload`) | ✅ `PdfParagraphFragmentRenderHandler` | ✅ `PptxParagraphFragmentRenderHandler` (one absolute, wrap-disabled frame per measured line) | ⚠️ semantic paragraphs (`DocxSemanticBackend`) — each run keeps its own style, falling back to the paragraph's when it has none; a centred or right-aligned left-to-right line of its own, of text alone and untracked, that Word sets a point or more wider or narrower at its half-point size has its letters spaced by the difference (`w:spacing`) and its room reckoned from the page's width; a `linkTarget` becomes a `w:hyperlink`, with a relationship for an address or `w:anchor` for one of the document's own anchors, and a run's own link wins over the paragraph's; a paragraph seated off its baseline (`TextVerticalAlign`) has its runs raised or lowered in the line (`w:position`) by the PDF backend's own correction (`ParagraphSeating`), one shift for the paragraph where the page seats each line by its own; Word and LibreOffice stand an exact line's baseline four fifths of the way down it whatever the face, where the page sets it the face's ascent down, so a paragraph whose face puts the two half a point or more apart — Spectral's, not Lato's — has its text moved to the page's baseline in the same position, matched at its middle line (not yet a list item's or a table text cell's; a picture among it moves with it in Word and stays on its own baseline in LibreOffice); lines a container stacks over one another tighter than their face each end halfway between their letters and the next line's (Word draws an exact line's text on screen only inside the line; its PDF export does not cut it), and the last layer of a shape container on one page, where its line runs past the foot, ends at the foot or below its letters; letters two lines share are split halfway so the page does not move, and a stack that holds a picture keeps its lines' own heights; a `bulletOffset` of spaces becomes the paragraph's indent (`w:ind` left, hanging or first line, by `indentStrategy`) in the flow and in cells, not yet over the flow, in an overlay's left-and-right pair, as a badge's initials or in a header or footer; one with letters in it is not written, its wrapped lines still set after the spaces that cover it; an auto-sized paragraph's text is written at its style's size, not the one the page fits it to; a paragraph a session reads as markdown — the default, unless `markdown(false)` — is written as the page sets it, read through the page's own parser, one run a piece in the face, family, colour, tracking and size the page's laid-out lines hold (an auto-sized one's at its style's size), its marks dropped, wherever the lines hold the pieces' letters so; where they are not read, or hold other letters or none (text the parser reads into nothing, which the page sets as nothing), as authored, its marks as letters; a `bookmark(...)` is Word's `HeadingN`, which Word's outline lists by the text of its Word paragraph — an overlay's pair's whole line, one level for both sides — at no level past the ninth. Outside a header or footer, the paragraph's report note (`ParagraphNode`) names each of these where it moves or renames something: the prefix's letters, and the room a path that writes no prefix leaves out where it moves a line; the size written and the size the page fits the text to, to Word's half point; the marks of a paragraph the page read as markdown and the file holds as letters, where its laid-out lines hold fewer of them than its text, not measured where its lines are not read; a markdown heading written taller than the line the page sets it in, which Word cuts on screen; an outline title that is not the text Word lists, a level past the ninth that shares it with another, and the right side's entry where the left holds the line's level |
| List hanging indent — a marker column and a content column (`ListBuilder.hangingIndent(true)`, `markerGap(...)`) | ✅ marker and content emitted as separate `ParagraphFragmentPayload` fragments at the resolved `markerX` / `contentX` | ✅ the same fragments — the fixed-layout pipeline resolves the geometry before either backend sees it | ⚠️ the top level only. `DocxSemanticBackend` exports a list as a real Word list — `numbering.xml`, `w:numPr` per item, the level carrying the marker — or, with rich items or a drawn marker, as paragraphs; content and nesting are unaffected. With the flag, the top level's marker column is the layout's — the marker's width and `markerGap`, the text and its wrapped lines where the page sets them — where the gap covers what Word may set the marker wider: a picture at its written size, its edges included, or text in the page's face (embedded, or a standard one Word sets in the same widths) grown to its half-point size, half a point clear. A Word list's level then indents and hangs by that column; a list of paragraphs writes the marker, a tab to a stop there, and hangs the item there. Word places content at absolute indents and has no relative-advance primitive, so without the layout's measure the gap could not be honoured; a Word list without the flag that the layout placed and that does not nest takes the page's column too, the spaces the page sets its wrapped lines after, its marker followed by a space (`w:suff`) and an item that wraps measured at Word's half-point size; a list that nests items, a list built as a tree of items (laid out flattened), and a marker the gap does not clear keep the stated column (180 twips, plus 120 per nesting level) — except, in a list of paragraphs, a nested rich item with no marker, which stands where the layout set its text, its measure weighed at Word's half-point sizes, where the layout's items are matched to the list's; a list that nests only such items sets its top level at the page's column too. The report counts, on the list, the items that stand at a stated column, a space past their marker or two spaces a level in, and names a centred or right-aligned list written flush left, a lineSpacing not written where the layout's items are not the list's own and one wraps (in a list composed in a table cell, its wrapping not measured), a continuationIndent not written where an item of a markerless list or a tree of items without the flag wraps or its wrapping is not measured, the rows the page draws as a marker alone for blank items of a flagged list, which are not written, and items the page reads as markdown, whose marks are written as letters, not yet as the page sets them |
| Inline code/badge chips (`InlineBackground` on text spans) | ✅ `PdfParagraphFragmentRenderHandler` | ✅ `PptxParagraphFragmentRenderHandler` | ⚠️ `DocxSemanticBackend` — the fill becomes the run's own `w:shd`, in a paragraph and in a list item alike, so a badge still reads as a badge. What Word has no way to say is the shape: shading covers the glyph box, so the corner radius and the padding above and below the letters are not in the file, and the export records them. The padding beside the letters is written as the room it takes (`spaceAfterTheLastLetter`): character spacing after the chip's last letter, shaded with it, and after the letter before the chip, unshaded. A chip opening its line or following a picture has no letter before it, so its left padding is not in the file; no space is written after right-to-left letters or after a symbol or emoji. The export records, chip by chip, how each side was written. LibreOffice sets no spacing after a line's last letter, so it does not apply the right padding of a chip that ends a line. A `w:shd` fill is opaque, so a translucent chip is flattened first against what Word paints underneath it — the paragraph's shading, the cell's, or else the colour the page paints under the paragraph, a page background included — so the chip agrees with the file it is in and shows the colour the PDF shows. It stops being translucent, and that is recorded with the rest |
| Inline images (`ParagraphImageSpan`) | ✅ `PdfParagraphFragmentRenderHandler` | ✅ `PptxParagraphFragmentRenderHandler` | ✅ `DocxSemanticBackend.writeInlinePicture` (a picture in its own run where it sits among the words, at its size, inside the run's or the paragraph's link; raised or lowered by `w:position` to where the page's alignment and `baselineOffset` put it, from the layout's measure of the paragraph's first line — in a list, the list's text on a line as tall as the item's own tallest picture; LibreOffice ignores `w:position` on a picture and stands it on the baseline, so a picture the export draws itself (icon, emoji, shape) that the page raises carries the rise as transparent rows and needs no `w:position`, while one the page lowers stands in LibreOffice higher than on the page by as much as the page lowers it — up to the text's descent for a centred icon as tall as its line; the editor clips a picture to an exact line height, so a paragraph holding a picture that leaves its text — past the ascent or the descent, in Word's placement or on the baseline — has its lines written at least the height the picture reaches, grown by the editor rather than clipped, every line of the paragraph since Word has one line height for it, and each as tall as the editor's font makes it — for 14pt text about 2.5pt taller than the page's in LibreOffice; a picture inside its text in both editors keeps the exact height; a paragraph of one line of text in a Word paragraph of its own, with room above for its pictures' reach, keeps an exact line at the page's height of it, the pictures set in it where the page puts them in Word and what their ink reaches past it taken from the gaps around it, and in LibreOffice a lowered picture there stands higher and loses what passes the line's top; its description is the text it stands for or empty) |
diff --git a/docs/recipes/docx-export.md b/docs/recipes/docx-export.md
index 4bf7447d4..11382c7f6 100644
--- a/docs/recipes/docx-export.md
+++ b/docs/recipes/docx-export.md
@@ -67,7 +67,7 @@ creation date is real metadata.
| Document node | DOCX output |
|---|---|
-| Paragraphs | Word paragraphs with alignment, font, size, colour (a translucent one with its transparency, as Word's text fill — see "Translucent colours"), bold/italic/underline; inline runs preserved; a `\n` in the text is a line break (`w:br`), which Word would otherwise read as a space; a table cell's `text("a\nb")` stays one line, as the page sets it; a `bulletOffset` of spaces is an indent — the wrapped lines (`FROM_SECOND_LINE`), the first (`FIRST_LINE`) or all of them start that far in, measured in the paragraph's style as the page measures it; a prefix with letters in it is not written, but the wrapped lines still start after the spaces the page covers it with; no prefix is applied to a paragraph written over the flow, as one of an overlay's left-and-right pair, as a badge's initials, or in a header or footer; the report names a prefix's letters, and the room a prefix sets lines in by where a path that writes none leaves it out and it moves a line — on the paragraph, or in a header or footer on the zone's note. A session reads markdown unless told not to (`markdown(false)`): it sets a paragraph of plain text's emphasis marks as the style of the text they mark, a heading line bold (the first three levels larger), and drops its marks; the export writes the text as the page sets it — read line by line through the page's own parser, one run a piece in the face, family, colour and size the page sets it in, `**Java**` a bold run reading `Java`, a linked paragraph's pieces in one link, Word's outline listing a heading by the text written — wherever it writes a paragraph, a page zone's and a badge's included (a badge's initials counted as the page sets them, and written in the flow where they stand in two faces, as initials in two runs' faces are). The page's laid-out lines decide it: the pieces are written where those lines hold their letters in their faces, families and colours, at their sizes or, for an auto-sized paragraph, at sizes in the same proportion, and a session that reads no markdown has its marks written as they stand. The page sets a markdown heading in a line as tall as the paragraph's own and draws its letters past it; written in that exact line, Word cuts their tops on screen, and the report names it. The faces are the page's: where the session reads markdown its parser sets every piece in a face of its own, the paragraph's left aside, so a bold paragraph's `Senior_Engineer` is written regular, as the page sets it. A paragraph whose lines are not read — with no layout, composed in a table cell whose text, as authored or as the page reads it, no line of its table carries, or a page zone's the layout shows none of — or that the page sets in other letters than its text, as Arabic, which the page shapes before it reads the marks, or that is marks alone, a lone `*` or a rule of `***`, which the page sets as nothing, is written as authored, marks and all, and the report names it, saying where the lines are not read that whether the page reads its marks is not measured, where the page's parser drops one. An auto-sized paragraph (`autoSize`) is written at its style's size, not the one the page fits its text to, and the report names both sizes where Word, to its half point, holds them apart, on the paragraph or, in a header or footer, on the zone's note. A body paragraph the layout moves to a new page keeps the space the layout leaves above its text there — its own top edge, the edges of the containers opening with it, and the gap before it where the gap did not fit at the foot of the page above — as an empty line that tall before it, kept with it, since Word drops a paragraph's space above at the top of a page and keeps a line's height; the rest of the space owed stays above that line, so Word breaks the page where it did, and the gap between the paragraph's lines comes off the line as it would off its space above. A paragraph in a table cell, an overlay or a list is left to its container |
+| Paragraphs | Word paragraphs with alignment, font, size, colour (a translucent one with its transparency, as Word's text fill — see "Translucent colours"), bold/italic/underline; inline runs preserved; a `\n` in the text is a line break (`w:br`), which Word would otherwise read as a space; a table cell's `text("a\nb")` stays one line, as the page sets it; a `bulletOffset` of spaces is an indent — the wrapped lines (`FROM_SECOND_LINE`), the first (`FIRST_LINE`) or all of them start that far in, measured in the paragraph's style as the page measures it; a prefix with letters in it is not written, but the wrapped lines still start after the spaces the page covers it with; no prefix is applied to a paragraph written over the flow, as one of an overlay's left-and-right pair, as a badge's initials, or in a header or footer; the report names a prefix's letters, and the room a prefix sets lines in by where a path that writes none leaves it out and it moves a line — on the paragraph, or in a header or footer on the zone's note. A session reads markdown unless told not to (`markdown(false)`): it sets a paragraph of plain text's emphasis marks as the style of the text they mark, a heading line bold (the first three levels larger), and drops its marks; the export writes the text as the page sets it — read line by line through the page's own parser, one run a piece in the face, family, colour, tracking and size the page sets it in, `**Java**` a bold run reading `Java`, a linked paragraph's pieces in one link, Word's outline listing a heading by the text written — wherever it writes a paragraph, a page zone's and a badge's included (a badge's initials counted as the page sets them, and written in the flow where they stand in two faces or as a heading, as initials in two runs' faces are). The page's laid-out lines decide it: the pieces are written where those lines hold their letters in their faces, families, colours and tracking, at their sizes or, for an auto-sized paragraph, at sizes in the same proportion; a session that reads no markdown has its marks written as they stand, and text the parser changes nothing of, an underscore inside a word, is written as it stands too. A paragraph composed in a table cell takes only lines that set its pieces so. The page sets a markdown heading in a line as tall as the paragraph's own and draws its letters past it; written in that exact line, Word cuts their tops on screen, and the report names a heading written taller than its line. The faces are the page's: where the session reads markdown its parser sets every piece in a face of its own, the paragraph's left aside, so a bold paragraph's `Senior_Engineer` is written regular, as the page sets it. A paragraph whose lines are not read — with no layout, composed in a table cell whose text, as authored or as the page reads it, no line of its table carries, or a page zone's the layout shows none of — or that the page sets in other letters than its text, as Arabic, which the page shapes before it reads the marks, or that the parser reads into nothing — a lone `*`, an empty list item; `***`, a rule; a line set four spaces in, a block of code — which the page sets as nothing, is written as authored, marks and all, and the report names it, saying where the lines are not read that whether the page reads its marks is not measured, where the page's parser drops one. An auto-sized paragraph (`autoSize`) is written at its style's size, not the one the page fits its text to, and the report names both sizes where Word, to its half point, holds them apart, on the paragraph or, in a header or footer, on the zone's note. A body paragraph the layout moves to a new page keeps the space the layout leaves above its text there — its own top edge, the edges of the containers opening with it, and the gap before it where the gap did not fit at the foot of the page above — as an empty line that tall before it, kept with it, since Word drops a paragraph's space above at the top of a page and keeps a line's height; the rest of the space owed stays above that line, so Word breaks the page where it did, and the gap between the paragraph's lines comes off the line as it would off its space above. A paragraph in a table cell, an overlay or a list is left to its container |
| Lists | Real Word lists: a `numbering.xml` definition per list, `w:numPr` on each item, and the authored marker as the level's text. Nesting is a list level, so Enter continues the list and Tab demotes an item. See "What a list becomes" below for the kinds that stay plain paragraphs |
| Tables | Word tables, one cell per cell. Each cell states its own padding, on all four sides, so a row is as tall as the page draws it: as `w:tcMar`, and above and below partly in its paragraphs. Word and LibreOffice give every cell of a row the largest top and bottom margin of any cell in it, so a row's cells are written with its smallest, and the rest of a cell's padding above and below is space above its first paragraph and below its last (measured: a row whose day cells were padded 5.5pt above and 10.25pt below beside a label padded 0.75pt stood 60.3pt tall in both editors, where its tallest cell came to 46). A cell opening with a table has no paragraph above it to hold its padding, and a cell in a vertical merge has its bottom edge in another row: these keep their margins, and the row's comes down no lower than the largest of them. Its padding above and below gives up the room Word makes for the table's horizontal rules — half of a rule between two rows to each, the lower row's rule where the two differ, the rule above the table and the one below it whole to their row — which the page does not (measured: a 0.75pt rule made each row 0.75pt taller). A row held at the page's height is written less its margins and those rules too — a rule and a half in the first row and the last, two in a table of one row — since both editors read a row's written height as its cells' content (measured: held less one rule, a table ruled at 0.75pt stood 0.46pt taller in its first row and 0.36pt in its last). A cell that holds nothing but an empty line — a row that is only a rule, its thickness the empty cell's font — has that line cut to the room its row leaves it, the page's row less the cell's own margins and border: the page draws the rule's borders across the line, and Word and LibreOffice keep them outside it and grow the row (measured: `CobaltRota`'s two rules under a 0.9pt border stood 0.9pt taller each). A line with letters, a picture or a paragraph border of its own keeps its height. A table that states no rule is written with the engine's default 1pt black rule, as the page draws it, not left on Word's thinner grid. Its `textAnchor` becomes `w:vAlign` and the paragraph's `w:jc`, with the engine's default — the vertical middle, on the left — where Word's is the top, so a line beside a taller neighbour sits where the page puts it and an amount column stays right-aligned. A cell with no style of its own is set in the engine's default cell face rather than the document's Normal. A cell's lines are one paragraph with line breaks; when its style's `lineSpacing(...)` is above zero and it has more than one line, they are a paragraph each, with the spacing after every line but the last, since Word has no space between the lines of one paragraph but a taller line. A column sized to its content gets a point more than the page gives it, so the editor's font substitute cannot wrap its widest cell. The width is written when the document states one or every column is fixed; otherwise Word sizes the table — see "What falls back". A table breaks across pages where the layout breaks it: every row the layout placed is kept whole (`w:cantSplit`), `repeatHeader(n)` rows repeat on each page (`w:tblHeader`) and stay with the row under them. Two tables in a row — rows included, since a row is carried as a table — are kept apart by a paragraph a tenth of a point tall, holding the rest of the gap between them: an editor joins two tables with nothing between them into one. A table or a row the layout moves to a new page keeps its own top edge there, as the page does — written as a line that tall, kept with it, since Word drops a paragraph's space above at the top of a page — while the gap between it and the block before stays at the foot of the page above, where it fits there; a gap the layout carries onto the new page, because it did not fit at the foot of the page above, is not yet held above a table (body paragraphs and spacers: see their rows). A table's margin is its indent and the space round it, and its padding on the sides holds its rows in as the page draws them: its left side is in the indent too, and both are out of the room its columns are given |
| Composed cells (`DocumentTableCell.node(...)`) | Written by the same writers that write that node anywhere else, so a cell built from an image, a list or a table carries it. A nested table is a real `w:tbl` followed by the paragraph Word requires a cell to end with — a hairline, which the paragraph written next in the cell takes over, so no empty line opens under the table, and whose mark is hidden where it is left at the cell's end holding nothing and no space, since LibreOffice lays it out — and takes the width of the column it sits in, less its own margins and padding — the column's, not the one the page gives it, because the layout reports a composed cell's content under the owner's path |
diff --git a/render-docx/README.md b/render-docx/README.md
index 3cb590b29..34f9d75f4 100644
--- a/render-docx/README.md
+++ b/render-docx/README.md
@@ -135,10 +135,10 @@ What is not written — each one is named in the export report
- the marks of a paragraph the session reads as markdown (the default; `markdown(false)` turns
it off), where they are written as letters: the paragraph is written as the page sets it —
its marks dropped, each piece in the page's face and size — wherever the page's lines show
- how, and as authored where they do not (no layout, a cell whose text no line carries), hold
- other letters (Arabic, which the page shapes first) or hold none (marks alone, which the page
- sets as nothing); and a markdown heading the page
- sets larger than its line, which Word cuts on screen;
+ how, and as authored where they do not (no layout, a cell whose text no line carries, a
+ page zone's part the layout shows none of), hold other letters (Arabic, which the page shapes
+ first) or hold none (text the parser reads into nothing, which the page sets as nothing); and
+ a markdown heading written taller than its line, which Word cuts on screen;
- the letters of a `bulletOffset` prefix; and, where it moves a line, the room a prefix sets
lines in by in a paragraph written over the flow, as a side of an overlay's left-and-right
pair or as a badge's initials;
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
index ea7b3e124..f86b6066d 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxLayoutMetrics.java
@@ -1106,9 +1106,9 @@ private List textFragmentsOf(DocumentNode node) {
* Gothic A1 line the page sets at 9.1 — and every row of such a table ran taller than the
* page's. Cells are laid out in order, and so is what each holds, so the paragraphs are
* walked in that order and each takes the first fragment not yet taken whose text is
- * its own as authored; those left then take one whose text is theirs as the page reads their
- * markdown, marks dropped. A paragraph whose text no fragment carries is left without one, as
- * before.
+ * its own as authored; those left then take the first whose lines set their markdown as the
+ * page reads it, marks dropped, in its faces and sizes. A paragraph whose text no fragment
+ * carries so is left without one, as before.
*/
private Map matchComposedText() {
Map matched = new IdentityHashMap<>();
@@ -1142,14 +1142,18 @@ private Map matchComposedText() {
}
}
// Then by its text as the page reads its markdown, where its session does: marks
- // dropped. Only once every paragraph has taken its own text, so that one read as
- // markdown takes no fragment a paragraph of that text as authored stands in.
+ // dropped, and only a fragment whose lines set the pieces as they are read. A paragraph
+ // of the same text as authored may have taken this one's own fragment: the one left
+ // is in another face or size, and taken, its lines would cut this one's letters.
for (ParagraphNode paragraph : unmatched) {
if (DocxMarkdown.mayRead(paragraph) && paragraph.textStyle() != null) {
- java.util.ArrayDeque waiting = waitingFor(byText,
- DocxMarkdown.text(DocxMarkdown.read(paragraph.text(), paragraph.textStyle())));
- if (waiting != null) {
- matched.put(paragraph, waiting.poll());
+ List pieces = DocxMarkdown.read(paragraph.text(), paragraph.textStyle());
+ java.util.ArrayDeque waiting = waitingFor(byText, DocxMarkdown.text(pieces));
+ PlacedFragment fragment = waiting == null ? null : waiting.stream()
+ .filter(candidate -> setsThePieces(paragraph, pieces, candidate)).findFirst().orElse(null);
+ if (fragment != null) {
+ waiting.remove(fragment);
+ matched.put(paragraph, fragment);
}
}
}
@@ -1157,6 +1161,16 @@ private Map matchComposedText() {
return matched;
}
+ /**
+ * Whether a fragment's lines set a paragraph's markdown pieces as they are read
+ * ({@link DocxMarkdown#laidOutIn}). A fragment whose lines lead with a prefix's letters carries
+ * other text than the pieces, and is never offered.
+ */
+ private static boolean setsThePieces(ParagraphNode paragraph, List pieces, PlacedFragment fragment) {
+ return DocxMarkdown.laidOutIn(pieces, ((ParagraphFragmentPayload) fragment.payload()).lines(), "",
+ paragraph.autoSize() != null);
+ }
+
/** The fragments of a text still waiting for a paragraph, or {@code null} where none is. */
private static java.util.ArrayDeque waitingFor(Map> byText,
String text) {
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java
index 06671c97f..5eefe2d8c 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdown.java
@@ -29,7 +29,7 @@
* it reads in a face of its own — bold, italic, both or neither, whatever the paragraph's style
* — and a heading's at a multiple of the size. The pieces are read here the same way
* ({@link #read}), and are the text the page sets only where its laid-out lines hold the same
- * letters in the same faces, families, colours and sizes ({@link #laidOutIn}): a session that
+ * letters in the same faces, families, colours, tracking and sizes ({@link #laidOutIn}): a session that
* reads no markdown lays the marks out, and the pieces hold none.
*/
final class DocxMarkdown {
@@ -160,15 +160,17 @@ static String text(List pieces) {
/**
* Whether the page laid the pieces out in its lines: the lines' letters, less a prefix's
- * leading them, are the pieces' letters, each in the same face, family and colour, and at the
- * same size — or, where the page fits the text to a size of its own, at sizes in the same
- * proportion. Pieces of no letter — marks alone, which the page sets as nothing — are not taken
- * for the page's: written, the paragraph would be blank, and a blank paragraph is what the
- * export writes elsewhere for no line at all. White space is not compared: the page drops it
- * where it breaks a line.
+ * leading them, are the pieces' letters, each in the same face, family, colour and tracking,
+ * and at the same size, to {@link #SIZE_CLEARANCE} — or, where the page fits the text to a
+ * size of its own, at sizes in the same proportion, its tracking, resolved at that size, not
+ * compared. Pieces of no letter
+ * — text the parser reads into nothing, which the page sets as nothing — are not taken for the
+ * page's: written, the paragraph would be blank, and a blank paragraph is what the export
+ * writes elsewhere for no line at all. White space is not compared: the page drops it where it
+ * breaks a line.
*
* @param pieces the pieces read off the text
- * @param lines the lines the page laid the text out in, at least one
+ * @param lines the lines the page laid the text out in; none never holds the pieces
* @param prefix the prefix the page sets before the first line, empty where it sets none
* @param fitted whether the page fits the text to a size of its own, an auto-sized paragraph's
*/
@@ -179,7 +181,8 @@ static boolean laidOutIn(List pieces, List lines, String p
List written = new ArrayList<>();
for (Piece piece : pieces) {
DocumentTextStyle style = piece.style();
- lettersOf(piece.text(), pageFace(style.decoration()), style.size(), style.fontName(),
+ double tracking = style.letterSpacing() == null ? 0 : style.letterSpacing().resolve(style.size());
+ lettersOf(piece.text(), pageFace(style.decoration()), style.size(), tracking, style.fontName(),
style.color() == null ? null : style.color().color(), written);
}
List laid = new ArrayList<>();
@@ -189,11 +192,12 @@ static boolean laidOutIn(List pieces, List lines, String p
return false;
}
TextStyle style = text.textStyle();
- lettersOf(text.text(), style.decoration(), style.size(), style.fontName(), style.color(), laid);
+ lettersOf(text.text(), style.decoration(), style.size(), style.letterSpacing(), style.fontName(),
+ style.color(), laid);
}
}
List leading = new ArrayList<>();
- lettersOf(prefix, null, 0, null, null, leading);
+ lettersOf(prefix, null, 0, 0, null, null, leading);
if (written.isEmpty() || laid.size() != leading.size() + written.size()) {
return false;
}
@@ -207,26 +211,27 @@ static boolean laidOutIn(List pieces, List lines, String p
Letter page = laid.get(leading.size() + index);
Letter file = written.get(index);
if (page.codePoint() != file.codePoint() || page.face() != file.face()
- || !Objects.equals(page.family(), file.family()) || page.rgb() != file.rgb()
- || Math.abs(page.size() - file.size() * ratio) > SIZE_CLEARANCE) {
+ || !Objects.equals(page.family(), file.family()) || page.argb() != file.argb()
+ || Math.abs(page.size() - file.size() * ratio) > SIZE_CLEARANCE
+ || !fitted && Math.abs(page.tracking() - file.tracking()) > SIZE_CLEARANCE) {
return false;
}
}
return true;
}
- /** One letter as the page or the file sets it, its colour as packed ARGB. */
- private record Letter(int codePoint, TextDecoration face, double size, FontName family, int rgb) {
+ /** One letter as the page or the file sets it: its tracking in points, its colour as packed ARGB. */
+ private record Letter(int codePoint, TextDecoration face, double size, double tracking, FontName family, int argb) {
}
- private static void lettersOf(String text, TextDecoration face, double size, FontName family, Color color,
- List into) {
+ private static void lettersOf(String text, TextDecoration face, double size, double tracking, FontName family,
+ Color color, List into) {
if (text == null) {
return;
}
- int rgb = color == null ? 0 : color.getRGB();
+ int argb = color == null ? 0 : color.getRGB();
text.codePoints()
.filter(codePoint -> !Character.isWhitespace(codePoint))
- .forEach(codePoint -> into.add(new Letter(codePoint, face, size, family, rgb)));
+ .forEach(codePoint -> into.add(new Letter(codePoint, face, size, tracking, family, argb)));
}
}
diff --git a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
index 0605dfde7..4a27d43df 100644
--- a/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
+++ b/render-docx/src/main/java/com/demcha/compose/document/backend/semantic/docx/DocxSemanticBackend.java
@@ -2192,7 +2192,7 @@ private List zoneParagraphLost(ParagraphNode node, List paragraphLost(ParagraphNode node, boolean roomLost, String
if (marks != null) {
lost.add(marks);
}
- String heading = markdownHeadingCut(node, "its");
+ String heading = markdownHeadingCut(node, layout.lines(node), "its");
if (heading != null) {
lost.add(heading);
}
@@ -7516,12 +7516,12 @@ private static String autoSizeLost(ParagraphNode node, ListThe pieces are the page's own: where its session reads markdown, the parser sets each
* one in its own face, the paragraph's left aside — a bold paragraph's {@code file_name}
@@ -7535,34 +7535,50 @@ private static List markdownPieces(ParagraphNode node,
return null;
}
List pieces = DocxMarkdown.read(node.text(), node.textStyle());
+ // Pieces that change nothing — every mark kept, every piece in the paragraph's style —
+ // leave the text to be written as it stands, its white space and all.
+ if (pieces.stream().allMatch(piece -> piece.style().equals(node.textStyle()))
+ && markdownMarksIn(DocxMarkdown.text(pieces)) == markdownMarksIn(node.text())) {
+ return null;
+ }
return DocxMarkdown.laidOutIn(pieces, lines, setsAPrefixBeforeTheFirstLine(node) ? node.bulletOffset() : "",
node.autoSize() != null) ? pieces : null;
}
/**
* What a paragraph written in the pieces the page reads its markdown into
- * ({@link #markdownPieces}) loses of them: a heading the page sets larger than the
- * paragraph's line, the line its own — it draws the heading's letters past it, where Word
- * cuts their tops on screen in the exact line the paragraph is written in.
+ * ({@link #markdownPieces}) loses of them: a heading written taller than the line the page
+ * sets it in, by more than {@link #HEADING_CLEARANCE}. The page sets a heading line as tall
+ * as the paragraph's own and draws the heading's letters past it; Word cuts their tops on
+ * screen in the exact line the paragraph is written in. An auto-sized paragraph's heading,
+ * written at a multiple of its style's size, may fit the line the page fits its text to.
*
+ * @param lines the lines the page laid the paragraph out in
* @param whose whose heading the phrase names
- * @return the phrase, or {@code null} where no piece is larger than the paragraph's text
+ * @return the phrase, or {@code null} where every piece fits its line
*/
- private String markdownHeadingCut(ParagraphNode node, String whose) {
+ private String markdownHeadingCut(ParagraphNode node, List lines,
+ String whose) {
List pieces = markdownWritten.get(node);
- if (pieces == null || node.textStyle() == null) {
+ if (pieces == null || node.textStyle() == null || lines.isEmpty()) {
return null;
}
+ double line = lines.stream().mapToDouble(com.demcha.compose.document.layout.payloads.ParagraphLine::lineHeight)
+ .max().orElse(0);
for (DocxMarkdown.Piece piece : pieces) {
- if (piece.style().size() > node.textStyle().size()) {
- return whose + " markdown heading is written at the size the page sets it, " + pointsOf(piece.style().size())
- + "pt, in a line only as tall as the paragraph's own: the page draws its letters past the line, "
- + "and Word cuts their tops on screen";
+ if (piece.style().size() > node.textStyle().size()
+ && styleLineHeight(piece.style()) > line + HEADING_CLEARANCE) {
+ return whose + " markdown heading is written at " + pointsOf(piece.style().size()) + "pt in a line "
+ + pointsOf(line) + "pt tall, as tall as the paragraph's own line on the page: the page draws its "
+ + "letters past the line, and Word cuts their tops on screen";
}
}
return null;
}
+ /** How much taller than its line a heading's own line may be before Word cuts its letters, in points. */
+ private static final double HEADING_CLEARANCE = 0.5;
+
/**
* What a paragraph read as markdown loses where it is written as authored, not in the pieces
* the page sets it in ({@link #markdownPieces}): the page sets the text its marks style, and
@@ -7578,13 +7594,13 @@ private String markdownLost(ParagraphNode node, List line.text().isBlank()) && !node.text().isBlank()
&& DocxMarkdown.text(DocxMarkdown.read(node.text(), node.textStyle() == null
? DocumentTextStyle.DEFAULT : node.textStyle())).isBlank()) {
- return whose + " markdown marks are written as letters, where the page reads them as marks alone and "
- + "sets nothing";
+ return whose + " text is written as authored, where the page reads it as markdown and sets nothing";
}
return marksDropped(whose, lines, markdownMarksIn(node.text()),
setsAPrefixBeforeTheFirstLine(node) ? markdownMarksIn(node.bulletOffset()) : 0, List.of(node.text()));
@@ -9751,9 +9767,12 @@ private ParagraphNode textBadgeParagraph(ShapeContainerNode badge) {
.anyMatch(run -> run instanceof InlineTextRun text && text.linkTarget() != null)) {
return null;
}
- // Read as the page reads its markdown, where it does: in one face, as a run's text is.
+ // Read as the page reads its markdown, where it does: in one face, as a run's text is, and
+ // at the paragraph's size — a heading the page draws past its line is written in the
+ // flow, its line the page's, and named there.
List pieces = badgePieces(paragraph);
- if (pieces != null && pieces.stream().map(DocxMarkdown.Piece::style).distinct().count() > 1) {
+ if (pieces != null && (pieces.stream().map(DocxMarkdown.Piece::style).distinct().count() > 1
+ || pieces.get(0).style().size() != paragraph.textStyle().size())) {
return null;
}
String text = pieces != null ? DocxMarkdown.text(pieces) : badgeTextOf(paragraph);
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
index 67eb7f2a4..c7d1c2cdd 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownReportTest.java
@@ -41,9 +41,10 @@ class DocxMarkdownReportTest {
void aParagraphThePageReadsAsMarkdownIsWrittenSoAndNotNamed() throws Exception {
assertThat(paragraphNotes(true, page -> page.addParagraph("Some **bold** and `code` text"))).isEmpty();
assertThat(paragraphNotes(true, page -> page.addParagraph("# Title *now*"))).as("a heading, past its line")
- .containsExactly("written as a paragraph; its markdown heading is written at the size the page sets it, "
- + "28pt, in a line only as tall as the paragraph's own: the page draws its letters past "
- + "the line, and Word cuts their tops on screen");
+ .singleElement().asString()
+ .startsWith("written as a paragraph; its markdown heading is written at 28pt in a line ")
+ .endsWith("pt tall, as tall as the paragraph's own line on the page: the page draws its letters past "
+ + "the line, and Word cuts their tops on screen");
assertThat(paragraphNotes(true, page -> page.addParagraph("`x`"))).as("a code span alone").isEmpty();
// A prefix with a mark of its own is laid out with it, and leads the letters the page sets.
assertThat(paragraphNotes(true, page -> page.addParagraph(p -> p.text("Some *emphasis* here")
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
index b59d94212..4942a3002 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxMarkdownTest.java
@@ -114,6 +114,10 @@ void thePiecesAreThePagesWhereItsLinesHoldThemSoAndNotOtherwise() {
assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some ", TextDecoration.DEFAULT, 10),
span("bold", TextDecoration.BOLD, 10), span(" text", TextDecoration.DEFAULT, 10, BODY.fontName(),
Color.RED))), "", false)).as("another colour").isFalse();
+ assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some ", TextDecoration.DEFAULT, 10),
+ span("bold", TextDecoration.BOLD, 10), new ParagraphTextSpan(" text",
+ new TextStyle(BODY.fontName(), 10, TextDecoration.DEFAULT, BODY.color().color(), 0.5), 25, 10,
+ null, null, false))), "", false)).as("another tracking").isFalse();
assertThat(DocxMarkdown.laidOutIn(pieces, List.of(line(span("Some ", TextDecoration.DEFAULT, 10),
span("bolt", TextDecoration.BOLD, 10), span(" text", TextDecoration.DEFAULT, 10))), "", false))
.as("another letter").isFalse();
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxNodeFieldLedgerTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxNodeFieldLedgerTest.java
index 72f4572c9..eec0df990 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxNodeFieldLedgerTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxNodeFieldLedgerTest.java
@@ -196,7 +196,7 @@ private record Entry(Fate fate, String note) {
"margin:REPORTED:the sides its alignment sets it from");
node(ParagraphNode.class, "name:INERT",
"text:REPORTED:where the page reads it as markdown, written as the page sets it where its lines "
- + "hold the pieces read so, a heading larger than its line named; its marks written as letters "
+ + "hold the pieces read so, a heading written taller than its line named; its marks written as letters "
+ "where the lines hold other letters or none, and not measured where they are not read and the "
+ "page's parser drops a mark; any other is written",
"inlineRuns:REPORTED:a chip's translucent fill, flattened against the colour under it, with what "
diff --git a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionMarkdownTest.java b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionMarkdownTest.java
index b3acdd0ef..4fc1fe341 100644
--- a/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionMarkdownTest.java
+++ b/render-docx/src/test/java/com/demcha/compose/document/backend/semantic/docx/DocxSessionMarkdownTest.java
@@ -44,16 +44,19 @@
* A paragraph the page reads as markdown is written as the page sets it: its marks dropped, the
* text they mark in the face — and a heading at the size — the page sets it in, and nothing named
* but a heading the page draws past its line. The page's own lines say whether it reads the
- * paragraph so; where they hold the marks, or other letters than the pieces, the paragraph is
- * written as authored and named.
+ * paragraph so: where they hold the marks, as a session that reads no markdown lays them out, it
+ * is written as it stands; where they hold other letters, it is written as authored and named.
*/
class DocxSessionMarkdownTest {
private static final String MARKS = "its markdown marks are written as letters, where the page sets the text "
+ "they mark and drops them";
- private static final String HEADING_CUT = "its markdown heading is written at the size the page sets it, 20pt, "
- + "in a line only as tall as the paragraph's own: the page draws its "
- + "letters past the line, and Word cuts their tops on screen";
+ private static final String HEADING_CUT = "pt tall, as tall as the paragraph's own line on the page: the page draws "
+ + "its letters past the line, and Word cuts their tops on screen";
+ private static final String NOTHING = "its text is written as authored, where the page reads it as markdown and "
+ + "sets nothing";
+ private static final String MARKS_UNMEASURED = "markdown marks are written as letters — whether the page reads them "
+ + "is not measured";
@Test
void aParagraphIsWrittenInThePiecesThePageSetsIt() throws Exception {
@@ -79,7 +82,9 @@ void aHeadingLineIsWrittenBoldAtTheSizeThePageSetsItAndEachLineBreaks() throws E
assertThat(paragraph.getCTP().xmlText()).as("a line break where the page starts a line").contains("");
assertThat(paragraph.getText()).contains("- dash").doesNotContain("#").doesNotContain("*");
// The page sets the heading in a line as tall as the paragraph's own, and draws it past.
- assertThat(export.notes("ParagraphNode")).containsExactly("written as a paragraph; " + HEADING_CUT);
+ assertThat(export.notes("ParagraphNode")).singleElement().asString()
+ .startsWith("written as a paragraph; its markdown heading is written at 20pt in a line ")
+ .endsWith(HEADING_CUT);
// The paragraph's mark closes its last line, sized as the piece that ends it.
XWPFParagraph heading = export(true, page -> page.addParagraph(p -> p.text("# Title *x*")
@@ -118,13 +123,13 @@ void aParagraphThePageBreaksAcrossPagesIsWrittenAsThePageSetsIt() throws Excepti
}
@Test
- void marksAloneThePageSetsAsNothingAreWrittenAsTheyStandAndNamed() throws Exception {
- // The page reads `***` as a rule and a lone `*` as an empty list item, and sets nothing.
- for (String marks : List.of("***", "*")) {
- Export export = export(true, page -> page.addParagraph("Above").addParagraph(marks).addParagraph("Below"));
- assertThat(export.document().getDocument().xmlText()).contains(">" + marks + "<");
- assertThat(export.notes("ParagraphNode")).containsExactly("written as a paragraph; its markdown marks are "
- + "written as letters, where the page reads them as marks alone and sets nothing");
+ void textThePageReadsIntoNothingIsWrittenAsItStandsAndNamed() throws Exception {
+ // The page reads `***` as a rule, a lone `*` as an empty list item and a line set four
+ // spaces in as a block of code it keeps no text of, and sets nothing.
+ for (String text : List.of("***", "*", " code *x*")) {
+ Export export = export(true, page -> page.addParagraph("Above").addParagraph(text).addParagraph("Below"));
+ assertThat(export.document().getDocument().xmlText()).contains(">" + text + "<");
+ assertThat(export.notes("ParagraphNode")).as(text).containsExactly("written as a paragraph; " + NOTHING);
}
Export off = export(false, page -> page.addParagraph("***"));
assertThat(off.document().getDocument().xmlText()).as("markdown off, the page sets the marks").contains(">***<");
@@ -138,6 +143,41 @@ void aSessionThatReadsNoMarkdownHasItsMarksWrittenAsTheyStand() throws Exception
assertThat(paragraph.getText()).isEqualTo("Some **bold** text");
assertThat(paragraph.getRuns()).singleElement().satisfies(run -> assertThat(run.isBold()).isFalse());
assertThat(export.notes("ParagraphNode")).isEmpty();
+ // Text the parser keeps whole, its white space and a tab among it, as it stands in either session.
+ for (boolean markdown : List.of(false, true)) {
+ assertThat(export(markdown, page -> page.addParagraph("a_b\tc")).paragraphWith("a_b").getText())
+ .as("markdown " + markdown).isEqualTo("a_b\tc");
+ }
+ }
+
+ @Test
+ void textTheParserReadsAsThePageDoesIsWrittenSoWhateverItHolds() throws Exception {
+ // Each of these the page reads through its parser; were the pieces read here otherwise
+ // than the page reads them, the paragraph would be written as authored and named.
+ DocumentTextStyle tracked = DocumentTextStyle.builder().size(10)
+ .letterSpacing(com.demcha.compose.document.style.DocumentLetterSpacing.ofFontSize(0.05)).build();
+ for (String text : List.of("1. a *b*", "> *q* r", "\\*x\\* *y*", "***nested** emph*", "`a_b` *c*",
+ "[*x*](https://example.org) y", "a & *b*", "x *y* z", "## Sub *x*\nbody *y*")) {
+ Export export = export(true, page -> page.addParagraph(p -> p.text(text).textStyle(tracked)));
+ assertThat(export.notes("ParagraphNode")).as(text)
+ .noneMatch(note -> note.contains("markdown marks are written as letters"));
+ }
+ }
+
+ @Test
+ void anAutoSizedParagraphIsWrittenInThePiecesAtItsStylesSize() throws Exception {
+ Export export = export(true, page -> page.addParagraph(p -> p.text("Fit **this** text")
+ .textStyle(DocumentTextStyle.builder().size(10).build()).autoSize(20, 6)));
+ XWPFParagraph paragraph = export.paragraphWith("this");
+ assertThat(paragraph.getRuns()).extracting(XWPFRun::text, XWPFRun::isBold).containsExactly(
+ org.assertj.core.groups.Tuple.tuple("Fit ", false), org.assertj.core.groups.Tuple.tuple("this", true),
+ org.assertj.core.groups.Tuple.tuple(" text", false));
+ assertThat(export.notes("ParagraphNode")).noneMatch(note -> note.contains("markdown"));
+ // A heading at a multiple of the style's size fits the taller line the page fits the text to.
+ Export heading = export(true, page -> page.addParagraph(p -> p.text("# Big_x")
+ .textStyle(DocumentTextStyle.builder().size(10).build()).autoSize(30, 6)));
+ assertThat(heading.paragraphWith("Big").getText()).isEqualTo("Big_x");
+ assertThat(heading.notes("ParagraphNode")).noneMatch(note -> note.contains("markdown"));
}
@Test
@@ -181,10 +221,10 @@ void aZoneParagraphIsWrittenAsThePageSetsIt() throws Exception {
// A heading in a zone stands taller than the zone's line, as in the body.
zoneFooter("# Confidential *now*", report);
assertThat(report.get().bySubject().get("page zone")).extracting(DocxExportReport.Note::detail)
- .containsExactly("a footer written as one line of Word's footer; a paragraph's markdown heading is "
- + "written at the size the page sets it, 16pt, in a line only as tall as the "
- + "paragraph's own: the page draws its letters past the line, and Word cuts their "
- + "tops on screen");
+ .singleElement().asString()
+ .startsWith("a footer written as one line of Word's footer; a paragraph's markdown heading is written "
+ + "at 16pt in a line ")
+ .endsWith(HEADING_CUT);
}
private static String zoneFooter(String text, AtomicReference report) throws Exception {
@@ -230,6 +270,26 @@ void aCellsParagraphKeepsItsOwnLinesBesideOneReadAsMarkdown() throws Exception {
.isEqualTo(org.openxmlformats.schemas.wordprocessingml.x2006.main.STLineSpacingRule.EXACT);
}
+ @Test
+ void aCellReadAsMarkdownTakesNoLineThatSetsItsPiecesOtherwise() throws Exception {
+ // A paragraph of the same text as authored after it takes its line first; the one left is
+ // in a smaller, regular face, and held to it, its bold letters would be cut. It takes none:
+ // Word's own line, its marks named as not measured.
+ Export export = export(true, page -> page.add(new com.demcha.compose.document.dsl.TableBuilder().name("Sums")
+ .columns(DocumentTableColumn.fixed(110), DocumentTableColumn.fixed(110))
+ .rowCells(DocumentTableCell.node(new ParagraphBuilder().name("Bold").text("**Total**").build()),
+ DocumentTableCell.node(new ParagraphBuilder().name("Small").text("Total")
+ .textStyle(DocumentTextStyle.DEFAULT.withSize(6)).build()))
+ .build()));
+ XWPFParagraph bold = export.document().getTables().get(0).getRow(0).getCell(0).getParagraphs().get(0);
+ org.openxmlformats.schemas.wordprocessingml.x2006.main.CTPPr properties = bold.getCTP().getPPr();
+ assertThat(properties != null && properties.isSetSpacing() && properties.getSpacing().isSetLineRule()
+ && properties.getSpacing().getLineRule()
+ == org.openxmlformats.schemas.wordprocessingml.x2006.main.STLineSpacingRule.EXACT)
+ .as("held to no line of another's").isFalse();
+ assertThat(export.notes("ParagraphNode")).containsExactly("written as a paragraph; its " + MARKS_UNMEASURED);
+ }
+
@Test
void aBadgesInitialsAreWrittenAsThePageSetsThem() throws Exception {
Export export = export(true, page -> page.add(badge("*JR*")));
@@ -245,6 +305,11 @@ void aBadgesInitialsAreWrittenAsThePageSetsThem() throws Exception {
assertThat(twoFaces.paragraphWith("JR").getRuns()).filteredOn(run -> !run.text().isEmpty())
.extracting(XWPFRun::text, XWPFRun::isItalic)
.containsExactly(org.assertj.core.groups.Tuple.tuple("J", true), org.assertj.core.groups.Tuple.tuple("R", false));
+ // A heading the page draws past its line is written in the flow, where its line is the
+ // page's and the cut is named; in the shape Word would grow the line instead.
+ Export heading = export(true, page -> page.add(badge("# *J*")));
+ assertThat(heading.document().getDocument().xmlText()).doesNotContain("");
+ assertThat(heading.notes("ParagraphNode")).singleElement().asString().contains("markdown heading").endsWith(HEADING_CUT);
}
@Test
@@ -276,8 +341,10 @@ void textOverTheFlowIsWrittenAsThePageSetsIt() throws Exception {
.position(new ParagraphBuilder().name("Monogram").text("**LM**").build(), 10, 10, LayerAlign.TOP_LEFT, 1)
.build()).addParagraph("Masthead"));
String body = export.document().getDocument().xmlText();
- assertThat(body).contains("").contains(">LM<").doesNotContain("**LM**");
- assertThat(export.notes("ParagraphNode")).allSatisfy(note -> assertThat(note).doesNotContain("markdown"));
+ assertThat(body).contains("").doesNotContain("**LM**")
+ .matches("(?s).*(?:(?!).)*LM.*");
+ assertThat(export.notes("ParagraphNode")).isNotEmpty()
+ .allSatisfy(note -> assertThat(note).doesNotContain("markdown"));
}
private static com.demcha.compose.document.node.DocumentNode badge(String initials) {
@@ -289,6 +356,7 @@ private static com.demcha.compose.document.node.DocumentNode badge(String initia
void theFacesThePiecesAreSetInTravelWithTheDocument() throws Exception {
Export export = export(true, page -> page.addParagraph(p -> p.text("Set **in** Lato *here*")
.textStyle(DocumentTextStyle.builder().fontName(FontName.LATO).build())));
+ assertThat(export.paragraphWith("Lato").getText()).as("written in its pieces").isEqualTo("Set in Lato here");
String table = partXml(export.document(), "/word/fontTable.xml");
assertThat(table).contains("