diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParse.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParse.groovy index bb81c901..d38dea60 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParse.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParse.groovy @@ -6,10 +6,17 @@ //! --- package com.quadient.migration.example.docx +import groovy.transform.Field +import org.slf4j.Logger +import org.slf4j.LoggerFactory + import static com.quadient.migration.example.common.util.InitMigration.initMigration import static com.quadient.migration.example.docx.parser.DocxTemplateParser.parseDocxFiles def migration = initMigration(this.binding) +@Field static Logger log = LoggerFactory.getLogger(this.class.name) -println("\nStarting Parse step...\n") +log.info "\nStarting Parse step...\n" parseDocxFiles(migration) +log.info "\nApplying persisted mapping...\n" +migration.mappingRepository.applyAll() diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParseWithApply.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParseWithApply.groovy deleted file mode 100644 index 5bcfe12d..00000000 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/DocxParseWithApply.groovy +++ /dev/null @@ -1,17 +0,0 @@ -//! --- -//! displayName: Parse DOCX and Apply Mapping -//! category: Parser -//! description: Parses input DOCX files specified in the project settings, translates their contents into the migration model, stores the resulting objects in the database and applies persisted mappings. -//! sourceFormat: DOCX -//! --- -package com.quadient.migration.example.docx - -import static com.quadient.migration.example.common.util.InitMigration.initMigration -import static com.quadient.migration.example.docx.parser.DocxTemplateParser.parseDocxFiles - -def migration = initMigration(this.binding) - -println("\nStarting Parse step...\n") -parseDocxFiles(migration) -println("\nApplying persisted mapping...\n") -migration.mappingRepository.applyAll() diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreas.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreas.groovy index 214f1600..2f69bda5 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreas.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreas.groovy @@ -18,6 +18,7 @@ import org.openxmlformats.schemas.drawingml.x2006.picture.CTPicture import org.openxmlformats.schemas.drawingml.x2006.wordprocessingDrawing.CTAnchor import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTP import org.w3c.dom.Node +import org.w3c.dom.Element import org.xml.sax.InputSource import javax.xml.parsers.DocumentBuilder @@ -40,6 +41,7 @@ class DocxAnchoredAreas { private static final String W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" private static final String WPS_SHAPE_URI = "http://schemas.microsoft.com/office/word/2010/wordprocessingShape" private static final String PIC_NS = "http://schemas.openxmlformats.org/drawingml/2006/picture" + private static final String VML_NS = "urn:schemas-microsoft-com:vml" private static final double BACKGROUND_COVERAGE_THRESHOLD = 0.6 final Map> backgroundAreasByPage = [:] @@ -67,7 +69,10 @@ class DocxAnchoredAreas { DocxAnchoredAreas result = new DocxAnchoredAreas(migration, doc, fileName) // Backgrounds are resolved over the whole document first so the same picture is never emitted as floating too. pages.each { DocxPage page -> result.eachUniqueAnchor(page) { result.addBackgroundArea(it, page) } } - pages.each { DocxPage page -> result.eachUniqueAnchor(page) { result.addFloatingArea(it, page) } } + pages.each { DocxPage page -> + result.eachUniqueAnchor(page) { result.addFloatingArea(it, page) } + result.eachUniqueVmlTextBox(page) { result.addVmlTextBoxArea(it, page) } + } return result } @@ -84,6 +89,21 @@ class DocxAnchoredAreas { } } + // Older Word documents use VML w:pict/v:shape text boxes rather than DrawingML wp:anchor shapes. + private void eachUniqueVmlTextBox(DocxPage page, Closure handler) { + Set seenShapeIds = [] + page.paragraphs().each { XWPFParagraph paragraph -> + paragraph.runs.each { XWPFRun run -> + findVmlTextBoxes(run).each { Element shape -> + String shapeId = vmlShapeId(shape) + if (seenShapeIds.add(shapeId ?: shape.textContent)) { + handler(shape) + } + } + } + } + } + private void addBackgroundArea(CTAnchor anchor, DocxPage page) { if (!isPicture(anchor) || !coversPage(anchor, page)) { return @@ -116,6 +136,13 @@ class DocxAnchoredAreas { } } + private void addVmlTextBoxArea(Element shape, DocxPage page) { + Area area = buildVmlTextBoxArea(shape, page) + if (area != null) { + floatingAreasByPage.computeIfAbsent(page.index) { [] }.add(area) + } + } + private static boolean isPicture(CTAnchor anchor) { return anchor.graphic?.graphicData?.uri == PIC_NS } @@ -132,6 +159,18 @@ class DocxAnchoredAreas { return found.collect { XmlObject o -> o instanceof CTAnchor ? o : CTAnchor.Factory.parse(o.xmlText()) } } + private List findVmlTextBoxes(XWPFRun run) { + // XMLBeans' XPath support can require optional Saxon classes. DOM traversal keeps this parser self-contained. + Element runDom = documentBuilder.parse(new InputSource(new StringReader(run.CTR.xmlText()))).documentElement + def shapes = runDom.getElementsByTagNameNS(VML_NS, 'shape') + return (0.. 0 } + } + + private static String vmlShapeId(Element shape) { + return shape.getAttribute('id') ?: null + } + private static String extractBlipEmbedId(CTAnchor anchor) { XmlObject[] pics = anchor.selectPath("declare namespace pic='${PIC_NS}' .//pic:pic") if (pics.length == 0) { @@ -161,32 +200,75 @@ class DocxAnchoredAreas { if (!cursor.toFirstChild()) { return null } - def shapeDom = documentBuilder.parse(new InputSource(new StringReader(cursor.xmlText()))) - def txbxContentNodes = shapeDom.getElementsByTagNameNS(W_NS, "txbxContent") - if (txbxContentNodes.length == 0) { - return null - } - List contentItems = [] - def children = txbxContentNodes.item(0).childNodes - for (int i = 0; i < children.length; i++) { - Node child = children.item(i) - if (child.nodeType == Node.ELEMENT_NODE && child.localName == 'p') { - XWPFParagraph paragraph = new XWPFParagraph(domParagraphToCtp(child), doc) - contentItems.add(parseParagraph(migration, paragraph, fileName)) - } - } - if (contentItems.isEmpty()) { - return null - } - String blockId = "${page.id(fileName)}_textbox${++textBoxes}" - String firstText = contentItems.findResult { extractParagraphText(it)?.trim() ?: null } - DocumentObjectRef blockRef = upsertBlock(migration, blockId, blockName(firstText, "text box", blockId), contentItems, fileName) - return new AreaBuilder().content([blockRef]).position(resolveAnchorPosition(anchor, page)).build() + return buildTextBoxArea(cursor.xmlText(), resolveAnchorPosition(anchor, page), page) } finally { cursor.dispose() } } + private Area buildVmlTextBoxArea(Element shape, DocxPage page) { + return buildTextBoxArea(shape, resolveVmlShapePosition(shape, page), page) + } + + private Area buildTextBoxArea(String shapeXml, Position position, DocxPage page) { + def shapeDom = documentBuilder.parse(new InputSource(new StringReader(shapeXml))) + return buildTextBoxArea(shapeDom.documentElement, position, page) + } + + private Area buildTextBoxArea(Element shapeDom, Position position, DocxPage page) { + def txbxContentNodes = shapeDom.getElementsByTagNameNS(W_NS, "txbxContent") + if (txbxContentNodes.length == 0) { + return null + } + List contentItems = [] + def children = txbxContentNodes.item(0).childNodes + for (int i = 0; i < children.length; i++) { + Node child = children.item(i) + if (child.nodeType == Node.ELEMENT_NODE && child.localName == 'p') { + XWPFParagraph paragraph = new XWPFParagraph(domParagraphToCtp(child), doc) + contentItems.add(parseParagraph(migration, paragraph, fileName)) + } + } + if (contentItems.isEmpty()) { + return null + } + String blockId = "${page.id(fileName)}_textbox${++textBoxes}" + String firstText = contentItems.findResult { extractParagraphText(it)?.trim() ?: null } + DocumentObjectRef blockRef = upsertBlock(migration, blockId, blockName(firstText, "text box", blockId), contentItems, fileName) + return new AreaBuilder().content([blockRef]).position(position).build() + } + + private Position resolveVmlShapePosition(Element shape, DocxPage page) { + Map style = vmlStyle(shape.getAttribute('style')) + double x = page.contentPosition.x.toPoints() + vmlPoints(style['margin-left']) + double y = page.contentPosition.y.toPoints() + vmlPoints(style['margin-top']) + return new Position(Size.ofPoints(x), Size.ofPoints(y), Size.ofPoints(vmlPoints(style['width'])), Size.ofPoints(vmlPoints(style['height']))) + } + + private static Map vmlStyle(String value) { + return value.split(';').collectEntries { String property -> + int separator = property.indexOf(':') + separator < 0 ? [:] : [(property.substring(0, separator).trim().toLowerCase(Locale.ROOT)): property.substring(separator + 1).trim()] + } + } + + // Word's VML geometry is CSS-like; point values dominate its generated documents, with the common alternatives + // handled here as well so that the resulting Area geometry stays in points. + private static double vmlPoints(String value) { + def matcher = value =~ /^([+-]?(?:\d+(?:\.\d*)?|\.\d+))(pt|in|cm|mm|px)?$/ + if (!matcher.matches()) { + return 0d + } + double number = matcher.group(1) as double + return switch (matcher.group(2)?.toLowerCase(Locale.ROOT)) { + case 'in' -> number * 72d + case 'cm' -> number * 72d / 2.54d + case 'mm' -> number * 72d / 25.4d + case 'px' -> number * 72d / 96d + default -> number + } + } + private CTP domParagraphToCtp(Node pNode) { StringWriter sw = new StringWriter() Transformer transformer = transformerFactory.newTransformer() diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxBodyContent.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxBodyContent.groovy index 9cf22b3d..5dfc5d9c 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxBodyContent.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxBodyContent.groovy @@ -2,7 +2,9 @@ package com.quadient.migration.example.docx.parser import com.quadient.migration.api.Migration import com.quadient.migration.api.dto.migrationmodel.DocumentContent +import com.quadient.migration.api.dto.migrationmodel.ColumnLayout import com.quadient.migration.api.dto.migrationmodel.builder.TableBuilder +import com.quadient.migration.shared.ColumnApplyTo import org.apache.poi.xwpf.usermodel.XWPFTable import org.apache.poi.xwpf.usermodel.XWPFTableCell import org.apache.xmlbeans.XmlObject @@ -94,18 +96,40 @@ class DocxBodyContent { if (topLevel == null) { return content } - List> sections = [[]] + List result = [] + List section = [] + int blockNumber = 0 + ColumnLayout pendingColumnLayout + Closure flushSection = { + if (section.isEmpty()) { + return + } + String id = "${pageId}_section${++blockNumber}" + String heading = headingLevels[section[0]] == topLevel ? extractParagraphText(section[0]) : null + List blockContent = pendingColumnLayout == null ? section : [new ColumnLayout( + pendingColumnLayout.numberOfColumns, pendingColumnLayout.gutterWidth, pendingColumnLayout.balancingType, + ColumnApplyTo.ThisBlockOnly)] + section + pendingColumnLayout = null + result.add(upsertBlock(migration, id, blockName(heading, null, id), blockContent, fileName)) + section.clear() + } content.each { DocumentContent item -> - if (headingLevels[item] == topLevel && !sections.last().isEmpty()) { - sections << [] + // A marker before a generated block belongs to that block, so its emitted scope is ThisBlockOnly. + if (item instanceof ColumnLayout) { + flushSection() + pendingColumnLayout = item + } else { + if (headingLevels[item] == topLevel && !section.isEmpty()) { + flushSection() + } + section << item } - sections.last() << item } - return sections.withIndex().collect { List items, int i -> - String id = "${pageId}_section${i + 1}" - String heading = headingLevels[items[0]] == topLevel ? extractParagraphText(items[0]) : null - upsertBlock(migration, id, blockName(heading, null, id), items, fileName) + flushSection() + if (pendingColumnLayout != null) { + result.add(pendingColumnLayout) } + return result } // Text of the first non-empty cell of the table's first row (typically the heading of the row). Word stores the diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxContentControls.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxContentControls.groovy new file mode 100644 index 00000000..2d36e5c3 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxContentControls.groovy @@ -0,0 +1,28 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.Migration +import com.quadient.migration.api.dto.migrationmodel.Paragraph +import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphBuilder +import org.apache.poi.xwpf.usermodel.XWPFSDT + +class DocxContentControls { + static String variableId(XWPFSDT control) { + return control.tag?.trim() ?: control.title?.trim() + } + + static boolean addInline(Migration migration, List textBuilders, XWPFSDT control, + String styleId, String fileName) { + String id = variableId(control) + if (!id) return false + DocxVariablePatterns.addVariable(migration, textBuilders, id, styleId, fileName) + return true + } + + static Paragraph parseBlock(Migration migration, XWPFSDT control, String fileName) { + String id = variableId(control) + if (!id) return null + List content = [] + DocxVariablePatterns.addVariable(migration, content, id, null, fileName) + return new ParagraphBuilder().content(content).build() + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFooters.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFooters.groovy index 72523cc3..8b2cc028 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFooters.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFooters.groovy @@ -54,11 +54,12 @@ class DocxHeaderFooters { XWPFHeaderFooter part = selectPart(headerOnly ? parts.headers : parts.footers, page) if (part != null) { Position position = headerOnly ? headerPosition(page) : footerPosition(page) - Set directImageEmbedIds = anchoredEmbedIds(part) + inlineEmbedIds(part) + Set directImageEmbedIds = anchoredEmbedIds(part) + inlineEmbedIds(part) + vmlEmbedIds(part) Area area = buildArea(migration, part, fileName, position, directImageEmbedIds) if (area != null) result.add(area) result.addAll(inlineImageAreas(migration, doc, part, fileName, position)) result.addAll(anchoredImageAreas(migration, doc, part, fileName, page, position)) + result.addAll(vmlImageAreas(migration, doc, part, fileName, position)) } return result } @@ -140,12 +141,20 @@ class DocxHeaderFooters { List areas = [] source.paragraphs.each { XWPFParagraph paragraph -> paragraph.runs.each { XWPFRun run -> + Set anchorEmbedIds = findAnchors(run).collect { CTAnchor anchor -> extractBlipEmbedId(anchor) } + .findAll().toSet() + boolean hasInlinePicture = false run.embeddedPictures.each { picture -> CTPicture ctPicture = picture.CTPicture - XWPFPictureData data = picture.pictureData ?: pictureData(doc, source, ctPicture?.blipFill?.blip?.embed) + String embedId = ctPicture?.blipFill?.blip?.embed + // POI can surface a wp:anchor through embeddedPictures. It is emitted below by + // anchoredImageAreas with its anchor offsets, so do not also treat it as an inline image. + if (embedId && anchorEmbedIds.contains(embedId)) return + XWPFPictureData data = picture.pictureData ?: pictureData(doc, source, embedId) addInlineImageArea(areas, migration, data, ctPicture, fileName, flowPosition) + hasInlinePicture = true } - if (!run.embeddedPictures.isEmpty()) return + if (hasInlinePicture) return findInlines(run).each { XmlObject inline -> CTPicture picture = inlinePicture(inline) addInlineImageArea(areas, migration, pictureData(doc, source, picture?.blipFill?.blip?.embed), picture, fileName, flowPosition) @@ -166,6 +175,29 @@ class DocxHeaderFooters { return areas } + private static List vmlImageAreas(Migration migration, XWPFDocument doc, XWPFHeaderFooter source, String fileName, + Position flowPosition) { + List areas = [] + source.paragraphs.each { XWPFParagraph paragraph -> + paragraph.runs.each { XWPFRun run -> + DocxVmlImages.extract(run).each { VmlImageSource vmlImage -> + if (!vmlImage.embedId) return + XWPFPictureData data = pictureData(doc, source, vmlImage.embedId) + if (data == null) return + String imageId = registerImageData(migration, data, fileName, vmlImage.options) + if (imageId != null) { + Size width = vmlImage.options?.resizeWidth + Size height = vmlImage.options?.resizeHeight + Position position = width != null && height != null + ? new Position(flowPosition.x, flowPosition.y, width, height) : flowPosition + areas.add(new AreaBuilder().imageRef(imageId).position(position).build()) + } + } + } + } + return areas + } + private static List findAnchors(XWPFRun run) { XmlObject[] found = run.CTR.selectPath("declare namespace wp='${WP_NS}' .//wp:anchor") return found.collect { XmlObject o -> o instanceof CTAnchor ? o : CTAnchor.Factory.parse(o.xmlText()) } @@ -201,6 +233,18 @@ class DocxHeaderFooters { return ids } + private static Set vmlEmbedIds(XWPFHeaderFooter source) { + Set ids = [] + source.paragraphs.each { XWPFParagraph paragraph -> + paragraph.runs.each { XWPFRun run -> + DocxVmlImages.extract(run).each { VmlImageSource image -> + if (image.embedId) ids.add(image.embedId) + } + } + } + return ids + } + private static void addInlineImageArea(List areas, Migration migration, XWPFPictureData data, CTPicture picture, String fileName, Position flowPosition) { if (data == null || picture == null) return diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxImages.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxImages.groovy index 6e8ea0cb..0f4d839d 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxImages.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxImages.groovy @@ -11,6 +11,10 @@ import org.apache.poi.common.usermodel.PictureType import org.apache.poi.xwpf.usermodel.XWPFPicture import org.apache.poi.xwpf.usermodel.XWPFPictureData import org.apache.poi.xwpf.usermodel.XWPFRun +import org.slf4j.Logger +import org.slf4j.LoggerFactory + +@Field static Logger log = LoggerFactory.getLogger(this.class.name) @Field static Map imageIdByChecksum = [:] @@ -22,6 +26,10 @@ static void resetImageState() { imageCounter = 0 } +static boolean hasRunImages(XWPFRun run) { + return !run.embeddedPictures.isEmpty() || DocxVmlImages.hasImages(run) +} + static void processRunImages(Migration migration, XWPFRun run, String fileName, List textBuilders, Set excludedEmbedIds = Collections.emptySet()) { run.getEmbeddedPictures().each { XWPFPicture picture -> XWPFPictureData data = picture.getPictureData() @@ -32,13 +40,27 @@ static void processRunImages(Migration migration, XWPFRun run, String fileName, if (embedId && excludedEmbedIds.contains(embedId)) { return } - String imageId = registerImageData(migration, data, fileName, resolveOptions(picture)) - if (imageId == null) { - return + addImageRef(textBuilders, registerImageData(migration, data, fileName, resolveOptions(picture))) + } + + DocxVmlImages.extract(run).each { VmlImageSource source -> + String imageId + if (source.embedId) { + if (excludedEmbedIds.contains(source.embedId)) { + return + } + XWPFPictureData data = run.document.getPictureDataByID(source.embedId) + imageId = data == null ? null : registerImageData(migration, data, fileName, source.options) + } else { + imageId = registerImageBytes(migration, source.bytes, fileName, source.imageType, source.options, source.checksum) } - ParagraphBuilder.TextBuilder textBuilder = new ParagraphBuilder.TextBuilder() - textBuilder.imageRef(imageId) - textBuilders.add(textBuilder) + addImageRef(textBuilders, imageId) + } +} + +private static void addImageRef(List textBuilders, String imageId) { + if (imageId != null) { + textBuilders.add(new ParagraphBuilder.TextBuilder().imageRef(imageId)) } } @@ -47,20 +69,25 @@ static String registerImageData(Migration migration, XWPFPictureData data, Strin if (checksum != null && imageIdByChecksum.containsKey(checksum)) { return imageIdByChecksum[checksum] } - ImageType imageType = toImageType(data.getPictureTypeEnum()) if (imageType == ImageType.Unknown) { - // Unsupported/unrecognized format (e.g. EMF/WMF vector metafiles) - skip rather than upsert unusable data. - println " Warning: Skipping embedded image with unsupported type: ${data.getPictureTypeEnum()}" + log.warn " Warning: Skipping embedded image with unsupported type: ${data.getPictureTypeEnum()}" return null } + return registerImageBytes(migration, data.getData(), fileName, imageType, options, checksum) +} + +private static String registerImageBytes(Migration migration, byte[] imageBytes, String fileName, ImageType imageType, + ImageOptions options, Long checksum) { + if (checksum != null && imageIdByChecksum.containsKey(checksum)) { + return imageIdByChecksum[checksum] + } String imageId = "${fileName}_img_${++imageCounter}" String storagePath = "${imageId}${imageType.extension()}" // Storage has both String and byte[] overloads. Keep the declared type so a malformed/empty picture payload // cannot make Groovy select neither overload at runtime. - byte[] imageBytes = data.getData() migration.storage.write(storagePath, imageBytes) def imageBuilder = new ImageBuilder(imageId) diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxMergeFields.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxMergeFields.groovy index 87a94da9..f7ae4b7f 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxMergeFields.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxMergeFields.groovy @@ -21,6 +21,8 @@ import org.apache.poi.xwpf.usermodel.XWPFTable import org.apache.xmlbeans.XmlObject import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTFldChar import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTText +import org.slf4j.Logger +import org.slf4j.LoggerFactory import static com.quadient.migration.example.docx.util.DocxUtils.sha256Hex @@ -133,6 +135,19 @@ class ParsedIfField { @Field static final char PLACEHOLDER = (char) 0xE000 +@Field static Logger log = LoggerFactory.getLogger(this.class.name) + +// Word-managed fields are deliberately separate from authored MERGEFIELD data. Add supported fields here rather +// than creating one-off parsing branches, and keep their source-independent model IDs in the system* namespace. +@Field +static final Map SYSTEM_FIELD_VARIABLE_IDS = [ + PAGE : 'systemPageNumber', + NUMPAGES : 'systemTotalPages', + SECTIONPAGES: 'systemSectionPages', + DATE : 'systemCurrentDate', + TIME : 'systemCurrentDateTime', +].asImmutable() + static List fieldChildren(XWPFRun run) { return run.CTR.selectPath("./*").findAll { it.domNode.localName in ["fldChar", "instrText"] } } @@ -207,12 +222,20 @@ private static void resolveField(Migration migration, FieldParseState state, Lis addMergeField(migration, textBuilders, fileName, instruction, field.styleId) return } + String systemVariableId = systemFieldVariableId(instruction) + if (systemVariableId) { + // Word-managed values remain ordinary migration variables. Their system* IDs prevent collisions with + // MERGEFIELD data and leave the migration author free to bind them to target system variables. + flushPendingIfFields(migration, state, textBuilders, fileName) + addSystemField(migration, textBuilders, fileName, systemVariableId, field.styleId) + return + } ParsedIfField parsed = parseIfField(field) if (parsed == null) { flushPendingIfFields(migration, state, textBuilders, fileName) if (field.containsTable()) { - println " Warning: Unsupported field wrapping a table, its tables are dropped: '${instruction.trim()}'" + log.warn " Warning: Unsupported field wrapping a table, its tables are dropped: '${instruction.trim()}'" } return } @@ -264,7 +287,7 @@ private static void emitBranch(Migration migration, FieldParseState state, List< if (state.conditionalTableHandler != null) { state.conditionalTableHandler.call(part.table, upsertDisplayRule(migration, path, fileName)) } else { - println " Warning: Table inside IF field is not supported in this context and is dropped." + log.warn " Warning: Table inside IF field is not supported in this context and is dropped." } } else if (part instanceof WordField) { emitNestedField(migration, state, textBuilders, fileName, part, path, blockLevel) @@ -288,7 +311,7 @@ private static void emitNestedField(Migration migration, FieldParseState state, } ParsedIfField nested = parseIfField(field) if (nested == null) { - println " Warning: Unsupported nested field inside IF branch is dropped: '${instruction.trim()}'" + log.warn " Warning: Unsupported nested field inside IF branch is dropped: '${instruction.trim()}'" return } ensureVariable(migration, nested.variableId, fileName) @@ -343,6 +366,14 @@ static void addMergeField(Migration migration, List textBuilders, String fileName, + String variableId, String textStyleId) { + ensureVariable(migration, variableId, fileName) + textBuilders.add(new ParagraphBuilder.TextBuilder() + .variableRef(variableId) + .styleRef(textStyleId)) +} + private static void ensureVariable(Migration migration, String variableId, String fileName) { if (migration.variableRepository.find(variableId) == null) { migration.variableRepository.upsert(new VariableBuilder(variableId) @@ -565,6 +596,11 @@ static boolean isMergeField(String fieldInstruction) { return fieldInstruction?.trim()?.toUpperCase(Locale.ROOT)?.startsWith("MERGEFIELD") } +static String systemFieldVariableId(String fieldInstruction) { + String keyword = fieldInstruction?.trim()?.tokenize()?.first()?.toUpperCase(Locale.ROOT) + return SYSTEM_FIELD_VARIABLE_IDS[keyword] +} + static String extractMergeFieldName(String fieldInstruction) { String instr = fieldInstruction?.trim() if (!instr) { diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageLayout.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageLayout.groovy index 2854d2f7..233a4212 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageLayout.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageLayout.groovy @@ -1,5 +1,7 @@ package com.quadient.migration.example.docx.parser +import com.quadient.migration.api.dto.migrationmodel.ColumnLayout +import com.quadient.migration.shared.ColumnApplyTo import com.quadient.migration.shared.Position import com.quadient.migration.shared.Size import groovy.transform.Field @@ -45,6 +47,21 @@ static List resolvePageSize(CTSectPr sectPr) { resolveDimension(sectPr?.pgSz?.h, DEFAULT_PAGE_SIZE[1])] } +/** + * Converts the section-wide Word column settings to flow content. A single column is Word's default and needs no + * model marker; placing a multi-column marker before the page flow makes the layout apply from the first paragraph. + */ +static ColumnLayout resolveColumnLayout(CTSectPr sectPr) { + def columns = sectPr?.cols + int numberOfColumns = columns?.num?.intValue() ?: 1 + if (numberOfColumns <= 1) { + return null + } + Double gutterPoints = twipsToPoints(columns.space) + return new ColumnLayout(numberOfColumns, gutterPoints != null ? Size.ofPoints(gutterPoints) : null, null, + ColumnApplyTo.WholeTemplate) +} + private static Size resolveDimension(Object twips, Size fallback) { Double points = twipsToPoints(twips) // Zero is a valid explicit margin, so do not use Groovy's truth-based fallback. diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageSections.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageSections.groovy index cb0a989b..2f5fa7cb 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageSections.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxPageSections.groovy @@ -58,51 +58,17 @@ class DocxPageSections { static List groupIntoPages(List sections) { List pages = [] sections.eachWithIndex { DocxSection section, int i -> - // A section can contain Word's cached pagination marker. It is not an authored page break, so only - // honor it where it begins a paragraph; using a marker after text would move already-laid-out content. - splitAtRenderedPageBreaks(section).eachWithIndex { DocxSection fragment, int fragmentIndex -> - if ((i == 0 && fragmentIndex == 0) || fragmentIndex > 0 || startsNewPage(section.sectPr)) { - pages.add(newPage(pages.size(), fragment)) - } else { - pages.last().sections.add(fragment) - } + // lastRenderedPageBreak is Word's cached layout output, not an authored page boundary. Mapping it to + // fixed design-time pages makes a single flowing template repeat headers only for the cached pages. + if (i == 0 || startsNewPage(section.sectPr)) { + pages.add(newPage(pages.size(), section)) + } else { + pages.last().sections.add(section) } } return pages } - private static List splitAtRenderedPageBreaks(DocxSection section) { - if (!section.elements.any { it instanceof XWPFParagraph && beginsAfterRenderedPageBreak(it) }) { - // Preserve the original section object when no cached pagination is involved. - return [section] - } - List fragments = [] - DocxSection current = new DocxSection(sectPr: section.sectPr) - section.elements.each { IBodyElement element -> - if (element instanceof XWPFParagraph && beginsAfterRenderedPageBreak(element) && !current.elements.isEmpty()) { - fragments.add(current) - current = new DocxSection(sectPr: section.sectPr) - } - current.elements.add(element) - } - if (!current.elements.isEmpty()) { - fragments.add(current) - } - return fragments - } - - private static boolean beginsAfterRenderedPageBreak(XWPFParagraph paragraph) { - String xml = paragraph.CTP.xmlText() - def marker = xml =~ /<(?:[A-Za-z_][\w.-]*:)?lastRenderedPageBreak(?=[\s\/>])/ - if (!marker.find()) { - return false - } - // These are the WordprocessingML elements that produce visible paragraph content. The marker is safe to - // treat as a page boundary only when none of them precedes it. - def visibleContent = xml =~ /<(?:[A-Za-z_][\w.-]*:)?(?:t|instrText|drawing|tab|br|object)(?=[\s\/>])/ - return !visibleContent.find() || marker.start() < visibleContent.start() - } - private static DocxPage newPage(int index, DocxSection firstSection) { List size = DocxPageLayout.resolvePageSize(firstSection.sectPr) return new DocxPage( diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParser.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParser.groovy index 7d759457..9c1d5175 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParser.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParser.groovy @@ -3,15 +3,23 @@ package com.quadient.migration.example.docx.parser import com.quadient.migration.api.Migration import com.quadient.migration.api.dto.migrationmodel.Paragraph import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphBuilder +import groovy.transform.Field import org.apache.poi.xwpf.usermodel.XWPFParagraph import org.apache.poi.xwpf.usermodel.XWPFFieldRun +import org.apache.poi.xwpf.usermodel.XWPFHyperlinkRun import org.apache.poi.xwpf.usermodel.XWPFRun +import org.apache.poi.xwpf.usermodel.XWPFSDT +import org.apache.poi.xwpf.usermodel.IRunElement import org.apache.xmlbeans.XmlObject import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTSimpleField +import org.slf4j.Logger +import org.slf4j.LoggerFactory import static com.quadient.migration.example.docx.style.DocxParagraphStyles.captureParagraphStyle import static com.quadient.migration.example.docx.style.DocxTextStyles.captureTextStyle +@Field static Logger log = LoggerFactory.getLogger(this.class.name) + class ParagraphContentCollector { private final Migration migration private final String fileName @@ -38,20 +46,36 @@ class ParagraphContentCollector { } void addRun(XWPFRun run, String styleId) { - if (!run.embeddedPictures.isEmpty()) { + if (DocxImages.hasRunImages(run)) { flushText() flushFields() DocxImages.processRunImages(migration, run, fileName, textBuilders, excludedImageEmbedIds) } + if (run instanceof XWPFHyperlinkRun) { + String url = hyperlinkUrl(run as XWPFHyperlinkRun) + String displayText = run.text() + if (url && displayText) { + flushText() + flushFields() + textBuilders.add(new ParagraphBuilder.TextBuilder().hyperlink(url, displayText, null).styleRef(styleId)) + return + } + } + CTSimpleField simpleField = run instanceof XWPFFieldRun ? (run as XWPFFieldRun).CTField : null if (simpleField != null && !resolvedSimpleFields.contains(simpleField)) { String instruction = simpleField.instr - if (DocxMergeFields.isMergeField(instruction)) { + String systemVariableId = DocxMergeFields.systemFieldVariableId(instruction) + if (DocxMergeFields.isMergeField(instruction) || systemVariableId) { resolvedSimpleFields.add(simpleField) flushText() flushFields() - DocxMergeFields.addMergeField(migration, textBuilders, fileName, instruction, styleId) + if (systemVariableId) { + DocxMergeFields.addSystemField(migration, textBuilders, fileName, systemVariableId, styleId) + } else { + DocxMergeFields.addMergeField(migration, textBuilders, fileName, instruction, styleId) + } return } } @@ -71,6 +95,14 @@ class ParagraphContentCollector { DocxMergeFields.handleFieldChildren(migration, fieldChildren, styleId, fieldState, textBuilders, fileName) } + void addContentControl(XWPFSDT control, String styleId) { + flushText() + flushFields() + if (!DocxContentControls.addInline(migration, textBuilders, control, styleId, fileName)) { + appendText(control.content.text, styleId) + } + } + List finish() { flushText() flushFields() @@ -104,6 +136,11 @@ class ParagraphContentCollector { private void flushFields() { DocxMergeFields.flushPendingIfFields(migration, fieldState, textBuilders, fileName) } + + private static String hyperlinkUrl(XWPFHyperlinkRun run) { + String externalUrl = run.getHyperlink(run.document)?.URL + return externalUrl ?: (run.anchor ? "#${run.anchor}" : null) + } } static Paragraph parseParagraph(Migration migration, XWPFParagraph paragraph, String fileName, String context = null, @@ -121,8 +158,12 @@ static Paragraph parseFlowParagraph(Migration migration, XWPFParagraph paragraph String paragraphStyleId = paragraph.styleID ?: "unknown" ParagraphContentCollector collector = new ParagraphContentCollector(migration, fileName, excludedImageEmbedIds, fieldState) - paragraph.runs.each { XWPFRun run -> - collector.addRun(run, captureTextStyle(migration, run, fileName, paragraphStyleId, context)) + paragraph.getIRuns().each { IRunElement run -> + if (run instanceof XWPFRun) { + collector.addRun(run as XWPFRun, captureTextStyle(migration, run as XWPFRun, fileName, paragraphStyleId, context)) + } else if (run instanceof XWPFSDT) { + collector.addContentControl(run as XWPFSDT, null) + } } List content = collector.finish() @@ -134,7 +175,7 @@ static Paragraph parseFlowParagraph(Migration migration, XWPFParagraph paragraph static void warnUnterminatedField(FieldParseState fieldState, String location) { if (fieldState.depth > 0) { - println " Warning: Unterminated complex field (missing fldChar end) in ${location}" + log.warn " Warning: Unterminated complex field (missing fldChar end) in ${location}" fieldState.stack.clear() } } diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTableParser.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTableParser.groovy index f5509485..2d37ca7e 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTableParser.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTableParser.groovy @@ -2,14 +2,20 @@ package com.quadient.migration.example.docx.parser import com.quadient.migration.api.Migration import com.quadient.migration.api.dto.migrationmodel.DocumentContent +import com.quadient.migration.api.dto.migrationmodel.ImageRef +import com.quadient.migration.api.dto.migrationmodel.Paragraph +import com.quadient.migration.api.dto.migrationmodel.StringValue import com.quadient.migration.api.dto.migrationmodel.Table import com.quadient.migration.api.dto.migrationmodel.builder.TableBuilder import com.quadient.migration.shared.TableAlignment import com.quadient.migration.shared.TablePdfTaggingRule import groovy.transform.Field +import org.apache.poi.xwpf.usermodel.XWPFParagraph import org.apache.poi.xwpf.usermodel.XWPFTable import org.apache.poi.xwpf.usermodel.XWPFTableCell import org.apache.poi.xwpf.usermodel.XWPFTableRow +import org.slf4j.Logger +import org.slf4j.LoggerFactory import static com.quadient.migration.example.docx.parser.DocxParagraphParser.parseFlowParagraph import static com.quadient.migration.example.docx.parser.DocxParagraphParser.parseParagraph @@ -22,6 +28,8 @@ import static com.quadient.migration.example.docx.util.DocxUtils.isHorizontallyM @Field static final long GRID_TOLERANCE_TWIPS = 30 +@Field static Logger log = LoggerFactory.getLogger(this.class.name) + static Table parseTable(Migration migration, XWPFTable table, String fileName) { return buildTable(migration, table, table.rows, fileName, resolveExpectedColumnCount(table), true) } @@ -64,7 +72,7 @@ static void addRows(Migration migration, TableBuilder tableBuilder, XWPFTable so int expectedColumnCount, boolean useHeader, String displayRuleId = null) { rows.eachWithIndex { XWPFTableRow row, int ri -> if (row.tableCells.size() != expectedColumnCount) { - println " Warning: Row ${ri + 1} has ${row.tableCells.size()} cells, expected ${expectedColumnCount}." + log.warn " Warning: Row ${ri + 1} has ${row.tableCells.size()} cells, expected ${expectedColumnCount}." } boolean isHeader = useHeader && ri == 0 && rows.size() > 1 TableBuilder.Row rowBuilder = isHeader ? tableBuilder.addFirstHeaderRow() : tableBuilder.addRow() @@ -85,7 +93,7 @@ static void addRows(Migration migration, TableBuilder tableBuilder, XWPFTable so } int missingCells = expectedColumnCount - rowBuilder.cells.size() if (missingCells > 0) { - println " Adding ${missingCells} empty cells to row ${ri + 1} to match expected column count." + log.warn " Adding ${missingCells} empty cells to row ${ri + 1} to match expected column count." missingCells.times { rowBuilder.addCell().mergeLeft = true } } } @@ -94,10 +102,79 @@ static void addRows(Migration migration, TableBuilder tableBuilder, XWPFTable so // Paragraphs of one cell share the field state so an IF whose instruction and result sit in different paragraphs resolves. private static List parseCellParagraphs(Migration migration, XWPFTableCell cell, String fileName, String context) { FieldParseState fieldState = new FieldParseState() - List paragraphs = cell.paragraphs.findResults { parseFlowParagraph(migration, it, fileName, fieldState, context) } + List parsedParagraphs = cell.paragraphs.findResults { XWPFParagraph paragraph -> + Paragraph parsed = parseFlowParagraph(migration, paragraph, fileName, fieldState, context) + parsed == null ? null : [source: paragraph, parsed: parsed] + } warnUnterminatedField(fieldState, "table cell: '${cell.text}'") + List paragraphs = mergeVmlImageAnchorParagraphs(parsedParagraphs) if (paragraphs.isEmpty() && cell.paragraphs) { paragraphs.add(parseParagraph(migration, cell.paragraphs.first(), fileName, context)) } return paragraphs } + +private static List mergeVmlImageAnchorParagraphs(List paragraphs) { + List result = [] + for (int i = 0; i < paragraphs.size(); i++) { + Map current = paragraphs[i] + Paragraph images = isVmlImageAnchorParagraph(current.source as XWPFParagraph) + ? keepOnlyImages(current.parsed as Paragraph) + : null + Map next = i + 1 < paragraphs.size() ? paragraphs[i + 1] : null + if (images != null && next != null && !isVmlImageAnchorParagraph(next.source as XWPFParagraph) + && hasVisibleContent(next.parsed as Paragraph)) { + result.add(prependImagesWithNaturalSpacing(images, next.parsed as Paragraph)) + i++ + } else { + result.add(current.parsed as Paragraph) + } + } + return result +} + +private static boolean isVmlImageAnchorParagraph(XWPFParagraph paragraph) { + return DocxVmlImages.hasAbsolutelyPositionedImage(paragraph) && !paragraph.text?.trim() +} + +private static Paragraph keepOnlyImages(Paragraph paragraph) { + List images = paragraph.content.findResults { Paragraph.Text text -> + def imageRefs = text.content.findAll { it instanceof ImageRef } + imageRefs ? new Paragraph.Text(imageRefs, text.styleRef, text.displayRuleRef) : null + } + return images.isEmpty() ? null : new Paragraph(images, paragraph.styleRef, paragraph.displayRuleRef) +} + +private static boolean hasVisibleContent(Paragraph paragraph) { + return paragraph.content.any { Paragraph.Text text -> + text.content.any { !(it instanceof StringValue) || it.value?.trim() } + } +} + +private static Paragraph prependImagesWithNaturalSpacing(Paragraph images, Paragraph paragraph) { + return new Paragraph(images.content + replaceLeadingWhitespaceWithSingleSpace(paragraph.content), + paragraph.styleRef, paragraph.displayRuleRef) +} + +private static List replaceLeadingWhitespaceWithSingleSpace(List texts) { + boolean leading = true + return texts.findResults { Paragraph.Text text -> + List trimmedContent = [] + text.content.each { item -> + if (leading && item instanceof StringValue) { + String value = item.value.replaceFirst(/^\s+/, '') + if (value) { + trimmedContent.add(new StringValue(" ${value}")) + leading = false + } + } else { + if (leading) { + trimmedContent.add(new StringValue(" ")) + } + trimmedContent.add(item) + leading = false + } + } + trimmedContent.isEmpty() ? null : new Paragraph.Text(trimmedContent, text.styleRef, text.displayRuleRef) + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTemplateParser.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTemplateParser.groovy index 2e9b1cf0..c817bd6d 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTemplateParser.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxTemplateParser.groovy @@ -2,6 +2,7 @@ package com.quadient.migration.example.docx.parser import com.quadient.migration.api.Migration import com.quadient.migration.api.dto.migrationmodel.Area +import com.quadient.migration.api.dto.migrationmodel.ColumnLayout import com.quadient.migration.api.dto.migrationmodel.DocumentObject import com.quadient.migration.api.dto.migrationmodel.DocumentObjectRef import com.quadient.migration.api.dto.migrationmodel.PageOptions @@ -13,9 +14,13 @@ import com.quadient.migration.api.dto.migrationmodel.builder.TableBuilder import com.quadient.migration.api.dto.migrationmodel.builder.documentcontent.AreaBuilder import com.quadient.migration.shared.DocumentObjectType import groovy.io.FileType +import groovy.transform.Field import org.apache.poi.xwpf.usermodel.XWPFDocument import org.apache.poi.xwpf.usermodel.XWPFParagraph import org.apache.poi.xwpf.usermodel.XWPFTable +import org.apache.poi.xwpf.usermodel.XWPFSDT +import org.slf4j.Logger +import org.slf4j.LoggerFactory import java.util.regex.Pattern @@ -32,6 +37,8 @@ import static com.quadient.migration.example.docx.parser.DocxTableParser.createT import static com.quadient.migration.example.docx.parser.DocxTableParser.parseTable import static com.quadient.migration.example.docx.parser.DocxTableParser.resolveExpectedColumnCount +@Field static Logger log = LoggerFactory.getLogger(this.class.name) + static void parseDocxFiles(Migration migration) { List inputFiles = [] new File(migration.projectConfig.inputDataPath).eachFileRecurse(FileType.FILES) { File file -> @@ -47,7 +54,7 @@ static void parseDocxFile(Migration migration, File file) { String documentType = file.parentFile.name String relativePath = new File(migration.projectConfig.inputDataPath).toPath().relativize(file.toPath()).toString() - println("=== Processing: " + relativePath + " ===") + log.info("=== Processing: " + relativePath + " ===") DocumentObjectBuilder builder = new DocumentObjectBuilder(fileName, DocumentObjectType.Template) .name(fileName) .originLocations([relativePath]) @@ -84,26 +91,36 @@ static List parsePages(Migration migration, File docxFile, Docum DocxBodyContent body = new DocxBodyContent(migration, fileName, pageId) FieldParseState fieldState = new FieldParseState(conditionalTableHandler: body.&addConditionalTable) int floatingTables = 0 - page.bodyElements().each { elem -> - if (elem instanceof XWPFParagraph) { - Paragraph paragraph = parseFlowParagraph(migration, elem, fileName, fieldState, null, anchoredAreas.consumedEmbedIds) - if (paragraph != null) { - body.add(paragraph, headingLevel(elem)) - } - } else if (elem instanceof XWPFTable) { - if (fieldState.depth > 0) { - // The table sits inside an open IF field; it is emitted (with a display rule) once the field resolves. - handleTableInsideField(fieldState, elem) - } else if (DocxFloatingTables.isFloating(elem)) { - Area area = buildFloatingTableArea(migration, elem, page, fileName, "${pageId}_floating_table${++floatingTables}") - if (area != null) { - anchoredAreas.floatingAreasByPage.computeIfAbsent(page.index) { [] }.add(area) + page.sections.each { DocxSection section -> + ColumnLayout columnLayout = DocxPageLayout.resolveColumnLayout(section.sectPr) + if (columnLayout != null) { + // A continuous section stays on the current page, so retain its position in the page flow. + body.add(columnLayout) + } + section.elements.each { elem -> + if (elem instanceof XWPFParagraph) { + Paragraph paragraph = parseFlowParagraph(migration, elem, fileName, fieldState, null, anchoredAreas.consumedEmbedIds) + if (paragraph != null) { + body.add(paragraph, headingLevel(elem)) + } + } else if (elem instanceof XWPFSDT) { + Paragraph paragraph = DocxContentControls.parseBlock(migration, elem as XWPFSDT, fileName) + if (paragraph != null) body.add(paragraph) + } else if (elem instanceof XWPFTable) { + if (fieldState.depth > 0) { + // The table sits inside an open IF field; it is emitted (with a display rule) once the field resolves. + handleTableInsideField(fieldState, elem) + } else if (DocxFloatingTables.isFloating(elem)) { + Area area = buildFloatingTableArea(migration, elem, page, fileName, "${pageId}_floating_table${++floatingTables}") + if (area != null) { + anchoredAreas.floatingAreasByPage.computeIfAbsent(page.index) { [] }.add(area) + } + } else { + addBodyTable(migration, body, elem, fileName) } } else { - addBodyTable(migration, body, elem, fileName) + otherElements++ } - } else { - otherElements++ } } warnUnterminatedField(fieldState, "page ${page.index + 1} body") diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxVariablePatterns.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxVariablePatterns.groovy index b64f5217..29c07be1 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxVariablePatterns.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxVariablePatterns.groovy @@ -4,6 +4,8 @@ import com.quadient.migration.api.Migration import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphBuilder import com.quadient.migration.api.dto.migrationmodel.builder.VariableBuilder import com.quadient.migration.shared.DataType +import org.slf4j.Logger +import org.slf4j.LoggerFactory import java.util.regex.Pattern @@ -13,6 +15,7 @@ import java.util.regex.Pattern */ class DocxVariablePatterns { static final String CONTEXT_KEY = "docxVariablePatterns" + private static final Logger log = LoggerFactory.getLogger(DocxVariablePatterns) static void addText(Migration migration, List textBuilders, String text, String styleId, String fileName) { @@ -48,6 +51,15 @@ class DocxVariablePatterns { addLiteral(textBuilders, text.substring(offset), styleId) } + static void addVariable(Migration migration, List textBuilders, String variableId, + String styleId, String fileName) { + if (!variableId) return + ensureVariable(migration, variableId, fileName) + ParagraphBuilder.TextBuilder builder = new ParagraphBuilder.TextBuilder().variableRef(variableId) + if (styleId) builder.styleRef(styleId) + textBuilders.add(builder) + } + private static List configuredPatterns(Migration migration) { def configured = migration.projectConfig.context?.get(CONTEXT_KEY) Collection values = configured instanceof Collection ? configured : configured == null ? [] : [configured] @@ -55,12 +67,12 @@ class DocxVariablePatterns { try { Pattern pattern = value instanceof Pattern ? value : Pattern.compile(value.toString()) if (pattern.matcher("").groupCount() < 1) { - println " Warning: DOCX variable pattern '${value}' has no capture group; it is ignored." + log.warn " Warning: DOCX variable pattern '${value}' has no capture group; it is ignored." return null } pattern } catch (Exception e) { - println " Warning: Invalid DOCX variable pattern '${value}': ${e.message}" + log.warn " Warning: Invalid DOCX variable pattern '${value}': ${e.message}" null } }.findAll() diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxVmlImages.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxVmlImages.groovy new file mode 100644 index 00000000..4e8de404 --- /dev/null +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/parser/DocxVmlImages.groovy @@ -0,0 +1,191 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.shared.ImageOptions +import com.quadient.migration.shared.ImageType +import com.quadient.migration.shared.Size +import groovy.transform.Field +import org.apache.poi.xwpf.usermodel.XWPFParagraph +import org.apache.poi.xwpf.usermodel.XWPFRun +import org.apache.xmlbeans.XmlCursor +import org.apache.xmlbeans.XmlObject + +import javax.xml.namespace.QName +import java.nio.charset.StandardCharsets +import java.util.zip.CRC32 + +@Field +private static final String VML_NS = "urn:schemas-microsoft-com:vml" +@Field +private static final String REL_NS = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" + +class VmlImageSource { + String embedId + byte[] bytes + ImageType imageType + ImageOptions options + Long checksum +} + +static boolean hasImages(XWPFRun run) { + return !findShapes(run).isEmpty() +} + +static boolean hasAbsolutelyPositionedImage(XWPFParagraph paragraph) { + return paragraph.runs.any { XWPFRun run -> + findShapes(run).any { XmlObject shape -> + String style = attribute(shape, "style") + style && (style =~ /(?i)(?:^|;)\s*position\s*:\s*absolute\s*(?:;|$)/).find() + } + } +} + +static List extract(XWPFRun run) { + return findShapes(run).findResults { XmlObject shape -> toImageSource(shape) } +} + +private static VmlImageSource toImageSource(XmlObject shape) { + ImageOptions options = resolveOptions(attribute(shape, "style")) + XmlObject[] imageData = shape.selectPath("declare namespace v='${VML_NS}' .//v:imagedata") + if (imageData.length > 0) { + String embedId = attribute(imageData[0], "id", REL_NS) + return embedId ? new VmlImageSource(embedId: embedId, options: options) : null + } + + String svg = shapeToSvg(shape) + if (svg == null) { + return null + } + byte[] bytes = svg.getBytes(StandardCharsets.UTF_8) + CRC32 crc = new CRC32() + crc.update(bytes) + return new VmlImageSource(bytes: bytes, imageType: ImageType.Svg, options: options, checksum: crc.value) +} + +private static List findShapes(XWPFRun run) { + return run.CTR.selectPath("declare namespace v='${VML_NS}' .//v:shape").toList() +} + +private static ImageOptions resolveOptions(String style) { + Double width = styleSizeInPoints(style, "width") + Double height = styleSizeInPoints(style, "height") + return width != null && height != null && width > 0 && height > 0 + ? new ImageOptions(Size.ofPoints(width), Size.ofPoints(height)) + : null +} + +private static Double styleSizeInPoints(String style, String property) { + if (!style) { + return null + } + def matcher = style =~ /(?i)(?:^|;)\s*${property}\s*:\s*([+-]?(?:\d+(?:\.\d*)?|\.\d+))\s*(pt|px|in|cm|mm)?\s*(?:;|$)/ + if (!matcher.find()) { + return null + } + double value = matcher.group(1) as double + switch ((matcher.group(2) ?: "pt").toLowerCase()) { + case "px": return value * 0.75d + case "in": return value * 72.0d + case "cm": return value * 72.0d / 2.54d + case "mm": return value * 72.0d / 25.4d + default: return value + } +} + +private static String shapeToSvg(XmlObject shape) { + String path = vmlPathToSvgPath(attribute(shape, "path")) + List coordSize = coordinatePair(attribute(shape, "coordsize")) + if (!path || coordSize == null || coordSize.any { it <= 0 }) { + return null + } + List coordOrigin = coordinatePair(attribute(shape, "coordorigin")) ?: [0.0d, 0.0d] + String fill = attribute(shape, "filled") == "f" ? "none" : safeColor(attribute(shape, "fillcolor"), "white") + String stroke = attribute(shape, "stroked") == "f" ? "none" : safeColor(attribute(shape, "strokecolor"), "black") + return "" +} + +static String vmlPathToSvgPath(String vmlPath) { + if (!vmlPath) { + return null + } + StringBuilder svg = new StringBuilder() + int index = 0 + while (index < vmlPath.length()) { + while (index < vmlPath.length() && (Character.isWhitespace(vmlPath.charAt(index)) || vmlPath.charAt(index) == ',')) index++ + if (index >= vmlPath.length()) break + if (!Character.isLetter(vmlPath.charAt(index))) return null + String command = Character.toString(vmlPath.charAt(index++)).toLowerCase() + int argsStart = index + while (index < vmlPath.length() && !Character.isLetter(vmlPath.charAt(index))) index++ + List args = coordinates(vmlPath.substring(argsStart, index)) + + switch (command) { + case "m": + case "l": + case "t": + case "r": + if (!appendPairs(svg, command == "m" ? "M" : command == "l" ? "L" : command == "t" ? "m" : "l", args, command == "m")) return null + break + case "c": + case "v": + if (!appendGroups(svg, command == "c" ? "C" : "c", args, 6)) return null + break + case "x": + appendCommand(svg, "Z", []) + break + case "e": + return svg.toString() + default: + return null + } + } + return svg.toString() +} + +private static boolean appendPairs(StringBuilder svg, String command, List args, boolean move) { + if (args.isEmpty() || args.size() % 2 != 0) return false + args.collate(2).eachWithIndex { List pair, int i -> + appendCommand(svg, move && i > 0 ? "L" : command, pair) + } + return true +} + +private static boolean appendGroups(StringBuilder svg, String command, List args, int groupSize) { + if (args.isEmpty() || args.size() % groupSize != 0) return false + args.collate(groupSize).each { appendCommand(svg, command, it) } + return true +} + +private static void appendCommand(StringBuilder svg, String command, List args) { + if (svg.length() > 0) svg.append(' ') + svg.append(command) + if (!args.isEmpty()) svg.append(' ').append(args.collect { number(it) }.join(' ')) +} + +private static List coordinates(String text) { + String normalized = text.trim().replaceAll(/\s+/, ',') + if (!normalized) return [] + return normalized.split(',', -1).collect { it ? it as double : 0.0d } +} + +private static List coordinatePair(String text) { + if (!text) return null + List values = coordinates(text) + return values.size() == 2 ? values : null +} + +private static String safeColor(String value, String fallback) { + return value ==~ /(?i)(?:#[0-9a-f]{3,8}|[a-z]+|none)/ ? value : fallback +} + +private static String number(double value) { + return value == Math.rint(value) ? Long.toString(value as long) : BigDecimal.valueOf(value).stripTrailingZeros().toPlainString() +} + +private static String attribute(XmlObject object, String localName, String namespace = "") { + XmlCursor cursor = object.newCursor() + try { + return cursor.getAttributeText(new QName(namespace, localName)) + } finally { + cursor.dispose() + } +} diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxTextStyles.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxTextStyles.groovy index 64781a0e..4ae2fcb4 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxTextStyles.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/DocxTextStyles.groovy @@ -19,7 +19,7 @@ static String captureTextStyle(Migration migration, XWPFRun run, String fileName // Attribute order and fallbacks feed the style id hash; changing them invalidates persisted mappings. LinkedHashMap styleAttributes = [ name : styleId, - fontName: resolveFirst(rPrChain, StyleChainResolver.&resolveFontName), + fontName: StyleChainResolver.resolveEffectiveFontName(rPrChain), fontSize: (resolveFirst(rPrChain, StyleChainResolver.&resolveFontSize) ?: -1) as double, bold : resolveFirst(rPrChain, StyleChainResolver.&resolveBold) ?: false, italic : resolveFirst(rPrChain, StyleChainResolver.&resolveItalic) ?: false, diff --git a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/StyleChainResolver.groovy b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/StyleChainResolver.groovy index 2a626b0a..30af8b37 100644 --- a/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/StyleChainResolver.groovy +++ b/migration-examples/src/main/groovy/com/quadient/migration/example/docx/style/StyleChainResolver.groovy @@ -50,11 +50,23 @@ private static List buildPropertyChain(T directProperties, XWPFStyles doc static String resolveFontName(CTRPr rPr) { if (rPr?.getRFontsList()) { def font = rPr.getRFontsList().get(0) - return font.getAscii() ?: font.getHAnsi() ?: font.getEastAsia() + return font.getAscii() ?: font.getHAnsi() } return null } +// Word font slots are script-specific. A direct East Asian override must not replace an inherited Latin font for +// English text (as in 01CVRPG0222.docx); use it only when the complete style chain lacks ascii/hAnsi data. +static String resolveEffectiveFontName(List chain) { + String latinFont = resolveFirst(chain, StyleChainResolver.&resolveFontName) + if (latinFont) { + return latinFont + } + return resolveFirst(chain) { CTRPr rPr -> + rPr?.getRFontsList() ? rPr.getRFontsList().get(0).getEastAsia() : null + } +} + static Double resolveFontSize(CTRPr rPr) { if (rPr?.getSzList()) { def val = rPr.getSzList().get(0).getVal() diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/DocxFieldFixtures.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/DocxFieldFixtures.groovy index e12db769..d291a48b 100644 --- a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/DocxFieldFixtures.groovy +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/DocxFieldFixtures.groovy @@ -15,6 +15,16 @@ class DocxFieldFixtures { fieldCharacter(paragraph.createRun(), STFldCharType.END) } + static void appendPageField(paragraph, String instructionText, String cachedResult = null) { + fieldCharacter(paragraph.createRun(), STFldCharType.BEGIN) + instruction(paragraph.createRun(), instructionText) + if (cachedResult != null) { + fieldCharacter(paragraph.createRun(), STFldCharType.SEPARATE) + paragraph.createRun().setText(cachedResult) + } + fieldCharacter(paragraph.createRun(), STFldCharType.END) + } + static void appendEqualityIfField(paragraph, String variable, String operand, String result) { fieldCharacter(paragraph.createRun(), STFldCharType.BEGIN) instruction(paragraph.createRun(), " IF ") diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreasTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreasTest.groovy index ec5a1df4..f3c5536c 100644 --- a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreasTest.groovy +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxAnchoredAreasTest.groovy @@ -2,11 +2,14 @@ package com.quadient.migration.example.docx.parser import com.quadient.migration.shared.Position import com.quadient.migration.shared.Size +import com.quadient.migration.api.dto.migrationmodel.DocumentObject +import com.quadient.migration.api.dto.migrationmodel.VariableRef import org.apache.poi.common.usermodel.PictureType import org.apache.poi.xwpf.usermodel.XWPFDocument import org.apache.poi.xwpf.usermodel.XWPFPictureData import org.apache.xmlbeans.XmlObject import org.junit.jupiter.api.Test +import org.mockito.ArgumentCaptor import org.junit.jupiter.params.ParameterizedTest import org.junit.jupiter.params.provider.CsvSource import org.openxmlformats.schemas.drawingml.x2006.wordprocessingDrawing.CTAnchor @@ -16,9 +19,42 @@ import org.openxmlformats.schemas.drawingml.x2006.wordprocessingDrawing.STRelFro import static org.mockito.Mockito.mock import static org.mockito.Mockito.when +import static org.mockito.Mockito.verify import static com.quadient.migration.example.Utils.mockMigration class DocxAnchoredAreasTest { + @Test + void "legacy VML text box becomes a positioned area with parsed merge fields"() { + // given: the VML w:pict/v:shape form used by KB47 - Velkommen til GF Grænsen_Følgebrev.docx + new XWPFDocument().withCloseable { document -> + document.createStyles() + def paragraph = document.createParagraph() + def run = paragraph.createRun() + run.CTR.set(XmlObject.Factory.parse(''' + + + MERGEFIELD customer_name + Customer name + + + ''')) + def page = page() + page.sections = [new DocxSection(elements: [paragraph])] + def migration = mockMigration() + + // when + def result = DocxAnchoredAreas.extract(migration, document, 'sample', [page]) + + // then: VML CSS geometry is relative to the content origin and the cached merge value is ignored + def area = result.floatingAreasByPage[0][0] + assert [area.position.x, area.position.y, area.position.width, area.position.height]*.toPoints() == [52d, 78d, 120d, 36d] + def blockCaptor = ArgumentCaptor.forClass(DocumentObject) + verify(migration.documentObjectRepository).upsert(blockCaptor.capture()) + assert blockCaptor.value.content[0].content[0].content[0] == new VariableRef('customer_name') + } + } + @Test void "page-sized image is a background and the same embed is suppressed from floating areas across pages"() { // given: the same image appears as a small anchor and a duplicated background on another page diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxBodyContentTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxBodyContentTest.groovy index 576264e4..d33b9f37 100644 --- a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxBodyContentTest.groovy +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxBodyContentTest.groovy @@ -1,9 +1,13 @@ package com.quadient.migration.example.docx.parser import com.quadient.migration.api.dto.migrationmodel.DisplayRule +import com.quadient.migration.api.dto.migrationmodel.ColumnLayout import com.quadient.migration.api.dto.migrationmodel.DocumentObject import com.quadient.migration.api.dto.migrationmodel.DocumentObjectRef import com.quadient.migration.api.dto.migrationmodel.Table +import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphBuilder +import com.quadient.migration.shared.ColumnApplyTo +import com.quadient.migration.shared.Size import org.apache.poi.xwpf.usermodel.XWPFDocument import org.apache.poi.xwpf.usermodel.XWPFParagraph import org.apache.poi.xwpf.usermodel.XWPFTable @@ -11,12 +15,34 @@ import org.junit.jupiter.api.Test import org.mockito.ArgumentCaptor import org.openxmlformats.schemas.wordprocessingml.x2006.main.STFldCharType +import static org.mockito.Mockito.times import static org.mockito.Mockito.verify import static com.quadient.migration.example.docx.DocxFieldFixtures.* import static com.quadient.migration.example.Utils.mockMigration class DocxBodyContentTest { + @Test + void "column layout before a generated section block is scoped to that block"() { + // given: a continuous section starts between two heading-derived blocks + def migration = mockMigration() + def body = new DocxBodyContent(migration, 'sample', 'sample_page1') + body.add(new ParagraphBuilder().string('Before columns').build(), 1) + body.add(new ColumnLayout(2, Size.ofPoints(17), null, ColumnApplyTo.WholeTemplate)) + body.add(new ParagraphBuilder().string('In columns').build()) + + // when + def content = body.sectionBlocks() + + // then: the marker is moved into the following referenced block with the narrower scope + assert content*.id == ['sample_page1_section1', 'sample_page1_section2'] + def blockCaptor = ArgumentCaptor.forClass(DocumentObject) + verify(migration.documentObjectRepository, times(2)).upsert(blockCaptor.capture()) + DocumentObject columnBlock = blockCaptor.allValues.find { it.id == 'sample_page1_section2' } + assert columnBlock.content[0] instanceof ColumnLayout + assert (columnBlock.content[0] as ColumnLayout).applyTo == ColumnApplyTo.ThisBlockOnly + } + @Test void "tables wrapped in body-level IF fields become conditional rows of the preceding table"() { // given diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxColumnLayoutTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxColumnLayoutTest.groovy new file mode 100644 index 00000000..f80fad75 --- /dev/null +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxColumnLayoutTest.groovy @@ -0,0 +1,76 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.dto.migrationmodel.Area +import com.quadient.migration.api.dto.migrationmodel.ColumnLayout +import com.quadient.migration.api.dto.migrationmodel.DocumentObject +import com.quadient.migration.api.dto.migrationmodel.builder.DocumentObjectBuilder +import com.quadient.migration.shared.DocumentObjectType +import org.apache.poi.xwpf.usermodel.XWPFDocument +import org.junit.jupiter.api.Test +import org.mockito.ArgumentCaptor +import org.openxmlformats.schemas.wordprocessingml.x2006.main.STSectionMark + +import java.nio.file.Files + +import static com.quadient.migration.example.Utils.mockMigration +import static org.mockito.Mockito.atLeastOnce +import static org.mockito.Mockito.verify + +class DocxColumnLayoutTest { + + @Test + void "terms and conditions page flow begins with its two-column layout"() { + // given: the supplied sample switches to two columns on its second page + File sample = new File(getClass().getResource('/exampleResources/docx/02_terms_and_conditions.docx').toURI()) + + // when + def migration = mockMigration() + def pages = DocxTemplateParser.parsePages(migration, sample, + new DocumentObjectBuilder('terms', DocumentObjectType.Template), 'terms') + + // then: the marker begins the generated block that contains the second page's flow content + Area secondPageFlow = pages[1].content.find { it instanceof Area } as Area + assert !(secondPageFlow.content[0] instanceof ColumnLayout) + def blockCaptor = ArgumentCaptor.forClass(DocumentObject) + verify(migration.documentObjectRepository, atLeastOnce()).upsert(blockCaptor.capture()) + DocumentObject columnBlock = blockCaptor.allValues.find { it.content && it.content[0] instanceof ColumnLayout } + ColumnLayout layout = columnBlock.content[0] as ColumnLayout + assert layout.numberOfColumns == 2 + assert layout.gutterWidth.toPoints() == 17d + assert layout.applyTo.name() == 'ThisBlockOnly' + } + + @Test + void "continuous two-column section is inserted at its position within the page flow"() { + File sample = Files.createTempFile('continuous-columns', '.docx').toFile() + try { + new XWPFDocument().withCloseable { doc -> + doc.createStyles() + doc.createParagraph().createRun().setText('Single-column content') + def boundary = doc.createParagraph() + boundary.createRun().setText('End of first section') + boundary.CTP.addNewPPr().addNewSectPr().addNewCols().space = BigInteger.valueOf(720) + doc.createParagraph().createRun().setText('Two-column content') + def finalSection = doc.document.body.addNewSectPr() + finalSection.addNewType().val = STSectionMark.CONTINUOUS + def columns = finalSection.addNewCols() + columns.num = BigInteger.valueOf(2) + columns.space = BigInteger.valueOf(720) + Files.newOutputStream(sample.toPath()).withCloseable { doc.write(it) } + } + + // when + def pages = DocxTemplateParser.parsePages(mockMigration(), sample, + new DocumentObjectBuilder('continuous', DocumentObjectType.Template), 'continuous') + + // then: the final section shares the page but introduces its layout only after the preceding content + assert pages.size() == 1 + Area flow = pages[0].content.find { it instanceof Area } as Area + int layoutIndex = flow.content.findIndexOf { it instanceof ColumnLayout } + assert layoutIndex > 0 + assert (flow.content[layoutIndex] as ColumnLayout).numberOfColumns == 2 + } finally { + Files.deleteIfExists(sample.toPath()) + } + } +} diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxContentControlsTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxContentControlsTest.groovy new file mode 100644 index 00000000..51573b0e --- /dev/null +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxContentControlsTest.groovy @@ -0,0 +1,31 @@ +package com.quadient.migration.example.docx.parser + +import com.quadient.migration.api.dto.migrationmodel.Variable +import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphBuilder +import org.apache.poi.xwpf.usermodel.XWPFDocument +import org.apache.poi.xwpf.usermodel.XWPFSDT +import org.junit.jupiter.api.Test +import org.mockito.ArgumentCaptor +import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTSdtRun + +import static com.quadient.migration.example.Utils.mockMigration +import static org.mockito.Mockito.verify + +class DocxContentControlsTest { + @Test + void "content-control tag becomes a string variable reference"() { + def source = CTSdtRun.Factory.newInstance() + source.addNewSdtPr().addNewTag().val = 'F_SINIESTRO' + def document = new XWPFDocument() + def control = new XWPFSDT(source, document) + def migration = mockMigration() + def content = [] as List + + assert DocxContentControls.addInline(migration, content, control, null, 'sample.docx') + assert content[0].content[0].id == 'F_SINIESTRO' + def captor = ArgumentCaptor.forClass(Variable) + verify(migration.variableRepository).upsert(captor.capture()) + assert captor.value.id == 'F_SINIESTRO' + document.close() + } +} diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFootersTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFootersTest.groovy index 7a9fcbc5..dc22679d 100644 --- a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFootersTest.groovy +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxHeaderFootersTest.groovy @@ -4,12 +4,53 @@ import com.quadient.migration.shared.Position import com.quadient.migration.shared.Size import org.apache.poi.xwpf.model.XWPFHeaderFooterPolicy import org.apache.poi.xwpf.usermodel.XWPFDocument +import org.apache.poi.xwpf.usermodel.XWPFRelation +import org.apache.xmlbeans.XmlObject +import org.junit.jupiter.api.AfterEach +import org.junit.jupiter.api.BeforeEach import org.junit.jupiter.api.Test import org.openxmlformats.schemas.wordprocessingml.x2006.main.STHdrFtr +import org.apache.poi.common.usermodel.PictureType +import org.apache.poi.openxml4j.opc.TargetMode + import static com.quadient.migration.example.Utils.mockMigration class DocxHeaderFootersTest { + @BeforeEach + @AfterEach + void resetImages() { DocxImages.resetImageState() } + + @Test + void 'legacy VML header image is resolved from the header relationship and emitted as a fixed area'() { + new XWPFDocument().withCloseable { document -> + document.createStyles() + def header = new XWPFHeaderFooterPolicy(document).createHeader(STHdrFtr.DEFAULT) + document.addPictureData([1, 2, 3] as byte[], PictureType.PNG) + String headerImageId = 'rIdHeaderImage' + header.packagePart.addRelationship(document.allPictures[0].packagePart.partName, TargetMode.INTERNAL, + XWPFRelation.IMAGES.relation, headerImageId) + def run = header.createParagraph().createRun() + setVml(run, """ + + """) + def bytes = new ByteArrayOutputStream() + document.write(bytes) + new XWPFDocument(new ByteArrayInputStream(bytes.toByteArray())).withCloseable { reopened -> + def page = new DocxPage(index: 0, + sections: [new DocxSection(sectPr: reopened.document.body.sectPr)], + width: Size.ofPoints(600), height: Size.ofPoints(800), + contentPosition: new Position(Size.ofPoints(40), Size.ofPoints(60), Size.ofPoints(500), Size.ofPoints(680))) + + def areas = DocxHeaderFooters.headerAreas(mockMigration(), reopened, page, 'sample') + + assert areas.size() == 1 + assert areas[0].content[0].id == 'sample_img_1' + assert [areas[0].position.width, areas[0].position.height]*.toPoints() == [122.25d, 35.25d] + } + } + } + @Test void 'default header and footer become fixed areas at the page edges'() { new XWPFDocument().withCloseable { document -> @@ -44,4 +85,13 @@ class DocxHeaderFootersTest { } } } + + private static void setVml(def run, String shapeXml) { + run.CTR.set(XmlObject.Factory.parse(""" + ${shapeXml} + """)) + } } diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxImagesTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxImagesTest.groovy index 80767fdf..c952b920 100644 --- a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxImagesTest.groovy +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxImagesTest.groovy @@ -5,6 +5,8 @@ import com.quadient.migration.shared.ImageOptions import com.quadient.migration.shared.Size import org.apache.poi.common.usermodel.PictureType import org.apache.poi.xwpf.usermodel.XWPFPictureData +import org.apache.poi.xwpf.usermodel.XWPFDocument +import org.apache.xmlbeans.XmlObject import org.junit.jupiter.api.AfterEach import org.junit.jupiter.api.BeforeEach import org.junit.jupiter.api.Test @@ -69,6 +71,125 @@ class DocxImagesTest { assert DocxImages.registerImageData(migration, picture(PictureType.PNG, 2L), 'sample', null) == 'sample_img_1' } + @Test + void "legacy VML relationship image is emitted inline with its shape dimensions"() { + // given: the legacy picture form used by MV0407GX2.docx inside table-cell paragraphs + def migration = mockMigration() + byte[] bytes = [1, 2, 3] as byte[] + new XWPFDocument().withCloseable { document -> + String embedId = document.addPictureData(bytes, PictureType.PNG) + def run = document.createParagraph().createRun() + setVml(run, """ + + """) + List builders = [] + + // when + DocxImages.processRunImages(migration, run, 'sample', builders) + + // then: position offsets are deliberately ignored and the picture becomes paragraph content + assert builders[0].build().content[0].id == 'sample_img_1' + verify(migration.storage).write('sample_img_1.png', bytes) + def captor = ArgumentCaptor.forClass(Image) + verify(migration.imageRepository).upsert(captor.capture()) + assert [captor.value.options.resizeWidth, captor.value.options.resizeHeight]*.toPoints() == [22.7d, 22.7d] + } + } + + @Test + void "paragraph collector keeps a legacy VML image before its table-cell title"() { + // given: an anchored VML icon followed by its title, as stored in the sample's cells + def migration = mockMigration() + new XWPFDocument().withCloseable { document -> + String embedId = document.addPictureData([1, 2, 3] as byte[], PictureType.PNG) + def paragraph = document.createParagraph() + def imageRun = paragraph.createRun() + setVml(imageRun, """ + + """) + def titleRun = paragraph.createRun() + titleRun.setText('DATOS COTIZACIÓN') + def collector = new ParagraphContentCollector(migration, 'sample') + + // when + collector.addRun(imageRun, 'title') + collector.addRun(titleRun, 'title') + def content = collector.finish()*.build() + + // then + assert content[0].content[0].id == 'sample_img_1' + assert content[1].content[0].value == 'DATOS COTIZACIÓN' + } + } + + @Test + void "absolute VML image paragraph merges with the following cell title and drops layout spaces"() { + // given: the separate image and title paragraphs used by MV0407GX2.docx + def migration = mockMigration() + new XWPFDocument().withCloseable { document -> + document.createStyles() + String embedId = document.addPictureData([1, 2, 3] as byte[], PictureType.PNG) + def table = document.createTable(1, 1) + def cell = table.getRow(0).getCell(0) + def imageParagraph = cell.paragraphs[0] + def imageRun = imageParagraph.createRun() + setVml(imageRun, """ + + """) + imageParagraph.createRun().setText(' ') + cell.addParagraph().createRun().setText(' DATOS TOMADOR') + + // when + def parsed = DocxTableParser.parseTable(migration, table, 'sample') + def content = parsed.rows[0].cells[0].content + + // then: one paragraph contains the icon, one natural space, and the title + assert content.size() == 1 + assert content[0].content[0].content[0].id == 'sample_img_1' + assert content[0].content[1].content[0].value == ' DATOS TOMADOR' + } + } + + @Test + void "legacy VML vector shape is converted to an inline SVG image"() { + // given: the legacy person-icon form used by MV0407GX2.docx + def migration = mockMigration() + new XWPFDocument().withCloseable { document -> + def run = document.createParagraph().createRun() + setVml(run, '') + List builders = [] + + // when + DocxImages.processRunImages(migration, run, 'sample', builders) + + // then + assert builders[0].build().content[0].id == 'sample_img_1' + def write = mockingDetails(migration.storage).invocations.find { + it.method.name == 'write' && it.arguments[0] == 'sample_img_1.svg' + } + assert write != null + String svg = new String(write.arguments[1] as byte[], 'UTF-8') + assert svg.contains('viewBox="0 0 243 265"') + assert svg.contains('d="M 0 0 L 243 0 L 243 265 L 0 265 Z"') + assert svg.contains('fill="#016557" stroke="none"') + } + } + + @Test + void "VML relative cubic paths preserve omitted zero coordinates"() { + assert DocxVmlImages.vmlPathToSvgPath('m243,220v,14,-4,25,-12,33xe') == + 'M 243 220 c 0 14 -4 25 -12 33 Z' + } + + private static void setVml(def run, String shapeXml) { + run.CTR.set(XmlObject.Factory.parse(""" + ${shapeXml} + """)) + } + private static XWPFPictureData picture(PictureType type, Long checksum) { def data = mock(XWPFPictureData) when(data.pictureTypeEnum).thenReturn(type) diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxPageLayoutTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxPageLayoutTest.groovy index 55302071..02e3b848 100644 --- a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxPageLayoutTest.groovy +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxPageLayoutTest.groovy @@ -1,6 +1,7 @@ package com.quadient.migration.example.docx.parser import com.quadient.migration.shared.Size +import com.quadient.migration.shared.ColumnApplyTo import org.junit.jupiter.api.Test import org.junit.jupiter.params.ParameterizedTest import org.junit.jupiter.params.provider.CsvSource @@ -20,6 +21,32 @@ class DocxPageLayoutTest { assert [position.x, position.y, position.width, position.height]*.toPoints() == [18d, 36d, 522d, 702d] } + @Test + void "multi-column section is converted to a page-flow column layout"() { + // given: Word's section properties declare two columns with a 340-twip gutter + def section = CTSectPr.Factory.newInstance() + def columns = section.addNewCols() + columns.num = BigInteger.valueOf(2) + columns.space = BigInteger.valueOf(340) + + // when + def layout = DocxPageLayout.resolveColumnLayout(section) + + // then: the page parser can prepend it before the first body element + assert layout.numberOfColumns == 2 + assert layout.gutterWidth.toPoints() == 17d + assert layout.applyTo == ColumnApplyTo.WholeTemplate + } + + @Test + void "single-column and absent settings do not emit a column layout"() { + // given / when / then: Word's implicit and explicit single-column layouts need no flow marker + assert DocxPageLayout.resolveColumnLayout(null) == null + def section = CTSectPr.Factory.newInstance() + section.addNewCols().num = BigInteger.ONE + assert DocxPageLayout.resolveColumnLayout(section) == null + } + @Test void "missing section geometry uses defaults"() { // given: absent or empty section properties diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxPageSectionsTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxPageSectionsTest.groovy index 4b5ecfac..c0472ad3 100644 --- a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxPageSectionsTest.groovy +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxPageSectionsTest.groovy @@ -62,31 +62,6 @@ class DocxPageSectionsTest { } } - @Test - void "a rendered page break at the start of a paragraph creates a new page"() { - // given: Word has cached a page boundary before the second paragraph's visible text - new XWPFDocument().withCloseable { doc -> - def first = doc.createParagraph() - first.createRun().setText('First-page content') - def second = doc.createParagraph() - second.createRun().CTR.addNewLastRenderedPageBreak() - second.createRun().setText('Second-page content') - def third = doc.createParagraph() - third.createRun().setText('More second-page content') - doc.document.body.addNewSectPr() - ByteArrayOutputStream serialized = new ByteArrayOutputStream() - doc.write(serialized) - - new XWPFDocument(new ByteArrayInputStream(serialized.toByteArray())).withCloseable { reloaded -> - // when - def pages = DocxPageSections.groupIntoPages(DocxPageSections.splitIntoSections(reloaded)) - - // then: the paragraph carrying the marker and subsequent content belong to the next page - assert pages*.bodyElements()*.collect { it.text } == [['First-page content'], ['Second-page content', 'More second-page content']] - } - } - } - @Test void "a rendered page break after paragraph content is not used as a boundary"() { // given: a marker after content cannot be represented without splitting a paragraph diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParserTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParserTest.groovy index a9667ec7..84d3e966 100644 --- a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParserTest.groovy +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/DocxParagraphParserTest.groovy @@ -1,6 +1,7 @@ package com.quadient.migration.example.docx.parser import com.quadient.migration.api.dto.migrationmodel.DisplayRule +import com.quadient.migration.api.dto.migrationmodel.Hyperlink import com.quadient.migration.api.dto.migrationmodel.VariableRef import com.quadient.migration.example.docx.util.DocxUtils import org.apache.poi.xwpf.usermodel.XWPFDocument @@ -16,6 +17,28 @@ import static com.quadient.migration.example.Utils.mockMigration class DocxParagraphParserTest { + @Test + void "external hyperlinks become model hyperlinks while surrounding text remains plain"() { + // given: a normal Word external hyperlink between unlinked text runs + new XWPFDocument().withCloseable { doc -> + doc.createStyles() + def paragraph = doc.createParagraph() + paragraph.createRun().setText('Read ') + paragraph.createHyperlinkRun('https://example.com/terms').setText('the terms') + paragraph.createRun().setText(' before proceeding.') + + // when + def parsed = DocxParagraphParser.parseParagraph(mockMigration(), paragraph, 'sample') + + // then: only the linked text is represented by Hyperlink content, in source order + assert parsed.content*.content.flatten().collect { it instanceof Hyperlink ? it : it.value } == [ + 'Read ', + new Hyperlink('https://example.com/terms', 'the terms', null), + ' before proceeding.' + ] + } + } + @Test void "adjacent equal formatting coalesces while style changes preserve text order"() { // given: adjacent plain runs, a bold run and a final plain run diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/ParagraphContentCollectorTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/ParagraphContentCollectorTest.groovy index aeadf3eb..9b0ef0c8 100644 --- a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/ParagraphContentCollectorTest.groovy +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/parser/ParagraphContentCollectorTest.groovy @@ -108,6 +108,37 @@ class ParagraphContentCollectorTest { document.close() } + @Test + void "collector emits recognized Word system fields as migration variables rather than their cached values"() { + // given: page fields plus DATE and TIME from the GF documents + def migration = mockMigration() + def document = new XWPFDocument() + def paragraph = document.createParagraph() + paragraph.createRun().setText('Página ') + appendPageField(paragraph, ' PAGE ', '1') + paragraph.createRun().setText(' de ') + appendPageField(paragraph, ' SECTIONPAGES ', '1') + paragraph.createRun().setText(' af ') + appendPageField(paragraph, ' NUMPAGES ', '2') + paragraph.createRun().setText(' den ') + appendPageField(paragraph, ' DATE \\@ "d. MMMM yyyy" ', '1. oktober 2026') + paragraph.createRun().setText(' kl. ') + appendPageField(paragraph, ' TIME \\@ "HH:mm" ', '13:45') + + // when + def collector = new ParagraphContentCollector(migration, 'sample') + paragraph.runs.each { collector.addRun(it, 'footer') } + def content = collector.finish()*.build() + + // then: the cached values are ignored and migration authors can bind the variables as needed + assert content.collect { it.content[0].hasProperty('id') ? it.content[0].id : it.content[0].value } == + ['Página ', 'systemPageNumber', ' de ', 'systemSectionPages', ' af ', 'systemTotalPages', + ' den ', 'systemCurrentDate', ' kl. ', 'systemCurrentDateTime'] + verify(migration.variableRepository, times(5)).upsert(any()) + + document.close() + } + @Test void "adjacent exclusive equality fields become one styled FirstMatch in source order"() { // given diff --git a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/style/StyleChainResolverTest.groovy b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/style/StyleChainResolverTest.groovy index 5a801a06..433a99ef 100644 --- a/migration-examples/src/test/groovy/com/quadient/migration/example/docx/style/StyleChainResolverTest.groovy +++ b/migration-examples/src/test/groovy/com/quadient/migration/example/docx/style/StyleChainResolverTest.groovy @@ -40,6 +40,27 @@ class StyleChainResolverTest { } } + @Test + void "East Asian-only run font does not replace an inherited Latin font"() { + // given: the BodyText/Times New Roman split stored by 01CVRPG0222.docx + new XWPFDocument().withCloseable { doc -> + doc.createStyles() + def bodyText = style(doc, 'BodyText', null) + bodyText.addNewRPr().addNewRFonts().ascii = 'Arial' + def paragraph = doc.createParagraph() + paragraph.style = 'BodyText' + def run = paragraph.createRun() + run.CTR.addNewRPr().addNewRFonts().eastAsia = 'Times New Roman' + + // when: resolving an English run with an East Asian-only direct override + def chain = StyleChainResolver.resolveRunPropertyChain(run, 'BodyText') + + // then: Latin text keeps Arial, while East Asian is still available as a final fallback + assert StyleChainResolver.resolveEffectiveFontName(chain) == 'Arial' + assert StyleChainResolver.resolveEffectiveFontName([run.CTR.RPr]) == 'Times New Roman' + } + } + @Test void "paragraph chain terminates cycles and preserves zero overrides"() { // given: cyclic parent styles and a direct zero-spacing override