Skip to content
Original file line number Diff line number Diff line change
Expand Up @@ -6,10 +6,17 @@
//! ---
package com.quadient.migration.example.docx

import groovy.transform.Field
import org.slf4j.Logger
import org.slf4j.LoggerFactory

import static com.quadient.migration.example.common.util.InitMigration.initMigration
import static com.quadient.migration.example.docx.parser.DocxTemplateParser.parseDocxFiles

def migration = initMigration(this.binding)
@Field static Logger log = LoggerFactory.getLogger(this.class.name)

println("\nStarting Parse step...\n")
log.info "\nStarting Parse step...\n"
parseDocxFiles(migration)
log.info "\nApplying persisted mapping...\n"
migration.mappingRepository.applyAll()

This file was deleted.

Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@ import org.openxmlformats.schemas.drawingml.x2006.picture.CTPicture
import org.openxmlformats.schemas.drawingml.x2006.wordprocessingDrawing.CTAnchor
import org.openxmlformats.schemas.wordprocessingml.x2006.main.CTP
import org.w3c.dom.Node
import org.w3c.dom.Element
import org.xml.sax.InputSource

import javax.xml.parsers.DocumentBuilder
Expand All @@ -40,6 +41,7 @@ class DocxAnchoredAreas {
private static final String W_NS = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
private static final String WPS_SHAPE_URI = "http://schemas.microsoft.com/office/word/2010/wordprocessingShape"
private static final String PIC_NS = "http://schemas.openxmlformats.org/drawingml/2006/picture"
private static final String VML_NS = "urn:schemas-microsoft-com:vml"
private static final double BACKGROUND_COVERAGE_THRESHOLD = 0.6

final Map<Integer, List<Area>> backgroundAreasByPage = [:]
Expand Down Expand Up @@ -67,7 +69,10 @@ class DocxAnchoredAreas {
DocxAnchoredAreas result = new DocxAnchoredAreas(migration, doc, fileName)
// Backgrounds are resolved over the whole document first so the same picture is never emitted as floating too.
pages.each { DocxPage page -> result.eachUniqueAnchor(page) { result.addBackgroundArea(it, page) } }
pages.each { DocxPage page -> result.eachUniqueAnchor(page) { result.addFloatingArea(it, page) } }
pages.each { DocxPage page ->
result.eachUniqueAnchor(page) { result.addFloatingArea(it, page) }
result.eachUniqueVmlTextBox(page) { result.addVmlTextBoxArea(it, page) }
}
return result
}

Expand All @@ -84,6 +89,21 @@ class DocxAnchoredAreas {
}
}

// Older Word documents use VML w:pict/v:shape text boxes rather than DrawingML wp:anchor shapes.
private void eachUniqueVmlTextBox(DocxPage page, Closure<Void> handler) {
Set<Object> seenShapeIds = []
page.paragraphs().each { XWPFParagraph paragraph ->
paragraph.runs.each { XWPFRun run ->
findVmlTextBoxes(run).each { Element shape ->
String shapeId = vmlShapeId(shape)
if (seenShapeIds.add(shapeId ?: shape.textContent)) {
handler(shape)
}
}
}
}
}

private void addBackgroundArea(CTAnchor anchor, DocxPage page) {
if (!isPicture(anchor) || !coversPage(anchor, page)) {
return
Expand Down Expand Up @@ -116,6 +136,13 @@ class DocxAnchoredAreas {
}
}

private void addVmlTextBoxArea(Element shape, DocxPage page) {
Area area = buildVmlTextBoxArea(shape, page)
if (area != null) {
floatingAreasByPage.computeIfAbsent(page.index) { [] }.add(area)
}
}

private static boolean isPicture(CTAnchor anchor) {
return anchor.graphic?.graphicData?.uri == PIC_NS
}
Expand All @@ -132,6 +159,18 @@ class DocxAnchoredAreas {
return found.collect { XmlObject o -> o instanceof CTAnchor ? o : CTAnchor.Factory.parse(o.xmlText()) }
}

private List<Element> findVmlTextBoxes(XWPFRun run) {
// XMLBeans' XPath support can require optional Saxon classes. DOM traversal keeps this parser self-contained.
Element runDom = documentBuilder.parse(new InputSource(new StringReader(run.CTR.xmlText()))).documentElement
def shapes = runDom.getElementsByTagNameNS(VML_NS, 'shape')
return (0..<shapes.length).collect { shapes.item(it) as Element }
.findAll { it.getElementsByTagNameNS(W_NS, 'txbxContent').length > 0 }
}

private static String vmlShapeId(Element shape) {
return shape.getAttribute('id') ?: null
}

private static String extractBlipEmbedId(CTAnchor anchor) {
XmlObject[] pics = anchor.selectPath("declare namespace pic='${PIC_NS}' .//pic:pic")
if (pics.length == 0) {
Expand Down Expand Up @@ -161,32 +200,75 @@ class DocxAnchoredAreas {
if (!cursor.toFirstChild()) {
return null
}
def shapeDom = documentBuilder.parse(new InputSource(new StringReader(cursor.xmlText())))
def txbxContentNodes = shapeDom.getElementsByTagNameNS(W_NS, "txbxContent")
if (txbxContentNodes.length == 0) {
return null
}
List<DocumentContent> contentItems = []
def children = txbxContentNodes.item(0).childNodes
for (int i = 0; i < children.length; i++) {
Node child = children.item(i)
if (child.nodeType == Node.ELEMENT_NODE && child.localName == 'p') {
XWPFParagraph paragraph = new XWPFParagraph(domParagraphToCtp(child), doc)
contentItems.add(parseParagraph(migration, paragraph, fileName))
}
}
if (contentItems.isEmpty()) {
return null
}
String blockId = "${page.id(fileName)}_textbox${++textBoxes}"
String firstText = contentItems.findResult { extractParagraphText(it)?.trim() ?: null }
DocumentObjectRef blockRef = upsertBlock(migration, blockId, blockName(firstText, "text box", blockId), contentItems, fileName)
return new AreaBuilder().content([blockRef]).position(resolveAnchorPosition(anchor, page)).build()
return buildTextBoxArea(cursor.xmlText(), resolveAnchorPosition(anchor, page), page)
} finally {
cursor.dispose()
}
}

private Area buildVmlTextBoxArea(Element shape, DocxPage page) {
return buildTextBoxArea(shape, resolveVmlShapePosition(shape, page), page)
}

private Area buildTextBoxArea(String shapeXml, Position position, DocxPage page) {
def shapeDom = documentBuilder.parse(new InputSource(new StringReader(shapeXml)))
return buildTextBoxArea(shapeDom.documentElement, position, page)
}

private Area buildTextBoxArea(Element shapeDom, Position position, DocxPage page) {
def txbxContentNodes = shapeDom.getElementsByTagNameNS(W_NS, "txbxContent")
if (txbxContentNodes.length == 0) {
return null
}
List<DocumentContent> contentItems = []
def children = txbxContentNodes.item(0).childNodes
for (int i = 0; i < children.length; i++) {
Node child = children.item(i)
if (child.nodeType == Node.ELEMENT_NODE && child.localName == 'p') {
XWPFParagraph paragraph = new XWPFParagraph(domParagraphToCtp(child), doc)
contentItems.add(parseParagraph(migration, paragraph, fileName))
}
}
if (contentItems.isEmpty()) {
return null
}
String blockId = "${page.id(fileName)}_textbox${++textBoxes}"
String firstText = contentItems.findResult { extractParagraphText(it)?.trim() ?: null }
DocumentObjectRef blockRef = upsertBlock(migration, blockId, blockName(firstText, "text box", blockId), contentItems, fileName)
return new AreaBuilder().content([blockRef]).position(position).build()
}

private Position resolveVmlShapePosition(Element shape, DocxPage page) {
Map<String, String> style = vmlStyle(shape.getAttribute('style'))
double x = page.contentPosition.x.toPoints() + vmlPoints(style['margin-left'])
double y = page.contentPosition.y.toPoints() + vmlPoints(style['margin-top'])
return new Position(Size.ofPoints(x), Size.ofPoints(y), Size.ofPoints(vmlPoints(style['width'])), Size.ofPoints(vmlPoints(style['height'])))
}

private static Map<String, String> vmlStyle(String value) {
return value.split(';').collectEntries { String property ->
int separator = property.indexOf(':')
separator < 0 ? [:] : [(property.substring(0, separator).trim().toLowerCase(Locale.ROOT)): property.substring(separator + 1).trim()]
}
}

// Word's VML geometry is CSS-like; point values dominate its generated documents, with the common alternatives
// handled here as well so that the resulting Area geometry stays in points.
private static double vmlPoints(String value) {
def matcher = value =~ /^([+-]?(?:\d+(?:\.\d*)?|\.\d+))(pt|in|cm|mm|px)?$/
if (!matcher.matches()) {
return 0d
}
double number = matcher.group(1) as double
return switch (matcher.group(2)?.toLowerCase(Locale.ROOT)) {
case 'in' -> number * 72d
case 'cm' -> number * 72d / 2.54d
case 'mm' -> number * 72d / 25.4d
case 'px' -> number * 72d / 96d
default -> number
}
}

private CTP domParagraphToCtp(Node pNode) {
StringWriter sw = new StringWriter()
Transformer transformer = transformerFactory.newTransformer()
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,9 @@ package com.quadient.migration.example.docx.parser

import com.quadient.migration.api.Migration
import com.quadient.migration.api.dto.migrationmodel.DocumentContent
import com.quadient.migration.api.dto.migrationmodel.ColumnLayout
import com.quadient.migration.api.dto.migrationmodel.builder.TableBuilder
import com.quadient.migration.shared.ColumnApplyTo
import org.apache.poi.xwpf.usermodel.XWPFTable
import org.apache.poi.xwpf.usermodel.XWPFTableCell
import org.apache.xmlbeans.XmlObject
Expand Down Expand Up @@ -94,18 +96,40 @@ class DocxBodyContent {
if (topLevel == null) {
return content
}
List<List<DocumentContent>> sections = [[]]
List<DocumentContent> result = []
List<DocumentContent> section = []
int blockNumber = 0
ColumnLayout pendingColumnLayout
Closure flushSection = {
if (section.isEmpty()) {
return
}
String id = "${pageId}_section${++blockNumber}"
String heading = headingLevels[section[0]] == topLevel ? extractParagraphText(section[0]) : null
List<DocumentContent> blockContent = pendingColumnLayout == null ? section : [new ColumnLayout(
pendingColumnLayout.numberOfColumns, pendingColumnLayout.gutterWidth, pendingColumnLayout.balancingType,
ColumnApplyTo.ThisBlockOnly)] + section
pendingColumnLayout = null
result.add(upsertBlock(migration, id, blockName(heading, null, id), blockContent, fileName))
section.clear()
}
content.each { DocumentContent item ->
if (headingLevels[item] == topLevel && !sections.last().isEmpty()) {
sections << []
// A marker before a generated block belongs to that block, so its emitted scope is ThisBlockOnly.
if (item instanceof ColumnLayout) {
flushSection()
pendingColumnLayout = item
} else {
if (headingLevels[item] == topLevel && !section.isEmpty()) {
flushSection()
}
section << item
}
sections.last() << item
}
return sections.withIndex().collect { List<DocumentContent> items, int i ->
String id = "${pageId}_section${i + 1}"
String heading = headingLevels[items[0]] == topLevel ? extractParagraphText(items[0]) : null
upsertBlock(migration, id, blockName(heading, null, id), items, fileName)
flushSection()
if (pendingColumnLayout != null) {
result.add(pendingColumnLayout)
}
return result
}

// Text of the first non-empty cell of the table's first row (typically the heading of the row). Word stores the
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
package com.quadient.migration.example.docx.parser

import com.quadient.migration.api.Migration
import com.quadient.migration.api.dto.migrationmodel.Paragraph
import com.quadient.migration.api.dto.migrationmodel.builder.ParagraphBuilder
import org.apache.poi.xwpf.usermodel.XWPFSDT

class DocxContentControls {
static String variableId(XWPFSDT control) {
return control.tag?.trim() ?: control.title?.trim()
}

static boolean addInline(Migration migration, List<ParagraphBuilder.TextBuilder> textBuilders, XWPFSDT control,
String styleId, String fileName) {
String id = variableId(control)
if (!id) return false
DocxVariablePatterns.addVariable(migration, textBuilders, id, styleId, fileName)
return true
}

static Paragraph parseBlock(Migration migration, XWPFSDT control, String fileName) {
String id = variableId(control)
if (!id) return null
List<ParagraphBuilder.TextBuilder> content = []
DocxVariablePatterns.addVariable(migration, content, id, null, fileName)
return new ParagraphBuilder().content(content).build()
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -54,11 +54,12 @@ class DocxHeaderFooters {
XWPFHeaderFooter part = selectPart(headerOnly ? parts.headers : parts.footers, page)
if (part != null) {
Position position = headerOnly ? headerPosition(page) : footerPosition(page)
Set<String> directImageEmbedIds = anchoredEmbedIds(part) + inlineEmbedIds(part)
Set<String> directImageEmbedIds = anchoredEmbedIds(part) + inlineEmbedIds(part) + vmlEmbedIds(part)
Area area = buildArea(migration, part, fileName, position, directImageEmbedIds)
if (area != null) result.add(area)
result.addAll(inlineImageAreas(migration, doc, part, fileName, position))
result.addAll(anchoredImageAreas(migration, doc, part, fileName, page, position))
result.addAll(vmlImageAreas(migration, doc, part, fileName, position))
}
return result
}
Expand Down Expand Up @@ -140,12 +141,20 @@ class DocxHeaderFooters {
List<Area> areas = []
source.paragraphs.each { XWPFParagraph paragraph ->
paragraph.runs.each { XWPFRun run ->
Set<String> anchorEmbedIds = findAnchors(run).collect { CTAnchor anchor -> extractBlipEmbedId(anchor) }
.findAll().toSet()
boolean hasInlinePicture = false
run.embeddedPictures.each { picture ->
CTPicture ctPicture = picture.CTPicture
XWPFPictureData data = picture.pictureData ?: pictureData(doc, source, ctPicture?.blipFill?.blip?.embed)
String embedId = ctPicture?.blipFill?.blip?.embed
// POI can surface a wp:anchor through embeddedPictures. It is emitted below by
// anchoredImageAreas with its anchor offsets, so do not also treat it as an inline image.
if (embedId && anchorEmbedIds.contains(embedId)) return
XWPFPictureData data = picture.pictureData ?: pictureData(doc, source, embedId)
addInlineImageArea(areas, migration, data, ctPicture, fileName, flowPosition)
hasInlinePicture = true
}
if (!run.embeddedPictures.isEmpty()) return
if (hasInlinePicture) return
findInlines(run).each { XmlObject inline ->
CTPicture picture = inlinePicture(inline)
addInlineImageArea(areas, migration, pictureData(doc, source, picture?.blipFill?.blip?.embed), picture, fileName, flowPosition)
Expand All @@ -166,6 +175,29 @@ class DocxHeaderFooters {
return areas
}

private static List<Area> vmlImageAreas(Migration migration, XWPFDocument doc, XWPFHeaderFooter source, String fileName,
Position flowPosition) {
List<Area> areas = []
source.paragraphs.each { XWPFParagraph paragraph ->
paragraph.runs.each { XWPFRun run ->
DocxVmlImages.extract(run).each { VmlImageSource vmlImage ->
if (!vmlImage.embedId) return
XWPFPictureData data = pictureData(doc, source, vmlImage.embedId)
if (data == null) return
String imageId = registerImageData(migration, data, fileName, vmlImage.options)
if (imageId != null) {
Size width = vmlImage.options?.resizeWidth
Size height = vmlImage.options?.resizeHeight
Position position = width != null && height != null
? new Position(flowPosition.x, flowPosition.y, width, height) : flowPosition
areas.add(new AreaBuilder().imageRef(imageId).position(position).build())
}
}
}
}
return areas
}

private static List<CTAnchor> findAnchors(XWPFRun run) {
XmlObject[] found = run.CTR.selectPath("declare namespace wp='${WP_NS}' .//wp:anchor")
return found.collect { XmlObject o -> o instanceof CTAnchor ? o : CTAnchor.Factory.parse(o.xmlText()) }
Expand Down Expand Up @@ -201,6 +233,18 @@ class DocxHeaderFooters {
return ids
}

private static Set<String> vmlEmbedIds(XWPFHeaderFooter source) {
Set<String> ids = []
source.paragraphs.each { XWPFParagraph paragraph ->
paragraph.runs.each { XWPFRun run ->
DocxVmlImages.extract(run).each { VmlImageSource image ->
if (image.embedId) ids.add(image.embedId)
}
}
}
return ids
}

private static void addInlineImageArea(List<Area> areas, Migration migration, XWPFPictureData data, CTPicture picture,
String fileName, Position flowPosition) {
if (data == null || picture == null) return
Expand Down
Loading
Loading