Add logic to apply formatting to text

This commit is contained in:
Veronica K. B. Olsen
2021-01-27 19:36:34 +01:00
parent 4df294f0a2
commit 8715ab9322
4 changed files with 246 additions and 42 deletions
+3 -3
View File
@@ -236,12 +236,12 @@ class ToHtml(Tokenizer):
if parStyle is None: if parStyle is None:
parStyle = hStyle parStyle = hStyle
for xPos, xLen, xFmt in reversed(tFormat): for xPos, xLen, xFmt in reversed(tFormat):
tTemp = tTemp[:xPos]+htmlTags[xFmt]+tTemp[xPos+xLen:] tTemp = tTemp[:xPos] + htmlTags[xFmt] + tTemp[xPos+xLen:]
if tText.endswith(" "): if tText.endswith(" "):
thisPar.append(tTemp.rstrip()+"<br/>") thisPar.append(tTemp.rstrip() + "<br/>")
hasHardBreak = True hasHardBreak = True
else: else:
thisPar.append(tTemp.rstrip()+" ") thisPar.append(tTemp.rstrip() + " ")
elif tType == self.T_SYNOPSIS and self.doSynopsis: elif tType == self.T_SYNOPSIS and self.doSynopsis:
tmpResult.append(self._formatSynopsis(tText)) tmpResult.append(self._formatSynopsis(tText))
+199 -36
View File
@@ -26,7 +26,6 @@ along with this program. If not, see <https://www.gnu.org/licenses/>.
import nw import nw
import logging import logging
import os
from lxml import etree from lxml import etree
from hashlib import sha256 from hashlib import sha256
@@ -45,11 +44,14 @@ XML_NS = {
"fo" : "urn:oasis:names:tc:opendocument:xmlns:xsl-fo-compatible:1.0", "fo" : "urn:oasis:names:tc:opendocument:xmlns:xsl-fo-compatible:1.0",
} }
X_BR = "{%s}line-break" % XML_NS["text"]
X_TAB = "{%s}tab" % XML_NS["text"]
class ToOdt(Tokenizer): class ToOdt(Tokenizer):
X_BLD = 0x01
X_ITA = 0x02
X_DEL = 0x04
X_BRK = 0x08
X_TAB = 0x10
def __init__(self, theProject, theParent): def __init__(self, theProject, theParent):
Tokenizer.__init__(self, theProject, theParent) Tokenizer.__init__(self, theProject, theParent)
@@ -208,7 +210,17 @@ class ToOdt(Tokenizer):
""" """
self.theResult = "" self.theResult = ""
odtTags = {
self.FMT_B_B : "_B",
self.FMT_B_E : "b_",
self.FMT_I_B : "I",
self.FMT_I_E : "i",
self.FMT_D_B : "_S",
self.FMT_D_E : "s_",
}
thisPar = [] thisPar = []
thisFmt = []
parStyle = None parStyle = None
hasHardBreak = False hasHardBreak = False
for tType, tLine, tText, tFormat, tStyle in self.theTokens: for tType, tLine, tText, tFormat, tStyle in self.theTokens:
@@ -238,10 +250,16 @@ class ToOdt(Tokenizer):
if hasHardBreak and parStyle is not None: if hasHardBreak and parStyle is not None:
if self.doJustify: if self.doJustify:
parStyle.setTextAlign("left") parStyle.setTextAlign("left")
if len(thisPar) > 0: if len(thisPar) > 0:
tTemp = "".join(thisPar) tTemp = "".join(thisPar)
self._addTextPar("Text_Body", parStyle, tTemp.rstrip()) fTemp = "".join(thisFmt)
tTxt = tTemp.rstrip()
tFmt = fTemp[:len(tTxt)]
self._addTextPar("Text_Body", parStyle, tTxt, theFmt=tFmt)
thisPar = [] thisPar = []
thisFmt = []
parStyle = None parStyle = None
hasHardBreak = False hasHardBreak = False
@@ -275,21 +293,31 @@ class ToOdt(Tokenizer):
tTemp = tText tTemp = tText
if parStyle is None: if parStyle is None:
parStyle = oStyle parStyle = oStyle
# for xPos, xLen, xFmt in reversed(tFormat):
# tTemp = tTemp[:xPos] + htmlTags[xFmt] + tTemp[xPos+xLen:] tFmt = " "*len(tTemp)
for xPos, xLen, xFmt in tFormat:
tFmt = tFmt[:xPos] + odtTags[xFmt] + tFmt[xPos+xLen:]
tTxt = tTemp.rstrip()
tFmt = tFmt[:len(tTxt)]
if tText.endswith(" "): if tText.endswith(" "):
thisPar.append(tTemp.rstrip()+"\n") thisPar.append(tTxt + "\n")
thisFmt.append(tFmt + " ")
hasHardBreak = True hasHardBreak = True
else: else:
thisPar.append(tTemp.rstrip()+" ") thisPar.append(tTxt + " ")
thisFmt.append(tFmt + " ")
return return
def closeDocument(self): def closeDocument(self):
"""Return the serialised XML document """Return the serialised XML document
""" """
# Build the auto-generated styles
for styleName, styleObj in self._autoPara.values(): for styleName, styleObj in self._autoPara.values():
styleObj.packXML(self._xAuto, styleName) styleObj.packXML(self._xAuto, styleName)
for styleName, styleObj in self._autoText.values():
styleObj.packXML(self._xAuto, styleName)
self.theResult = etree.tostring( self.theResult = etree.tostring(
self._xRoot, self._xRoot,
@@ -298,17 +326,13 @@ class ToOdt(Tokenizer):
xml_declaration = True xml_declaration = True
) )
cacheFile = os.path.join(os.path.expanduser("~"), "Temp", "odtGen.fodt")
with open(cacheFile, mode="wb") as outFile:
outFile.write(self.theResult)
return return
## ##
# Internal Functions # Internal Functions
## ##
def _addTextPar(self, styleName, oStyle, theText, isHead=False, oLevel=None): def _addTextPar(self, styleName, oStyle, theText, theFmt="", isHead=False, oLevel=None):
"""Add a text paragraph to the text XML element. """Add a text paragraph to the text XML element.
""" """
tAttr = {} tAttr = {}
@@ -322,36 +346,85 @@ class ToOdt(Tokenizer):
if not theText: if not theText:
return return
if "\t" not in theText and "\n" not in theText: ##
xElem.text = theText # Process Formatting
return ##
# Process tabs and line breaks if len(theText) != len(theFmt):
tTemp = "" # Genrate dummu format if there isn't any
theFmt = " "*len(theText)
# XML functions
xTail = None xTail = None
for c in theText:
if c == "\t": def appendText(tText):
nonlocal xElem, xTail
if tText:
if xTail is None: if xTail is None:
xElem.text = tTemp xElem.text = tText
else: else:
xTail.tail = tTemp xTail.tail = tText
def appendSpan(tText, tFmt):
nonlocal xElem, xTail
if tText:
xTail = etree.SubElement(xElem, _mkTag("text", "span"), attrib={
_mkTag("text", "style-name"): self._textStyle(tFmt)
})
xTail.text = tText
# The formatting loop
tTemp = "" tTemp = ""
xTail = etree.SubElement(xElem, X_TAB) xFmt = 0x00
elif c == "\n": pFmt = 0x00
if xTail is None:
xElem.text = tTemp for i, c in enumerate(theText):
else:
xTail.tail = tTemp if theFmt[i] == "_":
tTemp = "" continue
xTail = etree.SubElement(xElem, X_BR) elif theFmt[i] == "B":
else: xFmt |= self.X_BLD
elif theFmt[i] == "b":
xFmt ^= self.X_BLD
elif theFmt[i] == "I":
xFmt |= self.X_ITA
elif theFmt[i] == "i":
xFmt ^= self.X_ITA
elif theFmt[i] == "S":
xFmt |= self.X_DEL
elif theFmt[i] == "s":
xFmt ^= self.X_DEL
if c == "\n":
xFmt |= self.X_BRK
c = ""
elif c == "\t":
xFmt |= self.X_TAB
c = ""
if theFmt[i] == " ":
tTemp += c tTemp += c
if tTemp != "": if xFmt != pFmt:
if xTail is None: if pFmt == 0x00:
xElem.text = tTemp appendText(tTemp)
tTemp = ""
else: else:
xTail.tail = tTemp appendSpan(tTemp, pFmt)
tTemp = ""
if xFmt & self.X_BRK:
xTail = etree.SubElement(xElem, _mkTag("text", "line-break"))
xFmt ^= self.X_BRK
if xFmt & self.X_TAB:
xTail = etree.SubElement(xElem, _mkTag("text", "tab"))
xFmt ^= self.X_TAB
pFmt = xFmt
# Save what remains in the buffer
appendText(tTemp)
return return
@@ -376,6 +449,26 @@ class ToOdt(Tokenizer):
return newName return newName
def _textStyle(self, styleCode):
"""Return a text style for a given style code.
"""
if styleCode in self._autoText:
return self._autoText[styleCode][0]
newName = "T%d" % (len(self._autoText) + 1)
newStyle = ODTTextStyle()
if styleCode & self.X_BLD:
newStyle.setFontWeight("bold")
if styleCode & self.X_ITA:
newStyle.setFontStyle("italic")
if styleCode & self.X_DEL:
newStyle.setStrikeStyle("solid")
newStyle.setStrikeType("single")
self._autoText[styleCode] = (newName, newStyle)
return newName
def _emToCm(self, emVal): def _emToCm(self, emVal):
"""Converts an em value to centimetres. """Converts an em value to centimetres.
""" """
@@ -803,6 +896,76 @@ class ODTParagraphStyle():
# END Class ODTParagraphStyle # END Class ODTParagraphStyle
class ODTTextStyle():
"""Wrapper class for the text style setting used by the exporter.
Only the used settings are exposed here to keep the class minimal
and fast.
"""
VALID_WEIGHT = ["normal", "inherit", "bold"]
VALID_STYLE = ["normal", "inherit", "italic"]
VALID_LSTYLE = ["none", "solid"]
VALID_LTYPE = ["none", "single", "double"]
def __init__(self):
# Text Attributes
self._tAttr = {
"font-weight": ["fo", None],
"font-style": ["fo", None],
"text-line-through-style": ["style", None],
"text-line-through-type": ["style", None],
}
return
##
# Setters
##
def setFontWeight(self, theValue):
if theValue in self.VALID_WEIGHT:
self._tAttr["font-weight"][1] = str(theValue)
return
def setFontStyle(self, theValue):
if theValue in self.VALID_STYLE:
self._tAttr["font-style"][1] = str(theValue)
return
def setStrikeStyle(self, theValue):
if theValue in self.VALID_LSTYLE:
self._tAttr["text-line-through-style"][1] = str(theValue)
return
def setStrikeType(self, theValue):
if theValue in self.VALID_LTYPE:
self._tAttr["text-line-through-type"][1] = str(theValue)
return
##
# Methods
##
def packXML(self, xParent, xName):
"""Pack the content into an xml element.
"""
theAttr = {}
theAttr[_mkTag("style", "name")] = xName
theAttr[_mkTag("style", "family")] = "text"
xEntry = etree.SubElement(xParent, _mkTag("style", "style"), attrib=theAttr)
theAttr = {}
for aName, (aNm, aVal) in self._tAttr.items():
if aVal is not None:
theAttr[_mkTag(aNm, aName)] = aVal
if theAttr:
etree.SubElement(xEntry, _mkTag("style", "text-properties"), attrib=theAttr)
return
# END Class ODTTextStyle
# =============================================================================================== # # =============================================================================================== #
# Local Functions # Local Functions
# =============================================================================================== # # =============================================================================================== #
+41
View File
@@ -83,4 +83,45 @@ def testCoreToOdt_Convert(tmpConf, dummyGUI):
'</office:text>' '</office:text>'
) )
# Nested Text
theDoc.theText = "Some ~~nested **bold** and _italics_ text~~ text.\nNo format\n"
theDoc.tokenizeText()
theDoc.initDocument()
theDoc.doConvert()
theDoc.closeDocument()
assert xmlToText(theDoc._xText) == (
'<office:text>'
'<text:p text:style-name="Text_Body">Some '
'<text:span text:style-name="T1">nested </text:span>'
'<text:span text:style-name="T2">bold</text:span>'
'<text:span text:style-name="T1"> and </text:span>'
'<text:span text:style-name="T3">italics</text:span>'
'<text:span text:style-name="T1"> text</text:span> text. No format</text:p>'
'</office:text>'
)
# Hard Break
theDoc.theText = "Some text. \nNext line\n"
theDoc.tokenizeText()
theDoc.initDocument()
theDoc.doConvert()
theDoc.closeDocument()
assert xmlToText(theDoc._xText) == (
'<office:text>'
'<text:p text:style-name="Text_Body">Some text.<text:line-break/>Next line</text:p>'
'</office:text>'
)
# Tab
theDoc.theText = "\tItem 1\tItem 2\n"
theDoc.tokenizeText()
theDoc.initDocument()
theDoc.doConvert()
theDoc.closeDocument()
assert xmlToText(theDoc._xText) == (
'<office:text>'
'<text:p text:style-name="Text_Body"><text:tab/>Item 1<text:tab/>Item 2</text:p>'
'</office:text>'
)
# END Test testCoreToOdt_Convert # END Test testCoreToOdt_Convert