From 73951757bb0a5bb13958d9b23f869d72aa4ed465 Mon Sep 17 00:00:00 2001 From: "Veronica K. B. Olsen" <1619840+vkbo@users.noreply.github.com> Date: Mon, 28 Oct 2019 11:05:47 +0100 Subject: [PATCH] Added call to escape unicode in LaTeX export. Fails with a flag if unsuccessful --- nw/convert/text/tolatex.py | 34 +++++++++++++++------------------- requirements.txt | 1 + 2 files changed, 16 insertions(+), 19 deletions(-) diff --git a/nw/convert/text/tolatex.py b/nw/convert/text/tolatex.py index 2002f8d9..345a988e 100644 --- a/nw/convert/text/tolatex.py +++ b/nw/convert/text/tolatex.py @@ -12,6 +12,7 @@ import textwrap import logging +import codecs import re import nw @@ -23,20 +24,7 @@ class ToLaTeX(Tokenizer): def __init__(self, theProject, theParent): Tokenizer.__init__(self, theProject, theParent) - return - - def doAutoReplace(self): - Tokenizer.doAutoReplace(self) - - repDict = { - "\u2013" : "--", - "\u2014" : "---", - "\u2500" : "---", - "\u2026" : "...", - } - xRep = re.compile("|".join([re.escape(k) for k in repDict.keys()]), flags=re.DOTALL) - self.theText = xRep.sub(lambda x: repDict[x.group(0)], self.theText) - + self.texCodecFail = False return def doConvert(self): @@ -122,17 +110,17 @@ class ToLaTeX(Tokenizer): elif tType == self.T_HEAD1: self.theResult += begText - self.theResult += "{\\Huge %s}\n" % tText + self.theResult += "{\\Huge %s}\n" % self._escapeUnicode(tText) self.theResult += endText elif tType == self.T_HEAD2: - self.theResult += "\\chapter*{%s}\n\n" % tText + self.theResult += "\\chapter*{%s}\n\n" % self._escapeUnicode(tText) elif tType == self.T_HEAD3: - self.theResult += "\\section*{%s}\n\n" % tText + self.theResult += "\\section*{%s}\n\n" % self._escapeUnicode(tText) elif tType == self.T_HEAD4: - self.theResult += "\\subsection*{%s}\n\n" % tText + self.theResult += "\\subsection*{%s}\n\n" % self._escapeUnicode(tText) elif tType == self.T_SEP: self.theResult += begText @@ -140,7 +128,7 @@ class ToLaTeX(Tokenizer): self.theResult += endText elif tType == self.T_TEXT: - thisPar.append(tText) + thisPar.append(self._escapeUnicode(tText)) elif tType == self.T_PBREAK: self.theResult += "\\newpage\n\n" @@ -153,4 +141,12 @@ class ToLaTeX(Tokenizer): return + def _escapeUnicode(self, theText): + try: + import latexcodec + return codecs.encode(theText, "ulatex") + except: + self.texCodecFail = True + return theText + # END Class ToLaTeX diff --git a/requirements.txt b/requirements.txt index a83517dd..14cd2a0c 100644 --- a/requirements.txt +++ b/requirements.txt @@ -3,3 +3,4 @@ appdirs lxml pyenchant pycountry +latexcodec