Remove the tokenizer post processing stage and instead run it where needed (Issue #1412)

This commit is contained in:
Veronica Berglyd Olsen
2023-04-15 18:30:37 +02:00
parent 6d0455bf28
commit 38f31d6a64
7 changed files with 29 additions and 35 deletions
+2 -2
View File
@@ -26,7 +26,7 @@ along with this program. If not, see <https://www.gnu.org/licenses/>.
import logging
from novelwriter.constants import nwKeyWords, nwLabels, nwHtmlUnicode
from novelwriter.core.tokenizer import Tokenizer
from novelwriter.core.tokenizer import Tokenizer, stripEscape
logger = logging.getLogger(__name__)
@@ -269,7 +269,7 @@ class ToHtml(Tokenizer):
parStyle = hStyle
for xPos, xLen, xFmt in reversed(tFormat):
tTemp = tTemp[:xPos] + htmlTags[xFmt] + tTemp[xPos+xLen:]
thisPar.append(tTemp.rstrip())
thisPar.append(stripEscape(tTemp.rstrip()))
elif tType == self.T_SYNOPSIS and self._doSynopsis:
tmpResult.append(self._formatSynopsis(tText))
+11 -17
View File
@@ -40,6 +40,17 @@ from novelwriter.constants import nwConst, nwRegEx, nwUnicode
logger = logging.getLogger(__name__)
def stripEscape(text):
"""Helper function to strip escaped markdown characters from
paragraph text.
"""
if "\\" in text:
# Checking first is slightly slower when there are escaped
# characters in the text, but significantly faster when not
return text.replace(r"\*", "*").replace(r"\~", "~").replace(r"\_", "_")
return text
class Tokenizer(ABC):
# In-Text Format
@@ -340,23 +351,6 @@ class Tokenizer(ABC):
return
def doPostProcessing(self):
"""Do some postprocessing. Overloaded by subclasses. This just
does the standard escaped characters.
"""
escapeDict = {
r"\*": "*",
r"\~": "~",
r"\_": "_",
}
escReplace = re.compile(
"|".join([re.escape(k) for k in escapeDict.keys()]), flags=re.DOTALL
)
self._theResult = escReplace.sub(
lambda x: escapeDict[x.group(0)], self._theResult
)
return
def tokenizeText(self):
"""Scan the text for either lines starting with specific
characters that indicate headers, comments, commands etc, or
+5 -4
View File
@@ -32,7 +32,7 @@ from zipfile import ZipFile
from datetime import datetime
from novelwriter.constants import nwKeyWords, nwLabels
from novelwriter.core.tokenizer import Tokenizer
from novelwriter.core.tokenizer import Tokenizer, stripEscape
logger = logging.getLogger(__name__)
@@ -1356,16 +1356,17 @@ class XMLParagraph:
return
def appendText(self, tText):
def appendText(self, text):
"""Append text to the XML element. We do this one character at
the time in order to be able to process line breaks, tabs and
spaces separately. Multiple spaces above one are concatenated
into a single tag, and must therefore be processed separately.
"""
text = stripEscape(text)
nSpaces = 0
self._rawTxt += tText
self._rawTxt += text
for c in tText:
for c in text:
if c == " ":
nSpaces += 1
continue
-1
View File
@@ -188,7 +188,6 @@ class GuiDocViewer(QTextBrowser):
aDoc.doPreProcessing()
aDoc.tokenizeText()
aDoc.doConvert()
aDoc.doPostProcessing()
except Exception:
logger.error("Failed to generate preview for document with handle '%s'", tHandle)
logException()
-1
View File
@@ -771,7 +771,6 @@ class GuiBuildNovel(QDialog):
bldObj.doHeaders()
if doConvert:
bldObj.doConvert()
bldObj.doPostProcessing()
except Exception:
logger.error("Failed to build document '%s'", tItem.itemHandle)
-4
View File
@@ -574,10 +574,6 @@ def testCoreToHtml_Methods(mockGUI):
assert theHtml.theMarkdown[-1] == (
"Text with <brackets> &amp; short&ndash;dash, long&mdash;dash &hellip;\n\n"
)
theHtml.doPostProcessing()
assert theHtml.theMarkdown[-1] == (
"Text with <brackets> &amp; short&ndash;dash, long&mdash;dash &hellip;\n\n"
)
# Result Size
assert theHtml.getFullResultSize() == 147
+11 -6
View File
@@ -24,7 +24,7 @@ import pytest
from tools import C, buildTestProject, readFile
from novelwriter.core.project import NWProject
from novelwriter.core.tokenizer import Tokenizer
from novelwriter.core.tokenizer import Tokenizer, stripEscape
class BareTokenizer(Tokenizer):
@@ -203,11 +203,6 @@ def testCoreToken_TextOps(monkeypatch, mockGUI, mockRnd, fncPath):
theToken.doPreProcessing()
assert theToken._theText == docTextR
# Post Processing
theToken._theResult = r"This is text with escapes: \** \~~ \__"
theToken.doPostProcessing()
assert theToken.theResult == "This is text with escapes: ** ~~ __"
# Save File
savePath = fncPath / "dump.nwd"
theToken.saveRawMarkdown(savePath)
@@ -223,6 +218,16 @@ def testCoreToken_TextOps(monkeypatch, mockGUI, mockRnd, fncPath):
# END Test testCoreToken_TextOps
@pytest.mark.core
def testCoreToken_StripEscape():
"""Test the stripEscape helper function.
"""
text = r"This is text with escapes: \** \~~ \__"
assert stripEscape(text) == "This is text with escapes: ** ~~ __"
return
@pytest.mark.core
def testCoreToken_HeaderFormat(mockGUI):
"""Test the tokenization of header formats in the Tokenizer class.