Fix escaped markdown not being cleaned up for HTML and ODT (#1418)

This commit is contained in:
Veronica Berglyd Olsen
2023-04-15 18:55:30 +02:00
committed by GitHub
8 changed files with 60 additions and 39 deletions
+2 -2
View File
@@ -26,7 +26,7 @@ along with this program. If not, see <https://www.gnu.org/licenses/>.
import logging import logging
from novelwriter.constants import nwKeyWords, nwLabels, nwHtmlUnicode from novelwriter.constants import nwKeyWords, nwLabels, nwHtmlUnicode
from novelwriter.core.tokenizer import Tokenizer from novelwriter.core.tokenizer import Tokenizer, stripEscape
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -269,7 +269,7 @@ class ToHtml(Tokenizer):
parStyle = hStyle parStyle = hStyle
for xPos, xLen, xFmt in reversed(tFormat): for xPos, xLen, xFmt in reversed(tFormat):
tTemp = tTemp[:xPos] + htmlTags[xFmt] + tTemp[xPos+xLen:] tTemp = tTemp[:xPos] + htmlTags[xFmt] + tTemp[xPos+xLen:]
thisPar.append(tTemp.rstrip()) thisPar.append(stripEscape(tTemp.rstrip()))
elif tType == self.T_SYNOPSIS and self._doSynopsis: elif tType == self.T_SYNOPSIS and self._doSynopsis:
tmpResult.append(self._formatSynopsis(tText)) tmpResult.append(self._formatSynopsis(tText))
+11 -17
View File
@@ -40,6 +40,17 @@ from novelwriter.constants import nwConst, nwRegEx, nwUnicode
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
def stripEscape(text):
"""Helper function to strip escaped markdown characters from
paragraph text.
"""
if "\\" in text:
# Checking first is slightly slower when there are escaped
# characters in the text, but significantly faster when not
return text.replace(r"\*", "*").replace(r"\~", "~").replace(r"\_", "_")
return text
class Tokenizer(ABC): class Tokenizer(ABC):
# In-Text Format # In-Text Format
@@ -340,23 +351,6 @@ class Tokenizer(ABC):
return return
def doPostProcessing(self):
"""Do some postprocessing. Overloaded by subclasses. This just
does the standard escaped characters.
"""
escapeDict = {
r"\*": "*",
r"\~": "~",
r"\_": "_",
}
escReplace = re.compile(
"|".join([re.escape(k) for k in escapeDict.keys()]), flags=re.DOTALL
)
self._theResult = escReplace.sub(
lambda x: escapeDict[x.group(0)], self._theResult
)
return
def tokenizeText(self): def tokenizeText(self):
"""Scan the text for either lines starting with specific """Scan the text for either lines starting with specific
characters that indicate headers, comments, commands etc, or characters that indicate headers, comments, commands etc, or
+5 -4
View File
@@ -32,7 +32,7 @@ from zipfile import ZipFile
from datetime import datetime from datetime import datetime
from novelwriter.constants import nwKeyWords, nwLabels from novelwriter.constants import nwKeyWords, nwLabels
from novelwriter.core.tokenizer import Tokenizer from novelwriter.core.tokenizer import Tokenizer, stripEscape
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
@@ -1356,16 +1356,17 @@ class XMLParagraph:
return return
def appendText(self, tText): def appendText(self, text):
"""Append text to the XML element. We do this one character at """Append text to the XML element. We do this one character at
the time in order to be able to process line breaks, tabs and the time in order to be able to process line breaks, tabs and
spaces separately. Multiple spaces above one are concatenated spaces separately. Multiple spaces above one are concatenated
into a single tag, and must therefore be processed separately. into a single tag, and must therefore be processed separately.
""" """
text = stripEscape(text)
nSpaces = 0 nSpaces = 0
self._rawTxt += tText self._rawTxt += text
for c in tText: for c in text:
if c == " ": if c == " ":
nSpaces += 1 nSpaces += 1
continue continue
-1
View File
@@ -188,7 +188,6 @@ class GuiDocViewer(QTextBrowser):
aDoc.doPreProcessing() aDoc.doPreProcessing()
aDoc.tokenizeText() aDoc.tokenizeText()
aDoc.doConvert() aDoc.doConvert()
aDoc.doPostProcessing()
except Exception: except Exception:
logger.error("Failed to generate preview for document with handle '%s'", tHandle) logger.error("Failed to generate preview for document with handle '%s'", tHandle)
logException() logException()
-1
View File
@@ -771,7 +771,6 @@ class GuiBuildNovel(QDialog):
bldObj.doHeaders() bldObj.doHeaders()
if doConvert: if doConvert:
bldObj.doConvert() bldObj.doConvert()
bldObj.doPostProcessing()
except Exception: except Exception:
logger.error("Failed to build document '%s'", tItem.itemHandle) logger.error("Failed to build document '%s'", tItem.itemHandle)
+14 -7
View File
@@ -378,7 +378,7 @@ def testCoreToHtml_ConvertDirect(mockGUI):
@pytest.mark.core @pytest.mark.core
def testCoreToHtml_SpecialCases(mockGUI): def testCoreToHtml_SpecialCases(mockGUI):
"""Test some special cases that has caused errors in the past. """Test some special cases that have caused errors in the past.
""" """
theProject = NWProject(mockGUI) theProject = NWProject(mockGUI)
theHtml = ToHtml(theProject) theHtml = ToHtml(theProject)
@@ -415,8 +415,8 @@ def testCoreToHtml_SpecialCases(mockGUI):
"<p>Test &gt; text <em>&lt;<strong>bold</strong>&gt;</em> and more.</p>\n" "<p>Test &gt; text <em>&lt;<strong>bold</strong>&gt;</em> and more.</p>\n"
) )
# Test for bug #950 # Test for issue #950
# ================= # ===================
# See: https://github.com/vkbo/novelWriter/issues/950 # See: https://github.com/vkbo/novelWriter/issues/950
theHtml.setComments(True) theHtml.setComments(True)
@@ -436,6 +436,17 @@ def testCoreToHtml_SpecialCases(mockGUI):
"<h1 style='page-break-before: always;'>Heading &lt;1&gt;</h1>\n" "<h1 style='page-break-before: always;'>Heading &lt;1&gt;</h1>\n"
) )
# Test for issue #1412
# ====================
# See: https://github.com/vkbo/novelWriter/issues/1412
theHtml._theText = "Test text \\**_bold_** and more.\n"
theHtml.tokenizeText()
theHtml.doConvert()
assert theHtml.theResult == (
"<p>Test text **<em>bold</em>** and more.</p>\n"
)
# END Test testCoreToHtml_SpecialCases # END Test testCoreToHtml_SpecialCases
@@ -574,10 +585,6 @@ def testCoreToHtml_Methods(mockGUI):
assert theHtml.theMarkdown[-1] == ( assert theHtml.theMarkdown[-1] == (
"Text with <brackets> &amp; short&ndash;dash, long&mdash;dash &hellip;\n\n" "Text with <brackets> &amp; short&ndash;dash, long&mdash;dash &hellip;\n\n"
) )
theHtml.doPostProcessing()
assert theHtml.theMarkdown[-1] == (
"Text with <brackets> &amp; short&ndash;dash, long&mdash;dash &hellip;\n\n"
)
# Result Size # Result Size
assert theHtml.getFullResultSize() == 147 assert theHtml.getFullResultSize() == 147
+13 -6
View File
@@ -24,7 +24,7 @@ import pytest
from tools import C, buildTestProject, readFile from tools import C, buildTestProject, readFile
from novelwriter.core.project import NWProject from novelwriter.core.project import NWProject
from novelwriter.core.tokenizer import Tokenizer from novelwriter.core.tokenizer import Tokenizer, stripEscape
class BareTokenizer(Tokenizer): class BareTokenizer(Tokenizer):
@@ -203,11 +203,6 @@ def testCoreToken_TextOps(monkeypatch, mockGUI, mockRnd, fncPath):
theToken.doPreProcessing() theToken.doPreProcessing()
assert theToken._theText == docTextR assert theToken._theText == docTextR
# Post Processing
theToken._theResult = r"This is text with escapes: \** \~~ \__"
theToken.doPostProcessing()
assert theToken.theResult == "This is text with escapes: ** ~~ __"
# Save File # Save File
savePath = fncPath / "dump.nwd" savePath = fncPath / "dump.nwd"
theToken.saveRawMarkdown(savePath) theToken.saveRawMarkdown(savePath)
@@ -223,6 +218,18 @@ def testCoreToken_TextOps(monkeypatch, mockGUI, mockRnd, fncPath):
# END Test testCoreToken_TextOps # END Test testCoreToken_TextOps
@pytest.mark.core
def testCoreToken_StripEscape():
"""Test the stripEscape helper function.
"""
text1 = "This is text with escapes: \\** \\~~ \\__"
text2 = "This is text with escapes: ** ~~ __"
assert stripEscape(text1) == "This is text with escapes: ** ~~ __"
assert stripEscape(text2) == "This is text with escapes: ** ~~ __"
# END Test testCoreToken_StripEscape
@pytest.mark.core @pytest.mark.core
def testCoreToken_HeaderFormat(mockGUI): def testCoreToken_HeaderFormat(mockGUI):
"""Test the tokenization of header formats in the Tokenizer class. """Test the tokenization of header formats in the Tokenizer class.
+15 -1
View File
@@ -221,7 +221,21 @@ def testCoreToOdt_TextFormatting(mockGUI):
"</office:text>" "</office:text>"
) )
# Tabs and Breaks # Test for issue #1412
# ====================
# See: https://github.com/vkbo/novelWriter/issues/1412
theDoc.initDocument()
theTxt = "Test text \\**_bold_** and more."
theFmt = " I i "
theDoc._addTextPar("Standard", oStyle, theTxt, theFmt=theFmt)
assert theDoc.getErrors() == []
assert xmlToText(theDoc._xText) == (
"<office:text>"
"<text:p text:style-name=\"Standard\">Test text **<text:span text:style-name=\"T2\">"
"bold</text:span>** and more.</text:p>"
"</office:text>"
)
# END Test testCoreToOdt_TextFormatting # END Test testCoreToOdt_TextFormatting