# -*- coding: utf-8 -*- """novelWriter Text Tokenizer novelWriter – Text Tokenizer ============================== Splits a piece of nW markdown text into its elements File History: Created: 2019-05-05 [0.0.1] """ import logging import nw from operator import itemgetter from PyQt5.QtCore import QRegularExpression from nw.project.document import NWDoc logger = logging.getLogger(__name__) class Tokenizer(): FMT_B_B = 1 # Begin Bold FMT_B_E = 2 # End Bold FMT_I_B = 3 # Begin Italics FMT_I_E = 4 # End Italics FMT_U_B = 5 # Begin Underline FMT_U_E = 6 # End Underline def __init__(self, theProject, theParent): self.mainConf = nw.CONFIG self.theProject = theProject self.theParent = theParent self.theText = None self.theHandle = None self.theItem = None self.theTokens = None self.theResult = None return def setText(self, theHandle, theText=None): self.theHandle = theHandle self.theItem = self.theProject.getItem(theHandle) if theText is not None: # If the text is set, just use that self.theText = theText else: # Otherwise, load it from file theDocument = NWDoc(self.theProject, self.theParent) self.theText = theDocument.openDocument(theHandle) return def tokenizeText(self): """Scan the text for either lines starting with specific characters that indicate headers, comments, commands etc, or just contains plain text. in the case of plain text, apply the same RegExes that the syntax highlighter uses and save the locations of these formatting tags into the token array. """ # RegExes for adding formatting tags within text lines # Keep in sync with the DocHighlighter class rxFormats = [( QRegularExpression(r"(?