# -*- coding: utf-8 -*- """novelWriter Text Tokenizer novelWriter – Text Tokenizer ============================== Splits a piece of nW markdown text into its elements File History: Created: 2019-05-05 [0.0.1] """ import textwrap import logging import re import nw from operator import itemgetter from PyQt5.QtCore import QRegularExpression from nw.project.document import NWDoc from nw.tools.translate import numberToWord from nw.enum import nwItemLayout logger = logging.getLogger(__name__) class Tokenizer(): FMT_B_B = 1 # Begin bold FMT_B_E = 2 # End bold FMT_I_B = 3 # Begin italics FMT_I_E = 4 # End italics FMT_U_B = 5 # Begin underline FMT_U_E = 6 # End underline T_EMPTY = 1 # Empty line (new paragraph) T_COMMENT = 2 # Comment line T_COMMAND = 3 # Command line T_HEAD1 = 4 # Header 1 (title) T_HEAD2 = 5 # Header 2 (chapter) T_HEAD3 = 6 # Header 3 (scene) T_HEAD4 = 7 # Header 4 T_TEXT = 8 # Text line T_SEP = 9 # Scene separator A_LEFT = 1 # Left aligned A_RIGHT = 2 # Right aligned A_CENTRE = 3 # Centred A_JUSTIFY = 4 # Justified def __init__(self, theProject, theParent): self.mainConf = nw.CONFIG self.theProject = theProject self.theParent = theParent self.theText = None self.theHandle = None self.theItem = None self.theTokens = None self.theResult = None self.wordWrap = 80 self.doComments = False self.doCommands = False self.fmtTitle = "%title%" self.fmtChapter = "Chapter %numword%: %title%" self.fmtUnNum = "%title%" self.fmtScene = "* * *" self.fmtSection = "%title%" self.noSection = True self.numChapter = 0 self.firstScene = False return ## # Setters ## def setComments(self, doComments): self.doComments = doComments return def setCommands(self, doCommands): self.doCommands = doCommands return def setWordWrap(self, wordWrap): if wordWrap >= 0: self.wordWrap = wordWrap else: self.wordWrap = 0 return def setTitleFormat(self, fmtTitle): self.fmtTitle = fmtTitle return def setChapterFormat(self, fmtChapter): self.fmtChapter = fmtChapter return def setUnNumberedFormat(self, fmtUnNum): self.fmtUnNum = fmtUnNum return def setSceneFormat(self, fmtScene): self.fmtScene = fmtScene return def setSectionFormat(self, fmtSection): self.fmtSection = fmtSection return ## # Class Methods ## def setText(self, theHandle, theText=None): self.theHandle = theHandle self.theItem = self.theProject.getItem(theHandle) if theText is not None: # If the text is set, just use that self.theText = theText else: # Otherwise, load it from file theDocument = NWDoc(self.theProject, self.theParent) self.theText = theDocument.openDocument(theHandle) return def doAutoReplace(self): if len(self.theProject.autoReplace) > 0: repDict = {} for aKey, aVal in self.theProject.autoReplace.items(): repDict["<%s>" % aKey] = aVal xRep = re.compile("|".join([re.escape(k) for k in repDict.keys()]), flags=re.DOTALL) self.theText = xRep.sub(lambda x: repDict[x.group(0)], self.theText) return def tokenizeText(self): """Scan the text for either lines starting with specific characters that indicate headers, comments, commands etc, or just contains plain text. in the case of plain text, apply the same RegExes that the syntax highlighter uses and save the locations of these formatting tags into the token array. """ # RegExes for adding formatting tags within text lines # Keep in sync with the DocHighlighter class rxFormats = [( QRegularExpression(r"(? 0: tWrap = textwrap.TextWrapper( width = self.wordWrap, initial_indent = "", subsequent_indent = "", expand_tabs = True, replace_whitespace = True, fix_sentence_endings = False, break_long_words = True, drop_whitespace = True, break_on_hyphens = True, tabsize = 8, max_lines = None ) self.theResult = "" thisPar = [] for tType, tText, tFormat, tAlign in self.theTokens: # First check if we have a comment or plain text, as they need some # extra replacing before we proceed to wrapping and final formatting. if tType == self.T_COMMENT: tText = "[%s]" % tText elif tType == self.T_TEXT: tTemp = tText for xPos, xLen, xFmt in reversed(tFormat): tTemp = tTemp[:xPos]+tTemp[xPos+xLen:] tText = tTemp tLen = len(tText) # The text can now be word wrapped, if we have requested this and it's needed. if tAlign == self.A_CENTRE: if self.wordWrap > 0: if tLen > self.wordWrap: aText = tWrap.wrap(tText) for n in range(len(aText)): aText[n] = self._centreText(aText[n],self.wordWrap) tText = "\n".join(aText) else: tText = self._centreText(tText,self.wordWrap) else: if self.wordWrap > 0 and tLen > self.wordWrap: tText = tWrap.fill(tText) # Then the text can receive final formatting before we append it to the results. # We also store text lines in a buffer and merge them only when we find an empty line, # indicating a new paragraph. if tType == self.T_EMPTY: if len(thisPar) > 0: self.theResult += "%s\n\n" % " ".join(thisPar) thisPar = [] elif tType == self.T_HEAD1: uLine = "="*min(tLen,self.wordWrap) if tAlign == self.A_CENTRE: uLine = self._centreText(uLine,self.wordWrap) self.theResult += "%s\n%s\n\n" % (tText,uLine) elif tType == self.T_HEAD2: uLine = "~"*min(tLen,self.wordWrap) self.theResult += "%s\n%s\n\n" % (tText,uLine) elif tType == self.T_HEAD3: uLine = "-"*min(tLen,self.wordWrap) self.theResult += "%s\n%s\n\n" % (tText,uLine) elif tType == self.T_HEAD4: self.theResult += "%s\n\n" % tText elif tType == self.T_SEP: if self.wordWrap > 0 and tLen < self.wordWrap: tText = self._centreText(tText,self.wordWrap) self.theResult += "%s\n\n" % tText elif tType == self.T_TEXT: thisPar.append(tText) elif tType == self.T_COMMENT and self.doComments: self.theResult += "%s\n\n" % tText elif tType == self.T_COMMAND and self.doCommands: self.theResult += "%s\n\n" % tText return def windowsEndings(self): self.theResult = self.theResult.replace("\n","\r\n") return ## # Internal Functions ## def _doFormatTitle(self, theText): theTitle = self.fmtTitle theTitle = theTitle.replace("%title%", theText) return theTitle def _doFormatChapter(self, theText, noNum): if noNum: theTitle = self.fmtUnNum theTitle = theTitle.replace("%title%", theText) else: theTitle = self.fmtChapter theTitle = theTitle.replace("%title%", theText) theTitle = theTitle.replace("%num%", str(self.numChapter)) theTitle = theTitle.replace("%numword%", numberToWord(self.numChapter,"en")) return theTitle def _doFormatScene(self, theText): theTitle = self.fmtScene theTitle = theTitle.replace("%title%", theText) return theTitle def _doFormatSection(self, theText): theTitle = self.fmtSection theTitle = theTitle.replace("%title%", theText) return theTitle def _centreText(self, theText, theWidth): tLen = len(theText) if tLen < theWidth: return " "*int((theWidth-tLen)/2) + theText return theText # END Class Tokenizer