Refactor core and base classes (#931)
* Improve the NWIndex class * Improve the NWItem and NWDoc classes * Make some optimisations in projects and update tests * Minor changes to string formatting in main source file * Add some more protection to core classes * Make all converter class attributes private
This commit is contained in:
committed by
GitHub
parent
2f75d88694
commit
5e2ab4f612
+45
-74
@@ -27,7 +27,6 @@ along with this program. If not, see <https://www.gnu.org/licenses/>.
|
||||
import os
|
||||
import json
|
||||
import logging
|
||||
import novelwriter
|
||||
|
||||
from time import time
|
||||
|
||||
@@ -49,10 +48,10 @@ class NWIndex():
|
||||
|
||||
def __init__(self, theProject):
|
||||
|
||||
self.theProject = theProject
|
||||
|
||||
# Internal
|
||||
self.mainConf = novelwriter.CONFIG
|
||||
self.theProject = theProject
|
||||
self.indexBroken = False
|
||||
self._indexBroken = False
|
||||
|
||||
# Indices
|
||||
self._tagIndex = {}
|
||||
@@ -67,6 +66,10 @@ class NWIndex():
|
||||
|
||||
return
|
||||
|
||||
@property
|
||||
def indexBroken(self):
|
||||
return self._indexBroken
|
||||
|
||||
##
|
||||
# Public Methods
|
||||
##
|
||||
@@ -88,11 +91,7 @@ class NWIndex():
|
||||
"""
|
||||
logger.debug("Removing item '%s' from the index", tHandle)
|
||||
|
||||
delTags = []
|
||||
for tTag in self._tagIndex:
|
||||
if self._tagIndex[tTag][1] == tHandle:
|
||||
delTags.append(tTag)
|
||||
|
||||
delTags = list(filter(lambda x: self._tagIndex[x][1] == tHandle, self._tagIndex))
|
||||
for tTag in delTags:
|
||||
self._tagIndex.pop(tTag, None)
|
||||
|
||||
@@ -157,7 +156,7 @@ class NWIndex():
|
||||
except Exception:
|
||||
logger.error("Failed to load index file")
|
||||
logException()
|
||||
self.indexBroken = True
|
||||
self._indexBroken = True
|
||||
return False
|
||||
|
||||
self._tagIndex = theData.get("tagIndex", {})
|
||||
@@ -214,17 +213,17 @@ class NWIndex():
|
||||
self._checkRefIndex()
|
||||
self._checkFileIndex()
|
||||
self._checkFileMeta()
|
||||
self.indexBroken = False
|
||||
self._indexBroken = False
|
||||
|
||||
except Exception:
|
||||
logger.error("Error while checking index")
|
||||
logException()
|
||||
self.indexBroken = True
|
||||
self._indexBroken = True
|
||||
|
||||
logger.verbose("Index check took %.3f ms", (time() - tStart)*1000)
|
||||
logger.debug("Index check complete")
|
||||
|
||||
if self.indexBroken:
|
||||
if self._indexBroken:
|
||||
self.clearIndex()
|
||||
|
||||
return
|
||||
@@ -237,7 +236,8 @@ class NWIndex():
|
||||
"""Scan a piece of text associated with a handle. This will
|
||||
update the indices accordingly. This function takes the handle
|
||||
and text as separate inputs as we want to primarily scan the
|
||||
files before we save them, unless we're rebuilding the index.
|
||||
files before we save them in which case we already have the
|
||||
text.
|
||||
"""
|
||||
theItem = self.theProject.projTree[tHandle]
|
||||
theRoot = self.theProject.projTree.getRootItem(tHandle)
|
||||
@@ -259,7 +259,7 @@ class NWIndex():
|
||||
cC, wC, pC = countWords(theText)
|
||||
self._fileMeta[tHandle] = ["H0", cC, wC, pC]
|
||||
|
||||
# If the file is archived or trashed, we don't index the file itself
|
||||
# If the file is archived or in trash, we don't index the content
|
||||
if self.theProject.projTree.isTrashRoot(theItem.itemParent):
|
||||
logger.debug("Not indexing trash item '%s'", tHandle)
|
||||
return False
|
||||
@@ -276,16 +276,12 @@ class NWIndex():
|
||||
self._refIndex.pop(tHandle, None)
|
||||
self._fileIndex[tHandle] = {}
|
||||
|
||||
# Also clear references to file in tag index
|
||||
clearTags = []
|
||||
for aTag in self._tagIndex:
|
||||
if self._tagIndex[aTag][1] == tHandle:
|
||||
clearTags.append(aTag)
|
||||
# Also clear references to the file in the tags index
|
||||
clearTags = list(filter(lambda x: self._tagIndex[x][1] == tHandle, self._tagIndex))
|
||||
for aTag in clearTags:
|
||||
self._tagIndex.pop(aTag)
|
||||
|
||||
# Scan the text content
|
||||
nLine = 0
|
||||
nTitle = 0
|
||||
theLines = theText.splitlines()
|
||||
for nLine, aLine in enumerate(theLines, start=1):
|
||||
@@ -355,7 +351,7 @@ class NWIndex():
|
||||
hText = aLine[5:].strip()
|
||||
elif aLine.startswith("#! "):
|
||||
hDepth = "H1"
|
||||
hText = aLine[2:].strip()
|
||||
hText = aLine[3:].strip()
|
||||
elif aLine.startswith("##! "):
|
||||
hDepth = "H2"
|
||||
hText = aLine[4:].strip()
|
||||
@@ -374,6 +370,8 @@ class NWIndex():
|
||||
}
|
||||
|
||||
if self._fileMeta[tHandle][0] == "H0":
|
||||
# Since this initialises to H0, this ensures that only the
|
||||
# first header level is recorded in the file meta index
|
||||
self._fileMeta[tHandle][0] = hDepth
|
||||
|
||||
return True
|
||||
@@ -542,8 +540,7 @@ class NWIndex():
|
||||
hCount = [0, 0, 0, 0, 0]
|
||||
for tHandle in self._listNovelHandles(skipExcluded):
|
||||
for sTitle in self._fileIndex[tHandle]:
|
||||
theData = self._fileIndex[tHandle][sTitle]
|
||||
iLevel = H_LEVEL.get(theData["level"], 0)
|
||||
iLevel = H_LEVEL.get(self._fileIndex[tHandle][sTitle]["level"], 0)
|
||||
hCount[iLevel] += 1
|
||||
|
||||
return hCount
|
||||
@@ -551,45 +548,29 @@ class NWIndex():
|
||||
def getHandleWordCounts(self, tHandle):
|
||||
"""Get all header word counts for a specific handle.
|
||||
"""
|
||||
theCounts = []
|
||||
hRecord = self._fileIndex.get(tHandle, None)
|
||||
if hRecord is None:
|
||||
return theCounts
|
||||
|
||||
for sTitle, sData in hRecord.items():
|
||||
theCounts.append((f"{tHandle}:{sTitle}", sData["wCount"]))
|
||||
|
||||
return theCounts
|
||||
hRecord = self._fileIndex.get(tHandle, {})
|
||||
return [(f"{tHandle}:{sTitle}", sData["wCount"]) for sTitle, sData in hRecord.items()]
|
||||
|
||||
def getHandleHeaders(self, tHandle):
|
||||
"""Get all headers for a specific handle.
|
||||
"""
|
||||
theHeaders = []
|
||||
hRecord = self._fileIndex.get(tHandle, None)
|
||||
if hRecord is None:
|
||||
return theHeaders
|
||||
|
||||
for sTitle, sData in hRecord.items():
|
||||
theHeaders.append((sTitle, sData["level"], sData["title"]))
|
||||
|
||||
return theHeaders
|
||||
hRecord = self._fileIndex.get(tHandle, {})
|
||||
return [(sTitle, sData["level"], sData["title"]) for sTitle, sData in hRecord.items()]
|
||||
|
||||
def getHandleHeaderLevel(self, tHandle):
|
||||
"""Get the header level of the first header of a handle.
|
||||
"""
|
||||
if tHandle in self._fileMeta:
|
||||
return self._fileMeta[tHandle][0]
|
||||
return "H0"
|
||||
return self._fileMeta.get(tHandle, ["H0"])[0]
|
||||
|
||||
def getTableOfContents(self, maxDepth, skipExcluded=True):
|
||||
"""Generate a table of contents up to a maxiumum depth.
|
||||
"""Generate a table of contents up to a maximum depth.
|
||||
"""
|
||||
tOrder = []
|
||||
tData = {}
|
||||
pKey = None
|
||||
for tHandle in self._listNovelHandles(skipExcluded):
|
||||
for sTitle in sorted(self._fileIndex[tHandle]):
|
||||
tKey = "%s:%s" % (tHandle, sTitle)
|
||||
tKey = f"{tHandle}:{sTitle}"
|
||||
theData = self._fileIndex[tHandle][sTitle]
|
||||
iLevel = H_LEVEL.get(theData["level"], 0)
|
||||
if iLevel > maxDepth:
|
||||
@@ -605,19 +586,17 @@ class NWIndex():
|
||||
"words": theData["wCount"],
|
||||
}
|
||||
|
||||
theToC = []
|
||||
for tKey in tOrder:
|
||||
theToC.append((
|
||||
tKey,
|
||||
tData[tKey]["level"],
|
||||
tData[tKey]["title"],
|
||||
tData[tKey]["words"],
|
||||
))
|
||||
theToC = [(
|
||||
tKey,
|
||||
tData[tKey]["level"],
|
||||
tData[tKey]["title"],
|
||||
tData[tKey]["words"]
|
||||
) for tKey in tOrder]
|
||||
|
||||
return theToC
|
||||
|
||||
def getCounts(self, tHandle, sTitle=None):
|
||||
"""Returns the counts for a file, or a section of a file
|
||||
"""Return the counts for a file, or a section of a file,
|
||||
starting at title sTitle if it is provided.
|
||||
"""
|
||||
cC = 0
|
||||
@@ -640,12 +619,9 @@ class NWIndex():
|
||||
|
||||
def getReferences(self, tHandle, sTitle=None):
|
||||
"""Extract all references made in a file, and optionally title
|
||||
section. sTitle must be a string.
|
||||
section.
|
||||
"""
|
||||
theRefs = {}
|
||||
for tKey in nwKeyWords.KEY_CLASS:
|
||||
theRefs[tKey] = []
|
||||
|
||||
theRefs = {x: [] for x in nwKeyWords.KEY_CLASS}
|
||||
if tHandle not in self._refIndex:
|
||||
return theRefs
|
||||
|
||||
@@ -669,15 +645,11 @@ class NWIndex():
|
||||
"""Build a list of files referring back to our file, specified
|
||||
by tHandle.
|
||||
"""
|
||||
theRefs = {}
|
||||
if tHandle is None:
|
||||
return theRefs
|
||||
|
||||
theTags = set()
|
||||
for tTag in self._tagIndex:
|
||||
if tHandle == self._tagIndex[tTag][1]:
|
||||
theTags.add(tTag)
|
||||
return {}
|
||||
|
||||
theRefs = {}
|
||||
theTags = set(filter(lambda x: self._tagIndex[x][1] == tHandle, self._tagIndex))
|
||||
if theTags:
|
||||
for tHandle in self._refIndex:
|
||||
for sTitle in self._refIndex[tHandle]:
|
||||
@@ -690,10 +662,9 @@ class NWIndex():
|
||||
def getTagSource(self, theTag):
|
||||
"""Return the source location of a given tag.
|
||||
"""
|
||||
if theTag in self._tagIndex:
|
||||
theRef = self._tagIndex[theTag]
|
||||
if len(theRef) == 4:
|
||||
return theRef[1], theRef[0], theRef[3]
|
||||
theRef = self._tagIndex.get(theTag, [])
|
||||
if len(theRef) == 4:
|
||||
return theRef[1], theRef[0], theRef[3]
|
||||
return None, 0, "T000000"
|
||||
|
||||
##
|
||||
@@ -859,9 +830,9 @@ def countWords(theText):
|
||||
return charCount, wordCount, paraCount
|
||||
|
||||
# We need to treat dashes as word separators for counting words.
|
||||
# The check+replace apprach is much faster that direct replace for
|
||||
# The check+replace approach is much faster than direct replace for
|
||||
# large texts, and a bit slower for small texts, but in the latter
|
||||
# case it doesn't matter.
|
||||
# case it doesn't really matter.
|
||||
if nwUnicode.U_ENDASH in theText:
|
||||
theText = theText.replace(nwUnicode.U_ENDASH, " ")
|
||||
if nwUnicode.U_EMDASH in theText:
|
||||
|
||||
Reference in New Issue
Block a user