Move word split to regex pattern class
This commit is contained in:
@@ -23,8 +23,6 @@ along with this program. If not, see <https://www.gnu.org/licenses/>.
|
|||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from re import UNICODE, compile
|
|
||||||
|
|
||||||
from PyQt5.QtCore import QT_TRANSLATE_NOOP, QCoreApplication
|
from PyQt5.QtCore import QT_TRANSLATE_NOOP, QCoreApplication
|
||||||
|
|
||||||
from novelwriter.enum import (
|
from novelwriter.enum import (
|
||||||
@@ -62,16 +60,13 @@ class nwConst:
|
|||||||
|
|
||||||
class nwRegEx:
|
class nwRegEx:
|
||||||
|
|
||||||
|
WORDS = r"\b[^\s\-\+\/–—\[\]:]+\b"
|
||||||
FMT_EI = r"(?<![\w\\])(_)(?![\s_])(.+?)(?<![\s\\])(\1)(?!\w)"
|
FMT_EI = r"(?<![\w\\])(_)(?![\s_])(.+?)(?<![\s\\])(\1)(?!\w)"
|
||||||
FMT_EB = r"(?<![\w\\])(\*{2})(?![\s\*])(.+?)(?<![\s\\])(\1)(?!\w)"
|
FMT_EB = r"(?<![\w\\])(\*{2})(?![\s\*])(.+?)(?<![\s\\])(\1)(?!\w)"
|
||||||
FMT_ST = r"(?<![\w\\])(~{2})(?![\s~])(.+?)(?<![\s\\])(\1)(?!\w)"
|
FMT_ST = r"(?<![\w\\])(~{2})(?![\s~])(.+?)(?<![\s\\])(\1)(?!\w)"
|
||||||
FMT_SC = r"(?i)(?<!\\)(\[[\/\!]?(?:b|i|s|u|m|sup|sub)\])"
|
FMT_SC = r"(?i)(?<!\\)(\[[\/\!]?(?:b|i|s|u|m|sup|sub)\])"
|
||||||
FMT_SV = r"(?i)(?<!\\)(\[(?:footnote):)(.+?)(?<!\\)(\])"
|
FMT_SV = r"(?i)(?<!\\)(\[(?:footnote):)(.+?)(?<!\\)(\])"
|
||||||
|
|
||||||
RX_WORDS = compile(r"\b[^\s\-\+\/–—\[\]:]+\b", UNICODE)
|
|
||||||
RX_FMT_SC = compile(r"(?i)(?<!\\)(\[[\/\!]?(?:b|i|s|u|m|sup|sub)\])", UNICODE)
|
|
||||||
RX_FMT_SV = compile(r"(?i)(?<!\\)(\[(?:footnote):)(.+?)(?<!\\)(\])", UNICODE)
|
|
||||||
|
|
||||||
|
|
||||||
class nwShortcode:
|
class nwShortcode:
|
||||||
|
|
||||||
|
|||||||
@@ -37,13 +37,17 @@ from PyQt5.QtGui import (
|
|||||||
|
|
||||||
from novelwriter import CONFIG, SHARED
|
from novelwriter import CONFIG, SHARED
|
||||||
from novelwriter.common import checkInt
|
from novelwriter.common import checkInt
|
||||||
from novelwriter.constants import nwHeaders, nwRegEx, nwUnicode
|
from novelwriter.constants import nwHeaders, nwUnicode
|
||||||
from novelwriter.core.index import processComment
|
from novelwriter.core.index import processComment
|
||||||
from novelwriter.enum import nwComment
|
from novelwriter.enum import nwComment
|
||||||
from novelwriter.text.patterns import REGEX_PATTERNS
|
from novelwriter.text.patterns import REGEX_PATTERNS
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
RX_WORDS = REGEX_PATTERNS.wordSplit
|
||||||
|
RX_FMT_SC = REGEX_PATTERNS.shortcodePlain
|
||||||
|
RX_FMT_SV = REGEX_PATTERNS.shortcodeValue
|
||||||
|
|
||||||
BLOCK_NONE = 0
|
BLOCK_NONE = 0
|
||||||
BLOCK_TEXT = 1
|
BLOCK_TEXT = 1
|
||||||
BLOCK_META = 2
|
BLOCK_META = 2
|
||||||
@@ -479,7 +483,7 @@ class TextBlockData(QTextBlockUserData):
|
|||||||
"""
|
"""
|
||||||
if "[" in text:
|
if "[" in text:
|
||||||
# Strip shortcodes
|
# Strip shortcodes
|
||||||
for rX in [nwRegEx.RX_FMT_SC, nwRegEx.RX_FMT_SV]:
|
for rX in [RX_FMT_SC, RX_FMT_SV]:
|
||||||
for match in re.finditer(rX, text[offset:]):
|
for match in re.finditer(rX, text[offset:]):
|
||||||
iS = match.start(0) + offset
|
iS = match.start(0) + offset
|
||||||
iE = match.end(0) + offset
|
iE = match.end(0) + offset
|
||||||
@@ -488,7 +492,7 @@ class TextBlockData(QTextBlockUserData):
|
|||||||
|
|
||||||
self._spellErrors = []
|
self._spellErrors = []
|
||||||
checker = SHARED.spelling
|
checker = SHARED.spelling
|
||||||
for match in re.finditer(nwRegEx.RX_WORDS, text[offset:].replace("_", " ")):
|
for match in re.finditer(RX_WORDS, text[offset:].replace("_", " ")):
|
||||||
if (
|
if (
|
||||||
(word := match.group(0))
|
(word := match.group(0))
|
||||||
and not (word.isnumeric() or word.isupper() or checker.checkWord(word))
|
and not (word.isnumeric() or word.isupper() or checker.checkWord(word))
|
||||||
|
|||||||
@@ -32,12 +32,18 @@ from novelwriter.constants import nwRegEx
|
|||||||
class RegExPatterns:
|
class RegExPatterns:
|
||||||
|
|
||||||
# Static RegExes
|
# Static RegExes
|
||||||
|
_rxWords = re.compile(nwRegEx.WORDS, re.UNICODE)
|
||||||
_rxItalic = re.compile(nwRegEx.FMT_EI, re.UNICODE)
|
_rxItalic = re.compile(nwRegEx.FMT_EI, re.UNICODE)
|
||||||
_rxBold = re.compile(nwRegEx.FMT_EB, re.UNICODE)
|
_rxBold = re.compile(nwRegEx.FMT_EB, re.UNICODE)
|
||||||
_rxStrike = re.compile(nwRegEx.FMT_ST, re.UNICODE)
|
_rxStrike = re.compile(nwRegEx.FMT_ST, re.UNICODE)
|
||||||
_rxSCPlain = re.compile(nwRegEx.FMT_SC, re.UNICODE)
|
_rxSCPlain = re.compile(nwRegEx.FMT_SC, re.UNICODE)
|
||||||
_rxSCValue = re.compile(nwRegEx.FMT_SV, re.UNICODE)
|
_rxSCValue = re.compile(nwRegEx.FMT_SV, re.UNICODE)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def wordSplit(self) -> re.Pattern:
|
||||||
|
"""Split text into words."""
|
||||||
|
return self._rxWords
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def markdownItalic(self) -> re.Pattern:
|
def markdownItalic(self) -> re.Pattern:
|
||||||
"""Markdown italic style."""
|
"""Markdown italic style."""
|
||||||
|
|||||||
@@ -40,6 +40,52 @@ def allMatches(regEx: re.Pattern, text: str) -> list[list[str]]:
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.core
|
||||||
|
def testTextPatterns_Words():
|
||||||
|
"""Test the word split regex."""
|
||||||
|
regEx = REGEX_PATTERNS.wordSplit
|
||||||
|
|
||||||
|
# Spaces
|
||||||
|
assert allMatches(regEx, "one two three") == [
|
||||||
|
[("one", 0, 3)], [("two", 4, 7)], [("three", 8, 13)]
|
||||||
|
]
|
||||||
|
|
||||||
|
# Hyphens
|
||||||
|
assert allMatches(regEx, "one-two-three") == [
|
||||||
|
[("one", 0, 3)], [("two", 4, 7)], [("three", 8, 13)]
|
||||||
|
]
|
||||||
|
|
||||||
|
# Em Dashes
|
||||||
|
assert allMatches(regEx, "one\u2014two\u2014three") == [
|
||||||
|
[("one", 0, 3)], [("two", 4, 7)], [("three", 8, 13)]
|
||||||
|
]
|
||||||
|
|
||||||
|
# Em Dashes
|
||||||
|
assert allMatches(regEx, "one\u2014two\u2014three") == [
|
||||||
|
[("one", 0, 3)], [("two", 4, 7)], [("three", 8, 13)]
|
||||||
|
]
|
||||||
|
|
||||||
|
# Plus
|
||||||
|
assert allMatches(regEx, "one+two+three") == [
|
||||||
|
[("one", 0, 3)], [("two", 4, 7)], [("three", 8, 13)]
|
||||||
|
]
|
||||||
|
|
||||||
|
# Slash
|
||||||
|
assert allMatches(regEx, "one/two/three") == [
|
||||||
|
[("one", 0, 3)], [("two", 4, 7)], [("three", 8, 13)]
|
||||||
|
]
|
||||||
|
|
||||||
|
# Brackets
|
||||||
|
assert allMatches(regEx, "one[two]three") == [
|
||||||
|
[("one", 0, 3)], [("two", 4, 7)], [("three", 8, 13)]
|
||||||
|
]
|
||||||
|
|
||||||
|
# Colon
|
||||||
|
assert allMatches(regEx, "one:two:three") == [
|
||||||
|
[("one", 0, 3)], [("two", 4, 7)], [("three", 8, 13)]
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.core
|
@pytest.mark.core
|
||||||
def testTextPatterns_Markdown():
|
def testTextPatterns_Markdown():
|
||||||
"""Test the markdown pattern regexes."""
|
"""Test the markdown pattern regexes."""
|
||||||
|
|||||||
Reference in New Issue
Block a user