From 3c474987b29ba94ae729c8e4e24e8d58a95feafd Mon Sep 17 00:00:00 2001 From: Veronica Berglyd Olsen <1619840+vkbo@users.noreply.github.com> Date: Mon, 23 Jun 2025 23:24:20 +0200 Subject: [PATCH 1/4] Check indent and justify after lines are combined in tokenizer, not before (#2426) --- novelwriter/formats/tokenizer.py | 34 ++++++----- tests/test_formats/test_fmt_tokenizer.py | 74 ++++++++++++++++++++++++ tests/test_formats/test_fmt_toodt.py | 2 +- 3 files changed, 93 insertions(+), 17 deletions(-) diff --git a/novelwriter/formats/tokenizer.py b/novelwriter/formats/tokenizer.py index e6d2d563..e0587002 100644 --- a/novelwriter/formats/tokenizer.py +++ b/novelwriter/formats/tokenizer.py @@ -880,31 +880,21 @@ class Tokenizer(ABC): if nBlock[0] != BlockTyp.TEXT: # Next block is not text, so we add the buffer to blocks nLines = len(pLines) - cStyle = pLines[0][4] - if firstIndent and not (self._noIndent or cStyle & BlockFmt.ALIGNED): - # If paragraph indentation is enabled, not temporarily - # turned off, and the block is not aligned, we add the - # text indentation flag - cStyle |= BlockFmt.IND_T + tFmt: T_Formats = [] + pTxt = "" + cStyle = BlockFmt.NONE if nLines == 1: - # The paragraph contains a single line, so we just save - # that directly to the blocks list. If justify is - # enabled, and there is no alignment, we apply it. - if doJustify and not cStyle & BlockFmt.ALIGNED: - cStyle |= BlockFmt.JUSTIFY - + # The paragraph contains a single line + tFmt = pLines[0][3] pTxt = pLines[0][2].translate(transMapB) - sBlocks.append(( - BlockTyp.TEXT, pLines[0][1], pTxt, pLines[0][3], cStyle - )) + cStyle = pLines[0][4] elif nLines > 1: # The paragraph contains multiple lines, so we need to # join them according to the line break policy, and # recompute all the formatting markers tTxt = "" - tFmt: T_Formats = [] for aBlock in pLines: tLen = len(tTxt) tTxt += f"{aBlock[2]}{lineSep}" @@ -912,6 +902,18 @@ class Tokenizer(ABC): cStyle |= aBlock[4] pTxt = tTxt[:-1].translate(transMapB) + + if nLines: + isAligned = cStyle & BlockFmt.ALIGNED + if firstIndent and not (self._noIndent or isAligned): + # If paragraph indentation is enabled, not temporarily + # turned off, and the block is not aligned, we add the + # text indentation flag + cStyle |= BlockFmt.IND_T + + if doJustify and not isAligned: + cStyle |= BlockFmt.JUSTIFY + sBlocks.append(( BlockTyp.TEXT, pLines[0][1], pTxt, tFmt, cStyle )) diff --git a/tests/test_formats/test_fmt_tokenizer.py b/tests/test_formats/test_fmt_tokenizer.py index 7795601c..58106cdd 100644 --- a/tests/test_formats/test_fmt_tokenizer.py +++ b/tests/test_formats/test_fmt_tokenizer.py @@ -1074,6 +1074,80 @@ def testFmtToken_Paragraphs(mockGUI): ] +@pytest.mark.core +def testFmtToken_BreakAlignIndent(mockGUI): + """Test the splitting of paragraphs with alignment.""" + project = NWProject() + tokens = BareTokenizer(project) + tokens._handle = TMH + + for text in [ + "This is text <<\nspanning multiple\nlines", + "This is text\nspanning multiple <<\nlines", + "This is text\nspanning multiple\nlines <<", + ]: + # Preserve Breaks + tokens.setKeepLineBreaks(True) + tokens._text = text + tokens.tokenizeText() + assert tokens._blocks == [ + (BlockTyp.TEXT, "", "This is text\nspanning multiple\nlines", [], BlockFmt.LEFT), + ] + + # Don't Preserve Breaks + tokens.setKeepLineBreaks(False) + tokens._text = text + tokens.tokenizeText() + assert tokens._blocks == [ + (BlockTyp.TEXT, "", "This is text spanning multiple lines", [], BlockFmt.LEFT), + ] + + # With Justify + # This should disable justify + tokens.setKeepLineBreaks(True) + tokens.setJustify(True) + tokens._text = text + tokens.tokenizeText() + assert tokens._blocks == [ + (BlockTyp.TEXT, "", "This is text\nspanning multiple\nlines", [], BlockFmt.LEFT), + ] + + # With Indent + # This should disable indent + tokens.setKeepLineBreaks(True) + tokens.setFirstLineIndent(True, 1.0, False) + tokens._text = text + tokens.tokenizeText() + assert tokens._blocks == [ + (BlockTyp.TEXT, "", "This is text\nspanning multiple\nlines", [], BlockFmt.LEFT), + ] + + +@pytest.mark.core +def testFmtToken_BreakJustify(mockGUI): + """Test the of processing of justify with breaks.""" + project = NWProject() + tokens = BareTokenizer(project) + tokens._handle = TMH + tokens.setJustify(True) + + # Applied to all lines when breaks are preserved + tokens._text = "This is text\nspanning multiple\nlines" + tokens.setKeepLineBreaks(True) + tokens.tokenizeText() + assert tokens._blocks == [ + (BlockTyp.TEXT, "", "This is text\nspanning multiple\nlines", [], BlockFmt.JUSTIFY), + ] + + # Turning off breaks should make no difference (see issue #2426) + tokens._text = "This is text\nspanning multiple\nlines" + tokens.setKeepLineBreaks(False) + tokens.tokenizeText() + assert tokens._blocks == [ + (BlockTyp.TEXT, "", "This is text spanning multiple lines", [], BlockFmt.JUSTIFY), + ] + + @pytest.mark.core def testFmtToken_TextFormat(mockGUI): """Test the tokenization of text formats in the Tokenizer class.""" diff --git a/tests/test_formats/test_fmt_toodt.py b/tests/test_formats/test_fmt_toodt.py index 70b72874..eded1eeb 100644 --- a/tests/test_formats/test_fmt_toodt.py +++ b/tests/test_formats/test_fmt_toodt.py @@ -733,7 +733,7 @@ def testFmtToOdt_ConvertParagraphs(mockGUI): '' 'Scene' 'Regular paragraph' - 'withbreak' + 'withbreak' 'Left Align' '' ) From 979a55d24a1e3261cb04cc92e06ee523baa24927 Mon Sep 17 00:00:00 2001 From: Veronica Berglyd Olsen <1619840+vkbo@users.noreply.github.com> Date: Tue, 24 Jun 2025 00:07:55 +0200 Subject: [PATCH 2/4] Add alignment precedence test to HTML tests --- tests/test_formats/test_fmt_tohtml.py | 51 +++++++++++++++++++++++++-- 1 file changed, 48 insertions(+), 3 deletions(-) diff --git a/tests/test_formats/test_fmt_tohtml.py b/tests/test_formats/test_fmt_tohtml.py index fb05ebfd..28b8e3e5 100644 --- a/tests/test_formats/test_fmt_tohtml.py +++ b/tests/test_formats/test_fmt_tohtml.py @@ -147,9 +147,6 @@ def testFmtToHtml_ConvertParagraphs(mockGUI): html._isNovel = True html._isFirst = True - # Paragraphs - # ========== - # Text html._text = "Some **nested bold and _italic_ and ~~strikethrough~~ text** here\n" html.tokenizeText() @@ -280,6 +277,54 @@ def testFmtToHtml_ConvertParagraphs(mockGUI): ) +@pytest.mark.core +def testFmtToHtml_Alignment(mockGUI): + """Test paragraph alignment in the ToHtml class.""" + project = NWProject() + html = ToHtml(project) + html.initDocument() + + # Left + html._text = "This is text <<\nspanning multiple\nlines" + html.tokenizeText() + html.doConvert() + assert html._pages[-1] == ( + "

This is text
spanning multiple
lines

\n" + ) + + # Right + html._text = ">> This is text\nspanning multiple\nlines" + html.tokenizeText() + html.doConvert() + assert html._pages[-1] == ( + "

This is text
spanning multiple
lines

\n" + ) + + # Centre + html._text = ">> This is text <<\nspanning multiple\nlines" + html.tokenizeText() + html.doConvert() + assert html._pages[-1] == ( + "

This is text
spanning multiple
lines

\n" + ) + + # Left before Right + html._text = ">> This is text\nspanning multiple <<\nlines" + html.tokenizeText() + html.doConvert() + assert html._pages[-1] == ( + "

This is text
spanning multiple
lines

\n" + ) + + # Right before Centre + html._text = ">> This is text <<\n>> spanning multiple\nlines" + html.tokenizeText() + html.doConvert() + assert html._pages[-1] == ( + "

This is text
spanning multiple
lines

\n" + ) + + @pytest.mark.core def testFmtToHtml_Dialog(mockGUI): """Test paragraph formats in the ToHtml class.""" From a210a665fbd10863e615363c05697281ebd785b8 Mon Sep 17 00:00:00 2001 From: Veronica Berglyd Olsen <1619840+vkbo@users.noreply.github.com> Date: Tue, 24 Jun 2025 00:08:08 +0200 Subject: [PATCH 3/4] Update documentation --- docs/source/usage/advanced_formatting.rst | 14 ++++++ docs/source/usage/alignment_and_indent.rst | 55 ++++++++++++++++++++-- 2 files changed, 65 insertions(+), 4 deletions(-) diff --git a/docs/source/usage/advanced_formatting.rst b/docs/source/usage/advanced_formatting.rst index 557db1a7..94bfec0f 100644 --- a/docs/source/usage/advanced_formatting.rst +++ b/docs/source/usage/advanced_formatting.rst @@ -56,6 +56,20 @@ activated by clicking the left-most icon button in the editor header. .. versionadded:: 2.2 +.. _docs_usage_formatting_shortcodes_break: + +Forced Line Break +----------------- + +Inserting ``[br]`` in the text will ensure a line break is always inserted in that place, even if +you turn off **Preserve Hard Line breaks** in your **Manuscript Build Settings**. + +You can add a manual line break after it too, for a better visual representation in the editor, but +keep in mind that this line break is removed before the text is processed, so the text on either +side will be considered as belonging to the same line. This can affect how alignment is treated. +See :ref:`docs_usage_align_indent_forced` for more details. + + .. _docs_usage_formatting_breaks: Vertical Space and Page Breaks diff --git a/docs/source/usage/alignment_and_indent.rst b/docs/source/usage/alignment_and_indent.rst index 2494ea32..dea1b171 100644 --- a/docs/source/usage/alignment_and_indent.rst +++ b/docs/source/usage/alignment_and_indent.rst @@ -53,18 +53,33 @@ the entire paragraph. For the following text, all lines will be centred: .. code-block:: md - >> I am the very model of a modern Major-General + >> I am the very model of a modern Major-General << I've information vegetable, animal, and mineral I know the kings of England, and I quote the fights historical - From Marathon to Waterloo, in order categorical << + From Marathon to Waterloo, in order categorical + +If you have multiple conflicting alignments on a paragraph, only one is applied. The order of +precedence is: + +#. Left alignment +#. Right alignment +#. Centred text +#. Justified text + +.. note:: + + It is strongly recommended that you keep the **Preserve Hard Line Breaks** setting enabled in + your **Manuscript Build Settings**. This setting assumes all single line breaks in your text are + intended. Turning this off makes adding line breaks much more complicated, but it is still + possible. See :ref:`docs_usage_align_indent_forced`. Alignment with First Line Indent ================================ If you have first line indent enabled in your manuscript build settings, you probably want to -disable it for text in verses. Adding any alignment tags will cause the first line indent to be -switched off for that paragraph. +disable it for text in verses. Adding any alignment tags on a paragraph will cause the first +line indent to be switched off for that paragraph. :bdg-info:`Example` @@ -76,3 +91,35 @@ The following text will always be aligned against the left margin: I've information vegetable, animal, and mineral I know the kings of England, and I quote the fights historical From Marathon to Waterloo, in order categorical + + +.. _docs_usage_align_indent_forced: + +Alignment with Forced Line Breaks +================================= + +If you turn off **Preserve Hard Line Breaks** in **Manuscript Build Settings**, you can still force +line breaks in paragraphs using the ``[br]`` shortcode. For clarity in the text, you can add a line +break after it as well. It doesn't result in two line breaks. + +Keep in mind that when the text is processed, these lines on either side of a ``[br]`` shortcode +are combined, and a trailing hard line break is *ignored*. This means that when such a paragraph is +processed, these line breaks count as the same line. This affects hiw alignment tags are handled. +For instance, this text becomes centred instead of left aligned. + +.. code-block:: md + + >> I am the very model of a modern Major-General[br] + I've information vegetable, animal, and mineral[br] + I know the kings of England, and I quote the fights historical[br] + From Marathon to Waterloo, in order categorical << + +Since this is understood as one line, this is the only way you can actually centre this paragraph. + +.. caution:: + + Due to this difference in how text with ``[br]`` tags are processed, it is generally better to + stick with the **Preserve Hard Line Breaks** setting enabled. It ensures a better correspondence + between what you see in the editor and what output you get. + +See also :ref:`docs_usage_formatting_shortcodes_break`. From c1dd2ebb92ab52f061aa7afd235d2d05c84e587a Mon Sep 17 00:00:00 2001 From: Veronica Berglyd Olsen <1619840+vkbo@users.noreply.github.com> Date: Tue, 24 Jun 2025 00:22:35 +0200 Subject: [PATCH 4/4] Fix typos and inconsistencies --- docs/source/usage/advanced_formatting.rst | 6 +++--- docs/source/usage/alignment_and_indent.rst | 18 +++++++++--------- 2 files changed, 12 insertions(+), 12 deletions(-) diff --git a/docs/source/usage/advanced_formatting.rst b/docs/source/usage/advanced_formatting.rst index 94bfec0f..556d6f1f 100644 --- a/docs/source/usage/advanced_formatting.rst +++ b/docs/source/usage/advanced_formatting.rst @@ -62,12 +62,12 @@ Forced Line Break ----------------- Inserting ``[br]`` in the text will ensure a line break is always inserted in that place, even if -you turn off **Preserve Hard Line breaks** in your **Manuscript Build Settings**. +you turn off **Preserve Hard Line Breaks** in your manuscript build settings. You can add a manual line break after it too, for a better visual representation in the editor, but keep in mind that this line break is removed before the text is processed, so the text on either -side will be considered as belonging to the same line. This can affect how alignment is treated. -See :ref:`docs_usage_align_indent_forced` for more details. +side of the ``[br]`` shortcode will be considered as belonging to the same line. This can affect +how alignment is treated. See :ref:`docs_usage_align_indent_forced` for more details. .. _docs_usage_formatting_breaks: diff --git a/docs/source/usage/alignment_and_indent.rst b/docs/source/usage/alignment_and_indent.rst index dea1b171..60949b9a 100644 --- a/docs/source/usage/alignment_and_indent.rst +++ b/docs/source/usage/alignment_and_indent.rst @@ -69,9 +69,9 @@ precedence is: .. note:: It is strongly recommended that you keep the **Preserve Hard Line Breaks** setting enabled in - your **Manuscript Build Settings**. This setting assumes all single line breaks in your text are - intended. Turning this off makes adding line breaks much more complicated, but it is still - possible. See :ref:`docs_usage_align_indent_forced`. + your manuscript build settings. This setting assumes all single line breaks in your text are + intended. Turning this off makes adding line breaks more complicated, but it is still possible. + See :ref:`docs_usage_align_indent_forced`. Alignment with First Line Indent @@ -98,13 +98,13 @@ The following text will always be aligned against the left margin: Alignment with Forced Line Breaks ================================= -If you turn off **Preserve Hard Line Breaks** in **Manuscript Build Settings**, you can still force -line breaks in paragraphs using the ``[br]`` shortcode. For clarity in the text, you can add a line -break after it as well. It doesn't result in two line breaks. +If you turn off **Preserve Hard Line Breaks** in your manuscript build settings, you can still +force line breaks in paragraphs using the ``[br]`` shortcode. For clarity in the text, you can add +a line break after it as well. It doesn't result in two line breaks. -Keep in mind that when the text is processed, these lines on either side of a ``[br]`` shortcode -are combined, and a trailing hard line break is *ignored*. This means that when such a paragraph is -processed, these line breaks count as the same line. This affects hiw alignment tags are handled. +Keep in mind that when the text is processed, the lines on either side of a ``[br]`` shortcode are +combined, and any trailing hard line break is *ignored*. This means that when such a paragraph is +processed, these line breaks count as the same line. This affects how alignment tags are handled. For instance, this text becomes centred instead of left aligned. .. code-block:: md