diff --git a/ConvertOneNote2MarkDown-v2.Tests.ps1 b/ConvertOneNote2MarkDown-v2.Tests.ps1 index 72d1ae7..4ab10d7 100644 --- a/ConvertOneNote2MarkDown-v2.Tests.ps1 +++ b/ConvertOneNote2MarkDown-v2.Tests.ps1 @@ -1201,6 +1201,9 @@ Describe 'New-SectionGroupConversionConfig' -Tag 'Unit' { $fakeMarkdownContent = @" hello world$( [char]0x00A0 ) +press the$( [char]0x00A0 ) Windows$( [char]0x00A0 )key +$( [char]0x00A0 )$( [char]0x00A0 )no leading non-breaking spaces + $( [char]0x00A0 )keeps its indentation - foo @@ -1244,11 +1247,14 @@ some more } } - # Should remove extra newline between unordered and ordered lists, remove non-breaking spaces, and '>' from ordered lists. Ignore first 8 lines for page header + # Should remove extra newline between unordered and ordered lists, remove non-breaking spaces at line boundaries but keep them as spaces between words, and '>' from ordered lists. Ignore first 8 lines for page header $split = $mutated -split "`n" $expectedBody = $split[8..($split.Count - 1)] -join "`n" $expectedBody | Should -Be $( @" hello world +press the Windows key +no leading non-breaking spaces + keeps its indentation - foo - foo1 diff --git a/ConvertOneNote2MarkDown-v2.ps1 b/ConvertOneNote2MarkDown-v2.ps1 index 3668cba..abf9bc3 100644 --- a/ConvertOneNote2MarkDown-v2.ps1 +++ b/ConvertOneNote2MarkDown-v2.ps1 @@ -152,7 +152,7 @@ Whether to include page timestamp and separator at top of document } keepspaces = @{ description = @' -Whether to clear extra newlines between unordered (bullet) and ordered (numbered) list items, non-breaking spaces from blank lines, and `>` after unordered lists +Whether to clear extra newlines between unordered (bullet) and ordered (numbered) list items, non-breaking spaces from blank lines and line ends (other non-breaking spaces become normal spaces), and `>` after unordered lists 1: Clear - Default 2: Don't clear '@ @@ -933,11 +933,24 @@ Function New-SectionGroupConversionConfig { @{ description = 'Clear extra newlines between unordered (bullet) and ordered (numbered) list items, non-breaking spaces from blank lines, and `>` after unordered lists' replacements = @( - # Remove non-breaking spaces + # Non-breaking spaces: web-clipped content is full of them, often as the only space between two words + # (e.g. 'theWindowskey'). Deleting them all glued those words together ('Windowskey'), so: + # remove them at the end of a line (incl. lines that only contain them), since converting them there could + # create a trailing-double-space hard line break, @{ - searchRegex = [regex]::Escape([char]0x00A0) + searchRegex = '(?m)[ \t\u00A0]*\u00A0[ \t\u00A0]*$' replacement = '' } + # remove them at the start of a line, keeping its indentation, since converting them there could turn the line into an indented code block, + @{ + searchRegex = '(?m)^([ \t]*)\u00A0[ \t\u00A0]*' + replacement = '$1' + } + # and turn any other run of them (with adjacent spaces) into a single space + @{ + searchRegex = '[ \t]*\u00A0[ \t\u00A0]*' + replacement = ' ' + } # Remove an extra newline between each occurrence of '- some unordered list item' @{ searchRegex = '(\s*)- ([^\r\n]*)\r*\n\r*\n(?=\s*-)' diff --git a/README.md b/README.md index c23006f..5816860 100644 --- a/README.md +++ b/README.md @@ -29,7 +29,7 @@ The powershell script `ConvertOneNote2MarkDown-v2.ps1` will utilize the OneNote * `markdown_phpextra` (PHP Markdown Extra) * `markdown_strict` (original unextended Markdown) * Improved headers, with title now as a `#` heading, standardized `DateTime` format for created and modified dates, and horizontal line to separate from rest of document -* Choose whether to clear extra newlines between unordered (bullet) and ordered (numbered) list items, non-breaking spaces from blank lines, and `>` after unordered lists, which are created when converting with Pandoc +* Choose whether to clear extra newlines between unordered (bullet) and ordered (numbered) list items, non-breaking spaces from blank lines and line ends (other non-breaking spaces become normal spaces), and `>` after unordered lists, which are created when converting with Pandoc * Choose whether to remove `\` escape symbol that are created when converting with Pandoc * Choose whether to use Line Feed (`LF`) or Carriage Return + Line Feed (`CRLF`) for new lines * Choose whether to include a `.pdf` export alongside the `.md` file. `.md` does not preserve `InkDrawing` (i.e. overlayed drawings, highlights, pen marks) absolute positions within a page, but a `.pdf` export is a complete page snapshot that preserves `InkDrawing` absolute positions within a page.